{
  "version": "1.0",
  "last_substantive_review": "2026-08-07",
  "license": "CC BY 4.0",
  "kind": "vendor-question-bank",
  "fields": [
    "id",
    "category",
    "question",
    "evidence",
    "red_flag",
    "gate",
    "applies_to"
  ],
  "items": [
    {
      "id": "q01",
      "category": "purpose-and-scope",
      "question": "What exact hiring problem and workflow decision is the evaluated system intended to change?",
      "evidence": "A written intended-use statement mapped to the buyer's proposed workflow.",
      "red_flag": "The answer is a feature list or a claim that the system is suitable for every recruiting stage.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q02",
      "category": "purpose-and-scope",
      "question": "Which recruiting stages and decisions are explicitly outside the evaluated scope?",
      "evidence": "A production workflow diagram with included and excluded actions.",
      "red_flag": "Roadmap functions or adjacent products are presented as current evaluated capability.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q03",
      "category": "purpose-and-scope",
      "question": "Who is the accountable owner on the vendor side and the expected owner on the buyer side?",
      "evidence": "Named operational, technical, candidate-support, and incident responsibilities.",
      "red_flag": "Responsibility moves between sales, customer success, the employer, and a model provider without an owner.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q04",
      "category": "purpose-and-scope",
      "question": "Which roles, geographies, languages, volumes, and candidate populations were the system designed and tested for?",
      "evidence": "Intended-use documentation and evaluation slices with explicit exclusions.",
      "red_flag": "One aggregate result is generalized to roles or regions that were not represented.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q05",
      "category": "purpose-and-scope",
      "question": "Which decisions remain human and what evidence does that human receive?",
      "evidence": "A live demonstration of review, correction, override, and escalation on a difficult case.",
      "red_flag": "Human in the loop means only a confirmation click or a policy statement.",
      "gate": true,
      "applies_to": "screening, ranking, and agent systems"
    },
    {
      "id": "q06",
      "category": "purpose-and-scope",
      "question": "What system, model, data, and workflow changes require the buyer to re-evaluate the deployment?",
      "evidence": "Versioning and change-notification policy tied to retest triggers.",
      "red_flag": "The vendor may change material components without notice or a reproducible version record.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q07",
      "category": "system-and-evidence",
      "question": "Can the vendor diagram models, retrieval, rules, tools, integrations, humans, and external actions in the production workflow?",
      "evidence": "A current architecture and data-flow diagram for the contracted configuration.",
      "red_flag": "A single AI box hides third-party models, human work, or consequential tool calls.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q08",
      "category": "system-and-evidence",
      "question": "What evidence supports each material accuracy, quality, efficiency, fairness, or outcome claim?",
      "evidence": "Claim register with task, population, sample, labels, metric, denominator, date, and limitations.",
      "red_flag": "Claims rely on unattributed percentages, selected customers, or a benchmark unrelated to the buyer's task.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q09",
      "category": "system-and-evidence",
      "question": "Will the vendor show failed, borderline, rejected, and insufficient-information cases as well as successes?",
      "evidence": "Buyer-selected test cases and access to unselected or lower-ranked examples.",
      "red_flag": "Only curated outputs or a vendor-controlled demo environment are available.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q10",
      "category": "system-and-evidence",
      "question": "How were ground-truth labels or evaluation judgments created and adjudicated?",
      "evidence": "Rubric, rater qualifications, agreement results, disagreements, and adjudication record.",
      "red_flag": "Another model or one untrained reviewer is treated as unquestionable ground truth.",
      "gate": false,
      "applies_to": "ranking and screening systems"
    },
    {
      "id": "q11",
      "category": "system-and-evidence",
      "question": "What known limitations, unsupported uses, and negative results are documented?",
      "evidence": "Current model card, risk assessment, test report, release notes, and limitation register.",
      "red_flag": "Documentation lists only benefits or says the system is bias-free, objective, or always accurate.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q12",
      "category": "system-and-evidence",
      "question": "Can the buyer reproduce the evaluation with its own role, cases, configuration, and reviewers?",
      "evidence": "A bounded test protocol, exportable outputs, version identifiers, and buyer-visible logs.",
      "red_flag": "Results cannot be reproduced outside a vendor presentation or depend on undisclosed tuning.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q13",
      "category": "data-and-provenance",
      "question": "Which candidate, job, interaction, and outcome data sources are used, licensed, inferred, generated, or supplied by the buyer?",
      "evidence": "Field-level provenance and a data inventory for the evaluated workflow.",
      "red_flag": "Source facts, vendor inference, and generated summaries are indistinguishable.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q14",
      "category": "data-and-provenance",
      "question": "What role, geography, language, career-stage, and source coverage gaps are known?",
      "evidence": "Coverage analysis for the relevant population with missing and weak segments.",
      "red_flag": "A total profile or record count is used as the only coverage evidence.",
      "gate": false,
      "applies_to": "sourcing and ranking systems"
    },
    {
      "id": "q15",
      "category": "data-and-provenance",
      "question": "How is freshness shown and how are stale, contradictory, duplicate, or merged records handled?",
      "evidence": "Field dates, confidence or provenance, deduplication tests, and correction workflow.",
      "red_flag": "Generated text looks current even when underlying evidence has no date.",
      "gate": false,
      "applies_to": "sourcing, enrichment, and workflow systems"
    },
    {
      "id": "q16",
      "category": "data-and-provenance",
      "question": "Which protected, sensitive, biometric, or proxy characteristics are collected or inferred?",
      "evidence": "Data dictionary, purpose, access, retention, and evaluation of necessity for each field.",
      "red_flag": "The system infers sensitive traits because direct collection was avoided or uses them without a defined purpose.",
      "gate": true,
      "applies_to": "screening and candidate-data systems"
    },
    {
      "id": "q17",
      "category": "data-and-provenance",
      "question": "Where does data flow, including model providers, subprocessors, analytics, support tools, exports, and backups?",
      "evidence": "Data-flow map, subprocessor list, regions, contractual roles, and access controls.",
      "red_flag": "The privacy page does not match the contracted configuration or omits human service access.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q18",
      "category": "data-and-provenance",
      "question": "How can the buyer and affected person access, correct, restrict, delete, suppress, or export relevant data?",
      "evidence": "Demonstrated request workflow, service levels, propagation behavior, and audit record.",
      "red_flag": "Corrections do not reach derived data or deleted records reappear from another source.",
      "gate": true,
      "applies_to": "candidate-data and engagement systems"
    },
    {
      "id": "q19",
      "category": "candidate-impact",
      "question": "What are candidates told about AI use, evaluated information, timing, and consequential decisions?",
      "evidence": "Candidate-facing notice shown in the actual journey before the relevant action.",
      "red_flag": "Notice is hidden, generic, delivered after assessment, or inconsistent with the workflow.",
      "gate": true,
      "applies_to": "candidate-facing and screening systems"
    },
    {
      "id": "q20",
      "category": "candidate-impact",
      "question": "How does a candidate request an accommodation and reach an accountable human?",
      "evidence": "End-to-end test of request, response, alternative path, scoring, and downstream handling.",
      "red_flag": "Only technical support exists or requesting accommodation removes the person from normal consideration.",
      "gate": true,
      "applies_to": "candidate-facing systems"
    },
    {
      "id": "q21",
      "category": "candidate-impact",
      "question": "Has the complete candidate journey been tested for keyboard, screen reader, mobile, timing, language, interruption, and error recovery?",
      "evidence": "Accessibility test report, user testing, defects, remediation, and retest evidence.",
      "red_flag": "The vendor cites a generic compliance statement without the evaluated flow or alternative path.",
      "gate": true,
      "applies_to": "candidate-facing systems"
    },
    {
      "id": "q22",
      "category": "candidate-impact",
      "question": "Can candidates correct material data, contest a consequential result, or receive reassessment?",
      "evidence": "Visible correction and escalation workflow with ownership and record updates.",
      "red_flag": "There is no route to change an incorrect inference, status, score, or duplicate record.",
      "gate": true,
      "applies_to": "screening, ranking, and candidate-data systems"
    },
    {
      "id": "q23",
      "category": "candidate-impact",
      "question": "What candidate feedback, completion, drop-off, complaint, opt-out, and accommodation signals are monitored?",
      "evidence": "Metric definitions, reporting path visibility, review cadence, and incident examples.",
      "red_flag": "No complaints is treated as satisfaction even when candidates cannot identify or report AI-related issues.",
      "gate": false,
      "applies_to": "candidate-facing systems"
    },
    {
      "id": "q24",
      "category": "candidate-impact",
      "question": "What happens when the system has insufficient or conflicting evidence about a candidate?",
      "evidence": "Demonstration of unknown state, human follow-up, alternative evidence, and no-guess behavior.",
      "red_flag": "The model produces a confident score or rejection from missing information.",
      "gate": true,
      "applies_to": "ranking and screening systems"
    },
    {
      "id": "q25",
      "category": "controls-and-integrations",
      "question": "Which records and fields can each component read, write, export, message, or schedule?",
      "evidence": "Permission matrix and live credential configuration using least privilege.",
      "red_flag": "Broad administrative access is required for a narrow recruiting task.",
      "gate": true,
      "applies_to": "integrated and agent systems"
    },
    {
      "id": "q26",
      "category": "controls-and-integrations",
      "question": "Which actions require approval and is that policy enforced at the tool or workflow boundary?",
      "evidence": "Rejected unauthorized call, approval record, and execution tied to the approved object version.",
      "red_flag": "Approval is a prompt instruction, optional reviewer habit, or one click covering changed content.",
      "gate": true,
      "applies_to": "agent and external-action systems"
    },
    {
      "id": "q27",
      "category": "controls-and-integrations",
      "question": "How are sender identity, channel, message, timing, follow-up, suppression, and opt-out controlled?",
      "evidence": "Production configuration, suppression test, approval log, and stopped queued action.",
      "red_flag": "The vendor cannot show who authorized the sender or how a new suppression affects queued work.",
      "gate": true,
      "applies_to": "engagement systems"
    },
    {
      "id": "q28",
      "category": "controls-and-integrations",
      "question": "Can the vendor demonstrate field mappings, duplicates, retries, idempotency, partial writes, and reconciliation in the buyer's actual integration?",
      "evidence": "Sandbox or pilot tests against the named ATS, CRM, calendar, and communication systems.",
      "red_flag": "An integration logo or generic API documentation replaces a production-depth demonstration.",
      "gate": true,
      "applies_to": "integrated systems"
    },
    {
      "id": "q29",
      "category": "controls-and-integrations",
      "question": "What happens when a tool is unavailable, a credential is revoked, or a downstream write is rejected?",
      "evidence": "Failure-injection test showing containment, status, safe retry, escalation, and manual completion.",
      "red_flag": "The agent silently retries, duplicates an action, loses state, or invents success.",
      "gate": true,
      "applies_to": "agent and integrated systems"
    },
    {
      "id": "q30",
      "category": "controls-and-integrations",
      "question": "Can the buyer pause one action, candidate, role, channel, integration, agent, or the full environment?",
      "evidence": "Buyer-visible kill switches, queue behavior, service levels, and tested resumption.",
      "red_flag": "Only the vendor can stop work or a pause leaves scheduled retries active.",
      "gate": true,
      "applies_to": "agent and external-action systems"
    },
    {
      "id": "q31",
      "category": "controls-and-integrations",
      "question": "Can a human finish the workflow with the approved brief, evidence, conversation state, suppressions, and pending commitments?",
      "evidence": "Observed manual takeover and later reconciliation back into the system.",
      "red_flag": "Critical context exists only in model memory, vendor operations, or inaccessible logs.",
      "gate": true,
      "applies_to": "agent and managed-service systems"
    },
    {
      "id": "q32",
      "category": "controls-and-integrations",
      "question": "How is untrusted resume, message, or linked content prevented from overriding policy or expanding tool access?",
      "evidence": "Threat model and tests showing data cannot grant permissions, expose records, or bypass workflow rules.",
      "red_flag": "Prompt instructions are the only boundary between candidate content and privileged actions.",
      "gate": true,
      "applies_to": "agent systems"
    },
    {
      "id": "q33",
      "category": "operations-and-security",
      "question": "What audit trail connects source evidence, system recommendation, approval, tool call, external effect, human edit, and final outcome?",
      "evidence": "A reconstructed real or test case with stable identifiers and versioned events.",
      "red_flag": "Generated summaries overwrite source facts or actions cannot be tied to an approver.",
      "gate": true,
      "applies_to": "all consequential systems"
    },
    {
      "id": "q34",
      "category": "operations-and-security",
      "question": "How are logs minimized, protected, retained, exported, deleted, and made available during an incident?",
      "evidence": "Logging design, role access, retention settings, redaction, and incident retrieval test.",
      "red_flag": "Observability copies full candidate data broadly or is unavailable to the accountable buyer.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q35",
      "category": "operations-and-security",
      "question": "What incident categories, notification timelines, evidence preservation, correction, and return-to-service rules apply?",
      "evidence": "Incident plan, named contacts, contractual commitments, and a tabletop or prior example.",
      "red_flag": "Model-quality, candidate, message, or integration incidents fall outside the vendor's security process and have no owner.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q36",
      "category": "operations-and-security",
      "question": "How are production quality, overrides, errors, candidate signals, and drift monitored by version and role?",
      "evidence": "Monitoring definitions, thresholds, review cadence, version history, and investigation examples.",
      "red_flag": "Monitoring covers uptime and latency but not hiring quality, candidate impact, or changed behavior.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q37",
      "category": "operations-and-security",
      "question": "How are vendor staff, support access, managed-service operators, and subprocessors authorized and audited?",
      "evidence": "Access roles, approval, logging, training, review, revocation, and subprocessor obligations.",
      "red_flag": "Human operations can bypass controls that are enforced only in the product interface.",
      "gate": true,
      "applies_to": "managed-service and support-enabled systems"
    },
    {
      "id": "q38",
      "category": "commercial-and-exit",
      "question": "What fixed, variable, minimum, overage, credit, renewal, and expiration terms govern the evaluated workflow?",
      "evidence": "Order form and governing terms using the same unit definitions as the evaluation.",
      "red_flag": "Headline price excludes required data, model, message, implementation, or service units.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q39",
      "category": "commercial-and-exit",
      "question": "Which implementation, integration, operation, review, QA, support, and candidate-service labor belongs to the vendor or buyer?",
      "evidence": "Responsibility matrix and time record from references or the pilot.",
      "red_flag": "Automation claims exclude substantial buyer cleanup or hidden vendor operation.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q40",
      "category": "commercial-and-exit",
      "question": "How are role, candidate, qualified, interested, introduction, interview, usage, and remedy defined?",
      "evidence": "Consistent definitions in product documentation, pilot protocol, order form, and terms.",
      "red_flag": "Marketing outcome language becomes a different activity or unit in the agreement.",
      "gate": true,
      "applies_to": "outcome and managed-service systems"
    },
    {
      "id": "q41",
      "category": "commercial-and-exit",
      "question": "What remedy applies when the promised deliverable or service level is missed?",
      "evidence": "Contract language for rerun, replacement, credit, support, refund, and exclusions.",
      "red_flag": "The only remedy is more usage of the same failed process or a nonbinding sales promise.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q42",
      "category": "commercial-and-exit",
      "question": "How can the buyer export data, configurations, evidence, logs, and workflow context and exit the service?",
      "evidence": "Test export, documented formats, timing, fees, deletion, transition support, and continuity plan.",
      "red_flag": "Data export exists but decisions, approvals, conversations, or suppressions cannot be operationally transferred.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q43",
      "category": "pilot-and-outcomes",
      "question": "Will the vendor accept a pilot with a frozen brief, buyer-controlled baseline, representative cases, and pre-agreed decision rule?",
      "evidence": "Signed pilot protocol with scope, roles, metric definitions, owners, and end date.",
      "red_flag": "The pilot uses only vendor-selected roles or success criteria are chosen after results.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q44",
      "category": "pilot-and-outcomes",
      "question": "Will the pilot sample selected, rejected, borderline, stale, duplicate, and insufficient-information cases?",
      "evidence": "Sampling frame and completed review record across the defined cases.",
      "red_flag": "Only top-ranked or successful outputs are available for review.",
      "gate": true,
      "applies_to": "ranking and screening systems"
    },
    {
      "id": "q45",
      "category": "pilot-and-outcomes",
      "question": "Which outcome, quality, time, labor, candidate-impact, and reliability metrics will be collected with explicit denominators?",
      "evidence": "Measurement plan, data sources, collection owners, and interpretation limits.",
      "red_flag": "A dashboard counts activity but omits review labor, errors, candidate impact, or denominator definitions.",
      "gate": false,
      "applies_to": "all systems"
    },
    {
      "id": "q46",
      "category": "pilot-and-outcomes",
      "question": "What stop conditions and recovery actions apply before increasing candidate or operational exposure?",
      "evidence": "Observable triggers, owner, kill-switch test, queue handling, candidate correction, and fallback.",
      "red_flag": "The pilot cannot stop without losing context or affecting other hiring workflows.",
      "gate": true,
      "applies_to": "candidate-facing, integrated, and agent systems"
    },
    {
      "id": "q47",
      "category": "pilot-and-outcomes",
      "question": "How will material changes during the pilot be frozen, versioned, or treated as a new phase?",
      "evidence": "Configuration register and separate pre-change and post-change evaluation results.",
      "red_flag": "Models, sources, prompts, thresholds, or approvals change silently and results are blended.",
      "gate": true,
      "applies_to": "all systems"
    },
    {
      "id": "q48",
      "category": "pilot-and-outcomes",
      "question": "What exact boundary may expand if the pilot passes, and what still requires new evidence?",
      "evidence": "Closeout decision naming roles, populations, permissions, volume, controls, exclusions, and re-test triggers.",
      "red_flag": "Success on one role or component is used to approve every feature, geography, or autonomous action.",
      "gate": true,
      "applies_to": "all systems"
    }
  ]
}
