{
  "schema_version": 1,
  "id": "ai-structured-intake-human-judgment-boundary-2026",
  "canonical_url": "https://brali-lifeos.github.io/evidence/ai-structured-intake-human-judgment-boundary-2026/",
  "json_url": "https://brali-lifeos.github.io/evidence/ai-structured-intake-human-judgment-boundary-2026/index.json",
  "decision": "propose-protocol",
  "decision_label": "Protocol candidate",
  "reviewed_at": "2026-08-29",
  "reviewed_by": "Brali Evidence Reviewer",
  "source": {
    "title": "Voice AI in Firms: A Natural Field Experiment on Automated Job Interviews",
    "url": "https://arxiv.org/html/2607.28222",
    "type": "primary-study",
    "doi": "10.2139/ssrn.5395709",
    "citation_text": "Jabarian B, Henkel L. Voice AI in Firms: A Natural Field Experiment on Automated Job Interviews. SSRN Working Paper, posted 2025; revised 2026.",
    "study_design": "Preregistered natural field experiment at PSG Global Solutions. Of 70,884 applications received during the experiment, 67,056 eligible applications were randomized to an AI interviewer, a human interviewer, or a choice condition. The direct causal comparison changed who conducted the information-collection interview while human recruiters evaluated applications and made every final hiring decision.",
    "population": "Applicants for 48 entry-level customer-service job postings across 41 client accounts, processed at 26 sites in 19 cities in the Philippines. Most applicants were aged 20-30 and had prior customer-service experience.",
    "intervention_or_exposure": "A voice AI agent conducted a structured but adaptive initial interview instead of a human recruiter. Human recruiters later evaluated interview information and standardized test results and retained final hiring authority.",
    "outcomes": [
      "Job offer rate",
      "Job starts",
      "Worker retention",
      "Measured worker productivity",
      "Interview structure and consistency",
      "Applicant experience",
      "Technical and refusal failures"
    ]
  },
  "supported_claim": "A defensible way to divide some repetitive high-volume workflows is to automate structured information collection while keeping consequential evaluation with a human. In this field experiment, that division improved several downstream hiring outcomes without a measured decline in worker productivity. Transcript evidence is consistent with greater standardization and comparability as a mechanism, but does not prove that mechanism independently. Brali should treat this as a task-allocation pattern to test, not as evidence that AI should make final hiring or other high-stakes decisions.",
  "unsupported_or_overstated_claims": [
    "AI interviewers are generally better than human interviewers.",
    "AI should make final hiring decisions.",
    "The result generalizes to specialized, relationship-heavy, tacit-knowledge, executive, clinical, legal, or other high-stakes work.",
    "Automating information collection removes discrimination or guarantees fairness.",
    "The same voice-AI system will produce the same results in other firms, languages, cultures, or labor markets.",
    "Human oversight automatically prevents automation bias.",
    "The controlled-variance mechanism is causally proven by the experiment."
  ],
  "limitations": [
    "The source is a working paper rather than a peer-reviewed journal article.",
    "The experiment was conducted with one recruitment-process outsourcing firm and entry-level customer-service hiring in the Philippines.",
    "Five percent of AI interviews ended because applicants were unwilling to continue with AI and seven percent experienced technical failure.",
    "The transcript-based mechanism analysis is associative even though interviewer assignment was randomized.",
    "The candidate-experience survey had a low response rate and may not represent all applicants.",
    "Applicants who were allowed to choose showed negative sorting into AI, limiting interpretation of the choice condition.",
    "Employment selection has legal, fairness, accessibility, and accountability requirements that this study does not resolve."
  ],
  "target_hack_ids": [
    "automate-intake-keep-judgment"
  ],
  "target_protocol_ids": [
    "brali:automate-intake-keep-judgment"
  ],
  "risk_flags": [],
  "notes": "Protocol direction: decompose a workflow into information collection and consequential evaluation. Consider AI for the repetitive collection stage only when inputs can be structured, auditable, and failure-handled; preserve an explicit human judgment stage, expose source material and uncertainty, and measure both process variance and downstream outcomes. Do not infer that the human stage is safe merely because it exists."
}
