{
  "schema_version": 1,
  "id": "ai-review-correction-friction-boundary-2026",
  "canonical_url": "https://brali-lifeos.github.io/evidence/ai-review-correction-friction-boundary-2026/",
  "json_url": "https://brali-lifeos.github.io/evidence/ai-review-correction-friction-boundary-2026/index.json",
  "decision": "propose-protocol",
  "decision_label": "Protocol candidate",
  "reviewed_at": "2026-08-29",
  "reviewed_by": "Brali Evidence Reviewer",
  "source": {
    "title": "Bias in the Loop: How Humans Evaluate AI-Generated Suggestions",
    "url": "https://hdsr.mitpress.mit.edu/pub/nrcn4h7d/release/2",
    "type": "primary-study",
    "doi": "10.1162/99608f92.0e98898d",
    "citation_text": "Beck J, Eckman S, Kern C, Kreuter F. Harvard Data Science Review. 2026;8(2).",
    "study_design": "Factorial randomized experiment with 2,784 U.S.-based Prolific participants reviewing ten corporate greenhouse-gas tables pre-annotated by an AI system. The experiment manipulated the correctness of the first three AI suggestions, whether rejecting an AI suggestion required entering a corrected value, and whether high accuracy received a performance bonus. A subset had completed an AI-attitudes survey one week earlier.",
    "population": "Adult U.S.-based crowdworkers on Prolific. Participants were generally experienced online task workers but were not selected for greenhouse-gas accounting expertise.",
    "intervention_or_exposure": "Human reviewers judged whether AI-extracted values were correct. In one randomized condition, flagging an AI error also required typing the corrected value, making rejection more effortful than acceptance.",
    "outcomes": [
      "Annotation accuracy",
      "Correction rate",
      "Undercorrection",
      "Overcorrection",
      "Annotation time"
    ]
  },
  "supported_claim": "When humans review AI-generated suggestions, the review interface itself can bias behavior. In this experiment, adding repair work to the act of rejecting an AI suggestion reduced correction activity and increased undercorrection. Brali can therefore justify a bounded workflow rule: make it cheap to flag or reject an AI output, and separate validation from repair when the repair burden would otherwise make acceptance the path of least resistance.",
  "unsupported_or_overstated_claims": [
    "Making correction easier will always increase overall accuracy.",
    "People who distrust AI are universally better reviewers.",
    "Performance bonuses cannot improve AI review in other settings.",
    "The same effect size applies to expert, medical, legal, financial, or safety-critical review.",
    "A human-in-the-loop label by itself guarantees reliable oversight.",
    "AI suggestions should be hidden from reviewers."
  ],
  "limitations": [
    "The experiment used crowdworkers rather than domain experts.",
    "The task was limited to ten greenhouse-gas reporting tables and one pre-annotation workflow.",
    "Several difficult items required domain knowledge that many annotators lacked.",
    "The randomized correction-burden manipulation changed correction behavior, but the regression analysis did not show a clear overall accuracy loss because overcorrections also decreased.",
    "AI attitudes predicted behavior observationally rather than through randomized manipulation.",
    "The study did not include a no-AI baseline and could not analyze item-order effects."
  ],
  "target_hack_ids": [
    "make-ai-rejection-cheap"
  ],
  "target_protocol_ids": [
    "brali:make-ai-rejection-cheap"
  ],
  "risk_flags": [],
  "notes": "Protocol direction: design AI review so Accept and Flag/Reject have comparable friction. Let the reviewer flag an output first; route correction, rewriting, or remediation into a separate step or queue when doing both at once would create asymmetric effort. Track undercorrection and overcorrection separately rather than treating 'human reviewed' as a quality guarantee."
}
