{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/route-cases-using-both-human-and-ai-confidence-only-after-both-are-calibrated",
  "id": "MHC-D-RESEARCH-0505",
  "version": "0.1.0",
  "title": "Route cases using both human and AI confidence only after both are calibrated",
  "summary": "Comparing two uncalibrated confidence numbers creates a precise-looking coin toss.",
  "kind": "principle",
  "body": "Before using relative confidence to assign authority, test whether human confidence and AI confidence each discriminate correctness on the relevant task. If one signal is weak, use task class, evidence or independent review instead. Relative-confidence routing is valuable only when the inputs carry reliable metacognitive information.",
  "limits": [
    "The theoretical complementarity model assumes confidence signals more disciplined than most ad hoc workplace ratings."
  ],
  "topics": [
    "union-ai-reliance-metacognition"
  ],
  "intents": [],
  "source_ids": [
    "RS-45C4F121413B315D"
  ],
  "evidence": [
    {
      "claim": "Metacognitive sensitivity concerns how well confidence distinguishes correct from incorrect decisions, which is different from average confidence or simple calibration.",
      "source_id": "RS-45C4F121413B315D",
      "role": "supports",
      "note": "Formal metacognitive metrics require enough labeled decisions; a single confidence value cannot establish sensitivity.",
      "locator": "Abstract and theoretical model"
    },
    {
      "claim": "The 2026 mathematical model shows that human and AI metacognitive sensitivity jointly affect achievable combined accuracy when confidence is used to combine decisions.",
      "source_id": "RS-45C4F121413B315D",
      "role": "supports",
      "note": "The Bayes-optimal assumptions are stronger than ordinary workplace decision support.",
      "locator": "Analytic results"
    }
  ],
  "use_when": [
    "A team wants a rule such as 'AI decides when AI is more confident; human decides otherwise.'"
  ],
  "avoid_when": [
    "The theoretical complementarity model assumes confidence signals more disciplined than most ad hoc workplace ratings."
  ],
  "example": "Do not let an LLM override an experienced operator merely because it outputs 0.94 and the operator says 80% until those scales are validated.",
  "check": "Confidence-based routing outperforms or meaningfully complements simpler task-specific routing on held-out cases.",
  "sources": [
    {
      "id": "RS-45C4F121413B315D",
      "title": "Modeling the joint impact of human and AI metacognitive sensitivity on human-AI collaboration",
      "url": "https://www.sciencedirect.com/science/article/pii/S0022249626000192"
    }
  ],
  "relations": [],
  "collections": [
    {
      "id": "RC-39B535F35706B375",
      "title": "Calibrate when to rely on AI instead of measuring trust as a feeling",
      "url": "/collections/calibrate-when-to-rely-on-ai-instead-of-measuring-trust-as-a-feeling"
    }
  ]
}
