{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/measure-appropriate-reliance-as-two-errors-not-one-trust-score",
  "id": "MHC-D-RESEARCH-0506",
  "version": "0.1.0",
  "title": "Measure appropriate reliance as two errors, not one trust score",
  "summary": "Good reliance means knowing both when to listen and when not to.",
  "kind": "protocol",
  "body": "Measure two behavioral mistakes separately: rejecting correct AI advice and accepting incorrect AI advice. Add task-specific costs because one error may be much worse than the other. Overall agreement can rise while appropriate reliance gets worse.",
  "limits": [
    "Some outputs are not objectively binary correct/incorrect; define adjudication and uncertainty carefully."
  ],
  "topics": [
    "union-ai-reliance-metacognition"
  ],
  "intents": [],
  "source_ids": [
    "RS-BE3598079F5EB695",
    "RS-E5AA539FACACCC98"
  ],
  "evidence": [
    {
      "claim": "The same 2026 study found decision performance depended on AI recommendation quality and that adopting poor advice could reduce performance.",
      "source_id": "RS-BE3598079F5EB695",
      "role": "supports",
      "note": "Viewing advice without adopting it is behaviorally different from relying on it.",
      "locator": "Abstract highlights"
    },
    {
      "claim": "Reliance quality should distinguish accepting correct advice from accepting incorrect advice rather than measuring trust or overall agreement alone.",
      "source_id": "RS-E5AA539FACACCC98",
      "role": "supports",
      "note": "The best metric depends on task costs and whether false acceptance and false rejection have different consequences.",
      "locator": "Appropriate reliance and behavioral calibration results"
    }
  ],
  "use_when": [
    "A human-AI workflow is evaluated with one question such as 'Do users trust the AI?' or 'How often do they agree?'"
  ],
  "avoid_when": [
    "Some outputs are not objectively binary correct/incorrect; define adjudication and uncertainty carefully."
  ],
  "example": "In data deletion support, accepting one wrong recommendation may matter more than rejecting several correct suggestions, so agreement rate is a poor primary metric.",
  "check": "The evaluation can distinguish under-reliance from over-reliance and connect each to real task cost.",
  "steps": [
    "The evaluation can distinguish under-reliance from over-reliance and connect each to real task cost."
  ],
  "sources": [
    {
      "id": "RS-BE3598079F5EB695",
      "title": "Who listens to ChatGPT and when should they? A two-study examination of AI-assisted decision making",
      "url": "https://www.sciencedirect.com/science/article/pii/S2949882126000344"
    },
    {
      "id": "RS-E5AA539FACACCC98",
      "title": "More is not better: Visual uncertainty cues and the fragility of trust calibration in LLM-assisted decision making",
      "url": "https://doi.org/10.1016/j.chbah.2026.100307"
    }
  ],
  "relations": [
    {
      "from": "MHC-D-RESEARCH-0506",
      "to": "MHC-D-RESEARCH-0382",
      "type": "useful_with",
      "url": "/knowledge/measure-ai-speed-and-quality-as-separate-outcomes"
    }
  ],
  "collections": [
    {
      "id": "RC-39B535F35706B375",
      "title": "Calibrate when to rely on AI instead of measuring trust as a feeling",
      "url": "/collections/calibrate-when-to-rely-on-ai-instead-of-measuring-trust-as-a-feeling"
    }
  ]
}
