{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/give-the-second-check-evidence-the-first-answer-did-not-create",
  "id": "MHC-D-RESEARCH-0996",
  "version": "0.1.0",
  "title": "Give the second check evidence the first answer did not create",
  "summary": "Another confident paragraph is not automatically another line of evidence.",
  "kind": "protocol",
  "body": "Choose a check with a different failure route: execute a calculation, inspect the cited passage, compare with a controlled fixture or ask a qualified reviewer to assess the original evidence. Self-correction can help, and training can improve it, but a repeated assurance is not proof that an external check occurred.",
  "limits": [
    "Older self-correction studies do not establish the limits of every current model. The external test can also be wrong, so inspect its assumptions."
  ],
  "topics": [
    "work-03-ai-assurance"
  ],
  "intents": [],
  "source_ids": [
    "RS-63AF4652B6CE726C",
    "RS-49018FAD4FA1C49C"
  ],
  "evidence": [
    {
      "claim": "The cited ICLR study found that intrinsic reasoning self-correction without external feedback could fail or degrade performance in its tested settings.",
      "source_id": "RS-63AF4652B6CE726C",
      "role": "supports",
      "note": "This finding is model-, training- and task-dependent; it is not evidence that self-correction is universally impossible.",
      "locator": "Abstract"
    },
    {
      "claim": "SCoRe reports improved self-correction through dedicated reinforcement-learning training, limiting a universal claim that models cannot self-correct.",
      "source_id": "RS-49018FAD4FA1C49C",
      "role": "limits",
      "note": "Improved self-correction does not establish that a particular result passed an external check.",
      "locator": "Abstract"
    }
  ],
  "use_when": [
    "You are tempted to verify an AI result by asking the same assistant whether it is sure."
  ],
  "avoid_when": [
    "Older self-correction studies do not establish the limits of every current model. The external test can also be wrong, so inspect its assumptions."
  ],
  "example": "For a generated reconciliation formula, use a small hand-checked dataset with missing, extra and duplicate IDs.",
  "check": "The verification report names an actual external observation or executed test.",
  "steps": [
    "Identify the exact claim or output that matters.",
    "Choose a check capable of finding its likely failure independently of the generated explanation.",
    "Record the observed result and any disagreement instead of asking for reassurance again."
  ],
  "sources": [
    {
      "id": "RS-63AF4652B6CE726C",
      "title": "Large Language Models Cannot Self-Correct Reasoning Yet",
      "url": "https://arxiv.org/abs/2310.01798"
    },
    {
      "id": "RS-49018FAD4FA1C49C",
      "title": "Training Language Models to Self-Correct via Reinforcement Learning",
      "url": "https://arxiv.org/abs/2409.12917"
    }
  ],
  "relations": [
    {
      "from": "MHC-D-RESEARCH-0996",
      "to": "MHC-D-RESEARCH-0993",
      "type": "useful_with",
      "url": "/knowledge/test-what-should-stay-unchanged-when-the-input-changes"
    }
  ],
  "collections": [
    {
      "id": "RC-620AACFCDEA000D1",
      "title": "Make AI-assisted work earn your trust",
      "url": "/collections/make-ai-assisted-work-earn-your-trust"
    }
  ]
}
