{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/put-uncertainty-around-consequential-eval-differences",
  "id": "MHC-D-RESEARCH-1165",
  "version": "0.1.0",
  "title": "Put uncertainty around consequential eval differences",
  "summary": "A two-point score gap is not a verdict until you know how noisy the estimate is.",
  "kind": "principle",
  "body": "Report sample size and an appropriate uncertainty estimate around important metrics. Pair the aggregate with error counts and failure categories. A small apparent improvement should not drive a consequential release if the evaluation cannot distinguish it from sampling variation or grader noise.",
  "limits": [
    "Statistical uncertainty does not repair biased sampling, weak labels or an irrelevant metric."
  ],
  "topics": [
    "union-ai-runtime-verification-2026"
  ],
  "intents": [],
  "source_ids": [
    "RS-49EB9A55618CBC03"
  ],
  "evidence": [
    {
      "claim": "Consequential evaluation metrics should be reported with uncertainty so small observed differences are not treated as certain product improvements.",
      "source_id": "RS-49EB9A55618CBC03",
      "role": "supports",
      "note": "An uncertainty interval does not repair biased samples, weak labels or the wrong metric; it quantifies uncertainty conditional on the evaluation design.",
      "locator": "21:09-26:46, treat the judge as a classifier and put uncertainty around consequential numbers"
    }
  ],
  "use_when": [
    "Two agent versions have close evaluation scores and the result may change a launch, model-routing or safety decision."
  ],
  "avoid_when": [
    "Statistical uncertainty does not repair biased sampling, weak labels or an irrelevant metric."
  ],
  "example": "Two support agents score 91% and 93% on a small eval. Before switching, inspect confidence around the difference and whether severe escalation failures changed.",
  "check": "The release note reports uncertainty and meaningful error changes, not only a single point estimate.",
  "sources": [
    {
      "id": "RS-49EB9A55618CBC03",
      "title": "Build Evals That Actually Matter",
      "url": "https://ai.engineer/talks/3z2uT5aDx_Y-build-evals-that-actually-matter"
    }
  ],
  "relations": [],
  "collections": [
    {
      "id": "RC-ACD6385FB70207D3",
      "title": "Operate AI agents as systems you can replay, verify and constrain",
      "url": "/collections/operate-ai-agents-as-systems-you-can-replay-verify-and-constrain"
    }
  ]
}
