{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/separate-calibration-from-sharpness",
  "id": "MHC-D-RESEARCH-0406",
  "version": "0.1.0",
  "title": "Separate calibration from sharpness",
  "summary": "A forecast can be bold without being calibrated, and calibrated by being timid.",
  "kind": "principle",
  "body": "Ask two different questions. Calibration: when you say 70%, do events in that class occur about 70% of the time? Sharpness or resolution: do your forecasts meaningfully distinguish higher-risk from lower-risk cases instead of clustering near the base rate? Improve confidence only while preserving calibration.",
  "limits": [
    "Both diagnostics need adequate sample size and comparable cases; small bins can create unstable impressions."
  ],
  "topics": [
    "union-decision-analysis-calibration"
  ],
  "intents": [],
  "source_ids": [
    "RS-58E31CFFFB569015",
    "RS-38A3FC4F5DD1B370"
  ],
  "evidence": [
    {
      "claim": "Forecast evaluation distinguishes calibration, which concerns statistical compatibility between forecast probabilities and outcomes, from discrimination or resolution, which concerns separating cases with different outcomes.",
      "source_id": "RS-58E31CFFFB569015",
      "role": "supports",
      "note": "Reliable calibration assessment needs enough comparable forecasts; small samples can look well or poorly calibrated by chance.",
      "locator": "Scoring-rule decompositions"
    },
    {
      "claim": "A 2025 interview study found that decision-makers want uncertainty information but differ in the level and form of detail they can use, and complex probabilistic communication can be hard to interpret.",
      "source_id": "RS-38A3FC4F5DD1B370",
      "role": "supports",
      "note": "The study is qualitative and context-dependent; it does not imply that probabilities should be avoided.",
      "locator": "Abstract and uncertainty-communication findings"
    }
  ],
  "use_when": [
    "A forecaster looks impressive because predictions are confident or because most favored outcomes occur."
  ],
  "avoid_when": [
    "Both diagnostics need adequate sample size and comparable cases; small bins can create unstable impressions."
  ],
  "example": "Always saying 50% may be well calibrated in a balanced environment but tells the decision-maker almost nothing about which cases differ.",
  "check": "Forecast review reports calibration and discriminatory sharpness as separate qualities rather than calling one number 'accuracy.'",
  "sources": [
    {
      "id": "RS-58E31CFFFB569015",
      "title": "Proper Scoring Rules for Estimation and Forecast Evaluation",
      "url": "https://www.annualreviews.org/content/journals/10.1146/annurev-statistics-042424-050626"
    },
    {
      "id": "RS-38A3FC4F5DD1B370",
      "title": "From scientific models to decisions: exploring uncertainty communication gaps between scientists and decision-makers",
      "url": "https://link.springer.com/article/10.1007/s10669-025-10039-w"
    }
  ],
  "relations": [],
  "collections": [
    {
      "id": "RC-3EA4A5EF5790DD81",
      "title": "Model the decision before buying more certainty",
      "url": "/collections/model-the-decision-before-buying-more-certainty"
    }
  ]
}
