{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/own-the-task-eval-before-shopping-for-a-better-model",
  "id": "MHC-D-RESEARCH-0720",
  "version": "0.1.0",
  "title": "Own the task eval before shopping for a better model",
  "summary": "Without your own test, a model leaderboard is somebody else's job description.",
  "kind": "principle",
  "body": "Build a small representative task suite with the quality, latency and cost measures that matter to your workflow. Run candidate models through the same harness and starting state. Use public benchmarks as context, not as a substitute for local evidence. Keep the suite so future model releases can be tested quickly instead of restarting the comparison from opinion.",
  "limits": [
    "Small suites can overfit current work; refresh them as the task distribution changes."
  ],
  "topics": [
    "union-ai-eval-learning-workflow"
  ],
  "intents": [],
  "source_ids": [
    "RS-ED3CA2F877BE88CA"
  ],
  "evidence": [
    {
      "claim": "Anthropic argues that task evals let teams compare model or system changes against stable requirements instead of relying on impressions.",
      "source_id": "RS-ED3CA2F877BE88CA",
      "role": "supports",
      "note": "A stable eval can become stale or saturated and requires maintenance.",
      "locator": "Evals for model upgrades and baselines"
    }
  ],
  "use_when": [
    "A team is comparing models because an AI workflow feels unreliable or slow."
  ],
  "avoid_when": [
    "Small suites can overfit current work; refresh them as the task distribution changes."
  ],
  "example": "For SAP incident analysis, compare models on anonymized diagnostic cases with required evidence and false-cause penalties rather than generic coding scores.",
  "check": "A model choice can be explained from task-level evidence relevant to your workflow.",
  "sources": [
    {
      "id": "RS-ED3CA2F877BE88CA",
      "title": "Demystifying evals for AI agents",
      "url": "https://www.anthropic.com/engineering/demystifying-evals-for-ai-agents"
    }
  ],
  "relations": [
    {
      "from": "MHC-D-RESEARCH-0720",
      "to": "MHC-D-RESEARCH-0371",
      "type": "useful_with",
      "url": "/knowledge/re-benchmark-ai-when-the-model-or-task-changes"
    }
  ],
  "collections": [
    {
      "id": "RC-ABC17A288CB87C85",
      "title": "Make AI work improve from failures instead of accumulating rituals",
      "url": "/collections/make-ai-work-improve-from-failures-instead-of-accumulating-rituals"
    }
  ]
}
