{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/turn-subjective-quality-into-evaluator-criteria-before-looping-on-it",
  "id": "MHC-D-RESEARCH-0714",
  "version": "0.1.0",
  "title": "Turn subjective quality into evaluator criteria before looping on it",
  "summary": "An evaluator cannot enforce taste that nobody has made legible.",
  "kind": "protocol",
  "body": "Before adding an evaluator loop, write a small rubric for the dimensions that matter, with observable examples or failure anchors. Separate hard requirements from preference. Let the evaluator point to specific violations and evidence, not emit one opaque score. Periodically compare evaluator judgments with human review so the loop does not optimize a distorted proxy.",
  "limits": [
    "Some quality remains irreducibly judgmental; a rubric makes it discussable but does not turn taste into objective truth."
  ],
  "topics": [
    "union-ai-coding-harness-work"
  ],
  "intents": [],
  "source_ids": [
    "RS-4CF3167EB387E6BE"
  ],
  "evidence": [
    {
      "claim": "Anthropic's evaluator work operationalized subjective quality by writing concrete grading criteria before using an evaluator agent to drive iteration.",
      "source_id": "RS-4CF3167EB387E6BE",
      "role": "supports",
      "note": "Rubrics encode judgment imperfectly and can be gamed; periodic human calibration remains important.",
      "locator": "Generator-evaluator loop and evaluator criteria"
    }
  ],
  "use_when": [
    "An agent is told to make a design, report or interface 'better' and keeps polishing without a stable target."
  ],
  "avoid_when": [
    "Some quality remains irreducibly judgmental; a rubric makes it discussable but does not turn taste into objective truth."
  ],
  "example": "For a dashboard, grade information hierarchy, task completion and responsive behavior separately instead of asking a judge model whether it 'looks professional.'",
  "check": "A reviewer can explain why the evaluator passed or failed the artifact without appealing to its confidence.",
  "steps": [
    "A reviewer can explain why the evaluator passed or failed the artifact without appealing to its confidence."
  ],
  "sources": [
    {
      "id": "RS-4CF3167EB387E6BE",
      "title": "Harness design for long-running application development",
      "url": "https://www.anthropic.com/engineering/harness-design-long-running-apps"
    }
  ],
  "relations": [
    {
      "from": "MHC-D-RESEARCH-0714",
      "to": "MHC-D-RESEARCH-0715",
      "type": "use_before",
      "url": "/knowledge/add-planner-builder-evaluator-roles-only-after-the-simple-loop-hits-a-ceiling"
    }
  ],
  "collections": [
    {
      "id": "RC-A76353E5E9A33FE5",
      "title": "Make coding agents work in inspectable increments",
      "url": "/collections/make-coding-agents-work-in-inspectable-increments"
    }
  ]
}
