{
  "schema": "vedokrok.public-item.v1",
  "release_id": "MHC-RPUB-20260920-75ad787a",
  "url": "/knowledge/monitor-the-agent-after-deployment",
  "id": "MHC-D-RESEARCH-0321",
  "version": "0.1.0",
  "title": "Monitor the agent after deployment",
  "summary": "Deployment is the first time the model meets all the mess your test set forgot.",
  "kind": "protocol",
  "body": "Define a small set of post-deployment signals tied to real failure modes: invalid actions, corrections, refusals, escalation, unexpected tool use, latency or other task-specific outcomes. Review incidents and field behavior for conditions not represented in predeployment tests. Feed verified failures back into controls and evaluation rather than treating monitoring as a dashboard decoration.",
  "limits": [
    "NIST notes that AI monitoring practice is still developing; a metric set should not be presented as a complete safety methodology."
  ],
  "topics": [
    "union-ai-agent-control"
  ],
  "intents": [],
  "source_ids": [
    "RS-06C6740A8538704F"
  ],
  "evidence": [
    {
      "claim": "NIST recommends production monitoring of AI behavior, and its 2026 monitoring report explains why controlled pre-deployment evaluations cannot capture all real-world variability and unexpected consequences.",
      "source_id": "RS-06C6740A8538704F",
      "role": "supports",
      "note": "The 2026 report explicitly notes that monitoring methods and terminology remain nascent and scattered.",
      "locator": "Abstract"
    }
  ],
  "use_when": [
    "An AI workflow is useful enough to run repeatedly in real conditions."
  ],
  "avoid_when": [
    "NIST notes that AI monitoring practice is still developing; a metric set should not be presented as a complete safety methodology."
  ],
  "example": "Track rejected tool calls and manual corrections after an agent goes live; cluster repeated causes and add representative failures to the eval set.",
  "check": "A real-world failure has a route from detection to investigation and, when warranted, to a changed control or evaluation.",
  "steps": [
    "Choose signals linked to concrete failure or harm modes.",
    "Capture enough context to investigate without collecting unnecessary sensitive data.",
    "Review unexpected behavior and user corrections on a defined cadence or trigger.",
    "Turn confirmed new failure modes into a control, test case or product decision."
  ],
  "sources": [
    {
      "id": "RS-06C6740A8538704F",
      "title": "Challenges to the monitoring of deployed AI systems: Center for AI Standards and Innovation",
      "url": "https://www.nist.gov/publications/challenges-monitoring-deployed-ai-systems-center-ai-standards-and-innovation"
    }
  ],
  "relations": [
    {
      "from": "MHC-D-RESEARCH-0321",
      "to": "MHC-D-RESEARCH-0319",
      "type": "useful_with",
      "url": "/knowledge/keep-an-eval-set-that-can-embarrass-the-agent"
    }
  ],
  "collections": [
    {
      "id": "RC-EEF5F4530C11FD17",
      "title": "Let AI do useful work without giving it accidental authority",
      "url": "/collections/let-ai-do-useful-work-without-giving-it-accidental-authority"
    }
  ]
}
