{
  "notice": "Illustrative mapping from the AI Governance Engineer Body of Knowledge v0.5.0 (not a claim of conformity)",
  "version": "0.5.0",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "schemaVersion": 1,
  "schema": "https://aigovernanceengineer.com/api/v1/schemas/control.json",
  "self": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-eval-007.json",
  "source": "https://aigovernanceengineer.com/controls/evaluation-environment/aige-ctl-eval-007",
  "citation": {
    "title": "AI Governance Engineering: The Thesis & Body of Knowledge",
    "authors": [
      "Jorge García Aibar"
    ],
    "parentDoi": "https://doi.org/10.5281/zenodo.22956197",
    "conceptDoi": "https://doi.org/10.5281/zenodo.22857084"
  },
  "control": {
    "id": "AIGE-CTL-EVAL-007",
    "profile": "evaluation-environment",
    "url": "https://aigovernanceengineer.com/controls/evaluation-environment/aige-ctl-eval-007",
    "json": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-eval-007.json",
    "title": "Incident Evidence Preservation",
    "version": "0.2",
    "status": "draft",
    "reviewerStatus": "open",
    "depth": "specified",
    "objective": "When a run produces an incident, its traces, configuration and outputs are frozen before anything is fixed, so the record can be reviewed as it was.",
    "failureModes": [
      "Records of a run are changed or deleted after an incident was declared.",
      "The environment is reset before its state and traces were captured.",
      "Part of a run's transcript is lost when a container is reset, and the gap is not recorded.",
      "An incident record does not link to the run it came from or to the hashes of the frozen records."
    ],
    "scope": "Evaluation runs that produce an incident or a result disputed after the fact, and the records they leave. Reporting to authorities follows the incident process of chapter 17.",
    "enforcementPoints": [
      "runtime",
      "periodic"
    ],
    "verification": [
      {
        "kind": "inspect",
        "text": "Inspect the evidence store and the harness configuration: the transcripts, traces, configuration and outputs of every run are written as they are produced to write-once storage outside the environment, with a retention period recorded, and the harness snapshots the environment before any reset."
      },
      {
        "kind": "test",
        "text": "Drill the freeze on a schedule: declare a test incident on a live run, then check that the environment snapshot, trace, transcript and configuration were captured with their hashes before the environment was reset, and that an attempt to delete or overwrite them is refused."
      },
      {
        "kind": "observe",
        "text": "For each real incident, read the incident record: it names the run, lists every frozen artefact with its hash, the hashes still match the stored artefacts, and every gap in the transcript is recorded with its cause."
      }
    ],
    "evidence": [
      {
        "artefact": "Incident record naming the run and listing the frozen artefacts in its supporting materials",
        "schemaId": "incident-record",
        "schema": "https://aigovernanceengineer.com/schemas/incident-record.v1.json",
        "layer": 5
      },
      {
        "artefact": "Freeze record: hashes of the environment snapshot, trace, transcript and configuration, with the time of capture and the actor",
        "schemaId": "evidence-record",
        "schema": "https://aigovernanceengineer.com/schemas/evidence-record.v1.json",
        "layer": 5
      },
      {
        "artefact": "Freeze drill and hash check, filed as an observation of this control",
        "schemaId": "control-observation",
        "schema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
        "layer": 5
      }
    ],
    "failureResponse": {
      "effect": "alert",
      "text": "A missing snapshot, a hash mismatch or an unrecorded gap alerts the incident owner and is entered in the incident record. Until the freeze is complete the environment is not reset or reused, and the fix is made on a new version, not in place."
    },
    "layer": 5,
    "secondaryLayers": [],
    "patterns": [
      {
        "slug": "incident-pipeline",
        "title": "Incident Pipeline",
        "url": "https://aigovernanceengineer.com/patterns/incident-pipeline"
      },
      {
        "slug": "machine-readable-evidence-oscal",
        "title": "Machine-Readable Evidence (OSCAL)",
        "url": "https://aigovernanceengineer.com/patterns/machine-readable-evidence-oscal"
      }
    ],
    "seeds": [
      {
        "id": "traces",
        "title": "Traces",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#telemetry-with-the-opentelemetry-genai-conventions"
      },
      {
        "id": "otel-telemetry",
        "title": "Telemetry on the OpenTelemetry GenAI conventions",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#telemetry-with-the-opentelemetry-genai-conventions"
      },
      {
        "id": "ai-act-high-risk",
        "title": "EU AI Act hooks for a high-risk purpose",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#eu-ai-act-hooks-for-agents"
      }
    ],
    "derivedFrom": [],
    "mappings": {
      "obligations": [
        {
          "id": "AIGE-OBL-EUAIA-ART73",
          "name": "EU AI Act Art. 73 serious-incident reporting",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art73"
        },
        {
          "id": "AIGE-OBL-EUAIA-ART12",
          "name": "EU AI Act Art. 12 record-keeping and logging",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art12"
        },
        {
          "id": "AIGE-OBL-EUAIA-ART26-6",
          "name": "EU AI Act Art. 26(6) deployer retention of automatically generated logs",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art26-6"
        },
        {
          "id": "AIGE-OBL-GPAICOP-SAFETY-C9",
          "name": "Safety and Security Commitment 9: serious-incident reporting",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-gpaicop-safety-c9"
        },
        {
          "id": "AIGE-OBL-ISO42001-A8",
          "name": "A.8 Information for interested parties",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-iso42001-a8"
        }
      ],
      "iso42001": [
        {
          "id": "A.8.4",
          "title": "Communication of incidents"
        },
        {
          "id": "A.6.2.8",
          "title": "AI system recording of event logs"
        }
      ],
      "nistAiRmf": [
        {
          "id": "MANAGE 4.3",
          "title": "Incidents and errors are communicated to relevant AI actors, including affected communities."
        }
      ],
      "owasp": [],
      "atlas": [],
      "aiuc1": [
        "E015"
      ],
      "csaAicm": [],
      "other": [
        {
          "framework": "NIST SP 800-53 Rev. 5",
          "ref": "IR-4",
          "note": "Incident Handling"
        },
        {
          "framework": "NIST SP 800-53 Rev. 5",
          "ref": "AU-9",
          "note": "Protection of Audit Information"
        }
      ]
    },
    "references": [
      {
        "n": 51,
        "title": "Incidents, issues and root causes",
        "text": "Incidents, issues and root causes (AI Governance Engineering Body of Knowledge v0.5.0, chapter 17, section \"Freeze before you fix\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/incidents#freeze-before-you-fix",
        "verified": "primary"
      },
      {
        "n": 52,
        "title": "Incidents, issues and root causes",
        "text": "Incidents, issues and root causes (AI Governance Engineering Body of Knowledge v0.5.0, chapter 17, section \"The incident record\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/incidents#the-incident-record",
        "verified": "primary"
      },
      {
        "n": 53,
        "title": "Incidents, issues and root causes",
        "text": "Incidents, issues and root causes (AI Governance Engineering Body of Knowledge v0.5.0, chapter 17, section \"The overlapping clocks\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/incidents#the-overlapping-clocks",
        "verified": "primary"
      },
      {
        "n": 5,
        "title": "Brief independent investigation of agents' behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident",
        "text": "Brief independent investigation of agents' behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident (METR states that agents \"meant to be fully isolated from one another\" communicated through an internal package repository, and that one agent found working Hugging Face credentials exposed on the internet and posted them to the agents' board; it reports spoofed tool calls in at least 96 transcripts and transcripts missing components after container resets). METR. 2026-08-26.",
        "url": "https://metr.org/blog/2026-08-26-openai-hugging-face-incident-investigation/",
        "verified": "primary"
      },
      {
        "n": 31,
        "title": "Guidelines for capability elicitation",
        "text": "Guidelines for capability elicitation (task bugs such as \"The automatic scoring is incorrect\" or a crashed environment are spurious failures to fix before reporting; models get \"the best available scaffolding + tooling\"). METR. 2024-03-15.",
        "url": "https://metr.org/blog/2024-03-15-guidelines-for-capability-elicitation/",
        "verified": "primary"
      },
      {
        "n": 7,
        "title": "NIST SP 800-53 Rev. 5, Security and Privacy Controls for Information Systems and Organizations",
        "text": "NIST SP 800-53 Rev. 5, Security and Privacy Controls for Information Systems and Organizations (control catalogue cited by control id; publication page of Revision 5 with update 1 of 10 Dec 2020). NIST. 2020-12-10.",
        "url": "https://csrc.nist.gov/pubs/sp/800/53/r5/upd1/final",
        "verified": "primary"
      },
      {
        "n": 54,
        "title": "Our framework for reporting model misalignment",
        "text": "Our framework for reporting model misalignment (each full report describes the behavior observed, its severity and any external impact, the setting, the date, when it was discovered and the models involved). OpenAI. 2026-09-16.",
        "url": "https://openai.com/index/model-misalignment-reporting-framework/",
        "verified": "primary"
      },
      {
        "n": 55,
        "title": "Frontier Governance Framework",
        "text": "Frontier Governance Framework (section 2.6: OpenAI \"maintains an AI Safety Incident Response Plan (AIRP)\" for identifying and responding to AI safety incidents). OpenAI. 2026-05-28.",
        "url": "https://cdn.openai.com/pdf/e37d949b-8c9f-4d76-b99e-4272f4631a7e/openai-frontier-governance-framework.pdf",
        "verified": "primary"
      },
      {
        "n": 10,
        "title": "AIUC-1 requirements",
        "text": "AIUC-1 requirements (public requirement index, A001 to F002, each requirement on its own page (E007 and E014 marked retired); AIUC-1 is a standard of the Artificial Intelligence Underwriting Company; this site is not affiliated with AIUC, and a mapping here is not an AIUC-1 certificate or audit). Artificial Intelligence Underwriting Company. 2026-09-24.",
        "url": "https://standard.aiuc-1.com/llms.txt",
        "verified": "primary"
      },
      {
        "n": 12,
        "title": "OpenAI Hugging Face Incident Technical Report",
        "text": "OpenAI Hugging Face Incident Technical Report (OpenAI states that high-risk workloads are \"prohibited via technical controls from receiving direct or transitive Internet access\", protected by \"two separate layers of network isolation\", and that it is building tooling to \"identify and halt evaluation workloads\"). OpenAI. 2026-08-26.",
        "url": "https://cdn.openai.com/pdf/67869394-cb91-4c12-888c-5cbd85c7814c/OpenAI-Hugging-Face%20Incident-Technical-Report.pdf",
        "verified": "primary"
      }
    ],
    "implementationNotes": [
      "Freeze before you fix, as chapter 17 sets out: snapshot the system as it was (model version and hash, system prompt, policy and tool scopes, configuration), seal the traces of the incident window in tamper-evident storage, record who touched what, and fix on a new version, not in place. OpenAI's technical report on the Hugging Face incident states that it took the compromised Artifactory deployment offline, \"preserved its storage\", and revoked agent credentials.",
      "Write transcripts and traces to write-once storage as they are produced, not at the end of the run, and have the harness snapshot the environment before any reset. METR's investigation of the OpenAI/Hugging Face incident reports that many transcripts were \"missing small components, often due to container resets\".",
      "Link the incident record to the run: the run id, the frozen artefacts and their hashes go in the record's supporting materials, so a reviewer can find the run as it was. OpenAI's framework for reporting model misalignment states that each full report describes the behavior, its severity and any external impact, the setting, the date, when it was discovered and the models involved, and its Frontier Governance Framework refers to an AI Safety Incident Response Plan; a frozen run record gives such a report something to point to.",
      "Keep what an independent reviewer will need. METR states that OpenAI shared \"over a thousand unredacted transcripts\" for its investigation of the Hugging Face incident: a review of that kind depends on the transcripts having been kept whole."
    ],
    "openQuestions": [
      "How long should the records of an evaluation run be kept when the run produced no incident?",
      "Which parts of a frozen run record can be shared with an independent reviewer without exposing the task set, and in what format?"
    ],
    "observation": {
      "subjectKind": "eval-run",
      "expected": "After an incident is declared, the environment snapshot, trace, transcript and configuration of the run are frozen with their hashes before any reset, the incident record links them, and every transcript gap is recorded.",
      "observedExample": "Incident on run 88262: snapshot and trace frozen before the reset, but 14 minutes of transcript lost in a container reset with no gap recorded: fail."
    },
    "observationSchema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
    "examples": [
      {
        "status": "pass",
        "url": "https://aigovernanceengineer.com/controls/examples/control-observation.aige-ctl-eval-007.pass.json"
      },
      {
        "status": "fail",
        "url": "https://aigovernanceengineer.com/controls/examples/control-observation.aige-ctl-eval-007.fail.json"
      }
    ]
  }
}
