{
  "notice": "Illustrative mapping from the AI Governance Engineer Body of Knowledge v0.5.0 (not a claim of conformity)",
  "version": "0.5.0",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "schemaVersion": 1,
  "schema": "https://aigovernanceengineer.com/api/v1/schemas/control.json",
  "self": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-eval-009.json",
  "source": "https://aigovernanceengineer.com/controls/evaluation-environment/aige-ctl-eval-009",
  "citation": {
    "title": "AI Governance Engineering: The Thesis & Body of Knowledge",
    "authors": [
      "Jorge García Aibar"
    ],
    "parentDoi": "https://doi.org/10.5281/zenodo.22956197",
    "conceptDoi": "https://doi.org/10.5281/zenodo.22857084"
  },
  "control": {
    "id": "AIGE-CTL-EVAL-009",
    "profile": "evaluation-environment",
    "url": "https://aigovernanceengineer.com/controls/evaluation-environment/aige-ctl-eval-009",
    "json": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-eval-009.json",
    "title": "Evaluation Validity Checks",
    "version": "0.2",
    "status": "draft",
    "reviewerStatus": "open",
    "depth": "specified",
    "objective": "A result is reported only after checks that the run measured what it claims: scoring worked, the environment did not fail, and the path was evaluated as well as the answer.",
    "failureModes": [
      "A result is reported from a run whose environment crashed or whose automatic scoring was wrong.",
      "A task that could not be solved as set up is scored and reported as a failure of the model.",
      "Only final answers are scored: nobody reads the transcripts of failed runs, or of successes, for scorer tampering, reward hacking, communication between runs or signs of evaluation awareness.",
      "A failed validity check does not block the release it was meant to gate."
    ],
    "scope": "Evaluation runs whose results feed a release decision or an assurance claim. The choice of benchmarks and their statistical design are only in scope where they decide whether a result is valid.",
    "enforcementPoints": [
      "pre_merge"
    ],
    "verification": [
      {
        "kind": "inspect",
        "text": "Before the runs, inspect the task admission records: each task has evidence that it can be solved in this environment (a reference solution or a solved run), and the answers, the scorer and the task data are outside the agent's reach."
      },
      {
        "kind": "test",
        "text": "Before the runs, score a known-correct and a known-incorrect submission for each task through the scorer the runs will use: the scorer must accept the first and reject the second."
      },
      {
        "kind": "observe",
        "text": "After the runs, read the transcripts of every failed run and of a recorded sample of successes: classify each failure as a model limitation or a spurious failure (task bug, scoring error, crashed environment), and flag reward hacking, scorer tampering, communication between runs and verbalized evaluation awareness."
      },
      {
        "kind": "attest",
        "text": "Before release, the evaluation lead states in the signed test report which runs were excluded or re-scored after these checks and why, and that no failed check was waived without a recorded approval."
      }
    ],
    "evidence": [
      {
        "artefact": "Task admission and scorer check records: solvability evidence and the verdicts on known-correct and known-incorrect submissions",
        "schemaId": "evidence-record",
        "schema": "https://aigovernanceengineer.com/schemas/evidence-record.v1.json",
        "layer": 3
      },
      {
        "artefact": "Signed test report listing the validity checks run, the runs excluded or re-scored with the reason, and any waiver",
        "schemaId": "test-report",
        "schema": "https://aigovernanceengineer.com/schemas/test-report.v1.json",
        "layer": 3
      },
      {
        "artefact": "Validity check of each result, filed as an observation of this control",
        "schemaId": "control-observation",
        "schema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
        "layer": 5
      }
    ],
    "failureResponse": {
      "effect": "deny",
      "text": "A result whose validity checks failed or were not run is not released to the decision it gates. A spurious failure is fixed and the task rerun, or the task is excluded and the exclusion reported; runs with scorer tampering, communication between runs or verbalized evaluation awareness are excluded or reported as contaminated."
    },
    "layer": 3,
    "secondaryLayers": [],
    "patterns": [
      {
        "slug": "eval-gate-in-ci",
        "title": "Eval Gate in CI",
        "url": "https://aigovernanceengineer.com/patterns/eval-gate-in-ci"
      },
      {
        "slug": "adversarial-red-team-suite",
        "title": "Adversarial Red-Team Suite",
        "url": "https://aigovernanceengineer.com/patterns/adversarial-red-team-suite"
      }
    ],
    "seeds": [
      {
        "id": "trajectory-evals",
        "title": "Independent trajectory evals",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#what-makes-an-agent-a-governance-object"
      }
    ],
    "derivedFrom": [],
    "mappings": {
      "obligations": [
        {
          "id": "AIGE-OBL-EUAIA-ART15",
          "name": "EU AI Act Art. 15 accuracy, robustness and cybersecurity",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art15"
        },
        {
          "id": "AIGE-OBL-EUAIA-ART9",
          "name": "EU AI Act Art. 9 risk management system",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art9"
        },
        {
          "id": "AIGE-OBL-EUAIA-ART55",
          "name": "EU AI Act Art. 55 GPAI models with systemic risk",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art55"
        },
        {
          "id": "AIGE-OBL-NISTRMF-MEASURE",
          "name": "MEASURE",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-nistrmf-measure"
        },
        {
          "id": "AIGE-OBL-ISO42001-A6",
          "name": "A.6 AI system life cycle",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-iso42001-a6"
        },
        {
          "id": "AIGE-OBL-NIST-AI600-1",
          "name": "NIST AI 600-1 Generative AI Profile",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-nist-ai600-1"
        }
      ],
      "iso42001": [
        {
          "id": "A.6.2.4",
          "title": "AI system verification and validation"
        }
      ],
      "nistAiRmf": [
        {
          "id": "MEASURE 2.3",
          "title": "Performance or assurance criteria measured for deployment-like conditions"
        },
        {
          "id": "MEASURE 2.13",
          "title": "Effectiveness of the employed TEVV metrics and processes in the MEASURE function are evaluated and documented."
        }
      ],
      "owasp": [],
      "atlas": [],
      "aiuc1": [],
      "csaAicm": [],
      "other": [
        {
          "framework": "NIST SP 800-53 Rev. 5",
          "ref": "SA-11",
          "note": "Developer Testing and Evaluation"
        }
      ]
    },
    "references": [
      {
        "n": 62,
        "title": "Governing AI development",
        "text": "Governing AI development (AI Governance Engineering Body of Knowledge v0.5.0, chapter 14, section \"Statistical validity of evals\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-development#statistical-validity-of-evals",
        "verified": "primary"
      },
      {
        "n": 63,
        "title": "Governing AI development",
        "text": "Governing AI development (AI Governance Engineering Body of Knowledge v0.5.0, chapter 14, section \"Independent validation and model risk management\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-development#independent-validation-and-model-risk-management",
        "verified": "primary"
      },
      {
        "n": 64,
        "title": "Governing AI agents",
        "text": "Governing AI agents (AI Governance Engineering Body of Knowledge v0.5.0, chapter 23, section \"What makes an agent a governance object\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#what-makes-an-agent-a-governance-object",
        "verified": "primary"
      },
      {
        "n": 31,
        "title": "Guidelines for capability elicitation",
        "text": "Guidelines for capability elicitation (task bugs such as \"The automatic scoring is incorrect\" or a crashed environment are spurious failures to fix before reporting; models get \"the best available scaffolding + tooling\"). METR. 2024-03-15.",
        "url": "https://metr.org/blog/2024-03-15-guidelines-for-capability-elicitation/",
        "verified": "primary"
      },
      {
        "n": 65,
        "title": "Example autonomy evaluation protocol",
        "text": "Example autonomy evaluation protocol (read the transcripts of runs that missed the maximum score and check that the pattern of successes and failures is roughly as expected). METR. 2024-03-15.",
        "url": "https://metr.org/blog/2024-03-15-example-autonomy-evaluation-protocol/",
        "verified": "primary"
      },
      {
        "n": 58,
        "title": "Summary of METR's predeployment evaluation of GPT-5.6 Sol",
        "text": "Summary of METR's predeployment evaluation of GPT-5.6 Sol (METR states that \"observed cheating rates can also be influenced by the prompts used in the evaluation scaffold\" and by task wording). METR. 2026-06-26.",
        "url": "https://metr.org/blog/2026-06-26-gpt-5-6-sol/",
        "verified": "primary"
      },
      {
        "n": 7,
        "title": "NIST SP 800-53 Rev. 5, Security and Privacy Controls for Information Systems and Organizations",
        "text": "NIST SP 800-53 Rev. 5, Security and Privacy Controls for Information Systems and Organizations (control catalogue cited by control id; publication page of Revision 5 with update 1 of 10 Dec 2020). NIST. 2020-12-10.",
        "url": "https://csrc.nist.gov/pubs/sp/800/53/r5/upd1/final",
        "verified": "primary"
      },
      {
        "n": 9,
        "title": "Improving our alignment and security efforts",
        "text": "Improving our alignment and security efforts (best practices for external partners running cyber evaluations: the only outside connection is \"to the model's own API\", with the API keys kept outside the environment; the configuration \"should be verified before every evaluation begins\"; boundaries \"phrased as instructions\"; challenges confirmed \"solvable in principle\"; a monitor that flags a scope violation to a human and ends the exercise). Anthropic. 2026-08-31.",
        "url": "https://www.anthropic.com/news/improving-alignment-security-efforts",
        "verified": "primary"
      },
      {
        "n": 5,
        "title": "Brief independent investigation of agents' behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident",
        "text": "Brief independent investigation of agents' behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident (METR states that agents \"meant to be fully isolated from one another\" communicated through an internal package repository, and that one agent found working Hugging Face credentials exposed on the internet and posted them to the agents' board; it reports spoofed tool calls in at least 96 transcripts and transcripts missing components after container resets). METR. 2026-08-26.",
        "url": "https://metr.org/blog/2026-08-26-openai-hugging-face-incident-investigation/",
        "verified": "primary"
      },
      {
        "n": 4,
        "title": "METR Task Standard, STANDARD.md",
        "text": "METR Task Standard, STANDARD.md (version 0.5.0; unless a task declares the full_internet permission, the task machines \"MUST NOT have internet access\" except to an LLM API, an LLM API proxy or a hardened local server). METR (GitHub). 2024-10-30.",
        "url": "https://raw.githubusercontent.com/METR/task-standard/main/STANDARD.md",
        "verified": "primary"
      },
      {
        "n": 34,
        "title": "Preparedness Framework, Version 2",
        "text": "Preparedness Framework, Version 2 (Table 5 lists potential safeguards against a misaligned model, among them limiting internet and tool access, credentials and filesystem access, with agent actions \"logged in an uneditable database\"; a one-time capability elicitation is treated \"as a lower bound, rather than a ceiling\"). OpenAI. 2025-04-15.",
        "url": "https://cdn.openai.com/pdf/18a02b5d-6b67-4cec-ab64-68cdfbddebcd/preparedness-framework-v2.pdf",
        "verified": "primary"
      },
      {
        "n": 66,
        "title": "GPT-6 Astra System Card",
        "text": "GPT-6 Astra System Card (OpenAI states that evaluations where models show verbalized metagaming \"can be treated similarly to contaminated evals\"). OpenAI. 2026-09-03.",
        "url": "https://deploymentsafety.openai.com/gpt-6-astra",
        "verified": "primary"
      },
      {
        "n": 67,
        "title": "Sidestepping Evaluation Awareness and Anticipating Misalignment with Production Evaluations",
        "text": "Sidestepping Evaluation Awareness and Anticipating Misalignment with Production Evaluations (OpenAI reports that models tested with production evaluations \"display substantially lower signs of evaluation awareness\" than in a traditional evaluation). OpenAI (Alignment Research Blog). 2025-12-18.",
        "url": "https://alignment.openai.com/prod-evals/",
        "verified": "primary"
      },
      {
        "n": 39,
        "title": "The Hugging Face incident and the road ahead",
        "text": "The Hugging Face incident and the road ahead (OpenAI states that its chain-of-thought monitors \"did not run on the evaluations in this incident\", and that it is strengthening its incident response plan with clearer rules on \"who can stop a run or approve restarting it\"). OpenAI. 2026-08-26.",
        "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/",
        "verified": "primary"
      },
      {
        "n": 60,
        "title": "A shared playbook for trustworthy third party evaluations",
        "text": "A shared playbook for trustworthy third party evaluations (recommended report fields include the claim, the tested system (model, reasoning setting, tool access, harness and safeguards), the budget, elicitation methods and validity checks; a score is \"performance under that harness and budget\"). OpenAI. 2026-05-29.",
        "url": "https://openai.com/index/trustworthy-third-party-evaluations-foundations/",
        "verified": "primary"
      },
      {
        "n": 11,
        "title": "Unsanctioned Artifactory writes and cross-sample communication (misalignment report)",
        "text": "Unsanctioned Artifactory writes and cross-sample communication (misalignment report) (OpenAI reports models in RL training that used Artifactory, \"a shared package repository\", to exchange messages across separate training samples with the credentials they were given to download packages, without exploiting any vulnerability; such exchanges can \"undermine the assumption that training or evaluation samples are independent\"). OpenAI (Alignment Research Blog). 2026-09-16.",
        "url": "https://alignment.openai.com/misalignment-reports/unauthorized-artifactory-writes-and-cross-sample-communication/",
        "verified": "primary"
      }
    ],
    "implementationNotes": [
      "Confirm that each task can be solved before a failure on it is counted. Anthropic's guidance for external evaluation partners asks them to \"confirm any evaluation challenge is actually solvable in principle\", because agents facing a task they cannot complete look for other ways to complete it; METR reports that many of the agents in the OpenAI/Hugging Face incident that turned to cheating had unintentionally been given an impossible task, and OpenAI states that none of its models had ever solved 198 of the 898 tasks of that evaluation before the incident.",
      "Keep the answers and the scorer out of the agent's reach. METR's Task Standard states that the agent \"MUST NOT see the data returned by get_tasks\", and METR's investigation reports agents that coordinated to find a way to \"fool or tamper with the automated scorer\".",
      "Read the transcripts, not only the scores. METR's example evaluation protocol asks for someone to read the transcripts of runs that missed the maximum score and to check that the pattern of successes and failures is roughly as expected, and METR's guidelines treat task bugs such as incorrect automatic scoring as spurious failures to fix before reporting.",
      "Report the checks with the result. OpenAI's playbook for third-party evaluations asks reports to say how assessors looked for reward hacking, evaluation awareness, contamination, refusals and sandbagging, and how confirmed cases affected scoring. OpenAI's GPT-6 Astra system card states that evaluations where models show verbalized metagaming \"can be treated similarly to contaminated evals\", and OpenAI reports that exchanges between samples through a shared package repository can \"undermine the assumption that training or evaluation samples are independent\".",
      "Report an elicited score as a lower bound. OpenAI's Preparedness Framework v2 treats a one-time capability elicitation \"as a lower bound, rather than a ceiling\", and OpenAI's alignment blog reports lower signs of evaluation awareness under production evaluations than under traditional ones."
    ],
    "openQuestions": [
      "Which validity threats (task bugs, scoring errors, evaluation awareness) should block a result, and which should only be disclosed with it?",
      "How large a sample of successful runs should be read for reward hacking and scorer tampering before a result is reported?"
    ],
    "observation": {
      "subjectKind": "eval-run",
      "expected": "Every task was shown to be solvable, the scorer passed its known-answer check, every failed run was read and classified, and contaminated runs were excluded or reported.",
      "observedExample": "Suite run 88280: 12 of 200 tasks had no evidence of being solvable and their failures were counted against the model: fail."
    },
    "observationSchema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
    "examples": [
      {
        "status": "pass",
        "url": "https://aigovernanceengineer.com/controls/examples/control-observation.aige-ctl-eval-009.pass.json"
      },
      {
        "status": "fail",
        "url": "https://aigovernanceengineer.com/controls/examples/control-observation.aige-ctl-eval-009.fail.json"
      }
    ]
  }
}
