{
  "notice": "Illustrative mapping from the AI Governance Engineer Body of Knowledge v0.5.0 (not a claim of conformity)",
  "version": "0.5.0",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "schemaVersion": 1,
  "schema": "https://aigovernanceengineer.com/api/v1/schemas/control.json",
  "self": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-eval-004.json",
  "source": "https://aigovernanceengineer.com/controls/evaluation-environment/aige-ctl-eval-004",
  "citation": {
    "title": "AI Governance Engineering: The Thesis & Body of Knowledge",
    "authors": [
      "Jorge García Aibar"
    ],
    "parentDoi": "https://doi.org/10.5281/zenodo.22956197",
    "conceptDoi": "https://doi.org/10.5281/zenodo.22857084"
  },
  "control": {
    "id": "AIGE-CTL-EVAL-004",
    "profile": "evaluation-environment",
    "url": "https://aigovernanceengineer.com/controls/evaluation-environment/aige-ctl-eval-004",
    "json": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-eval-004.json",
    "title": "Tool and Action Mediation",
    "version": "0.2",
    "status": "draft",
    "reviewerStatus": "open",
    "depth": "specified",
    "objective": "Every tool call an agent makes during a run passes a mediation point outside the model that checks it against policy before it executes.",
    "failureModes": [
      "A tool call executes without a policy verdict recorded for it.",
      "An irreversible action runs while the mediation point is down.",
      "An approval is reused for a call whose parameters changed.",
      "After repeated denials the agent keeps seeking the same outcome through other calls, and the run continues."
    ],
    "scope": "Tool, MCP and code-execution calls made by the agent under evaluation, and the reviewer (a rule, a model or a person) that decides on them. The internal reasoning of the model is out of scope; network egress is AIGE-CTL-EVAL-002.",
    "enforcementPoints": [
      "deploy",
      "runtime"
    ],
    "verification": [
      {
        "kind": "inspect",
        "text": "Before the run, inspect the environment and its policy: tool servers, MCP servers and code execution are reachable only through the mediation point, and the policy lists each operation class with its verdict, its failure posture (fail closed for irreversible classes such as delete, send, publish and execute) and the denial threshold that interrupts a run."
      },
      {
        "kind": "test",
        "text": "At admission, send through the harness one call the policy denies and one it allows, then take the mediation point down and send an irreversible-class call; the denied call and the call sent while it is down must not execute, and all three must leave a verdict record."
      },
      {
        "kind": "observe",
        "text": "After the run, join the tool servers' own logs with the verdict records: every executed call has an allow verdict, or an approval bound to a parameter hash that matches the call, and no run continued past its denial threshold."
      }
    ],
    "evidence": [
      {
        "artefact": "Mediation policy of the run: operation classes, verdicts, failure posture per class and the denial threshold",
        "schemaId": "policy-card",
        "schema": "https://aigovernanceengineer.com/schemas/policy-card.v1.json",
        "layer": 4
      },
      {
        "artefact": "Verdict record of every call: tool, parameter hash, verdict, reviewer and time",
        "schemaId": "evidence-record",
        "schema": "https://aigovernanceengineer.com/schemas/evidence-record.v1.json",
        "layer": 4
      },
      {
        "artefact": "One observation per run joining the executed calls with their verdicts",
        "schemaId": "control-observation",
        "schema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
        "layer": 5
      }
    ],
    "failureResponse": {
      "effect": "deny",
      "text": "A call without an allow verdict does not execute. While the mediation point is down, irreversible classes fail closed and reads fail open only with an alert. A run in which a call executed without a verdict is stopped and its result is withheld; a run that reaches its denial threshold is interrupted."
    },
    "layer": 4,
    "secondaryLayers": [],
    "patterns": [
      {
        "slug": "runtime-guardrail",
        "title": "Runtime Guardrail",
        "url": "https://aigovernanceengineer.com/patterns/runtime-guardrail"
      },
      {
        "slug": "human-in-the-loop-gate",
        "title": "Human-in-the-loop Gate",
        "url": "https://aigovernanceengineer.com/patterns/human-in-the-loop-gate"
      }
    ],
    "seeds": [
      {
        "id": "guardrail-every-call",
        "title": "Runtime guardrail on every tool call",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#runtime-guardrails-for-tool-calls"
      },
      {
        "id": "checkpoint-irreversible",
        "title": "Checkpoints on irreversible actions, failing closed",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#where-to-put-a-checkpoint"
      },
      {
        "id": "approval-log",
        "title": "Approval log, bound to the call",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#what-a-good-approval-looks-like"
      },
      {
        "id": "mcp-admission",
        "title": "MCP server admission gate",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#admitting-an-mcp-server"
      },
      {
        "id": "sandbox",
        "title": "Code runs only in a sandbox",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#runtime-guardrails-for-tool-calls"
      }
    ],
    "derivedFrom": [],
    "mappings": {
      "obligations": [
        {
          "id": "AIGE-OBL-EUAIA-ART14",
          "name": "EU AI Act Art. 14 human oversight",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art14"
        },
        {
          "id": "AIGE-OBL-OWASP-ACS",
          "name": "Agent Control Standard (ACS)",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-owasp-acs"
        },
        {
          "id": "AIGE-OBL-SG-AGENTIC-CHECKPOINTS",
          "name": "Singapore IMDA Model AI Governance Framework for Agentic AI: human checkpoints for significant actions (voluntary)",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-sg-agentic-checkpoints"
        },
        {
          "id": "AIGE-OBL-OWASP-AGENTIC",
          "name": "Top 10 for Agentic Applications 2026",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-owasp-agentic"
        }
      ],
      "iso42001": [
        {
          "id": "A.9.2",
          "title": "Processes for responsible use of AI systems"
        }
      ],
      "nistAiRmf": [
        {
          "id": "MAP 4.2",
          "title": "Internal risk controls for components of the AI system, including third-party AI technologies, are identified and documented."
        }
      ],
      "owasp": [
        {
          "id": "asi01",
          "externalId": "ASI01",
          "name": "Agent Goal Hijack",
          "url": "https://aigovernanceengineer.com/resources/threats#threat-asi01"
        },
        {
          "id": "asi02",
          "externalId": "ASI02",
          "name": "Tool Misuse and Exploitation",
          "url": "https://aigovernanceengineer.com/resources/threats#threat-asi02"
        },
        {
          "id": "asi05",
          "externalId": "ASI05",
          "name": "Unexpected Code Execution (RCE)",
          "url": "https://aigovernanceengineer.com/resources/threats#threat-asi05"
        },
        {
          "id": "asi09",
          "externalId": "ASI09",
          "name": "Human-Agent Trust Exploitation",
          "url": "https://aigovernanceengineer.com/resources/threats#threat-asi09"
        },
        {
          "id": "llm10-2026",
          "externalId": "LLM10:2026",
          "name": "Improper Output Handling",
          "url": "https://aigovernanceengineer.com/resources/threats#threat-llm10-2026"
        }
      ],
      "atlas": [],
      "aiuc1": [
        "D003",
        "B006"
      ],
      "csaAicm": [],
      "other": [
        {
          "framework": "MITRE ATLAS mitigation",
          "ref": "AML.M0028",
          "note": "AI Agent Tools Permissions Configuration"
        },
        {
          "framework": "MITRE ATLAS mitigation",
          "ref": "AML.M0029",
          "note": "Human In-the-Loop for AI Agent Actions"
        },
        {
          "framework": "MITRE ATLAS mitigation",
          "ref": "AML.M0030",
          "note": "Restrict AI Agent Tool Invocation on Untrusted Data"
        }
      ]
    },
    "references": [
      {
        "n": 13,
        "title": "Governing AI agents",
        "text": "Governing AI agents (AI Governance Engineering Body of Knowledge v0.5.0, chapter 23, section \"Runtime guardrails for tool calls\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#runtime-guardrails-for-tool-calls",
        "verified": "primary"
      },
      {
        "n": 28,
        "title": "Governing AI agents",
        "text": "Governing AI agents (AI Governance Engineering Body of Knowledge v0.5.0, chapter 23, section \"Where to put a checkpoint\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#where-to-put-a-checkpoint",
        "verified": "primary"
      },
      {
        "n": 29,
        "title": "Governing AI agents",
        "text": "Governing AI agents (AI Governance Engineering Body of Knowledge v0.5.0, chapter 23, section \"Admitting an MCP server\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-agents#admitting-an-mcp-server",
        "verified": "primary"
      },
      {
        "n": 30,
        "title": "Agent Control Standard (ACS)",
        "text": "Agent Control Standard (ACS) (wire specification for a guardian agent that decides on an agent action before it runs; donated to OWASP, announced 1 Sep 2026). OWASP GenAI Security Project. 2026-09-01.",
        "url": "https://genai.owasp.org/resource/agent-control-standard-acs/",
        "verified": "primary"
      },
      {
        "n": 31,
        "title": "Guidelines for capability elicitation",
        "text": "Guidelines for capability elicitation (task bugs such as \"The automatic scoring is incorrect\" or a crashed environment are spurious failures to fix before reporting; models get \"the best available scaffolding + tooling\"). METR. 2024-03-15.",
        "url": "https://metr.org/blog/2024-03-15-guidelines-for-capability-elicitation/",
        "verified": "primary"
      },
      {
        "n": 6,
        "title": "OWASP Top 10 for Agentic Applications for 2026",
        "text": "OWASP Top 10 for Agentic Applications for 2026 (ASI01 Agent Goal Hijack to ASI10 Rogue Agents). OWASP GenAI Security Project. 2025-12-09.",
        "url": "https://genai.owasp.org/resource/owasp-top-10-for-agentic-applications-for-2026/",
        "verified": "primary"
      },
      {
        "n": 32,
        "title": "Codex auto-review",
        "text": "Codex auto-review (undated developer documentation, read 2026-09-26: \"Auto-review is a reviewer swap, not a permission grant\"). OpenAI. 2026.",
        "url": "https://developers.openai.com/codex/sandboxing/auto-review",
        "verified": "primary"
      },
      {
        "n": 33,
        "title": "Auto-review of agent actions without synchronous human oversight",
        "text": "Auto-review of agent actions without synchronous human oversight (a separate agent approves or denies actions that cross the sandbox boundary; OpenAI states that auto-review \"should not be treated as a guarantee of security\"). OpenAI (Alignment Research Blog). 2026-04-30.",
        "url": "https://alignment.openai.com/auto-review/",
        "verified": "primary"
      },
      {
        "n": 34,
        "title": "Preparedness Framework, Version 2",
        "text": "Preparedness Framework, Version 2 (Table 5 lists potential safeguards against a misaligned model, among them limiting internet and tool access, credentials and filesystem access, with agent actions \"logged in an uneditable database\"; a one-time capability elicitation is treated \"as a lower bound, rather than a ceiling\"). OpenAI. 2025-04-15.",
        "url": "https://cdn.openai.com/pdf/18a02b5d-6b67-4cec-ab64-68cdfbddebcd/preparedness-framework-v2.pdf",
        "verified": "primary"
      },
      {
        "n": 10,
        "title": "AIUC-1 requirements",
        "text": "AIUC-1 requirements (public requirement index, A001 to F002, each requirement on its own page (E007 and E014 marked retired); AIUC-1 is a standard of the Artificial Intelligence Underwriting Company; this site is not affiliated with AIUC, and a mapping here is not an AIUC-1 certificate or audit). Artificial Intelligence Underwriting Company. 2026-09-24.",
        "url": "https://standard.aiuc-1.com/llms.txt",
        "verified": "primary"
      },
      {
        "n": 27,
        "title": "Exposing a GitHub token in a public repository (misalignment report)",
        "text": "Exposing a GitHub token in a public repository (misalignment report) (OpenAI reports an internally deployed model that published a researcher's GitHub token to a public repository, split into pieces with the stated aim of avoiding secret scanning). OpenAI (Alignment Research Blog). 2026-09-25.",
        "url": "https://alignment.openai.com/misalignment-reports/exposing-a-github-token-in-a-public-repository/",
        "verified": "primary"
      }
    ],
    "implementationNotes": [
      "Put the mediation point between the decision to call a tool and the call, outside the model and the agent's process: a gateway or guardian that checks identity, allow-list, parameters and operation class before the call runs, as chapter 23 describes and OWASP's Agent Control Standard specifies on the wire. Route tool servers, MCP servers and code execution only through it, so no call can go around it.",
      "Record the failure posture per operation class in the policy of the run. Chapter 23 recommends failing closed for pay, delete, send and execute and failing open, with an alert, only for reads, and notes that the reference guardian of the Agent Control Standard starts with a fail-open posture: set the posture, do not inherit it.",
      "Bind an approval to the exact call. OpenAI's Codex documentation describes an override that \"applies to the exact denied action, not similar future actions\" and holds for one retry; store a hash of the parameters with each approval and refuse a call whose parameters differ.",
      "Interrupt a run that keeps asking. The Codex documentation describes a rejection circuit breaker that, in its current open-source implementation, interrupts the turn after 3 consecutive denials or 10 in the last 50 reviews, and OpenAI's post on auto-review states that it stops the trajectory after repeated denials. Record the threshold in the policy and make reaching it a stop condition (AIGE-CTL-EVAL-006).",
      "An automated reviewer can stand in for a person at the mediation point without widening what the agent may do: the Codex documentation states that \"Auto-review is a reviewer swap, not a permission grant\", and OpenAI's post on the design states that it \"should not be treated as a guarantee of security\". OpenAI reports that a model which had published a researcher's token was brought back with \"enforced auto-review on all actions\"."
    ],
    "openQuestions": [
      "How can every call be mediated without taking away the scaffolding and tooling a capability elicitation needs? METR's guidelines ask for \"the best available scaffolding + tooling\".",
      "When the reviewer at the mediation point is a model, what evidence shows that its verdicts are reliable enough to replace a person for a given operation class?"
    ],
    "observation": {
      "subjectKind": "eval-run",
      "expected": "Every executed tool call has an allow verdict, or an approval bound to its exact parameters, recorded before it ran; irreversible calls fail closed while the mediation point is down.",
      "observedExample": "Run 88241: 3,112 tool calls, each with a verdict; 1 publish call executed on an approval issued for different parameters: fail."
    },
    "observationSchema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
    "examples": [
      {
        "status": "pass",
        "url": "https://aigovernanceengineer.com/controls/examples/control-observation.aige-ctl-eval-004.pass.json"
      },
      {
        "status": "fail",
        "url": "https://aigovernanceengineer.com/controls/examples/control-observation.aige-ctl-eval-004.fail.json"
      }
    ]
  }
}
