{
  "notice": "Illustrative mapping from the AI Governance Engineer Body of Knowledge v0.5.0 (not a claim of conformity)",
  "version": "0.5.0",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "schemaVersion": 1,
  "schema": "https://aigovernanceengineer.com/api/v1/schemas/control.json",
  "self": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-009.json",
  "source": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-009",
  "citation": {
    "title": "AI Governance Engineering: The Thesis & Body of Knowledge",
    "authors": [
      "Jorge García Aibar"
    ],
    "parentDoi": "https://doi.org/10.5281/zenodo.22956197",
    "conceptDoi": "https://doi.org/10.5281/zenodo.22857084"
  },
  "control": {
    "id": "AIGE-CTL-DATA-009",
    "profile": "data-admission-and-privacy",
    "url": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-009",
    "json": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-009.json",
    "title": "Signed Snapshot Integrity",
    "version": "0.1",
    "status": "draft",
    "reviewerStatus": "open",
    "depth": "derived",
    "objective": "Only a content-addressed, signed snapshot is admitted, its hash is re-verified when a job reads it, and new or appended data is checked for anomalies before it is admitted.",
    "failureModes": [
      "A model trains on a snapshot someone altered after admission.",
      "Poisoned data enters through an appended batch that was never checked, so the model still looks functional while carrying a bias, a weakness or a backdoor.",
      "The admission record names a hash that no job ever compares with the data it reads."
    ],
    "scope": "Snapshots admitted to training, fine-tuning, validation, testing, evaluation and retrieval-index pipelines, and data appended to them. The integrity of the trained model artefact is covered by the Model Artefact Integrity pattern.",
    "enforcementPoints": [
      "deploy",
      "runtime"
    ],
    "verification": [
      {
        "kind": "test",
        "text": "When a job reads an admitted snapshot, the hash of what it reads is compared with the content hash on the admission record, and a mismatch stops the read."
      }
    ],
    "evidence": [
      {
        "artefact": "Content hash of the admitted snapshot and the signature over the admission record",
        "schemaId": "dataset-admission-record",
        "schema": "https://aigovernanceengineer.com/schemas/dataset-admission-record.v1.json",
        "layer": 1
      },
      {
        "artefact": "Read-time hash verification and anomaly-check results",
        "schemaId": null,
        "schema": null,
        "layer": 1
      }
    ],
    "failureResponse": {
      "effect": "deny",
      "text": "A snapshot whose hash does not match its admission record is not read; new or appended data that fails the anomaly checks is not admitted."
    },
    "layer": 2,
    "secondaryLayers": [
      1
    ],
    "patterns": [
      {
        "slug": "dataset-admission-gate",
        "title": "Dataset Admission Gate",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate"
      },
      {
        "slug": "aibom",
        "title": "AIBOM",
        "url": "https://aigovernanceengineer.com/patterns/aibom"
      }
    ],
    "seeds": [],
    "derivedFrom": [
      {
        "kind": "pattern",
        "ref": "dataset-admission-gate",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate"
      },
      {
        "kind": "schema",
        "ref": "dataset-admission-record",
        "url": "https://aigovernanceengineer.com/resources/templates#schema-dataset-admission-record"
      },
      {
        "kind": "chapter",
        "ref": "governing-development",
        "url": "https://aigovernanceengineer.com/bok/governing-development"
      }
    ],
    "mappings": {
      "obligations": [
        {
          "id": "AIGE-OBL-EUAIA-ART10",
          "name": "EU AI Act Art. 10 data and data governance",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art10"
        },
        {
          "id": "AIGE-OBL-OWASP-LLM",
          "name": "Top 10 for LLM Applications 2026",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-owasp-llm"
        }
      ],
      "iso42001": [
        {
          "id": "A.7.5",
          "title": "Data provenance"
        }
      ],
      "nistAiRmf": [],
      "owasp": [
        {
          "id": "llm05-2026",
          "externalId": "LLM05:2026",
          "name": "Data and Model Poisoning",
          "url": "https://aigovernanceengineer.com/resources/threats#threat-llm05-2026"
        }
      ],
      "atlas": [],
      "aiuc1": [],
      "csaAicm": [],
      "other": [
        {
          "framework": "MITRE ATLAS",
          "ref": "AML.M0007",
          "note": "Sanitize Training Data (mitigation)"
        },
        {
          "framework": "MITRE ATLAS",
          "ref": "AML.M0025",
          "note": "Maintain AI Dataset Provenance (mitigation)"
        }
      ]
    },
    "references": [
      {
        "n": 1,
        "title": "Dataset Admission Gate",
        "text": "Dataset Admission Gate (AI Governance Engineering Body of Knowledge v0.5.0, pattern catalogue (chapter 05): a job may read a dataset version only if a complete, signed admission record admits it for that use). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate",
        "verified": "primary"
      },
      {
        "n": 25,
        "title": "Governing AI development",
        "text": "Governing AI development (AI Governance Engineering Body of Knowledge v0.5.0, chapter 14, section \"Quality, quantity, representativeness and fitness for purpose\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-development#quality-quantity-representativeness-and-fitness-for-purpose",
        "verified": "primary"
      },
      {
        "n": 6,
        "title": "OWASP GenAI LLM Top 10 2026",
        "text": "OWASP GenAI LLM Top 10 2026 (LLM01:2026 Prompt Injection to LLM10:2026 Improper Output Handling; resource page dated 3 Aug 2026). OWASP GenAI Security Project. 2026-08-03.",
        "url": "https://genai.owasp.org/resource/owasp-genai-llm-top-10-2026/",
        "verified": "primary"
      },
      {
        "n": 27,
        "title": "MITRE ATLAS data, release 2026.09",
        "text": "MITRE ATLAS data, release 2026.09 (modified 2026-09-15; AML.T0020 Training Data Poisoning; mitigations AML.M0007 Sanitize Training Data and AML.M0025 Maintain AI Dataset Provenance). MITRE (atlas-data repository). 2026-09-15.",
        "url": "https://github.com/mitre-atlas/atlas-data",
        "verified": "primary"
      },
      {
        "n": 7,
        "title": "ISO/IEC 42001:2023, AI management systems, Annex A",
        "text": "ISO/IEC 42001:2023, AI management systems, Annex A (reference control objectives and controls A.2 to A.10, cited by id and short title). ISO/IEC. 2023.",
        "url": "https://www.iso.org/standard/81230.html",
        "verified": "secondary"
      }
    ],
    "implementationNotes": [
      "Sign the admission record (for example \"ed25519:<base64>\") so it is tamper-evident in the evidence store, and record the digest of the snapshot that was checked in its input_hash.",
      "Training data is an attack surface: ATLAS catalogues training data poisoning (AML.T0020) and lists \"Sanitize Training Data\" (AML.M0007) and \"Maintain AI Dataset Provenance\" (AML.M0025) among its mitigations."
    ],
    "openQuestions": [
      "Which anomaly checks on appended data are strong enough to catch planted triggers, and which belong instead in the regression evals of the trained model?"
    ],
    "observation": null,
    "observationSchema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
    "examples": []
  }
}
