{
  "notice": "Illustrative mapping from the AI Governance Engineer Body of Knowledge v0.5.0 (not a claim of conformity)",
  "version": "0.5.0",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "schemaVersion": 1,
  "schema": "https://aigovernanceengineer.com/api/v1/schemas/control.json",
  "self": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-001.json",
  "source": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-001",
  "citation": {
    "title": "AI Governance Engineering: The Thesis & Body of Knowledge",
    "authors": [
      "Jorge García Aibar"
    ],
    "parentDoi": "https://doi.org/10.5281/zenodo.22956197",
    "conceptDoi": "https://doi.org/10.5281/zenodo.22857084"
  },
  "control": {
    "id": "AIGE-CTL-DATA-001",
    "profile": "data-admission-and-privacy",
    "url": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-001",
    "json": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-001.json",
    "title": "Dataset Admission Gate at Read Time",
    "version": "0.1",
    "status": "draft",
    "reviewerStatus": "open",
    "depth": "derived",
    "objective": "A training, fine-tuning, validation, testing, evaluation or retrieval-index job reads a dataset version only if a signed admission record for that version admits the job's pipeline for its use case and target system, and the rights, quality, representativeness, bias and integrity checks passed or were waived by someone entitled to waive them.",
    "failureModes": [
      "A job reads a dataset version that has no admission record for its pipeline, for example a training job pointed at a whole warehouse admitted for nothing.",
      "A model trains on data outside its consent scope, on a sample that misses the population it will serve, on labels nobody audited or on a snapshot someone altered, and the problem surfaces in production or in an audit, when the fix is a retrain.",
      "The person who wants the dataset used is the only person who decides it may be: the requester signs the admission, or a check is waived by someone not entitled to waive it.",
      "A missing admission field produces a reminder in a wiki instead of a failed run."
    ],
    "scope": "Every job that reads data to train, fine-tune, validate, test, evaluate or build a retrieval index, and the dataset versions it reads. The checks the gate runs are specified in AIGE-CTL-DATA-002 to 009; a sandbox pipeline with its own lighter admission is out of scope.",
    "enforcementPoints": [
      "runtime"
    ],
    "verification": [
      {
        "kind": "inspect",
        "text": "Each admission record validates against dataset-admission-record.v1 and carries what the pattern lists: subject, pipeline, target system, linked data card, decision, the checks with the obligation each enforces, the content hash of the admitted snapshot, the actor and a signature."
      },
      {
        "kind": "test",
        "text": "A job that presents a use-case id or a pipeline the record does not admit is denied the read."
      }
    ],
    "evidence": [
      {
        "artefact": "Admission record per dataset version and permitted pipeline, signed by the data owner",
        "schemaId": "dataset-admission-record",
        "schema": "https://aigovernanceengineer.com/schemas/dataset-admission-record.v1.json",
        "layer": 1
      },
      {
        "artefact": "Read-time verdicts of the gate: the use-case id presented and the admit or deny decision",
        "schemaId": null,
        "schema": null,
        "layer": 1
      }
    ],
    "failureResponse": {
      "effect": "deny",
      "text": "The policy denies the read unless the record admits that pipeline for that use; a job with no admitted dataset does not start. A new version, a new source, quality drift, a licence change, an erasure request or a new use case reopens admission."
    },
    "layer": 1,
    "secondaryLayers": [
      2
    ],
    "patterns": [
      {
        "slug": "dataset-admission-gate",
        "title": "Dataset Admission Gate",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate"
      },
      {
        "slug": "policy-card",
        "title": "Policy Card",
        "url": "https://aigovernanceengineer.com/patterns/policy-card"
      }
    ],
    "seeds": [],
    "derivedFrom": [
      {
        "kind": "pattern",
        "ref": "dataset-admission-gate",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate"
      },
      {
        "kind": "schema",
        "ref": "dataset-admission-record",
        "url": "https://aigovernanceengineer.com/resources/templates#schema-dataset-admission-record"
      },
      {
        "kind": "chapter",
        "ref": "governing-development",
        "url": "https://aigovernanceengineer.com/bok/governing-development"
      }
    ],
    "mappings": {
      "obligations": [
        {
          "id": "AIGE-OBL-EUAIA-ART10",
          "name": "EU AI Act Art. 10 data and data governance",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art10"
        },
        {
          "id": "AIGE-OBL-GDPR-ART5-1B",
          "name": "GDPR Art. 5(1)(b) and 6(4) purpose limitation",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-gdpr-art5-1b"
        },
        {
          "id": "AIGE-OBL-ISO42001-A7",
          "name": "A.7 Data for AI systems",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-iso42001-a7"
        }
      ],
      "iso42001": [
        {
          "id": "A.7.2",
          "title": "Data for development and enhancement of AI system"
        },
        {
          "id": "A.7.4",
          "title": "Quality of data for AI systems"
        },
        {
          "id": "A.7.5",
          "title": "Data provenance"
        }
      ],
      "nistAiRmf": [
        {
          "id": "MAP 2.3",
          "title": "Scientific integrity and TEVV considerations are identified and documented, including those related to experimental design, data collection and selection (e.g., availability, representativeness, suitability), system trustworthiness, and construct validation."
        },
        {
          "id": "MAP 4.1",
          "title": "Approaches for mapping AI technology and legal risks of its components – including the use of third-party data or software – are in place, followed, and documented, as are risks of infringement of a third party’s intellectual property or other rights."
        }
      ],
      "owasp": [
        {
          "id": "llm05-2026",
          "externalId": "LLM05:2026",
          "name": "Data and Model Poisoning",
          "url": "https://aigovernanceengineer.com/resources/threats#threat-llm05-2026"
        }
      ],
      "atlas": [],
      "aiuc1": [],
      "csaAicm": [],
      "other": []
    },
    "references": [
      {
        "n": 1,
        "title": "Dataset Admission Gate",
        "text": "Dataset Admission Gate (AI Governance Engineering Body of Knowledge v0.5.0, pattern catalogue (chapter 05): a job may read a dataset version only if a complete, signed admission record admits it for that use). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate",
        "verified": "primary"
      },
      {
        "n": 2,
        "title": "Governing AI development",
        "text": "Governing AI development (AI Governance Engineering Body of Knowledge v0.5.0, chapter 14, section \"Data for training and testing\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-development#data-for-training-and-testing",
        "verified": "primary"
      },
      {
        "n": 3,
        "title": "Governing AI development",
        "text": "Governing AI development (AI Governance Engineering Body of Knowledge v0.5.0, chapter 14, section \"Owners, stewards and the admission gate\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-development#owners-stewards-and-the-admission-gate",
        "verified": "primary"
      },
      {
        "n": 4,
        "title": "Regulation (EU) 2024/1689 (AI Act), consolidated text of 2026-07-27, Art. 10",
        "text": "Regulation (EU) 2024/1689 (AI Act), consolidated text of 2026-07-27, Art. 10 (data and data governance: 10(2) practices, including origin, preparation, bias examination and mitigation, and data gaps; 10(3) relevant, sufficiently representative, free of errors and complete; 10(4) specific setting of use). Publications Office of the EU (EUR-Lex). 2026-07-27.",
        "url": "https://eur-lex.europa.eu/eli/reg/2024/1689/2026-07-27/eng#art_10",
        "verified": "primary"
      },
      {
        "n": 5,
        "title": "Opinion 28/2024 on certain data protection aspects related to the processing of personal data in the context of AI models",
        "text": "Opinion 28/2024 on certain data protection aspects related to the processing of personal data in the context of AI models (anonymity of models; legitimate interest; consequences of unlawful processing in development). European Data Protection Board. 2024-12.",
        "url": "https://www.edpb.europa.eu/documents/opinion-of-the-board-art-64/opinion-282024-on-certain-data-protection-aspects-related-to_en",
        "verified": "primary"
      },
      {
        "n": 6,
        "title": "OWASP GenAI LLM Top 10 2026",
        "text": "OWASP GenAI LLM Top 10 2026 (LLM01:2026 Prompt Injection to LLM10:2026 Improper Output Handling; resource page dated 3 Aug 2026). OWASP GenAI Security Project. 2026-08-03.",
        "url": "https://genai.owasp.org/resource/owasp-genai-llm-top-10-2026/",
        "verified": "primary"
      },
      {
        "n": 7,
        "title": "ISO/IEC 42001:2023, AI management systems, Annex A",
        "text": "ISO/IEC 42001:2023, AI management systems, Annex A (reference control objectives and controls A.2 to A.10, cited by id and short title). ISO/IEC. 2023.",
        "url": "https://www.iso.org/standard/81230.html",
        "verified": "secondary"
      },
      {
        "n": 8,
        "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1",
        "text": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1 (GOVERN 6.1 third-party risks incl. infringement of intellectual property or other rights; MAP 1.1 intended purposes documented; MAP 2.3 data collection and selection considerations identified and documented; MAP 3.3 targeted application scope; MAP 4.1 legal risks of components incl. third-party data; MEASURE 2.10 privacy risk examined and documented; MANAGE 1.4 negative residual risks to downstream acquirers and end users documented). NIST. 2023-01-26.",
        "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-1.pdf",
        "verified": "primary"
      }
    ],
    "implementationNotes": [
      "Write one admission record per dataset version and per permitted pipeline on the published dataset-admission-record schema, and make every data-reading job present its use-case id and target system at read time; the check is policy-as-code in the pipeline, not a reminder in a wiki.",
      "Separate the duties: the data owner is accountable and signs, the data steward operates the checks, and a small review board settles contested admissions. The minimum checklist lives as code, so adding a check is a reviewed change.",
      "Give waivers an owner and an expiry, or they become the norm; record conditions on the admission record (for example \"collect islands-region claims before the next retrain\") so the next retrain cannot start until they are closed."
    ],
    "openQuestions": [
      "What lighter admission should a sandbox pipeline for exploratory work carry, and how is data kept from leaving the sandbox into a training job?",
      "Which enforcement point fits a gate that decides at read time inside a data pipeline: the platform's access layer, the job scheduler or the storage policy engine?"
    ],
    "observation": null,
    "observationSchema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
    "examples": []
  }
}
