{
  "notice": "Illustrative mapping from the AI Governance Engineer Body of Knowledge v0.5.0 (not a claim of conformity)",
  "version": "0.5.0",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "schemaVersion": 1,
  "schema": "https://aigovernanceengineer.com/api/v1/schemas/control.json",
  "self": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-008.json",
  "source": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-008",
  "citation": {
    "title": "AI Governance Engineering: The Thesis & Body of Knowledge",
    "authors": [
      "Jorge García Aibar"
    ],
    "parentDoi": "https://doi.org/10.5281/zenodo.22956197",
    "conceptDoi": "https://doi.org/10.5281/zenodo.22857084"
  },
  "control": {
    "id": "AIGE-CTL-DATA-008",
    "profile": "data-admission-and-privacy",
    "url": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-008",
    "json": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-008.json",
    "title": "Fitness-for-Purpose Checks Before Admission",
    "version": "0.1",
    "status": "draft",
    "reviewerStatus": "open",
    "depth": "derived",
    "objective": "Before a dataset version is admitted, its quality (label accuracy, completeness, consistency, timeliness), its quantity per class and per group, its representativeness against the deployment population and a proxy and bias examination are checked and recorded, each passing or waived in writing by someone entitled to waive it.",
    "failureModes": [
      "A dataset large enough but drawn from the wrong population is admitted: quantity is taken for representativeness.",
      "A group falls below the minimum cell size in the test plan and nobody records it.",
      "No bias examination is recorded, or a failed check is waived with no signer, no condition and no expiry.",
      "The data measures a proxy rather than what the use case needs, and the assumption is never written down."
    ],
    "scope": "Training, validation and testing datasets for high-risk systems, where Art. 10 applies, and any dataset admitted under the organisation's own policy. Fairness testing of the trained model is covered by the Fairness Eval Suite pattern, not here.",
    "enforcementPoints": [
      "deploy"
    ],
    "verification": [
      {
        "kind": "inspect",
        "text": "The admission record holds a result for each quality, quantity, representativeness and bias check, with the obligation it enforces; every waived check names a waiver signed by the data owner and every condition is listed."
      }
    ],
    "evidence": [
      {
        "artefact": "Quality, representativeness and bias check results on the admission record, with waivers and conditions",
        "schemaId": "dataset-admission-record",
        "schema": "https://aigovernanceengineer.com/schemas/dataset-admission-record.v1.json",
        "layer": 1
      },
      {
        "artefact": "Quality checks, populations covered and known gaps on the dataset card",
        "schemaId": "dataset-card",
        "schema": "https://aigovernanceengineer.com/schemas/dataset-card.v1.json",
        "layer": 2
      }
    ],
    "failureResponse": {
      "effect": "require_approval",
      "text": "A failed check blocks admission unless someone entitled to waive it signs a waiver; the waiver and its conditions appear in the model card, and the next retrain cannot start until the conditions are closed."
    },
    "layer": 2,
    "secondaryLayers": [
      3
    ],
    "patterns": [
      {
        "slug": "dataset-admission-gate",
        "title": "Dataset Admission Gate",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate"
      },
      {
        "slug": "fairness-eval-suite",
        "title": "Fairness Eval Suite",
        "url": "https://aigovernanceengineer.com/patterns/fairness-eval-suite"
      }
    ],
    "seeds": [],
    "derivedFrom": [
      {
        "kind": "pattern",
        "ref": "dataset-admission-gate",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate"
      },
      {
        "kind": "schema",
        "ref": "dataset-admission-record",
        "url": "https://aigovernanceengineer.com/resources/templates#schema-dataset-admission-record"
      },
      {
        "kind": "chapter",
        "ref": "governing-development",
        "url": "https://aigovernanceengineer.com/bok/governing-development"
      }
    ],
    "mappings": {
      "obligations": [
        {
          "id": "AIGE-OBL-EUAIA-ART10",
          "name": "EU AI Act Art. 10 data and data governance",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-euaia-art10"
        },
        {
          "id": "AIGE-OBL-ISO42001-A7",
          "name": "A.7 Data for AI systems",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-iso42001-a7"
        }
      ],
      "iso42001": [
        {
          "id": "A.7.4",
          "title": "Quality of data for AI systems"
        }
      ],
      "nistAiRmf": [
        {
          "id": "MAP 2.3",
          "title": "Scientific integrity and TEVV considerations are identified and documented, including those related to experimental design, data collection and selection (e.g., availability, representativeness, suitability), system trustworthiness, and construct validation."
        }
      ],
      "owasp": [],
      "atlas": [],
      "aiuc1": [],
      "csaAicm": [],
      "other": []
    },
    "references": [
      {
        "n": 25,
        "title": "Governing AI development",
        "text": "Governing AI development (AI Governance Engineering Body of Knowledge v0.5.0, chapter 14, section \"Quality, quantity, representativeness and fitness for purpose\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-development#quality-quantity-representativeness-and-fitness-for-purpose",
        "verified": "primary"
      },
      {
        "n": 1,
        "title": "Dataset Admission Gate",
        "text": "Dataset Admission Gate (AI Governance Engineering Body of Knowledge v0.5.0, pattern catalogue (chapter 05): a job may read a dataset version only if a complete, signed admission record admits it for that use). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate",
        "verified": "primary"
      },
      {
        "n": 4,
        "title": "Regulation (EU) 2024/1689 (AI Act), consolidated text of 2026-07-27, Art. 10",
        "text": "Regulation (EU) 2024/1689 (AI Act), consolidated text of 2026-07-27, Art. 10 (data and data governance: 10(2) practices, including origin, preparation, bias examination and mitigation, and data gaps; 10(3) relevant, sufficiently representative, free of errors and complete; 10(4) specific setting of use). Publications Office of the EU (EUR-Lex). 2026-07-27.",
        "url": "https://eur-lex.europa.eu/eli/reg/2024/1689/2026-07-27/eng#art_10",
        "verified": "primary"
      },
      {
        "n": 26,
        "title": "ISO/IEC 5259 series, Data quality for analytics and machine learning (ML)",
        "text": "ISO/IEC 5259 series, Data quality for analytics and machine learning (ML) (Part 1 overview, terminology and examples; Part 2 data quality measures; Part 3 data quality management requirements and guidelines; Part 4 data quality process framework; Part 5 data quality governance framework). ISO/IEC. 2024-2025.",
        "url": "https://www.iso.org/standard/81088.html",
        "verified": "primary"
      },
      {
        "n": 7,
        "title": "ISO/IEC 42001:2023, AI management systems, Annex A",
        "text": "ISO/IEC 42001:2023, AI management systems, Annex A (reference control objectives and controls A.2 to A.10, cited by id and short title). ISO/IEC. 2023.",
        "url": "https://www.iso.org/standard/81230.html",
        "verified": "secondary"
      },
      {
        "n": 8,
        "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1",
        "text": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1 (GOVERN 6.1 third-party risks incl. infringement of intellectual property or other rights; MAP 1.1 intended purposes documented; MAP 2.3 data collection and selection considerations identified and documented; MAP 3.3 targeted application scope; MAP 4.1 legal risks of components incl. third-party data; MEASURE 2.10 privacy risk examined and documented; MANAGE 1.4 negative residual risks to downstream acquirers and end users documented). NIST. 2023-01-26.",
        "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-1.pdf",
        "verified": "primary"
      }
    ],
    "implementationNotes": [
      "Use the vocabulary of the ISO/IEC 5259 series for the quality measures, and evidence each dimension with the test chapter 14 names: an audit of a labelled sample and inter-annotator agreement for labels, null rates per field and segment for completeness, cell counts against the minimum in the test plan for quantity, a distribution comparison against a reference for representativeness.",
      "Keep an assumption register for what the data is meant to measure (Art. 10(2)(d)) and a proxy analysis for fitness for purpose."
    ],
    "openQuestions": [
      "Who sets the thresholds each check is held to, and how are they reviewed when the deployment population changes?"
    ],
    "observation": null,
    "observationSchema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
    "examples": []
  }
}
