{
  "notice": "Illustrative mapping from the AI Governance Engineer Body of Knowledge v0.5.0 (not a claim of conformity)",
  "version": "0.5.0",
  "license": "CC BY 4.0",
  "licenseUrl": "https://creativecommons.org/licenses/by/4.0/",
  "schemaVersion": 1,
  "schema": "https://aigovernanceengineer.com/api/v1/schemas/control.json",
  "self": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-004.json",
  "source": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-004",
  "citation": {
    "title": "AI Governance Engineering: The Thesis & Body of Knowledge",
    "authors": [
      "Jorge García Aibar"
    ],
    "parentDoi": "https://doi.org/10.5281/zenodo.22956197",
    "conceptDoi": "https://doi.org/10.5281/zenodo.22857084"
  },
  "control": {
    "id": "AIGE-CTL-DATA-004",
    "profile": "data-admission-and-privacy",
    "url": "https://aigovernanceengineer.com/controls/data-admission-and-privacy#aige-ctl-data-004",
    "json": "https://aigovernanceengineer.com/api/v1/controls/aige-ctl-data-004.json",
    "title": "Lawful Basis and Assessment per Processing Stage",
    "version": "0.1",
    "status": "draft",
    "reviewerStatus": "open",
    "depth": "derived",
    "objective": "Each dataset carries, for each processing stage (training, fine-tuning, retrieval, inference, log monitoring), its lawful basis, its purpose and a pointer to the assessment behind it, and processing likely to result in a high risk has a versioned DPIA or a recorded decision that none was needed.",
    "failureModes": [
      "One basis is picked for \"the model\", although training, indexing and answering a live customer are different activities that each need their own basis.",
      "A training job runs on a dataset whose recorded basis does not cover training, or on data whose basis was never recorded.",
      "A legitimate-interest assessment points at mitigations that have been switched off, and the registry does not show that it went stale.",
      "No DPIA exists and no decision that one was not needed was recorded."
    ],
    "scope": "Personal data in datasets used for training, fine-tuning and retrieval, and the stages that process it. Transfers, automated decision-making and the rights path are outside this control; the fundamental rights impact assessment, which complements the DPIA rather than repeating it, is left to the FRIA-as-Code pattern.",
    "enforcementPoints": [
      "runtime",
      "periodic"
    ],
    "verification": [
      {
        "kind": "test",
        "text": "The training job reads the basis registry before it runs and refuses to run on a dataset whose basis does not cover training."
      }
    ],
    "evidence": [
      {
        "artefact": "Lawful basis for this purpose on the dataset card",
        "schemaId": "dataset-card",
        "schema": "https://aigovernanceengineer.com/schemas/dataset-card.v1.json",
        "layer": 2
      },
      {
        "artefact": "Basis registry entry per dataset and stage, with the reference of its assessment (for example a versioned LIA)",
        "schemaId": null,
        "schema": null,
        "layer": 2
      },
      {
        "artefact": "Versioned AI DPIA addendum, or the recorded decision that no DPIA was needed",
        "schemaId": "impact-assessment",
        "schema": "https://aigovernanceengineer.com/schemas/impact-assessment.v1.json",
        "layer": 2
      }
    ],
    "failureResponse": {
      "effect": "deny",
      "text": "The job refuses to run on a dataset whose basis does not cover the stage. Where the DPIA shows high residual risk, the controller consults the supervisory authority before the processing."
    },
    "layer": 2,
    "secondaryLayers": [],
    "patterns": [
      {
        "slug": "dataset-admission-gate",
        "title": "Dataset Admission Gate",
        "url": "https://aigovernanceengineer.com/patterns/dataset-admission-gate"
      },
      {
        "slug": "training-data-rights-ledger",
        "title": "Training-Data Rights Ledger",
        "url": "https://aigovernanceengineer.com/patterns/training-data-rights-ledger"
      },
      {
        "slug": "fria-as-code",
        "title": "FRIA-as-Code",
        "url": "https://aigovernanceengineer.com/patterns/fria-as-code"
      }
    ],
    "seeds": [],
    "derivedFrom": [
      {
        "kind": "schema",
        "ref": "dataset-card",
        "url": "https://aigovernanceengineer.com/resources/templates#schema-dataset-card"
      },
      {
        "kind": "schema",
        "ref": "impact-assessment",
        "url": "https://aigovernanceengineer.com/resources/templates#schema-impact-assessment"
      },
      {
        "kind": "chapter",
        "ref": "privacy-and-ai",
        "url": "https://aigovernanceengineer.com/bok/privacy-and-ai"
      }
    ],
    "mappings": {
      "obligations": [
        {
          "id": "AIGE-OBL-GDPR-ART6",
          "name": "GDPR Art. 6 lawful basis per processing moment",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-gdpr-art6"
        },
        {
          "id": "AIGE-OBL-GDPR-ART35-36",
          "name": "GDPR Arts. 35–36 DPIA and prior consultation",
          "url": "https://aigovernanceengineer.com/obligations/aige-obl-gdpr-art35-36"
        }
      ],
      "iso42001": [],
      "nistAiRmf": [
        {
          "id": "MEASURE 2.10",
          "title": "Privacy risk of the AI system – as identified in the MAP function – is examined and documented."
        }
      ],
      "owasp": [],
      "atlas": [],
      "aiuc1": [],
      "csaAicm": [],
      "other": []
    },
    "references": [
      {
        "n": 16,
        "title": "Privacy and data protection law applied to AI",
        "text": "Privacy and data protection law applied to AI (AI Governance Engineering Body of Knowledge v0.5.0, chapter 19, section \"Lawful basis for training versus inference\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/privacy-and-ai#lawful-basis-for-training-versus-inference",
        "verified": "primary"
      },
      {
        "n": 17,
        "title": "Privacy and data protection law applied to AI",
        "text": "Privacy and data protection law applied to AI (AI Governance Engineering Body of Knowledge v0.5.0, chapter 19, section \"The DPIA for AI systems\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/privacy-and-ai#the-dpia-for-ai-systems",
        "verified": "primary"
      },
      {
        "n": 12,
        "title": "Governing AI development",
        "text": "Governing AI development (AI Governance Engineering Body of Knowledge v0.5.0, chapter 14, section \"The right to use the data\"). AI Governance Engineer (Jorge García Aibar). 2026-09.",
        "url": "https://aigovernanceengineer.com/bok/governing-development#the-right-to-use-the-data",
        "verified": "primary"
      },
      {
        "n": 18,
        "title": "Regulation (EU) 2016/679 (GDPR)",
        "text": "Regulation (EU) 2016/679 (GDPR) (Art. 5 principles, incl. 5(1)(b) purpose limitation and 5(1)(c) minimisation; Art. 6 lawful basis and 6(4) compatibility; Art. 7 consent; Art. 9 special categories; Arts. 15 to 17 and 21 rights; Art. 25 data protection by design and by default; Art. 30 records of processing; Arts. 35 and 36 DPIA and prior consultation). Publications Office of the EU (EUR-Lex). 2016-04-27.",
        "url": "https://eur-lex.europa.eu/eli/reg/2016/679/oj/eng",
        "verified": "primary"
      },
      {
        "n": 5,
        "title": "Opinion 28/2024 on certain data protection aspects related to the processing of personal data in the context of AI models",
        "text": "Opinion 28/2024 on certain data protection aspects related to the processing of personal data in the context of AI models (anonymity of models; legitimate interest; consequences of unlawful processing in development). European Data Protection Board. 2024-12.",
        "url": "https://www.edpb.europa.eu/documents/opinion-of-the-board-art-64/opinion-282024-on-certain-data-protection-aspects-related-to_en",
        "verified": "primary"
      },
      {
        "n": 8,
        "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1",
        "text": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1 (GOVERN 6.1 third-party risks incl. infringement of intellectual property or other rights; MAP 1.1 intended purposes documented; MAP 2.3 data collection and selection considerations identified and documented; MAP 3.3 targeted application scope; MAP 4.1 legal risks of components incl. third-party data; MEASURE 2.10 privacy risk examined and documented; MANAGE 1.4 negative residual risks to downstream acquirers and end users documented). NIST. 2023-01-26.",
        "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-1.pdf",
        "verified": "primary"
      }
    ],
    "implementationNotes": [
      "Keep the basis registry attached to the system's registry entry: each dataset and stage carries its basis, purpose and a pointer to the assessment behind it, and the training job reads it.",
      "Make the legitimate-interest assessment a versioned artefact that points each mitigation at the control implementing it (the opt-out endpoint, the filter rule id, the scraping allow-list); switch a mitigation off and the LIA goes stale.",
      "Generate most of the AI DPIA from the registry: processing moments from the registry, bases from the basis registry, with the DPO still writing and signing the risk judgement. The impact-assessment schema carries a dpia_addendum type for the AI-specific fields."
    ],
    "openQuestions": [
      "Should the admission gate itself refuse a dataset whose stage has no DPIA and no recorded \"no DPIA\" decision, or is that a periodic check over the registry?"
    ],
    "observation": null,
    "observationSchema": "https://aigovernanceengineer.com/schemas/control-observation.v1.json",
    "examples": []
  }
}
