# Vercy AI instruction - YAML 1.2 (JSON-compatible) { "vercy": "1.0-draft", "publication": { "status": "published", "adjudicationStatus": "reviewable-draft", "publishableCanonical": false, "generatedAt": "2026-08-29T18:19:21Z", "synthesisSha256": "22b3de4d03b6df6127f3b166ce9837777f04bbc265f3b993d6b894d0b7170222", "providerMode": "single-provider-waiver", "providers": [ "Claude" ], "waivedProviders": [ "Grok" ] }, "metaModel": { "id": "WM-DAT-006", "registryId": "vr.wm-dat-006", "name": "Data Lineage", "version": "0.3.0-research.1", "previousVersions": [], "entryKind": "pattern", "family": "World Models", "category": "Information and virtual systems", "industry": [ "Cross-industry" ], "domain": [ "INF.DAT.LIN" ], "tags": [ "data", "lineage", "inf.dat.lin" ], "status": "published" }, "canonicalUrl": "https://ver.cy/models/wm-dat-006-data-lineage/", "sourceUrl": "https://github.com/ver-cy/world-models/tree/feat/mega-model-registry/research/runs/wm-dat-006", "model": { "registry_id": "vr.wm-dat-006", "model_id": "WM-DAT-006", "name": "Data Lineage", "entry_kind": "pattern", "purpose": "Provide a format-neutral, composable pattern for asserting, evidencing and governing how datasets and their fields were derived from other datasets and fields through processing activities, so that an agent can reconstruct derivation paths, perform impact analysis and evidence provenance obligations.", "scope_statement": "WM-DAT-006 models the lineage assertion itself: the identity of the nodes it connects (dataset, distribution, field, activity, run, agent), the derivation edge and its transformation characterisation, the time and version bindings that make the assertion interpretable, the evidence and capture method that make it falsifiable, and the governance of the lineage record. Per the known-relation ledger it is a pattern composed into WM-DAT-001 (Dataset) and WM-DAT-005 (Pipeline); it carries references and bindings to datasets, jobs and runs but never reproduces their definition, schema, orchestration, execution or run lifecycle. It is storage- and interface-neutral: property graphs, event streams, catalog tables, RDF, JSON documents and narrative statements are projections of the same semantics.", "in_scope": [ "Identity and resolution of lineage nodes: dataset, distribution, field or column, processing activity, run reference and responsible agent", "Derivation edges between nodes, including direct and indirect field-level dependency and characterisation of the transformation applied", "Granularity declaration (dataset, field, partition or collection level) and the declared perimeter separating observed from opaque segments", "Time semantics: event time of derivation versus observation or ingestion time of the assertion, nominal time, and the validity window of an assertion", "Version and snapshot pinning of referenced dataset states, and continuity of lineage across schema change", "Capture method, producer, confidence grading and supporting evidence for each assertion, including provenance of the lineage record itself", "Structural validation of the lineage instance and measurement of lineage coverage and reconciliation against declared sources", "Sensitivity classification, disclosure scoping requirements, retention and disposition of lineage records", "Alignment crosswalks and exchange payload contracts for interchange with PROV, OpenLineage and DCAT consumers" ], "out_of_scope": [ "Dataset definition, schema authoring, distribution publication and catalog lifecycle, which are owned by WM-DAT-001", "Pipeline orchestration, scheduling, task dependency resolution, retry and failure handling, and run state transitions, which are owned by WM-DAT-005", "Execution of processing itself; this model records assertions about processing, it does not run or trigger it", "Audit-trail and log-record semantics, including tamper-evidence and log retention for regulatory logging obligations, which belong to a logging or audit-record model", "Access-control policy evaluation and enforcement; this model declares required scoping and defers evaluation to the access model", "Data quality rule definition, profiling and fitness scoring of the data content itself, which belongs to a data quality sibling", "Software build provenance and artifact attestation for code and container images", "Master data matching, entity resolution of business records, and semantic mapping of business terms", "Lineage visualisation, graph query languages and storage engines, which are interface and storage projections" ], "boundary_notes": [ { "neighbor": "WM-DAT-001 Dataset", "distinction": "WM-DAT-001 owns what a dataset is: its definition, schema, distributions, classification and catalog lifecycle. WM-DAT-006 carries only a resolvable reference plus a version or snapshot pin to a dataset state, and asserts derivation between such references. A schema description reproduced inside a lineage record is a cached binding, not an authoritative schema.", "source_refs": [ "SRC-009", "SRC-004", "SRC-013" ] }, { "neighbor": "WM-DAT-005 Pipeline", "distinction": "WM-DAT-005 owns job definition, scheduling, orchestration, run execution and run state transitions such as START, RUNNING, COMPLETE, ABORT and FAIL. WM-DAT-006 records the run identifier and job reference under which a derivation was observed, plus the terminal outcome as a carried attribute for interpreting completeness; it does not model, drive or validate the run lifecycle.", "source_refs": [ "SRC-007", "SRC-004" ] }, { "neighbor": "Logging and audit-record model", "distinction": "Regulatory logging obligations require automatic recording of events over the lifetime of a system. Lineage may be an input to such records but does not own log capture, tamper-evidence, log retention or audit-trail semantics; a reference to an audit record never transfers those semantics to this model.", "source_refs": [ "SRC-010", "SRC-011" ] }, { "neighbor": "Software build provenance (SLSA)", "distinction": "SLSA provenance attests how a software artifact was built, covering build definition, resolved dependencies and builder identity. It does not describe dataset or field derivation. WM-DAT-006 may reference a build attestation as evidence for the code version of a processing activity, but the two provenance graphs are distinct and must not be merged into one node space.", "source_refs": [ "SRC-014" ] }, { "neighbor": "Data quality management", "distinction": "Quality of the data content (accuracy, completeness of values, conformance to rules) is a sibling concern. WM-DAT-006 owns only quality of the lineage assertion: capture method, confidence, coverage and reconciliation status. Quality metric facets attached to a dataset in an exchange payload are carried references to the quality model.", "source_refs": [ "SRC-012", "SRC-008" ] }, { "neighbor": "Access control and data classification model", "distinction": "WM-DAT-006 declares that a lineage record has a sensitivity classification and which disclosure scopes apply, because lineage exposes schema names, business logic and personal-data footprints. Evaluation and enforcement of the resulting access decision, and any enforcement audit trail, remain with the access-control model.", "source_refs": [ "SRC-011", "SRC-006" ] }, { "neighbor": "Catalog and discovery (DCAT)", "distinction": "DCAT owns catalog records, publication and discovery of datasets and dataset series. WM-DAT-006 aligns to DCAT provenance and versioning terms for interchange but does not own catalog entries, publisher metadata or distribution access URLs.", "source_refs": [ "SRC-009" ] } ] }, "sources": [ { "id": "SRC-001", "title": "PROV-DM: The PROV Data Model", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/prov-dm/", "version_or_date": "W3C Recommendation, 30 April 2013", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-29T09:05:00Z", "relevance": "Normative conceptual core for provenance: Entity, Activity, Agent; the relations used, wasGeneratedBy, wasDerivedFrom, wasInformedBy, wasAttributedTo, wasAssociatedWith, actedOnBehalfOf, wasStartedBy, wasEndedBy, wasInvalidatedBy; plus bundles, plans, roles, collections, alternate and specialization, and optional time annotations." }, { "id": "SRC-002", "title": "PROV-O: The PROV Ontology", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/prov-o/", "version_or_date": "W3C Recommendation, 30 April 2013", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-29T09:05:00Z", "relevance": "Governed IRI namespace http://www.w3.org/ns/prov# and the qualified-influence pattern (Generation, Usage, Derivation, Association, Attribution, Communication, Start, End, Invalidation) that lets a derivation edge carry its own attributes, roles and times." }, { "id": "SRC-003", "title": "Constraints of the PROV Data Model", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/prov-constraints/", "version_or_date": "W3C Recommendation, 30 April 2013", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-29T09:06:00Z", "relevance": "Defines validity, normalization and equivalence for provenance instances via uniqueness, event-ordering, impossibility and typing constraints; supplies the normative basis for structural validation of a lineage graph." }, { "id": "SRC-004", "title": "OpenLineage Object Model", "organization": "OpenLineage project (LF AI & Data Foundation)", "url": "https://openlineage.io/docs/spec/object-model", "version_or_date": "Specification documentation 1.52.0, accessed 29 August 2026", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-29T09:07:00Z", "relevance": "Defines Job, Run and Dataset as the core lineage entities, RunEvent, JobEvent and DatasetEvent, run identifiers as client-generated UUIDs (preferably UUIDv7), dataset identity derived from physical location as namespace and name, and the facet extension mechanism." }, { "id": "SRC-005", "title": "OpenLineage Naming Conventions", "organization": "OpenLineage project (LF AI & Data Foundation)", "url": "https://openlineage.io/docs/spec/naming", "version_or_date": "Specification documentation 1.52.0, accessed 29 August 2026", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-29T09:07:00Z", "relevance": "Source-type-specific dataset namespace and name construction (postgres, s3, gcs, kafka, bigquery, snowflake, hdfs, local file), hierarchical dotted job names unique within a namespace, and the explicit warning that changing a namespace format disconnects existing lineage nodes." }, { "id": "SRC-006", "title": "ColumnLineageDatasetFacet", "organization": "OpenLineage project (LF AI & Data Foundation)", "url": "https://openlineage.io/docs/spec/facets/dataset-facets/column_lineage_facet", "version_or_date": "Facet schema 1-2-0 (documentation 1.52.0), accessed 29 August 2026", "source_type": "schema", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-29T09:08:00Z", "relevance": "Field-level lineage structure: per-output-field inputFields with namespace, name and field; transformation type DIRECT or INDIRECT; subtypes IDENTITY, TRANSFORMATION, AGGREGATION, JOIN, FILTER, SORT, WINDOW; a masking boolean; and a dataset-level dependency array to avoid cartesian expansion." }, { "id": "SRC-007", "title": "OpenLineage Run Cycle", "organization": "OpenLineage project (LF AI & Data Foundation)", "url": "https://openlineage.io/docs/spec/run-cycle", "version_or_date": "Specification documentation 1.52.0, accessed 29 August 2026", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-29T09:08:00Z", "relevance": "Run states START, RUNNING, COMPLETE, ABORT, FAIL and OTHER; terminal-event rule; accumulative versus complete-snapshot event patterns; and differing lifecycles for BATCH, STREAMING and SERVICE processing. Cited here as the boundary owned by WM-DAT-005 and as the merge semantics for late-arriving lineage." }, { "id": "SRC-008", "title": "OpenLineage Facets and custom facet rules", "organization": "OpenLineage project (LF AI & Data Foundation)", "url": "https://openlineage.io/docs/spec/facets/", "version_or_date": "Specification documentation 1.52.0, accessed 29 August 2026", "source_type": "schema", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-29T09:09:00Z", "relevance": "Facet categories (run, job, input dataset, output dataset), custom facet naming with a project-derived prefix, mandatory _producer and _schemaURL fields, and the rule that a versioned schema URL must be an immutable pointer to a specific tag or commit." }, { "id": "SRC-009", "title": "Data Catalog Vocabulary (DCAT) - Version 3", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/vocab-dcat-3/", "version_or_date": "W3C Recommendation, 22 August 2024", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-29T09:10:00Z", "relevance": "Dataset, Distribution and DatasetSeries; versioning terms dcat:version, dcat:previousVersion, dcat:hasVersion, dcat:hasCurrentVersion; dcat:qualifiedRelation and dcat:Relationship; dcterms:provenance and dcterms:source; issued and modified dates; and spdx:checksum for distribution integrity." }, { "id": "SRC-010", "title": "Regulation (EU) 2024/1689 laying down harmonised rules on artificial intelligence (AI Act)", "organization": "European Parliament and Council of the European Union (EUR-Lex)", "url": "https://eur-lex.europa.eu/eli/reg/2024/1689/oj/eng", "version_or_date": "Regulation (EU) 2024/1689 of 13 June 2024, OJ published 12 July 2024", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-29T09:11:00Z", "relevance": "Article 10 requires documentation of data collection processes and the origin of data, the original purpose of collection for personal data, and data-preparation operations such as annotation, labelling, cleaning, updating, enrichment and aggregation; Article 11 and Annex IV require technical documentation of datasets and their provenance; Article 12 imposes record-keeping obligations that are adjacent to but distinct from lineage." }, { "id": "SRC-011", "title": "Regulation (EU) 2016/679 General Data Protection Regulation", "organization": "European Parliament and Council of the European Union (EUR-Lex)", "url": "https://eur-lex.europa.eu/legal-content/EN/TXT/HTML/?uri=CELEX:32016R0679", "version_or_date": "Regulation (EU) 2016/679, OJ L 119, 4 May 2016", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-29T09:12:00Z", "relevance": "Article 5(1)(e) storage limitation; Article 17 right to erasure including the obligation to inform other controllers about links to, or copies or replications of, personal data; Article 30 records of processing activities. These drive retention, tombstoning and downstream-propagation duties that lineage must support without owning their execution." }, { "id": "SRC-012", "title": "Principles for effective risk data aggregation and risk reporting (BCBS 239)", "organization": "Basel Committee on Banking Supervision, Bank for International Settlements", "url": "https://www.bis.org/publ/bcbs239.pdf", "version_or_date": "January 2013", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-29T09:13:00Z", "relevance": "Supervisory expectations for governance and accountability over risk data, documented data architecture and metadata, accuracy and integrity including reconciliation and controlled treatment of manual workarounds, and traceable flow from source system through processing to report." }, { "id": "SRC-013", "title": "Apache Iceberg Table Specification", "organization": "Apache Software Foundation", "url": "https://raw.githubusercontent.com/apache/iceberg/main/format/spec.md", "version_or_date": "Format versions 1, 2 and 3 adopted; version 4 in development (main branch, accessed 29 August 2026)", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-29T09:14:00Z", "relevance": "Concrete grounding for version and field identity: table-uuid, current-snapshot-id, snapshots with snapshot-id, parent-snapshot-id, sequence-number, timestamp-ms, summary, manifest-list and schema-id; snapshot-log and metadata-log; schema field IDs that are unique in the table schema and immutable under rename; and v3 row lineage fields _row_id and _last_updated_sequence_number." }, { "id": "SRC-014", "title": "SLSA Provenance specification v1.1", "organization": "Open Source Security Foundation (OpenSSF), Linux Foundation", "url": "https://slsa.dev/spec/v1.1/provenance", "version_or_date": "Version 1.1", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-29T09:15:00Z", "relevance": "Build-provenance attestation with buildDefinition (buildType, externalParameters, internalParameters, resolvedDependencies) and runDetails (builder.id, invocationId, startedOn, finishedOn). Used to fix the boundary between software build provenance and dataset or field derivation lineage." }, { "id": "SRC-015", "title": "Commission Regulation (EC) No 1205/2008 implementing Directive 2007/2/EC as regards metadata, Annex Part B section 6 (Quality and validity)", "organization": "legislation.gov.uk (The National Archives), republishing Commission Regulation (EC) No 1205/2008", "url": "https://www.legislation.gov.uk/eur/2008/1205/annex/part/B/division/6/data.htm?view=plain", "version_or_date": "Commission Regulation (EC) No 1205/2008 of 3 December 2008; retained-EU-law text as published by legislation.gov.uk, accessed 29 August 2026", "source_type": "legislation", "primary_source": false, "authority_tier": 2, "accessed_at": "2026-08-29T09:16:00Z", "relevance": "Cross-domain legal precedent that Lineage is a mandatory metadata element defined as a statement on process history and/or overall quality of the data set, with a free-text value domain. Establishes the narrative-statement form of lineage and the conflict with machine-readable graph forms. The authentic text is EUR-Lex CELEX 32008R1205, which could not be retrieved during this research." } ], "structure": { "bundles": [ { "id": "subject-identity-and-perimeter", "name": "Lineage Subject Identity and Perimeter", "description": "What a lineage assertion is about: the identity and resolution of the dataset, field, activity and agent nodes it connects, and the declared granularity and coverage perimeter of the lineage graph.", "rationale": "Every provenance standard reviewed is identifier-first: PROV requires typed identified objects, OpenLineage identity is namespace plus name for datasets and a client-generated UUID for runs, and Iceberg gives immutable field IDs. Lineage that cannot resolve its endpoints is unusable, and a graph whose perimeter is undeclared silently misleads impact analysis.", "source_refs": [ "SRC-001", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ], "layers": [ { "id": "node-identity", "name": "Node Identity and Resolution", "description": "Identity rules for the four node kinds a lineage edge can touch: dataset or distribution, field, processing activity, and responsible agent.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-005", "SRC-013" ], "findings": [ { "id": "dataset-node-identity", "name": "Dataset and distribution node identity", "description": "How a lineage node that stands for a dataset or one of its physical distributions is identified, bound to a location-derived name, and resolved to the authoritative dataset record held by WM-DAT-001.", "source_refs": [ "SRC-004", "SRC-005", "SRC-009", "SRC-013" ], "questions": [ { "id": "dni-q1", "text": "Which authoritative master-system identifier identifies the dataset this node stands for, and which system of record issued it?", "kind": "identity", "answer_data": [ "Master-system dataset identifier value", "Issuing system-of-record reference", "Identifier scheme or type code" ] }, { "id": "dni-q2", "text": "What location-derived binding (namespace and name) is recorded for this node, and under which source-type convention was it constructed?", "kind": "interoperability", "answer_data": [ "Namespace string", "Name string", "Source-type convention code such as postgres, s3, bigquery or snowflake" ] }, { "id": "dni-q3", "text": "Does this node denote the abstract dataset or one specific distribution or physical materialisation of it?", "kind": "classification", "answer_data": [ "Node level code: dataset, distribution, or materialisation", "Reference to the parent dataset node when the node is a distribution" ] }, { "id": "dni-q4", "text": "How is a node re-resolved when the dataset is renamed, relocated or its namespace convention changes?", "kind": "exception", "answer_data": [ "Alternate or symlink identifier set", "Resolution rule reference", "Effective observation time of the rebinding" ] } ], "data_elements": [ { "id": "dni-master-id", "name": "Dataset master identifier", "description": "Authoritative identifier issued by the system of record for the referenced dataset, such as a catalog table UUID.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-013", "SRC-004" ] }, { "id": "dni-namespace-binding", "name": "Location-derived name binding", "description": "Namespace and name pair constructed under a declared source-type convention, retained as a resolvable binding rather than as primary identity.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "dni-node-level", "name": "Node level", "description": "Whether the node denotes an abstract dataset, a distribution, or a concrete materialisation.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-009" ] }, { "id": "dni-dataset-ref", "name": "Dataset model reference", "description": "Reference to the authoritative dataset record owned by WM-DAT-001; the lineage record carries the reference only.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-009" ] } ], "artifacts": [], "inline_only_rationale": "Node identity is pure reference data carried inside every lineage assertion. Materialising it as a separate artifact would create a second, competing dataset record and would duplicate the catalog entry that WM-DAT-001 owns." }, { "id": "field-node-identity", "name": "Field and column node identity", "description": "How a field-level lineage node is identified so that column lineage survives renames and schema evolution, given that column lineage facets key on field name while table formats assign immutable numeric field IDs.", "source_refs": [ "SRC-006", "SRC-013" ], "questions": [ { "id": "fni-q1", "text": "What stable field identifier does the source system assign to this column, and is it invariant under rename?", "kind": "identity", "answer_data": [ "Field identifier value", "Rename-invariance flag", "Assigning format or system reference" ] }, { "id": "fni-q2", "text": "Which field name was observed at capture time, and to which parent dataset node and schema version does that name belong?", "kind": "provenance", "answer_data": [ "Observed field name", "Parent dataset node reference", "Schema version reference" ] }, { "id": "fni-q3", "text": "How are nested, repeated or struct-typed fields addressed within a single field node?", "kind": "composition", "answer_data": [ "Field path expression", "Nesting depth", "Repeated or array indicator" ] }, { "id": "fni-q4", "text": "What happens to existing field-level edges when a column is renamed, retyped or dropped?", "kind": "constraint", "answer_data": [ "Continuity rule code", "Superseding assertion reference", "Impact statement reference" ] } ], "data_elements": [ { "id": "fni-field-id", "name": "Stable field identifier", "description": "Rename-invariant field identifier assigned by the authoritative format or system, such as an Iceberg schema field ID.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] }, { "id": "fni-field-name", "name": "Observed field name", "description": "Field name as observed at capture time, used as the interchange key by column lineage facets.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-006" ] }, { "id": "fni-field-path", "name": "Field path expression", "description": "Dotted or indexed path addressing a nested or repeated field within the parent dataset schema.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] } ], "artifacts": [], "inline_only_rationale": "A field node is an addressing tuple, not a document. The authoritative field definition lives in the dataset schema owned by WM-DAT-001, so emitting a separate field artifact here would duplicate that schema." }, { "id": "activity-and-agent-references", "name": "Processing activity and agent references", "description": "How the activity that performed a derivation and the agent responsible for it are referenced, keeping job and run definition with WM-DAT-005 and agent records with the adopting Dimension's party model.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-005" ], "questions": [ { "id": "aar-q1", "text": "Which processing activity is asserted to have generated the output node, and how is that activity identified?", "kind": "identity", "answer_data": [ "Activity identifier", "Job namespace and name reference", "Activity type code" ] }, { "id": "aar-q2", "text": "Which agent kind bears responsibility for the activity: a person, an organization or a software agent?", "kind": "classification", "answer_data": [ "Agent kind code", "Agent identifier", "Agent reference to the party or identity model" ] }, { "id": "aar-q3", "text": "Is the activity node internal to the observed perimeter or an opaque external activity known only by reference?", "kind": "relationship", "answer_data": [ "Perimeter membership flag", "External activity descriptor", "Reference to the perimeter declaration" ] } ], "data_elements": [ { "id": "aar-activity-id", "name": "Activity identifier", "description": "Identifier of the processing activity asserted to have produced the output, resolvable to the job record owned by WM-DAT-005.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-001" ] }, { "id": "aar-agent-ref", "name": "Responsible agent reference", "description": "Reference to the person, organization or software agent bearing responsibility, typed per the provenance agent taxonomy.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "aar-activity-kind", "name": "Activity kind", "description": "Classification of the activity such as batch transform, streaming transform, manual edit, copy or export.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-007", "SRC-012" ] } ], "artifacts": [], "inline_only_rationale": "Activity and agent nodes are references into models that own those records (WM-DAT-005 for jobs, the party or identity model for agents). Producing an artifact here would reproduce a job definition this model is explicitly forbidden to own." } ] }, { "id": "granularity-and-perimeter-frame", "name": "Granularity and Perimeter Frame", "description": "The declared resolution of the lineage graph and the explicit statement of what lies inside, outside and opaque within the observed perimeter.", "source_refs": [ "SRC-006", "SRC-012", "SRC-015", "SRC-013" ], "findings": [ { "id": "lineage-granularity-and-perimeter", "name": "Declared granularity and coverage perimeter", "description": "The explicit declaration of the level at which lineage is asserted and the boundary of what was observed, so that absence of an edge can be distinguished from absence of observation.", "source_refs": [ "SRC-006", "SRC-012", "SRC-013", "SRC-015" ], "questions": [ { "id": "lgp-q1", "text": "At which granularity levels is lineage asserted for this subject, and which levels are explicitly not asserted?", "kind": "definition", "answer_data": [ "Asserted granularity level codes: dataset, distribution, partition, field, row", "Excluded level codes", "Reason for exclusion" ] }, { "id": "lgp-q2", "text": "Which systems, zones or hops lie inside the observed perimeter, and which are recorded as opaque or black-box segments?", "kind": "composition", "answer_data": [ "In-perimeter system references", "Opaque segment descriptors", "Entry and exit boundary node references" ] }, { "id": "lgp-q3", "text": "Does the absence of an upstream edge for a node mean no derivation exists, or that no observation was made?", "kind": "quality", "answer_data": [ "Closed-world or open-world assumption code", "Per-node observation status", "Known gap list" ] }, { "id": "lgp-q4", "text": "Which manual steps, spreadsheets or user-side workarounds are declared as part of the flow rather than omitted?", "kind": "process", "answer_data": [ "Manual step descriptors", "Control status of each manual step", "Responsible role reference" ] } ], "data_elements": [ { "id": "lgp-granularity", "name": "Asserted granularity levels", "description": "Set of levels at which lineage edges are asserted for the subject.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-006", "SRC-013" ] }, { "id": "lgp-perimeter", "name": "Perimeter membership statement", "description": "Enumeration of in-perimeter systems and of opaque segments with their entry and exit boundary nodes.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-012" ] }, { "id": "lgp-world-assumption", "name": "World assumption code", "description": "Whether missing edges are interpreted as absence of derivation or as absence of observation.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003" ] }, { "id": "lgp-manual-steps", "name": "Declared manual steps", "description": "Manual interventions and workarounds recognised as lineage-bearing steps rather than silent gaps.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] } ], "artifacts": [ { "id": "perimeter-and-granularity-declaration", "name": "Lineage perimeter and granularity declaration", "description": "A standing declaration, per subject or per domain, of the granularity levels asserted, the systems inside the observed perimeter, the opaque segments, the world assumption and the recognised manual steps. It is the reference against which coverage measurements and completeness claims are judged.", "media_or_form": [ "structured declaration record", "narrative statement equivalent to a free-text lineage statement", "boundary node register" ], "serial": false, "identity_strategy": "Identified by the subject reference (dataset or domain master identifier) plus a monotonic revision number; superseded rather than edited, with the prior revision retained and linked.", "source_refs": [ "SRC-012", "SRC-015" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "derivation-semantics", "name": "Derivation and Transformation Semantics", "description": "The meaning of the edge itself: what kind of dependency it asserts, how the transformation is characterised, and what structural constraints keep the resulting graph coherent.", "rationale": "PROV distinguishes used, wasGeneratedBy and wasDerivedFrom and constrains their combination; the OpenLineage column lineage facet distinguishes DIRECT from INDIRECT dependency with a typed subtype and a masking flag. Without these distinctions an agent cannot tell a value-carrying dependency from a filter predicate, which changes both impact analysis and privacy reasoning.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-006" ], "layers": [ { "id": "derivation-edges", "name": "Derivation Edge Semantics", "description": "The typed dependency relations between nodes and the characterisation of the transformation that realises them.", "source_refs": [ "SRC-001", "SRC-002", "SRC-006" ], "findings": [ { "id": "derivation-edge-semantics", "name": "Edge relation typing and identity", "description": "Which relation an edge asserts, how the edge is itself identified so it can carry attributes, times and roles, and how directionality and multiplicity are constrained.", "source_refs": [ "SRC-001", "SRC-002", "SRC-006" ], "questions": [ { "id": "des-q1", "text": "Which relation does this edge assert between its endpoints, and what is the source of that typing?", "kind": "relationship", "answer_data": [ "Relation type code such as used, wasGeneratedBy, wasDerivedFrom or wasInformedBy", "Typing source reference", "Directionality" ] }, { "id": "des-q2", "text": "Is the dependency direct, meaning output values are derived from input values, or indirect, meaning the input influenced the output without contributing values?", "kind": "classification", "answer_data": [ "Dependency mode code: DIRECT or INDIRECT", "Justification text", "Facet or standard reference" ] }, { "id": "des-q3", "text": "Does this edge carry its own identifier and attributes, or is it an unqualified binary link?", "kind": "identity", "answer_data": [ "Edge identifier", "Qualification pattern indicator", "Edge attribute set" ] }, { "id": "des-q4", "text": "What role did the input node play in the activity that produced the output?", "kind": "authority", "answer_data": [ "Role code", "Role vocabulary reference", "Position or parameter binding" ] } ], "data_elements": [ { "id": "des-relation-type", "name": "Relation type", "description": "The provenance relation asserted by the edge, drawn from a governed relation vocabulary.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "des-dependency-mode", "name": "Dependency mode", "description": "Whether the dependency is direct value derivation or indirect influence such as filtering, joining, sorting or windowing.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-006" ] }, { "id": "des-edge-id", "name": "Edge identifier", "description": "Identifier for the qualified edge, enabling attributes, roles and times to be attached to the relationship itself.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "des-role", "name": "Input role", "description": "The role the input node played in the activity, using a governed role vocabulary.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001", "SRC-002" ] } ], "artifacts": [], "inline_only_rationale": "An edge is the atomic assertion this model is made of; it is inline structured data in every projection. Packaging individual edges as artifacts would fragment the graph and defeat traversal." }, { "id": "transformation-characterization", "name": "Transformation characterisation and logic evidence", "description": "How the transformation behind an edge is described: its typed subtype, whether it obfuscates values, and the reference to the expression or mapping that realises it.", "source_refs": [ "SRC-006", "SRC-010", "SRC-011" ], "questions": [ { "id": "tch-q1", "text": "Which transformation subtype best characterises this edge, and from which controlled vocabulary is it drawn?", "kind": "classification", "answer_data": [ "Subtype code such as IDENTITY, TRANSFORMATION, AGGREGATION, JOIN, FILTER, SORT or WINDOW", "Vocabulary reference", "Free-text description" ] }, { "id": "tch-q2", "text": "Does the transformation mask, hash, pseudonymise or otherwise obfuscate the input values?", "kind": "privacy", "answer_data": [ "Masking indicator", "Obfuscation technique code", "Reversibility statement" ] }, { "id": "tch-q3", "text": "Where is the executable logic that realises this transformation, and at which version was it observed?", "kind": "evidence", "answer_data": [ "Expression or query reference", "Logic version or commit reference", "Mapping specification reference" ] }, { "id": "tch-q4", "text": "Which data-preparation operation class does this transformation fall into for documentation obligations?", "kind": "requirement", "answer_data": [ "Preparation operation code such as annotation, labelling, cleaning, updating, enrichment or aggregation", "Obligation reference", "Applicability statement" ] } ], "data_elements": [ { "id": "tch-subtype", "name": "Transformation subtype", "description": "Controlled subtype refining the dependency mode of the edge.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-006" ] }, { "id": "tch-masking", "name": "Masking indicator", "description": "Whether the transformation obfuscates the input values, which changes downstream privacy reasoning.", "value_kind": "boolean", "cardinality": "1", "required": true, "source_refs": [ "SRC-006" ] }, { "id": "tch-logic-ref", "name": "Transformation logic reference", "description": "Version-pinned reference to the expression, query text or mapping rule that realises the transformation.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "tch-preparation-class", "name": "Data-preparation operation class", "description": "Classification of the operation against the documented preparation operations required for regulated dataset documentation.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-010" ] } ], "artifacts": [ { "id": "column-mapping-specification", "name": "Column mapping specification", "description": "A field-by-field mapping record for one output dataset: each output field with its contributing input fields, dependency mode, subtype, masking flag and the transformation description or expression reference. It is the evidence object behind field-level edges and the unit reviewed during regulated dataset documentation.", "media_or_form": [ "field mapping table", "annotated expression listing", "structured column-lineage record" ], "serial": false, "identity_strategy": "Identified by the output dataset master identifier plus the pinned output version reference; a new version pin yields a new specification rather than an in-place edit.", "source_refs": [ "SRC-006", "SRC-010" ] } ], "inline_only_rationale": null } ] }, { "id": "graph-structure-integrity", "name": "Graph Structure and Integrity", "description": "Constraints that make a lineage instance logically coherent, and the treatment of collections, partitions and aggregate nodes.", "source_refs": [ "SRC-001", "SRC-003", "SRC-009", "SRC-013" ], "findings": [ { "id": "graph-consistency-constraints", "name": "Structural consistency constraints", "description": "The uniqueness, typing, ordering and impossibility constraints a lineage instance must satisfy, and the normalisation rule that makes two differently expressed instances comparable.", "source_refs": [ "SRC-003", "SRC-001" ], "questions": [ { "id": "gcc-q1", "text": "Which uniqueness constraints apply, so that the same generation or usage is not asserted twice with conflicting attributes?", "kind": "constraint", "answer_data": [ "Uniqueness constraint set", "Conflict detection rule", "Merge or reject decision code" ] }, { "id": "gcc-q2", "text": "Which event-ordering constraints must hold between generation, usage, start, end and invalidation on this instance?", "kind": "temporal", "answer_data": [ "Ordering constraint set", "Violation severity", "Ordering evidence reference" ] }, { "id": "gcc-q3", "text": "Which node and edge combinations are declared impossible, including type disjointness and reflexive specialization?", "kind": "validation", "answer_data": [ "Impossibility rule set", "Type disjointness declaration", "Detected violation list" ] }, { "id": "gcc-q4", "text": "Are cycles permitted in the derivation graph, and how are recursive or self-referencing derivations represented?", "kind": "constraint", "answer_data": [ "Cycle policy code", "Version-discriminated node rule", "Recursive derivation representation" ] } ], "data_elements": [ { "id": "gcc-constraint-profile", "name": "Constraint profile reference", "description": "Reference to the constraint profile the instance claims to satisfy, including uniqueness, ordering, typing and impossibility rules.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-003" ] }, { "id": "gcc-normal-form", "name": "Normalisation status", "description": "Whether the instance has been normalised, and under which normalisation rule set, so that equivalence comparison is meaningful.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003" ] }, { "id": "gcc-cycle-policy", "name": "Cycle policy", "description": "Whether derivation cycles are rejected or permitted when nodes are discriminated by version.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003", "SRC-013" ] } ], "artifacts": [], "inline_only_rationale": "Constraints are declarative rules attached to the instance profile, not documents. The report produced by evaluating them is a separate artifact belonging to the validation finding, which keeps rule declaration and evaluation evidence apart." }, { "id": "collection-and-partition-lineage", "name": "Collection, series and partition lineage", "description": "How lineage is asserted for aggregates whose members change over time: dataset series, partitions, file sets and logical collections, and how member-level edges roll up to collection-level edges.", "source_refs": [ "SRC-001", "SRC-009", "SRC-013" ], "questions": [ { "id": "cpl-q1", "text": "Is this node a collection, and if so which membership relation defines its members at the asserted time?", "kind": "composition", "answer_data": [ "Collection indicator", "Membership relation reference", "Member node references" ] }, { "id": "cpl-q2", "text": "How do member-level derivation edges aggregate into an edge asserted at collection level?", "kind": "relationship", "answer_data": [ "Rollup rule code", "Coverage of members contributing to the rollup", "Residual member exceptions" ] }, { "id": "cpl-q3", "text": "How are partition-scoped derivations expressed when only part of an output was regenerated?", "kind": "state", "answer_data": [ "Partition predicate expression", "Affected partition list", "Unchanged partition assertion" ] }, { "id": "cpl-q4", "text": "When does a new member make a series a new node rather than an update to the existing collection node?", "kind": "lifecycle", "answer_data": [ "Series continuation rule", "New-node threshold criteria", "Reference to the series record owned by the dataset model" ] } ], "data_elements": [ { "id": "cpl-collection-kind", "name": "Collection kind", "description": "Whether the node is a dataset series, a partitioned table, a file set or another logical collection.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "cpl-membership", "name": "Membership assertion", "description": "Set of member node references with the time at which membership is asserted.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "cpl-partition-predicate", "name": "Partition predicate", "description": "Expression identifying the partitions affected by a derivation when the derivation is not whole-dataset.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] } ], "artifacts": [], "inline_only_rationale": "Collection and partition scoping is expressed as membership assertions and predicates inside the graph. The authoritative series and partition definitions belong to the dataset model, so no artifact is created here." } ] } ] }, { "id": "processing-context-and-responsibility", "name": "Processing Context and Responsibility", "description": "The context that makes a derivation reproducible and attributable: the run under which it was observed, the plan and configuration that governed it, and the chain of responsibility for both the processing and the lineage record.", "rationale": "PROV separates activity from the plan and from the agent associated with it and provides delegation via actedOnBehalfOf; OpenLineage attaches source code location, source code, SQL and parent-run facets; BCBS 239 requires named accountability for risk data. Reproducibility and accountability are distinct questions and both must be answerable from the lineage record.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-008", "SRC-012" ], "layers": [ { "id": "execution-reference-binding", "name": "Execution Reference Binding", "description": "How a lineage assertion is bound to the execution occurrence and configuration under which it was observed, without importing run lifecycle ownership.", "source_refs": [ "SRC-004", "SRC-007", "SRC-008", "SRC-014" ], "findings": [ { "id": "run-reference-binding", "name": "Run reference and observation binding", "description": "The carried reference to the execution occurrence under which the derivation was observed, including the run identifier, parent run reference and terminal outcome used only to interpret completeness of the assertion.", "source_refs": [ "SRC-004", "SRC-007" ], "questions": [ { "id": "rrb-q1", "text": "Under which run identifier was this derivation observed, and which model owns that run record?", "kind": "provenance", "answer_data": [ "Run identifier", "Owning model reference", "Job namespace and name reference" ] }, { "id": "rrb-q2", "text": "What terminal outcome did the referenced run report, and how does that outcome qualify the completeness of this assertion?", "kind": "state", "answer_data": [ "Carried terminal outcome code", "Completeness qualification code", "Interpretation rule reference" ] }, { "id": "rrb-q3", "text": "Is this assertion part of a nested run hierarchy, and which parent run does it belong to?", "kind": "composition", "answer_data": [ "Parent run reference", "Nesting depth", "Aggregation rule for parent-level edges" ] }, { "id": "rrb-q4", "text": "For a continuous or streaming activity with no terminal state, what bounded observation window does this assertion cover?", "kind": "temporal", "answer_data": [ "Observation window start and end", "Processing type code such as batch, streaming or service", "Snapshot or accumulative pattern code" ] } ], "data_elements": [ { "id": "rrb-run-ref", "name": "Run reference", "description": "Identifier of the execution occurrence under which the derivation was observed, resolvable to the run record owned by WM-DAT-005.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-004" ] }, { "id": "rrb-outcome", "name": "Carried run outcome", "description": "Terminal outcome reported by the owning pipeline model, carried read-only to qualify assertion completeness.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007" ] }, { "id": "rrb-window", "name": "Observation window", "description": "Bounded time window covered by the assertion for continuous processing where no terminal state occurs.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007" ] } ], "artifacts": [], "inline_only_rationale": "The run is owned by WM-DAT-005. This finding deliberately holds only a reference plus a read-only carried outcome; creating an artifact would materialise a run record and breach the composition contract with the pipeline model." }, { "id": "plan-and-configuration-provenance", "name": "Plan, code and configuration provenance", "description": "The version-pinned description of the logic and configuration that governed the derivation, sufficient to explain and re-derive it, expressed as references to code, parameters and engine identity.", "source_refs": [ "SRC-001", "SRC-008", "SRC-014", "SRC-010" ], "questions": [ { "id": "pcp-q1", "text": "Which plan or program governed the activity, and at which immutable version was it observed?", "kind": "provenance", "answer_data": [ "Plan or source code location reference", "Immutable version, tag or commit reference", "Plan description" ] }, { "id": "pcp-q2", "text": "Which externally supplied parameters and which platform-internal settings influenced the derivation?", "kind": "process", "answer_data": [ "External parameter set", "Internal parameter set", "Trust qualification per parameter class" ] }, { "id": "pcp-q3", "text": "Which processing engine and resolved dependencies were in effect, and how are they identified?", "kind": "interoperability", "answer_data": [ "Engine identifier and version", "Resolved dependency references", "Build attestation reference where one exists" ] }, { "id": "pcp-q4", "text": "Is the recorded plan sufficient to re-derive the output, and what is explicitly not captured?", "kind": "quality", "answer_data": [ "Reproducibility claim level", "Uncaptured factor list", "Non-determinism declaration" ] } ], "data_elements": [ { "id": "pcp-plan-ref", "name": "Plan reference", "description": "Version-pinned reference to the code, query or specification that governed the activity.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-008" ] }, { "id": "pcp-parameters", "name": "Parameter set", "description": "External and internal parameters in effect, kept distinct because they carry different trust qualifications.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "pcp-engine", "name": "Engine and dependency identity", "description": "Identifier and version of the processing engine plus resolved dependency references, optionally linked to a build attestation.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "pcp-reproducibility", "name": "Reproducibility claim", "description": "Declared level to which the captured plan and parameters permit re-derivation, with explicit non-determinism factors.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-014", "SRC-010" ] } ], "artifacts": [ { "id": "processing-context-manifest", "name": "Processing context manifest", "description": "A per-derivation manifest fixing the plan reference and version, the external and internal parameters, engine and dependency identity, and the reproducibility claim with its known non-determinism factors. It is the evidence object an agent inspects to decide whether an output can be reconstructed.", "media_or_form": [ "structured manifest record", "signed attestation reference", "parameter listing" ], "serial": true, "identity_strategy": "Identified by the run reference plus the output node master identifier and a monotonic sequence number; content-addressed by digest so that identical contexts deduplicate.", "source_refs": [ "SRC-014", "SRC-008" ] } ], "inline_only_rationale": null } ] }, { "id": "responsibility-and-accountability", "name": "Responsibility and Accountability", "description": "Attribution and delegation for the processing, and named accountability for the correctness of the lineage record itself.", "source_refs": [ "SRC-001", "SRC-002", "SRC-012" ], "findings": [ { "id": "agent-attribution-and-delegation", "name": "Attribution, association and delegation", "description": "Who is responsible for the derived entity, who was associated with the activity, and on whose behalf a software agent acted, keeping party records in the identity model.", "source_refs": [ "SRC-001", "SRC-002" ], "questions": [ { "id": "aad-q1", "text": "To which agent is the derived entity attributed, and on what basis was that attribution made?", "kind": "ownership", "answer_data": [ "Attributed agent reference", "Attribution basis code", "Attribution evidence reference" ] }, { "id": "aad-q2", "text": "Which agents were associated with the activity, and in which roles?", "kind": "authority", "answer_data": [ "Associated agent references", "Role per agent", "Association plan reference" ] }, { "id": "aad-q3", "text": "When a service account or automated agent acted, on whose behalf did it act and under which authorisation?", "kind": "authority", "answer_data": [ "Delegating agent reference", "Delegation chain", "Authorisation reference" ] }, { "id": "aad-q4", "text": "Does attribution transfer to downstream derived entities, or must it be re-asserted at each hop?", "kind": "relationship", "answer_data": [ "Transitivity rule code", "Re-assertion requirement", "Exception conditions" ] } ], "data_elements": [ { "id": "aad-attribution", "name": "Attribution assertion", "description": "Agent to whom the derived entity is ascribed, with the basis for the ascription.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "aad-delegation", "name": "Delegation chain", "description": "Ordered chain of agents acting on behalf of other agents for the activity.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "aad-transitivity", "name": "Attribution transitivity rule", "description": "Whether attribution propagates along derivation paths or must be re-asserted per hop.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] } ], "artifacts": [], "inline_only_rationale": "Attribution and delegation are relationship assertions whose party endpoints are owned by the adopting Dimension's identity or party model. Producing an artifact would create a shadow party record outside that model's control." }, { "id": "lineage-stewardship-and-attestation", "name": "Lineage stewardship and accuracy attestation", "description": "Named accountability for the lineage record itself: who maintains it, who attests that it reflects the real flow, and on what cadence that attestation is refreshed.", "source_refs": [ "SRC-012", "SRC-010" ], "questions": [ { "id": "lsa-q1", "text": "Which named role is accountable for the accuracy of this lineage record, distinct from the owner of the underlying data?", "kind": "ownership", "answer_data": [ "Accountable role reference", "Role scope statement", "Delegation or deputy reference" ] }, { "id": "lsa-q2", "text": "On what cadence, and against which evidence, is the lineage record attested as reflecting the actual flow?", "kind": "process", "answer_data": [ "Attestation cadence", "Evidence set referenced", "Last attestation observation time" ] }, { "id": "lsa-q3", "text": "What escalation applies when the accountable role rejects or cannot confirm the recorded lineage?", "kind": "exception", "answer_data": [ "Escalation path reference", "Interim confidence downgrade", "Blocking or non-blocking decision code" ] } ], "data_elements": [ { "id": "lsa-accountable-role", "name": "Accountable role reference", "description": "Named role accountable for lineage accuracy for the subject, resolvable to the adopting Dimension's role model.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-012" ] }, { "id": "lsa-attestation-time", "name": "Attestation observation time", "description": "Time at which the most recent accuracy attestation was recorded.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-012" ] }, { "id": "lsa-attestation-outcome", "name": "Attestation outcome", "description": "Whether the attestation confirmed, qualified or rejected the recorded lineage.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-012" ] } ], "artifacts": [ { "id": "lineage-accuracy-attestation", "name": "Lineage accuracy attestation", "description": "A periodic record in which the accountable role confirms, qualifies or rejects that the lineage held for a subject reflects the actual flow from source to consumption, listing the evidence reviewed and any qualifications carried forward.", "media_or_form": [ "signed attestation record", "review minute", "qualification register" ], "serial": true, "identity_strategy": "Identified by the subject master identifier plus the accountable role reference plus a monotonic attestation sequence number; the observation time is recorded as an attribute, never as the identifier.", "source_refs": [ "SRC-012" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "time-version-and-validity", "name": "Time, Version and Validity", "description": "Making a lineage assertion interpretable in time: separating when the derivation happened from when it was recorded, pinning the dataset states involved, and defining when an assertion stops being true.", "rationale": "PROV-CONSTRAINTS makes ordering of instantaneous events normative; OpenLineage carries eventTime and nominal time and permits accumulative, late-arriving events; Iceberg gives snapshot identifiers and sequence numbers that make a dataset state addressable. Without explicit time and version pinning, a lineage graph collapses distinct historical states into one misleading present.", "source_refs": [ "SRC-003", "SRC-004", "SRC-007", "SRC-009", "SRC-013" ], "layers": [ { "id": "time-semantics", "name": "Time Semantics", "description": "The distinct time axes a lineage assertion carries and the validity window over which it holds.", "source_refs": [ "SRC-001", "SRC-003", "SRC-004", "SRC-007" ], "findings": [ { "id": "event-and-observation-time", "name": "Event time versus observation time", "description": "The separation of when the derivation occurred from when the assertion about it was captured and recorded, plus nominal or scheduled time where it differs from both.", "source_refs": [ "SRC-001", "SRC-004", "SRC-007" ], "questions": [ { "id": "eot-q1", "text": "At what event time did the derivation occur, expressed with seconds and an explicit offset?", "kind": "temporal", "answer_data": [ "Event time value", "Time source reference", "Precision or granularity statement" ] }, { "id": "eot-q2", "text": "At what observation or ingestion time was this assertion captured and recorded, and by which producer?", "kind": "provenance", "answer_data": [ "Observation time value", "Ingestion time value where it differs", "Producer reference" ] }, { "id": "eot-q3", "text": "What nominal or logical time does the processed data represent, where that differs from wall-clock event time?", "kind": "temporal", "answer_data": [ "Nominal start and end time", "Logical partition or batch key", "Relationship to event time" ] }, { "id": "eot-q4", "text": "How are late-arriving or out-of-order assertions ordered and merged against already-recorded ones?", "kind": "process", "answer_data": [ "Ordering key", "Merge or supersession rule code", "Lateness tolerance" ] } ], "data_elements": [ { "id": "eot-event-time", "name": "Event time", "description": "Time at which the derivation occurred, as reported by the executing system.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-001" ] }, { "id": "eot-observation-time", "name": "Observation time", "description": "Time at which the lineage assertion was captured by its producer, recorded separately from event time.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-004" ] }, { "id": "eot-ingestion-time", "name": "Ingestion time", "description": "Time at which the assertion was accepted into the lineage store, recorded when it differs from observation time.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007" ] }, { "id": "eot-nominal-time", "name": "Nominal time window", "description": "Logical or scheduled time window the processed data represents, distinct from wall-clock event time.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] } ], "artifacts": [], "inline_only_rationale": "These are timestamp fields carried on every assertion. They are the classic case of inline context data; extracting them into an artifact would destroy the ability to order and merge assertions." }, { "id": "assertion-validity-and-supersession", "name": "Assertion validity, supersession and retraction", "description": "The lifecycle of the lineage assertion as a record: when it becomes effective, when it is superseded by a corrected assertion, and how a retraction is expressed without erasing history.", "source_refs": [ "SRC-001", "SRC-003", "SRC-009" ], "questions": [ { "id": "avs-q1", "text": "Over which validity window is this assertion held to be true, and what ends that window?", "kind": "lifecycle", "answer_data": [ "Validity start time", "Validity end time or open-ended flag", "Terminating event reference" ] }, { "id": "avs-q2", "text": "Which prior assertion does this one supersede, and what was the reason for the correction?", "kind": "provenance", "answer_data": [ "Superseded assertion identifier", "Correction reason code", "Correcting agent reference" ] }, { "id": "avs-q3", "text": "How is a retracted assertion represented so that consumers who already read it can detect the retraction?", "kind": "exception", "answer_data": [ "Retraction record identifier", "Retraction reason", "Notification scope" ] }, { "id": "avs-q4", "text": "Which state may a lineage record occupy, and which transitions are permitted?", "kind": "state", "answer_data": [ "State code such as asserted, confirmed, superseded, retracted or tombstoned", "Permitted transition set", "Transition authority reference" ] } ], "data_elements": [ { "id": "avs-validity-window", "name": "Assertion validity window", "description": "Interval over which the assertion is held to be true, with an explicit open-ended indicator.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "avs-supersedes", "name": "Supersedes reference", "description": "Identifier of the assertion this record replaces, with the correction reason.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "avs-record-state", "name": "Lineage record state", "description": "Current state of the lineage record within its own governed state set.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003" ] } ], "artifacts": [], "inline_only_rationale": "Validity and supersession are attributes and links on the assertion record itself. Introducing an artifact would imply a document-level correction workflow that belongs to the adopting Dimension's change process rather than to the assertion." } ] }, { "id": "version-binding-and-change", "name": "Version Binding and Schema Change", "description": "Pinning the dataset states an edge connects and preserving lineage continuity when structures change.", "source_refs": [ "SRC-009", "SRC-013", "SRC-005", "SRC-006" ], "findings": [ { "id": "version-and-snapshot-pinning", "name": "Version and snapshot pinning", "description": "How an edge names the specific state of the input and output datasets it relates, using the version or snapshot identity issued by the owning system, plus an integrity digest where available.", "source_refs": [ "SRC-013", "SRC-009", "SRC-004" ], "questions": [ { "id": "vsp-q1", "text": "Which specific dataset state does each endpoint of this edge refer to, and which identifier scheme expresses it?", "kind": "identity", "answer_data": [ "Version or snapshot identifier per endpoint", "Identifier scheme code", "Issuing system reference" ] }, { "id": "vsp-q2", "text": "What ordering key allows two states of the same dataset to be compared?", "kind": "temporal", "answer_data": [ "Sequence number or monotonic counter", "Parent state reference", "State timestamp" ] }, { "id": "vsp-q3", "text": "What content digest or checksum evidences the integrity of the referenced state?", "kind": "evidence", "answer_data": [ "Digest value", "Digest algorithm", "Digest scope statement" ] }, { "id": "vsp-q4", "text": "If the endpoint state is not addressable, what weaker binding is recorded and how is that weakness declared?", "kind": "quality", "answer_data": [ "Fallback binding descriptor", "Weakness code", "Impact on confidence grade" ] } ], "data_elements": [ { "id": "vsp-state-ref", "name": "Dataset state reference", "description": "Version or snapshot identifier naming the exact dataset state an edge endpoint refers to, issued by the owning system.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-013", "SRC-009" ] }, { "id": "vsp-sequence", "name": "State ordering key", "description": "Monotonic sequence number or parent-state link that orders successive states of the same dataset.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] }, { "id": "vsp-digest", "name": "State integrity digest", "description": "Checksum over the referenced state with its algorithm and scope.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009" ] } ], "artifacts": [], "inline_only_rationale": "Version pins are references issued and owned by the dataset or table-format system. Copying them into an artifact would create a competing version record that WM-DAT-001 owns." }, { "id": "schema-change-continuity", "name": "Schema and naming change continuity", "description": "Preserving lineage continuity when fields are renamed, retyped or dropped, or when a naming convention changes and location-derived node names would otherwise disconnect.", "source_refs": [ "SRC-005", "SRC-013", "SRC-006" ], "questions": [ { "id": "scc-q1", "text": "Which schema change occurred, and which fields and downstream edges does it affect?", "kind": "event", "answer_data": [ "Change type code such as rename, retype, add or drop", "Affected field references", "Affected downstream edge set" ] }, { "id": "scc-q2", "text": "Which identifier remained stable across the change, and which binding had to be re-established?", "kind": "identity", "answer_data": [ "Stable identifier reference", "Rebound name binding", "Rebinding evidence reference" ] }, { "id": "scc-q3", "text": "What is the risk that a namespace or naming-convention change silently disconnects existing lineage nodes?", "kind": "validation", "answer_data": [ "Disconnection risk assessment", "Detected orphan node count", "Remediation plan reference" ] }, { "id": "scc-q4", "text": "Which downstream consumers must be informed of the change, and by which route?", "kind": "decision", "answer_data": [ "Downstream consumer set from traversal", "Notification obligation code", "Reference to the notification-owning model" ] } ], "data_elements": [ { "id": "scc-change-type", "name": "Schema change type", "description": "Classification of the structural change affecting lineage continuity.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-013" ] }, { "id": "scc-rebinding", "name": "Rebinding record", "description": "Mapping from the prior name binding to the new one, anchored on the stable identifier.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "scc-impacted-edges", "name": "Impacted edge set", "description": "Edges whose interpretation changes or breaks as a result of the change.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-006" ] } ], "artifacts": [ { "id": "schema-change-impact-statement", "name": "Schema change impact statement", "description": "A per-change record naming the structural change, the stable identifiers that carried through, the rebindings performed, the edges whose interpretation changed and the downstream consumer set derived by traversal. It is the decision-support object handed to change approval, not the approval itself.", "media_or_form": [ "impact analysis record", "rebinding map", "downstream consumer list" ], "serial": true, "identity_strategy": "Identified by the affected dataset master identifier plus the target state reference plus a monotonic change sequence number; never identified by change date alone.", "source_refs": [ "SRC-005", "SRC-013" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "capture-evidence-and-assurance", "name": "Capture, Evidence and Assurance", "description": "Why a lineage assertion should be believed: how it was obtained, what supports it, how the instance is validated, and how coverage is measured and reconciled.", "rationale": "Provenance about provenance is explicitly supported by PROV bundles, and OpenLineage requires every facet to declare a _producer and an immutable _schemaURL. BCBS 239 requires reconciliation and controlled treatment of manual workarounds. A lineage graph asserted without capture method and confidence cannot be audited or safely relied on for impact analysis.", "source_refs": [ "SRC-001", "SRC-003", "SRC-008", "SRC-012" ], "layers": [ { "id": "capture-provenance", "name": "Capture Method and Provenance of the Lineage Record", "description": "How each assertion entered the record and how much weight it carries.", "source_refs": [ "SRC-001", "SRC-004", "SRC-008", "SRC-012" ], "findings": [ { "id": "capture-method-and-producer", "name": "Capture method and producing agent", "description": "The method by which the assertion was obtained (runtime observation, static parsing, declaration, inference or manual entry) and the identity and schema version of the producer that emitted it.", "source_refs": [ "SRC-001", "SRC-004", "SRC-008" ], "questions": [ { "id": "cmp-q1", "text": "By which method was this assertion obtained, and what does that method structurally fail to see?", "kind": "provenance", "answer_data": [ "Capture method code: observed, parsed, declared, inferred or manual", "Known blind spots of the method", "Method version reference" ] }, { "id": "cmp-q2", "text": "Which producer emitted the assertion, and against which immutable schema version?", "kind": "interoperability", "answer_data": [ "Producer identifier", "Immutable schema URL or version reference", "Emission time" ] }, { "id": "cmp-q3", "text": "Is provenance recorded about the lineage record itself, and where is that meta-provenance held?", "kind": "provenance", "answer_data": [ "Meta-provenance bundle reference", "Bundle attribution agent", "Bundle generation time" ] }, { "id": "cmp-q4", "text": "When two producers assert conflicting lineage for the same endpoints, which one prevails?", "kind": "exception", "answer_data": [ "Producer precedence order", "Conflict resolution rule reference", "Recorded dissent or both-retained flag" ] } ], "data_elements": [ { "id": "cmp-method", "name": "Capture method", "description": "Governed code for how the assertion was obtained, carrying the method's known blind spots.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-012" ] }, { "id": "cmp-producer", "name": "Producer reference", "description": "Identity of the emitting producer with the immutable schema version it emitted against.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-008" ] }, { "id": "cmp-bundle-ref", "name": "Meta-provenance bundle reference", "description": "Reference to the bundle holding provenance about this set of lineage assertions.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "cmp-precedence", "name": "Producer precedence", "description": "Ordering used to resolve conflicting assertions from different producers.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003" ] } ], "artifacts": [], "inline_only_rationale": "Capture method and producer are mandatory attributes on each assertion in every reviewed exchange format; they must travel with the assertion. An artifact would separate the claim from its warrant." }, { "id": "confidence-and-evidence-grading", "name": "Confidence grading and supporting evidence", "description": "The graded confidence attached to an assertion and the evidence references that justify the grade, including explicit flagging of manual or uncontrolled steps.", "source_refs": [ "SRC-012", "SRC-008", "SRC-010" ], "questions": [ { "id": "ceg-q1", "text": "What confidence grade is assigned to this assertion, and against which published grading scale?", "kind": "measurement", "answer_data": [ "Confidence grade value", "Grading scale reference", "Grader identity" ] }, { "id": "ceg-q2", "text": "Which concrete evidence items support the assertion, and are they retrievable?", "kind": "evidence", "answer_data": [ "Evidence item references", "Retrievability status", "Evidence retention expiry" ] }, { "id": "ceg-q3", "text": "Is any part of the asserted path dependent on a manual workaround or uncontrolled step?", "kind": "quality", "answer_data": [ "Manual dependency flag", "Control status of the step", "Compensating control reference" ] }, { "id": "ceg-q4", "text": "What would falsify this assertion, and has that test been run?", "kind": "validation", "answer_data": [ "Falsification test description", "Test execution status", "Result reference" ] } ], "data_elements": [ { "id": "ceg-confidence", "name": "Confidence grade", "description": "Graded confidence in the assertion against a published scale, with the grader identity.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-012" ] }, { "id": "ceg-evidence-refs", "name": "Evidence references", "description": "References to retrievable evidence items supporting the assertion.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "ceg-manual-flag", "name": "Manual dependency flag", "description": "Whether the asserted path depends on a manual or uncontrolled step, with its control status.", "value_kind": "boolean", "cardinality": "1", "required": true, "source_refs": [ "SRC-012" ] } ], "artifacts": [], "inline_only_rationale": "Confidence is a graded attribute that must be readable at the point of use during traversal. The evidence items it points to are artifacts owned by their own producing models, so this finding holds references rather than copies." } ] }, { "id": "validation-and-reconciliation", "name": "Validation and Reconciliation", "description": "Evaluating the lineage instance against constraints and measuring how much of the real flow it actually covers.", "source_refs": [ "SRC-003", "SRC-012", "SRC-004" ], "findings": [ { "id": "lineage-instance-validation", "name": "Lineage instance validation", "description": "Structural evaluation of a lineage instance against uniqueness, typing, ordering and impossibility constraints, plus referential checks that every node reference resolves.", "source_refs": [ "SRC-003", "SRC-004", "SRC-008" ], "questions": [ { "id": "liv-q1", "text": "Which constraint profile and schema version was the instance evaluated against?", "kind": "validation", "answer_data": [ "Constraint profile reference", "Schema version reference", "Evaluation engine reference" ] }, { "id": "liv-q2", "text": "Which violations were found, at which severity, and on which nodes or edges?", "kind": "quality", "answer_data": [ "Violation list with severity", "Offending node and edge references", "Rule reference per violation" ] }, { "id": "liv-q3", "text": "Which node references failed to resolve, leaving dangling or orphan nodes?", "kind": "validation", "answer_data": [ "Unresolved reference list", "Orphan node count", "Suspected cause code" ] }, { "id": "liv-q4", "text": "Does a failing validation block acceptance of the assertions, and who decides?", "kind": "decision", "answer_data": [ "Blocking policy code", "Deciding role reference", "Waiver reference where granted" ] } ], "data_elements": [ { "id": "liv-profile", "name": "Evaluated profile reference", "description": "Constraint profile and schema version the instance was evaluated against.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-003" ] }, { "id": "liv-violations", "name": "Violation set", "description": "Violations detected, each with rule reference, severity and offending node or edge.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-003" ] }, { "id": "liv-outcome", "name": "Validation outcome", "description": "Overall pass, pass-with-qualification or fail result for the evaluated instance.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003" ] } ], "artifacts": [ { "id": "lineage-validation-report", "name": "Lineage validation report", "description": "The evidence record of one validation run over a lineage instance: profile and schema version evaluated, violations with severity and rule references, unresolved references and orphan nodes, and the overall outcome with any waiver reference.", "media_or_form": [ "structured validation report", "violation register", "machine-readable result set" ], "serial": true, "identity_strategy": "Identified by the evaluated instance identifier plus the constraint profile reference plus a monotonic run sequence number; the evaluation time is an attribute of the report, not part of its identifier.", "source_refs": [ "SRC-003", "SRC-004" ] } ], "inline_only_rationale": null }, { "id": "coverage-and-reconciliation", "name": "Coverage measurement and reconciliation", "description": "Quantifying how much of the declared perimeter is actually covered by asserted lineage and reconciling asserted paths against independent evidence of the real flow.", "source_refs": [ "SRC-012", "SRC-010" ], "questions": [ { "id": "crc-q1", "text": "What proportion of the declared perimeter has asserted lineage at each granularity level?", "kind": "measurement", "answer_data": [ "Coverage ratio per granularity level", "Denominator definition", "Measurement observation time" ] }, { "id": "crc-q2", "text": "Which asserted paths were reconciled against independent evidence, and which diverged?", "kind": "evidence", "answer_data": [ "Reconciled path set", "Divergence list with magnitude", "Independent evidence reference" ] }, { "id": "crc-q3", "text": "Which known gaps remain unclosed, and what is the accepted remediation horizon?", "kind": "quality", "answer_data": [ "Open gap register", "Accepted risk statement", "Remediation target reference" ] }, { "id": "crc-q4", "text": "How is coverage reported to the accountable role and to supervisory or assurance consumers?", "kind": "process", "answer_data": [ "Reporting cadence", "Recipient role references", "Report scope statement" ] } ], "data_elements": [ { "id": "crc-coverage-ratio", "name": "Coverage ratio", "description": "Measured proportion of the declared perimeter with asserted lineage, per granularity level and with an explicit denominator.", "value_kind": "quantity", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-012" ] }, { "id": "crc-divergences", "name": "Reconciliation divergences", "description": "Differences between asserted lineage and independent evidence of the actual flow.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] }, { "id": "crc-gap-register", "name": "Open gap register", "description": "Known unclosed lineage gaps with accepted risk and remediation horizon.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012", "SRC-010" ] } ], "artifacts": [ { "id": "coverage-and-reconciliation-report", "name": "Lineage coverage and reconciliation report", "description": "A periodic measurement record giving coverage ratios per granularity level with explicit denominators, reconciliation results against independent evidence, divergences with magnitude, and the open gap register with remediation horizons.", "media_or_form": [ "measurement report", "gap register", "reconciliation result set" ], "serial": true, "identity_strategy": "Identified by the perimeter declaration reference plus a monotonic reporting sequence number; period start and end are recorded as attributes with explicit offsets.", "source_refs": [ "SRC-012" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "lineage-record-governance", "name": "Lineage Record Governance", "description": "Governance of the lineage record as an object in its own right: how sensitive it is, who may see which parts, how long it is kept, and which external obligations bind it.", "rationale": "Lineage discloses schema names, business logic and personal-data footprints, so it is not automatically less sensitive than the data it describes. GDPR imposes storage limitation and erasure duties including informing controllers about copies and replications, while the AI Act and supervisory expectations require provenance documentation to be retained. These pull in opposite directions and must be reconciled explicitly rather than assumed.", "source_refs": [ "SRC-010", "SRC-011", "SRC-012", "SRC-006" ], "layers": [ { "id": "sensitivity-and-access", "name": "Sensitivity and Disclosure Scoping", "description": "Classification of the lineage record and the scoping requirements it places on disclosure, with evaluation and enforcement left to the access model.", "source_refs": [ "SRC-011", "SRC-006", "SRC-012" ], "findings": [ { "id": "lineage-record-sensitivity", "name": "Sensitivity of the lineage record", "description": "Classification of a lineage record in its own right, recognising that node names, transformation expressions and masking flags can disclose confidential logic or reveal where personal data flows.", "source_refs": [ "SRC-011", "SRC-006", "SRC-010" ], "questions": [ { "id": "lrs-q1", "text": "What sensitivity classification applies to this lineage record, and is it higher or lower than that of the data it describes?", "kind": "classification", "answer_data": [ "Classification code", "Comparison to subject data classification", "Classification authority reference" ] }, { "id": "lrs-q2", "text": "Which elements of the record could disclose personal data footprints or confidential business logic if released?", "kind": "privacy", "answer_data": [ "Disclosive element list", "Disclosure risk statement", "Redaction requirement per element" ] }, { "id": "lrs-q3", "text": "Which elements must be redacted or generalised in a lower-trust projection of the record?", "kind": "security", "answer_data": [ "Redaction rule set", "Generalisation target level", "Reference to the enforcing access model" ] }, { "id": "lrs-q4", "text": "Does propagating a masking flag along a derivation path change the classification of downstream nodes?", "kind": "constraint", "answer_data": [ "Propagation rule code", "Downstream classification effect", "Reference to the classification-owning model" ] } ], "data_elements": [ { "id": "lrs-classification", "name": "Lineage record classification", "description": "Sensitivity classification of the lineage record itself, referencing the classification scheme owned by the classification model.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-011" ] }, { "id": "lrs-disclosive-elements", "name": "Disclosive element list", "description": "Elements of the record identified as capable of disclosing personal data footprints or confidential logic.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-006", "SRC-011" ] }, { "id": "lrs-redaction-rules", "name": "Redaction requirements", "description": "Declared redaction or generalisation requirements per element for lower-trust projections, to be applied by the access model.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-011" ] } ], "artifacts": [], "inline_only_rationale": "Classification codes and redaction requirements are declarations carried on the record and consumed by the access-control model. Emitting an artifact would look like a policy document and risk implying that this model owns policy evaluation and enforcement, which it does not." }, { "id": "disclosure-scoping-and-exceptions", "name": "Disclosure scoping and exceptions", "description": "The scope levels at which lineage may be disclosed, the default denial posture, and the named exceptions such as regulatory or incident access, with the evaluation decision left to the access model.", "source_refs": [ "SRC-011", "SRC-010", "SRC-012" ], "questions": [ { "id": "dse-q1", "text": "At which scope may a requester read lineage: whole bundle, layer, individual finding or a specific artifact?", "kind": "access", "answer_data": [ "Permitted scope code", "Requester role reference", "Scope justification" ] }, { "id": "dse-q2", "text": "Which named exceptions permit broader disclosure than the default rule, and for how long?", "kind": "exception", "answer_data": [ "Exception name", "Triggering condition", "Time-bounded validity of the exception" ] }, { "id": "dse-q3", "text": "Which model evaluates and enforces the disclosure decision, and what does this model contribute to it?", "kind": "authority", "answer_data": [ "Enforcing model reference", "Attributes contributed by this model", "Explicit non-ownership statement" ] } ], "data_elements": [ { "id": "dse-scope", "name": "Disclosure scope", "description": "Scope level at which a lineage read is permitted, expressed as bundle, layer, finding or artifact.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-011" ] }, { "id": "dse-exception", "name": "Disclosure exception", "description": "Named, time-bounded exception permitting broader disclosure, with its triggering condition.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-010", "SRC-012" ] }, { "id": "dse-enforcer-ref", "name": "Enforcing model reference", "description": "Reference to the access model that evaluates and enforces the disclosure decision.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-011" ] } ], "artifacts": [], "inline_only_rationale": "This finding declares scoping requirements and defers the decision to the access model. Producing an artifact would create an access-decision record, which is precisely the evaluation and audit-trail semantics the relation ledger keeps outside this model." } ] }, { "id": "retention-and-obligation-binding", "name": "Retention and External Obligation Binding", "description": "How long lineage records are kept, how deletion is represented without destroying traceability, and which external obligations bind the lineage held for a subject.", "source_refs": [ "SRC-010", "SRC-011", "SRC-012", "SRC-015" ], "findings": [ { "id": "retention-and-disposition", "name": "Retention, tombstoning and disposition", "description": "The retention rule for lineage records, the tombstone representation used when a referenced node or record must be removed, and the propagation duty toward downstream copies, with execution owned elsewhere.", "source_refs": [ "SRC-011", "SRC-010", "SRC-012" ], "questions": [ { "id": "rdd-q1", "text": "What retention period applies to this lineage record, and which obligation sets it?", "kind": "retention", "answer_data": [ "Retention period value", "Setting obligation reference", "Retention start trigger" ] }, { "id": "rdd-q2", "text": "When a referenced node must be erased, is the lineage record deleted or tombstoned, and what remains readable?", "kind": "lifecycle", "answer_data": [ "Disposition mode code: delete, tombstone or redact", "Residual readable fields", "Tombstone record identifier" ] }, { "id": "rdd-q3", "text": "Which downstream holders of copies or replications must be informed of an erasure, and how are they identified?", "kind": "relationship", "answer_data": [ "Downstream holder set derived by traversal", "Notification obligation reference", "Notification-owning model reference" ] }, { "id": "rdd-q4", "text": "Which party actually executes the deletion in the storage substrate, and where is that execution recorded?", "kind": "authority", "answer_data": [ "Executing party or model reference", "Execution evidence reference", "Explicit non-ownership statement for this model" ] } ], "data_elements": [ { "id": "rdd-retention-period", "name": "Retention period", "description": "Duration for which the lineage record is retained, with the obligation that sets it and the retention start trigger.", "value_kind": "duration", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-010" ] }, { "id": "rdd-disposition-mode", "name": "Disposition mode", "description": "Whether disposition is by deletion, tombstoning or field-level redaction.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-011" ] }, { "id": "rdd-downstream-holders", "name": "Downstream holder set", "description": "Holders of copies or replications derived by downstream traversal, to be notified of an erasure by the model that owns notification.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-011" ] } ], "artifacts": [ { "id": "lineage-disposition-record", "name": "Lineage disposition record", "description": "A record of one disposition action over lineage records: the scope disposed, the mode applied, the fields left readable in a tombstone, the derived downstream holder set handed to the notifying model, and the reference to the party that executed the removal in the storage substrate.", "media_or_form": [ "disposition record", "tombstone entry", "downstream holder list" ], "serial": true, "identity_strategy": "Identified by the disposed scope reference plus a monotonic disposition sequence number; the tombstone retains the original record identifier so that prior readers can detect the disposition.", "source_refs": [ "SRC-011", "SRC-012" ] } ], "inline_only_rationale": null }, { "id": "regulatory-obligation-binding", "name": "External obligation binding", "description": "Which external obligations require lineage for a given subject, what each requires, and which lineage elements satisfy the requirement, recorded as binding parameters rather than as a reproduction of the obligation.", "source_refs": [ "SRC-010", "SRC-012", "SRC-015", "SRC-011" ], "questions": [ { "id": "rob-q1", "text": "Which external obligations require documented provenance for this subject, and under which jurisdiction?", "kind": "requirement", "answer_data": [ "Obligation reference with article or principle", "Jurisdiction code", "Applicability determination" ] }, { "id": "rob-q2", "text": "Which specific lineage elements are relied on to satisfy each obligation?", "kind": "evidence", "answer_data": [ "Obligation to element mapping", "Sufficiency assessment", "Residual gap statement" ] }, { "id": "rob-q3", "text": "Does the obligation require a narrative lineage statement, a machine-readable graph, or both?", "kind": "interoperability", "answer_data": [ "Required form code", "Dual-representation requirement flag", "Consistency rule between forms" ] }, { "id": "rob-q4", "text": "Does the obligation require recording the original purpose of collection for personal data upstream of a derivation?", "kind": "privacy", "answer_data": [ "Original purpose statement reference", "Upstream collection process reference", "Personal data indicator" ] } ], "data_elements": [ { "id": "rob-obligation-ref", "name": "Obligation reference", "description": "Citation of the external obligation requiring provenance for the subject, with jurisdiction and applicability determination.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-010", "SRC-012", "SRC-015" ] }, { "id": "rob-element-map", "name": "Obligation to element mapping", "description": "Mapping from each obligation requirement to the lineage elements relied on to satisfy it, with a sufficiency assessment.", "value_kind": "object", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-010" ] }, { "id": "rob-required-form", "name": "Required representation form", "description": "Whether the obligation expects a narrative statement, a machine-readable graph or both.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-015", "SRC-010" ] }, { "id": "rob-original-purpose", "name": "Original purpose reference", "description": "Reference to the recorded original purpose of collection for personal data upstream of the derivation, held by the processing-records owner.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-010", "SRC-011" ] } ], "artifacts": [ { "id": "obligation-binding-register", "name": "Obligation binding register", "description": "A register, per subject or domain, of the external obligations that require lineage, the required representation form for each, the lineage elements relied on to satisfy them, the sufficiency assessment and any residual gaps. It binds obligations to lineage; it does not reproduce or interpret the obligations themselves.", "media_or_form": [ "binding register", "obligation to element crosswalk", "sufficiency assessment record" ], "serial": false, "identity_strategy": "Identified by the subject master identifier plus the obligation reference set plus a monotonic revision number; each obligation entry cites its authoritative source reference rather than restating its text.", "source_refs": [ "SRC-010", "SRC-012", "SRC-015" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "interoperability-and-exchange", "name": "Interoperability and Exchange", "description": "Making lineage portable: alignment crosswalks to external provenance standards, governance of namespaces and extensions, resolution of nodes across system boundaries, and the integrity of exchanged payloads.", "rationale": "Three independent standard families describe lineage with different identity models and different assumptions about structure. Alignment must be recorded as a crosswalk with declared conformance evidence rather than asserted, and node identity must be resolvable across producers, otherwise a merged graph silently fabricates or severs paths.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-005", "SRC-008", "SRC-009", "SRC-015" ], "layers": [ { "id": "standard-alignment", "name": "Standard Alignment and Extension Governance", "description": "Recorded crosswalks to external provenance vocabularies and the rules governing namespaces and custom extensions.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-008", "SRC-009", "SRC-015" ], "findings": [ { "id": "standard-alignment-crosswalk", "name": "Alignment crosswalk and conformance claims", "description": "Element-by-element mapping between this model and external provenance vocabularies, with explicit lossiness notes and evidenced conformance claims rather than assumed compatibility.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-015" ], "questions": [ { "id": "sac-q1", "text": "Which external vocabulary elements does each local element map to, and in which direction is the mapping lossless?", "kind": "interoperability", "answer_data": [ "Local to external element mapping", "Direction of losslessness", "Lossy element list with the information lost" ] }, { "id": "sac-q2", "text": "Which conformance claims are made against which standard version, and what evidence supports each claim?", "kind": "validation", "answer_data": [ "Conformance claim statement", "Standard version reference", "Supporting test or evidence reference" ] }, { "id": "sac-q3", "text": "Where do the aligned standards conflict, and which one governs in this Dimension?", "kind": "exception", "answer_data": [ "Conflict description", "Governing standard decision", "Deviation record" ] }, { "id": "sac-q4", "text": "How is a narrative lineage statement reconciled with the machine-readable graph for the same subject?", "kind": "quality", "answer_data": [ "Derivation rule from graph to statement", "Consistency check result", "Divergence handling rule" ] } ], "data_elements": [ { "id": "sac-mapping", "name": "Element mapping", "description": "Directed mapping between a local element and an external vocabulary term, with lossiness annotation.", "value_kind": "object", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-004" ] }, { "id": "sac-conformance", "name": "Conformance claim", "description": "Claim of conformance to a named standard version, with the evidence supporting it; absent evidence the claim is recorded as an alignment only.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-003" ] }, { "id": "sac-conflict", "name": "Recorded conflict", "description": "Documented conflict between aligned standards, with the governing decision for this Dimension.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-015", "SRC-005" ] } ], "artifacts": [ { "id": "alignment-crosswalk", "name": "Standard alignment crosswalk", "description": "The versioned mapping table between this model's elements and external provenance vocabularies, annotated with direction of losslessness, evidenced conformance claims, recorded conflicts and the governing decision for each conflict.", "media_or_form": [ "crosswalk table", "mapping specification", "conformance evidence index" ], "serial": false, "identity_strategy": "Identified by the pair of aligned specification references plus their version identifiers plus a monotonic revision number; a new standard version yields a new crosswalk revision rather than an edit.", "source_refs": [ "SRC-002", "SRC-004", "SRC-009" ] } ], "inline_only_rationale": null }, { "id": "namespace-and-extension-governance", "name": "Namespace and extension governance", "description": "Rules for allocating namespaces used in node identity and for defining custom extensions, so that identifiers stay stable and extensions do not collide with governed terms.", "source_refs": [ "SRC-005", "SRC-008", "SRC-002" ], "questions": [ { "id": "neg-q1", "text": "Who allocates a namespace used in node identity, and what makes an allocation immutable once used?", "kind": "authority", "answer_data": [ "Allocating authority reference", "Immutability rule", "Registered namespace list" ] }, { "id": "neg-q2", "text": "What prefix and naming rule prevents a custom extension from colliding with governed terms?", "kind": "constraint", "answer_data": [ "Prefix allocation rule", "Naming pattern", "Collision check procedure" ] }, { "id": "neg-q3", "text": "How is an extension schema version referenced so that a consumer can resolve exactly the version emitted?", "kind": "interoperability", "answer_data": [ "Immutable schema pointer", "Version pinning rule such as a tag or commit reference", "Resolution fallback" ] }, { "id": "neg-q4", "text": "What is the promotion path from a local extension to a governed term, and what changes on promotion?", "kind": "lifecycle", "answer_data": [ "Promotion criteria", "Deprecation handling for the local term", "Backward compatibility commitment" ] } ], "data_elements": [ { "id": "neg-namespace", "name": "Registered namespace", "description": "Namespace registered for use in node identity, with its allocating authority and immutability status.", "value_kind": "identifier", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-005", "SRC-002" ] }, { "id": "neg-extension-prefix", "name": "Extension prefix", "description": "Project-derived prefix allocated to custom extensions to prevent collision with governed terms.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "neg-schema-pointer", "name": "Immutable schema pointer", "description": "Version-pinned pointer to the exact extension schema version emitted, resolvable to a tag or commit rather than a moving branch.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008" ] } ], "artifacts": [ { "id": "namespace-registry-entry", "name": "Namespace and extension registry entry", "description": "A registry entry recording an allocated namespace or extension prefix, its allocating authority, its immutable schema pointer, its status, and any promotion or deprecation decision. It is the reference an agent consults before minting identifiers or defining extensions.", "media_or_form": [ "registry entry", "allocation record", "schema pointer index" ], "serial": false, "identity_strategy": "Identified by the namespace or prefix string itself, which is immutable once allocated and used; superseded entries retain the original string with a status change rather than being reissued.", "source_refs": [ "SRC-005", "SRC-008" ] } ], "inline_only_rationale": null } ] }, { "id": "cross-system-exchange", "name": "Cross-system Resolution and Exchange", "description": "Stitching lineage graphs produced by different systems and exchanging them with verifiable integrity.", "source_refs": [ "SRC-001", "SRC-004", "SRC-005", "SRC-007", "SRC-009" ], "findings": [ { "id": "cross-system-node-resolution", "name": "Cross-system node resolution and stitching", "description": "How nodes asserted independently by different producers are recognised as the same thing or as related specialisations, and what evidence is required before two graphs are joined.", "source_refs": [ "SRC-001", "SRC-002", "SRC-005" ], "questions": [ { "id": "cnr-q1", "text": "On what evidence are two independently asserted nodes judged to denote the same thing?", "kind": "identity", "answer_data": [ "Matching evidence set", "Match confidence grade", "Matching rule reference" ] }, { "id": "cnr-q2", "text": "Is the relationship between two nodes identity, alternate presentation, or specialisation of a more general node?", "kind": "relationship", "answer_data": [ "Relation code: same, alternate or specialization", "Discriminating attributes", "Assertion source" ] }, { "id": "cnr-q3", "text": "How are symlink or mirror identifiers for the same physical dataset recorded without creating duplicate nodes?", "kind": "composition", "answer_data": [ "Symlink identifier set", "Canonical node designation", "Precedence rule" ] }, { "id": "cnr-q4", "text": "What is done when a stitching decision is later found to be wrong?", "kind": "exception", "answer_data": [ "Unstitch procedure reference", "Affected edge set", "Superseding assertion identifiers" ] } ], "data_elements": [ { "id": "cnr-match-evidence", "name": "Match evidence", "description": "Evidence and confidence supporting a judgement that two nodes denote the same thing.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "cnr-relation-code", "name": "Node correspondence code", "description": "Whether nodes are identical, alternates of one another, or one is a specialisation of the other.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "cnr-canonical-node", "name": "Canonical node designation", "description": "Designated canonical node for a set of symlink or mirror identifiers, with the precedence rule applied.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005" ] } ], "artifacts": [], "inline_only_rationale": "Stitching decisions are assertions inside the graph with their own evidence and confidence, and they must be individually reversible. An artifact would batch them into a document and obscure which single decision needs to be superseded." }, { "id": "exchange-payload-integrity", "name": "Exchange payload contract and integrity", "description": "The serialization-neutral contract for a unit of lineage handed to another system, including merge semantics for partial or accumulative delivery and the integrity evidence that binds payload to producer.", "source_refs": [ "SRC-004", "SRC-007", "SRC-008", "SRC-009" ], "questions": [ { "id": "epi-q1", "text": "What is the minimum element set a payload must carry to be independently interpretable?", "kind": "requirement", "answer_data": [ "Mandatory element list", "Optional element list", "Rejection rule for incomplete payloads" ] }, { "id": "epi-q2", "text": "Is the payload a complete self-contained snapshot or an accumulative increment to be merged?", "kind": "process", "answer_data": [ "Payload pattern code: snapshot or accumulative", "Merge key set", "Merge conflict rule" ] }, { "id": "epi-q3", "text": "What integrity evidence binds the payload to its producer and its declared schema version?", "kind": "security", "answer_data": [ "Content digest and algorithm", "Producer identifier", "Declared schema version reference" ] }, { "id": "epi-q4", "text": "How does a consumer detect that a payload it already merged has since been superseded or retracted?", "kind": "state", "answer_data": [ "Supersession signal", "Consumer notification route", "Reference to the notification-owning model" ] } ], "data_elements": [ { "id": "epi-mandatory-set", "name": "Mandatory payload elements", "description": "Minimum element set required for a payload to be interpretable without out-of-band context.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-004", "SRC-008" ] }, { "id": "epi-pattern", "name": "Payload pattern", "description": "Whether the payload is a complete snapshot for its period or an accumulative increment requiring merge.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-007" ] }, { "id": "epi-digest", "name": "Payload digest", "description": "Content digest with algorithm, computed over the canonical serialization and bound to producer and schema version.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-009", "SRC-008" ] } ], "artifacts": [ { "id": "lineage-exchange-payload", "name": "Lineage exchange payload", "description": "A transferable unit of lineage carrying its mandatory element set, its payload pattern and merge keys, its producer identity and declared schema version, and its content digest. It is deliberately serialization-neutral: the same payload can be projected as an event, a document, a graph fragment or a table extract.", "media_or_form": [ "serialization-neutral payload record", "event envelope projection", "graph fragment projection" ], "serial": true, "identity_strategy": "Identified by producer identifier plus a monotonic emission sequence number, content-addressed by the payload digest; the emission time is carried as an attribute with an explicit offset and is never the identifier.", "source_refs": [ "SRC-004", "SRC-008", "SRC-009" ] } ], "inline_only_rationale": null } ] } ] } ] }, "functions": [ { "id": "assert-lineage-edge", "name": "Assert a lineage edge", "description": "Record a new derivation assertion between resolved nodes, carrying its relation type, dependency mode, transformation characterisation, time axes, version pins, capture method, producer and confidence grade.", "inputs": [ "Resolved input and output node references", "Relation type and dependency mode", "Event time and observation time", "Capture method and producer reference" ], "outputs": [ "Persisted lineage assertion with an assigned identifier", "Provisional validation outcome" ], "preconditions": [ "Both endpoint references resolve under the identity priority rules", "The producer is registered and declares an immutable schema version", "The subject falls inside a declared perimeter" ], "effects": [ "A new assertion enters the asserted state", "Coverage measurements for the affected perimeter become stale and are marked for recomputation" ], "source_refs": [ "SRC-001", "SRC-004", "SRC-008" ] }, { "id": "resolve-lineage-node", "name": "Resolve a lineage node reference", "description": "Resolve an identifier, location-derived name or symlink to a canonical lineage node, applying the identity priority order and recording the resolution evidence.", "inputs": [ "Candidate identifier, namespace and name pair, or symlink", "Resolution context such as source-type convention and observation time" ], "outputs": [ "Canonical node reference", "Resolution evidence with confidence grade", "Unresolved status with cause code when no match is found" ], "preconditions": [ "The namespace is registered or explicitly accepted as unregistered with a recorded weakness" ], "effects": [ "A resolution record is attached to the node", "Repeated unresolved outcomes raise an entry in the open gap register" ], "source_refs": [ "SRC-005", "SRC-002", "SRC-004" ] }, { "id": "traverse-lineage", "name": "Traverse lineage upstream or downstream", "description": "Return the set of nodes and edges reachable from a starting node in a given direction, at a given granularity, as of a specified time and version pin, with per-edge confidence and perimeter status.", "inputs": [ "Starting node reference", "Direction and maximum depth", "As-of time and granularity level", "Minimum confidence threshold" ], "outputs": [ "Reachable node and edge set", "Per-edge confidence and capture method", "Perimeter status flags marking opaque segments and unresolved references" ], "preconditions": [ "The starting node resolves", "A perimeter declaration exists for the subject so absence of edges is interpretable" ], "effects": [ "No change to stored assertions; traversal is read-only", "A traversal record may be retained for reuse by impact analysis" ], "source_refs": [ "SRC-001", "SRC-003", "SRC-006" ] }, { "id": "analyse-change-impact", "name": "Analyse impact of a proposed change", "description": "Produce the downstream consumer and field set affected by a proposed schema, semantic or retention change, as decision support for the change process owned elsewhere.", "inputs": [ "Proposed change description and affected node or field references", "Traversal parameters including granularity and confidence threshold" ], "outputs": [ "Impacted downstream node and field set", "Confidence and coverage qualification of the impact set", "Schema change impact statement" ], "preconditions": [ "Field-level lineage is asserted at the required granularity or the shortfall is declared", "Coverage for the affected perimeter has been measured" ], "effects": [ "A schema change impact statement artifact is produced", "No approval, notification or enforcement is performed; those remain with the change and notification owners" ], "source_refs": [ "SRC-005", "SRC-013", "SRC-006" ] }, { "id": "validate-lineage-instance", "name": "Validate a lineage instance", "description": "Evaluate a lineage instance against its declared constraint profile and schema version, checking uniqueness, typing, event ordering, impossibility and referential resolution, and emit a validation report.", "inputs": [ "Lineage instance or scope reference", "Constraint profile reference and schema version" ], "outputs": [ "Validation outcome code", "Violation set with severity and rule references", "Lineage validation report" ], "preconditions": [ "The instance has been normalised under the declared normalisation rule set" ], "effects": [ "A validation report artifact is created", "Assertions failing a blocking rule are held from acceptance pending a waiver decision by the accountable role" ], "source_refs": [ "SRC-003", "SRC-004" ] }, { "id": "measure-lineage-coverage", "name": "Measure coverage and reconcile", "description": "Compute coverage ratios per granularity level against the declared perimeter, reconcile asserted paths against independent evidence, and record divergences and open gaps.", "inputs": [ "Perimeter declaration reference", "Reporting period with explicit offsets", "Independent evidence references" ], "outputs": [ "Coverage ratios with explicit denominators", "Divergence list", "Coverage and reconciliation report" ], "preconditions": [ "A perimeter and granularity declaration exists for the subject", "Independent evidence is retrievable for at least the reconciled path sample" ], "effects": [ "A coverage and reconciliation report artifact is created", "The open gap register is updated with new or closed gaps" ], "source_refs": [ "SRC-012", "SRC-010" ] }, { "id": "supersede-lineage-assertion", "name": "Supersede or retract an assertion", "description": "Replace an incorrect assertion with a corrected one, or retract it, preserving the superseded record and making the change detectable by prior readers.", "inputs": [ "Assertion identifier to supersede", "Corrected assertion content or retraction reason", "Correcting agent reference" ], "outputs": [ "New assertion identifier with a supersedes link", "Updated state on the prior record", "Supersession signal for consumers" ], "preconditions": [ "The correcting agent holds the authority recorded for the subject", "The prior assertion exists and is not already tombstoned" ], "effects": [ "The prior record moves to superseded or retracted state and is retained, not erased", "Downstream consumers are made able to detect the change; issuing the notification remains with the notification owner" ], "source_refs": [ "SRC-001", "SRC-009", "SRC-003" ] }, { "id": "export-lineage-payload", "name": "Export a lineage exchange payload", "description": "Project a scoped set of assertions into a serialization-neutral exchange payload under a named alignment profile, with producer identity, declared schema version and content digest.", "inputs": [ "Scope selector and as-of parameters", "Target alignment profile reference", "Redaction level for the destination trust tier" ], "outputs": [ "Lineage exchange payload with digest", "Lossiness note listing elements dropped by the profile" ], "preconditions": [ "An alignment crosswalk exists for the target profile version", "The redaction requirements for the destination have been supplied by the access model" ], "effects": [ "An immutable payload is emitted and recorded with its digest", "Elements not expressible in the target profile are reported as lossy rather than silently dropped" ], "source_refs": [ "SRC-004", "SRC-008", "SRC-009", "SRC-002" ] }, { "id": "apply-lineage-disposition", "name": "Apply retention disposition to lineage records", "description": "Mark lineage records for deletion, tombstoning or field-level redaction on expiry of their retention period or on a lawful erasure trigger, and hand the derived downstream holder set to the notifying model.", "inputs": [ "Scope reference and disposition trigger", "Applicable retention rule and obligation reference", "Disposition mode" ], "outputs": [ "Lineage disposition record", "Tombstone entries retaining original record identifiers", "Downstream holder set derived by traversal" ], "preconditions": [ "No overriding retention obligation requires continued retention of the same records", "The accountable role has authorised the disposition mode" ], "effects": [ "Records enter tombstoned or redacted state within this model's own store", "Physical removal in the storage substrate and any notification to downstream holders are executed by the adopting Dimension's storage and notification owners, not by this model" ], "source_refs": [ "SRC-011", "SRC-010", "SRC-012" ] } ], "composition": [ { "target": "WM-XCT-012", "relation": "CHILD", "purpose": "WM-DAT-006 is registered as a child of the cross-cutting traceability parent and inherits its generic provenance vocabulary, agent typing and evidence grading conventions. This model specialises them for dataset and field derivation only; generic identity, authority and conflict-resolution machinery is not duplicated here.", "required": true, "source_refs": [ "SRC-001", "SRC-002" ] }, { "target": "WM-DAT-001", "relation": "COMPOSE", "purpose": "Per the known-relation ledger, WM-DAT-001 composes this lineage pattern so that a dataset carries its lineage context. In that direction WM-DAT-006 supplies derivation assertions keyed on dataset and field references plus version pins; it does not supply, override or restate dataset definition, schema, distribution or catalog lifecycle, which remain owned by WM-DAT-001.", "required": true, "source_refs": [ "SRC-009", "SRC-013", "SRC-004" ] }, { "target": "WM-DAT-005", "relation": "COMPOSE", "purpose": "Per the known-relation ledger, WM-DAT-005 composes this pattern so that a pipeline emits lineage. WM-DAT-006 receives job and run references and a read-only carried run outcome used to qualify assertion completeness; run state transitions, scheduling, retries, failure handling and operational monitoring remain owned by WM-DAT-005.", "required": true, "source_refs": [ "SRC-007", "SRC-004" ] }, { "target": "W3C PROV-DM and PROV-O (http://www.w3.org/ns/prov#)", "relation": "ALIGN", "purpose": "Alignment to the provenance conceptual core and its IRI namespace for Entity, Activity, Agent, the derivation relations and the qualified-influence pattern. Recorded as a crosswalk with declared lossiness; no conformance is claimed without evidence recorded in the alignment crosswalk artifact.", "required": false, "source_refs": [ "SRC-001", "SRC-002", "SRC-003" ] }, { "target": "OpenLineage specification (Job, Run, Dataset, facets)", "relation": "ALIGN", "purpose": "Alignment to the runtime lineage exchange model for dataset and job naming, run identifiers, column lineage and facet extension rules. This model adopts the facet extension discipline and naming conventions as bindings; it does not adopt OpenLineage run-state semantics as its own lifecycle.", "required": false, "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-008", "SRC-007" ] }, { "target": "W3C DCAT 3 catalog vocabulary", "relation": "ALIGN", "purpose": "Alignment for interchange of dataset, distribution, dataset series, versioning, qualified relation, provenance statement and checksum terms with catalog consumers. Catalog records, publication and discovery remain outside this model.", "required": false, "source_refs": [ "SRC-009" ] }, { "target": "Logging and audit-record model (registry identifier unresolved)", "relation": "REFERENCE", "purpose": "Lineage assertions may be cited by regulatory logging and audit records, and disposition actions may need to be evidenced there. This model carries the reference and the binding parameters only; log capture, tamper-evidence, log retention and audit-trail semantics are owned entirely by the target and are listed in out_of_scope.", "required": false, "source_refs": [ "SRC-010", "SRC-011" ] }, { "target": "Access control and data classification model (registry identifier unresolved)", "relation": "REFERENCE", "purpose": "This model declares the sensitivity classification of a lineage record, the disclosive elements and the required redaction and scoping parameters. Evaluation of an access request, enforcement of the decision and any enforcement audit trail are owned by the target model.", "required": false, "source_refs": [ "SRC-011", "SRC-006" ] } ], "serviceLayers": { "dimension": { "owner_package_requirements": [ "The adopting Dimension must name a data product owner or data steward accountable for lineage accuracy per subject, distinct from the owner of the underlying data, and record the escalation path when an attestation is rejected.", "The adopting Dimension must publish a perimeter and granularity declaration before any coverage or completeness claim is made, including the world assumption applied to missing edges and the list of recognised manual steps.", "The adopting Dimension must register every namespace and extension prefix used in node identity, with an allocating authority and an immutable schema pointer, before identifiers minted under them are exchanged.", "The adopting Dimension must state, per subject class, which external obligations require lineage, in which representation form, and which retention rule governs the resulting records.", "The adopting Dimension must name the storage and notification owners that execute physical deletion and downstream erasure notification, because this model marks disposition but does not execute it." ], "namespace_guidance": "Node identifiers are minted under registered namespaces. Where a governed IRI scheme exists it is preferred for cross-organisation exchange; where identity is location-derived, the namespace must follow the declared source-type convention and be treated as a resolvable binding rather than as primary identity. A namespace string is immutable once used in an exchanged assertion: changing its format disconnects existing nodes, so a format migration must be executed as an explicit rebinding with retained alternate identifiers rather than as an in-place rename. Local extension terms carry a Dimension-specific prefix that cannot collide with governed vocabulary terms, and each extension declares an immutable, version-pinned schema pointer.", "registry_links": [ "Model registry entry vr.wm-dat-006 with nav path NAV.INF.DAT.LIN and parent WM-XCT-012", "Namespace and extension registry entries governing allocated namespaces, prefixes and immutable schema pointers", "Obligation binding register linking subject classes to the external obligations that require lineage and their required representation form", "Alignment crosswalk register holding versioned mappings and evidenced conformance claims against external provenance vocabularies" ] }, "canon_and_patch": { "canonicalization_rules": [ "Normalise a lineage instance before comparison or storage: expand shorthand relations into their qualified form, apply the declared inference set, and enforce uniqueness so that the same generation or usage is stated once. Two instances are equivalent when their normal forms correspond node for node and edge for edge.", "Canonical identity for every node is the highest-priority identifier available under the identity priority order; location-derived names, symlinks and display labels are retained as alternate bindings on the canonical node and never promoted to primary identity.", "All timestamps are normalised to RFC 3339 with seconds and an explicit offset; values received as local civil time without an offset are rejected rather than assumed to be UTC.", "Codes for relation type, dependency mode, transformation subtype, capture method and disposition mode are canonicalised against the registered vocabulary version in force at observation time, and the vocabulary version is recorded with the value." ], "patch_rules": [ "Assertions are append-only. A correction is expressed as a new assertion carrying a supersedes link, a correction reason and the correcting agent; the superseded record is retained in superseded state and remains readable to consumers that already read it.", "A retraction is a first-class record, not a deletion. It preserves the retracted assertion identifier so that prior readers can detect the retraction, and it names the affected edge set.", "Accumulative payloads are merged on the declared merge key set; late-arriving assertions are ordered by event time with observation time as the tiebreaker, and a merge that would violate an event-ordering constraint is rejected with a violation record rather than silently reordered.", "Emitted exchange payloads are immutable once released. A changed payload is a new emission with a new sequence number and digest; the prior emission is superseded, never edited in place." ], "compatibility_rules": [ "Adding an optional element, a new vocabulary code or a new extension facet is a compatible change. Removing an element, narrowing a cardinality, changing the meaning of an existing code or changing a namespace format is a breaking change requiring a new major model version and a migration crosswalk.", "A conformance claim is valid only against a named external standard version with recorded evidence; if the external standard issues a new version the claim reverts to an alignment until re-evidenced.", "Consumers must tolerate unknown extension facets and preserve them unchanged when relaying a payload, so that intermediary systems do not silently strip provenance.", "Deprecated vocabulary codes remain resolvable for at least the longest retention period applicable to records that used them, with a documented replacement mapping." ] }, "artifact_rules": { "identity_priority": [ "Authoritative master-system identifier issued by the system of record for the referenced object, such as the catalog table identifier for a dataset, the table-format field identifier for a column, or the orchestrator run identifier for a run; this is the first choice whenever it exists.", "Governed global identifier or IRI from a normative registry or namespace, such as a provenance IRI, a catalog dataset IRI, or a namespace and name pair minted under a registered namespace.", "UUID or ULID minted by the adopting Dimension, recorded together with the resolution evidence that binds it to a master-system identifier once one becomes available.", "Never an identifier: a date, a load or emission timestamp, a file path fragment, a display label or a human-readable title. Location-derived names are bindings and must remain resolvable to one of the three identifier classes above." ], "timestamp_rule": "All time values are RFC 3339 date-time values with explicit seconds and an explicit numeric offset or the literal Z; local civil time without an offset is rejected. Event time (when the derivation actually occurred) and observation or ingestion time (when the assertion was captured by its producer and when it was accepted into the store) are held in separate fields and are never conflated or defaulted to one another. Nominal or scheduled time, where it differs from both, is a third distinct field. Where only a coarser precision is known, the precision is declared alongside the value rather than padded with fabricated seconds.", "serial_naming_rule": "Serial artifacts (processing context manifests, accuracy attestations, schema change impact statements, validation reports, coverage and reconciliation reports, disposition records and exchange payloads) are named as artifact-kind plus the subject or scope master identifier plus a monotonic sequence number, optionally followed by the producer identifier. A date or period is never part of the name; period start and end and emission time are recorded as attributes with explicit offsets, and ordering is carried by the sequence number so that late-arriving items do not disturb the sequence.", "integrity_rule": "Every artifact carries a content digest with a named algorithm computed over its canonical serialization, bound to the producer identifier and the declared schema version. Artifacts are immutable once emitted: corrections are issued as a new artifact with a supersedes link, never as an in-place edit. A consumer that cannot recompute the digest treats the artifact as unverified and must downgrade any confidence derived from it." }, "policies": [ "Lineage is asserted, not assumed. Every edge records its capture method, producer and confidence grade, and a graph without a perimeter declaration may not be used to support a completeness or coverage claim.", "This model records assertions about processing; it never executes, triggers, approves or enforces anything. Impact analysis, validation and coverage measurement produce decision-support evidence for owners elsewhere, and disposition marks records without performing physical removal.", "Referenced objects are carried as references and version pins only. Cached copies of a schema, job definition or run record inside a lineage assertion are bindings with an observation time, are never authoritative, and must not be read as the owning model's current state.", "A lineage record is classified on its own merits, because node names, expressions and masking flags can disclose confidential logic and personal-data footprints independently of the data described.", "Alignment to an external standard is recorded as a crosswalk with declared lossiness. Conformance is claimed only with recorded evidence against a named standard version, and conflicts between aligned standards are documented with an explicit governing decision." ], "crud": { "read": [ "Reads are scoped: a requester may be granted bundle, layer, finding or artifact scope, and traversal results are filtered to the granted scope with omitted segments marked as withheld rather than silently absent.", "Every traversal read returns per-edge capture method, confidence grade and perimeter status so that a consumer can distinguish a confirmed path from an inferred or unobserved one.", "As-of reads accept a time and a version pin; a read without an as-of parameter returns the current validity window and must state that it did so.", "Reads of redacted projections apply the redaction requirements declared on the record; the redaction level applied is reported with the result." ], "create": [ "An assertion may be created only when both endpoint references resolve under the identity priority order, or when the shortfall is recorded as an unresolved reference with a cause code.", "Creation requires a registered producer, an immutable schema version reference, an event time and an observation time.", "Creation outside a declared perimeter is permitted but is flagged, and it does not count toward coverage until the perimeter declaration is revised.", "Bulk creation from an accumulative payload merges on the declared merge key set and rejects, with a violation record, any assertion that would breach an event-ordering or uniqueness constraint." ], "update": [ "Assertions are not updated in place. A correction creates a new assertion with a supersedes link, a correction reason and the correcting agent, and moves the prior record to superseded state.", "Governance attributes that are not part of the assertion content, such as confidence regrade, attestation outcome and classification, may be updated with a new observation time and the updating agent recorded.", "A stitching decision found to be wrong is reversed by superseding the correspondence assertion; the edges it joined are re-evaluated and any resulting orphan is recorded.", "Any update that would change the meaning of an already-exchanged payload requires a new payload emission rather than a silent revision." ], "delete": [ "Lineage records are retained for the period set by the applicable obligation recorded in the obligation binding register; where two obligations conflict, the longer retention prevails and the conflict is documented.", "Default disposition is tombstoning, not deletion. A tombstone retains the original record identifier, the relation type, the time axes and the disposition reason so that traversal remains structurally sound and prior readers can detect the change, while identifying or disclosive content is removed or redacted.", "Physical deletion is permitted only where no retention obligation applies, or where a lawful erasure trigger overrides retention; in that case field-level redaction of the identifying elements is preferred over removal of the edge, so that the derivation path itself survives.", "Execution of deletion is outside this model's boundary. This model marks disposition and derives the downstream holder set by traversal; physical removal in the storage substrate is executed by the adopting Dimension's storage owner under its retention policy, notification of downstream holders of copies or replications is executed by the notification owner, and any evidencing of that execution belongs to the logging and audit-record model.", "A disposition record is created for every disposition action and is itself retained beyond the disposed records, since it is the only remaining evidence that the records existed." ] }, "roles": [ { "name": "Data product owner or data steward", "responsibilities": [ "Hold accountability for the accuracy of the lineage recorded for the subject", "Approve the perimeter and granularity declaration and its revisions", "Sign or reject the periodic lineage accuracy attestation and accept residual gaps with a stated remediation horizon", "Authorise the disposition mode when retention expires or an erasure trigger applies" ] }, { "name": "Lineage producer operator", "responsibilities": [ "Register the producer and maintain its immutable schema version references", "Emit assertions with event time, observation time, capture method and confidence grade populated", "Declare the known blind spots of the capture method and report degradation in capture coverage", "Re-emit corrected assertions as supersessions rather than editing prior emissions" ] }, { "name": "Lineage registrar", "responsibilities": [ "Allocate and record namespaces and extension prefixes and enforce their immutability once used", "Maintain the alignment crosswalk register and the evidence supporting each conformance claim", "Adjudicate producer precedence when producers assert conflicting lineage for the same endpoints", "Operate the promotion and deprecation path for local extension terms" ] }, { "name": "Lineage assurance reviewer", "responsibilities": [ "Run instance validation against the declared constraint profile and publish the validation report", "Measure coverage per granularity level and reconcile asserted paths against independent evidence", "Maintain the open gap register and report unresolved references and orphan nodes", "Recommend blocking or waiver decisions on failing validations to the accountable role" ] }, { "name": "Privacy and disclosure controller", "responsibilities": [ "Classify lineage records and identify disclosive elements independently of the classification of the described data", "Specify redaction and generalisation requirements for lower-trust projections", "Define named, time-bounded disclosure exceptions and their triggering conditions", "Refer every access request to the access model for evaluation and enforcement rather than deciding it here" ] } ], "access": { "default_rule": "Deny by default. Read access to lineage is granted per scope to an identified requester with a stated purpose; the default granted scope is the narrowest that satisfies the purpose, and withheld segments are marked as withheld so the requester cannot mistake redaction for absence of derivation. Write access is restricted to registered producers for assertion creation and to the accountable role or registrar for governance attributes. All decisions are evaluated and enforced by the access model; this model supplies classification, disclosive-element and redaction parameters only.", "scopes": [ "bundle", "layer", "finding", "artifact" ], "exceptions": [ "Regulatory or supervisory access: broader scope granted to a named authority for a stated obligation, time-bounded, with the obligation reference recorded on the grant.", "Incident and breach investigation: temporary downstream-traversal access granted to identify holders of copies or replications, limited to node references and edges, with expressions and payload content withheld.", "Data subject request handling: scoped traversal to locate personal-data footprints, returning node and edge references only, never the underlying data values.", "Assurance and validation review: read access across scopes for the assurance reviewer, limited to the perimeter under review and logged as a review-purpose grant.", "Emergency break-glass for a production data incident: time-bounded, requiring after-the-fact ratification by the accountable role; a grant that is not ratified within the declared window is recorded as an unratified exception." ], "audit_requirements": [ "Every access grant, denial and exception invocation must be recorded with requester identity, scope, purpose, redaction level applied and RFC 3339 grant time with an explicit offset. Capture, storage, tamper-evidence and retention of those records are owned by the logging and audit-record model, not by this model.", "Every disclosure made under an exception must be traceable to the exception name, its triggering condition and its expiry, and unratified break-glass grants must be reportable to the accountable role.", "This model must expose, for each read, the redaction level applied and the withheld-segment markers, so that the enforcing model's records are reconcilable against what was actually returned.", "Requests for evidence of enforcement are answered by reference to the logging and audit-record model; this model does not maintain an independent enforcement audit trail." ] }, "agents_bootstrap": { "filename": "AGENTS.md", "required_fields": [ "Name", "Type", "Specification URL", "Storage type URL", "Interface URL", "Processes URL", "Model ID and registry identifier", "Owner or maintainer role", "Parent and composition links", "Alignment crosswalk URL", "Namespace registry URL", "Retention and disposition policy URL" ], "read_order": [ "AGENTS.md at the package root, to establish Name, Type and the four resolvable URLs before any other read", "Specification URL, for the model scope statement, boundary notes and the in-scope and out-of-scope lists that constrain what may be written here", "Storage type URL, for the concrete projection in use (graph store, event log, catalog table, document store or repository files) and its canonicalization rules", "Interface URL, for the callable surface, scope enforcement and the access parameters the interface expects", "Processes URL, for assertion, supersession, validation, coverage measurement and disposition procedures and their owning roles", "Namespace registry URL and alignment crosswalk URL, before minting any identifier or emitting any exchange payload" ] } }, "coverage": { "claim": "Audited as a single-provider reviewable draft. Internal arithmetic reconciles exactly: 7 bundles, 14 layers (2 per bundle), 28 findings, 109 questions (kind histogram sums to 109), 12 artifacts, 9 functions, 15 sources — all matching the declared provider_counts; the artifact-or-rationale rule (artifacts with null rationale, or empty array with substantive rationale, never both) holds for all 28 findings. Boundary separation from WM-DAT-001 (schema/catalog), WM-DAT-005 (orchestration and run lifecycle), SLSA build provenance, the audit-record model and the access-control model survives adversarial reading: run state appears only as a carried read-only outcome, and the two access findings are deliberately artifact-free. Coverage is claimed only for the derivation-assertion surface grounded in the 15 cited sources. Row-level, streaming-window and ML-feature lineage remain declared gaps; ISO 19115-1 and ISO/IEC 11179 were not retrieved; non-EU jurisdictions are unenumerated. No completeness is claimed against those gaps, against unverified live source versions, or against whatever an independent second provider would have surfaced.", "confidence": "medium", "checklist": [ { "dimension": "identity", "status": "covered", "notes": "Four node kinds each have an identity finding, with the identity priority order fixed in artifact_rules and location-derived names explicitly demoted to bindings. Grounded in PROV identified objects, OpenLineage namespace and name plus run UUIDs, and Iceberg table UUID and immutable field IDs." }, { "dimension": "lifecycle", "status": "covered", "notes": "Lifecycle is modelled for the lineage record itself (asserted, confirmed, superseded, retracted, tombstoned) and for extension terms. Pipeline run lifecycle is deliberately excluded and carried only as a read-only outcome, per the WM-DAT-005 relation rationale." }, { "dimension": "relationships", "status": "covered", "notes": "Edge typing, dependency mode, roles, collections and membership, delegation chains, cross-system correspondence and rollup rules are all modelled, with PROV relations and the OpenLineage column lineage facet as the primary basis." }, { "dimension": "temporal", "status": "covered", "notes": "Event time, observation time, ingestion time and nominal time are separated; validity windows, supersession, event-ordering constraints and late-arrival merge rules are modelled. RFC 3339 with seconds and explicit offset is mandated and local civil time without offset is rejected." }, { "dimension": "provenance", "status": "covered", "notes": "Provenance of the lineage record itself is explicit: capture method with declared blind spots, producer identity with immutable schema version, meta-provenance bundle reference, and producer precedence for conflicting assertions." }, { "dimension": "ownership", "status": "covered", "notes": "Attribution, association and delegation for the processing are separated from named accountability for the lineage record, with a periodic accuracy attestation artifact and a defined escalation on rejection. Grounded in PROV agent relations and supervisory governance expectations." }, { "dimension": "validation", "status": "covered", "notes": "Structural validation against uniqueness, typing, ordering and impossibility constraints with a normalisation prerequisite, plus referential resolution checks, a validation report artifact and an explicit blocking-versus-waiver decision owner." }, { "dimension": "access", "status": "covered", "notes": "Deny-by-default with the four scopes, named time-bounded exceptions, withheld-segment marking so redaction is not read as absence, and explicit deferral of evaluation and enforcement to the access model. Enforcement audit records are referenced, not owned." }, { "dimension": "retention and deletion", "status": "covered", "notes": "Retention set by the obligation register with longest-prevails conflict handling, tombstoning as default disposition preserving identifier and structure, field-level redaction preferred over edge removal, a retained disposition record, and explicit assignment of physical deletion to the storage owner and downstream notification to the notification owner." }, { "dimension": "interoperability", "status": "covered", "notes": "Evidenced crosswalks to PROV, OpenLineage and DCAT with declared lossiness, namespace and extension governance with immutable schema pointers, unknown-facet preservation on relay, and a serialization-neutral payload contract with merge semantics and digest binding." }, { "dimension": "granularity and perimeter", "status": "covered", "notes": "Granularity levels and the observed perimeter are a first-class declaration with an explicit world assumption, so that a missing edge can be distinguished from an unobserved one and coverage claims have a defined denominator." }, { "dimension": "measurement", "status": "covered", "notes": "Coverage ratios per granularity level with explicit denominators, reconciliation against independent evidence, divergence recording and an open gap register with remediation horizons." }, { "dimension": "security", "status": "covered", "notes": "Payload integrity through digest bound to producer and schema version, immutability of emitted artifacts, unverified-artifact downgrade rule, and redaction requirements for lower-trust projections." }, { "dimension": "privacy", "status": "covered", "notes": "Lineage records classified independently of the described data; disclosive-element identification; masking-flag propagation questions; original purpose of collection referenced rather than restated; data subject request traversal returning references only, never values." }, { "dimension": "authority", "status": "covered", "notes": "Namespace allocation authority, producer precedence adjudication, disposition authorisation, waiver decisions and delegation chains are each assigned to a named role, with evaluation and enforcement of access decisions explicitly held outside this model." }, { "dimension": "row-level lineage", "status": "gap", "notes": "Row and record level derivation is acknowledged as a granularity level but has no cross-system normative grounding. The only primary support retrieved is table-format specific (Iceberg v3 _row_id and _last_updated_sequence_number), which does not generalise across systems, so no canonical structure is asserted for it." } ], "known_omissions": [ "Row-level and record-level lineage semantics are not modelled beyond a declarable granularity level; the only primary evidence retrieved is table-format specific and does not generalise.", "Streaming and continuous-processing lineage is handled only through an observation window and a snapshot-versus-accumulative pattern code; windowing, watermark and reprocessing semantics are not modelled and would need a dedicated treatment.", "Machine-learning feature and model lineage (feature stores, training-run to model-artifact derivation, evaluation dataset lineage) is not modelled; MLCommons Croissant and the SPDX AI and dataset profiles were not consulted in this research pass.", "ISO 19115-1 LI_Lineage, LI_ProcessStep and LI_Source could not be retrieved directly because the ISO catalogue returned HTTP 403, and the INSPIRE lineage requirement is cited via a retained-EU-law republication rather than the authentic EUR-Lex text; the structured process-step model those standards define is therefore represented only indirectly.", "ISO/IEC 11179 data element registration was not consulted, so field-node identity is grounded in table-format field IDs and column lineage facet naming rather than in a registry-theoretic account of data element identity.", "Registry identifiers for the logging and audit-record model and for the access control and classification model are unresolved; both composition targets are named descriptively and must be rebound once those registry entries exist.", "Confidence grading for inferred lineage has no normative source in the retrieved material. The grading scale is declared as a required element but its content is left to the adopting Dimension and is marked as a gap rather than presented as canonical.", "Cost, latency and volume propagation along lineage paths, and lineage-driven access propagation, are excluded as they belong to operational and access models respectively." ], "conflicts": [ "PROV-CONSTRAINTS treats a provenance instance as a consistent history subject to event-ordering constraints, while OpenLineage explicitly permits accumulative, partial and out-of-order events that consumers combine. A store that validates strictly on arrival will reject legitimate late-arriving lineage; this model resolves it by validating after merge and normalisation and by recording ordering violations as violations rather than silently reordering.", "INSPIRE and the ISO tradition define lineage as a free-text statement on process history and overall quality, while PROV and OpenLineage define a machine-readable graph. A single lineage field cannot satisfy both. This model requires the obligation binding register to declare the required form and, where both are required, a consistency rule deriving the statement from the graph.", "OpenLineage dataset identity is derived from physical location (namespace plus name), whereas this model's identity priority puts the authoritative master-system identifier first. The OpenLineage documentation itself warns that switching a namespace format disconnects existing lineage nodes, confirming the fragility; this model resolves the conflict by treating location-derived names as retained bindings rather than as identity.", "Table formats assign field IDs that are unique and immutable under rename, while the column lineage facet keys input and output fields by name. A rename therefore preserves lineage in the table format but breaks it in the exchange format unless a rebinding is recorded. This model requires both a stable field identifier and the observed name.", "GDPR storage limitation and the right to erasure, including the duty to inform other controllers about copies and replications, pull toward removing lineage records, while AI Act technical documentation and supervisory expectations on risk data pull toward retaining them. The retrieved sources do not resolve this; this model resolves it operationally through tombstoning and field-level redaction and by requiring the conflict to be documented, which is a design decision and not a conformance claim.", "DCAT offers both a narrative dcterms:provenance statement and structured prov:wasGeneratedBy links for the same resource, with no rule preventing them from disagreeing. This model requires the narrative form to be derived from the structured form when both are published." ], "regional_assumptions": [ "The legal anchors used are EU instruments (the AI Act, GDPR and the INSPIRE metadata implementing regulation). Equivalent obligations in other jurisdictions were not enumerated, and the obligation binding register is the intended extension point for them.", "BCBS 239 is a supervisory standard whose scope and implementation timetable are set by national supervisors and which primarily binds globally and domestically systemically important banks. Its use here as evidence for governance, reconciliation and manual-workaround control is a generalisation of supervisory expectation, not a claim of universal applicability.", "The INSPIRE lineage requirement is cited from a retained-EU-law republication by a national authority; the authentic EU text should be substituted before the citation is relied on in an EU regulatory context.", "Time handling assumes RFC 3339 offsets are available from source systems. Systems that record only local civil time, or that change offset across a daylight-saving boundary during a processing window, require an explicit offset resolution rule that this model requires but does not supply.", "The identity priority assumes a system of record exists that issues stable identifiers. In estates dominated by file drops and ad hoc extracts, only the third priority is achievable and the resulting weakness must be recorded on each affected node." ], "adversarial_checks": [ "Checked every bundle, layer, finding and function against the WM-DAT-005 relation rationale. Run states appear only as a carried read-only outcome inside run-reference-binding, with an explicit non-ownership statement; no finding models scheduling, retries, task dependency resolution or run state transitions, and no function triggers or controls execution.", "Checked every node against the WM-DAT-001 relation rationale. Dataset and field findings hold references, version pins and observed names only. An earlier draft carried a schema description finding; it was removed because it reproduced the dataset schema, and the residual need is met by a cached binding with an observation time plus a boundary note.", "Checked that no finding or function claims audit-trail ownership. Access audit requirements are written as obligations discharged by the logging and audit-record model, and the disclosure finding produces no access-decision record, because a reference to an audit record does not transfer audit semantics.", "Checked that no finding claims policy evaluation or enforcement. lineage-record-sensitivity and disclosure-scoping-and-exceptions were deliberately kept artifact-free so that they cannot be mistaken for policy documents, and both name the enforcing model.", "Rejected an attractive lineage-based data quality scoring layer. Propagating quality scores along edges would import data content quality, which belongs to a sibling; only quality of the lineage assertion (capture method, confidence, coverage, reconciliation) is retained.", "Rejected a lineage graph storage and query layer covering traversal indexes, graph databases and query languages. These are storage and interface projections, and including them would breach the format-neutrality rule.", "Tested the identity rule against a counterexample: OpenLineage identity is location-derived and its own documentation warns that a namespace format change disconnects nodes. This falsifies location-derived naming as primary identity, so it was demoted to a binding and the identity priority order was left unchanged.", "Tested the deletion rule against the conflicting retention obligations and confirmed that this model marks disposition and derives the downstream holder set but assigns physical removal to the storage owner and notification to the notification owner, with the disposition record retained beyond the records it disposes.", "Verified that every local identifier is lower-kebab-case, unique across bundles, layers, findings, questions, data elements, artifacts and functions, and contains no date-like component; verified that every finding either declares artifacts with a null rationale or declares an empty artifact array with a substantive rationale, never both." ] }, "researchAdjudication": { "providerMode": "single-provider-waiver", "activeProviders": [ "claude" ], "waivedProviders": [ "grok" ], "providerPolicy": { "contract_version": "1.0.0", "mode": "single-provider-waiver", "effective_at": "2026-08-29T09:06:27Z", "scope": "Queued subject-model research from WM-XCT-013 onward", "active_providers": [ "claude" ], "waived_providers": [ { "provider": "grok", "authorized_by": "repository owner", "authorized_at": "2026-08-29T09:06:27Z", "reason": "The repository owner explicitly instructed the research queue to continue without Grok after repeated structured-output failures." } ], "review_rule": "Claude-only results require a separate no-tools adversarial audit and remain reviewable drafts with a visible single-provider hold." }, "boundaryDecision": { "entry_kind": "pattern", "status": "accepted", "rationale": "The pattern kind survives challenge: the frozen ledger gives two COMPOSE parents (WM-DAT-001 dataset includes lineage context, WM-DAT-005 pipeline emits lineage), the model is explicitly storage- and serialization-neutral, and its aggregate root is the lineage assertion rather than any business entity with its own domain lifecycle. The strongest counter-argument is bundle six, Lineage Record Governance, which treats the lineage record as a managed object with sensitivity, retention, disposition and named stewardship — surface that reads more like an entity than a composable pattern. It does not force reclassification here because those findings hold declarations only and defer evaluation, enforcement, physical deletion and notification to named neighbours. A future split into an assertion pattern plus a generic governed-record pattern is recorded as deferred, not executed. Registry review_state remains boundary-review-required and both relations remain candidate, so the entry kind publishes as accepted-candidate, not settled." }, "decisions": [ { "concept": "Aggregate root is the lineage assertion (edge), not the dataset, the run or the graph store", "disposition": "accepted with mandatory vocabulary disambiguation", "rationale": "Edge-as-root is what keeps WM-DAT-001 and WM-DAT-005 outside the boundary and it is consistently applied in the derivation bundle. But the text uses assertion, lineage record, lineage instance, bundle and exchange payload for materially different scopes — validation runs on an instance, retention and sensitivity apply to a record, disclosure applies to a bundle. The synthesizer must require an explicit statement defining record, instance, bundle and payload as named scopes over assertions before review." }, { "concept": "Entry kind pattern versus splitting out the Lineage Record Governance bundle", "disposition": "accepted; split deferred", "rationale": "Two COMPOSE parents plus declared format-neutrality fit the pattern kind. Bundle six governs the lineage record as a managed object, which strains that classification, but its findings declare and defer rather than own. Record the potential split against a future generic governed-record pattern instead of executing it in single-provider mode." }, { "concept": "Composition contract with WM-DAT-001 and WM-DAT-005", "disposition": "accepted as candidate only", "rationale": "Non-ownership holds on inspection: no finding models scheduling, retries, task dependency resolution or run state transitions, and no function triggers execution; dataset and field nodes carry references, version pins and observed names only. Both ledger relations are review_state candidate and the registry is boundary-review-required, so they may not be published as settled boundary closure." }, { "concept": "Descriptive-only neighbours: logging and audit-record model, access-control and classification model, party or identity model", "disposition": "deferred — registry rebinding required", "rationale": "Three boundary notes and at least four inline_only rationales discharge duties to models that have no registry identifier. Until those identifiers exist the deferrals are prose, not enforceable relations, and cannot be counted as evidence that access evaluation, audit-trail semantics or party records sit outside this model." }, { "concept": "Frozen registry parent WM-XCT-012 is not acknowledged anywhere in the result", "disposition": "deferred — reconcile result against registry record", "rationale": "The frozen record declares parent_ids WM-XCT-012, yet no bundle, scope statement, boundary note or source reference mentions it. Either the parent linkage is stale or the scope statement is incomplete; a boundary claim cannot be published while the model and its own registry record disagree about the parent." }, { "concept": "SRC-013 Apache Iceberg cited from the moving main branch", "disposition": "rejected as a stable citation", "rationale": "A raw main-branch URL qualified only by an access date cannot be deterministically re-verified, and it contradicts this model's own rule that referenced states carry immutable version pins. Require a tag- or commit-pinned citation before the field-ID and snapshot-pinning grounding is relied on." }, { "concept": "SRC-015 retained-EU-law republication carrying the INSPIRE narrative lineage requirement", "disposition": "accepted provisionally, publication-held", "rationale": "It is the sole evidence for the narrative-statement obligation and for one of the six declared conflicts, and the result honestly marks it non-primary. A national republication still cannot carry an EU regulatory claim, so substitution with the authentic EUR-Lex text is a precondition of that citation being relied upon." }, { "concept": "Artifact serialisation flags on perimeter-and-granularity-declaration, alignment-crosswalk and column-mapping-specification (serial false)", "disposition": "reclassified — flag for owner adjudication", "rationale": "Serial coverage and reconciliation reports pin their denominator to a non-serial, revised-in-place perimeter declaration, so a coverage figure cannot be tied to the perimeter version it was measured against. Crosswalks are likewise bound to external standard versions. A standing revisable artifact contradicts the model's own time-and-version pinning doctrine." }, { "concept": "dse-q1 expresses disclosure scope in Vercy meta-vocabulary (bundle, layer, finding, artifact)", "disposition": "reclassified — reword to subject vocabulary", "rationale": "Disclosure scoping of a lineage record should be expressible as graph, subgraph, path, node, edge, field-level detail and transformation expression. Borrowing the authoring container names leaks the meta-model frame into the subject model and makes the only access-kind question unusable outside Vercy's own document structure." }, { "concept": "ceg-q1 presupposes a published confidence grading scale", "disposition": "reclassified — reword", "rationale": "Known omission seven states that no normative grading source was retrieved and that the scale's content is left to the adopting Dimension. Asking which published scale applies asserts an authority the evidence does not support; ask instead which declared scale is in force and which role issued it." }, { "concept": "epi-q3 and export-lineage-payload require integrity evidence binding payload to producer and schema version", "disposition": "accepted with a source-support flag", "rationale": "No source cited on the exchange-payload finding (SRC-004, 007, 008, 009) defines digest or signature binding. Either attach grounding already present in the source list (DCAT checksum, SLSA attestation as technique rather than merged node space) or mark the requirement explicitly as design-derived rather than standards-derived." }, { "concept": "Function name apply-lineage-disposition asserts execution the model explicitly does not perform", "disposition": "reclassified — rename to marking semantics", "rationale": "The description marks records for deletion, tombstoning or redaction and hands the downstream holder set to the notifying model, while physical removal is assigned to the storage owner. The imperative Apply invites a reader to attribute enforcement to this model, which is precisely the boundary it claims to hold." }, { "concept": "Five artifacts have no producing function: perimeter declaration, column mapping specification, obligation binding register, namespace registry entry, accuracy attestation", "disposition": "deferred to owner", "rationale": "Either declaration, registration and attestation functions are missing from the nine-function set, or those artifacts are authored outside the model and the findings should say so. Single-provider mode forbids adding functions, so the asymmetry is recorded for adjudication rather than silently closed." }, { "concept": "pcp-q3 resolved dependencies and engine identity sit adjacent to SLSA build provenance", "disposition": "accepted with an explicit carried-reference constraint", "rationale": "The SLSA boundary note forbids merging the build and data provenance node spaces, yet resolved dependencies is SLSA's own vocabulary. The question is legitimate only if the answer is a reference to a build attestation; the draft must state that constraint inline rather than relying on the boundary note alone." }, { "concept": "Row-level lineage left as a declared gap rather than modelled from Iceberg-specific evidence", "disposition": "accepted — rejection of an invented canon", "rationale": "The only retrieved grounding is table-format specific and does not generalise across systems. Declaring the gap in both the checklist and the omissions list is the correct outcome; asserting a cross-system row-level structure would manufacture canon the sources do not support." }, { "concept": "Orthographic and naming drift between identifiers and prose", "disposition": "accepted identifiers, normalise prose", "rationale": "Identifier transformation-characterization and question text characterisation diverge, as do specialization in gcc-q3 and specialisation in cnr-q2; identifier regulatory-obligation-binding diverges from its name External obligation binding. Identifiers are stability surfaces and should not churn, so normalise reader-facing prose to one variant and record the id-versus-name divergence in the registry." }, { "concept": "Whether the waived second-provider review constitutes a critical conflict", "disposition": "rejected as a conflict; recorded as a governance hold", "rationale": "The waiver is an owner-authorised process condition, not a contradiction in the evidence, and the six conflicts the provider declares are each resolved by an explicit design rule. Fabricating a blocking conflict from the waiver would misreport the state of the work, so critical_conflicts is returned empty and the waiver appears in the publication holds." } ], "publicationHolds": [ "Single-provider hold: independent second-provider review by Grok was waived by the repository owner at 2026-08-29T09:06:27Z after repeated structured-output failures. No corroborating provider result exists — comparison shows zero common sources, zero matching bundles, layers, findings or questions, and entry_kind_agreement status waived with grok null. Every published artifact must carry a visible single-provider notice, remain a reviewable draft, and must not be represented as cross-provider validated.", "Live source and version verification is outstanding for all 15 sources. This audit ran with no tools and could not confirm that any URL resolves or still carries the cited version. Priority checks: the OpenLineage 1.52.0 documentation set (SRC-004 through SRC-008) and the ColumnLineageDatasetFacet 1-2-0 schema; SRC-013 Iceberg, cited from a moving main branch and requiring a tag- or commit-pinned replacement; SRC-014 SLSA v1.1; SRC-009 DCAT-3 dated 22 August 2024.", "SRC-015 must be replaced by the authentic EUR-Lex text of Commission Regulation (EC) No 1205/2008 before the INSPIRE narrative lineage requirement is relied on in an EU regulatory context. It is currently the only evidence for the narrative-versus-graph conflict and is a non-primary national republication.", "Registry hold: the frozen record carries status candidate and review_state boundary-review-required, and both COMPOSE relations from WM-DAT-001 and WM-DAT-005 carry review_state candidate. Publish as a candidate boundary, never as settled. The unreconciled parent_ids value WM-XCT-012, which the result never references, must be resolved in the same pass.", "Unresolved composition targets: the logging and audit-record model and the access-control and classification model are named descriptively only, with no registry identifiers. The deferral of audit-trail semantics, access evaluation and enforcement is therefore asserted in prose and cannot be treated as an enforceable boundary until both entries exist and are rebound.", "The coverage checklist asserts normative content that is not present in the audited structure payload — the identity priority order said to be fixed in artifact_rules, the RFC 3339 offset mandate, the four disclosure scopes with deny-by-default, tombstoning as default disposition, longest-prevails retention conflict handling, and the unverified-artifact downgrade rule. Confirm these exist in the full record or downgrade the checklist notes before publication, because as supplied the checklist claims more than the evidence shows.", "The GDPR-versus-AI-Act retention resolution (tombstoning plus field-level redaction) and the confidence grading scale must be published as adopting-Dimension design decisions, explicitly not conformance claims. The result already says so for the retention conflict; the same label must be applied wherever the longest-prevails retention rule is stated, since it cannot serve as a lawful default against an erasure request.", "Independent second-provider review was explicitly waived by the repository owner; this Claude-only result remains a reviewable draft." ], "deferredResearch": [ "Retrieve ISO 19115-1 LI_Lineage, LI_ProcessStep and LI_Source through an accessible route after the ISO catalogue returned HTTP 403, so the structured process-step account is grounded directly rather than indirectly through a retained-EU-law republication.", "Consult ISO/IEC 11179 data element registration to give field-node identity a registry-theoretic basis instead of relying solely on table-format field IDs and column lineage facet naming.", "Consult MLCommons Croissant and the SPDX AI and dataset profiles to determine whether machine-learning feature, training-run and model-artifact lineage belongs in this model or in a sibling.", "Find cross-system normative grounding for row-level and record-level lineage; the only retrieved evidence is Iceberg-specific (_row_id, _last_updated_sequence_number) and does not generalise.", "Model streaming and continuous-processing lineage properly: windowing, watermark and reprocessing semantics currently reduce to an observation window plus a snapshot-versus-accumulative pattern code.", "Locate or commission a normative confidence grading scale for inferred lineage, which is currently declared as a required element with no source and left to the adopting Dimension.", "Determine whether a tombstoned lineage record can itself retain personal data through location-derived node names, file paths or field names; no question in retention-and-disposition covers residue in the tombstone, which weakens the erasure resolution.", "Enumerate non-EU obligation equivalents to populate the obligation binding register, since all legal anchors are currently EU instruments and BCBS 239 is used as a generalisation of supervisory expectation.", "Decide whether declaration, registration and attestation functions belong in the function set, given that five artifacts — perimeter declaration, column mapping specification, obligation binding register, namespace registry entry and accuracy attestation — have no producing function." ] }, "statistics": { "sources": 15, "bundles": 7, "layers": 14, "findings": 28, "questions": 109, "artifacts": 12, "functions": 9 } }