# Vercy AI instruction - YAML 1.2 (JSON-compatible) { "vercy": "1.0-draft", "publication": { "status": "published", "adjudicationStatus": "reviewable-draft", "publishableCanonical": false, "generatedAt": "2026-08-23T14:14:37Z", "synthesisSha256": "f244c50fcd7af13add229181b617533c85f710f8285cb75b496e8f3e9c646ed4", "providerMode": "dual-provider", "providers": [ "Claude", "Grok" ], "waivedProviders": [] }, "metaModel": { "id": "WM-XCT-026", "registryId": "vr.wm-xct-026", "name": "Quality / Confidence", "version": "0.3.0-research.1", "previousVersions": [], "entryKind": "mixin", "family": "World Models", "category": "Cross-cutting context", "industry": [ "Cross-industry" ], "domain": [ "XCT.QLT" ], "tags": [ "quality", "confidence", "xct.qlt" ], "status": "published" }, "canonicalUrl": "https://ver.cy/models/wm-xct-026-quality-confidence/", "sourceUrl": "https://github.com/ver-cy/world-models/tree/feat/mega-model-registry/research/runs/wm-xct-026", "model": { "registry_id": "vr.wm-xct-026", "model_id": "WM-XCT-026", "name": "Quality / Confidence", "entry_kind": "mixin", "purpose": "Provide a format-neutral, attachable structure for asserting, computing, evidencing and governing quality dimensions, measures, scores, assessment methods, uncertainty and confidence about any subject, so that an agent can decide whether the subject is fit for a stated purpose and can defend that decision with evidence.", "scope_statement": "WM-XCT-026 models the quality/confidence ASSERTION as a first-class, separately identified, attributable and time-stamped record about some other subject. It covers: what is assessed and at what granularity; which dimension and which registered measure are used; the method, sampling and reference basis; the result value, its scale and its uncertainty; a confidence statement about the assertion itself, its declared scale and its evidential basis; conformance against a stated requirement; declared fitness for purpose and limitations; validity in time; lifecycle and supersession; attribution and provenance; comparability, disagreement, access and retention. It does NOT model the subject's own semantics, nor units, provenance graphs, identifier minting, constraint languages, risk of harm or agent trust, all of which belong to composable sibling models. Storage and interface (JSON, YAML, Markdown, RDF, Git, MCP, MongoDB) are projections of this structure, never its semantics. W3C DQV is used as the primary structural anchor precisely because it deliberately refuses to prescribe a single definition of quality and instead standardises the shape of comparable assessments.", "in_scope": [ "Quality assertions attached to any host entity, event, dataset, model output, artifact or process step", "Dimension/category vocabularies and registered quality measures with parameters, value type and value structure", "Assessment method: procedure, evaluation environment, full inspection versus sampling, reference/ground-truth basis", "Result expression: value, scale type, value structure, unit reference, aggregation and composite scores", "Measurement uncertainty and error models, including coverage interval, coverage probability and coverage factor", "Confidence as an explicit statement about the assertion, with a declared scale, evidential basis and calibration claim", "Requirements, acceptance thresholds, decision rules, conformance verdicts and severity", "Declared intended use, fitness-for-purpose judgement and explicit limitations, caveats and known defects", "Temporal validity, currency, re-assessment cadence, lifecycle states, supersession and retraction", "Attribution, competence, independence of the assessor, and provenance/lineage of derived assertions", "Comparability rules, competing assessments, disagreement and rectification", "Access classification, disclosure, retention and deletion of quality statements and their evidence" ], "out_of_scope": [ "The subject's own domain semantics, payload or business rules", "Definition of units, quantity kinds and unit conversion (Measurement & Units sibling model)", "General provenance graph semantics and activity modelling (Provenance & Attribution sibling model)", "Identifier minting policy and identifier scheme registration (Identifier sibling model)", "Constraint/shape languages and the authoring of validation rules (Validation & Constraint sibling model)", "Risk, hazard and likelihood of harm; confidence in an assertion is not a risk score (Risk sibling model)", "Trust, reputation and accreditation of agents as standalone subjects (Agent/Party and Certification sibling models)", "Statistical and machine-learning method definitions themselves (Method/Analysis sibling model)", "Licensing, rights and terms of use of the subject", "Orchestration of monitoring pipelines and job scheduling (Process sibling model)", "Sensitive-data classification schemes as such; this model only carries the classification assigned to a quality statement" ], "boundary_notes": [ { "neighbor": "Validation & Conformance (SHACL-style constraint checking)", "distinction": "SHACL yields binary conformance: a focus node either conforms or produces results, and sh:resultSeverity (Violation/Warning/Info) is organisational rather than a graded score. WM-XCT-026 carries graded, scaled and uncertain assessments. A validation report is therefore an INPUT artifact to a quality assertion, not a substitute for it; conversely a quality score must never be presented as conformance without an explicit requirement and decision rule.", "source_refs": [ "SRC-007", "SRC-003" ] }, { "neighbor": "Provenance & Attribution", "distinction": "DQV itself reuses PROV for who produced a quality statement, when, and from what. This model references prov:wasAttributedTo / prov:wasGeneratedBy / prov:wasDerivedFrom rather than redefining provenance; it owns only the quality-specific slots (assessor competence, independence, method, evidence).", "source_refs": [ "SRC-001", "SRC-006" ] }, { "neighbor": "Measurement & Units", "distinction": "Uncertainty semantics (standard/expanded uncertainty, coverage interval, coverage probability, coverage factor, metrological traceability) are consumed from JCGM GUM/VIM; unit identifiers are referenced (DQV recommends sdmx-attribute:unitMeasure with dereferenceable unit definitions) and never minted here.", "source_refs": [ "SRC-004", "SRC-005", "SRC-001" ] }, { "neighbor": "Risk & Harm", "distinction": "Confidence is a statement about the strength of an assessment, not a probability of harm and not an expected loss. Frameworks that grade certainty of evidence (GRADE) keep the certainty rating strictly separate from the effect estimate; this model preserves that separation and forbids arithmetic between confidence grades and risk scores.", "source_refs": [ "SRC-015", "SRC-017" ] }, { "neighbor": "Evidence & Attestation / Verifiable Credentials", "distinction": "A signed attestation carries issuer, proof, status and an evidence property, but W3C VC Data Model 2.0 defines no confidence property; assurance is derived by the verifier from issuer, proof, evidence and status. This model supplies the confidence semantics and treats the credential as a transport and integrity wrapper.", "source_refs": [ "SRC-018" ] }, { "neighbor": "Sensing systems and system capability (SSN/SOSA)", "distinction": "ssn-system states device capabilities (Accuracy, Precision, Resolution, Drift, Latency) qualified by Condition, i.e. quality of an instrument under stated conditions. That is a capability declaration about a system, not an assessment of a produced result. WM-XCT-026 may cite such a capability as method context but keeps assertion-level results separate.", "source_refs": [ "SRC-008" ] }, { "neighbor": "Catalogue/Dataset description (DCAT-style records)", "distinction": "A catalogue record describes availability and access; quality metadata is attached to it via a quality-metadata container rather than embedded as descriptive fields, so that quality statements can carry their own provenance, validity and access rules.", "source_refs": [ "SRC-001", "SRC-019" ] } ] }, "sources": [ { "id": "SRC-001", "title": "Data on the Web Best Practices: Data Quality Vocabulary (DQV)", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/vocab-dqv/", "version_or_date": "W3C Working Group Note, 15 December 2016", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:05:00Z", "relevance": "Primary structural anchor: Category, Dimension, Metric, QualityMeasurement (subclass of qb:Observation), QualityAnnotation, QualityCertificate, UserQualityFeedback, QualityPolicy, QualityMeasurementDataset, QualityMetadata; properties isMeasurementOf, computedOn, value, inDimension, inCategory, expectedDataType, hasQualityMeasurement, hasQualityAnnotation, hasQualityMetadata; PROV reuse and sdmx-attribute:unitMeasure for units. Explicitly refuses to prescribe one definition of quality." }, { "id": "SRC-002", "title": "ISO/IEC 25012:2008 Software engineering — Software product Quality Requirements and Evaluation (SQuaRE) — Data quality model", "organization": "ISO/IEC", "url": "https://committee.iso.org/es/sites/isoorg/contents/data/standard/03/57/35736.html", "version_or_date": "Edition 1, published December 2008 (status: published/confirmed)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:12:00Z", "relevance": "Establishes a general data quality model for structured data with fifteen characteristics viewed from two points of view, inherent and system-dependent. Used here as an alignable dimension vocabulary, not as a hardcoded list: the catalogue record confirms the two-viewpoint structure but does not enumerate the characteristic names, so the enumeration is treated as unverified." }, { "id": "SRC-003", "title": "ISO 19157-1:2023 Geographic information — Data quality — Part 1: General requirements", "organization": "ISO/TC 211", "url": "https://committee.iso.org/es/sites/isoorg/contents/data/standard/07/89/78900.html", "version_or_date": "Edition 1, published 19 April 2023 (replaces ISO 19157:2013)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:14:00Z", "relevance": "Components and content structure of data quality measures, general procedures for evaluating quality (including scope and evaluation method), and principles for reporting. Explicitly does not define minimum acceptable quality levels — those live in product specifications — which grounds the separation of measure, result and requirement in this model." }, { "id": "SRC-004", "title": "JCGM publications: Guide to the expression of uncertainty in measurement (JCGM 100:2008 and GUM-series), Supplements 1–2, and JCGM 106:2012 The role of measurement uncertainty in conformity assessment", "organization": "Joint Committee for Guides in Metrology (JCGM) / BIPM", "url": "https://www.bipm.org/en/committees/jc/jcgm/publications", "version_or_date": "JCGM 100:2008 with Amendment 1:2026; JCGM 101:2008; JCGM 102:2011; JCGM 106:2012; JCGM GUM-1:2023, GUM-5:2026, GUM-6:2020", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:08:00Z", "relevance": "Normative basis for uncertainty evaluation and for the role of uncertainty in conformity assessment (decision rules, acceptance limits, guard bands). Establishes that a bare score without an uncertainty statement is not a defensible conformity decision." }, { "id": "SRC-005", "title": "International Vocabulary of Metrology — Basic and general concepts and associated terms (VIM3), annotated online edition (JCGM 200:2012)", "organization": "Joint Committee for Guides in Metrology (JCGM) / BIPM", "url": "https://jcgm.bipm.org/vim/en/index.html", "version_or_date": "3rd edition, JCGM 200:2012; annotated edition last updated 29 April 2017", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:20:00Z", "relevance": "Definitions used verbatim as alignment targets: measurement accuracy (2.13), trueness (2.14), precision (2.15), measurement uncertainty (2.26), standard (2.30) and combined standard (2.31) and expanded (2.35) uncertainty, coverage interval (2.36), coverage probability (2.37), coverage factor (2.38), calibration (2.39), metrological traceability (2.41), verification (2.44), validation (2.45)." }, { "id": "SRC-006", "title": "PROV-O: The PROV Ontology", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/prov-o/", "version_or_date": "W3C Recommendation, 30 April 2013", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:06:00Z", "relevance": "Entity/Activity/Agent plus wasGeneratedBy, wasAttributedTo, wasDerivedFrom, wasAssociatedWith, used, startedAtTime/endedAtTime/generatedAtTime and the qualification pattern (qualifiedAttribution, qualifiedDerivation, atTime). Supplies attribution and lineage for quality assertions without duplicating them." }, { "id": "SRC-007", "title": "Shapes Constraint Language (SHACL)", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/shacl/", "version_or_date": "W3C Recommendation, 20 July 2017", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:16:00Z", "relevance": "ValidationReport with sh:conforms plus ValidationResult (focusNode, resultPath, value, sourceShape, sourceConstraintComponent, resultSeverity, resultMessage). Confirms that constraint validation is binary conformance with organisational severities, establishing the boundary against graded quality." }, { "id": "SRC-008", "title": "Semantic Sensor Network Ontology (SSN/SOSA)", "organization": "W3C and Open Geospatial Consortium (Spatial Data on the Web Working Group)", "url": "https://www.w3.org/TR/vocab-ssn/", "version_or_date": "W3C Recommendation, 19 October 2017", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:18:00Z", "relevance": "ssn-system SystemCapability/SystemProperty (Accuracy, Precision, Resolution, Sensitivity, Selectivity, DetectionLimit, Drift, Latency, ResponseTime, MeasurementRange, OperatingRange, SurvivalRange) qualified by Condition; and the sosa distinction between phenomenonTime and resultTime, which grounds separate event and observation timestamps." }, { "id": "SRC-009", "title": "ISO 8000-8:2015 Data quality — Part 8: Information and data quality: Concepts and measuring", "organization": "ISO", "url": "https://committee.iso.org/es/sites/isoorg/contents/data/standard/06/08/60805.html", "version_or_date": "Edition 1, published 10 November 2015 (stage 90.93, confirmed 16 August 2022)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:22:00Z", "relevance": "Specifies prerequisites for measuring information and data quality within quality management processes — i.e. that a measurement is only meaningful when its prerequisites (specification, scope, method) are in place. Widely reported to distinguish syntactic, semantic and pragmatic quality with verification versus validation; that internal terminology was not verified from the normative text and is recorded as an evidence gap." }, { "id": "SRC-010", "title": "ISO/IEC 5259-2:2024 Artificial intelligence — Data quality for analytics and machine learning (ML) — Part 2: Data quality measures", "organization": "ISO/IEC JTC 1/SC 42", "url": "https://committee.iso.org/es/sites/isoorg/contents/data/standard/08/18/81860.html", "version_or_date": "Edition 1, published 5 November 2024", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:24:00Z", "relevance": "Data quality model and measurable characteristics for analytics and ML, covering structured and unstructured data and building on ISO/IEC 25012 and ISO 8000. Evidence that dimension vocabularies are plural and domain-scoped, so the mixin must bind a vocabulary rather than embed one." }, { "id": "SRC-011", "title": "ISO/IEC TS 4213:2022 Information technology — Artificial intelligence — Assessment of machine learning classification performance", "organization": "ISO/IEC JTC 1/SC 42", "url": "https://committee.iso.org/es/sites/isoorg/contents/data/standard/07/97/79799.html", "version_or_date": "Edition 1, published 13 October 2022 (under revision as ISO/IEC DIS 4213)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:26:00Z", "relevance": "Methodology for assessing classification performance: control criteria for evaluation setup, ground-truth definition, data splits and cross-validation, documented algorithm/hyperparameters, evaluation environment and baselines — the method slots that make a score interpretable and reproducible." }, { "id": "SRC-012", "title": "ISO/FDIS 19157-3 Geographic information — Data quality — Part 3: Data quality measures register", "organization": "ISO/TC 211", "url": "https://committee.iso.org/es/sites/isoorg/contents/data/standard/08/70/87032.html", "version_or_date": "Final Draft International Standard, stage 50.60 (voting closed 8 July 2026) — not yet published", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-23T10:34:00Z", "relevance": "Specifies establishing, maintaining and publishing a register of data quality measures in compliance with ISO 19135-1:2015, including register content structure, registration and maintenance procedures and a machine-readable implementation. Supports the measure-registration finding, but as a draft it cannot be treated as settled; the registration governance node is therefore marked as partially supported." }, { "id": "SRC-013", "title": "Regulation (EU) 2024/1689 laying down harmonised rules on artificial intelligence (Artificial Intelligence Act)", "organization": "European Parliament and Council of the European Union", "url": "https://eur-lex.europa.eu/legal-content/EN/TXT/HTML/?uri=OJ:L_202401689", "version_or_date": "Published in the Official Journal, 12 July 2024", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:28:00Z", "relevance": "Article 10 requires training/validation/testing data sets to be relevant, sufficiently representative, to the best extent possible free of errors and complete, with appropriate statistical properties and bias examination; Article 15 requires declared accuracy levels and accuracy metrics in the instructions for use; Article 12 logging and Article 11/Annex IV technical documentation require retained performance and validation records. Legal grounding for declared measures, declared thresholds and retained evidence." }, { "id": "SRC-014", "title": "Regulation (EU) 2016/679 (General Data Protection Regulation)", "organization": "European Parliament and Council of the European Union", "url": "https://eur-lex.europa.eu/legal-content/EN/TXT/HTML/?uri=CELEX:32016R0679", "version_or_date": "27 April 2016", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:30:00Z", "relevance": "Article 5(1)(d) makes accuracy a legal obligation with a duty to erase or rectify; Article 16 grants rectification; Article 17 grants erasure; Article 5(1)(e) imposes storage limitation; Article 5(2) imposes accountability. Grounds the rectification/dispute path and the retention and deletion rules for quality statements about personal data." }, { "id": "SRC-015", "title": "GRADE Working Group — the GRADE approach to rating certainty of evidence", "organization": "GRADE Working Group", "url": "https://www.gradeworkinggroup.org/", "version_or_date": "Site content updated 2023-05", "source_type": "scientific", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-23T10:17:00Z", "relevance": "An operational, widely adopted example of certainty/confidence as an explicit ordinal grade (high, moderate, low, very low) derived from named domains that lower certainty (risk of bias, imprecision, inconsistency, indirectness, publication bias) and raise it (large effect, dose-response, plausible confounding). Demonstrates that confidence must be reason-decomposed, not asserted as a bare number." }, { "id": "SRC-016", "title": "European Statistics Code of Practice", "organization": "Eurostat / European Statistical System Committee", "url": "https://ec.europa.eu/eurostat/web/quality/european-quality-standards/european-statistics-code-of-practice", "version_or_date": "Revised edition adopted November 2017 (16 principles, 84 indicators)", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:32:00Z", "relevance": "Output quality principles — relevance, accuracy and reliability, timeliness and punctuality, coherence and comparability, accessibility and clarity — showing that quality is multi-dimensional, that relevance is defined by user need, and that comparability is itself a governed dimension rather than an assumption." }, { "id": "SRC-017", "title": "AI Risk Management Framework (AI RMF 1.0)", "organization": "National Institute of Standards and Technology (NIST)", "url": "https://www.nist.gov/itl/ai-risk-management-framework", "version_or_date": "AI RMF 1.0, released 26 January 2023", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:11:00Z", "relevance": "Core functions GOVERN, MAP, MEASURE, MANAGE, with MEASURE as an explicit, separately governed activity for trustworthiness characteristics. Supports treating assessment as a governed function with named roles rather than an incidental by-product of processing." }, { "id": "SRC-018", "title": "Verifiable Credentials Data Model v2.0", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/vc-data-model-2.0/", "version_or_date": "W3C Recommendation, 15 May 2025", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:10:00Z", "relevance": "evidence, issuer, proof/securing mechanism, credentialStatus, refreshService, validFrom/validUntil. Confirms that the core data model defines no confidence property, so a transport credential cannot by itself carry a confidence claim — the confidence semantics must come from this model." }, { "id": "SRC-019", "title": "FAIR Principles", "organization": "GO FAIR International Support and Coordination Office", "url": "https://www.go-fair.org/fair-principles/", "version_or_date": "FAIR Guiding Principles as published in Scientific Data (2016); page current at access date", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-23T10:29:00Z", "relevance": "R1 (rich description with a plurality of accurate and relevant attributes), R1.2 (detailed provenance), R1.3 (domain-relevant community standards) and I3 (qualified references) motivate machine-actionable, provenance-bearing, vocabulary-bound quality metadata rather than free-text quality notes." }, { "id": "SRC-020", "title": "ISO 19157 Data Quality Measures (DQM) XML schema, version 1.2.0", "organization": "ISO/TC 211 XML schema repository", "url": "https://schemas.isotc211.org/19157/-/dqm/1.2.0/", "version_or_date": "dqm 1.2.0, status: current (normative schema dqm.xsd)", "source_type": "schema", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T10:36:00Z", "relevance": "Evidence that quality measure descriptions are published as normative machine-readable schemas independent of any single application, supporting the requirement that a measure be dereferenceable and versioned rather than named inline." }, { "id": "SRC-021", "title": "ISO/IEC 25012:2008 Software engineering — Software product Quality Requirements and Evaluation (SQuaRE) — Data quality model", "organization": "International Organization for Standardization / International Electrotechnical Commission", "url": "https://www.iso.org/standard/35736.html", "version_or_date": "2008 (confirmed current catalogue entry)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T16:12:00Z", "relevance": "Defines fifteen data-quality characteristics from inherent and system-dependent viewpoints; DQV and ISO/IEC 5259-2 treat this as the general data-quality characteristic catalogue." }, { "id": "SRC-022", "title": "ISO 19157-1:2023 Geographic information — Data quality — Part 1: General requirements", "organization": "International Organization for Standardization", "url": "https://www.iso.org/standard/78900.html", "version_or_date": "First edition 2023-04", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T16:15:00Z", "relevance": "Authoritative structure for data-quality units, elements (completeness, logical consistency, positional accuracy, temporal quality, thematic quality), measures, evaluation procedures, reporting and metaquality; does not set minimum acceptable quality." }, { "id": "SRC-023", "title": "ISO 8000-1:2022 Data quality — Part 1: Overview", "organization": "International Organization for Standardization", "url": "https://www.iso.org/standard/81745.html", "version_or_date": "2022-04", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T16:18:00Z", "relevance": "Establishes principles of information and data quality, the path to data quality, and the series architecture; treats relevant characteristics as purpose-dependent rather than absolute." }, { "id": "SRC-024", "title": "JCGM 100:2008 Evaluation of measurement data — Guide to the expression of uncertainty in measurement (GUM 1995 with minor corrections)", "organization": "Joint Committee for Guides in Metrology (BIPM, IEC, ILAC, ISO, IUPAC, IUPAP, IFCC, OIML)", "url": "https://www.bipm.org/documents/20126/2071204/JCGM_100_2008_E.pdf", "version_or_date": "First edition 2008, corrected version 2010", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T16:40:00Z", "relevance": "Normative rules for standard uncertainty, Type A and Type B evaluation, combined and expanded uncertainty, coverage factor and coverage probability or level of confidence attached to a measured result." }, { "id": "SRC-025", "title": "NIST AI Risk Management Framework 1.0 — AI Risks and Trustworthiness and AI RMF Core (Measure)", "organization": "National Institute of Standards and Technology (United States)", "url": "https://airc.nist.gov/airmf-resources/airmf/3-sec-characteristics/", "version_or_date": "AI RMF 1.0, 2023 (AIRC excerpts accessed 2026-08-23; revision in progress)", "source_type": "public-authority", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-23T16:25:00Z", "relevance": "Requires accuracy measurements with documented test sets and methods; Measure function demands associated measures of uncertainty, independent review, recalibration and documented TEVV; distinguishes validity, reliability, robustness and contextual trade-offs." }, { "id": "SRC-026", "title": "ISO/IEC 5259-2:2024 Artificial intelligence — Data quality for analytics and machine learning (ML) — Part 2: Data quality measures", "organization": "ISO/IEC JTC 1/SC 42", "url": "https://www.iso.org/standard/81860.html", "version_or_date": "First edition 2024-11", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-23T16:28:00Z", "relevance": "Specifies a data-quality model, measures and reporting guidance for analytics and ML, building on ISO 8000, ISO/IEC 25012 and ISO/IEC 25024, and adding ML-oriented characteristics and measure identifiers such as Acc-ML-1/2." } ], "structure": { "bundles": [ { "id": "assertion-foundations", "name": "Quality assertion foundations", "description": "What a quality/confidence record IS: a separately identified assertion about some other subject, bound to a declared dimension and a registered, dereferenceable measure.", "rationale": "DQV models quality as its own resource (QualityMeasurement/QualityAnnotation) linked to the assessed resource by computedOn and to a Metric by isMeasurementOf, and organises metrics into Dimensions and Categories. ISO 19157-1 likewise separates the measure, the scope and the evaluation from the reported result. Without this separation a score cannot carry its own provenance, validity or access rules.", "source_refs": [ "SRC-001", "SRC-003", "SRC-002" ], "layers": [ { "id": "assertion-identity-and-attachment", "name": "Assertion identity and attachment", "description": "Identity of the assertion itself and the precise unit of the subject it covers.", "source_refs": [ "SRC-001", "SRC-003" ], "findings": [ { "id": "quality-assertion-identity", "name": "Identity of the quality assertion as a distinct resource", "description": "The quality/confidence record is a resource in its own right, distinct from the subject, from the measure and from the assessing activity. It must be independently identifiable so that it can be cited, superseded, disputed, embargoed and retained on its own schedule. DQV expresses this as a typed resource (QualityMeasurement, QualityAnnotation, QualityCertificate, UserQualityFeedback, QualityPolicy) that is linked to, rather than embedded in, the assessed resource; PROV then attaches generation and attribution to that resource.", "source_refs": [ "SRC-001", "SRC-006", "SRC-019" ], "questions": [ { "id": "q-assertion-identifier", "text": "Which identifier identifies this quality assertion itself, as distinct from the subject it describes and the measure it applies?", "kind": "identity", "answer_data": [ "assertion identifier value", "identifier kind: authoritative master-system id, governed IRI, or UUID/ULID minted by the adopting Dimension", "issuing authority or namespace" ] }, { "id": "q-assertion-class", "text": "Is this record a computed measurement, a human quality annotation, a certificate, a policy statement, or user feedback?", "kind": "classification", "answer_data": [ "assertion class code", "class vocabulary reference (e.g. DQV class IRI)", "whether the class implies an automated or human origin" ] }, { "id": "q-assertion-sameness", "text": "What natural key determines whether two records are the same assertion rather than two independent assessments of the same subject?", "kind": "identity", "answer_data": [ "natural-key components (subject reference, measure, scope, method version, assessment time)", "deduplication rule", "policy on re-running the same measure at the same time" ] }, { "id": "q-mint-authority", "text": "Who is authorised to mint and retire identifiers for quality assertions within the adopting Dimension?", "kind": "authority", "answer_data": [ "identifier-minting role", "delegation record", "retirement/tombstone policy" ] } ], "data_elements": [ { "id": "assertion-id", "name": "Assertion identifier", "description": "Stable identifier of the quality assertion, resolved by the identity priority rule (master-system id, then governed IRI, then UUID/ULID).", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-019" ] }, { "id": "assertion-class", "name": "Assertion class", "description": "Typed kind of quality statement (measurement, annotation, certificate, policy, user feedback), bound to an external class vocabulary.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "assertion-natural-key", "name": "Assertion natural key basis", "description": "Declared component list used to decide assertion sameness and to deduplicate repeated runs.", "value_kind": "collection", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001", "SRC-003" ] } ], "artifacts": [ { "id": "quality-assertion-record", "name": "Quality assertion record", "description": "The serialised assertion carrying identity, subject reference, measure reference, result, confidence, validity and attribution. Format-neutral: an RDF resource, a document, a row or a message are equally valid projections.", "media_or_form": [ "structured record in any serialisation", "RDF resource/named graph", "database row or document", "message payload" ], "serial": true, "identity_strategy": "Identified by the assertion identifier; storage location and file naming are projections and never the identity.", "source_refs": [ "SRC-001", "SRC-006" ] } ], "inline_only_rationale": null }, { "id": "subject-attachment-and-assessment-scope", "name": "Subject attachment and assessment scope", "description": "An assertion is meaningless unless the assessed unit is explicit: the whole resource, one distribution, a subset, a feature type, a single instance, one attribute, or a population represented by a sample. ISO 19157-1 makes scope an explicit component of a data quality evaluation, and DQV's computedOn names the exact resource assessed. Ambiguous scope is the most common way an honest score becomes a misleading one.", "source_refs": [ "SRC-003", "SRC-001", "SRC-011" ], "questions": [ { "id": "q-assessed-unit", "text": "What exactly was assessed: the whole resource, a distribution, a subset, a field, or a single instance?", "kind": "composition", "answer_data": [ "subject reference", "scope level code", "selector expression identifying the assessed part" ] }, { "id": "q-scope-restriction", "text": "Does the assertion apply only within a restricted extent, population segment, or operating condition, and how is that restriction expressed?", "kind": "constraint", "answer_data": [ "extent restriction (temporal, spatial, categorical)", "condition statement", "statement of what is explicitly not covered" ] }, { "id": "q-sample-vs-census", "text": "Was the whole scope inspected or was a sample used to infer a property of a population?", "kind": "measurement", "answer_data": [ "inspection mode (full inspection or sampling)", "sampling frame reference", "sample size and selection method" ] }, { "id": "q-multi-subject", "text": "May one assertion cover several subjects at once, and if so how is per-subject traceability preserved?", "kind": "relationship", "answer_data": [ "cardinality policy for subject references", "aggregation flag", "link to per-subject sub-assertions" ] } ], "data_elements": [ { "id": "subject-reference", "name": "Assessed subject reference", "description": "Reference to the resource assessed, at the granularity actually evaluated.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-001", "SRC-003" ] }, { "id": "scope-selector", "name": "Scope selector", "description": "Expression identifying the assessed part of the subject (path, feature type, attribute, query, or partition key).", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003" ] }, { "id": "scope-conditions", "name": "Applicability conditions", "description": "Conditions under which the assertion holds, following the pattern of stating capability relative to declared conditions.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008", "SRC-003" ] } ], "artifacts": [], "inline_only_rationale": "Scope is a structural qualifier of the assertion record itself and yields no separately retrievable object; where sampling is used, the retrievable object is the sample manifest recorded under the evaluation-procedure finding, and where a partition is used, it is the subject's own partition definition owned by a sibling model." } ] }, { "id": "dimension-and-measure-vocabulary", "name": "Dimension and measure vocabulary", "description": "The graded property being assessed (dimension/category) and the concrete, registered, dereferenceable procedure that produces a value (measure/metric).", "source_refs": [ "SRC-001", "SRC-002", "SRC-020" ], "findings": [ { "id": "quality-dimension-and-category-vocabulary", "name": "Dimension and category vocabulary binding", "description": "DQV separates Dimension (a quality-related characteristic that matters to consumers) from Metric (an abstract procedure for assessing it) and groups dimensions into Categories, but deliberately declines to prescribe one authoritative list. Multiple valid vocabularies exist in parallel — ISO/IEC 25012's fifteen characteristics under inherent and system-dependent viewpoints, ISO/IEC 5259-2 for analytics and ML data, ISO 19157-1's geographic quality components, and the European Statistics Code of Practice output principles. The mixin therefore binds a vocabulary per subject class rather than embedding one, and records that binding as a governed profile.", "source_refs": [ "SRC-001", "SRC-002", "SRC-010", "SRC-003", "SRC-016" ], "questions": [ { "id": "q-dimension-vocab", "text": "Which dimension vocabulary is bound for this subject class, at which version, and by whose authority?", "kind": "interoperability", "answer_data": [ "vocabulary identifier and version", "dimension term identifier", "binding authority and effective date" ] }, { "id": "q-dimension-viewpoint", "text": "Is the dimension inherent to the data itself or dependent on the surrounding system and context of use?", "kind": "classification", "answer_data": [ "viewpoint code (inherent / system-dependent / both)", "justification", "dependency on the evaluating system" ] }, { "id": "q-required-profile", "text": "Which dimensions are mandatory for this subject class before it may be published or relied upon, and which are optional?", "kind": "requirement", "answer_data": [ "quality profile identifier", "mandatory dimension list", "optional dimension list", "consequence of an unassessed mandatory dimension" ] }, { "id": "q-dimension-overlap", "text": "Where two bound vocabularies define overlapping or contradictory dimensions, which one governs?", "kind": "exception", "answer_data": [ "precedence rule", "overlap mapping", "recorded conflict note" ] } ], "data_elements": [ { "id": "dimension-ref", "name": "Quality dimension reference", "description": "Reference to the dimension term in a governed external vocabulary, never a free-text label.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002" ] }, { "id": "category-ref", "name": "Quality category reference", "description": "Optional grouping of dimensions sharing a common type of information.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "quality-profile-ref", "name": "Quality profile reference", "description": "Identifier of the profile that declares which dimensions and measures are required for a given subject class in the adopting Dimension.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013", "SRC-016" ] } ], "artifacts": [ { "id": "quality-profile-definition", "name": "Quality profile definition", "description": "Governed declaration binding a subject class to required dimensions, measures, thresholds and re-assessment cadence. This is what makes the mixin attachable in a specific Dimension.", "media_or_form": [ "profile document", "machine-readable profile record", "vocabulary binding table" ], "serial": false, "identity_strategy": "Profile identifier plus version; superseded profiles retained and referenced by effective date range.", "source_refs": [ "SRC-001", "SRC-013", "SRC-016" ] } ], "inline_only_rationale": null }, { "id": "quality-measure-definition-and-registration", "name": "Measure definition and registration", "description": "A score is uninterpretable unless the measure that produced it is itself a registered, versioned, dereferenceable object with a definition, parameters, a value type and a value structure. ISO 19157-1 specifies the components and content structure of data quality measures; ISO/TC 211 publishes these as normative machine-readable schemas; ISO/FDIS 19157-3 specifies a register of measures maintained under ISO 19135-1 registration procedures. That register standard is still at FDIS stage, so registration governance is adopted here as a defensible pattern rather than a settled normative requirement.", "source_refs": [ "SRC-003", "SRC-020", "SRC-012", "SRC-001" ], "questions": [ { "id": "q-measure-identity", "text": "Which registered measure produced this value, at which version, and where is its definition dereferenceable?", "kind": "identity", "answer_data": [ "measure identifier", "measure version", "definition location", "register or vocabulary of origin" ] }, { "id": "q-measure-parameters", "text": "What parameters, tolerances or thresholds were bound when the measure was applied, and what are their values?", "kind": "constraint", "answer_data": [ "parameter name/value pairs", "parameter units or code lists", "defaults applied when unspecified" ] }, { "id": "q-measure-value-type", "text": "What value type and value structure does the measure emit — a single value, a ratio, a count, a matrix, or a coverage?", "kind": "measurement", "answer_data": [ "expected data type", "value structure code", "permitted value domain" ] }, { "id": "q-measure-lifecycle", "text": "How is a measure superseded or retired, and what happens to results already produced by the retired version?", "kind": "lifecycle", "answer_data": [ "register item status (valid, superseded, retired)", "supersession reference", "restatement policy for historical results" ] } ], "data_elements": [ { "id": "measure-ref", "name": "Measure reference", "description": "Dereferenceable identifier of the applied quality measure, including version.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-003", "SRC-020" ] }, { "id": "measure-parameter-binding", "name": "Measure parameter binding", "description": "Parameters actually bound for this application of the measure.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-003" ] }, { "id": "measure-expected-datatype", "name": "Expected data type of the measure", "description": "Declared type of the value the measure emits, so consumers can validate and compare results.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-003" ] }, { "id": "measure-register-status", "name": "Measure register status", "description": "Registration status of the measure item (valid, superseded, retired) with supersession pointer.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-012" ] } ], "artifacts": [ { "id": "measure-register-entry", "name": "Quality measure register entry", "description": "Registered, versioned description of a measure: identifier, name and alias, associated quality element, basic measure, definition, description, parameters, value type, value structure, source reference and example.", "media_or_form": [ "register entry", "machine-readable measure schema", "specification section" ], "serial": true, "identity_strategy": "Register item identifier assigned by the register manager; version and status carried on the item, never inferred from a filename.", "source_refs": [ "SRC-012", "SRC-020", "SRC-003" ] } ], "inline_only_rationale": null }, { "id": "domain-specific-dimension-extension", "name": "Domain-specific dimension extension", "description": "Additional dimensions (for example ML-specific characteristics in ISO/IEC 5259-2, linked-data interlinking, or community uniqueness) are first-class local concepts with provenance of who added them. The full list of extra ISO/IEC 5259-2 characteristics is not fully extractable from freely published text and remains a documented gap.", "source_refs": [ "SRC-026", "SRC-022", "SRC-001" ], "questions": [ { "id": "domain-specific-dimension-extension-q01", "text": "What identifier and definition distinguish this extension dimension from standard catalogue codes?", "kind": "identity", "answer_data": [ "extension_dimension_id", "definition_text", "not_a_core_iso_code_flag" ] }, { "id": "domain-specific-dimension-extension-q02", "text": "Who authorised the extension and against which standard extension clause?", "kind": "authority", "answer_data": [ "authorising_agent_id", "extension_clause_ref", "approval_time" ] }, { "id": "domain-specific-dimension-extension-q03", "text": "If an ISO/IEC 5259-2 additional ML characteristic is claimed, is the official characteristic name cited from the standard text rather than inferred?", "kind": "evidence", "answer_data": [ "official_name_citation", "gap_flag", "measure_table_id" ] } ], "data_elements": [ { "id": "domain-specific-dimension-extension-data01", "name": "Extension dimension identifier", "description": "Local IRI for a non-core dimension.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-022" ] }, { "id": "domain-specific-dimension-extension-data02", "name": "Evidence gap flag", "description": "True if the official characteristic list was not available from a primary extract.", "value_kind": "boolean", "cardinality": "1", "required": true, "source_refs": [ "SRC-026" ] } ], "artifacts": [], "inline_only_rationale": "Extension dimensions are concept records stored in the dimension registry; no separate binary artefact is required beyond the registry entry already defined for dimensions." } ] } ] }, { "id": "assessment-method-and-result", "name": "Assessment method and result", "description": "How the assessment was carried out and how its numeric or categorical outcome is expressed, including uncertainty.", "rationale": "ISO 19157-1 defines general evaluation procedures; ISO/IEC TS 4213 fixes the control criteria that make a performance score reproducible; JCGM GUM and VIM define how a value and its uncertainty must be reported. A score reported without method, scale and uncertainty is not falsifiable.", "source_refs": [ "SRC-003", "SRC-011", "SRC-004", "SRC-005" ], "layers": [ { "id": "assessment-method", "name": "Assessment method", "description": "Procedure, environment, sampling design and the reference against which the subject was compared.", "source_refs": [ "SRC-003", "SRC-011", "SRC-005" ], "findings": [ { "id": "evaluation-procedure-sampling-and-environment", "name": "Evaluation procedure, sampling and environment", "description": "The executed procedure must be recorded with enough fidelity to be repeated: full inspection versus sampling, the sampling design, the data splits used, the executing agent (human, automated or hybrid), the tool and its version, and the evaluation environment. ISO/IEC TS 4213 makes evaluation setup, preprocessing, training/test/validation splits, cross-validation, documented algorithm and hyperparameters, environment and baselines explicit control criteria; ISO 19157-1 makes the evaluation method a reported component.", "source_refs": [ "SRC-011", "SRC-003", "SRC-017" ], "questions": [ { "id": "q-procedure-identity", "text": "Which documented procedure and which tool version executed this assessment?", "kind": "process", "answer_data": [ "procedure identifier and version", "tool/software identifier and version", "execution environment description" ] }, { "id": "q-sampling-design", "text": "If sampling was used, what was the sampling design, the achieved sample size, and the resulting representativeness claim?", "kind": "measurement", "answer_data": [ "sampling method", "frame definition", "achieved sample size and response/coverage rate", "representativeness statement" ] }, { "id": "q-executor-mode", "text": "Was the assessment automated, human-judged, or hybrid, and where did human judgement enter?", "kind": "provenance", "answer_data": [ "execution mode code", "points of human judgement", "inter-rater agreement where multiple judges were used" ] }, { "id": "q-preprocessing", "text": "What preprocessing, exclusion or filtering was applied before measurement, and could it bias the result?", "kind": "quality", "answer_data": [ "preprocessing steps", "exclusion criteria and counts excluded", "known or suspected bias direction" ] } ], "data_elements": [ { "id": "method-ref", "name": "Assessment method reference", "description": "Identifier and version of the documented evaluation procedure.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-003", "SRC-011" ] }, { "id": "inspection-mode", "name": "Inspection mode", "description": "Whether the scope was fully inspected or sampled.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003" ] }, { "id": "execution-environment", "name": "Execution environment", "description": "Environment record: tooling, versions, hardware/runtime where material to the result.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-011" ] } ], "artifacts": [ { "id": "sample-manifest", "name": "Sample and split manifest", "description": "Enumeration or reproducible specification of the units assessed, including data splits, seeds and exclusions.", "media_or_form": [ "tabular manifest", "reproducible query or seed specification", "split definition file" ], "serial": true, "identity_strategy": "Manifest identifier referenced by the assertion; content hash recorded for integrity.", "source_refs": [ "SRC-011", "SRC-003" ] } ], "inline_only_rationale": null }, { "id": "reference-basis-and-metrological-traceability", "name": "Reference basis, ground truth and traceability", "description": "Every quality claim is a comparison against something. That reference must be named: a specification, a reference dataset, an adjudicated ground truth, a calibrated standard, or an expert panel. VIM defines accuracy, trueness and precision strictly relative to a true or reference value, and defines metrological traceability as the property that relates a result to a reference through a documented unbroken chain of calibrations. Where no defensible reference exists, the assertion must say so rather than imply one.", "source_refs": [ "SRC-005", "SRC-011", "SRC-003" ], "questions": [ { "id": "q-reference-basis", "text": "Against which reference, specification or ground truth was the subject compared, and who adjudicated it?", "kind": "evidence", "answer_data": [ "reference identifier and version", "reference kind (specification, reference dataset, adjudicated labels, calibrated standard, expert panel)", "adjudicating agent" ] }, { "id": "q-traceability-chain", "text": "Is there an unbroken documented chain relating this result to a stated reference, and where is it recorded?", "kind": "provenance", "answer_data": [ "traceability chain reference", "calibration records", "date of last calibration" ] }, { "id": "q-reference-quality", "text": "What is known about the quality and error rate of the reference itself?", "kind": "quality", "answer_data": [ "reference error rate or known defects", "reference provenance", "recursive quality assertion about the reference" ] }, { "id": "q-no-reference-case", "text": "If no independent reference exists, how is that stated so the result is not read as accuracy?", "kind": "exception", "answer_data": [ "no-reference declaration", "substitute interpretation (e.g. internal consistency only)", "prohibited interpretations" ] } ], "data_elements": [ { "id": "reference-basis-ref", "name": "Reference basis reference", "description": "Identifier of the specification, reference dataset or standard used as the comparison basis.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005", "SRC-011" ] }, { "id": "reference-kind", "name": "Reference kind", "description": "Classification of the comparison basis, distinguishing calibrated standards from adjudicated labels from mere convention.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "traceability-statement", "name": "Traceability statement", "description": "Statement of the documented chain relating the result to the stated reference, or an explicit declaration that no such chain exists.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005" ] } ], "artifacts": [ { "id": "reference-dataset-or-calibration-record", "name": "Reference dataset or calibration record", "description": "The retained reference material or the calibration/adjudication record that establishes the comparison basis.", "media_or_form": [ "reference dataset snapshot", "calibration certificate", "adjudication protocol and label set" ], "serial": true, "identity_strategy": "Identifier plus version and content hash; snapshots retained for the life of assertions that depend on them.", "source_refs": [ "SRC-005", "SRC-011" ] } ], "inline_only_rationale": null } ] }, { "id": "result-expression", "name": "Result expression", "description": "The value produced, its scale and structure, and the uncertainty attached to it.", "source_refs": [ "SRC-001", "SRC-004", "SRC-005" ], "findings": [ { "id": "result-value-scale-and-structure", "name": "Result value, scale and value structure", "description": "The result carries a value plus everything needed to interpret it: expected data type, scale type (nominal, ordinal, interval, ratio), value structure (single value, ratio, count, matrix, coverage) and, where the value is a quantity, a dereferenceable unit reference. DQV recommends dereferenceable unit identifiers rather than unit strings. Ordinal grades must be marked as ordinal so that consumers do not average them.", "source_refs": [ "SRC-001", "SRC-003", "SRC-005" ], "questions": [ { "id": "q-result-value", "text": "What is the result value and what data type and value structure does it use?", "kind": "measurement", "answer_data": [ "value", "data type", "value structure code" ] }, { "id": "q-result-scale", "text": "What scale type applies, and which arithmetic operations are therefore invalid on this value?", "kind": "constraint", "answer_data": [ "scale type (nominal/ordinal/interval/ratio)", "permitted operations", "explicitly prohibited operations such as averaging ordinal grades" ] }, { "id": "q-result-unit", "text": "If the value is a quantity, which dereferenceable unit identifier applies?", "kind": "interoperability", "answer_data": [ "unit identifier IRI", "unit vocabulary reference", "conversion policy" ] }, { "id": "q-value-domain", "text": "What is the permitted value domain, and how are null, not-applicable and not-assessed distinguished from a poor score?", "kind": "validation", "answer_data": [ "value domain definition", "null/not-applicable/not-assessed codes", "rule forbidding substitution of nulls by zero" ] } ], "data_elements": [ { "id": "result-value", "name": "Result value", "description": "The measured or judged value emitted by the measure.", "value_kind": "other", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "result-scale-type", "name": "Scale type", "description": "Scale of measurement governing which operations and comparisons are valid.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-005", "SRC-003" ] }, { "id": "result-unit-ref", "name": "Unit reference", "description": "Dereferenceable identifier of the unit of measure, referenced from a sibling units model.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "result-status-code", "name": "Result status code", "description": "Distinguishes assessed values from not-assessed, not-applicable and failed-to-evaluate states.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003", "SRC-001" ] } ], "artifacts": [], "inline_only_rationale": "The value and its scale qualifiers are inline attributes of the assertion record already carried by the quality-assertion-record artifact; introducing a separate artifact would fragment an atomic result and invite value/unit drift between copies." }, { "id": "uncertainty-and-error-model", "name": "Uncertainty and error model", "description": "A value without an uncertainty statement cannot support a defensible conformity decision. JCGM GUM defines the evaluation of measurement uncertainty and its supplements cover distribution propagation and multiple output quantities; VIM defines standard, combined and expanded uncertainty, coverage interval, coverage probability and coverage factor. This model requires that any quantitative result declare either an uncertainty statement or an explicit, reasoned statement that none was evaluated. Uncertainty about the value is distinct from confidence in the assertion, and the two must not be merged.", "source_refs": [ "SRC-004", "SRC-005", "SRC-003" ], "questions": [ { "id": "q-uncertainty-value", "text": "What uncertainty is attached to the result, expressed as a standard uncertainty, an expanded uncertainty, or an interval?", "kind": "measurement", "answer_data": [ "uncertainty value and form", "coverage factor", "coverage interval bounds", "coverage probability" ] }, { "id": "q-uncertainty-method", "text": "How was the uncertainty evaluated, and which components dominate the budget?", "kind": "process", "answer_data": [ "evaluation approach (statistical from repeated observations, or from other information)", "named uncertainty components", "dominant contributor" ] }, { "id": "q-no-uncertainty", "text": "If no uncertainty was evaluated, why not, and what interpretation is thereby forbidden?", "kind": "exception", "answer_data": [ "reason for omission", "forbidden interpretations", "required caveat text" ] }, { "id": "q-uncertainty-vs-confidence", "text": "How is the uncertainty of the value kept distinct from the confidence in the assertion as a whole?", "kind": "definition", "answer_data": [ "separate fields for uncertainty and confidence", "rule prohibiting derivation of one from the other", "documented relationship where one exists" ] } ], "data_elements": [ { "id": "uncertainty-statement", "name": "Uncertainty statement", "description": "Structured expression of measurement uncertainty: form, value, coverage factor and coverage probability.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004", "SRC-005" ] }, { "id": "coverage-interval", "name": "Coverage interval", "description": "Interval associated with a stated coverage probability for the measurand.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "uncertainty-omission-reason", "name": "Uncertainty omission reason", "description": "Explicit reason when no uncertainty was evaluated, required so silence is not read as precision.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] } ], "artifacts": [ { "id": "uncertainty-budget", "name": "Uncertainty budget", "description": "Itemised list of uncertainty components, their evaluation type, magnitudes, sensitivity coefficients and combination, retained where a quantitative conformity decision depends on it.", "media_or_form": [ "tabular budget", "calculation record or notebook", "report annex" ], "serial": true, "identity_strategy": "Budget identifier referenced by the assertion; content hash recorded; superseded budgets retained.", "source_refs": [ "SRC-004", "SRC-005" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "confidence-and-evidence", "name": "Confidence and evidence", "description": "The explicit, scale-declared statement of how much weight the assertion itself can bear, and the evidence that backs it.", "rationale": "Confidence is the part of this model with the weakest cross-domain standardisation and therefore the highest risk of unsupported invention. GRADE demonstrates operational, reason-decomposed certainty grading; JCGM defines coverage probability as a strictly different construct; and W3C VC Data Model 2.0 confirms that credential formats carry evidence but define no confidence property. The structure here is deliberately minimal: declare the scale, decompose the basis, forbid cross-scale arithmetic.", "source_refs": [ "SRC-015", "SRC-004", "SRC-018", "SRC-001" ], "layers": [ { "id": "confidence-expression", "name": "Confidence expression", "description": "What the confidence statement is about, which scale it uses, and what reasons produced it.", "source_refs": [ "SRC-015", "SRC-005", "SRC-004" ], "findings": [ { "id": "confidence-object-and-semantics", "name": "Object and semantics of a confidence statement", "description": "Confidence in this model is a statement about the assertion — how much the assessment can be relied upon — not a property of the subject and not a probability that a proposition about the world is true. GRADE frames certainty as certainty that the true value lies on one side of a threshold or within a range, which is explicitly about the evidence base rather than about the estimate alone. Conflating confidence with a model's output probability, with coverage probability, or with a risk score is the principal failure mode this finding exists to prevent.", "source_refs": [ "SRC-015", "SRC-004", "SRC-005", "SRC-018" ], "questions": [ { "id": "q-confidence-object", "text": "What is this confidence statement about: the assertion, the underlying value, the method, or the subject?", "kind": "definition", "answer_data": [ "confidence target reference", "target kind code", "statement of what the confidence does not cover" ] }, { "id": "q-confidence-vs-probability", "text": "Is the confidence value a probability, and if not, what is it?", "kind": "classification", "answer_data": [ "confidence value kind (ordinal grade, probability, interval, qualitative)", "if probability, the proposition it is the probability of", "explicit non-probability declaration where applicable" ] }, { "id": "q-confidence-consumer-rule", "text": "What action is an agent permitted or forbidden to take at each confidence level?", "kind": "decision", "answer_data": [ "action-permission mapping by confidence level", "escalation or human-review trigger", "default behaviour on missing confidence" ] }, { "id": "q-model-output-confidence", "text": "Where the subject is a model output that carries its own score, is that score treated as confidence, and is it calibrated?", "kind": "quality", "answer_data": [ "source of the score", "calibration evidence or its absence", "rule preventing raw scores from being presented as governed confidence" ] } ], "data_elements": [ { "id": "confidence-target", "name": "Confidence target", "description": "Explicit reference to what the confidence statement qualifies.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-015" ] }, { "id": "confidence-value-kind", "name": "Confidence value kind", "description": "Whether the confidence is an ordinal grade, a probability, an interval or a qualitative judgement.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-015", "SRC-004" ] }, { "id": "confidence-value", "name": "Confidence value", "description": "The confidence value itself, interpretable only together with its declared scale.", "value_kind": "other", "cardinality": "1", "required": true, "source_refs": [ "SRC-015" ] } ], "artifacts": [], "inline_only_rationale": "The confidence statement is an inline component of the assertion record; separating it would allow a confidence value to circulate detached from the assertion it qualifies, which is exactly the misuse this finding forbids. Its supporting material is carried by the evidence artifacts finding." }, { "id": "confidence-scale-and-calibration", "name": "Confidence scale declaration and calibration", "description": "A confidence value is uninterpretable without its scale. The scale must be named, versioned and dereferenceable, with its ordered terms and any numeric mapping stated. GRADE's four-level certainty scale is one such governed scale; a model's softmax output is not. Where a numeric confidence is claimed to be calibrated, the calibration evidence — reference set, calibration method, date and observed calibration error — must be recorded, otherwise the value is reported as uncalibrated.", "source_refs": [ "SRC-015", "SRC-011", "SRC-004" ], "questions": [ { "id": "q-scale-identity", "text": "Which confidence scale is used, at which version, and where are its ordered terms defined?", "kind": "interoperability", "answer_data": [ "scale identifier and version", "ordered term list", "term definitions location" ] }, { "id": "q-scale-mapping", "text": "Is there an authorised numeric mapping for the scale terms, and is cross-scale comparison permitted?", "kind": "constraint", "answer_data": [ "term-to-number mapping if any", "cross-scale comparison rule", "prohibition on averaging terms across scales" ] }, { "id": "q-calibration-evidence", "text": "What evidence supports a claim that numeric confidence values are calibrated, and when was it last established?", "kind": "evidence", "answer_data": [ "calibration reference set", "calibration method", "calibration error measure and value", "calibration date" ] }, { "id": "q-uncalibrated-default", "text": "How is an uncalibrated confidence value marked so downstream agents do not treat it as a probability?", "kind": "quality", "answer_data": [ "calibration status flag", "required caveat", "downstream handling rule" ] } ], "data_elements": [ { "id": "confidence-scale-ref", "name": "Confidence scale reference", "description": "Dereferenceable identifier and version of the confidence scale in use.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-015" ] }, { "id": "calibration-status", "name": "Calibration status", "description": "Whether the confidence value is calibrated, uncalibrated, or calibration-unknown.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-004" ] }, { "id": "calibration-evidence-ref", "name": "Calibration evidence reference", "description": "Pointer to the calibration study or record supporting a calibrated claim.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-011" ] } ], "artifacts": [ { "id": "confidence-scale-definition", "name": "Confidence scale definition", "description": "Governed definition of a confidence scale: ordered terms, definitions, any numeric mapping, permitted uses and explicit non-comparability notes.", "media_or_form": [ "scale definition document", "code list / concept scheme", "register entry" ], "serial": false, "identity_strategy": "Scale identifier plus version; superseded versions retained so historical assertions remain interpretable.", "source_refs": [ "SRC-015", "SRC-001" ] } ], "inline_only_rationale": null }, { "id": "confidence-basis-evidence-and-agreement", "name": "Confidence basis: evidence and agreement", "description": "Confidence must be decomposed into named reasons rather than asserted. GRADE names domains that lower certainty (risk of bias, imprecision, inconsistency, indirectness, publication bias) and domains that raise it (large effect, dose-response gradient, plausible confounding that would only strengthen the finding). This model generalises that pattern: record the type, amount and consistency of evidence, the degree of agreement among independent assessments, and each rating-down or rating-up decision with its reason, so the grade is reconstructible and challengeable.", "source_refs": [ "SRC-015", "SRC-016", "SRC-009" ], "questions": [ { "id": "q-evidence-basis", "text": "What type and amount of evidence underpins this assertion, and how consistent is it?", "kind": "evidence", "answer_data": [ "evidence type codes", "evidence volume or count", "consistency assessment" ] }, { "id": "q-agreement", "text": "How much do independent assessors, methods or sources agree, and how was agreement measured?", "kind": "quality", "answer_data": [ "number of independent assessments", "agreement measure and value", "description of any dissent" ] }, { "id": "q-rating-adjustments", "text": "Which factors lowered or raised the confidence rating, and what is the recorded reason for each adjustment?", "kind": "decision", "answer_data": [ "rating-down domains applied with reasons", "rating-up domains applied with reasons", "net resulting grade" ] }, { "id": "q-basis-completeness", "text": "What evidence that would materially change the rating is known to be missing or inaccessible?", "kind": "exception", "answer_data": [ "known evidence gaps", "inaccessible sources and why", "expected direction of change if obtained" ] } ], "data_elements": [ { "id": "evidence-basis-summary", "name": "Evidence basis summary", "description": "Structured summary of evidence type, amount and consistency underpinning the confidence grade.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-015" ] }, { "id": "agreement-measure", "name": "Agreement measure", "description": "Measured or judged agreement across independent assessments or sources.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-015", "SRC-016" ] }, { "id": "rating-adjustment", "name": "Rating adjustment record", "description": "One recorded decision that lowered or raised the confidence grade, with the domain and reason.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-015" ] } ], "artifacts": [], "inline_only_rationale": "The basis is a set of structured reason records inline to the assertion; the underlying documents they cite are already modelled as evidence artifacts in the adjacent finding, and duplicating them here would create two competing evidence stores." }, { "id": "metaquality-of-the-evaluation", "name": "Metaquality", "description": "A quality report can itself be wrong, biased or unrepresentative. Metaquality elements record confidence in the quality evaluation, whether the evaluated sample represents the quality unit, and whether quality is homogeneous across the unit. This is quality-of-quality, not a second copy of the host score.", "source_refs": [ "SRC-022", "SRC-001" ], "questions": [ { "id": "metaquality-of-the-evaluation-q01", "text": "What metaquality confidence is assigned to this quality result, and on what evidence?", "kind": "quality", "answer_data": [ "metaquality_confidence", "meta_method", "meta_evidence_refs" ] }, { "id": "metaquality-of-the-evaluation-q02", "text": "How representative of the declared quality unit is the evaluated sample or census?", "kind": "evidence", "answer_data": [ "representativity_statement", "sample_frame_id", "coverage_of_unit" ] }, { "id": "metaquality-of-the-evaluation-q03", "text": "Is quality homogeneous across the unit, or do partitions differ enough that a single score is misleading?", "kind": "quality", "answer_data": [ "homogeneity_flag", "heterogeneous_partition_ids", "disaggregation_refs" ] }, { "id": "metaquality-of-the-evaluation-q04", "text": "Who assessed metaquality, and is that party independent of the original assessor?", "kind": "ownership", "answer_data": [ "meta_assessor_id", "independence_flag" ] } ], "data_elements": [ { "id": "metaquality-of-the-evaluation-data01", "name": "Metaquality confidence", "description": "Stated confidence in the quality evaluation.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-022" ] }, { "id": "metaquality-of-the-evaluation-data02", "name": "Representativity statement", "description": "How the evaluation represents the quality unit.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-022" ] }, { "id": "metaquality-of-the-evaluation-data03", "name": "Homogeneity flag", "description": "Whether quality is treated as homogeneous across the unit.", "value_kind": "boolean", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-022" ] } ], "artifacts": [ { "id": "metaquality-of-the-evaluation-artifact01", "name": "Metaquality statement", "description": "Record of confidence, representativity and homogeneity about a quality result.", "media_or_form": [ "application/json", "application/xml" ], "serial": true, "identity_strategy": "Metaquality assertion identifier referencing the evaluated quality assertion.", "source_refs": [ "SRC-022" ] } ], "inline_only_rationale": null } ] }, { "id": "evidence-and-reproducibility", "name": "Evidence and reproducibility", "description": "The retrievable material that substantiates the assertion and allows an independent party to re-run it.", "source_refs": [ "SRC-018", "SRC-013", "SRC-011" ], "findings": [ { "id": "evidence-artifacts-integrity-and-reproducibility", "name": "Evidence artifacts, integrity and re-execution", "description": "An assertion is only as defensible as the material a challenger can retrieve. W3C VC Data Model 2.0 provides an evidence property, cryptographic securing mechanisms and credential status for tamper-evidence and revocation; the EU AI Act requires technical documentation and retained logs that substantiate declared performance. This finding requires that each assertion either point to retrievable, integrity-protected evidence with re-execution instructions, or declare explicitly that it is an unsubstantiated judgement.", "source_refs": [ "SRC-018", "SRC-013", "SRC-011", "SRC-019" ], "questions": [ { "id": "q-evidence-items", "text": "Which retrievable evidence items substantiate this assertion, and what does each one show?", "kind": "evidence", "answer_data": [ "evidence item references", "evidence kind per item", "what each item establishes" ] }, { "id": "q-evidence-integrity", "text": "How is the integrity of each evidence item protected and verified at retrieval time?", "kind": "security", "answer_data": [ "content hash or digest algorithm and value", "signature or securing mechanism", "verification procedure" ] }, { "id": "q-reexecution", "text": "What is needed to re-execute this assessment and reproduce the result within stated tolerance?", "kind": "process", "answer_data": [ "input snapshot references", "procedure and tool versions", "seeds and configuration", "reproduction tolerance" ] }, { "id": "q-unsubstantiated", "text": "If no evidence is retained, is the assertion explicitly marked as an unsubstantiated judgement?", "kind": "quality", "answer_data": [ "substantiation flag", "reason evidence is absent", "restricted uses that follow" ] }, { "id": "q-evidence-access", "text": "Is any evidence item access-restricted, and what may a consumer be told when they cannot retrieve it?", "kind": "access", "answer_data": [ "access classification per item", "redacted summary available to non-privileged consumers", "contact or request path" ] } ], "data_elements": [ { "id": "evidence-ref", "name": "Evidence reference", "description": "Reference to a retrievable evidence item supporting the assertion.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-018", "SRC-013" ] }, { "id": "evidence-digest", "name": "Evidence digest", "description": "Cryptographic digest binding the assertion to the exact evidence content it was based on.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-018" ] }, { "id": "substantiation-flag", "name": "Substantiation flag", "description": "Whether the assertion is evidence-backed, partially backed, or an unsubstantiated judgement.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-013", "SRC-018" ] } ], "artifacts": [ { "id": "assessment-evidence-package", "name": "Assessment evidence package", "description": "Retained bundle of raw outputs, logs, validation reports, intermediate results and re-execution instructions for one assessment run.", "media_or_form": [ "log files", "raw result set", "validation report", "executable notebook or script", "signed archive" ], "serial": true, "identity_strategy": "Package identifier plus run identifier and content digest; retention governed by the retention finding rather than by storage convenience.", "source_refs": [ "SRC-013", "SRC-011", "SRC-018" ] }, { "id": "quality-certificate", "name": "Quality certificate or attestation", "description": "A statement by an authorised or accredited party associating the subject with a certification or attestation of quality, optionally secured as a verifiable credential with issuer, proof and status.", "media_or_form": [ "certificate document", "verifiable credential", "attestation record" ], "serial": true, "identity_strategy": "Issuer-assigned certificate identifier; status resolved through the issuer's revocation mechanism, never cached as final.", "source_refs": [ "SRC-001", "SRC-018" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "requirement-conformance-and-fitness", "name": "Requirement, conformance and fitness", "description": "Turning measured quality into a decision: thresholds, verdicts and a declared judgement of fitness for a stated purpose.", "rationale": "ISO 19157-1 explicitly declines to set minimum acceptable quality levels, locating them in product specifications; JCGM 106 addresses the role of measurement uncertainty in conformity assessment; the EU AI Act requires declared accuracy levels and metrics in the instructions for use. Requirement, verdict and fitness must therefore be modelled separately from the measurement itself.", "source_refs": [ "SRC-003", "SRC-004", "SRC-013", "SRC-016" ], "layers": [ { "id": "requirements-and-conformance", "name": "Requirements and conformance", "description": "Stated acceptance criteria and the verdict derived from measurement against them.", "source_refs": [ "SRC-003", "SRC-004", "SRC-007" ], "findings": [ { "id": "quality-requirement-and-acceptance-threshold", "name": "Quality requirement and acceptance threshold", "description": "A threshold is a governed object separate from both the measure and the result: it names the measure, the acceptance limit, the direction of the comparison, its owner and its effective period. Because measurement uncertainty affects conformity decisions, the decision rule must state how uncertainty is handled — for example acceptance limits with guard bands — rather than comparing a point estimate to a limit and calling it conformity.", "source_refs": [ "SRC-003", "SRC-004", "SRC-013" ], "questions": [ { "id": "q-threshold-definition", "text": "What acceptance limit applies to this measure, in which direction, and who set it?", "kind": "requirement", "answer_data": [ "measure reference", "acceptance limit value and direction", "owning authority", "effective period" ] }, { "id": "q-decision-rule", "text": "How does the decision rule account for measurement uncertainty when comparing the result to the limit?", "kind": "decision", "answer_data": [ "decision rule identifier", "guard band or acceptance interval", "risk of false acceptance/rejection accepted" ] }, { "id": "q-threshold-provenance", "text": "On what basis was the threshold chosen — regulation, contract, product specification, or internal convention?", "kind": "authority", "answer_data": [ "threshold basis code", "citation to the governing instrument", "approval record" ] }, { "id": "q-threshold-change", "text": "What happens to existing verdicts when a threshold changes?", "kind": "lifecycle", "answer_data": [ "re-evaluation policy", "verdict restatement rule", "effective-date handling" ] } ], "data_elements": [ { "id": "requirement-ref", "name": "Requirement reference", "description": "Identifier of the governing requirement or product specification clause.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003", "SRC-013" ] }, { "id": "acceptance-limit", "name": "Acceptance limit", "description": "The threshold value and comparison direction applied to the measure result.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003", "SRC-004" ] }, { "id": "decision-rule-ref", "name": "Decision rule reference", "description": "Documented rule for converting a measured value with uncertainty into a conformity decision.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] } ], "artifacts": [ { "id": "quality-requirement-specification", "name": "Quality requirement specification", "description": "Governed statement of required dimensions, measures, acceptance limits, decision rules and effective dates for a subject class.", "media_or_form": [ "specification document", "machine-readable requirement record", "contract annex" ], "serial": false, "identity_strategy": "Specification identifier plus version and effective date range; never identified by publication date alone.", "source_refs": [ "SRC-003", "SRC-013" ] } ], "inline_only_rationale": null }, { "id": "conformance-verdict-severity-and-decision-rule", "name": "Conformance verdict and severity", "description": "A verdict is a derived assertion that cites both a result and a requirement. Constraint validation supplies a genuinely binary conformance signal with organisational severities (violation, warning, informational); graded measurement supplies a score. The model keeps them distinct: a verdict must name the requirement it was evaluated against, and a score may never be presented as a verdict. Severity classifies consequence, not degree of failure.", "source_refs": [ "SRC-007", "SRC-003", "SRC-004" ], "questions": [ { "id": "q-verdict-value", "text": "What is the verdict, against which requirement, and using which result?", "kind": "validation", "answer_data": [ "verdict code (conforms, does not conform, conditional, not evaluated)", "requirement reference", "result reference" ] }, { "id": "q-verdict-severity", "text": "What severity is attached to each non-conformance, and what does that severity oblige the consumer to do?", "kind": "constraint", "answer_data": [ "severity code", "obligation or blocking behaviour", "escalation path" ] }, { "id": "q-verdict-derivation", "text": "Is the verdict reproducible from the recorded result, requirement and decision rule alone?", "kind": "provenance", "answer_data": [ "derivation record", "inputs cited", "recomputation check outcome" ] }, { "id": "q-partial-conformance", "text": "How are partial, conditional or waived conformance states represented without collapsing them into pass?", "kind": "exception", "answer_data": [ "conditional-state code", "waiver reference and approver", "expiry of the waiver" ] } ], "data_elements": [ { "id": "verdict-code", "name": "Verdict code", "description": "Conformance outcome against a named requirement.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-007", "SRC-003" ] }, { "id": "verdict-severity", "name": "Verdict severity", "description": "Consequence classification of a non-conformance, distinct from the magnitude of the deviation.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007" ] }, { "id": "waiver-ref", "name": "Waiver reference", "description": "Reference to an approved, time-bounded waiver permitting continued use despite non-conformance.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003", "SRC-013" ] } ], "artifacts": [ { "id": "conformance-report", "name": "Conformance report", "description": "Report enumerating verdicts, the requirements evaluated, the results cited and per-item severity, including machine-readable validation results where constraint checking was used.", "media_or_form": [ "validation report", "conformance report document", "structured result set" ], "serial": true, "identity_strategy": "Report identifier plus run identifier; each report immutable once issued and superseded rather than edited.", "source_refs": [ "SRC-007", "SRC-003" ] } ], "inline_only_rationale": null } ] }, { "id": "fitness-and-limitations", "name": "Fitness for purpose and limitations", "description": "The purpose-relative judgement and the explicit statement of what the assertion does not support.", "source_refs": [ "SRC-016", "SRC-013", "SRC-009" ], "findings": [ { "id": "intended-use-fitness-and-declared-limitations", "name": "Intended use, fitness for purpose and declared limitations", "description": "Quality is only decidable relative to a purpose: the European Statistics Code of Practice defines relevance by user need, ISO 19157-1 frames quality information as helping users decide whether data suffice for their particular application, and the EU AI Act requires providers to declare accuracy levels and metrics in the instructions for use. A fitness statement must therefore name the purpose it was judged against and, symmetrically, state the known defects, exclusions and uses the assertion does not support — otherwise absence of a caveat is read as absence of a limitation.", "source_refs": [ "SRC-016", "SRC-003", "SRC-013", "SRC-009" ], "questions": [ { "id": "q-declared-purpose", "text": "For which stated purpose was fitness judged, and by whom?", "kind": "decision", "answer_data": [ "purpose statement or identifier", "judging agent", "date of judgement" ] }, { "id": "q-fitness-verdict", "text": "Is the subject judged fit, fit with conditions, or unfit for that purpose, and on which assertions does the judgement rest?", "kind": "validation", "answer_data": [ "fitness verdict code", "conditions attached", "cited assertion references" ] }, { "id": "q-known-defects", "text": "What known defects, exclusions or systematic errors are present that a consumer must be told about?", "kind": "quality", "answer_data": [ "defect list with affected scope", "estimated magnitude or frequency", "mitigation or workaround" ] }, { "id": "q-forbidden-uses", "text": "Which uses does this assessment explicitly not support, and where is that recorded so it travels with the data?", "kind": "constraint", "answer_data": [ "out-of-scope use list", "required warning text", "propagation rule for derived products" ] }, { "id": "q-purpose-change", "text": "What happens to the fitness judgement when the intended purpose changes?", "kind": "lifecycle", "answer_data": [ "re-judgement trigger", "invalidation rule", "notification obligation" ] } ], "data_elements": [ { "id": "declared-purpose", "name": "Declared purpose", "description": "The stated use against which fitness is judged; required for any fitness verdict.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-016", "SRC-003" ] }, { "id": "fitness-verdict", "name": "Fitness verdict", "description": "Purpose-relative judgement: fit, fit with conditions, or unfit.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-003", "SRC-016" ] }, { "id": "known-defect", "name": "Known defect record", "description": "One recorded defect, exclusion or systematic error with its affected scope.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-013", "SRC-003" ] }, { "id": "forbidden-use", "name": "Non-supported use statement", "description": "Explicit statement of uses the assessment does not support, intended to travel with derived products.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-013" ] } ], "artifacts": [ { "id": "quality-statement-for-users", "name": "User-facing quality statement", "description": "Consumer-readable statement combining declared purpose, headline results, confidence, limitations and non-supported uses — the analogue of a quality report or instructions-for-use section.", "media_or_form": [ "quality report document", "dataset or model documentation section", "instructions-for-use section" ], "serial": true, "identity_strategy": "Statement identifier plus version, bound to the assertion identifiers it summarises; regenerated rather than hand-edited when assertions change.", "source_refs": [ "SRC-016", "SRC-013", "SRC-003" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "time-lifecycle-and-provenance", "name": "Time, lifecycle and provenance", "description": "When the assertion refers to, when it was produced, how long it remains valid, how it is superseded, and who stands behind it.", "rationale": "Quality decays. SOSA separates the time a phenomenon obtained from the time a result became available; PROV supplies attribution and derivation; DQV wraps quality statements in a container so they can carry joint provenance. Currency and timeliness are themselves governed quality dimensions in official statistics practice.", "source_refs": [ "SRC-008", "SRC-006", "SRC-001", "SRC-016" ], "layers": [ { "id": "time-and-validity", "name": "Time and validity", "description": "Distinct timestamps and the period over which an assertion may be relied on.", "source_refs": [ "SRC-008", "SRC-016", "SRC-018" ], "findings": [ { "id": "assessment-time-and-observation-time", "name": "Assessment time versus subject state time", "description": "At least three times must be distinguishable: the time of the subject state that was assessed, the time the assessment produced its result, and the time the assertion was recorded or ingested. SOSA's separation of phenomenonTime from resultTime is the canonical pattern. All times use RFC 3339 with seconds and an explicit offset or Z; a date alone is never sufficient and is never an identifier.", "source_refs": [ "SRC-008", "SRC-006", "SRC-001" ], "questions": [ { "id": "q-subject-state-time", "text": "Which state of the subject, at which instant or interval, does this assertion describe?", "kind": "temporal", "answer_data": [ "subject state timestamp or interval", "subject version reference", "timezone offset" ] }, { "id": "q-result-time", "text": "When was the assessment executed and when did its result become available?", "kind": "temporal", "answer_data": [ "assessment start and end timestamps", "result availability timestamp", "execution duration" ] }, { "id": "q-record-time", "text": "When was the assertion recorded or ingested into the adopting Dimension, and how does that differ from the result time?", "kind": "provenance", "answer_data": [ "record/ingestion timestamp", "recording agent", "lag between result and record" ] }, { "id": "q-time-precision", "text": "What time precision and clock source are used, and is the precision adequate for the decisions taken on this assertion?", "kind": "quality", "answer_data": [ "timestamp precision", "clock source or synchronisation claim", "precision adequacy statement" ] } ], "data_elements": [ { "id": "subject-state-time", "name": "Subject state time", "description": "Instant or interval of the subject state assessed, RFC 3339 with explicit offset or Z.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-008" ] }, { "id": "assessment-result-time", "name": "Assessment result time", "description": "Time the assessment result became available.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-008", "SRC-006" ] }, { "id": "record-time", "name": "Record/ingestion time", "description": "Time the assertion was recorded in the adopting Dimension, kept separate from result time.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-006" ] } ], "artifacts": [], "inline_only_rationale": "Timestamps are inline attributes of the assertion; an external time artifact would create a second source of truth for the very field used to order and supersede assertions." }, { "id": "validity-currency-and-reassessment", "name": "Validity period, currency and re-assessment", "description": "Assertions have a shelf life. Credential models express this with validFrom/validUntil; official statistics treat timeliness and punctuality as governed dimensions. This model requires an explicit validity policy per measure: a validity window or a decay statement, a re-assessment cadence, and defined consumer behaviour when an assertion is stale — never silent reuse of an expired score.", "source_refs": [ "SRC-018", "SRC-016", "SRC-013" ], "questions": [ { "id": "q-validity-window", "text": "From when until when may this assertion be relied upon?", "kind": "temporal", "answer_data": [ "valid-from timestamp", "valid-until timestamp or open-ended flag", "basis for the window" ] }, { "id": "q-staleness-behaviour", "text": "What must a consuming agent do when the assertion is past its validity window?", "kind": "state", "answer_data": [ "stale-handling rule", "fallback behaviour", "whether stale assertions may still be cited historically" ] }, { "id": "q-reassessment-cadence", "text": "How often must this measure be re-assessed, and what events force early re-assessment?", "kind": "process", "answer_data": [ "cadence specification", "triggering events (subject change, method change, threshold change, incident)", "owner of the cadence" ] }, { "id": "q-decay-model", "text": "Does quality on this dimension decay predictably over time, and is that decay modelled or merely assumed?", "kind": "measurement", "answer_data": [ "decay model or none", "evidence for the decay claim", "effect on confidence over time" ] } ], "data_elements": [ { "id": "valid-from", "name": "Valid from", "description": "Start of the period in which the assertion may be relied upon.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-018" ] }, { "id": "valid-until", "name": "Valid until", "description": "End of the reliance period; absence must be explicit rather than implied.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-018" ] }, { "id": "reassessment-cadence", "name": "Re-assessment cadence", "description": "Required interval or event-driven trigger for repeating the assessment.", "value_kind": "duration", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016", "SRC-013" ] } ], "artifacts": [], "inline_only_rationale": "Validity is inline policy metadata on the assertion; the governed cadence itself lives in the quality-profile-definition artifact and is referenced rather than duplicated." } ] }, { "id": "lifecycle-and-supersession", "name": "Lifecycle and supersession", "description": "States an assertion moves through and the rules that keep the historical record honest.", "source_refs": [ "SRC-006", "SRC-012", "SRC-014" ], "findings": [ { "id": "assertion-lifecycle-supersession-and-immutability", "name": "Assertion lifecycle, supersession and immutability", "description": "Once issued, a quality assertion is a historical fact about what was believed at a time and must not be silently edited. Register practice models item status (valid, superseded, retired) with explicit supersession pointers; credential practice adds revocation through status checking. This model requires a declared state set, an append-only supersession chain, and a distinction between correcting a defective assertion (retract plus reissue with reason) and producing a new assessment (new assertion, prior one remains valid for its period).", "source_refs": [ "SRC-012", "SRC-018", "SRC-006", "SRC-014" ], "questions": [ { "id": "q-state-set", "text": "What are the permitted lifecycle states of an assertion and which transitions are legal?", "kind": "lifecycle", "answer_data": [ "state list (e.g. draft, issued, superseded, retracted, disputed)", "legal transition map", "role authorised for each transition" ] }, { "id": "q-correction-vs-reassessment", "text": "Is this record a correction of a defective assertion or an independent new assessment?", "kind": "state", "answer_data": [ "change kind code", "retraction reason where applicable", "link to the superseded assertion" ] }, { "id": "q-immutability", "text": "What guarantees that an issued assertion is not altered after the fact, and how is that verifiable?", "kind": "security", "answer_data": [ "immutability mechanism", "content digest at issuance", "append-only audit trail reference" ] }, { "id": "q-retraction-effects", "text": "When an assertion is retracted, what must happen to decisions and derived assertions that relied on it?", "kind": "event", "answer_data": [ "downstream notification obligation", "re-evaluation requirement for derived assertions", "record of affected decisions" ] } ], "data_elements": [ { "id": "lifecycle-state", "name": "Lifecycle state", "description": "Current state of the assertion within the declared state set.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-012", "SRC-018" ] }, { "id": "supersedes-ref", "name": "Supersedes reference", "description": "Pointer to the assertion this record supersedes or retracts, forming an append-only chain.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012", "SRC-006" ] }, { "id": "change-reason", "name": "Change reason", "description": "Reason for supersession, correction or retraction, required for auditability.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014", "SRC-012" ] } ], "artifacts": [ { "id": "assertion-change-log", "name": "Assertion change log", "description": "Append-only record of lifecycle transitions with actor, timestamp, prior and new state, and reason.", "media_or_form": [ "append-only log", "event stream", "audit table" ], "serial": true, "identity_strategy": "Log entries keyed by assertion identifier plus RFC 3339 transition timestamp and a monotonic sequence number.", "source_refs": [ "SRC-012", "SRC-014", "SRC-013" ] } ], "inline_only_rationale": null } ] }, { "id": "attribution-and-provenance", "name": "Attribution and provenance", "description": "Who made the assertion, on whose authority, with what independence, and from which inputs.", "source_refs": [ "SRC-006", "SRC-001", "SRC-016" ], "findings": [ { "id": "assessor-attribution-authority-and-independence", "name": "Assessor attribution, competence, authority and independence", "description": "A quality assertion inherits credibility from its author. PROV supplies attribution and qualified attribution with roles; DQV distinguishes producer measurements, third-party certificates and user feedback precisely because their standing differs. This model additionally records competence or accreditation where claimed, the authority under which the assertion is issued, and independence or conflict of interest relative to the subject's owner — a self-assessment and an independent audit are not interchangeable and must not be presented as equivalent.", "source_refs": [ "SRC-006", "SRC-001", "SRC-016", "SRC-017" ], "questions": [ { "id": "q-assessor-identity", "text": "Which agent produced this assertion, in which role, and on whose behalf?", "kind": "ownership", "answer_data": [ "assessor agent reference", "role in the assessment", "commissioning or responsible organisation" ] }, { "id": "q-assessor-authority", "text": "Under what authority, accreditation or delegated mandate is this assertion issued?", "kind": "authority", "answer_data": [ "authority basis", "accreditation or certification reference", "scope of the mandate" ] }, { "id": "q-independence", "text": "What is the assessor's relationship to the subject's owner, and is any conflict of interest declared?", "kind": "provenance", "answer_data": [ "independence classification (self-assessment, second-party, independent third-party)", "declared conflicts", "mitigations applied" ] }, { "id": "q-competence-evidence", "text": "What evidence supports the assessor's competence for this specific measure?", "kind": "evidence", "answer_data": [ "qualification or training record", "prior calibration/proficiency results", "scope of demonstrated competence" ] } ], "data_elements": [ { "id": "assessor-ref", "name": "Assessor reference", "description": "Reference to the agent credited with the assertion, resolved in a sibling agent model.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-006", "SRC-001" ] }, { "id": "independence-class", "name": "Independence classification", "description": "Whether the assessment is self, second-party or independent third-party.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-016", "SRC-001" ] }, { "id": "authority-basis", "name": "Authority basis", "description": "Instrument, accreditation or delegation under which the assertion is issued.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-017", "SRC-013" ] } ], "artifacts": [], "inline_only_rationale": "Attribution fields are inline references into the sibling Agent/Party model; the accreditation documents themselves are that model's artifacts, and copying them here would duplicate an authoritative record outside its owning model." }, { "id": "provenance-lineage-and-derived-assertions", "name": "Provenance lineage and derived assertions", "description": "Assertions are frequently derived: rolled up across parts, weighted into composite indices, propagated to downstream products, or recomputed from other assertions. PROV's wasDerivedFrom, used and wasGeneratedBy make that lineage explicit, and DQV's quality-metadata container lets certificates, policies, measurements and annotations carry joint provenance as a group. Aggregation is a lineage operation with its own risks: composite scores hide dimension-level failures and are invalid across incompatible scales, so the aggregation function, weights and their justification must be recorded with the derived assertion.", "source_refs": [ "SRC-006", "SRC-001", "SRC-016", "SRC-003" ], "questions": [ { "id": "q-derivation-inputs", "text": "From which assertions, datasets or activities was this assertion derived?", "kind": "provenance", "answer_data": [ "input assertion references", "input dataset snapshots", "generating activity reference" ] }, { "id": "q-aggregation-function", "text": "If this is a composite or rolled-up score, what function and weights produced it and who approved them?", "kind": "composition", "answer_data": [ "aggregation function specification", "weight vector and rationale", "approving authority" ] }, { "id": "q-aggregation-validity", "text": "Are the inputs on compatible scales and scopes such that aggregation is meaningful?", "kind": "constraint", "answer_data": [ "scale compatibility check outcome", "scope alignment check outcome", "refusal condition when incompatible" ] }, { "id": "q-dimension-visibility", "text": "Can a consumer recover the dimension-level results behind a composite score?", "kind": "interoperability", "answer_data": [ "links to component assertions", "decomposition availability flag", "rule against publishing composites without components" ] }, { "id": "q-propagation", "text": "How does an assertion about an input propagate to products derived from that input?", "kind": "relationship", "answer_data": [ "propagation rule", "recomputation versus inheritance policy", "staleness handling for inherited assertions" ] } ], "data_elements": [ { "id": "derived-from-ref", "name": "Derived-from reference", "description": "Reference to an assertion or resource this assertion was derived from.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-006" ] }, { "id": "aggregation-spec", "name": "Aggregation specification", "description": "Function, weights and preconditions used to combine component results into a composite.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003", "SRC-016" ] }, { "id": "generating-activity-ref", "name": "Generating activity reference", "description": "Reference to the activity that generated this assertion, held in the sibling provenance model.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-006" ] } ], "artifacts": [ { "id": "quality-metadata-container", "name": "Quality metadata container", "description": "Grouping of related quality statements — measurements, annotations, certificates and policies about one subject — held together so the group can carry joint provenance, versioning and access rules.", "media_or_form": [ "named graph", "metadata bundle", "catalogue record section" ], "serial": true, "identity_strategy": "Container identifier distinct from both the subject identifier and the identifiers of the statements it contains.", "source_refs": [ "SRC-001", "SRC-006" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "interoperability-disagreement-and-governance", "name": "Interoperability, disagreement and governance", "description": "Exchanging quality statements without false equivalence, handling competing and disputed assessments, and governing access and retention.", "rationale": "Because no single quality vocabulary is universal, exchange requires explicit alignment and explicit non-comparability rules. DQV exists to make heterogeneous assessments comparable in shape while accepting that fitness for purpose varies; the Code of Practice treats comparability as a governed dimension; GDPR creates enforceable rectification and erasure duties over quality-relevant assertions about people.", "source_refs": [ "SRC-001", "SRC-016", "SRC-014", "SRC-013" ], "layers": [ { "id": "interoperability-and-comparability", "name": "Interoperability and comparability", "description": "Alignment to external vocabularies, and the rules that stop incomparable scores from being compared.", "source_refs": [ "SRC-001", "SRC-002", "SRC-016" ], "findings": [ { "id": "external-alignment-and-conflicts", "name": "External alignment, mappings and recorded conflicts", "description": "External standards are alignments, not conformance claims. This model maps its slots to DQV classes and properties, to ISO/IEC 25012 and ISO/IEC 5259-2 dimension vocabularies, to ISO 19157-1 measure components, to JCGM uncertainty terms and to PROV attribution, recording for each mapping whether it is exact, broader, narrower or merely related. Conformance to any of these is asserted only where tested evidence exists; recorded conflicts include the fact that constraint-validation severities are not quality grades and that credential formats carry no confidence property.", "source_refs": [ "SRC-001", "SRC-002", "SRC-010", "SRC-003", "SRC-005", "SRC-007", "SRC-018", "SRC-019" ], "questions": [ { "id": "q-mapping-target", "text": "To which external term does each slot of this model map, and at what mapping strength?", "kind": "interoperability", "answer_data": [ "source slot identifier", "target term IRI and standard version", "mapping strength (exact, broader, narrower, related)" ] }, { "id": "q-conformance-evidence", "text": "Is conformance to a cited standard claimed, and what test evidence supports the claim?", "kind": "evidence", "answer_data": [ "conformance claim status", "test or profile reference", "scope and limits of the claim" ] }, { "id": "q-mapping-conflicts", "text": "Which known conflicts or semantic mismatches exist between aligned standards, and how are they resolved?", "kind": "exception", "answer_data": [ "conflict description", "affected slots", "resolution rule and its owner" ] }, { "id": "q-mapping-versioning", "text": "How are mappings maintained when an aligned standard is revised?", "kind": "lifecycle", "answer_data": [ "mapping version and effective date", "revision monitoring owner", "re-mapping trigger" ] } ], "data_elements": [ { "id": "mapping-entry", "name": "Alignment mapping entry", "description": "One mapping from a model slot to an external term with strength, standard version and notes.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001", "SRC-019" ] }, { "id": "conformance-claim", "name": "Conformance claim", "description": "Explicit statement of whether conformance to a standard is claimed and on what evidence.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-003", "SRC-013" ] } ], "artifacts": [ { "id": "alignment-crosswalk", "name": "Alignment crosswalk", "description": "Maintained table of model slots against external standard terms with mapping strength, standard version, conflicts and open questions.", "media_or_form": [ "crosswalk table", "machine-readable mapping set", "specification annex" ], "serial": false, "identity_strategy": "Crosswalk identifier plus version, with per-row references to the exact standard version mapped.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003" ] } ], "inline_only_rationale": null }, { "id": "comparability-and-non-comparability-rules", "name": "Comparability and non-comparability rules", "description": "The most damaging misuse of quality data is comparing values that are not comparable. Coherence and comparability are treated as a governed output-quality principle in official statistics precisely because comparability must be established, not assumed. Two results are comparable only if the measure and its version, the parameter binding, the scope and population, the scale and unit, the reference basis and the validity period are compatible. Confidence grades on different scales are never comparable and never averageable.", "source_refs": [ "SRC-016", "SRC-003", "SRC-015", "SRC-001" ], "questions": [ { "id": "q-comparability-conditions", "text": "Under what stated conditions may two results of this measure be compared or trended?", "kind": "constraint", "answer_data": [ "comparability condition list", "tolerance for parameter differences", "break-in-series markers" ] }, { "id": "q-scale-mixing", "text": "Which combinations of scales, units or confidence vocabularies are explicitly forbidden from being combined?", "kind": "validation", "answer_data": [ "forbidden combination list", "detection rule", "required error or refusal behaviour" ] }, { "id": "q-series-breaks", "text": "How is a break in comparability caused by a method or threshold change signalled to consumers?", "kind": "event", "answer_data": [ "break marker and effective date", "cause description", "guidance on pre/post comparison" ] }, { "id": "q-normalisation", "text": "When results are normalised for comparison, what normalisation was applied and is it reversible?", "kind": "process", "answer_data": [ "normalisation function", "parameters used", "reversibility statement" ] } ], "data_elements": [ { "id": "comparability-key", "name": "Comparability key", "description": "Composite of measure version, parameter binding, scope, scale and reference basis used to decide whether two results may be compared.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-016", "SRC-003" ] }, { "id": "series-break-marker", "name": "Series break marker", "description": "Flag and effective date indicating that comparability with earlier results is broken.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] } ], "artifacts": [], "inline_only_rationale": "Comparability is expressed as derived rules over fields already carried by the assertion and the measure register entry; materialising it as a separate artifact would let a stale comparability claim outlive the fields it is computed from." } ] }, { "id": "disagreement-and-feedback", "name": "Disagreement, feedback and rectification", "description": "Multiple assessors, contested assessments and the enforceable right to have inaccurate assertions corrected.", "source_refs": [ "SRC-001", "SRC-014", "SRC-016" ], "findings": [ { "id": "competing-assessments-disagreement-and-rectification", "name": "Competing assessments, disagreement and rectification", "description": "Several parties may legitimately assess the same subject and disagree — DQV explicitly accommodates producer measurements, third-party certificates and user feedback side by side, since fitness for purpose varies by consumer. Disagreement must be represented, not silently resolved by last-writer-wins. Where assertions concern personal data, disagreement acquires legal force: the accuracy principle obliges rectification or erasure of inaccurate data, and data subjects hold rights to rectification and erasure, so a dispute channel with defined outcomes and deadlines is mandatory rather than optional.", "source_refs": [ "SRC-001", "SRC-014", "SRC-016" ], "questions": [ { "id": "q-competing-set", "text": "Which other assertions cover the same subject, scope and dimension, and do they conflict?", "kind": "relationship", "answer_data": [ "sibling assertion references", "conflict detection outcome", "magnitude and direction of disagreement" ] }, { "id": "q-precedence", "text": "What precedence rule applies when assertions conflict, and who owns that rule?", "kind": "decision", "answer_data": [ "precedence rule (authority, recency, independence, method strength)", "rule owner", "explicit prohibition of silent merging" ] }, { "id": "q-dispute-channel", "text": "How does an affected party challenge an assertion, and what outcomes and deadlines apply?", "kind": "exception", "answer_data": [ "dispute intake channel", "permitted outcomes (uphold, amend, retract, annotate)", "response deadline", "appeal path" ] }, { "id": "q-rectification-duty", "text": "Where the assertion concerns personal data, what rectification or erasure duty applies and how is it evidenced?", "kind": "privacy", "answer_data": [ "applicable legal basis and article", "action taken and date", "notification of recipients", "retained evidence of the action" ] }, { "id": "q-disputed-state-visibility", "text": "While a dispute is open, how is the assertion's disputed status made visible to consumers?", "kind": "state", "answer_data": [ "disputed-state flag", "consumer-facing notice", "restrictions on reliance during dispute" ] } ], "data_elements": [ { "id": "conflicting-assertion-ref", "name": "Conflicting assertion reference", "description": "Reference to another assertion covering the same subject, scope and dimension with a materially different result.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "dispute-record", "name": "Dispute record", "description": "Record of a challenge to an assertion: raiser, grounds, status, outcome and dates.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-014", "SRC-001" ] }, { "id": "precedence-rule-ref", "name": "Precedence rule reference", "description": "Governed rule determining which assertion prevails when several conflict.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016" ] } ], "artifacts": [ { "id": "user-quality-feedback", "name": "User quality feedback record", "description": "Consumer-originated opinion or complaint about the quality of a subject, retained as a distinct class of quality statement with lower default standing than a measured assertion.", "media_or_form": [ "feedback record", "annotation", "support ticket export" ], "serial": true, "identity_strategy": "Feedback identifier plus submitting party reference; never merged into a measurement record.", "source_refs": [ "SRC-001", "SRC-014" ] } ], "inline_only_rationale": null } ] }, { "id": "governance-access-and-retention", "name": "Governance: access and retention", "description": "Who may see quality statements and their evidence, and how long they are kept.", "source_refs": [ "SRC-014", "SRC-013", "SRC-016" ], "findings": [ { "id": "access-sensitivity-and-disclosure", "name": "Access sensitivity and disclosure of quality statements", "description": "A quality statement can be more sensitive than the subject it describes: a low accuracy score, an open dispute or an unresolved defect can be commercially or legally damaging, and evidence packages may contain personal or confidential data even when the subject does not. Access must therefore be classified independently at bundle, layer, finding and artifact level, with a defined redacted view for consumers who are entitled to know that a limitation exists but not to see the underlying evidence, and with a bias toward disclosing limitations that affect safety.", "source_refs": [ "SRC-014", "SRC-013", "SRC-016", "SRC-018" ], "questions": [ { "id": "q-access-classification", "text": "What access classification applies to this assertion, and does it differ from that of the subject?", "kind": "access", "answer_data": [ "classification code", "divergence from subject classification and reason", "classifying authority" ] }, { "id": "q-redacted-view", "text": "What may a consumer without full access be told, and what must never be inferable from the redacted view?", "kind": "security", "answer_data": [ "redacted view specification", "suppressed fields", "inference risks addressed" ] }, { "id": "q-mandatory-disclosure", "text": "Which limitations must be disclosed regardless of classification because they affect safety, rights or legal duties?", "kind": "requirement", "answer_data": [ "mandatory disclosure list", "legal or policy basis", "disclosure channel and timing" ] }, { "id": "q-embargo", "text": "Is the assertion under embargo before a scheduled release, and who may lift it?", "kind": "authority", "answer_data": [ "embargo end timestamp", "release schedule reference", "authorised releaser" ] } ], "data_elements": [ { "id": "access-classification", "name": "Access classification", "description": "Classification governing who may read the assertion, set independently of the subject's classification.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-014", "SRC-013" ] }, { "id": "redaction-profile-ref", "name": "Redaction profile reference", "description": "Specification of the reduced view served to consumers lacking full access.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "embargo-until", "name": "Embargo end time", "description": "Time before which the assertion must not be disclosed, RFC 3339 with explicit offset or Z.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016" ] } ], "artifacts": [], "inline_only_rationale": "Access classification is a governed inline attribute evaluated at request time; the policy documents that define classifications are owned by a sibling access-policy model and are referenced, not copied, so that a stale local copy can never grant access." }, { "id": "retention-deletion-and-legal-hold", "name": "Retention, deletion and legal hold of assertions and evidence", "description": "Retention of quality records is contested between opposing duties. Regulated AI providers must retain technical documentation, logs and performance evidence that substantiate declared accuracy; data protection law imposes storage limitation and grants erasure rights over personal data, including within evidence packages. This finding requires an explicit retention schedule per assertion class and per evidence artifact, a deletion mechanism that preserves the integrity of the historical record where the assertion itself must survive, and a legal-hold override that suspends scheduled deletion.", "source_refs": [ "SRC-013", "SRC-014", "SRC-012" ], "questions": [ { "id": "q-retention-schedule", "text": "How long must this assertion and each of its evidence artifacts be retained, and on what basis?", "kind": "retention", "answer_data": [ "retention period per class", "legal or policy basis", "retention clock start event" ] }, { "id": "q-deletion-mechanism", "text": "When evidence must be deleted but the assertion must survive, what is retained and what is destroyed?", "kind": "process", "answer_data": [ "tombstone or digest-only retention specification", "fields destroyed", "effect on reproducibility and how it is disclosed" ] }, { "id": "q-erasure-request", "text": "How is an erasure request affecting quality evidence executed and evidenced without falsifying the audit trail?", "kind": "privacy", "answer_data": [ "request handling procedure", "records of what was erased and when", "recipients notified" ] }, { "id": "q-legal-hold", "text": "What suspends scheduled deletion, who may impose and release a hold, and how is the hold recorded?", "kind": "authority", "answer_data": [ "hold trigger and scope", "imposing and releasing authority", "hold record with timestamps" ] }, { "id": "q-retention-conflict", "text": "How is a conflict between a mandatory retention duty and an erasure duty resolved and documented?", "kind": "exception", "answer_data": [ "conflict resolution decision and its legal reasoning", "approving role", "documented outcome and residual risk" ] } ], "data_elements": [ { "id": "retention-class", "name": "Retention class", "description": "Retention category assigned to the assertion or evidence artifact, driving the schedule.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-013", "SRC-014" ] }, { "id": "retention-until", "name": "Retention until", "description": "Computed end of the retention period, RFC 3339 with explicit offset or Z.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "legal-hold-flag", "name": "Legal hold flag", "description": "Indicator that scheduled deletion is suspended, with hold reference.", "value_kind": "boolean", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] }, { "id": "deletion-record", "name": "Deletion record", "description": "Tombstone recording what was deleted, when, by whom and under which authority, retained after content destruction.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-014", "SRC-012" ] } ], "artifacts": [ { "id": "retention-schedule", "name": "Retention and disposition schedule", "description": "Governed schedule mapping assertion classes and evidence artifact types to retention periods, disposition actions, legal bases and hold rules.", "media_or_form": [ "schedule table", "policy document", "machine-readable disposition rules" ], "serial": false, "identity_strategy": "Schedule identifier plus version and effective date range; superseded versions retained to explain past disposals.", "source_refs": [ "SRC-013", "SRC-014" ] } ], "inline_only_rationale": null } ] } ] } ] }, "functions": [ { "id": "register-quality-measure", "name": "Register a quality measure", "description": "Create or revise a registered, versioned, dereferenceable measure definition with its dimension, definition, parameters, expected data type and value structure, following register-style registration and maintenance procedures.", "inputs": [ "proposed measure definition", "dimension reference", "parameter and value-type specification", "submitting organisation" ], "outputs": [ "registered measure item with identifier, version and status", "register change record" ], "preconditions": [ "a dimension vocabulary is bound", "the register manager and submission procedure are defined" ], "effects": [ "the measure becomes citable by assertions", "earlier versions are marked superseded with a supersession pointer" ], "source_refs": [ "SRC-012", "SRC-003", "SRC-020", "SRC-001" ] }, { "id": "bind-quality-profile", "name": "Bind a quality profile to a subject class", "description": "Declare which dimensions and measures are mandatory or optional for a subject class, with thresholds, cadence and required confidence handling — the operation that makes this mixin attachable in a specific Dimension.", "inputs": [ "subject class reference", "dimension vocabulary and version", "measure references", "thresholds and cadence" ], "outputs": [ "quality profile definition", "binding effective-date record" ], "preconditions": [ "measures are registered", "an owner is designated for the profile" ], "effects": [ "assessments for that subject class become checkable for completeness against the profile", "unassessed mandatory dimensions become detectable" ], "source_refs": [ "SRC-013", "SRC-016", "SRC-002", "SRC-010" ] }, { "id": "plan-assessment", "name": "Plan an assessment scope and sampling design", "description": "Fix the assessed unit, extent restrictions, inspection mode, sampling design and reference basis before execution, so the resulting score is interpretable and not retrofitted.", "inputs": [ "subject reference", "profile or requirement reference", "population or frame definition" ], "outputs": [ "assessment plan", "sample and split manifest" ], "preconditions": [ "subject and scope are resolvable", "a reference basis is identified or explicitly declared absent" ], "effects": [ "scope and sampling become auditable inputs rather than post-hoc explanations" ], "source_refs": [ "SRC-003", "SRC-011" ] }, { "id": "execute-assessment", "name": "Execute an assessment and record a result", "description": "Run the measure over the planned scope in a recorded environment and emit a result with value, scale, value structure, unit reference and status code.", "inputs": [ "assessment plan", "registered measure and parameter binding", "subject snapshot" ], "outputs": [ "quality assertion with result value", "execution log and raw outputs" ], "preconditions": [ "measure version and parameters are bound", "execution environment is recorded" ], "effects": [ "a new assertion is created in draft state with subject state time, result time and record time set" ], "source_refs": [ "SRC-001", "SRC-003", "SRC-011" ] }, { "id": "evaluate-uncertainty", "name": "Evaluate and attach measurement uncertainty", "description": "Evaluate uncertainty for a quantitative result and attach a standard or expanded uncertainty with coverage factor and coverage probability, or record an explicit reasoned omission.", "inputs": [ "result value", "uncertainty components and their evaluation basis", "coverage probability required by the decision rule" ], "outputs": [ "uncertainty statement", "uncertainty budget" ], "preconditions": [ "the result is quantitative", "components of uncertainty have been identified" ], "effects": [ "the result becomes usable in a conformity decision rule", "omission of uncertainty becomes explicit rather than implied" ], "source_refs": [ "SRC-004", "SRC-005" ] }, { "id": "assign-confidence", "name": "Assign a scale-declared confidence with reasons", "description": "Attach a confidence value bound to a named, versioned scale, decomposed into recorded rating-down and rating-up reasons, evidence characterisation and agreement, with calibration status stated.", "inputs": [ "assertion reference", "confidence scale reference", "evidence basis summary", "agreement measure" ], "outputs": [ "confidence statement with reason records", "calibration status" ], "preconditions": [ "the confidence scale is registered", "the confidence target is explicit" ], "effects": [ "confidence becomes reconstructible and challengeable rather than asserted", "uncalibrated values are marked and cannot be read as probabilities" ], "source_refs": [ "SRC-015", "SRC-011", "SRC-004" ] }, { "id": "evaluate-conformance", "name": "Evaluate conformance against a requirement", "description": "Apply a documented decision rule that accounts for measurement uncertainty to produce a verdict against a named requirement, with severity and any waiver reference.", "inputs": [ "result with uncertainty", "requirement and acceptance limit", "decision rule" ], "outputs": [ "conformance verdict", "conformance report" ], "preconditions": [ "a requirement exists and is in force at the assessment time", "the decision rule is documented" ], "effects": [ "a derived verdict assertion is created citing its inputs", "the verdict is recomputable from recorded inputs" ], "source_refs": [ "SRC-004", "SRC-003", "SRC-007", "SRC-013" ] }, { "id": "issue-assertion", "name": "Issue an assertion immutably", "description": "Transition an assertion from draft to issued, fix its content digest, set validity, apply access classification and retention class, and optionally wrap it in a secured credential.", "inputs": [ "draft assertion", "validity policy", "access classification", "retention class" ], "outputs": [ "issued immutable assertion", "content digest", "optional signed credential" ], "preconditions": [ "mandatory fields for the assertion class are complete", "an authorised issuing role approves" ], "effects": [ "the assertion becomes citable and non-editable", "subsequent changes require supersession or retraction" ], "source_refs": [ "SRC-018", "SRC-012", "SRC-014" ] }, { "id": "supersede-or-retract", "name": "Supersede or retract an assertion", "description": "Create a superseding assertion or a retraction with a recorded reason, preserving the append-only chain and notifying dependants.", "inputs": [ "target assertion reference", "change kind", "reason", "new assertion where applicable" ], "outputs": [ "updated lifecycle state", "supersession or retraction record", "dependant notification list" ], "preconditions": [ "the target is in issued or disputed state", "the actor holds the transition role" ], "effects": [ "history is preserved rather than overwritten", "derived assertions are flagged for re-evaluation" ], "source_refs": [ "SRC-012", "SRC-006", "SRC-014" ] }, { "id": "reconcile-competing-assertions", "name": "Reconcile competing assertions", "description": "Detect assertions covering the same subject, scope and dimension that materially disagree, apply the governed precedence rule and record the disagreement instead of merging silently.", "inputs": [ "candidate assertion set", "comparability keys", "precedence rule" ], "outputs": [ "disagreement record", "prevailing assertion designation", "retained non-prevailing assertions" ], "preconditions": [ "comparability keys are computable for all candidates", "a precedence rule is in force" ], "effects": [ "disagreement becomes visible to consumers", "non-prevailing assessments remain retrievable and citable" ], "source_refs": [ "SRC-001", "SRC-016", "SRC-014" ] }, { "id": "decide-fitness-for-purpose", "name": "Decide fitness for a declared purpose", "description": "Combine cited assertions, verdicts, confidence and limitations into a purpose-relative fitness judgement with conditions and non-supported uses, recorded as its own assertion.", "inputs": [ "declared purpose", "cited assertions and verdicts", "known defects and limitations" ], "outputs": [ "fitness verdict", "user-facing quality statement" ], "preconditions": [ "the purpose is explicit", "required profile dimensions have been assessed or their absence is declared" ], "effects": [ "a citable, purpose-bound decision record is produced", "reliance outside the declared purpose is explicitly unsupported" ], "source_refs": [ "SRC-016", "SRC-003", "SRC-013" ] }, { "id": "expire-and-schedule-reassessment", "name": "Expire assertions and schedule re-assessment", "description": "Detect assertions past their validity window or invalidated by subject, method or threshold change, mark them stale and enqueue re-assessment according to the profile cadence.", "inputs": [ "validity windows", "change events on subject, measure or threshold", "cadence specification" ], "outputs": [ "staleness flags", "re-assessment work items", "consumer notifications" ], "preconditions": [ "validity or cadence is declared for the measure", "change events are observable" ], "effects": [ "stale scores stop being served as current", "silent reuse of expired assertions is prevented" ], "source_refs": [ "SRC-018", "SRC-016", "SRC-013" ] }, { "id": "export-aligned-quality-metadata", "name": "Export quality metadata in an alignment profile", "description": "Project assertions into an external target vocabulary using the maintained crosswalk, emitting only mappings of declared strength and attaching non-comparability notes.", "inputs": [ "assertion set", "target vocabulary and version", "alignment crosswalk" ], "outputs": [ "exported quality metadata", "mapping loss report" ], "preconditions": [ "crosswalk rows exist for the exported slots", "target vocabulary version is pinned" ], "effects": [ "external consumers receive interoperable statements", "unmappable content is reported rather than silently dropped" ], "source_refs": [ "SRC-001", "SRC-019", "SRC-002", "SRC-003" ] }, { "id": "apply-retention-disposition", "name": "Apply retention and disposition", "description": "Execute the retention schedule over assertions and evidence artifacts, honouring legal holds, producing tombstones where content is destroyed but the record must survive, and documenting retention/erasure conflicts.", "inputs": [ "retention schedule", "legal hold register", "erasure requests" ], "outputs": [ "disposition actions", "deletion records", "conflict decisions" ], "preconditions": [ "retention classes are assigned", "hold status is resolvable" ], "effects": [ "evidence is destroyed on schedule without falsifying the audit trail", "loss of reproducibility caused by disposal is disclosed on affected assertions" ], "source_refs": [ "SRC-013", "SRC-014", "SRC-012" ] }, { "id": "attach-quality-assertion", "name": "Attach quality assertion", "description": "Bind a new quality measurement, annotation, certificate reference or policy binding to a host using the host's authoritative identifier and a new assertion identifier.", "inputs": [ "host identifiers and types", "assertion class", "optional quality-metadata container", "attributed agent" ], "outputs": [ "quality assertion record with assertion_id and host binding" ], "preconditions": [ "Host identifier is resolvable in the host identity model.", "Caller is permitted to create quality records for that host." ], "effects": [ "A new quality entity is generated and attributed.", "The host inverse quality list is updated." ], "source_refs": [ "SRC-001", "SRC-006" ] }, { "id": "ingest-quality-feedback", "name": "Ingest quality feedback", "description": "Accept a user or aggregator annotation, including conflicting assessments.", "inputs": [ "author agent", "motivation", "body", "target host", "optional dimension" ], "outputs": [ "quality annotation record" ], "preconditions": [ "Motivation includes quality assessment.", "Privacy classification of the comment is set." ], "effects": [ "Disagreement with existing assessments is linked, not overwritten." ], "source_refs": [ "SRC-001" ] }, { "id": "derive-composite-score", "name": "Derive composite score", "description": "Compute a derived measurement from other measurements with disclosed operators and weights.", "inputs": [ "source assertion ids", "combination operator", "optional weight map", "decision agent" ], "outputs": [ "derived measurement", "derivation record" ], "preconditions": [ "All sources exist and share a compatible quality unit or documented roll-up.", "Weights are stored when a weighted operator is used." ], "effects": [ "prov:wasDerivedFrom links are written.", "Trade-off notes may be attached for management review." ], "source_refs": [ "SRC-001", "SRC-025", "SRC-006" ] }, { "id": "maintain-alignment-crosswalk", "name": "Crosswalk dimension", "description": "Record or update a mapping from a local dimension or field to external catalogues without implying conformance.", "inputs": [ "local dimension_id", "target catalogue codes", "conflict flags" ], "outputs": [ "crosswalk row", "updated catalogue version" ], "preconditions": [ "Target codes are cited from a named standard version." ], "effects": [ "Interoperability table version increments.", "Unmapped locals remain visible as gaps." ], "source_refs": [ "SRC-001", "SRC-021", "SRC-022", "SRC-026" ] } ], "composition": [ { "target": "Any host world-model entry that requires quality or confidence qualification", "relation": "MIX-IN", "purpose": "Attach quality assertions to any subject through a quality-metadata container and a subject reference, without modifying or duplicating the host's own semantics. DQV's pattern of linking quality statements to an assessed resource is the template.", "required": true, "source_refs": [ "SRC-001", "SRC-006" ] }, { "target": "Provenance & Attribution mixin (sibling; identifier to be assigned by the registry)", "relation": "REFERENCE", "purpose": "Resolve who generated a quality assertion, from what, and when. DQV itself delegates this to PROV rather than restating it, so this model references generation, attribution and derivation instead of redefining them.", "required": true, "source_refs": [ "SRC-001", "SRC-006" ] }, { "target": "Measurement, Units & Quantity Kinds model (sibling)", "relation": "REFERENCE", "purpose": "Resolve dereferenceable unit identifiers and quantity kinds for quantitative results; DQV recommends a dereferenceable unit reference rather than a unit string, and VIM supplies the metrological vocabulary.", "required": true, "source_refs": [ "SRC-001", "SRC-005" ] }, { "target": "Validation & Constraint results model (sibling)", "relation": "COMPOSE", "purpose": "Consume binary conformance results with focus node, source shape, constraint component and severity as evidence inputs to graded quality assertions, keeping validation authoring out of this model.", "required": false, "source_refs": [ "SRC-007", "SRC-003" ] }, { "target": "Agent / Party & Role model (sibling)", "relation": "REFERENCE", "purpose": "Resolve assessor identity, role, accreditation and organisational affiliation used for attribution, competence and independence classification.", "required": true, "source_refs": [ "SRC-006", "SRC-016" ] }, { "target": "Evidence & Attestation model (sibling)", "relation": "COMPOSE", "purpose": "Carry issued quality certificates and attestations with issuer, securing mechanism, status and evidence, providing tamper-evidence and revocation that this model does not define.", "required": false, "source_refs": [ "SRC-018", "SRC-001" ] }, { "target": "Quality measure register (nested registry within this model's governance scope)", "relation": "CHILD", "purpose": "Hold registered measure items with identifier, version, status and supersession, maintained under registration procedures; nested rather than sibling because measure semantics are inseparable from result semantics.", "required": true, "source_refs": [ "SRC-012", "SRC-020", "SRC-003" ] }, { "target": "Sampling & Statistical Population model (sibling)", "relation": "REFERENCE", "purpose": "Resolve frames, sampling designs, achieved samples and representativeness claims that determine whether a measured value generalises to the declared scope.", "required": false, "source_refs": [ "SRC-003", "SRC-011" ] }, { "target": "Risk & Harm model (sibling)", "relation": "ALIGN", "purpose": "Align confidence and quality inputs to risk assessment without merging them: measurement and evidence assessment is a distinct governed activity from risk treatment, and confidence grades must not be arithmetically combined with risk scores.", "required": false, "source_refs": [ "SRC-017", "SRC-015" ] }, { "target": "Access & Disclosure Policy model (sibling)", "relation": "REFERENCE", "purpose": "Resolve classification schemes, redaction profiles and authorisation decisions applied to quality statements and evidence packages, which this model classifies but does not define.", "required": true, "source_refs": [ "SRC-014", "SRC-013" ] }, { "target": "Records Retention & Disposition model (sibling)", "relation": "REFERENCE", "purpose": "Resolve retention classes, schedules, legal holds and disposition actions covering assertions and evidence under conflicting retention and erasure duties.", "required": true, "source_refs": [ "SRC-013", "SRC-014" ] }, { "target": "ISO/IEC 25012 and ISO/IEC 5259-2 data quality dimension vocabularies", "relation": "ALIGN", "purpose": "Bind dimension terms from published vocabularies rather than embedding a proprietary list; alignment strength is recorded per term and conformance is not claimed without test evidence.", "required": false, "source_refs": [ "SRC-002", "SRC-010", "SRC-001" ] }, { "target": "ISO 19157-1 data quality measure component structure", "relation": "ALIGN", "purpose": "Align the measure definition slots (identifier, name, definition, parameters, value type, value structure, source reference) and the scope/evaluation/reporting separation to a published component structure.", "required": false, "source_refs": [ "SRC-003", "SRC-020" ] }, { "target": "Sensing System Capability model (sibling, SSN/SOSA-aligned)", "relation": "ALIGN", "purpose": "Reference declared instrument capabilities qualified by operating conditions as method context, while keeping capability declarations distinct from assessments of produced results.", "required": false, "source_refs": [ "SRC-008" ] }, { "target": "Dataset / Catalogue record model (sibling)", "relation": "EXTEND", "purpose": "Extend catalogue records with attached quality metadata containers so that quality carries its own provenance, validity and access rules instead of becoming inert descriptive fields.", "required": false, "source_refs": [ "SRC-001", "SRC-019" ] } ], "serviceLayers": { "dimension": { "owner_package_requirements": [ "Designate a named owner for the model and a separate register manager for the quality measure register and the confidence scale definitions, with documented delegation and a successor rule.", "Publish a quality profile per subject class in scope, declaring mandatory dimensions, bound measures and versions, thresholds, decision rules, validity windows and re-assessment cadence, with effective-date ranges.", "Maintain the alignment crosswalk against pinned external standard versions, review it when an aligned standard is revised, and record conflicts rather than silently re-mapping.", "Assign access classification and retention class to every assertion class and evidence artifact type before the first assertion is issued.", "Operate a dispute and rectification channel with published outcomes and deadlines, including the legally required path where assertions concern personal data." ], "namespace_guidance": "Allocate one namespace for assertion identifiers, a separate namespace for measure register items, and a third for confidence scale definitions, so that a measure or a scale can be cited independently of any assertion that uses it. Namespaces must be resolvable and version-bearing; external terms (dimensions, units, PROV and DQV terms) are referenced in their owning namespaces and never re-minted locally. Local identifiers follow the identity priority rule; a date, a filename or a storage path is never an identifier.", "registry_links": [ "Registry entry vr.wm-xct-026 (WM-XCT-026, Quality / Confidence), nav path NAV.XCT.QLT, domain tag XCT.QLT, status candidate, review state boundary-review-required.", "Quality measure register (nested child registry) holding measure items with identifier, version, status and supersession pointers.", "Confidence scale register holding named, versioned scale definitions with ordered terms and non-comparability notes.", "Alignment crosswalk register mapping model slots to pinned external standard versions with mapping strength." ] }, "canon_and_patch": { "canonicalization_rules": [ "Canonical content is the semantic assertion graph — subject reference, scope, measure and version, parameter binding, result, uncertainty, confidence with scale, verdict, times, attribution, validity, classification — independent of serialisation; JSON, YAML, Markdown, RDF, Git objects, MCP messages and MongoDB documents are projections.", "Canonicalise before digesting: order fields by identifier, normalise identifiers to their resolved canonical form, normalise all timestamps to RFC 3339 with seconds and an explicit offset or Z, and represent numeric results with their declared precision rather than platform float formatting.", "Never canonicalise away a distinction the model requires: subject state time, result time and record time remain three fields even when equal; unassessed, not-applicable and null remain distinct from a zero score.", "Confidence values are canonicalised only together with their scale reference; a confidence value serialised without its scale is invalid, not merely incomplete." ], "patch_rules": [ "Issued assertions are immutable: changes are expressed as a superseding assertion or a retraction with a recorded reason, never as an in-place edit.", "Drafts may be patched freely; the transition to issued fixes the content digest and closes the record to modification.", "Patches to profiles, thresholds, measures and confidence scales carry an effective-date range and do not retroactively alter assertions already issued under earlier versions; where re-evaluation is required it is executed as new assertions.", "Every patch to a governed object records actor, RFC 3339 timestamp, prior and new version, and reason; unattributed patches are rejected.", "Deletion of evidence under a retention schedule replaces content with a tombstone that preserves the digest and the deletion record, and flags affected assertions as no longer reproducible." ], "compatibility_rules": [ "Adding an optional slot, a new dimension binding, a new registered measure or a new confidence scale is backward compatible.", "Changing the meaning, scale type, unit or parameter defaults of an existing measure is breaking and requires a new measure version with a supersession pointer.", "Changing a confidence scale's terms or their order is breaking; historical assertions keep resolving to the superseded scale version, which must remain published.", "Removing a slot, narrowing a value domain, or tightening cardinality from optional to required is breaking and requires a major version plus a migration note stating what becomes invalid.", "Conformance to an external standard is claimed only per pinned version with test evidence; a standard revision never automatically extends an existing conformance claim." ] }, "artifact_rules": { "identity_priority": [ "Authoritative master-system identifier assigned by the system of record for the subject, the measure, the certificate or the register item.", "Governed global identifier or IRI issued by a recognised registry or standards body, including external measure, dimension, unit and scale identifiers.", "UUID or ULID minted by the adopting Dimension, used only where neither of the above exists, and recorded together with the minting namespace.", "A date, a version label, a filename, a storage path or a content hash alone is never an identifier; a content digest is an integrity control that may accompany, but never replace, an identifier." ], "timestamp_rule": "All time values are recorded in RFC 3339 format with seconds and an explicit UTC offset or Z. Subject state time, assessment result time and record/ingestion time are stored as separate fields and are never collapsed, even when equal; validity, embargo, retention and lifecycle transition times follow the same rule.", "serial_naming_rule": "Serial artifacts (assertion records, assessment runs, evidence packages, conformance reports, change-log entries, feedback records) are keyed by their owning object identifier plus a monotonic sequence number and the RFC 3339 issuance timestamp. Sequence numbers are scoped to the owning object and are never reused after retraction; human-readable labels are display metadata only and carry no identity.", "integrity_rule": "Every artifact referenced by an issued assertion carries a content digest recorded on the assertion at issuance, with the digest algorithm named. Retrieval verifies the digest before the artifact may be used as evidence; a verification failure marks the assertion as unsubstantiated rather than silently degrading it. Where an artifact is destroyed under retention rules, the digest and a deletion record are retained so that past reliance remains auditable." }, "policies": [ "No bare scores: a result may not be published without its measure identifier and version, its scope, its scale, and either an uncertainty statement or an explicit reasoned omission.", "No unscaled confidence: a confidence value may not be published without a reference to a named, versioned confidence scale and a calibration status; uncalibrated values must not be presented or consumed as probabilities.", "Separation of constructs: measurement uncertainty, confidence in the assertion, conformance verdict and risk are four distinct fields; none may be derived from another by convention, and none may be arithmetically combined across scales.", "No silent merging: conflicting assertions about the same subject, scope and dimension are retained and surfaced as a disagreement, with a governed precedence rule; last-writer-wins is prohibited.", "Self-assessments are labelled: independence classification is mandatory, and a self-assessment may not be presented as equivalent to an independent third-party assessment.", "Limitations travel: declared non-supported uses and known defects propagate to derived products and to user-facing quality statements; silence is never a substitute for a caveat.", "No conformance by citation: alignment to an external standard is recorded as a mapping with strength; conformance is claimed only with pinned version and test evidence.", "Quality measurements are observations: append and invalidate, do not mutate historical values.", "Do not claim standard conformance without a conformance statement that cites measurements or an explicit not-evaluated status.", "Predictive or rater confidence SHALL NOT be labelled as GUM expanded uncertainty or statistical confidence interval without a documented calibration and distributional justification.", "Independent review is required before a Dimension treats a quality result as authoritative for high-impact decisions, consistent with NIST Measure guidance.", "Access to quality records defaults to need-to-know; publication of scores that reveal security or personal data requires a documented exception." ], "crud": { "read": [ "Reads resolve the assertion together with its measure version, scope, scale references and validity, so a value can never be read detached from its interpretation.", "Reads past the validity window return the assertion marked stale, with the staleness rule applied, rather than as a current value.", "Consumers without full access receive the declared redacted view; the existence of a restricted assertion is disclosed unless a specific legal basis forbids it.", "Every read of a restricted assertion or evidence package is logged with requester, purpose and RFC 3339 timestamp." ], "create": [ "Assertions are created in draft state with subject state time, result time and record time populated and the assessor attributed.", "Creation validates completeness against the bound quality profile for the subject class and rejects records missing mandatory slots for their assertion class.", "Creating an assertion that duplicates an existing natural key is resolved by the declared deduplication rule, not by overwriting." ], "update": [ "Only draft assertions are updatable in place; issued assertions are updated exclusively through supersession or retraction.", "Updates to profiles, measures, thresholds and confidence scales are versioned with effective-date ranges and never rewrite history.", "Any update records actor, timestamp, prior and new version, and reason; the change log is append-only." ], "delete": [ "Assertions are not deleted as a routine operation; they are retracted, with reason, and retained for the audit record.", "Physical deletion occurs only under the retention schedule or an enforceable erasure duty, and produces a tombstone retaining identifier, digest, deletion time, actor and authority.", "Deletion of evidence flags every dependent assertion as no longer reproducible and records the resulting loss of substantiation.", "Legal holds suspend all scheduled deletion; releasing a hold requires the authorised role and is itself recorded." ] }, "roles": [ { "name": "Model owner", "responsibilities": [ "Own the scope, boundary notes and out-of-scope list, and resolve boundary disputes with sibling models.", "Approve major versions and breaking changes, and publish migration notes.", "Maintain the registry entry, review state and the record of unresolved boundaries." ] }, { "name": "Register manager", "responsibilities": [ "Operate the quality measure register and the confidence scale register: accept submissions, assign identifiers, set status and supersession.", "Ensure every registered item is dereferenceable, versioned and accompanied by a definition, parameters, value type and value structure.", "Publish register change records and retain superseded items so historical assertions remain interpretable." ] }, { "name": "Assessor", "responsibilities": [ "Execute assessments according to the planned scope, sampling design and documented procedure, and record the execution environment.", "Record results with scale, unit reference and status codes, and evaluate uncertainty or record an explicit reasoned omission.", "Assign confidence against a named scale with decomposed rating-down and rating-up reasons, and declare independence and any conflict of interest." ] }, { "name": "Quality steward", "responsibilities": [ "Maintain quality profiles, thresholds, decision rules, validity windows and re-assessment cadence for subject classes in scope.", "Monitor staleness, trigger re-assessment on subject, method or threshold change, and manage series-break markers.", "Reconcile competing assertions under the governed precedence rule and keep disagreement visible." ] }, { "name": "Evidence and records custodian", "responsibilities": [ "Retain evidence packages with digests, enforce the retention and disposition schedule, and apply and release legal holds.", "Execute erasure requests without falsifying the audit trail, producing tombstones and deletion records.", "Document and escalate conflicts between retention duties and erasure duties." ] }, { "name": "Access authority", "responsibilities": [ "Assign and review access classifications for assertions and evidence independently of the subject's classification.", "Define and maintain redaction profiles and mandatory-disclosure lists for safety- and rights-affecting limitations.", "Authorise embargo release and audit access to restricted quality statements." ] }, { "name": "Dispute and rectification handler", "responsibilities": [ "Receive and triage challenges to assertions, set and enforce response deadlines, and record outcomes.", "Execute rectification, amendment, annotation or retraction and notify recipients of corrected assertions.", "Maintain the disputed-state visibility of assertions while a challenge is open." ] } ], "access": { "default_rule": "Deny by default. Access to a quality assertion, its evidence and its confidence basis is granted only to identified principals holding an authorised role for the declared purpose, evaluated at request time against the sibling access-policy model. The assertion's classification is evaluated independently of the subject's classification, because a quality statement can be more sensitive than what it describes.", "scopes": [ "bundle", "layer", "finding", "artifact" ], "exceptions": [ "Mandatory disclosure: limitations, known defects and non-conformances that affect safety, legal rights or regulatory duties are disclosed to affected parties regardless of the default classification, through the declared disclosure channel.", "Data subject access: individuals may obtain quality assertions concerning their personal data and may exercise rectification and erasure rights, with the evidence of the action retained.", "Regulator and auditor access: supervisory bodies and appointed auditors may access assertions, evidence packages and change logs within their mandate, including embargoed material.", "Redacted public view: consumers without full access receive a declared reduced view stating that a limitation exists and its severity, without the underlying evidence, provided this does not permit re-identification or disclosure of confidential material.", "Embargo: assertions under embargo are withheld until the declared release time, and may be released early only by the authorised releaser with a recorded decision.", "Public certificate summaries may be released without evidence packs.", "Personal comments in user feedback may be redacted or restricted even when the score is public.", "Security-relevant availability or vulnerability-related quality findings follow the security sibling model's disclosure timeline.", "Legal hold blocks deletion but does not by itself expand read access." ], "audit_requirements": [ "Log every read of a restricted assertion or evidence package with principal, purpose, scope and RFC 3339 timestamp.", "Log every lifecycle transition, supersession, retraction, threshold change, profile change and precedence decision with actor and reason in an append-only trail.", "Log every disposition action, deletion, legal hold imposition and release, and every erasure executed under a data subject right.", "Retain access and change logs at least as long as the assertions they describe, subject to their own retention class, and make them available to auditors and regulators.", "Periodically verify artifact digests and record verification failures as substantiation defects on affected assertions.", "Record create, invalidate, revoke, access-grant change and evidence-pack download events with RFC 3339 time, actor and assertion_id.", "Retain audit logs at least as long as the quality evidence they refer to." ] }, "agents_bootstrap": { "filename": "AGENTS.md", "required_fields": [ "Name", "Type", "Specification URL", "Storage type URL", "Interface URL", "Processes URL", "Registry ID", "Model ID", "Owner and register manager", "Version and effective date", "Bound dimension vocabularies and versions", "Confidence scale registry URL", "Alignment crosswalk URL", "Access and retention policy URL" ], "read_order": [ "Read AGENTS.md first to obtain Name, Type, Specification URL, Storage type URL, Interface URL and Processes URL; do not infer any of these from directory layout, filenames or storage engine.", "Read the Specification URL to load scope, boundaries, out-of-scope list and the bundle/layer/finding structure before interpreting any assertion.", "Read the bound quality profile and the measure register to resolve which dimensions and measure versions apply to the subject class at hand.", "Read the confidence scale definitions before interpreting any confidence value; refuse to compare or aggregate values across scales.", "Read the Storage type URL and Interface URL only to learn how to retrieve records; treat both as projections that never alter semantics.", "Read the Processes URL to learn the assessment, issuance, supersession, dispute, access and retention procedures before writing or superseding anything.", "Read the alignment crosswalk and recorded conflicts before exporting to or importing from an external vocabulary." ] } }, "coverage": { "claim": "Covers the quality/confidence assertion as a first-class attachable record: identity and subject scope, bound dimension vocabularies and registered measures, evaluation method with sampling and reference basis, result value with scale and uncertainty, confidence as a scale-declared reason-decomposed statement about the assertion, requirement/verdict/fitness, time, lifecycle, attribution, comparability, disagreement, access and retention. Grounded in W3C DQV for assertion shape, ISO 19157-1 and ISO/TC 211 schema practice for measure components, JCGM GUM/VIM for uncertainty, PROV-O for lineage, SHACL for the conformance boundary, GRADE for certainty decomposition, and EU AI Act/GDPR/Eurostat for declared metrics, rectification and retention. Does not claim a universal quality ontology, completeness of any dimension taxonomy, or conformance to any cited standard. Measure-register governance, confidence-scale semantics, non-numeric/qualitative quality and all domain profiles beyond structural slots remain partially supported.", "confidence": "medium", "checklist": [ { "dimension": "identity", "status": "covered", "notes": "The assertion is identified independently of subject and measure (quality-assertion-identity), with a stated identity priority, a natural-key rule for sameness, and separate namespaces for assertions, measures and confidence scales. Grounded in DQV's treatment of quality statements as first-class resources and in registration practice." }, { "dimension": "lifecycle", "status": "covered", "notes": "Declared state set, legal transitions, append-only supersession chain, correction versus new-assessment distinction, retraction effects on derived assertions, and register item status for measures and scales. Supported by register status practice, credential status/revocation and PROV derivation." }, { "dimension": "relationships", "status": "covered", "notes": "Subject attachment, measure and dimension references, derived-from lineage, aggregation into composites with recoverable components, propagation to derived products, conflicting-assertion links, and fifteen composition links to sibling models each with source support." }, { "dimension": "temporal", "status": "covered", "notes": "Three distinct times (subject state, result, record) modelled on the SOSA phenomenonTime/resultTime separation, RFC 3339 with seconds and explicit offset or Z enforced, plus validity windows, decay, re-assessment cadence, embargo and retention timestamps." }, { "dimension": "provenance", "status": "covered", "notes": "PROV-based generation, attribution and derivation are referenced rather than redefined, with a quality-metadata container for joint provenance of grouped statements, an execution environment record and a re-execution path." }, { "dimension": "ownership", "status": "covered", "notes": "Assessor attribution with role, commissioning organisation, authority basis, accreditation reference and a mandatory independence classification distinguishing self, second-party and independent third-party assessments; model owner and register manager are named roles." }, { "dimension": "validation", "status": "covered", "notes": "Requirement, acceptance limit, uncertainty-aware decision rule, verdict and severity are modelled separately from measurement; the SHACL boundary establishes that binary conformance is an input to, not a synonym for, graded quality; value-domain and null-handling rules are explicit." }, { "dimension": "access", "status": "covered", "notes": "Deny-by-default with classification assigned independently of the subject, four access scopes, redacted views, mandatory safety/rights disclosure, data-subject and regulator exceptions, embargo control and read logging." }, { "dimension": "retention and deletion", "status": "covered", "notes": "Retention classes and schedule, tombstone/digest-only retention when evidence is destroyed, disclosure of the resulting loss of reproducibility, legal holds, erasure execution with an intact audit trail, and an explicit procedure for documenting retention-versus-erasure conflicts." }, { "dimension": "interoperability", "status": "covered", "notes": "Alignment crosswalk with per-row mapping strength and pinned standard versions, a no-conformance-by-citation policy, an export function that reports mapping loss, and explicit non-comparability rules including a prohibition on cross-scale confidence comparison." }, { "dimension": "uncertainty and calibration", "status": "covered", "notes": "GUM/VIM-grounded uncertainty statement with coverage factor, interval and probability, an uncertainty budget artifact, mandatory reasoned omission when uncertainty is not evaluated, and a calibration-status field that prevents uncalibrated values being read as probabilities." }, { "dimension": "sampling and representativeness", "status": "covered", "notes": "Inspection mode, sampling design, frame, achieved sample and representativeness claim are required before execution, aligned with ISO 19157-1 evaluation procedures, TS 4213 control criteria and the AI Act's representativeness obligation." }, { "dimension": "comparability", "status": "covered", "notes": "A comparability key over measure version, parameters, scope, scale and reference basis, forbidden-combination rules, series-break markers and reversible normalisation, grounded in coherence and comparability as a governed output-quality principle." }, { "dimension": "evidence and reproducibility", "status": "covered", "notes": "Retrievable evidence with digests and securing mechanisms, re-execution inputs including snapshots, seeds and tolerances, an explicit unsubstantiated-judgement flag, and access handling for restricted evidence." }, { "dimension": "disagreement and dispute", "status": "covered", "notes": "Competing assessments are retained with a governed precedence rule, silent merging is prohibited, disputed state is visible to consumers, and the legally enforceable rectification/erasure path is modelled for assertions about personal data." }, { "dimension": "measurement semantics of non-numeric quality", "status": "gap", "notes": "Qualitative and narrative quality judgements (expert commentary, editorial assessment) are accommodated structurally as annotations with ordinal confidence, but no primary source was found that governs their internal structure, so this remains under-specified rather than canonical." } ], "known_omissions": [ "The specific fifteen ISO/IEC 25012 characteristic names and their inherent/system-dependent assignment were not verified from normative text; only the two-viewpoint structure and the count were confirmed from the ISO catalogue record. The model therefore binds the vocabulary by reference and does not enumerate it.", "The internal terminology of ISO 8000-8 (syntactic, semantic and pragmatic quality; verification versus validation) is widely reported in secondary sources but was not confirmed from the normative text; only the standard's scope was verified. The fitness-for-purpose finding is grounded instead on ISO 19157-1 and the European Statistics Code of Practice.", "A calibrated confidence-plus-likelihood exemplar from climate assessment practice was sought as a third independent grounding for ordinal confidence, but the authoritative documents were not retrievable at access time and were therefore excluded rather than cited from memory or from mirrors.", "The full text of JCGM 100 and JCGM 106 was not read; the publication index and the annotated VIM were. Specific clause-level requirements on decision rules and guard bands are represented at the level of the concepts named in those publications' titles and in VIM definitions.", "No domain-specific bindings are supplied for images, audio, code, physical goods or robotics sensing beyond the SSN capability alignment; the registry records a robotics factor of zero, and no robotics-specific quality node is asserted.", "Aggregation and composite-index methodology is modelled structurally (function, weights, validity preconditions) but no primary source is cited for any particular aggregation function; specific composite indices must be justified locally.", "Human inter-rater reliability statistics are referenced as an agreement measure without binding a specific coefficient, since no single normative choice was identified.", "Full clause-level text of ISO/IEC 25012, ISO 19157-1, ISO 8000 parts beyond Part 1 overview, ISO/IEC 25024 measure tables and ISO/IEC 5259-2 additional ML characteristic names was not available as free complete primary HTML; those names must be copied from licensed text before claiming a closed catalogue.", "JCGM GUM-1:2023 Bayesian/coverage-interval restatement is only partially cited via secondary catalogue notices; this model remains aligned to JCGM 100:2008/2010 as fetched.", "UNECE NQAF, SDMX quality reporting, DDI quality statements, HL7 FHIR/clinical quality measures and FAIR metrics are not modelled; they are likely specialised siblings.", "ISO 5725 accuracy of measurement methods, JCGM 200 VIM vocabulary and ISO/IEC 17025 calibration certificates are not fully incorporated.", "Wang and Strong 1996 academic dimension set and DAMA-DMBOK uniqueness/validity are used only as conflict notes, not as primary catalogues.", "Qualitative uncertainty for nominal classifications lacks a single primary standard comparable to GUM.", "Region-specific legal accuracy duties (for example GDPR accuracy principle) are not encoded as legislation objects." ], "conflicts": [ "Constraint validation produces binary conformance with organisational severities (violation, warning, informational) that are explicitly not graded outcomes, whereas quality measurement produces graded values. Treating severity as a quality grade, or a score as a verdict, is a category error; the model keeps requirement, result and verdict as separate objects.", "W3C Verifiable Credentials Data Model 2.0 defines evidence, issuer, proof and status but no confidence property, so a credential cannot convey a governed confidence claim on its own. Any confidence carried in a credential must reference a scale defined outside the credential format.", "There is no single authoritative data quality dimension taxonomy: ISO/IEC 25012, ISO/IEC 5259-2, ISO 19157-1 and the European Statistics Code of Practice each define overlapping but non-identical sets, and DQV deliberately refuses to prescribe one. The model requires a bound vocabulary per subject class and a documented precedence rule for overlaps.", "Retention duties conflict directly: AI Act technical documentation and logging obligations require retaining performance evidence, while GDPR storage limitation and erasure rights require deleting personal data. The model does not resolve this generically; it requires a documented, approved, reasoned conflict decision per case.", "ISO/FDIS 19157-3, the strongest available source for governing a register of data quality measures, remains at final-draft stage and is not published. Measure-register governance is therefore adopted as a defensible pattern and marked as a partially supported node rather than presented as canonical.", "Metrological vocabulary and data-quality vocabulary use 'accuracy' and 'precision' with different senses: VIM defines them relative to a true or reference value and to dispersion of repeated measurements, while data quality frameworks often use 'accuracy' as a record-level correctness rate. Cross-domain exchange must map through the crosswalk rather than by term name.", "DQV refuses a universal definition of quality; ISO/IEC 25012 enumerates fifteen characteristics as a data-quality model. Both are alignments.", "ISO 19157:2013 included a usability element; ISO 19157-1:2023 does not keep usability as a core element.", "Uniqueness is common in DAMA-style lists but is not an ISO/IEC 25012 characteristic.", "ISO 8000 treats relevant characteristics as purpose-dependent; some operational scorecards treat accuracy as intrinsic.", "GUM coverage probability or level of confidence is not the same construct as ISO/IEC 25012 credibility, ISO 19157 metaquality confidence, or ML/rater confidence scores.", "Precision in ISO/IEC 25012 and DQV (resolution) is not GUM precision/repeatability and not expanded uncertainty.", "ISO/IEC 25012 confidentiality, accessibility and availability overlap sibling access/privacy/availability models." ], "regional_assumptions": [ "Legal grounding for accuracy duties, rectification, erasure, storage limitation and declared accuracy metrics is drawn from EU instruments (Regulation (EU) 2016/679 and Regulation (EU) 2024/1689). Adopters in other jurisdictions must substitute equivalent instruments; the structural slots (retention class, legal basis, rectification outcome) are jurisdiction-neutral but their contents are not.", "Output-quality principles are drawn from the European Statistical System's Code of Practice; other statistical systems use comparable but differently named principle sets, so principle names are treated as vocabulary bindings rather than universals.", "NIST AI RMF is a voluntary United States framework, not a legal requirement; it is used only to support treating measurement as a separately governed function.", "GRADE originates in health evidence assessment; its certainty domains are used as an evidence-decomposition pattern, not as a claim that clinical certainty grading transfers unmodified to other domains.", "Time handling assumes RFC 3339 with explicit offsets throughout; adopters operating on local civil time or on calendar systems without a fixed offset must record the offset explicitly rather than relying on a local default.", "ISO, W3C, BIPM/JCGM and NIST sources are treated as globally citable technical authorities; they are not a substitute for jurisdictional quality-of-data law.", "GUM k approximately 2 for about 95 percent coverage assumes a near-normal case; other distributions need explicit justification.", "Positional accuracy, topological consistency and gridded accuracy apply when the host is geographic or otherwise spatially referenced.", "NIST AI RMF is a voluntary US public-authority framework; ISO/IEC 5259 is the international AI data-quality measures standard." ], "adversarial_checks": [ "Attempted to justify a canonical, universal list of quality dimensions inside the model. Rejected: DQV explicitly declines to prescribe a single definition of quality, and at least four published vocabularies disagree, so a fixed list would be attractive, comparable-looking and wrong. The model binds vocabularies instead.", "Attempted to fold confidence into measurement uncertainty as a single 'certainty' field. Rejected: coverage probability is a defined metrological construct about a value, while certainty grading in evidence appraisal is a reason-decomposed judgement about an assessment; merging them would let an uncalibrated grade masquerade as a probability. They are kept as separate fields with an explicit no-derivation rule.", "Attempted to treat validation severity levels as a ready-made quality scale. Rejected on the basis that SHACL conformance is binary and its severities are organisational, not graded; validation results are modelled as evidence inputs instead.", "Attempted to model a certificate or verifiable credential as the carrier of confidence. Rejected: the VC 2.0 core data model has no confidence property, so credentials are modelled as integrity and transport wrappers around assertions whose confidence semantics come from this model.", "Searched for a counterexample to the immutability rule and found the legitimate case of erasure duties over personal data inside evidence. Resolved by tombstone-plus-digest retention and by flagging affected assertions as no longer reproducible, rather than by weakening immutability.", "Tested whether a measure register is normatively required. Found the governing standard still at FDIS stage; downgraded the claim to a supported pattern with an explicit gap note rather than presenting registration governance as canonical.", "Tested whether composite quality scores should be first-class. Found no primary support for any aggregation function; retained aggregation only as a lineage operation with recorded function, weights, validity preconditions and mandatory component recoverability, and forbade publishing composites without their components.", "Checked whether this model duplicates a sibling provenance model. Confirmed that DQV itself delegates provenance to PROV, so attribution and lineage are referenced, and only quality-specific slots (assessor competence, independence, method, evidence, calibration) are owned here.", "Rejected treating a model softmax or star rating as GUM expanded uncertainty without calibration evidence.", "Rejected using a report date as the quality assertion identifier.", "Rejected silent in-place overwrite of measurement values.", "Rejected mapping ISO 19157:2013 usability as if it were an ISO 19157-1:2023 core element.", "Rejected claiming ISO/IEC 25012 uniqueness or treating confidentiality scores as access control.", "Rejected collapsing all dimensions into one undocumented composite score as the canonical record.", "Rejected recording untested required checks as pass.", "Rejected mixing host lineage with assessment provenance as a single undifferentiated provenance graph." ] }, "researchAdjudication": { "providerMode": "dual-provider", "activeProviders": [ "claude", "grok" ], "waivedProviders": [], "providerPolicy": {}, "boundaryDecision": { "entry_kind": "mixin", "status": "accepted", "rationale": "Both providers independently converged on mixin, and the base scope statement holds the boundary cleanly: the record attaches to any host without owning the host's semantics, identity minting, unit definitions, provenance graph or access policy, each of which is named as a delegated sibling. A split into separate Quality and Confidence models was considered and rejected because the confidence statement in this model is defined as being about the assertion itself; detaching it would leave a confidence value circulating without the assertion it qualifies, which is precisely the misuse both providers built the model to forbid. Base and source disagree on no boundary line, only on granularity within it." }, "decisions": [ { "concept": "Base provider selection", "disposition": "claude", "rationale": "Grok is larger in layers, findings and artifacts, so size is not the deciding factor. Claude carries the clearest complete boundary: eleven out-of-scope entries each naming the sibling model that owns the excluded concern, seven boundary notes with source refs, a declared coverage gap, and an adversarial-check log recording what was tried and rejected. Grok's boundaries are sound but partly rest on catalogue enumerations neither provider verified from normative text." }, { "concept": "Entry kind and model split", "disposition": "accepted as a single mixin", "rationale": "Both providers independently classified this as a mixin and agree on what is delegated to siblings. Splitting Quality from Confidence was weighed and rejected because confidence here is defined as a statement about the assertion, and separating the two would permit a detached confidence value, the exact failure mode both providers guard against." }, { "concept": "Confidence versus measurement uncertainty versus risk", "disposition": "base separation retained and reinforced", "rationale": "Base forbids arithmetic between confidence grades and risk scores and keeps coverage probability distinct from a reason-decomposed grade. Grok independently reaches the same separation from GUM's own caution that statistical confidence terms apply only under stated conditions. Convergence from two different source sets makes this the model's most defensible core rule." }, { "concept": "ISO/IEC 25012 fifteen-characteristic enumeration", "disposition": "rejected as structure, deferred as vocabulary profile", "rationale": "Grok enumerates the fifteen characteristics with inherent and system-dependent assignment, but the cited evidence is an ISO catalogue record, not normative text, and the base explicitly recorded this as an unverified omission. Embedding one taxonomy would also contradict the base design, which binds a vocabulary per subject class precisely because four published taxonomies disagree and DQV declines to prescribe one." }, { "concept": "ISO 19157-1 quality element enumeration and the dropped usability element", "disposition": "rejected as structure, routed to the alignment crosswalk", "rationale": "Same reasoning as the 25012 enumeration. The specific evidence-backed detail worth keeping is that usability was a core element in ISO 19157:2013 and is not retained in ISO 19157-1:2023; that belongs in the crosswalk conflict register as a version-mapping trap, not as a new finding." }, { "concept": "Host binding and evaluation target as a separate finding", "disposition": "rejected as duplicative", "rationale": "Base subject-attachment-and-assessment-scope already fixes the assessed unit, scope restriction, sample-versus-census and multi-subject traceability. The one genuinely distinct idea, a statement about a prior quality statement, is absorbed by the accepted metaquality node." }, { "concept": "Metric definition, measurement value, uncertainty and coverage interval findings", "disposition": "rejected as duplicative", "rationale": "Base measure-definition-and-registration, result-value-scale-and-structure and uncertainty-and-error-model cover identity, parameters, value type, scale, unit reference, uncertainty evaluation and coverage semantics with equal or stronger sourcing. Importing four parallel nodes would fragment an atomic result across competing records." }, { "concept": "Result polarity and empty-unit semantics", "disposition": "deferred to question-level enrichment of the base result finding", "rationale": "Grok asks whether a larger value means better quality and what the theoretical best and worst values are, and requires an exception code rather than a perfect score for an empty unit. Both are real interpretation hazards the base does not fully cover, but they sit inside an otherwise duplicative finding, so they are recorded for question-level enrichment rather than as new structure." }, { "concept": "Quality policy, threshold and conformance findings", "disposition": "rejected as duplicative", "rationale": "Base requirement-and-acceptance-threshold and conformance-verdict-severity already carry limit, direction, owner, threshold provenance, uncertainty-aware decision rule, severity and partial or waived states. Grok's ODRL policy anchoring is an alignment detail for the crosswalk, and its untested-is-not-pass rule is covered by the base's not-assessed value-domain requirement." }, { "concept": "Quality certificate as annotation versus document", "disposition": "rejected as a finding, routed to the crosswalk conflict register", "rationale": "Grok's observation that dqv:QualityCertificate is the annotation pointing at a certificate rather than the certificate file is a genuine interoperability trap, but the base already holds the certificate as an evidence artifact and handles revocation through credential status and retraction. It belongs among recorded mapping conflicts, not as new structure." }, { "concept": "Supporting evidence, assessment provenance, statement lifecycle and interoperability crosswalk findings", "disposition": "rejected as duplicative", "rationale": "Base evidence-artifacts-integrity-and-reproducibility, assessor-attribution, provenance-lineage, lifecycle-supersession-and-immutability and external-alignment-and-conflicts each cover the same ground with tighter rules, including append-only supersession, correction versus reassessment, and no-conformance-by-citation." }, { "concept": "Access, retention and legal hold", "disposition": "base retained", "rationale": "Base grounds retention conflict, erasure with intact audit trail, tombstone retention and mandatory safety disclosure in the AI Act and GDPR directly. Grok's combined access-retention-audit finding is thinner and adds only audit-event scope, which the base's change log already carries." }, { "concept": "DQV authority tier disagreement", "disposition": "publication hold", "rationale": "Base treats DQV as a tier-1 primary anchor; the source provider assigns tier 2. DQV is a W3C Working Group Note rather than a Recommendation, so the source's assessment is arguably the more accurate one. Since DQV is the base's primary structural anchor, the tier must be reconciled or the reliance justified before publication." }, { "concept": "Spatial extent of the quality unit", "disposition": "deferred to a geospatial domain profile", "rationale": "Grok carries two spatial questions and the base carries none. Spatial extent is real for geographic hosts but is domain-profile content, and the base's generic scope-restriction question covers restricted extent without committing the format-neutral mixin to a coordinate reference system it explicitly excludes." }, { "concept": "Function-set completeness", "disposition": "four operations imported", "rationale": "The base function set is producer-side and closed: it can plan, execute, evaluate, issue, supersede, reconcile and dispose, but cannot attach a non-computed assertion to a host, ingest an externally authored assessment, derive a composite it already models structurally, or maintain the crosswalk its own export function presupposes." } ], "publicationHolds": [ "Source verification is incomplete: normalise the base's ISO citations from the unusual committee.iso.org/es/sites/... paths to canonical iso.org/standard/NNNNN.html forms as used by the source provider, re-resolve every one of the twenty base URLs live, and re-pin access dates before publication.", "Verify the JCGM version pins actually asserted by the base (JCGM 100:2008 with Amendment 1:2026, GUM-1:2023, GUM-5:2026, GUM-6:2020, JCGM 106:2012). The base states it read the publications index and the annotated VIM but not the full text of JCGM 100 or 106, so all clause-level claims about decision rules and guard bands remain unverified and must not be published as normative guidance.", "Recheck the publication status of ISO/FDIS 19157-3, the strongest available anchor for measure-register governance, which the base records as still at final-draft stage. Until it is published, measure-register governance must remain marked as a defensible pattern rather than a normative requirement.", "Reconcile the authority tier assigned to W3C DQV. It is a Working Group Note, not a Recommendation; the base treats it as tier-1 and its primary structural anchor while the source provider assigns tier 2. Either downgrade the tier or state explicitly why a non-normative Note is load-bearing for the model's shape.", "Multi-profile validation is not complete. The model has been reasoned over data, geospatial and ML/analytics hosts only; the base records a robotics factor of zero and supplies no bindings for images, audio, code, physical goods or sensing. Validate against at least three concrete domain profiles before publication and state the untested profiles explicitly.", "Verify the ISO/IEC 25012 fifteen-characteristic set and the ISO 19157-1 element set from licensed normative text before publishing any bound vocabulary profile derived from them; both providers relied on catalogue records for these enumerations." ], "deferredResearch": [ "Verify the ISO/IEC 25012 fifteen characteristics and their inherent versus system-dependent assignment from normative text, then publish as a bound vocabulary profile artifact rather than as model structure.", "Verify the ISO 19157-1:2023 element set from normative text and confirm that usability, a core element in ISO 19157:2013, is not retained in the 2023 edition; record the outcome as a version-mapping conflict in the alignment crosswalk.", "Obtain the official ISO/IEC 5259-2 ML data-quality characteristic names from licensed text; both providers flag them as not extractable from freely published sources, so no closed ML quality catalogue may be claimed until then.", "Verify the ISO 8000 part selection and terminology: the base cites Part 8 (concepts and measuring) with its internal syntactic/semantic/pragmatic terminology unverified, while the source cites Part 1 (overview). Determine which part governs the fitness-for-purpose grounding.", "Resolve result polarity as question-level enrichment of the base result finding: whether a larger numeric value means better quality for a given measure, the theoretical best and worst values, and an exception code for an empty unit so that an empty sample is never reported as perfect completeness.", "Define currentness evaluation for streaming and continuously updated hosts, specifically the sliding window or watermark that bounds the assessed state; neither the base nor the accepted additions cover this.", "Find a primary source governing qualitative and nominal uncertainty comparable to GUM for quantitative results; both providers independently record this as an unresolved gap, and the base marks non-numeric quality semantics as its only declared coverage gap.", "Settle the terminology guardrail between GUM coverage probability, GUM level of confidence and statistical confidence interval, and record the permitted labelling rules in the crosswalk so a coverage figure is never published under a statistical label it does not support.", "Develop the spatial extent of the quality unit within a geospatial domain profile rather than the format-neutral mixin, including the spatial reference in which the extent is stated." ] }, "statistics": { "sources": 26, "bundles": 6, "layers": 14, "findings": 27, "questions": 112, "artifacts": 18, "functions": 18 } }