# Vercy AI instruction - YAML 1.2 (JSON-compatible) { "vercy": "1.0-draft", "publication": { "status": "published", "adjudicationStatus": "reviewable-draft", "publishableCanonical": false, "generatedAt": "2026-08-26T17:08:34Z", "synthesisSha256": "cd918159562d9850df7eccf5b216b85468c561b8434bfe8c68b9dbafda2872bf", "providerMode": "dual-provider", "providers": [ "Claude", "Grok" ], "waivedProviders": [] }, "metaModel": { "id": "WM-AI-003", "registryId": "vr.wm-ai-003", "name": "AI Model Evaluation", "version": "0.3.0-research.1", "previousVersions": [], "entryKind": "aggregate", "family": "World Models", "category": "Information and virtual systems", "industry": [ "Cross-industry" ], "domain": [ "INF.AI.EVL" ], "tags": [ "ai", "model", "evaluation", "inf.ai.evl" ], "status": "published" }, "canonicalUrl": "https://ver.cy/models/wm-ai-003-ai-model-evaluation/", "sourceUrl": "https://github.com/ver-cy/world-models/tree/feat/mega-model-registry/research/runs/wm-ai-003", "model": { "registry_id": "vr.wm-ai-003", "model_id": "WM-AI-003", "name": "AI Model Evaluation", "entry_kind": "aggregate", "purpose": "Provide the format-neutral context an agent needs to plan, execute, record, interpret, approve and disclose an evaluation of a specified AI model or system version, and to retain the resulting evidence.", "scope_statement": "One evaluation instance: a bounded, protocol-governed measurement applied to a version-pinned AI model or system under a declared configuration, producing runs, results, findings, decisions and retained evidence. The model owns the evaluation record graph and its governance; it references but does not own the dataset, the model artefact, the organisational management system, the risk register or the incident process.", "in_scope": [ "Evaluation mandate, trigger and consuming decision or gate", "Binding of the evaluation to a model or system version, configuration and inference settings", "Method-class selection (statistical, formal, empirical, human-subject, adversarial) and protocol design", "Task and dataset version binding, split discipline, ground-truth definition and contamination control", "Metric specification, aggregation, disaggregation factors and pre-registered acceptance thresholds", "Human-rater and red-team protocols including threat models and elicitation effort", "Run execution records, environment capture, determinism and replication", "Result records, per-item evidence, uncertainty and statistical validity", "Findings, severity, validity threats and declared limitations", "Approval, conditional approval, waiver, rejection and release gating", "Re-evaluation triggers, evaluation lifecycle states and cadence", "Evidence packaging, integrity, provenance and lineage", "Audience-tiered reporting, external and third-party evaluation arrangements", "Access control, sensitivity handling, retention, deletion and legal hold", "Declared alignments to external frameworks and exchange interoperability" ], "out_of_scope": [ "Construction, licensing, annotation and internal structure of evaluation datasets (belongs to the benchmark dataset model)", "Model training, fine-tuning, packaging and build provenance of the evaluated artefact (belongs to the model artefact model)", "Organisation-wide AI management system policy, competence programme and internal audit programme", "Risk identification, treatment selection and residual-risk ownership in a risk register", "Serious-incident capture, statutory reporting deadlines and corrective-action workflow", "Notified-body conformity assessment, declarations of conformity and certification records", "Continuous production telemetry and observability outside a bounded evaluation protocol", "Procurement scoring, vendor selection and commercial cost-benefit analysis", "Human-subject research ethics approval procedures (referenced, not defined)", "Compute infrastructure inventory and capacity management" ], "boundary_notes": [ { "neighbor": "WM-AI-009 benchmark dataset", "distinction": "Dataset creation, licensing, annotation workforce and internal record structure are owned there. This model retains only a version-pinned reference, checksum, split designation and contamination check — the minimum needed for reproducibility and comparability.", "source_refs": [ "SRC-008", "SRC-005", "SRC-002" ] }, { "neighbor": "WM-SFT-004 AI or software model artefact", "distinction": "Artefact identity, versioning, packaging and build provenance are owned there. This model records a subject binding (identifier, version, digest, configuration) and never mints artefact identifiers.", "source_refs": [ "SRC-010", "SRC-011", "SRC-007" ] }, { "neighbor": "AI management system (ISO/IEC 42001 AIMS)", "distinction": "Clause 9 duties to determine what is monitored, run internal audits and hold management reviews are organisational and belong to an AIMS model. This model covers a single evaluation instance and the evidence it yields to that system.", "source_refs": [ "SRC-006", "SRC-001" ] }, { "neighbor": "AI risk register and treatment model", "distinction": "Risk identification, treatment selection and residual-risk acceptance live in a risk model. Evaluation supplies measurement evidence and findings that reference risks; it does not own risk objects.", "source_refs": [ "SRC-001", "SRC-002" ] }, { "neighbor": "Serious incident and post-market monitoring model", "distinction": "Incident capture, reporting deadlines and corrective action belong to an incident model. Evaluation consumes incidents as re-evaluation triggers and emits findings, not incident reports.", "source_refs": [ "SRC-002", "SRC-004" ] }, { "neighbor": "Conformity assessment and certification records", "distinction": "Notified-body assessment and declarations of conformity are separate records. Evaluation evidence is an input; this model forbids presenting alignment as certified conformance.", "source_refs": [ "SRC-002", "SRC-006" ] }, { "neighbor": "Production observability and telemetry", "distinction": "Continuous telemetry is a monitoring concern. An evaluation instance is bounded by a registered protocol with thresholds fixed before results are observed, even when executed on production traffic.", "source_refs": [ "SRC-001", "SRC-002" ] } ] }, "sources": [ { "id": "SRC-001", "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0) — AI RMF Core", "organization": "National Institute of Standards and Technology (NIST)", "url": "https://airc.nist.gov/airmf-resources/airmf/5-sec-core/", "version_or_date": "AI RMF 1.0 (NIST AI 100-1), January 2023", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:15:00Z", "relevance": "MEASURE 1.1-4.3 supply the normative-style structure for metric selection, documented TEVV test sets/metrics/tools (2.1), deployment-condition measurement (2.3), generalisability limits (2.5), fairness (2.11), TEVV effectiveness review (2.13), emergent-risk tracking (3.1) and independent assessors (1.3). Explicitly voluntary." }, { "id": "SRC-002", "title": "Regulation (EU) 2024/1689 laying down harmonised rules on artificial intelligence (Artificial Intelligence Act)", "organization": "European Union — Publications Office (EUR-Lex)", "url": "https://eur-lex.europa.eu/legal-content/EN/TXT/HTML/?uri=OJ:L_202401689", "version_or_date": "OJ L, 2024/1689, 12 July 2024", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:18:00Z", "relevance": "Articles 9 (iterative risk management; testing against prior-defined metrics and probabilistic thresholds), 15 (accuracy, robustness, cybersecurity), 19 (log retention), 55 (GPAI model evaluation and documented adversarial testing), 60 (real-world testing), 72 (post-market monitoring) and Annex IV (validation and testing procedures, metrics, dated and signed test logs and reports)." }, { "id": "SRC-003", "title": "EU Artificial Intelligence Act — article and annex reference texts (Articles 9, 19, 55; Annex IV)", "organization": "Future of Life Institute", "url": "https://artificialintelligenceact.eu/annex/4/", "version_or_date": "Accessed 26 August 2026", "source_type": "secondary", "primary_source": false, "authority_tier": 3, "accessed_at": "2026-08-26T09:20:00Z", "relevance": "Used only to locate and cross-check specific AI Act clause wording (Annex IV point 2(g) metrics and signed test logs; Article 19 six-month minimum log retention; Article 9(6)-(8) testing). Superseded by SRC-002 where they differ." }, { "id": "SRC-004", "title": "The General-Purpose AI Code of Practice", "organization": "European Commission — DG CNECT / AI Office", "url": "https://digital-strategy.ec.europa.eu/en/policies/contents-code-gpai", "version_or_date": "Published 10 July 2025", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:22:00Z", "relevance": "Three chapters (Transparency, Copyright, Safety and Security). The Safety and Security chapter supplies the voluntary state-of-the-art route for Article 55: safety framework with evaluation triggers and risk tiers, acceptance determination before proceeding, external evaluations, and a Safety and Security Model Report before release." }, { "id": "SRC-005", "title": "ISO/IEC TS 4213:2022 — Information technology — Artificial intelligence — Assessment of machine learning classification performance", "organization": "ISO/IEC JTC 1/SC 42 (via IEC Webstore)", "url": "https://webstore.iec.ch/en/publication/79567", "version_or_date": "Edition 1.0, 2022-10-13 (confirmed 2025)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:25:00Z", "relevance": "Specifies methodologies for measuring classification performance of ML models, systems and algorithms, including evaluation setup, data representativeness, train/test/validation splits, ground-truth definition, environment and baselines. Full text is paywalled; alignment here is at scope level." }, { "id": "SRC-006", "title": "ISO/IEC 42001:2023 — Information technology — Artificial intelligence — Management system", "organization": "ISO/IEC JTC 1/SC 42", "url": "https://www.iso.org/standard/81230.html", "version_or_date": "First edition, December 2023", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:27:00Z", "relevance": "Clause 9 (performance evaluation: monitoring, measurement, analysis and evaluation; internal audit; management review) and Annex A controls for AI system verification, validation and operation frame the organisational duties that evaluation evidence feeds. Normative text is paywalled; alignment is asserted at clause-title level only." }, { "id": "SRC-007", "title": "PROV-O: The PROV Ontology", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/prov-o/", "version_or_date": "W3C Recommendation, 30 April 2013", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:29:00Z", "relevance": "Namespace http://www.w3.org/ns/prov#. Entity, Activity and Agent plus wasGeneratedBy, wasDerivedFrom, wasAttributedTo, wasAssociatedWith, used, startedAtTime and endedAtTime give format-neutral lineage semantics for datasets, subjects, runs, scorers, results and decisions." }, { "id": "SRC-008", "title": "Croissant Format Specification 1.0", "organization": "MLCommons", "url": "https://docs.mlcommons.org/croissant/docs/croissant-spec.html", "version_or_date": "Version 1.0, 1 March 2024", "source_type": "schema", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:31:00Z", "relevance": "Dataset-level metadata (name, description, license, url, version with MAJOR.MINOR.PATCH, citeAs, datePublished), FileObject/FileSet with contentUrl, encodingFormat and sha256 checksums, and RecordSet/Field with dataType — the reference vocabulary for version-pinning and integrity-checking referenced evaluation data." }, { "id": "SRC-009", "title": "MLPerf Inference Rules", "organization": "MLCommons", "url": "https://github.com/mlcommons/inference_policies/blob/master/inference_rules.adoc", "version_or_date": "master branch, accessed 26 August 2026", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:33:00Z", "relevance": "Closed/Open/Network divisions and Available/Preview/RDI categories define comparability conditions; SingleStream, MultiStream, Server and Offline scenarios define metrics; quality targets (e.g. 99% of FP32), five-significant-figure round-to-even reporting, seeded MT19937 non-determinism limits, mandatory replicability, prohibition of benchmark detection, and a two-day auditor process with a 90-day completion window." }, { "id": "SRC-010", "title": "Annotated Model Card Template", "organization": "Hugging Face", "url": "https://huggingface.co/docs/hub/model-card-annotated", "version_or_date": "Hub documentation, accessed 26 August 2026", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:35:00Z", "relevance": "Defines the Evaluation section as Testing Data, Factors, Metrics and Results with explicit guidance that evaluation is ideally disaggregated by task, domain and population subgroup; plus Uses/Out-of-Scope Use, Bias-Risks-Limitations, Societal Impact Assessment, Environmental Impact fields and role separation between developer, sociotechnic and project organizer." }, { "id": "SRC-011", "title": "Inspect AI — Eval Logs", "organization": "UK AI Security Institute", "url": "https://inspect.aisi.org.uk/eval-logs.html", "version_or_date": "Accessed 26 August 2026", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:37:00Z", "relevance": "Concrete evaluation-run record structure: status, eval header (run_id, task, task_id, task_version, model, dataset, config, revision, packages), plan, results, stats (token usage), error, samples (input, target, messages, output, scores, metadata) and reductions; score edit histories preserve original values with provenance and force recomputation of aggregates." }, { "id": "SRC-012", "title": "Adversarial Machine Learning: A Taxonomy and Terminology of Attacks and Mitigations (NIST AI 100-2 E2025)", "organization": "National Institute of Standards and Technology (NIST)", "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-2e2025.pdf", "version_or_date": "NIST AI 100-2 E2025, March 2025", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:39:00Z", "relevance": "Supplies the threat-model vocabulary for adversarial evaluation: attacker goals, capabilities and knowledge; lifecycle stages of attack; evasion, poisoning and privacy attacks for predictive AI; and generative-AI classes including indirect prompt injection and misaligned outputs, with corresponding mitigations." }, { "id": "SRC-013", "title": "ISO/IEC 25059:2023 — Software engineering — Systems and software Quality Requirements and Evaluation (SQuaRE) — Quality model for AI systems", "organization": "ISO/IEC JTC 1/SC 7 and SC 42", "url": "https://www.iso.org/standard/80655.html", "version_or_date": "First edition, 2023 (revision in progress as ISO/IEC DIS 25059)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:41:00Z", "relevance": "AI-specific extension of the SQuaRE quality model providing consistent terminology for specifying, measuring and evaluating AI system quality characteristics and sub-characteristics (including functional adaptability). Full text is paywalled and the edition is under revision, so it is used as vocabulary alignment only." }, { "id": "SRC-014", "title": "HELM code architecture documentation", "organization": "Stanford Center for Research on Foundation Models (CRFM)", "url": "https://crfm-helm.readthedocs.io/en/latest/code/", "version_or_date": "latest, accessed 26 August 2026", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 3, "accessed_at": "2026-08-26T09:43:00Z", "relevance": "RunSpec, Scenario/ScenarioSpec, Instance, Reference, AdaptationSpec, ScenarioState, RequestState, MetricSpec and Stat demonstrate the separation of task definition, adaptation, execution and scoring, and the artefacts produced (scenario_state.json, stats.json, per-instance stats)." }, { "id": "SRC-015", "title": "ISO/IEC 24029-2:2023 — Artificial intelligence (AI) — Assessment of the robustness of neural networks — Part 2: Methodology for the use of formal methods", "organization": "ISO/IEC JTC 1/SC 42", "url": "https://www.iso.org/standard/79804.html", "version_or_date": "First edition, August 2023", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:45:00Z", "relevance": "Provides methodology for selecting, applying and managing formal methods to prove robustness properties, and situates formal methods alongside statistical and empirical methods as distinct evaluation method classes. Full text is paywalled; used for method-class taxonomy only." }, { "id": "SRC-016", "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1", "organization": "National Institute of Standards and Technology", "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-1.pdf", "version_or_date": "NIST AI 100-1, January 2023", "source_type": "public-authority", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Defines Measure and TEVV as the evaluation function, trustworthy characteristics, separation of builders from validators, scientific-integrity and construct-validation expectations, and go or no-go commissioning decisions." }, { "id": "SRC-017", "title": "The TEVV-Athlon Framework for Evaluating AI Systems, NIST AI 200-2 ipd", "organization": "National Institute of Standards and Technology", "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.200-2.ipd.pdf", "version_or_date": "NIST AI 200-2 ipd, August 2026 (comment period through 2026-10-06)", "source_type": "public-authority", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Four-stage evaluation design (Articulate and Organize, Define and Construct, Apply and Measure, Synthesize and Interrogate); metrology blocks, events and tools; benchmarks, red teaming and field testing; Goodhart caution; TEVV-Athlon as the Measure implementation." }, { "id": "SRC-018", "title": "ISO/IEC TS 4213:2022 Information technology — Artificial intelligence — Assessment of machine learning classification performance", "organization": "ISO/IEC JTC 1/SC 42", "url": "https://www.vde-verlag.de/iec-normen/preview-pdf/info_isoiects4213%7Bed1.0%7Den.pdf", "version_or_date": "First edition 2022-10", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Normative process and control criteria for classification performance assessment, including representativeness, leakage, ground truth, evaluation environment, baselines, confusion-matrix metrics, computational complexity and statistical significance tests. Explicitly does not address benchmarking or use cases." }, { "id": "SRC-019", "title": "Regulation (EU) 2024/1689 of the European Parliament and of the Council (Artificial Intelligence Act)", "organization": "European Union", "url": "https://eur-lex.europa.eu/eli/reg/2024/1689/oj", "version_or_date": "OJ L, 12.7.2024; ELI eli/reg/2024/1689/oj", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Article 15 accuracy, robustness and cybersecurity including declared accuracy metrics, lifecycle consistency, feedback loops and adversarial or poisoning attacks; Articles 10, 43, 55, 72 and 92 for datasets, conformity assessment, GPAI systemic-risk evaluation, post-market monitoring and official evaluations." }, { "id": "SRC-020", "title": "Croissant Format Specification 1.1", "organization": "MLCommons Association", "url": "https://docs.mlcommons.org/croissant/docs/croissant-spec-1.1.html", "version_or_date": "Version 1.1, published 2026-01-29; IRI http://mlcommons.org/croissant/1.1", "source_type": "schema", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Dataset identity, FileObject integrity hashes, RecordSet fields, training/validation/test splits, PROV-O provenance and license metadata used when binding evaluation datasets without copying dataset-master semantics." }, { "id": "SRC-021", "title": "Inspect AI: Framework for Large Language Model Evaluations", "organization": "UK AI Security Institute", "url": "https://inspect.aisi.org.uk/", "version_or_date": "Software citation May 2024; documentation accessed 2026-08-26 including eval-logs and scoring pages", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Operational information model of a Task as dataset plus solver plus scorer; EvalLog fields (status, eval, plan, results, stats, samples, reductions); scoring, model grading, scanners, sandboxes and re-scoring workflow." }, { "id": "SRC-022", "title": "Holistic Evaluation of Language Models (HELM)", "organization": "Stanford Center for Research on Foundation Models", "url": "https://crfm.stanford.edu/2022/11/17/helm.html", "version_or_date": "2022-11-17; paper arXiv:2211.09110", "source_type": "scientific", "primary_source": true, "authority_tier": 3, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Scenario as task, domain and language; multi-metric measurement in the same context; standardized adaptation; explicit incompleteness of coverage; trade-offs among accuracy, calibration, robustness, fairness, bias, toxicity and efficiency." }, { "id": "SRC-023", "title": "Artificial Intelligence Risk Management Framework: Generative Artificial Intelligence Profile, NIST AI 600-1", "organization": "National Institute of Standards and Technology", "url": "https://www.nist.gov/publications/artificial-intelligence-risk-management-framework-generative-artificial-intelligence", "version_or_date": "NIST Trustworthy and Responsible AI 600-1, 2024-07-26", "source_type": "public-authority", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Cross-sector generative-AI profile of AI RMF for design, development, use and evaluation of GAI, including GAI-specific risks that evaluation campaigns must map to metrics and events." }, { "id": "SRC-024", "title": "ISO/IEC TR 29119-11:2020 Software and systems engineering — Software testing — Part 11: Guidelines on the testing of AI-based systems", "organization": "ISO/IEC JTC 1/SC 7 with ISO/IEC JTC 1/SC 42", "url": "https://www.vde-verlag.de/iec-normen/preview-pdf/info_isoiectr29119-11%7Bed1.0%7Den.pdf", "version_or_date": "First edition 2020-11", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Test-oracle problem, non-determinism, acceptance-criteria difficulty, black-box and neural-network white-box testing, test environments and lifecycle testing of AI-based systems." }, { "id": "SRC-025", "title": "Autonomous Systems Evaluation Standard", "organization": "UK AI Security Institute", "url": "https://ukgovernmentbeis.github.io/as-evaluation-standard", "version_or_date": "2024-10-31", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Minimum quality bar for submitting evaluations: Inspect-based implementation, repository structure, and recommended binary success metric as main scorer for autonomous-system tasks." } ], "structure": { "bundles": [ { "id": "b-scope-and-mandate", "name": "Evaluation Scope and Mandate", "description": "What is being evaluated, for what claim, under whose authority, and by whom.", "rationale": "MEASURE 1.1 requires measurement approaches to be selected for the most significant contextual risks; AI Act Article 9(8) requires testing against metrics appropriate to the intended purpose; Clause 9 requires the organisation to determine what is measured and by whom; MEASURE 1.3 requires internal experts and independent assessors.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-006" ], "layers": [ { "id": "l-subject-and-claim", "name": "Subject and Claim Scope", "description": "The exact object under test and the claim the evaluation is designed to support or refute.", "source_refs": [ "SRC-002", "SRC-010", "SRC-011" ], "findings": [ { "id": "f-subject-binding", "name": "Evaluation subject binding", "description": "The immutable binding between the evaluation and the model or system version, weights, configuration and inference settings under test.", "source_refs": [ "SRC-002", "SRC-009", "SRC-010", "SRC-011" ], "questions": [ { "id": "q-subject-identifier", "text": "Which authoritative identifier and version designate the exact model or system under evaluation?", "kind": "identity", "answer_data": [ "Master-system model artefact identifier", "Version label", "Content digest of weights or endpoint build" ] }, { "id": "q-subject-unit-boundary", "text": "Is the evaluated unit the bare model, a guarded endpoint, or the full deployed system including scaffolding and tools?", "kind": "classification", "answer_data": [ "Evaluated-unit type code", "Included components list", "Excluded components list" ] }, { "id": "q-subject-configuration", "text": "Which inference configuration, decoding parameters, system prompt, tooling and safety filters form part of the evaluated subject?", "kind": "composition", "answer_data": [ "Configuration object", "Prompt or template reference", "Filter and guardrail settings" ] }, { "id": "q-subject-pinning", "text": "How is a hosted or continuously updated endpoint pinned so results remain attributable to a fixed subject?", "kind": "temporal", "answer_data": [ "Endpoint pin or snapshot reference", "Pin establishment event time", "Detected-change handling rule" ] } ], "data_elements": [ { "id": "de-subject-model-ref", "name": "Subject model reference", "description": "Reference to the evaluated artefact in its master system of record.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-010" ] }, { "id": "de-subject-artifact-digest", "name": "Subject artefact digest", "description": "Content digest of weights, image or build identifying the exact evaluated binary.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-008", "SRC-009" ] }, { "id": "de-subject-unit-type", "name": "Evaluated unit type", "description": "Whether the evaluated unit is a model, guarded endpoint or full system.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-011" ] }, { "id": "de-subject-config", "name": "Subject configuration", "description": "Complete inference and scaffolding configuration treated as part of the subject.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-009" ] } ], "artifacts": [ { "id": "a-subject-binding-record", "name": "Subject binding record", "description": "Record fixing the evaluated subject, its version, digest and configuration for the whole evaluation instance.", "media_or_form": [ "structured record", "manifest entry" ], "serial": false, "identity_strategy": "Master-system artefact identifier plus version, with a locally minted binding identifier when the same artefact is bound more than once.", "source_refs": [ "SRC-011", "SRC-002" ] } ], "inline_only_rationale": null }, { "id": "f-purpose-and-claim-scope", "name": "Intended purpose and claim scope", "description": "The intended purpose, deployment context and the specific claim the evaluation supports, with explicit out-of-scope uses and generalisability limits.", "source_refs": [ "SRC-001", "SRC-002", "SRC-010" ], "questions": [ { "id": "q-claim-statement", "text": "What precise claim about capability, safety or quality is this evaluation designed to support or refute?", "kind": "definition", "answer_data": [ "Claim statement", "Property or quality characteristic addressed", "Direction of the claim" ] }, { "id": "q-intended-purpose-link", "text": "Which documented intended purpose, deployment context and affected populations bound the validity of the claim?", "kind": "relationship", "answer_data": [ "Intended-purpose reference", "Deployment context description", "Affected population list" ] }, { "id": "q-out-of-scope-use", "text": "Which uses, populations, languages or environments are explicitly outside the evaluated claim?", "kind": "constraint", "answer_data": [ "Out-of-scope use list", "Untested modality or language list", "Exclusion rationale" ] }, { "id": "q-generalisation-limit", "text": "What documented limits on generalisability apply outside the tested distribution?", "kind": "quality", "answer_data": [ "Generalisability limit statement", "Tested distribution description", "Extrapolation prohibition" ] } ], "data_elements": [ { "id": "de-claim-statement", "name": "Evaluation claim statement", "description": "The single claim the evaluation is designed to test.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-010" ] }, { "id": "de-intended-purpose-ref", "name": "Intended purpose reference", "description": "Link to the governed statement of intended purpose and deployment context.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-002" ] }, { "id": "de-out-of-scope-use", "name": "Out-of-scope use", "description": "Uses expressly not covered by the evaluated claim.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-010" ] }, { "id": "de-generalisation-limit", "name": "Generalisability limit", "description": "Documented limit on inference beyond the tested distribution.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] } ], "artifacts": [ { "id": "a-evaluation-scope-statement", "name": "Evaluation scope statement", "description": "Signed statement of claim, intended purpose, exclusions and generalisability limits governing the instance.", "media_or_form": [ "document", "structured record" ], "serial": false, "identity_strategy": "Locally minted scope-statement identifier bound to the evaluation instance identifier and version.", "source_refs": [ "SRC-001", "SRC-010" ] } ], "inline_only_rationale": null } ] }, { "id": "l-mandate-and-actors", "name": "Mandate, Actors and Independence", "description": "Why the evaluation exists, what triggered it, and who is competent and independent enough to perform it.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-006" ], "findings": [ { "id": "f-mandate-and-trigger", "name": "Evaluation mandate and trigger", "description": "The legal, contractual, policy or research basis obliging the evaluation, the event that triggered this instance, and the decision that consumes it.", "source_refs": [ "SRC-002", "SRC-004", "SRC-006" ], "questions": [ { "id": "q-mandate-basis", "text": "Under which legal, regulatory, contractual or internal-policy basis is this evaluation required or commissioned?", "kind": "authority", "answer_data": [ "Mandate basis code", "Cited clause or policy reference", "Obligated party" ] }, { "id": "q-trigger-event", "text": "Which event triggered this evaluation instance — release gate, capability threshold, incident, drift alarm or scheduled review?", "kind": "event", "answer_data": [ "Trigger type code", "Trigger event time", "Triggering record reference" ] }, { "id": "q-decision-consumer", "text": "Which decision or gate consumes the result, and who is accountable for that decision?", "kind": "decision", "answer_data": [ "Consuming decision reference", "Accountable role", "Required completion date" ] } ], "data_elements": [ { "id": "de-mandate-basis", "name": "Mandate basis", "description": "Classification of why the evaluation is performed.", "value_kind": "code", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-006" ] }, { "id": "de-trigger-event-type", "name": "Trigger event type", "description": "Kind of event that initiated this evaluation instance.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-002" ] }, { "id": "de-trigger-event-time", "name": "Trigger event time", "description": "When the triggering event occurred, distinct from when it was recorded.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-007", "SRC-002" ] }, { "id": "de-consuming-decision-ref", "name": "Consuming decision reference", "description": "The gate or decision that will rely on the result.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004", "SRC-006" ] } ], "artifacts": [ { "id": "a-evaluation-mandate-record", "name": "Evaluation mandate record", "description": "Record of basis, trigger, obligated party and consuming decision for the instance.", "media_or_form": [ "structured record", "register entry" ], "serial": true, "identity_strategy": "Evaluation instance identifier from the quality or records master system; sequence number when triggers recur.", "source_refs": [ "SRC-006", "SRC-002" ] } ], "inline_only_rationale": null }, { "id": "f-assessor-independence", "name": "Assessor identity, competence and independence", "description": "Who designed, executed, scored and reviewed the evaluation, their competence evidence, degree of independence from the development team, and declared conflicts.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009" ], "questions": [ { "id": "q-actor-roles", "text": "Which named organisations and roles designed, executed, scored and reviewed this evaluation?", "kind": "ownership", "answer_data": [ "Actor organisation identifier", "Role assignment per activity", "Named responsible person" ] }, { "id": "q-independence-degree", "text": "What degree of independence separates the evaluators from the development team, and how is it evidenced?", "kind": "classification", "answer_data": [ "Independence level code", "Reporting-line evidence", "Separation-of-duties control" ] }, { "id": "q-competence-evidence", "text": "What evidence establishes that assessors and human raters are competent for the domain and risk being measured?", "kind": "evidence", "answer_data": [ "Competence evidence reference", "Domain qualification", "Training record" ] }, { "id": "q-conflict-of-interest", "text": "Which conflicts of interest were declared and how were they managed?", "kind": "constraint", "answer_data": [ "Declared conflict", "Mitigation applied", "Declaration event time" ] } ], "data_elements": [ { "id": "de-actor-org-id", "name": "Actor organisation identifier", "description": "Authoritative identifier of each participating organisation.", "value_kind": "identifier", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-007", "SRC-002" ] }, { "id": "de-actor-role", "name": "Actor role assignment", "description": "Role held by each actor for each evaluation activity.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-010", "SRC-001" ] }, { "id": "de-independence-level", "name": "Independence level", "description": "Degree of separation between evaluators and developers.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-004" ] }, { "id": "de-coi-declaration", "name": "Conflict-of-interest declaration", "description": "Declared conflicts and their management.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009", "SRC-004" ] } ], "artifacts": [ { "id": "a-assessor-register", "name": "Assessor and role register", "description": "Register of participating actors, roles, competence evidence, independence level and declared conflicts.", "media_or_form": [ "structured record", "register entry" ], "serial": false, "identity_strategy": "Organisation and person identifiers from the identity master system; local register entry identifier per evaluation instance.", "source_refs": [ "SRC-001", "SRC-002" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "b-evaluation-design", "name": "Evaluation Design", "description": "The protocol: method class, task and dataset binding, metrics, disaggregation, acceptance criteria and human or adversarial procedures, all fixed before results are observed.", "rationale": "MEASURE 2.1 requires test sets, metrics and tools to be documented; Article 9(8) requires prior-defined metrics and probabilistic thresholds; Annex IV point 2(g) requires documented validation and testing procedures; TS 4213 sets control criteria for evaluation setup, ground truth and splits; the GPAI Code requires acceptance determination against defined risk tiers with safety margins.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-005" ], "layers": [ { "id": "l-method-and-task-design", "name": "Method Class and Task Binding", "description": "Which method class is used and how tasks, datasets, splits and ground truth are bound to the evaluation.", "source_refs": [ "SRC-005", "SRC-008", "SRC-014", "SRC-015" ], "findings": [ { "id": "f-method-class", "name": "Evaluation method class and baseline", "description": "Classification of the evaluation method — statistical, formal, empirical, human-subject or adversarial — with justification, declared blind spots and comparison baselines.", "source_refs": [ "SRC-015", "SRC-005", "SRC-002", "SRC-001" ], "questions": [ { "id": "q-method-class-choice", "text": "Which method class is applied, and why is it appropriate to the property being assessed?", "kind": "classification", "answer_data": [ "Method class code", "Property or characteristic targeted", "Selection rationale" ] }, { "id": "q-method-blind-spot", "text": "Which properties can this method class not evidence, and what complementary method covers that gap?", "kind": "constraint", "answer_data": [ "Non-evidenced property list", "Complementary method reference", "Residual coverage gap" ] }, { "id": "q-baseline-comparator", "text": "Against which baseline, reference implementation or comparator system are results interpreted?", "kind": "measurement", "answer_data": [ "Baseline reference", "Baseline version", "Comparator selection rationale" ] }, { "id": "q-state-of-the-art-basis", "text": "How is the method shown to reflect the state of the art at the time of execution?", "kind": "requirement", "answer_data": [ "State-of-the-art justification", "Cited protocol or tool version", "Assessment event time" ] } ], "data_elements": [ { "id": "de-method-class", "name": "Method class", "description": "Statistical, formal, empirical, human-subject or adversarial method classification.", "value_kind": "code", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-015" ] }, { "id": "de-method-rationale", "name": "Method selection rationale", "description": "Why the chosen class fits the property under assessment.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-005" ] }, { "id": "de-baseline-ref", "name": "Baseline reference", "description": "Reference implementation or comparator against which results are read.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-005", "SRC-009" ] }, { "id": "de-state-of-art-basis", "name": "State-of-the-art basis", "description": "Evidence that the protocol reflects current practice at execution time.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002", "SRC-004" ] } ], "artifacts": [ { "id": "a-evaluation-protocol", "name": "Registered evaluation protocol", "description": "The pre-registered protocol covering method class, tasks, metrics, analysis plan and acceptance criteria.", "media_or_form": [ "document", "structured record", "versioned specification" ], "serial": true, "identity_strategy": "Protocol identifier plus immutable version; supersession recorded by reference, never by in-place edit.", "source_refs": [ "SRC-002", "SRC-001", "SRC-005" ] } ], "inline_only_rationale": null }, { "id": "f-dataset-binding-and-contamination", "name": "Task and dataset binding with contamination control", "description": "Version-pinned reference to evaluation tasks, datasets and splits, the ground-truth definition, and controls against train/test contamination and held-out set erosion.", "source_refs": [ "SRC-005", "SRC-008", "SRC-002", "SRC-014" ], "questions": [ { "id": "q-dataset-version-ref", "text": "Which dataset version, snapshot or split identifiers, and which checksums, are bound to this evaluation?", "kind": "relationship", "answer_data": [ "Dataset reference", "Semantic version", "Content checksum per file or split" ] }, { "id": "q-ground-truth-definition", "text": "How is ground truth or the reference answer defined, produced and quality-assured for each item?", "kind": "definition", "answer_data": [ "Ground-truth production method", "Adjudication rule", "Known label-error rate" ] }, { "id": "q-contamination-control", "text": "What controls establish that the evaluation items were not present in training or tuning data?", "kind": "validation", "answer_data": [ "Contamination check method", "Check result and date", "Residual contamination risk" ] }, { "id": "q-holdout-governance", "text": "How is a held-out or canary split protected from repeated exposure that would invalidate future results?", "kind": "access", "answer_data": [ "Holdout access rule", "Exposure counter", "Rotation or retirement policy" ] } ], "data_elements": [ { "id": "de-dataset-ref", "name": "Dataset reference", "description": "Reference to the governed dataset entry, owned by the benchmark dataset model.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-008", "SRC-005" ] }, { "id": "de-dataset-version", "name": "Dataset version", "description": "Version or snapshot label of the bound dataset.", "value_kind": "text", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-008" ] }, { "id": "de-dataset-checksum", "name": "Dataset content checksum", "description": "Digest over dataset files or splits used for integrity verification.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "de-split-designation", "name": "Split designation", "description": "Whether the bound partition is training, validation, test, holdout or canary.", "value_kind": "code", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-005", "SRC-002" ] }, { "id": "de-ground-truth-method", "name": "Ground-truth method", "description": "How reference answers were produced and adjudicated.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-005", "SRC-014" ] }, { "id": "de-contamination-check", "name": "Contamination check result", "description": "Method, result and date of the train/test contamination check.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009", "SRC-005" ] } ], "artifacts": [ { "id": "a-dataset-binding-manifest", "name": "Dataset binding manifest", "description": "Manifest binding dataset versions, splits, checksums and contamination-check outcomes to the evaluation.", "media_or_form": [ "manifest", "structured record", "linked-data descriptor" ], "serial": false, "identity_strategy": "Dataset master-system identifier plus version and split key; local manifest identifier for the binding itself.", "source_refs": [ "SRC-008", "SRC-005" ] } ], "inline_only_rationale": null }, { "id": "f-scenario-and-adaptation", "name": "Scenario, task and adaptation protocol", "description": "A HELM scenario is a task plus domain plus language. Inspect Task composes dataset, solver or agent, and scorer, optionally with a sandbox and tools. Standardized few-shot or other adaptation is required so comparisons attribute performance to the subject. UK AISI autonomous-system submissions must be Inspect tasks and should expose a binary success scorer.", "source_refs": [ "SRC-021", "SRC-022", "SRC-025" ], "questions": [ { "id": "f-scenario-and-adaptation-q01", "text": "What task, domain, language or modality, and user-facing use case define this scenario, and which HELM or Inspect eval identifier does it correspond to?", "kind": "identity", "answer_data": [ "scenario_id", "task_type", "domain", "language_or_modality", "public_eval_name" ] }, { "id": "f-scenario-and-adaptation-q02", "text": "What adaptation protocol (zero-shot, few-shot, chain-of-thought, agent loop, tool use) is frozen, including prompt templates and in-context example policy?", "kind": "process", "answer_data": [ "adaptation_kind", "prompt_template_id", "n_shot", "agent_loop_limits" ] }, { "id": "f-scenario-and-adaptation-q03", "text": "If the task is agentic, which tools, MCP servers and sandbox (Docker, Kubernetes or other) isolate untrusted model actions, and what approval gates exist for tool calls?", "kind": "security", "answer_data": [ "sandbox_kind", "tool_list", "mcp_servers", "tool_approval_policy" ] }, { "id": "f-scenario-and-adaptation-q04", "text": "Does the scenario depend on a geographic, legal or language locale, and if so which locale constraints apply to items and raters?", "kind": "spatial", "answer_data": [ "locale_code", "jurisdiction", "language_tag", "rater_locale" ] } ], "data_elements": [ { "id": "f-scenario-and-adaptation-data01", "name": "Scenario identifier", "description": "Identifier of the task-domain-language triple or Inspect task name.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-021", "SRC-022" ] }, { "id": "f-scenario-and-adaptation-data02", "name": "Adaptation kind", "description": "Coded adaptation strategy frozen for comparability.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-022" ] }, { "id": "f-scenario-and-adaptation-data03", "name": "Sandbox kind", "description": "Isolation mechanism for untrusted code or tool use, if any.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-021" ] } ], "artifacts": [ { "id": "f-scenario-and-adaptation-artifact01", "name": "Task and adaptation specification", "description": "Inspect-style or equivalent task definition: dataset map, solver, scorer, sandbox and prompts.", "media_or_form": [ "structured task record", "source projection such as Python module" ], "serial": true, "identity_strategy": "Task or scenario identifier from the evaluation platform of record; prompt and solver revisions are serials.", "source_refs": [ "SRC-021", "SRC-025" ] } ], "inline_only_rationale": null } ] }, { "id": "l-metrics-and-acceptance", "name": "Metrics, Disaggregation and Acceptance", "description": "How results are computed, how they are broken down, and what counts as pass or fail.", "source_refs": [ "SRC-001", "SRC-002", "SRC-005", "SRC-009" ], "findings": [ { "id": "f-metric-specification", "name": "Metric specification", "description": "Formal specification of each metric: construct measured, computation, aggregation, reporting precision and the scorer implementation and version producing it.", "source_refs": [ "SRC-005", "SRC-009", "SRC-014", "SRC-013" ], "questions": [ { "id": "q-metric-definition", "text": "What is the exact computational definition, unit and aggregation rule for each reported metric?", "kind": "definition", "answer_data": [ "Metric identifier", "Computation formula or specification reference", "Unit and aggregation rule" ] }, { "id": "q-metric-construct-validity", "text": "Which construct does the metric claim to measure, and what evidence links the metric to that construct?", "kind": "quality", "answer_data": [ "Named construct or quality characteristic", "Construct-validity evidence", "Known proxy weakness" ] }, { "id": "q-metric-implementation", "text": "Which scorer implementation, version and configuration produced the metric values?", "kind": "provenance", "answer_data": [ "Scorer implementation reference", "Scorer version", "Scorer configuration" ] }, { "id": "q-metric-reporting-precision", "text": "What rounding, significant-figure and tie-breaking conventions govern reported values?", "kind": "constraint", "answer_data": [ "Significant figures", "Rounding mode", "Tie-breaking rule" ] } ], "data_elements": [ { "id": "de-metric-id", "name": "Metric identifier", "description": "Stable identifier for a metric definition; changing semantics requires a new identifier.", "value_kind": "identifier", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-005", "SRC-014" ] }, { "id": "de-metric-definition", "name": "Metric definition", "description": "Exact computational definition of the metric.", "value_kind": "text", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-005", "SRC-013" ] }, { "id": "de-scorer-impl-ref", "name": "Scorer implementation reference", "description": "Implementation and version that computed the metric.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-011", "SRC-014" ] }, { "id": "de-reporting-precision", "name": "Reporting precision rule", "description": "Significant figures, rounding mode and tie-breaking convention.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009" ] } ], "artifacts": [ { "id": "a-metric-definition-record", "name": "Metric definition record", "description": "Versioned catalogue entry defining a metric, its construct, aggregation and scorer binding.", "media_or_form": [ "structured record", "catalogue entry" ], "serial": false, "identity_strategy": "Metric identifier governed by the adopting Dimension's metric catalogue; semantic change forces a new identifier.", "source_refs": [ "SRC-005", "SRC-009" ] } ], "inline_only_rationale": null }, { "id": "f-disaggregation", "name": "Disaggregation factors and subgroup analysis", "description": "Factors along which results must be broken down, the sufficiency rules for subgroup reporting, and the disparity comparisons performed.", "source_refs": [ "SRC-001", "SRC-010", "SRC-005", "SRC-002" ], "questions": [ { "id": "q-disaggregation-factors", "text": "Which factors must results be disaggregated by, and how was that factor list justified?", "kind": "composition", "answer_data": [ "Factor list", "Justification for each factor", "Factors considered and rejected" ] }, { "id": "q-subgroup-sufficiency", "text": "What minimum sample size per subgroup is required before a disaggregated result is reportable?", "kind": "measurement", "answer_data": [ "Minimum subgroup sample size", "Suppression rule below threshold", "Achieved sample size per subgroup" ] }, { "id": "q-fairness-comparison", "text": "Which disparity comparison is applied across subgroups, and what gap is treated as a finding?", "kind": "validation", "answer_data": [ "Disparity measure", "Disparity threshold", "Comparison reference group" ] }, { "id": "q-subgroup-attribute-basis", "text": "How are protected attributes obtained or inferred for disaggregation, and under which lawful basis?", "kind": "privacy", "answer_data": [ "Attribute source code", "Lawful basis", "Inference method and error rate" ] } ], "data_elements": [ { "id": "de-factor", "name": "Disaggregation factor", "description": "Task, domain, language, difficulty or subgroup dimension for breakdown.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-010", "SRC-001" ] }, { "id": "de-factor-justification", "name": "Factor justification", "description": "Reason each factor was selected, and which were rejected.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-010" ] }, { "id": "de-subgroup-min-n", "name": "Minimum subgroup sample size", "description": "Threshold below which a disaggregated value is suppressed.", "value_kind": "number", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "de-disparity-measure", "name": "Disparity measure", "description": "Comparison used to express performance gaps between subgroups.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001", "SRC-010" ] }, { "id": "de-attribute-source", "name": "Protected attribute source", "description": "Whether attributes are self-declared, assigned, inferred or synthetic.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002", "SRC-001" ] } ], "artifacts": [ { "id": "a-disaggregated-result-table", "name": "Disaggregated result table", "description": "Result values broken down by declared factors, with suppression markers where subgroup samples are insufficient.", "media_or_form": [ "tabular record", "structured record" ], "serial": true, "identity_strategy": "Composite key of evaluation instance, metric identifier and factor tuple.", "source_refs": [ "SRC-010", "SRC-001" ] } ], "inline_only_rationale": null }, { "id": "f-acceptance-criteria", "name": "Acceptance criteria and thresholds", "description": "Pass/fail criteria, probabilistic thresholds, risk tiers and safety margins fixed and recorded before results are observed, with change control for later alteration.", "source_refs": [ "SRC-002", "SRC-004", "SRC-009", "SRC-001" ], "questions": [ { "id": "q-threshold-preregistration", "text": "Which acceptance thresholds and probabilistic criteria were fixed before results were observed, and where is that pre-registration recorded?", "kind": "requirement", "answer_data": [ "Threshold identifier and value", "Pre-registration event time", "Pre-registration record reference" ] }, { "id": "q-threshold-derivation", "text": "Who set each threshold, on what basis, and with what safety margin?", "kind": "authority", "answer_data": [ "Setting authority", "Derivation basis", "Safety margin applied" ] }, { "id": "q-threshold-change-control", "text": "How is a change to a threshold after results are known authorised, justified and disclosed?", "kind": "lifecycle", "answer_data": [ "Change record reference", "Authoriser", "Disclosure statement" ] }, { "id": "q-breach-consequence", "text": "What consequence follows automatically from breaching each threshold?", "kind": "decision", "answer_data": [ "Consequence per threshold", "Escalation path", "Blocking or non-blocking flag" ] } ], "data_elements": [ { "id": "de-threshold-id", "name": "Threshold identifier", "description": "Stable identifier for an acceptance criterion.", "value_kind": "identifier", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002" ] }, { "id": "de-threshold-value", "name": "Threshold value", "description": "Numeric or probabilistic bound with its unit and comparison direction.", "value_kind": "quantity", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-009" ] }, { "id": "de-prereg-time", "name": "Pre-registration time", "description": "Event time at which criteria were fixed, necessarily before first result observation.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-004" ] }, { "id": "de-safety-margin", "name": "Safety margin", "description": "Margin applied between the measured bound and the tolerated risk level.", "value_kind": "quantity", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-breach-consequence", "name": "Breach consequence", "description": "Action that follows automatically from breaching the criterion.", "value_kind": "text", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-004", "SRC-002" ] } ], "artifacts": [ { "id": "a-acceptance-criteria-record", "name": "Acceptance criteria record", "description": "Immutable pre-registration of thresholds, margins, authorities and breach consequences.", "media_or_form": [ "structured record", "signed document" ], "serial": false, "identity_strategy": "Protocol identifier plus criteria-set version; append-only with dated change records.", "source_refs": [ "SRC-002", "SRC-004" ] } ], "inline_only_rationale": null } ] }, { "id": "l-human-and-adversarial", "name": "Human and Adversarial Protocols", "description": "Procedures for evaluations whose validity depends on people: raters and red teams.", "source_refs": [ "SRC-001", "SRC-012", "SRC-004", "SRC-002" ], "findings": [ { "id": "f-human-evaluation-protocol", "name": "Human evaluation protocol", "description": "Recruitment, representativeness, instructions, blinding, agreement measurement and participant protection for human-rated evaluation.", "source_refs": [ "SRC-001", "SRC-010", "SRC-004" ], "questions": [ { "id": "q-rater-population", "text": "How were raters recruited and how does the rater population relate to the affected population?", "kind": "composition", "answer_data": [ "Recruitment method", "Rater demographic profile", "Representativeness argument" ] }, { "id": "q-rater-agreement", "text": "Which inter-rater agreement statistic is computed and what value is required for the result to be usable?", "kind": "measurement", "answer_data": [ "Agreement statistic", "Achieved value", "Minimum acceptable value" ] }, { "id": "q-participant-protection", "text": "Which human-subject protections, consent arrangements and ethics approvals govern the evaluation?", "kind": "privacy", "answer_data": [ "Consent mechanism", "Ethics approval reference", "Harmful-content exposure safeguard" ] }, { "id": "q-blinding-protocol", "text": "How are raters blinded to system identity and condition to prevent expectancy effects?", "kind": "process", "answer_data": [ "Blinding method", "Randomisation scheme", "Unblinding conditions" ] } ], "data_elements": [ { "id": "de-rater-count", "name": "Rater count", "description": "Number of human raters contributing judgements.", "value_kind": "number", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "de-rater-recruitment", "name": "Rater recruitment description", "description": "How raters were sourced, screened and compensated.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-010" ] }, { "id": "de-agreement-value", "name": "Inter-rater agreement value", "description": "Computed agreement statistic for the rating task.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-ethics-approval-ref", "name": "Ethics approval reference", "description": "Reference to the human-subject protection approval, governed elsewhere.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] } ], "artifacts": [ { "id": "a-human-evaluation-protocol", "name": "Human evaluation protocol and rater instructions", "description": "Protocol plus the exact instruction set shown to raters, retained so judgements can be interpreted later.", "media_or_form": [ "document", "structured record", "instruction set" ], "serial": false, "identity_strategy": "Protocol identifier plus instruction-set version; instruction changes create a new version.", "source_refs": [ "SRC-001", "SRC-010" ] } ], "inline_only_rationale": null }, { "id": "f-red-team-protocol", "name": "Adversarial and red-team protocol", "description": "Threat model, attacker capability and knowledge assumptions, attack-class coverage, elicitation effort and stopping rules for adversarial testing.", "source_refs": [ "SRC-012", "SRC-002", "SRC-004", "SRC-001" ], "questions": [ { "id": "q-threat-model", "text": "Which threat model, attacker goals, capabilities and knowledge assumptions define this adversarial test?", "kind": "security", "answer_data": [ "Attacker goal", "Attacker capability set", "Attacker knowledge level" ] }, { "id": "q-attack-class-coverage", "text": "Which attack classes — evasion, poisoning, privacy, prompt injection, misaligned output — are in and out of scope?", "kind": "classification", "answer_data": [ "In-scope attack class list", "Out-of-scope attack class list", "Lifecycle stage targeted" ] }, { "id": "q-elicitation-effort", "text": "How much elicitation effort, tooling and time was expended, and how is that effort evidenced as sufficient?", "kind": "measurement", "answer_data": [ "Person-hours expended", "Tooling and affordances granted", "Sufficiency argument" ] }, { "id": "q-stopping-rule", "text": "What stopping rule ends the exercise, and how are unresolved attack avenues recorded?", "kind": "constraint", "answer_data": [ "Stopping criterion", "Unresolved avenue list", "Follow-up commitment" ] } ], "data_elements": [ { "id": "de-threat-model", "name": "Threat model", "description": "Structured statement of attacker goals, capabilities, knowledge and access.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-012" ] }, { "id": "de-attack-class", "name": "Attack class in scope", "description": "Attack categories covered by the exercise.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-012" ] }, { "id": "de-elicitation-effort", "name": "Elicitation effort", "description": "Quantified effort expended to elicit the behaviour under test.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004", "SRC-002" ] }, { "id": "de-external-redteam-flag", "name": "External red team indicator", "description": "Whether independent external experts participated.", "value_kind": "boolean", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-004" ] } ], "artifacts": [ { "id": "a-red-team-exercise-record", "name": "Red-team protocol and exercise log", "description": "Threat model, scope, granted affordances, attempt log and unresolved avenues, held under restricted access.", "media_or_form": [ "document", "structured record", "append-only log" ], "serial": true, "identity_strategy": "Exercise identifier plus monotonic attempt sequence; hazardous detail held in a separately labelled partition.", "source_refs": [ "SRC-012", "SRC-004" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "b-execution-and-evidence", "name": "Execution, Results and Evidence", "description": "Run records, environment capture, result and item-level evidence, uncertainty, integrity and provenance.", "rationale": "MEASURE 2.1 requires documented test sets, metrics and tools; Annex IV requires test logs and dated signed test reports; MLPerf mandates replicability and bounded non-determinism; Inspect and HELM demonstrate the run/sample/score record structure; Croissant supplies checksum-based integrity; PROV-O supplies lineage semantics.", "source_refs": [ "SRC-001", "SRC-002", "SRC-007", "SRC-009", "SRC-011", "SRC-014" ], "layers": [ { "id": "l-run-execution", "name": "Run Execution and Reproducibility", "description": "The record of an individual execution and what is needed for an independent party to repeat it.", "source_refs": [ "SRC-009", "SRC-011", "SRC-014" ], "findings": [ { "id": "f-run-record", "name": "Run record, configuration and resource accounting", "description": "Identity of a single run, its complete configuration, timing, and the compute, token, latency and energy quantities attributable to it.", "source_refs": [ "SRC-011", "SRC-009", "SRC-010", "SRC-014" ], "questions": [ { "id": "q-run-identifier", "text": "Which identifier uniquely designates this run, and how does it relate to the campaign and to repeated epochs?", "kind": "identity", "answer_data": [ "Run identifier", "Campaign identifier", "Epoch or repetition index" ] }, { "id": "q-run-config-capture", "text": "Which configuration values — seeds, sampling parameters, batch, concurrency, epochs, prompt template — are captured with the run?", "kind": "composition", "answer_data": [ "Configuration object", "Random seed set", "Adaptation or prompt specification" ] }, { "id": "q-run-timing", "text": "What are the run start and end times, and how are they distinguished from result publication and record ingestion times?", "kind": "temporal", "answer_data": [ "Run start time", "Run end time", "Record ingestion time" ] }, { "id": "q-run-resource-accounting", "text": "What compute, token, latency, monetary and energy quantities are recorded for the run?", "kind": "measurement", "answer_data": [ "Token usage counts", "Latency percentiles", "Energy or carbon estimate with method" ] } ], "data_elements": [ { "id": "de-run-id", "name": "Run identifier", "description": "Unique identifier for one execution of the protocol.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-011" ] }, { "id": "de-run-config", "name": "Run configuration", "description": "Complete parameter set governing the execution.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-014" ] }, { "id": "de-random-seed", "name": "Random seed", "description": "Announced seed values bounding permitted non-determinism.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-run-start-time", "name": "Run start time", "description": "Event time at which execution began.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-007", "SRC-011" ] }, { "id": "de-run-end-time", "name": "Run end time", "description": "Event time at which execution completed or aborted.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007", "SRC-011" ] }, { "id": "de-token-usage", "name": "Token and compute usage", "description": "Model usage statistics attributable to the run.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-011" ] }, { "id": "de-energy-estimate", "name": "Energy or carbon estimate", "description": "Estimated energy use or emissions with the estimation method recorded.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-010", "SRC-001" ] } ], "artifacts": [ { "id": "a-evaluation-run-record", "name": "Evaluation run record", "description": "Header and status record for one run, carrying identity, configuration, timing, usage statistics and error state.", "media_or_form": [ "append-only log", "structured record" ], "serial": true, "identity_strategy": "Run identifier minted by the execution service, monotonic within a campaign; file-name timestamps are ordering hints only, never identity.", "source_refs": [ "SRC-011", "SRC-009" ] } ], "inline_only_rationale": null }, { "id": "f-reproducibility", "name": "Environment capture and replication", "description": "Software and hardware environment, dependency versions, bounded non-determinism, replication tolerance and declared irreproducible components.", "source_refs": [ "SRC-009", "SRC-011", "SRC-014", "SRC-005" ], "questions": [ { "id": "q-environment-capture", "text": "Which code revision, dependency versions, container image and hardware description are recorded for the run?", "kind": "provenance", "answer_data": [ "Code revision identifier", "Package and version list", "Hardware and accelerator description" ] }, { "id": "q-determinism-boundary", "text": "Which sources of non-determinism are permitted, and how are they bounded or seeded?", "kind": "constraint", "answer_data": [ "Permitted non-determinism list", "Seeding scheme", "Prohibited optimisation list" ] }, { "id": "q-replication-procedure", "text": "What exact procedure and inputs allow an independent party to replicate the run, and what tolerance defines success?", "kind": "process", "answer_data": [ "Replication procedure reference", "Required inputs and access", "Numeric tolerance for agreement" ] }, { "id": "q-irreproducible-component", "text": "Which parts of the run are irreproducible, such as a retired endpoint or a one-off rater panel, and how is that recorded?", "kind": "exception", "answer_data": [ "Irreproducible component list", "Reason", "Compensating evidence" ] } ], "data_elements": [ { "id": "de-code-revision", "name": "Code revision identifier", "description": "Revision of the harness and scorer code used.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-011" ] }, { "id": "de-package-versions", "name": "Package version list", "description": "Dependency versions captured with the run.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-011" ] }, { "id": "de-nondeterminism-source", "name": "Non-determinism source", "description": "Permitted sources of run-to-run variation.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-replication-tolerance", "name": "Replication tolerance", "description": "Numeric tolerance within which a replication counts as successful.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009", "SRC-005" ] }, { "id": "de-irreproducible-note", "name": "Irreproducibility note", "description": "Declared components that cannot be re-executed.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] } ], "artifacts": [ { "id": "a-environment-manifest", "name": "Environment manifest", "description": "Captured revision, dependency, container and hardware description enabling replication.", "media_or_form": [ "manifest", "structured record" ], "serial": true, "identity_strategy": "Digest of the manifest content, referenced from the run record.", "source_refs": [ "SRC-011", "SRC-009" ] } ], "inline_only_rationale": null } ] }, { "id": "l-results-and-uncertainty", "name": "Results, Item Evidence and Uncertainty", "description": "What was measured, the item-level evidence behind it, and how confident the numbers are.", "source_refs": [ "SRC-011", "SRC-014", "SRC-005", "SRC-009" ], "findings": [ { "id": "f-result-and-item-evidence", "name": "Result record, aggregation and item-level evidence", "description": "Reported metric values keyed by subject, task, split, metric and factor, the traceable aggregation path from per-item scores, and the retained per-item inputs, outputs, traces and scores.", "source_refs": [ "SRC-011", "SRC-014", "SRC-002", "SRC-005" ], "questions": [ { "id": "q-result-identity", "text": "How is a single reported result value uniquely identified across subject, task, split, metric and factor?", "kind": "identity", "answer_data": [ "Result identifier", "Scope key tuple", "Run reference" ] }, { "id": "q-aggregation-path", "text": "What is the traceable path from per-item scores through reducers to each headline value?", "kind": "composition", "answer_data": [ "Reducer chain", "Per-item score set reference", "Excluded or failed item count" ] }, { "id": "q-item-record-content", "text": "Which fields are retained per evaluated item, and which are required to re-score without re-running inference?", "kind": "evidence", "answer_data": [ "Retained item field list", "Re-scorable indicator", "Trace sampling rate" ] }, { "id": "q-result-supersession", "text": "When a result is recomputed or corrected, how are the prior value, the scorer version and the reason retained?", "kind": "state", "answer_data": [ "Superseded result reference", "Original and new value", "Correction reason and event time" ] } ], "data_elements": [ { "id": "de-result-id", "name": "Result identifier", "description": "Identifier for a single reported metric value in its full scope.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-014" ] }, { "id": "de-result-value", "name": "Result value", "description": "The measured value with unit and recorded precision.", "value_kind": "quantity", "cardinality": "1", "required": true, "source_refs": [ "SRC-009", "SRC-014" ] }, { "id": "de-result-scope-key", "name": "Result scope key", "description": "Composite key of subject, task, split, metric and factor tuple.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-010" ] }, { "id": "de-result-status", "name": "Result status", "description": "Provisional, validated, superseded or withdrawn.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-001" ] }, { "id": "de-item-id", "name": "Item identifier", "description": "Identifier of an individual evaluated instance.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-014", "SRC-011" ] }, { "id": "de-item-output", "name": "Item model output", "description": "Model response retained for the item.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-011" ] }, { "id": "de-item-score", "name": "Item score", "description": "Per-item score with scorer metadata and edit history.", "value_kind": "object", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-011", "SRC-014" ] } ], "artifacts": [ { "id": "a-result-set", "name": "Result set", "description": "Aggregated result values with scope keys, status and supersession links.", "media_or_form": [ "structured record", "tabular record" ], "serial": true, "identity_strategy": "Result identifier composed from the run identifier and the scope key; supersession by explicit reference.", "source_refs": [ "SRC-011", "SRC-014" ] }, { "id": "a-per-item-trace-log", "name": "Per-item trace log", "description": "Item-level inputs, targets, messages, outputs, scores and annotations enabling audit and re-scoring.", "media_or_form": [ "append-only log", "structured record" ], "serial": true, "identity_strategy": "Run identifier plus item identifier plus epoch index; retained subject to sensitivity labels and retention rules.", "source_refs": [ "SRC-011", "SRC-014", "SRC-002" ] } ], "inline_only_rationale": null }, { "id": "f-uncertainty", "name": "Uncertainty and statistical validity", "description": "Sample size adequacy, variance, interval estimates, significance rules and multiple-comparison control governing what the numbers may be said to show.", "source_refs": [ "SRC-005", "SRC-009", "SRC-001", "SRC-015" ], "questions": [ { "id": "q-uncertainty-quantification", "text": "Which uncertainty estimate accompanies each headline value, and by what procedure was it computed?", "kind": "measurement", "answer_data": [ "Uncertainty type", "Interval bounds and level", "Computation procedure" ] }, { "id": "q-sample-size-adequacy", "text": "How was the number of items and repetitions determined to be adequate for the claimed resolution?", "kind": "validation", "answer_data": [ "Sample size", "Repetition count", "Power or resolution argument" ] }, { "id": "q-difference-significance", "text": "Under which rule is a difference between two systems or two versions declared meaningful rather than noise?", "kind": "constraint", "answer_data": [ "Significance rule", "Minimum detectable difference", "Comparison scope" ] }, { "id": "q-multiplicity-control", "text": "How are multiple comparisons across metrics, factors and checkpoints controlled?", "kind": "quality", "answer_data": [ "Multiplicity method", "Number of comparisons", "Adjusted criterion" ] } ], "data_elements": [ { "id": "de-sample-size", "name": "Sample size", "description": "Number of evaluated items contributing to the value.", "value_kind": "number", "cardinality": "1", "required": true, "source_refs": [ "SRC-005", "SRC-009" ] }, { "id": "de-uncertainty-interval", "name": "Uncertainty interval", "description": "Interval estimate with its bounds, level and type.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005", "SRC-009" ] }, { "id": "de-significance-rule", "name": "Significance rule", "description": "Rule for declaring a difference meaningful.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005", "SRC-001" ] }, { "id": "de-multiplicity-method", "name": "Multiplicity control method", "description": "Adjustment applied across many simultaneous comparisons.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001", "SRC-005" ] } ], "artifacts": [ { "id": "a-uncertainty-statement", "name": "Uncertainty statement", "description": "Statement of sampling design, intervals, significance rules and multiplicity control accompanying the result set.", "media_or_form": [ "document", "structured record" ], "serial": false, "identity_strategy": "Bound to the result set identifier and version.", "source_refs": [ "SRC-005", "SRC-009" ] } ], "inline_only_rationale": null }, { "id": "f-scoring-and-judging", "name": "Scoring and judging", "description": "Inspect scoring maps output to a Score against a Target, with built-in match, multiple-choice, math, model-graded and perplexity scorers, custom scorers, multiple scorers and epoch reductions. Model grading uses another model as judge. Human raters and CJE-style calibration of judges against oracles are first-class. ISO/IEC TR 29119-11 highlights the oracle problem when expected results are hard to specify. Refusals may be logged separately from task success.", "source_refs": [ "SRC-021", "SRC-024" ], "questions": [ { "id": "f-scoring-and-judging-q01", "text": "Which scorer or judge configuration produced each sample score, including grader model, rubric, and whether scoring was live or deferred and later re-scored?", "kind": "process", "answer_data": [ "scorer_id", "grader_model_id", "rubric_id", "deferred_score_flag", "rescored_at" ] }, { "id": "f-scoring-and-judging-q02", "text": "For each sample, what were the input, output, target, score value, explanation and refusal or scanner flags?", "kind": "measurement", "answer_data": [ "sample_id", "input_ref", "output_ref", "target_ref", "score_value", "refusal_flag" ] }, { "id": "f-scoring-and-judging-q03", "text": "If a model or crowd judge was used, what agreement with gold oracles, inter-rater reliability and bias controls support treating those judgments as evidence?", "kind": "quality", "answer_data": [ "irr_statistic", "oracle_calibration", "judge_bias_notes", "human_subjects_approval" ] }, { "id": "f-scoring-and-judging-q04", "text": "Do sample inputs, outputs or rater notes contain personal data or prohibited content, and what minimization or redaction applies?", "kind": "privacy", "answer_data": [ "personal_data_flag", "redaction_method", "legal_basis", "retention_class" ] } ], "data_elements": [ { "id": "f-scoring-and-judging-data01", "name": "Sample identifier", "description": "Identifier of an evaluated item within the run.", "value_kind": "identifier", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-021" ] }, { "id": "f-scoring-and-judging-data02", "name": "Score value", "description": "Numeric, boolean or categorical score emitted by the scorer.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-021" ] }, { "id": "f-scoring-and-judging-data03", "name": "Grader model identifier", "description": "Model used as judge when scoring is model-graded.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-021" ] }, { "id": "f-scoring-and-judging-data04", "name": "Refusal flag", "description": "Whether the subject refused rather than attempting the task.", "value_kind": "boolean", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-021" ] } ], "artifacts": [ { "id": "f-scoring-and-judging-artifact01", "name": "Sample score table", "description": "Per-sample inputs, outputs, targets, scores, explanations and scanner flags, possibly reduced across epochs.", "media_or_form": [ "tabular records", "EvalSample collection" ], "serial": true, "identity_strategy": "Run master identifier plus sample identifier; re-score produces a new score serial, not a new sample identity.", "source_refs": [ "SRC-021" ] } ], "inline_only_rationale": null } ] }, { "id": "l-evidence-integrity", "name": "Evidence Integrity and Provenance", "description": "Packaging, sealing and lineage that make the evidence transferable and auditable.", "source_refs": [ "SRC-002", "SRC-007", "SRC-008", "SRC-004" ], "findings": [ { "id": "f-evidence-package", "name": "Evidence package and integrity", "description": "Bundling, manifesting, digesting and sealing of evaluation evidence so it can be transferred to auditors or authorities without loss of integrity.", "source_refs": [ "SRC-002", "SRC-008", "SRC-006", "SRC-004" ], "questions": [ { "id": "q-evidence-manifest", "text": "Which items constitute the sealed evidence package for this evaluation, and how is completeness asserted?", "kind": "composition", "answer_data": [ "Manifest entry list", "Completeness assertion", "Known omission list" ] }, { "id": "q-integrity-mechanism", "text": "Which digest, signature or timestamping mechanism protects the package, and who holds the signing authority?", "kind": "security", "answer_data": [ "Digest algorithm", "Signature and signatory role", "Seal event time" ] }, { "id": "q-tamper-detection", "text": "How is a later alteration of any packaged artefact detected and reported?", "kind": "validation", "answer_data": [ "Verification procedure and cadence", "Mismatch handling rule", "Verification log reference" ] }, { "id": "q-transfer-form", "text": "In what neutral form is the package transferred to an auditor or authority without loss of meaning?", "kind": "interoperability", "answer_data": [ "Exchange form", "Vocabulary or schema reference", "Resolution rules for external references" ] } ], "data_elements": [ { "id": "de-package-id", "name": "Evidence package identifier", "description": "Identifier of the sealed package.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-002" ] }, { "id": "de-manifest-entry", "name": "Manifest entry", "description": "One included artefact with its reference and digest.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-008" ] }, { "id": "de-content-digest", "name": "Content digest", "description": "Cryptographic digest over an included artefact.", "value_kind": "identifier", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-008" ] }, { "id": "de-seal-time", "name": "Seal time", "description": "Event time at which the package was sealed.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-007" ] }, { "id": "de-seal-signatory", "name": "Seal signatory", "description": "Person or service that signed the sealed package, with role.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002" ] } ], "artifacts": [ { "id": "a-evidence-package-manifest", "name": "Evidence package manifest", "description": "Signed manifest listing every included artefact with digests, completeness assertion and seal metadata.", "media_or_form": [ "manifest", "signed document", "structured record" ], "serial": true, "identity_strategy": "Package identifier plus version; the manifest digest itself is the integrity anchor.", "source_refs": [ "SRC-002", "SRC-008" ] } ], "inline_only_rationale": null }, { "id": "f-provenance-lineage", "name": "Provenance and lineage graph", "description": "Graph linking datasets, subject artefacts, runs, scorers, results, findings and decisions to responsible agents and times, using entity/activity/agent semantics.", "source_refs": [ "SRC-007", "SRC-011", "SRC-008", "SRC-001" ], "questions": [ { "id": "q-lineage-edges", "text": "Which generation, derivation, usage and attribution edges must be recorded between evaluation entities?", "kind": "relationship", "answer_data": [ "Generation edges", "Derivation edges", "Usage edges" ] }, { "id": "q-agent-attribution", "text": "Which agent — person, organisation or software — is recorded as responsible for each activity and artefact?", "kind": "ownership", "answer_data": [ "Agent identifier and type", "Association per activity", "Attribution per entity" ] }, { "id": "q-activity-timing", "text": "How are activity start and end times recorded and reconciled with the time the record reached the evaluation store?", "kind": "temporal", "answer_data": [ "Activity start time", "Activity end time", "Ingestion time and clock source" ] }, { "id": "q-cross-system-lineage", "text": "How is lineage preserved when datasets, models or scorers are governed in external systems?", "kind": "interoperability", "answer_data": [ "External identifier scheme", "Resolution endpoint or policy", "Snapshot fallback when resolution fails" ] } ], "data_elements": [ { "id": "de-prov-entity", "name": "Provenance entity reference", "description": "A dataset, artefact, run output or report participating in lineage.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-007" ] }, { "id": "de-prov-activity", "name": "Provenance activity reference", "description": "An execution, scoring, review or decision activity.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-007" ] }, { "id": "de-prov-agent", "name": "Provenance agent reference", "description": "Responsible person, organisation or software agent.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-007" ] }, { "id": "de-was-derived-from", "name": "Derivation edge", "description": "Link asserting that one entity was derived from another.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-007" ] }, { "id": "de-observation-time", "name": "Observation or ingestion time", "description": "When the record was observed or ingested by the evaluation store, distinct from event time.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-007", "SRC-011" ] } ], "artifacts": [ { "id": "a-provenance-graph", "name": "Evaluation provenance graph", "description": "Format-neutral lineage graph over evaluation entities, activities and agents.", "media_or_form": [ "graph record", "linked-data descriptor", "structured record" ], "serial": false, "identity_strategy": "Governed IRIs for entities, activities and agents where available; otherwise Dimension-minted identifiers recorded with the minting agent and time.", "source_refs": [ "SRC-007", "SRC-008" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "b-interpretation-and-decision", "name": "Interpretation, Findings and Decision", "description": "Turning results into findings with declared validity limits, and into authorised decisions with lifecycle control.", "rationale": "MEASURE 3.1 requires tracking of existing, unanticipated and emergent risks; MEASURE 2.13 requires review of TEVV effectiveness; Annex IV requires dated and signed test reports; the GPAI Code requires an acceptance determination before proceeding; Article 9(2) makes risk management a continuous iterative process and Article 72 requires post-market monitoring.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-006" ], "layers": [ { "id": "l-findings-and-validity", "name": "Findings and Validity", "description": "Atomic findings and the declared threats to their validity.", "source_refs": [ "SRC-001", "SRC-005", "SRC-010", "SRC-013" ], "findings": [ { "id": "f-finding-record", "name": "Evaluation finding record", "description": "An atomic, separately citable finding derived from results: what was observed, its severity and likelihood, the evidence behind it, and its state.", "source_refs": [ "SRC-001", "SRC-004", "SRC-002", "SRC-006" ], "questions": [ { "id": "q-finding-identity", "text": "How is a single finding identified and cited independently of the report that first presented it?", "kind": "identity", "answer_data": [ "Finding identifier", "Originating evaluation reference", "Citation form" ] }, { "id": "q-finding-evidence-link", "text": "Which specific results, items and traces substantiate the finding?", "kind": "evidence", "answer_data": [ "Result references", "Representative item references", "Reproduction instruction" ] }, { "id": "q-severity-classification", "text": "Which severity and likelihood scale classifies the finding, and who assigns it?", "kind": "classification", "answer_data": [ "Severity code and scale reference", "Likelihood code", "Assigning role" ] }, { "id": "q-finding-state", "text": "What states can a finding occupy — open, accepted, mitigated, disputed, withdrawn — and what moves it between them?", "kind": "state", "answer_data": [ "Current state", "Permitted transitions", "Transition event time and actor" ] } ], "data_elements": [ { "id": "de-finding-id", "name": "Finding identifier", "description": "Stable, citable identifier independent of any report.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-006" ] }, { "id": "de-finding-statement", "name": "Finding statement", "description": "The observation expressed so it can be confirmed or refuted.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "de-finding-evidence-ref", "name": "Finding evidence reference", "description": "Results, items and traces substantiating the finding.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-011" ] }, { "id": "de-severity", "name": "Severity", "description": "Severity classification on a declared scale.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-002" ] }, { "id": "de-finding-state", "name": "Finding state", "description": "Lifecycle state of the finding.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-006" ] }, { "id": "de-linked-risk-ref", "name": "Linked risk reference", "description": "Reference to a risk object owned by the risk model, not defined here.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001", "SRC-002" ] } ], "artifacts": [ { "id": "a-finding-record", "name": "Finding record", "description": "Citable record of one finding with evidence links, severity, state and transition history.", "media_or_form": [ "structured record", "register entry" ], "serial": true, "identity_strategy": "Finding identifier minted in the evaluation records system, monotonic and never reused after withdrawal.", "source_refs": [ "SRC-001", "SRC-006" ] } ], "inline_only_rationale": null }, { "id": "f-validity-threats", "name": "Validity threats and declared limitations", "description": "Threats to construct, internal and external validity, benchmark saturation or gaming, disclosure of inconclusive results, and periodic review of the measurement approach itself.", "source_refs": [ "SRC-001", "SRC-005", "SRC-010", "SRC-013" ], "questions": [ { "id": "q-validity-threat-inventory", "text": "Which threats to construct, internal and external validity are declared for this evaluation?", "kind": "quality", "answer_data": [ "Threat list by validity type", "Effect on inference", "Mitigation attempted" ] }, { "id": "q-benchmark-saturation", "text": "How is benchmark saturation, gaming or optimisation against the test set assessed and disclosed?", "kind": "validation", "answer_data": [ "Saturation indicator", "Headroom estimate", "Test-set retirement decision" ] }, { "id": "q-negative-result-disclosure", "text": "How are inconclusive or negative results recorded so they cannot be silently dropped?", "kind": "evidence", "answer_data": [ "Inconclusive result register entry", "Reason for inconclusiveness", "Disclosure destination" ] }, { "id": "q-measurement-effectiveness-review", "text": "How and when is the effectiveness of the measurement approach itself reviewed?", "kind": "process", "answer_data": [ "Review cadence", "Reviewer role", "Review outcome and change record" ] } ], "data_elements": [ { "id": "de-validity-threat", "name": "Validity threat", "description": "A declared threat with its validity type.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-001", "SRC-005" ] }, { "id": "de-threat-impact", "name": "Threat impact on inference", "description": "What the threat prevents the results from showing.", "value_kind": "text", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-001", "SRC-010" ] }, { "id": "de-inconclusive-flag", "name": "Inconclusive result indicator", "description": "Marks a result as inconclusive rather than passing or failing.", "value_kind": "boolean", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "de-method-review-time", "name": "Measurement method review time", "description": "Event time of the most recent review of measurement effectiveness.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001", "SRC-006" ] } ], "artifacts": [ { "id": "a-limitations-statement", "name": "Limitations and validity statement", "description": "Declared validity threats, saturation assessment and inconclusive results, published with every audience tier.", "media_or_form": [ "document", "structured record" ], "serial": false, "identity_strategy": "Bound to the evaluation instance identifier and report version.", "source_refs": [ "SRC-001", "SRC-010" ] } ], "inline_only_rationale": null } ] }, { "id": "l-decision-and-lifecycle", "name": "Decision, Waivers and Lifecycle", "description": "The authorised decision that consumes the evaluation, any conditional release, and the states and triggers that govern repetition.", "source_refs": [ "SRC-002", "SRC-004", "SRC-006", "SRC-001" ], "findings": [ { "id": "f-approval-and-waiver", "name": "Approval decision, conditions and waivers", "description": "The authorised approve, approve-with-conditions, reject or defer decision, its signatories, the evidence version reviewed, the authorised scope, and any time-bounded waiver with compensating controls.", "source_refs": [ "SRC-002", "SRC-004", "SRC-006", "SRC-001" ], "questions": [ { "id": "q-decision-authority", "text": "Who holds the authority to approve or reject on the basis of this evaluation, and under which delegation?", "kind": "authority", "answer_data": [ "Authorised role", "Delegation instrument reference", "Escalation level required" ] }, { "id": "q-decision-record", "text": "What decision was taken, on which sealed evidence version, and with what conditions attached?", "kind": "decision", "answer_data": [ "Decision outcome code", "Evidence package version reference", "Attached conditions" ] }, { "id": "q-decision-scope", "text": "Which deployment scope, markets, populations and time window does the authorisation cover?", "kind": "constraint", "answer_data": [ "Authorised deployment scope", "Geographic or market scope", "Authorisation expiry date" ] }, { "id": "q-waiver-basis", "text": "On what documented basis may release proceed when an acceptance criterion is not met, and when does that waiver expire?", "kind": "exception", "answer_data": [ "Unmet criterion reference", "Compensating controls", "Waiver expiry time and approver" ] } ], "data_elements": [ { "id": "de-decision-outcome", "name": "Decision outcome", "description": "Approve, approve with conditions, reject or defer.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-006" ] }, { "id": "de-decision-time", "name": "Decision event time", "description": "Event time at which the decision was signed.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-002" ] }, { "id": "de-signatory-ref", "name": "Decision signatory", "description": "Responsible person or persons signing the decision, with role.", "value_kind": "reference", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002" ] }, { "id": "de-evidence-version-ref", "name": "Reviewed evidence version", "description": "Exact sealed evidence package version the decision relies on.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-004" ] }, { "id": "de-authorised-scope", "name": "Authorised scope", "description": "Deployment, market, population and time bounds of the authorisation.", "value_kind": "object", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-004" ] }, { "id": "de-waiver-expiry-time", "name": "Waiver expiry time", "description": "When a conditional release ceases to be authorised.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004", "SRC-006" ] }, { "id": "de-compensating-control", "name": "Compensating control", "description": "Control or usage restriction mandated while a waiver is in force.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004", "SRC-002" ] } ], "artifacts": [ { "id": "a-approval-decision-record", "name": "Approval and waiver decision record", "description": "Dated, signed decision binding outcome, conditions, waivers and authorised scope to a specific evidence version.", "media_or_form": [ "signed document", "structured record" ], "serial": true, "identity_strategy": "Decision identifier from the quality or governance master system; immutable, superseded only by a new decision that references it.", "source_refs": [ "SRC-002", "SRC-006" ] } ], "inline_only_rationale": null }, { "id": "f-lifecycle-and-triggers", "name": "Evaluation lifecycle and re-evaluation triggers", "description": "States an evaluation instance passes through, the conditions requiring repetition or invalidation, and the cadence of periodic re-evaluation and in-production measurement.", "source_refs": [ "SRC-002", "SRC-001", "SRC-004", "SRC-006" ], "questions": [ { "id": "q-lifecycle-states", "text": "Which states does an evaluation instance pass through from planned to archived, and which transitions are irreversible?", "kind": "lifecycle", "answer_data": [ "State list", "Permitted transition map", "Irreversible transition list" ] }, { "id": "q-reevaluation-trigger", "text": "Which changes to model, data, threat landscape, scaffolding or deployment context oblige re-evaluation?", "kind": "event", "answer_data": [ "Trigger condition list", "Detection mechanism", "Response deadline" ] }, { "id": "q-cadence-and-monitoring", "text": "At what cadence is periodic re-evaluation and in-production measurement required?", "kind": "temporal", "answer_data": [ "Review cadence duration", "Next scheduled evaluation date", "Production monitoring frequency" ] }, { "id": "q-invalidation", "text": "Under what conditions is a completed evaluation declared invalid, and how are dependent approvals affected?", "kind": "state", "answer_data": [ "Invalidation reason", "Affected decision references", "Notification obligation" ] } ], "data_elements": [ { "id": "de-eval-state", "name": "Evaluation state", "description": "Current lifecycle state of the evaluation instance.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-006", "SRC-002" ] }, { "id": "de-state-entered-time", "name": "State entry time", "description": "Event time at which the current state was entered.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-007" ] }, { "id": "de-reevaluation-trigger", "name": "Re-evaluation trigger condition", "description": "Condition obliging repetition of the evaluation.", "value_kind": "collection", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-004" ] }, { "id": "de-review-cadence", "name": "Review cadence", "description": "Interval between scheduled re-evaluations.", "value_kind": "duration", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002", "SRC-006" ] }, { "id": "de-invalidation-reason", "name": "Invalidation reason", "description": "Why a completed evaluation was declared invalid.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001", "SRC-006" ] } ], "artifacts": [ { "id": "a-evaluation-state-log", "name": "Evaluation state log", "description": "Append-only log of state transitions, triggers, actors and event times for the evaluation instance.", "media_or_form": [ "append-only log", "structured record" ], "serial": true, "identity_strategy": "Evaluation instance identifier plus monotonic transition sequence.", "source_refs": [ "SRC-002", "SRC-006" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "b-assurance-and-interoperability", "name": "Assurance, Records Governance and Interoperability", "description": "Disclosure to differentiated audiences, external assurance, access and retention of records, and alignment and exchange with external frameworks.", "rationale": "The GPAI Code requires a model report before release and external evaluations; Annex IV and Articles 18-19 impose documentation and log retention; Article 55(1)(d) requires cybersecurity protection; MLPerf divisions define when results may be compared; Croissant and PROV-O supply exchange vocabularies.", "source_refs": [ "SRC-002", "SRC-004", "SRC-008", "SRC-009", "SRC-010" ], "layers": [ { "id": "l-reporting-and-external", "name": "Reporting and External Assurance", "description": "How results are disclosed to each audience and how external evaluators participate.", "source_refs": [ "SRC-004", "SRC-002", "SRC-010", "SRC-009" ], "findings": [ { "id": "f-report-tiering", "name": "Evaluation report and audience tiering", "description": "Report instances derived from the evaluation, tiered by audience, with rules preventing public claims from exceeding internal evidence, and versioning for corrections and retractions.", "source_refs": [ "SRC-004", "SRC-002", "SRC-010", "SRC-001" ], "questions": [ { "id": "q-report-audience-tier", "text": "Which audience tiers exist and what content is withheld or summarised in each?", "kind": "classification", "answer_data": [ "Audience tier code", "Withheld content list and reason", "Summarisation rule" ] }, { "id": "q-report-consistency", "text": "How is it ensured that public claims never exceed what the full internal evidence supports?", "kind": "validation", "answer_data": [ "Consistency check procedure", "Reviewer role", "Check event time" ] }, { "id": "q-report-versioning", "text": "How are report versions, corrections and retractions identified and communicated?", "kind": "lifecycle", "answer_data": [ "Report version", "Superseded report reference", "Notification recipients" ] }, { "id": "q-mandatory-disclosure", "text": "Which report content is mandated by regulation or code, and by what deadline?", "kind": "requirement", "answer_data": [ "Mandated content element", "Legal or code basis", "Submission deadline" ] } ], "data_elements": [ { "id": "de-report-id", "name": "Report identifier", "description": "Identifier of a report instance derived from the evaluation.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-002" ] }, { "id": "de-audience-tier", "name": "Audience tier", "description": "Internal, regulator, downstream deployer or public.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-010" ] }, { "id": "de-report-version", "name": "Report version", "description": "Version label of the report instance.", "value_kind": "text", "cardinality": "1", "required": true, "source_refs": [ "SRC-004" ] }, { "id": "de-withheld-content-reason", "name": "Withheld content reason", "description": "Why content is absent from a given tier.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004", "SRC-002" ] }, { "id": "de-publication-time", "name": "Publication time", "description": "Event time at which the report version was released to its tier.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] } ], "artifacts": [ { "id": "a-evaluation-report", "name": "Tiered evaluation report", "description": "Audience-specific report derived from sealed evidence, including the limitations statement.", "media_or_form": [ "document", "structured record", "published page" ], "serial": true, "identity_strategy": "Report identifier plus version and audience tier; each tier is a distinct addressable instance bound to one evidence version.", "source_refs": [ "SRC-004", "SRC-002" ] }, { "id": "a-model-card-evaluation-section", "name": "Model card evaluation section", "description": "Projection of testing data, factors, metrics and results into the model or system card for downstream deployers.", "media_or_form": [ "structured metadata", "document section" ], "serial": false, "identity_strategy": "Bound to the subject model identifier and the cited evidence version.", "source_refs": [ "SRC-010" ] } ], "inline_only_rationale": null }, { "id": "f-external-evaluation", "name": "External and third-party evaluation", "description": "Arrangements for independent evaluators: access tier, scope, safe harbour, publication rights, exemption basis and handling of dissent.", "source_refs": [ "SRC-004", "SRC-002", "SRC-001", "SRC-009" ], "questions": [ { "id": "q-external-access-tier", "text": "What level of access — black-box, grey-box, weights, fine-tuning or internal tooling — is granted to external evaluators?", "kind": "access", "answer_data": [ "Access tier code", "Granted affordances", "Access window" ] }, { "id": "q-external-scope-contract", "text": "Which contractual terms bound scope, timing, safe harbour and publication rights?", "kind": "constraint", "answer_data": [ "Agreement reference", "Safe harbour terms", "Publication right and embargo" ] }, { "id": "q-external-exemption", "text": "On what basis may an external evaluation be omitted, and how is that basis evidenced?", "kind": "exception", "answer_data": [ "Exemption basis", "Comparator model evidence", "Approving authority" ] }, { "id": "q-dissent-handling", "text": "How is disagreement between internal and external assessments recorded and resolved?", "kind": "process", "answer_data": [ "Dissent record", "Resolution procedure", "Unresolved disagreement disclosure" ] } ], "data_elements": [ { "id": "de-external-party", "name": "External evaluator reference", "description": "Independent organisation or expert performing the evaluation.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004", "SRC-002" ] }, { "id": "de-access-tier", "name": "External access tier", "description": "Depth of access granted to the external evaluator.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004" ] }, { "id": "de-safe-harbour-terms", "name": "Safe harbour terms", "description": "Protections offered to good-faith external testing.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-exemption-basis", "name": "External evaluation exemption basis", "description": "Justification for omitting an external evaluation.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-dissent-record", "name": "Dissent record", "description": "Recorded disagreement between assessments and its resolution.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009", "SRC-001" ] } ], "artifacts": [ { "id": "a-external-evaluation-agreement", "name": "External evaluation agreement and access grant", "description": "Agreement setting scope, access tier, safe harbour, publication rights and reporting obligations for third-party evaluation.", "media_or_form": [ "signed document", "structured record" ], "serial": true, "identity_strategy": "Agreement identifier from the contracts master system, referenced by the evaluation instance.", "source_refs": [ "SRC-004", "SRC-002" ] } ], "inline_only_rationale": null } ] }, { "id": "l-records-governance", "name": "Records Access and Retention", "description": "Who may see evaluation records, how sensitive detail is handled, and how long records live.", "source_refs": [ "SRC-002", "SRC-004", "SRC-012", "SRC-006" ], "findings": [ { "id": "f-access-and-sensitivity", "name": "Access control and hazardous detail handling", "description": "Default access rules per record scope, redaction of uplift-relevant or exploit-enabling detail, handling of personal data in traces, and access auditing.", "source_refs": [ "SRC-004", "SRC-002", "SRC-012", "SRC-001" ], "questions": [ { "id": "q-default-access-rule", "text": "What is the default access rule for each record scope, and which roles hold read or write rights?", "kind": "access", "answer_data": [ "Scope code", "Default rule", "Role-to-right mapping" ] }, { "id": "q-hazardous-detail-handling", "text": "Which uplift-relevant or exploit-enabling details must be redacted, held separately, or shared only with authorities?", "kind": "security", "answer_data": [ "Sensitivity label", "Redaction rule", "Authorised recipient list" ] }, { "id": "q-personal-data-handling", "text": "How are personal data in prompts, traces and rater records minimised, lawfully based and protected?", "kind": "privacy", "answer_data": [ "Personal data inventory", "Lawful basis", "Minimisation or pseudonymisation measure" ] }, { "id": "q-access-audit", "text": "Which access events are logged, and for how long are those logs themselves retained?", "kind": "evidence", "answer_data": [ "Logged event types", "Log fields including actor and purpose", "Log retention period" ] } ], "data_elements": [ { "id": "de-access-scope", "name": "Access scope", "description": "Whether the rule applies at bundle, layer, finding or artefact level.", "value_kind": "code", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-004", "SRC-006" ] }, { "id": "de-sensitivity-label", "name": "Sensitivity label", "description": "Classification driving access and redaction handling.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-012" ] }, { "id": "de-redaction-rule", "name": "Redaction rule", "description": "What must be removed before distribution at a given tier.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004", "SRC-012" ] }, { "id": "de-lawful-basis", "name": "Personal data lawful basis", "description": "Basis for processing personal data appearing in evaluation records.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002", "SRC-001" ] }, { "id": "de-access-log-entry", "name": "Access log entry", "description": "Recorded access event with actor, purpose, scope and event time.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002", "SRC-006" ] } ], "artifacts": [ { "id": "a-access-policy-and-audit-log", "name": "Access policy record and access audit log", "description": "Per-scope access rules with role mappings, plus the append-only log of grants, reads and revocations.", "media_or_form": [ "structured record", "append-only log" ], "serial": true, "identity_strategy": "Policy identifier plus version for rules; monotonic sequence per access event for the log.", "source_refs": [ "SRC-004", "SRC-002" ] } ], "inline_only_rationale": null }, { "id": "f-retention-and-deletion", "name": "Retention, deletion and legal hold", "description": "Retention periods and legal bases per record class, resolution of deletion-versus-evidence conflicts, legal-hold handling, and the tombstone that keeps prior claims auditable after deletion.", "source_refs": [ "SRC-002", "SRC-003", "SRC-006", "SRC-001" ], "questions": [ { "id": "q-retention-period", "text": "What minimum and maximum retention period applies to each evaluation record class, and on what legal basis?", "kind": "retention", "answer_data": [ "Record class", "Minimum retention duration", "Legal or policy basis" ] }, { "id": "q-deletion-conflict", "text": "How are conflicts resolved between data-minimisation deletion duties and evidence-retention duties?", "kind": "exception", "answer_data": [ "Conflict description", "Resolution rule", "Authorising role" ] }, { "id": "q-legal-hold", "text": "How is a legal hold applied, recorded and released across evaluation records?", "kind": "process", "answer_data": [ "Hold scope", "Hold start and release times", "Authorising instrument" ] }, { "id": "q-post-retention-summary", "text": "What summary or digest must survive after primary records are deleted so that prior claims remain auditable?", "kind": "requirement", "answer_data": [ "Tombstone content", "Retained digest", "Tombstone retention period" ] } ], "data_elements": [ { "id": "de-record-class", "name": "Record class", "description": "Category of evaluation record for retention purposes.", "value_kind": "code", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-006" ] }, { "id": "de-retention-minimum", "name": "Minimum retention duration", "description": "Shortest permitted retention for the record class.", "value_kind": "duration", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-003" ] }, { "id": "de-retention-basis", "name": "Retention basis", "description": "Legal or policy basis for the retention period.", "value_kind": "text", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-006" ] }, { "id": "de-legal-hold-flag", "name": "Legal hold indicator", "description": "Whether deletion is currently suspended for the record.", "value_kind": "boolean", "cardinality": "1", "required": true, "source_refs": [ "SRC-006", "SRC-002" ] }, { "id": "de-tombstone-digest", "name": "Tombstone digest", "description": "Digest of deleted material retained to keep prior claims verifiable.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-008", "SRC-002" ] } ], "artifacts": [ { "id": "a-retention-schedule-and-tombstone", "name": "Retention schedule and deletion tombstone", "description": "Per-class retention schedule with bases, plus tombstones recording identifier, class, deletion time, authoriser and digest of deleted material.", "media_or_form": [ "structured record", "register entry" ], "serial": true, "identity_strategy": "Record-class key for the schedule; deleted-record identifier reused as the tombstone key so references never dangle silently.", "source_refs": [ "SRC-002", "SRC-006" ] } ], "inline_only_rationale": null } ] }, { "id": "l-interoperability", "name": "Alignment and Exchange Interoperability", "description": "Declared alignments to external frameworks and the identifier and comparability conventions enabling exchange.", "source_refs": [ "SRC-002", "SRC-006", "SRC-008", "SRC-009" ], "findings": [ { "id": "f-alignment-and-exchange", "name": "Standards alignment, conformance limits and exchange interoperability", "description": "Which external framework clauses each record claims to address, how alignment is kept distinct from certified conformance, which identifier schemes are used, and when results from different organisations or rounds may be compared.", "source_refs": [ "SRC-002", "SRC-001", "SRC-006", "SRC-008", "SRC-009", "SRC-007" ], "questions": [ { "id": "q-alignment-declaration", "text": "Which external framework clauses does each evaluation record claim to address, and what evidence supports the claim?", "kind": "interoperability", "answer_data": [ "Alignment target and clause", "Supporting evidence reference", "Alignment assertion date" ] }, { "id": "q-conformance-claim-limit", "text": "How are alignment statements prevented from being read as certified conformance?", "kind": "constraint", "answer_data": [ "Conformance status code", "Required disclaimer wording", "Attached assessment record if any" ] }, { "id": "q-identifier-scheme", "text": "Which identifier schemes and namespaces are used for subjects, datasets, metrics and reports across organisations?", "kind": "identity", "answer_data": [ "Scheme per entity class", "Namespace authority", "Resolution mechanism" ] }, { "id": "q-comparability-condition", "text": "Under what conditions may results from two organisations or two rounds be compared without misleading readers?", "kind": "measurement", "answer_data": [ "Division or profile equivalence", "Held-constant factors", "Explicit non-comparability warning" ] }, { "id": "q-framework-conflict", "text": "Where two frameworks impose conflicting requirements, how is the conflict recorded and resolved?", "kind": "decision", "answer_data": [ "Conflict statement", "Precedence rule applied", "Deciding authority" ] } ], "data_elements": [ { "id": "de-alignment-target", "name": "Alignment target", "description": "External framework or clause the record claims to address.", "value_kind": "code", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-002", "SRC-001", "SRC-006" ] }, { "id": "de-conformance-status", "name": "Conformance status", "description": "Aligned, partially aligned, or independently assessed; never asserted as certified without an attached assessment record.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-006", "SRC-002" ] }, { "id": "de-identifier-scheme", "name": "Identifier scheme", "description": "Scheme governing identifiers for each exchanged entity class.", "value_kind": "code", "cardinality": "1..n", "required": true, "source_refs": [ "SRC-007", "SRC-008" ] }, { "id": "de-comparability-condition", "name": "Comparability condition", "description": "Conditions under which two result sets may be compared.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-known-conflict", "name": "Known framework conflict", "description": "Recorded conflict between framework requirements and its resolution.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002", "SRC-004" ] } ], "artifacts": [ { "id": "a-alignment-matrix", "name": "Alignment and comparability matrix", "description": "Matrix mapping evaluation records to external framework clauses with conformance status, evidence links, comparability conditions and recorded conflicts.", "media_or_form": [ "tabular record", "structured record" ], "serial": false, "identity_strategy": "Evaluation instance identifier plus alignment-target key; assertions are dated and withdrawable with reason.", "source_refs": [ "SRC-002", "SRC-006", "SRC-009" ] } ], "inline_only_rationale": null } ] } ] } ] }, "functions": [ { "id": "fn-register-evaluation-protocol", "name": "Register evaluation protocol", "description": "Fix the method, task and dataset bindings, metrics, disaggregation factors and acceptance thresholds as an immutable pre-registered protocol version before any result is observed.", "inputs": [ "Subject binding record", "Dataset binding manifest", "Metric definition records", "Acceptance criteria", "Mandate basis" ], "outputs": [ "Registered evaluation protocol version", "Pre-registration event time" ], "preconditions": [ "Subject identifier resolves in its master system", "Dataset version and split are pinned", "Thresholds are set by an authorised role" ], "effects": [ "Protocol becomes the immutable baseline for the instance", "Later changes require a dated, authorised change record", "Evaluation instance enters the planned state" ], "source_refs": [ "SRC-002", "SRC-004", "SRC-001", "SRC-005" ] }, { "id": "fn-bind-evaluation-subject", "name": "Bind evaluation subject", "description": "Pin the model or system version, configuration and digest that the evaluation will measure, including snapshotting a hosted endpoint.", "inputs": [ "Model artefact reference", "Version label", "Inference configuration" ], "outputs": [ "Subject binding record", "Endpoint pin or content digest" ], "preconditions": [ "An authoritative artefact identifier exists in the model artefact master system" ], "effects": [ "Results become attributable to a fixed subject", "Any detected subject change invalidates in-flight runs" ], "source_refs": [ "SRC-010", "SRC-011", "SRC-009", "SRC-002" ] }, { "id": "fn-bind-evaluation-dataset", "name": "Bind evaluation dataset and verify contamination", "description": "Pin dataset versions, splits and checksums by reference and record the train/test contamination check.", "inputs": [ "Dataset registry reference", "Version or snapshot label", "Split designation" ], "outputs": [ "Dataset binding manifest", "Contamination check result" ], "preconditions": [ "Dataset entry is governed in the benchmark dataset model", "Checksums are retrievable" ], "effects": [ "Dataset version is locked to the protocol", "Holdout exposure counter is incremented" ], "source_refs": [ "SRC-008", "SRC-005", "SRC-002" ] }, { "id": "fn-execute-evaluation-run", "name": "Execute evaluation run", "description": "Run the registered protocol against the bound subject and dataset, capturing configuration, environment, timing, usage and item-level traces.", "inputs": [ "Registered protocol version", "Subject binding record", "Dataset binding manifest", "Environment manifest" ], "outputs": [ "Evaluation run record", "Per-item trace log", "Raw item scores" ], "preconditions": [ "Protocol is registered and not superseded", "Execution service can capture revision and package versions" ], "effects": [ "Run record is appended immutably", "Token, latency and energy quantities are recorded", "Run start and end event times are recorded separately from ingestion time" ], "source_refs": [ "SRC-011", "SRC-009", "SRC-014", "SRC-001" ] }, { "id": "fn-compute-results", "name": "Compute and aggregate results", "description": "Apply scorers and reducers to item scores to produce disaggregated and headline values with uncertainty estimates.", "inputs": [ "Per-item scores", "Metric definitions", "Disaggregation factors", "Aggregation and reducer rules" ], "outputs": [ "Result set", "Disaggregated result table", "Uncertainty statement" ], "preconditions": [ "Scorer implementation and version are recorded", "Subgroup sample sizes are known" ], "effects": [ "Results are marked provisional until validated", "Subgroup values below the minimum sample size are suppressed" ], "source_refs": [ "SRC-011", "SRC-014", "SRC-005", "SRC-009" ] }, { "id": "fn-rescore-result", "name": "Re-score and supersede a result", "description": "Recompute a score or aggregate while preserving the original value, the scorer version and the reason for change.", "inputs": [ "Existing result or item score", "New scorer version", "Change reason" ], "outputs": [ "New result version", "Supersession link", "Recomputed dependent aggregates" ], "preconditions": [ "Original value and scorer metadata are retained", "Change is attributable to a named actor" ], "effects": [ "Prior value is preserved, never overwritten", "Dependent aggregates and published reports are flagged for review" ], "source_refs": [ "SRC-011", "SRC-001", "SRC-002" ] }, { "id": "fn-raise-finding", "name": "Raise evaluation finding", "description": "Create a citable finding from results, with evidence links, severity, likelihood and initial state.", "inputs": [ "Result references", "Item and trace references", "Severity scale", "Assessor judgement" ], "outputs": [ "Finding record", "Optional linked risk reference" ], "preconditions": [ "Evidence references resolve", "Assessor role is authorised to classify severity" ], "effects": [ "Finding enters the open state", "Hazardous-detail findings receive a restricted sensitivity label" ], "source_refs": [ "SRC-001", "SRC-004", "SRC-012", "SRC-006" ] }, { "id": "fn-seal-evidence-package", "name": "Seal evidence package", "description": "Assemble, digest and sign the set of artefacts constituting the evidence for a decision or a submission.", "inputs": [ "Artefact references", "Completeness assertion", "Signing authority" ], "outputs": [ "Evidence package manifest", "Content digests", "Seal event time" ], "preconditions": [ "All referenced artefacts are present and retrievable", "Signatory holds the recorded role" ], "effects": [ "Package becomes tamper-evident", "Any later digest mismatch suspends dependent conformance claims" ], "source_refs": [ "SRC-002", "SRC-008", "SRC-006", "SRC-004" ] }, { "id": "fn-decide-approval", "name": "Decide approval or rejection", "description": "Take and sign the authorised decision that consumes the evaluation, bound to a specific sealed evidence version and authorised scope.", "inputs": [ "Sealed evidence package version", "Findings", "Acceptance criteria", "Delegated authority" ], "outputs": [ "Approval and waiver decision record", "Authorised scope", "Attached conditions" ], "preconditions": [ "All mandatory criteria have been evaluated or waived", "Signatory is not the sole executor of the runs relied on" ], "effects": [ "Release gate opens, closes or defers", "Decision expiry and conditions become monitorable obligations" ], "source_refs": [ "SRC-002", "SRC-004", "SRC-006" ] }, { "id": "fn-issue-waiver", "name": "Issue time-bounded waiver", "description": "Authorise release despite an unmet acceptance criterion, with compensating controls and an explicit expiry.", "inputs": [ "Unmet criterion reference", "Proposed compensating controls", "Escalated approver" ], "outputs": [ "Waiver record", "Waiver expiry time" ], "preconditions": [ "Required escalation level is satisfied", "Compensating controls are implementable and monitorable" ], "effects": [ "Conditional release is permitted until expiry", "Expiry without renewal reverts the authorisation" ], "source_refs": [ "SRC-004", "SRC-006", "SRC-002" ] }, { "id": "fn-publish-tiered-report", "name": "Publish tiered evaluation report", "description": "Derive an audience-specific report from sealed evidence, applying redaction rules and a consistency check against the full internal record.", "inputs": [ "Sealed evidence package", "Audience tier", "Redaction rules", "Limitations statement" ], "outputs": [ "Tiered evaluation report version", "Model card evaluation section" ], "preconditions": [ "Consistency check confirms claims do not exceed evidence", "Sensitivity labels have been applied" ], "effects": [ "Public claim is bound to a named evidence version", "Corrections require a new report version and notification" ], "source_refs": [ "SRC-004", "SRC-002", "SRC-010" ] }, { "id": "fn-grant-scoped-access", "name": "Grant scoped evaluation access", "description": "Issue time-bounded access to evaluation records at a declared scope and tier, with purpose recorded and access logged.", "inputs": [ "Requester identity", "Requested scope and tier", "Stated purpose", "Agreement reference for external parties" ], "outputs": [ "Access grant", "Access audit log entry" ], "preconditions": [ "Sensitivity label and lawful basis have been checked", "Granting authority is recorded" ], "effects": [ "Access expires automatically at the recorded end time", "Every subsequent read of a restricted scope is logged" ], "source_refs": [ "SRC-004", "SRC-002", "SRC-006" ] }, { "id": "fn-trigger-reevaluation", "name": "Trigger re-evaluation", "description": "Open a new evaluation instance in response to a qualifying change, incident, drift alarm or scheduled review, linked to the prior instance.", "inputs": [ "Trigger condition", "Prior evaluation instance reference", "Trigger event time" ], "outputs": [ "New evaluation instance", "Link to superseded instance" ], "preconditions": [ "Trigger is classified against the declared trigger list" ], "effects": [ "Prior instance is marked under review or superseded", "Dependent approvals are re-examined against their expiry" ], "source_refs": [ "SRC-002", "SRC-001", "SRC-004", "SRC-006" ] }, { "id": "fn-apply-retention-action", "name": "Apply retention, hold or deletion action", "description": "Enforce the retention schedule for a record class, apply or release a legal hold, or delete with a tombstone.", "inputs": [ "Record class", "Retention schedule", "Legal hold state", "Authorising role" ], "outputs": [ "Retention action record", "Deletion tombstone with digest" ], "preconditions": [ "No active legal hold or statutory period blocks the action", "Authoriser is recorded" ], "effects": [ "Primary records are removed while prior claims remain verifiable via the tombstone digest", "References to deleted records resolve to the tombstone rather than dangling" ], "source_refs": [ "SRC-002", "SRC-003", "SRC-006", "SRC-008" ] }, { "id": "fn-scan-evaluation-integrity", "name": "Scan evaluation integrity", "description": "Review transcripts for refusals, evaluation awareness, leakage, harness failure and other threats to validity.", "inputs": [ "run log", "scanner or contamination-check configuration" ], "outputs": [ "contamination and leakage report", "items excluded or flagged" ], "preconditions": [ "Run log is readable" ], "effects": [ "Validity exceptions are attached to samples or the run" ], "source_refs": [ "SRC-018", "SRC-021" ] }, { "id": "fn-package-reproducibility-evidence", "name": "Package reproducibility evidence", "description": "Assemble pins, hashes and unreleased-component list so another party can attempt a re-run or audit.", "inputs": [ "campaign", "run logs", "dataset binding", "task spec", "subject binding" ], "outputs": [ "reproducibility package" ], "preconditions": [ "Content hashes for released components exist" ], "effects": [ "A serial package is linked to the campaign and runs" ], "source_refs": [ "SRC-020", "SRC-021" ] } ], "composition": [ { "target": "WM-AI-009 — Benchmark dataset", "relation": "REFERENCE", "purpose": "Version-pinned binding of evaluation tasks, splits, ground truth and checksums. This model stores the reference, version, split designation and contamination check only; dataset construction, licensing and internal structure remain there.", "required": true, "source_refs": [ "SRC-008", "SRC-005", "SRC-002" ] }, { "target": "WM-SFT-004 — AI or software model artefact", "relation": "REFERENCE", "purpose": "Identifies the evaluated subject artefact, its version and digest. This model never mints artefact identifiers and defers packaging and build provenance to the artefact model.", "required": true, "source_refs": [ "SRC-010", "SRC-011", "SRC-002" ] }, { "target": "NIST AI RMF 1.0 MEASURE function (NIST AI 100-1)", "relation": "ALIGN", "purpose": "Structuring alignment for documented TEVV test sets, metrics and tools, metric selection, independent assessment, deployment-condition measurement, fairness evaluation and measurement-effectiveness review. Voluntary and non-certifiable.", "required": false, "source_refs": [ "SRC-001" ] }, { "target": "Regulation (EU) 2024/1689 Articles 9, 15, 19, 55, 60, 72 and Annex IV", "relation": "ALIGN", "purpose": "Legal alignment for prior-defined metrics and probabilistic thresholds, documented adversarial testing, dated and signed test logs and reports, minimum log retention, real-world testing and post-market re-evaluation.", "required": false, "source_refs": [ "SRC-002", "SRC-003" ] }, { "target": "General-Purpose AI Code of Practice — Safety and Security chapter", "relation": "ALIGN", "purpose": "Voluntary compliance route supplying evaluation triggers, risk tiers with safety margins, acceptance determination before proceeding, external evaluation expectations and the model report before release.", "required": false, "source_refs": [ "SRC-004" ] }, { "target": "ISO/IEC 42001:2023 clause 9 and Annex A", "relation": "ALIGN", "purpose": "Alignment of monitoring, measurement, internal audit and management review duties to the evaluation record set. Asserted at clause-title level only because the normative text is paywalled.", "required": false, "source_refs": [ "SRC-006" ] }, { "target": "W3C PROV-O", "relation": "MIX-IN", "purpose": "Supplies entity, activity and agent classes with generation, derivation, attribution, association and timing properties for the evaluation lineage graph rather than redefining provenance semantics locally.", "required": false, "source_refs": [ "SRC-007" ] }, { "target": "MLCommons Croissant 1.0", "relation": "ALIGN", "purpose": "Interoperable description of referenced evaluation data, including semantic versioning, sha256 checksums and RecordSet or Field structure used for integrity verification of bound datasets.", "required": false, "source_refs": [ "SRC-008" ] }, { "target": "MLPerf Inference Rules", "relation": "ALIGN", "purpose": "Reference profile for comparability: divisions and categories, scenario-specific metrics, quality targets, reporting precision, bounded non-determinism, mandatory replicability and an independent audit process.", "required": false, "source_refs": [ "SRC-009" ] }, { "target": "Hugging Face model card evaluation section", "relation": "ALIGN", "purpose": "Projection target for downstream-deployer disclosure of testing data, factors, metrics, results, limitations and environmental impact.", "required": false, "source_refs": [ "SRC-010" ] }, { "target": "NIST AI 100-2 E2025 adversarial machine learning taxonomy", "relation": "ALIGN", "purpose": "Vocabulary for attacker goals, capabilities and knowledge, lifecycle stages of attack, and attack classes including evasion, poisoning, privacy, indirect prompt injection and misaligned outputs.", "required": false, "source_refs": [ "SRC-012" ] }, { "target": "ISO/IEC 25059:2023 quality model for AI systems", "relation": "ALIGN", "purpose": "Vocabulary alignment for the quality characteristics a metric claims to measure. Treated as provisional because the edition is under revision and the text is paywalled.", "required": false, "source_refs": [ "SRC-013" ] }, { "target": "Candidate sibling — AI incident and post-market monitoring model", "relation": "REFERENCE", "purpose": "Supplies re-evaluation triggers and receives findings. Not registered in the relations file, so this link is a proposal marked as a gap rather than a canonical relation.", "required": false, "source_refs": [ "SRC-002", "SRC-004" ] }, { "target": "Candidate sibling — AI risk register and treatment model", "relation": "REFERENCE", "purpose": "Receives evaluation findings as measurement evidence against identified risks and owns treatment and residual-risk acceptance. Not yet registered; proposed only.", "required": false, "source_refs": [ "SRC-001", "SRC-002" ] } ], "serviceLayers": { "dimension": { "owner_package_requirements": [ "Name an accountable owner for the evaluation record set and a separate approver role; the approver may not be the sole executor of the runs relied on.", "Declare which external master systems are authoritative for model artefact identifiers (WM-SFT-004) and dataset identifiers (WM-AI-009); this model mints no such identifiers and must resolve them by reference.", "Publish the Dimension's own severity scale, acceptance-threshold catalogue and independence-level codes, since no cited source fixes these values.", "Publish a retention schedule per record class with its jurisdictional basis, and a sensitivity-label scheme covering hazardous-capability detail and personal data in traces.", "State which external frameworks the Dimension declares alignment with, and confirm that no certified-conformance language is used without an attached assessment record." ], "namespace_guidance": "Use one namespace per accountable provider organisation, structured as /eval///. Local model IDs are stable lower-kebab-case. Instance identifiers must resolve within the owning Dimension, must be opaque with respect to time, and must never encode a date, a benchmark name or a score as identity. Metric identifiers live in a governed metric catalogue; a change in metric semantics forces a new identifier rather than a new version of the old one.", "registry_links": [ "vr.wm-ai-003 — this model entry (WM-AI-003, nav path NAV.INF.AI.EVL)", "WM-AI-009 — benchmark dataset entry, referenced for task and dataset binding", "WM-SFT-004 — parent model artefact entry, referenced as the evaluation subject", "planning/VERCY-MODEL-RELATIONS.csv — typed-edge relation records for this model" ] }, "canon_and_patch": { "canonicalization_rules": [ "The canonical form is the format-neutral record graph. JSON, YAML, Markdown, HTML, Git objects, MCP resources and MongoDB documents are projections and must round-trip without semantic loss.", "All time values canonicalise to RFC 3339 with seconds and an explicit offset or Z; durations canonicalise to ISO 8601 durations and are never substituted for an offset.", "Numeric result values canonicalise together with their declared precision and rounding convention; a five-significant-figure round-to-even profile is acceptable where the adopting Dimension declares it.", "Identifiers canonicalise to trimmed lower-case strings compared by exact match; no identifier is derived from a date, a file name or a display label.", "Sensitivity labels, state codes and severity codes canonicalise against the Dimension's published code lists, with unmapped values rejected rather than silently coerced." ], "patch_rules": [ "Registered protocols, run records, sealed evidence packages and signed decisions are append-only; corrections create a new version that references the superseded record.", "A score change must retain the original value, the new value, the scorer version, the actor and the reason, and must recompute every dependent aggregate and flag affected published reports.", "Any change to an acceptance threshold after results are known requires a change record naming the authoriser and stating explicitly that results were already observed.", "State transitions are recorded as appended entries carrying actor and event time; the current state is a derived projection, not an editable field.", "Deletion never removes a reference target silently: it leaves a tombstone carrying the identifier, class, deletion time, authoriser and content digest." ], "compatibility_rules": [ "Adding optional data elements, new metric definitions or new alignment declarations is backward compatible.", "Changing a metric definition, aggregation rule or threshold semantics is breaking and requires a new metric or threshold identifier, not an in-place redefinition.", "Removing or narrowing a declared disaggregation factor is breaking for fairness claims and invalidates comparability with prior rounds; the break must be disclosed in the affected reports.", "Alignment declarations may be withdrawn, but withdrawal must be dated, reasoned and reflected in every report version that relied on them.", "Consumers must tolerate unknown optional fields and must fail loudly on unknown values in code-list fields that drive access, retention or conformance behaviour." ] }, "artifact_rules": { "identity_priority": [ "Authoritative master-system identifier issued by the system of record for the entity — the model artefact registry for the evaluation subject, the dataset registry for evaluation data, and the quality or records management system for the evaluation instance, decision and finding.", "Governed global identifier or IRI where no master-system identifier exists — for example a PROV-O or JSON-LD IRI, a DOI, or a registry-governed URI under a declared namespace authority.", "UUID or ULID minted by the adopting Dimension, recorded together with the minting agent and the minting event time.", "A date, a run timestamp, a benchmark name, a file name or a metric score is never an identifier; such values may only be recorded as attributes." ], "timestamp_rule": "All time values use RFC 3339 with seconds and an explicit UTC offset or Z, for example 2026-08-26T09:15:00Z or 2026-08-26T11:15:00+02:00. Event time (run start, run end, trigger occurrence, seal, decision signature, publication) is recorded separately from observation or ingestion time (when the record reached the evaluation store) whenever the two can differ; where both exist, neither may be inferred from the other. Durations use ISO 8601 durations and never substitute for an offset, and the clock source is recorded where runs span systems.", "serial_naming_rule": "Serial artefacts — runs, per-item trace segments, findings, report versions, access-log entries and state transitions — are named :::, where the sequence is an opaque counter or ULID scoped to the parent. Human-readable timestamps may appear in file names purely as ordering hints and carry no identity; sequence numbers are never reused, including after withdrawal or deletion.", "integrity_rule": "Every artefact carries a content digest, such as SHA-256, recorded in its manifest entry. Sealed evidence packages additionally carry a signature and a seal event time bound to the signatory and their role. Verification is repeated before any submission or publication; a digest mismatch invalidates every downstream conformance claim and dependent decision until the artefact is re-verified or the decision is re-taken, and the mismatch itself is raised as a finding." }, "policies": [ "No public or downstream claim about capability, safety or quality may exceed what the sealed internal evidence for the cited evidence version supports.", "Acceptance thresholds, disaggregation factors and the analysis plan are pre-registered before any result is observed; post-hoc change is permitted only with a dated, authorised change record that discloses that results were known.", "Evaluators who executed the runs relied on may not be the sole approvers of the decision that consumes them.", "Alignment to an external framework is recorded as an alignment, never as certified conformance, unless a conformity-assessment record from a competent body is attached to the claim.", "Hazardous-capability and exploit-enabling detail is stored under a restricted sensitivity label with a documented redaction rule applied before any distribution beyond the authorised recipient list.", "Inconclusive and negative results are registered with the same identity discipline as positive results and cannot be discarded without an authorised record.", "Do not treat a single accuracy number as a complete evaluation; record absent metric classes and coverage gaps", "Keep builders and validators separated for high-risk or systemic-risk subjects unless a documented exception applies", "Do not use test-split items in training or prompt-tuning of the subject under the same campaign", "Red-team and vulnerability findings default to restricted access", "Human-subjects or field tests require documented consent and ethics approval before execution", "Metric targets are reviewed for Goodhart effects before they become gate rules" ], "crud": { "read": [ "Any authenticated role may read bundle-level and layer-level structure, the protocol summary and the audience-tiered report corresponding to its tier.", "Per-item traces, red-team exercise logs and findings labelled hazardous are readable only by named roles with a recorded purpose and an unexpired access grant.", "Every read of a restricted scope produces an access-log entry carrying actor, purpose, scope and event time.", "External evaluators read only within the scope, tier and window fixed by their agreement." ], "create": [ "Creating an evaluation instance requires a resolvable subject binding, a declared mandate basis and a registered protocol version.", "Run records may be created only by an execution service that captures code revision, package versions, configuration and run start time.", "Findings may be created by assessors, red-team leads and external evaluators operating under an access grant; hazardous findings are created directly into the restricted partition.", "Decisions and waivers may be created only by roles holding the recorded delegation, and must reference a sealed evidence version." ], "update": [ "Protocols, sealed packages, signed decisions and tombstones are immutable; an update creates a new version linked by an explicit supersession reference.", "Result values change only through a re-score activity that preserves the original value, the scorer version, the actor and the reason.", "Status, state and sensitivity-label fields are updated by appending a transition entry; the current value is derived from the appended history.", "Alignment declarations may be added or withdrawn, but withdrawal is dated, reasoned and propagated to affected report versions." ], "delete": [ "Deletion is prohibited while any legal hold or statutory retention period applies to the record class.", "Permitted deletion leaves a tombstone carrying the record identifier, class, deletion event time, authorising role and the content digest of the deleted material.", "Personal data appearing in prompts, outputs or rater records may be redacted or minimised before the retention period expires where a lawful basis requires it, with the redaction itself recorded as an appended entry.", "Bulk deletion across a record class requires an authorised retention action record and does not remove the corresponding access or state logs." ] }, "roles": [ { "name": "Evaluation Owner", "responsibilities": [ "Registers the protocol and its pre-registration time and maintains the mandate basis and scope statement", "Ensures the retention schedule, sensitivity scheme and code lists are applied to the instance", "Accountable for completeness of the sealed evidence package" ] }, { "name": "Evaluator or Test Engineer", "responsibilities": [ "Designs and executes runs, capturing environment, configuration, timing and item-level traces", "Computes metrics with a recorded scorer implementation and version", "Records replication tolerance and any irreproducible components; may not be the sole approver of the consuming decision" ] }, { "name": "Independent Assessor", "responsibilities": [ "Reviews method-class fit, validity threats, uncertainty treatment and evidence sufficiency", "Records dissent where internal and independent readings differ", "Confirms that reported claims do not exceed sealed evidence" ] }, { "name": "Approver or Accountable Deployer", "responsibilities": [ "Takes approve, approve-with-conditions, reject or defer decisions within a recorded delegation", "Signs decision records bound to a specific evidence version and authorised scope", "Grants, extends and revokes waivers with compensating controls and explicit expiry" ] }, { "name": "Red Team Lead", "responsibilities": [ "Owns the threat model, attack-class scope, granted affordances and elicitation effort record", "Applies restricted sensitivity labels to uplift-relevant detail before distribution", "Records unresolved attack avenues and follow-up commitments" ] }, { "name": "Records and Access Steward", "responsibilities": [ "Applies sensitivity labels, access grants, expiries and revocations across all four scopes", "Operates the retention schedule, legal holds, deletion tombstones and access audit logs", "Verifies evidence-package digests before submission or publication and raises findings on mismatch" ] } ], "access": { "default_rule": "Deny by default. Access is granted per scope to named roles with a recorded purpose and an explicit expiry; every grant, read of a restricted scope, extension and revocation is auditable and attributable to an actor.", "scopes": [ "bundle", "layer", "finding", "artifact" ], "exceptions": [ "Competent authorities and notified bodies receive full, unredacted access to sealed evidence packages on lawful request, without the redaction applied to public or deployer tiers.", "External evaluators receive time-bounded, tier-limited access under an agreement setting scope, safe harbour and publication rights.", "Emergency access to hazardous-capability findings for incident response may bypass the normal approval path but requires post-hoc ratification by the Approver within a defined window, recorded in the access log.", "Aggregated, non-identifying metric summaries may be published at the public tier without individual grants, provided the consistency check against sealed evidence has passed.", "Access may be denied notwithstanding an otherwise valid grant where the artefact is under an unresolved digest mismatch or an active legal hold that restricts disclosure.", "Break-glass auditor or market-surveillance access with recorded legal basis", "Notified-body or AI-Office evaluation access under a documented request", "Public aggregate leaderboards when the dataset license and sensitivity class allow", "Internal builder debugging of failed runs without permission to edit scores" ], "audit_requirements": [ "Every read of a restricted finding, per-item trace or red-team log is logged with actor, purpose, scope and event time, distinct from ingestion time.", "Every access grant, extension, revocation and emergency bypass is recorded with the granting authority and the ratification outcome.", "Access logs are retained at least as long as the underlying evaluation records, are themselves access-controlled, and are never deleted by a bulk retention action on the records they describe.", "Digest verification and seal-check events are logged, and any mismatch automatically raises a finding and suspends dependent conformance claims.", "Publication events record the evidence version cited, the redaction rule applied and the reviewer who confirmed claim consistency.", "Append-only provenance for score edits, tag edits, finding errata and decisions", "Retention of evidence packs for the documentation-keeping period applicable to the subject", "Record of who accessed restricted red-team findings" ] }, "agents_bootstrap": { "filename": "AGENTS.md", "required_fields": [ "Name", "Type", "Specification URL", "Storage type URL", "Interface URL", "Processes URL" ], "read_order": [ "AGENTS.md at the model root — resolve Name, Type, Specification URL, Storage type URL, Interface URL and Processes URL before any read or write, including when the store is MongoDB or an MCP server.", "Specification URL — bundle, layer, finding and question structure, data elements, identity priority, timestamp rule and code lists.", "Storage type URL — the concrete projection in use (filesystem, Git, MongoDB, MCP resource) and its canonicalisation and round-trip profile.", "Interface URL — supported operations, authentication, access scopes and audit-logging obligations.", "Processes URL — protocol registration, subject and dataset binding, run execution, sealing, approval, waiver, publication, access grant and retention procedures.", "Only then read instance records, entering through the evaluation instance index and resolving external references to the dataset and model artefact master systems." ] } }, "coverage": { "claim": "Covers the decision and operating surface of one AI model or system evaluation instance as structured by the base: mandate and subject binding, method class and task/adaptation design, dataset binding and contamination control, metrics, disaggregation and acceptance thresholds, human and adversarial protocols, run execution and reproducibility, scoring and judging, results, item evidence and uncertainty, evidence integrity and provenance, findings and validity threats, approval, waiver and lifecycle triggers, tiered reporting and external assurance, access, retention and standards alignment. Grounded in NIST AI RMF MEASURE, EU AI Act testing and documentation duties, the GPAI Code of Practice, ISO/IEC SC 42 assessment standards and first-party evaluation-harness specifications. No claim of universal completeness and no claim of conformance to any cited instrument; dataset construction, artefact identity, AI management systems, risk treatment, incident handling and notified-body conformity assessment remain with sibling models, and campaign-level grouping, test-oracle definition under non-deterministic expectations and continual-learning feedback-loop evaluation remain unresolved.", "confidence": "medium", "checklist": [ { "dimension": "identity", "status": "covered", "notes": "Subject binding, run, result, finding, decision, report and package identity are each addressed, with an explicit identity priority placing the authoritative master-system identifier first and forbidding dates, benchmark names and scores as identifiers." }, { "dimension": "lifecycle", "status": "covered", "notes": "Protocol registration through planned, running, results-provisional, validated, decided, published, superseded and archived states, with re-evaluation triggers, invalidation conditions and waiver expiry. Grounded in AI Act Article 9(2) iterative risk management and Article 72 post-market monitoring." }, { "dimension": "relationships", "status": "covered", "notes": "Typed references to the dataset model and the model artefact model, plus provenance edges between entities, activities and agents. Risk and incident objects are referenced, never redefined." }, { "dimension": "temporal", "status": "covered", "notes": "Trigger, run start, run end, seal, decision, publication and state-entry event times are separated from observation or ingestion time; cadence and expiry use durations. RFC 3339 with seconds and explicit offset is mandated." }, { "dimension": "provenance", "status": "covered", "notes": "PROV-O entity, activity and agent semantics with generation, derivation, attribution and association edges; score edit history preserves original values, scorer version and reason. Cross-system lineage resolution is addressed explicitly." }, { "dimension": "ownership", "status": "covered", "notes": "Actor and role register, agent attribution per activity, independence level, delegated decision authority, and a separation rule preventing executors from being sole approvers." }, { "dimension": "validation", "status": "covered", "notes": "Contamination control, ground-truth quality, replication tolerance, uncertainty and significance rules, multiplicity control, validity-threat inventory, saturation assessment, tamper detection and report-consistency checking." }, { "dimension": "access", "status": "covered", "notes": "Deny-by-default rule across bundle, layer, finding and artefact scopes; hazardous-detail redaction; external evaluator access tiers; authority and emergency exceptions with post-hoc ratification; full access auditing." }, { "dimension": "retention and deletion", "status": "covered", "notes": "Per-class retention with legal bases, minimum log retention grounded in AI Act Article 19, legal hold, redaction of personal data, and tombstones carrying the digest so prior claims stay verifiable. The deletion-versus-evidence conflict is recorded rather than silently resolved." }, { "dimension": "interoperability", "status": "covered", "notes": "Declared alignments with an explicit conformance-status element, identifier schemes per entity class, comparability conditions derived from MLPerf division rules, Croissant dataset descriptors, and recorded framework conflicts with a precedence rule." }, { "dimension": "classification", "status": "covered", "notes": "Method class, evaluated-unit type, attack class, severity, audience tier, sensitivity label, independence level and access tier are all explicit code-list elements owned by the adopting Dimension." }, { "dimension": "measurement and uncertainty", "status": "covered", "notes": "Metric specification with construct validity, aggregation and reporting precision; sample size, intervals, significance and multiplicity control; resource, latency and energy accounting at run level." }, { "dimension": "security", "status": "covered", "notes": "Threat models and attack-class coverage from the NIST AML taxonomy, evidence integrity and signing, cybersecurity alignment to AI Act Article 55(1)(d), and restricted handling of exploit-enabling detail." }, { "dimension": "privacy and human subjects", "status": "covered", "notes": "Lawful basis for protected attributes used in disaggregation, participant consent and ethics approval by reference, personal data in traces, and minimisation or pseudonymisation recorded as appended entries." }, { "dimension": "spatial and jurisdictional scope", "status": "gap", "notes": "Deployment geography determines which testing duties apply, but only EU obligations are grounded in the cited sources. Authorised scope carries a market or geographic field; no jurisdiction-to-requirement mapping is asserted, and adopters outside the EU must supply their own." }, { "dimension": "cost and resource economics", "status": "not-applicable", "notes": "Evaluation budget, procurement and commercial trade-off analysis are outside the scope statement; only technical resource accounting attributable to a run is retained, for reproducibility and environmental reporting." } ], "known_omissions": [ "Full normative text of ISO/IEC 42001:2023, ISO/IEC 25059:2023, ISO/IEC TS 4213:2022 and ISO/IEC 24029-2:2023 is paywalled. Alignment is asserted at scope or clause-title level only and must be re-verified against purchased copies before any conformance statement.", "Sector-specific evaluation regimes — medical device software, automotive functional safety, aviation, and financial model risk management such as SR 11-7 — are not modelled and may impose additional validation, independent review and documentation duties.", "Agentic and long-horizon tool-use evaluation (environment sandboxing, task success over many steps, autonomous replication and self-proliferation testing) is only partially covered by the run and trace findings; no cited primary source supplies a settled structure for it.", "Continual-learning and online evaluation, where the subject changes during the evaluation window, is acknowledged through endpoint pinning and lifecycle invalidation but not fully elaborated.", "Environmental accounting appears only as optional run-level data elements; no authoritative measurement methodology for AI evaluation energy or emissions is specified here.", "Human-rater labour conditions, compensation and psychological safeguards for exposure to harmful content are referenced through participant protection but not modelled in detail.", "Cross-organisation evaluation result exchange has no settled normative registry; the identifier-scheme guidance is a design position, not a cited requirement.", "ISO/IEC 24029 formal and statistical robustness methods are neighboring techniques not modelled in full; primary text was not retrieved beyond catalogue scope lines.", "ISO/IEC TS 42119-2:2025 testing-of-AI overview and the in-progress ISO/IEC CD 4213 expansion to regression, clustering and recommendation were not retrieved as full text.", "ISO/IEC 25059 AI quality model, ISO/IEC 42001 AIMS audit evidence and ISO/IEC DIS 24970 AI logging are not encoded.", "MLPerf, Hugging Face Evaluate, OpenAI Evals and EleutherAI lm-evaluation-harness schemas are widely used projections not treated as canonical.", "Model Cards (Mitchell et al.) and data cards are reporting formats; only finding and results artifacts are modelled here.", "Medical-device, aviation and SR 11-7-style financial model-validation regimes are out of scope and unmapped.", "China, sectoral and other non-EU, non-US evaluation statutes were not searched in this pass.", "Inter-laboratory reproducibility protocols and energy-measurement metrology beyond ISO/IEC TS 4213 mentions remain thin.", "Human-subjects statistical-power design for ARIA-style contextual evaluations is only pointed at via TEVV-Athlon field-testing tools." ], "conflicts": [ "EU AI Act Annex IV requires dated, signed test reports and Articles 18 and 19 impose long retention, while data-protection minimisation duties push toward deleting item-level traces containing personal data. The tombstone-plus-redaction resolution used here is a design choice, not a cited requirement.", "MLPerf Closed-division comparability forbids retraining and requires reference-equivalent pre- and post-processing, whereas HELM-style adaptation and red-teaming deliberately vary prompting and affordances. Results from the two regimes are not comparable, so comparability is modelled as conditional rather than assumed.", "The GPAI Code of Practice expects external evaluations unless a similarly-safe comparator is shown, while Article 55(1)(a) only provides that adversarial testing may involve independent external experts. The Code is a voluntary compliance route, not a legal minimum, and the model does not treat external evaluation as mandatory.", "NIST AI RMF is explicitly voluntary and non-prescriptive; its MEASURE subcategories are used here as structuring evidence, not as requirements, and no MEASURE subcategory is presented as an obligation.", "ISO/IEC 25059 is under revision as a DIS with a broader scope, so the quality-characteristic vocabulary used for construct-validity questions may shift; the alignment is marked provisional.", "Pre-registration of thresholds serves integrity but conflicts with genuinely exploratory evaluation, where the useful measurement is discovered during the work. The model resolves this by allowing authorised, dated post-hoc change records rather than by forbidding exploration, which is a governance choice not fixed by any source.", "ISO/IEC TS 4213 explicitly excludes benchmarking and use cases, while HELM, Inspect and TEVV-Athlon are built around scenarios, suites and events; this model aligns to both and must not claim 4213 conformance for generative or agentic suites.", "Single-score leaderboards conflict with HELM multi-metric-in-context measurement and with NIST's Goodhart warning.", "First-party vendor evals conflict with AI RMF separation of builders from validators and with EU notified-body testing.", "Model-as-judge scores conflict with gold-label oracles; both are allowed only when judge quality is recorded.", "NIST AI 200-2 is an initial public draft (comment period open as of August 2026) overlapping a published AI RMF 1.0; TEVV-Athlon terms are therefore provisional.", "Closed-model API evaluation cannot satisfy bit-level reproducibility required by Croissant file hashes for the subject itself.", "EU Article 15 'appropriate' accuracy is legal and context-dependent; ISO/IEC TS 4213 metrics are statistical and classification-specific." ], "regional_assumptions": [ "Regulatory duties modelled here are EU-centric under Regulation (EU) 2024/1689; the GPAI obligations in Articles 53 to 55 applied from 2 August 2025 and high-risk system obligations phase in on a later schedule.", "US material (NIST AI 100-1, AI 100-2 E2025) is voluntary guidance; no US federal evaluation mandate is assumed.", "UK AI Security Institute Inspect is used as an engineering reference for run and log structure, not as a regulatory requirement.", "Personal-data handling assumes a GDPR-like regime for lawful basis, minimisation and retention; other regimes may differ materially, particularly on collecting protected attributes for fairness disaggregation.", "No assumption is made about Chinese, Indian, Brazilian, Japanese, Korean or Canadian AI evaluation regimes; adopters in those jurisdictions must add local requirements to the mandate and retention findings.", "Digest algorithm choice (SHA-256 in the cited Croissant profile) is illustrative; jurisdictions or sectors with mandated cryptographic suites must substitute their own.", "EU Articles 10, 15, 43, 55, 72 and 92 apply to providers and deployers in the Union for in-scope systems and GPAI models; they are alignments, not global duties.", "NIST AI RMF and TEVV-Athlon are voluntary US measurement frameworks, not binding law.", "UK AISI Inspect and the Autonomous Systems Evaluation Standard are de facto frontier-evaluation engineering practice, not a statute.", "Croissant adoption is concentrated in Hugging Face, Kaggle and OpenML ecosystems and is not universal.", "Classification-performance controls in ISO/IEC TS 4213 are assumed transferable as hygiene controls (leakage, environment, baselines) even when the metric set does not apply." ], "adversarial_checks": [ "Counterexample test — a purely internal, unpublished experiment must still fit. The mandate finding admits internal research as a basis and no finding requires a regulator, so the model does not over-fit to regulated releases.", "Boundary test — dataset licensing, annotation workforce and collection consent were deliberately excluded and left to WM-AI-009; only version binding, checksum, split designation and contamination control remain, which is the minimum reproducibility needs.", "Over-claiming test — no finding asserts conformance to ISO/IEC 42001, ISO/IEC 25059 or the AI Act. Alignment carries an explicit conformance-status element and a policy forbidding certified-conformance language without an attached assessment record.", "Unsupported-structure test — an attractive evaluation maturity score layer and a universal severity scale were both rejected because no cited primary source defines such scales; both are pushed into Dimension owner-package requirements instead.", "Identity test — benchmark names, run timestamps and file names were rejected as identifiers because they are neither unique nor stable across re-runs; identity priority forces a master-system identifier first and treats a date strictly as an attribute.", "Falsifiability test — required flags were set to the minimum that Annex IV point 2(g), MEASURE 2.1 and PROV-O jointly demand. If a compliant evaluation can be executed without recording an element marked required, that element is over-specified and should be demoted.", "Duplication test — post-market monitoring, serious-incident reporting and risk treatment appear only as triggers and references, never as owned processes, so the model does not shadow an incident or risk sibling.", "Paywall-honesty test — every ISO alignment is flagged as scope-level only, and none is used as the sole support for a structural node; each ISO-supported node also carries a freely verifiable primary source.", "Checked that evaluation datasets-as-used are bindings rather than a second dataset-master model competing with WM-AI-009.", "Checked that the subject under evaluation is a reference to WM-SFT-004 plus interface configuration, not a duplicate model-card identity.", "Checked that JSON, YAML, Markdown, Git, MCP and MongoDB are treated as projections, with Inspect .eval and .json called out as such.", "Checked that dates are not used as identifiers and that event time is separated from ingestion time.", "Checked that rare paths (partial runs, re-score, retraction, red-team non-publication, continual-learning feedback loops, official evaluations, field-test consent) are present or listed as omissions.", "Checked that no node claims universal completeness; ISO 24029, 42119-2, Model Cards and non-Western law are explicit gaps." ] }, "researchAdjudication": { "providerMode": "dual-provider", "activeProviders": [ "claude", "grok" ], "waivedProviders": [], "providerPolicy": {}, "boundaryDecision": { "entry_kind": "aggregate", "status": "accepted", "rationale": "Both providers independently classify WM-AI-003 as an aggregate; neither models it as a service layer or a reference list, so entry kind is settled without reclassification. The root is fixed to the base definition — one bounded, protocol-governed evaluation instance applied to a version-pinned subject under a declared configuration — and not to grok's campaign that groups many tasks, datasets and runs. Campaign-level grouping is deferred as a possible sibling or parent object rather than folded into this root, because admitting it would silently change the aggregate's cardinality. Dataset master data stays with WM-AI-009 and artefact identity with WM-SFT-004, on which both providers agree." }, "decisions": [ { "concept": "Base provider selection", "disposition": "claude", "rationale": "Chosen on boundary clarity, not size: seven neighbour notes each with source refs and an explicit distinction, symmetrical in-scope/out-of-scope lists, and a coverage checklist that marks its own spatial gap honestly. It also owns whole surfaces grok lacks entirely — evidence integrity and sealing, PROV-O provenance, tiered reporting, external assurance, records governance and alignment/comparability." }, { "concept": "Aggregate root granularity: instance versus campaign", "disposition": "instance (base wording retained)", "rationale": "The base's one bounded evaluation instance yields stable identity, lifecycle and approval semantics. Grok's campaign groups many tasks, datasets and runs, which would change the cardinality of every downstream record. Grouping is deferred, not adopted." }, { "concept": "Adaptation protocol and agentic sandbox (grok scenario-and-adaptation)", "disposition": "accepted into l-method-and-task-design", "rationale": "Fills a gap the base itself declares as a known omission, with evidence the base never cited (UK AISI Autonomous Systems Evaluation Standard, HELM scenario/adaptation, Inspect task composition). Instance-scoped wording throughout, so a verbatim copy is safe." }, { "concept": "Scoring and judging including model-as-judge calibration (grok finding-scoring-and-judging)", "disposition": "accepted into l-results-and-uncertainty", "rationale": "Judge quality, gold-oracle agreement and deferred re-scoring are materially missing from the base. Accepted despite two of four questions partially overlapping existing base questions, because rejecting would lose the only coverage of automated judging." }, { "concept": "Campaign identity (grok evaluation-campaign-identity)", "disposition": "deferred", "rationale": "Its name, description and questions all assert a campaign that groups multiple tasks and runs, which contradicts the chosen instance root; a verbatim copy would reintroduce the rejected granularity. The genuine gap it exposes — the base never asks for a master identifier for the evaluation instance itself — is recorded as deferred research instead." }, { "concept": "Evaluation dataset binding (grok evaluation-dataset-binding)", "disposition": "rejected", "rationale": "Duplicates the base's f-dataset-binding-and-contamination, and its licence and redistribution question crosses into WM-AI-009, which both providers agree owns dataset licensing. Accepting it would breach a boundary both providers independently drew." }, { "concept": "Split leakage and contamination controls (grok split-leakage-and-contamination)", "disposition": "rejected as duplicative", "rationale": "Leakage, canary exposure and representativeness are already carried by the base's contamination-control and hold-out-governance questions. Only the test-oracle question is new, and it is not worth a duplicate node; it is recorded as deferred research." }, { "concept": "Subject under evaluation (grok subject-under-evaluation)", "disposition": "rejected as duplicative", "rationale": "Splits across the base's f-subject-binding and f-assessor-independence with no material addition; the open-weight versus closed-API access class is already reachable through the base's external-access-tier and irreproducible-component questions." }, { "concept": "Metric and metrology-block definition (grok metric-and-block-definition)", "disposition": "rejected as duplicative", "rationale": "The base's f-metric-specification plus f-uncertainty already carry definition, aggregation, construct validity, scorer provenance and significance. The TEVV-Athlon block vocabulary it introduces rests on an initial public draft still in comment period." }, { "concept": "Aggregate results and uncertainty (grok aggregate-results-and-uncertainty)", "disposition": "rejected as duplicative", "rationale": "Fully covered by the base's f-result-and-item-evidence and f-uncertainty, including the draft/corrected/retracted states via the supersession question." }, { "concept": "Evaluation finding (grok evaluation-finding)", "disposition": "rejected as duplicative", "rationale": "The base's f-finding-record and f-validity-threats already cover citable identity, evidence links, severity, state, limitations and non-suppression of negative results." }, { "concept": "Gate approval decision (grok gate-approval-decision)", "disposition": "rejected", "rationale": "Duplicates f-approval-and-waiver, and its notified-body and declaration-of-conformity content sits in a neighbour the base explicitly places out of scope. Verbatim copy would make the model appear to own conformity-assessment records." }, { "concept": "Re-evaluation and post-market monitoring (grok reevaluation-and-post-market)", "disposition": "deferred", "rationale": "Triggers and cadence are already in f-lifecycle-and-triggers, and the post-market monitoring plan and market-surveillance access questions cross into the incident and post-market neighbour the base excludes. The genuinely new continual-learning feedback-loop concern is held as deferred research." }, { "concept": "Access, retention and reproducibility (grok access-retention-and-reproducibility)", "disposition": "rejected as duplicative", "rationale": "Collapses into one node what the base separates across f-access-and-sensitivity, f-retention-and-deletion, f-reproducibility and f-evidence-package; adopting it would create four-way duplication and blur access from retention." }, { "concept": "Claims and acceptance criteria (grok claims-and-acceptance-criteria)", "disposition": "rejected as duplicative", "rationale": "The base's f-purpose-and-claim-scope and f-acceptance-criteria cover claim statement, pre-registered thresholds, authority, change control and breach consequence; the declared-metrics duty is reachable through q-mandatory-disclosure." }, { "concept": "Integrity scanning as an operation", "disposition": "accepted as function", "rationale": "No base function performs transcript-level validity scanning, although the base asserts the corresponding validity-threat and saturation findings. Adding the function closes an operation gap without adding structure." }, { "concept": "Reproducibility packaging as an operation", "disposition": "accepted as function", "rationale": "Distinct in purpose from sealing evidence for a decision: it exists to let an independent party attempt a re-run and to disclose what cannot be released for a closed subject." }, { "concept": "Source identifier remapping for accepted grok nodes", "disposition": "mandatory before synthesis", "rationale": "The two providers share no URL and their SRC numbering collides in meaning; copied source_refs would resolve to the wrong documents in the base list. Grok's Inspect, TR 29119-11 and AISI Autonomous Systems sources must be merged and renumbered first." } ], "publicationHolds": [ "Source verification: the two providers share zero common URLs while citing several of the same instruments through different endpoints (AI RMF, AI Act, ISO/IEC TS 4213, Croissant, Inspect, HELM). Every accepted source must be fetched live, de-duplicated against its twin, and pinned to a version before publication.", "Source-ID remapping verification: the deterministic synthesis merged Grok-only sources and rewrote every copied source_ref into the unified source namespace. Before canonical release, verify that the accepted Inspect, ISO/IEC TR 29119-11 and UK AISI Autonomous Systems nodes resolve to those merged source records and never to a same-numbered Claude source.", "Croissant version conflict: the base pins Specification 1.0 (1 March 2024) while grok pins 1.1 (29 January 2026). One live version must be selected and any split/hash semantics re-checked against it before the dataset-binding and reproducibility content is published.", "Paywalled ISO texts (42001, 25059, TS 4213, 24029-2, TR 29119-11) were read at scope or clause-title level only. Alignment statements must be re-verified against purchased copies and no conformance language may be published until then.", "Draft-standard dependency: NIST AI 200-2 ipd (TEVV-Athlon) is an initial public draft with a comment period open to 6 October 2026. No accepted node depends on it, and any residual TEVV-Athlon vocabulary carried into narrative text must be marked provisional.", "Multi-profile validation: the aggregate has not been exercised against distinct domain profiles. Validate against at least a regulated high-risk deployment profile, a frontier GPAI systemic-risk safety profile, and a purely internal unpublished research profile, confirming that no finding forces a regulator, a notified body or a public report to exist." ], "deferredResearch": [ "Evaluation-instance identity: the base never asks which authoritative master-system identifier designates the evaluation instance itself, only the subject, run, result, finding and report. Draft an instance-scoped identity node rather than importing grok's campaign-scoped wording.", "Campaign or programme grouping: decide whether a body of work spanning several evaluation instances is a sibling aggregate, a parent programme object or a service layer, and where the identifier for it lives. Grok's campaign node is held pending this decision.", "Test-oracle definition where expected results are unavailable or non-deterministic (ISO/IEC TR 29119-11 oracle problem). The base's ground-truth question presumes a reference answer exists; primary text must be retrieved before a node is drafted.", "Continual-learning and post-deployment feedback-loop evaluation (EU AI Act Article 15(4) and Article 72 monitoring plans), including where the boundary sits against the post-market monitoring and incident neighbour that the base excludes.", "Cross-organisation exchange of evaluation results: the base's identifier-scheme and comparability guidance is a design position with no cited normative registry behind it, and grok independently notes MLPerf, Evals and lm-evaluation-harness schemas are projections rather than canon.", "Jurisdictional coverage beyond the EU and voluntary US frameworks: the base marks spatial/jurisdictional scope as a gap and neither provider searched Chinese, Indian, Japanese, Korean, Canadian or Brazilian evaluation regimes, nor sector regimes such as medical device, aviation or SR 11-7 model validation." ] }, "statistics": { "sources": 25, "bundles": 5, "layers": 13, "findings": 28, "questions": 112, "artifacts": 30, "functions": 16 } }