# Vercy AI instruction - YAML 1.2 (JSON-compatible) { "vercy": "1.0-draft", "publication": { "status": "published", "adjudicationStatus": "reviewable-draft", "publishableCanonical": false, "generatedAt": "2026-09-06T12:06:45Z", "synthesisSha256": "cc86100cdb4c606db07f42e70c57eeb9ccd4ea0cf720987355d4dd63ce3b8be2", "providerMode": "single-provider-waiver", "providers": [ "Codex" ], "waivedProviders": [ "Claude", "Grok" ] }, "metaModel": { "id": "WM-AI-009", "registryId": "vr.wm-ai-009", "name": "Evaluation Dataset / Benchmark", "version": "0.3.0-research.1", "previousVersions": [], "entryKind": "aggregate", "family": "World Models", "category": "Information and virtual systems", "industry": [ "Cross-industry" ], "domain": [ "INF.AI.DAT" ], "tags": [ "evaluation", "dataset", "benchmark", "inf.ai.dat" ], "status": "published" }, "canonicalUrl": "https://ver.cy/models/wm-ai-009-evaluation-dataset-benchmark/", "sourceUrl": "https://github.com/ver-cy/world-models/tree/feat/mega-model-registry/publications/wm-ai-009-evaluation-dataset-benchmark", "model": { "registry_id": "vr.wm-ai-009", "model_id": "WM-AI-009", "name": "Evaluation Dataset / Benchmark", "entry_kind": "aggregate", "purpose": "Represent one governed, versioned evaluation dataset or benchmark definition that makes population, content, labels, splits, protocol, metrics, validity and release semantics machine-readable without absorbing evaluation runs or result masters.", "scope_statement": "Owns benchmark identity, family, suite, task and release; intended use and population; versioned items, schemas, labels, reference answers, rubrics, annotations, splits, sampling and hidden-test controls; source lineage, rights, privacy, security and access; protocol, environment, metrics, scoring, statistics, validity, contamination, subgroups, robustness, limitations, lifecycle, result bindings, retention, audit and projections. Evaluated models, evaluation runs, results, leaderboards, claims, policy, credential, provenance, audit and records masters remain external.", "in_scope": [ "Benchmark identity, release, purpose, task, population, content, items, labels, references, rubrics, annotations, splits, sampling, lineage, rights and access", "Protocol, interface, environment, metrics, scoring, uncertainty, quality, leakage, contamination, subgroup, robustness, safety, lifecycle, result bindings and projections" ], "out_of_scope": [ "Owning or executing evaluated models, evaluation runs, results, leaderboards, deployment decisions, policy, credentials, provenance, audit or records masters", "Treating a score, leaderboard position, checksum, split label or publication as proof of safety, fairness, validity or fitness", "Autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup" ], "boundary_notes": [ { "neighbor": "WM-DAT-001 Dataset", "distinction": "The catalogue parent signal identifies the dataset boundary, but no frozen relation-ledger edge grants inheritance or mutation authority. The benchmark adds task, split-role, protocol, metric and validity semantics while source datasets remain external masters.", "source_refs": [ "SRC-006", "SRC-007", "SRC-008", "SRC-011", "SRC-012", "SRC-013" ] }, { "neighbor": "WM-AI-003 Evaluation Run", "distinction": "The incoming candidate reference expresses evaluation reproducibility. A run owns execution inputs, environment, outputs and measurements; this model owns the benchmark definition and non-owning run or result bindings.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ] }, { "neighbor": "AI model, result, leaderboard and claim", "distinction": "Models are evaluated subjects; results and leaderboards are version-qualified observations and presentations. None becomes part of the benchmark definition merely by reference.", "source_refs": [ "SRC-001", "SRC-009", "SRC-010" ] }, { "neighbor": "DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS", "distinction": "These are catalog, dataset, supply-chain, card, benchmark, provenance and classification profiles with different scopes. Every mapping requires a release pin and information-loss declaration.", "source_refs": [ "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014" ] } ] }, "sources": [ { "id": "SRC-001", "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0)", "organization": "National Institute of Standards and Technology", "url": "https://www.nist.gov/publications/artificial-intelligence-risk-management-framework-ai-rmf-10", "version_or_date": "NIST AI 100-1, 26 January 2023; revision in progress at access", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Frames governed AI measurement, validation, documentation, risk and accountability." }, { "id": "SRC-002", "title": "Artificial Intelligence Risk Management Framework: Generative Artificial Intelligence Profile", "organization": "National Institute of Standards and Technology", "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.600-1.pdf", "version_or_date": "NIST AI 600-1, July 2024", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Adds dataset, provenance, privacy, security, testing and monitoring considerations for generative AI." }, { "id": "SRC-003", "title": "Adversarial Machine Learning: A Taxonomy and Terminology of Attacks and Mitigations", "organization": "National Institute of Standards and Technology", "url": "https://www.nist.gov/publications/adversarial-machine-learning-taxonomy-and-terminology-attacks-and-mitigations-0", "version_or_date": "NIST AI 100-2e2025, March 2025", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Provides current terminology for poisoning, evasion, privacy and misuse attacks relevant to benchmark integrity." }, { "id": "SRC-004", "title": "Regulation (EU) 2024/1689 Artificial Intelligence Act", "organization": "European Union", "url": "https://eur-lex.europa.eu/eli/reg/2024/1689/oj", "version_or_date": "13 June 2024 official journal text", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Provides an EU legal profile for training, validation and testing data, documentation and evaluation records." }, { "id": "SRC-005", "title": "Regulation (EU) 2016/679 General Data Protection Regulation", "organization": "European Union", "url": "https://eur-lex.europa.eu/eli/reg/2016/679/oj", "version_or_date": "27 April 2016; applicable from 25 May 2018", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Provides an EU privacy profile for lawful processing, minimization, rights, security and accountability." }, { "id": "SRC-006", "title": "Data Catalog Vocabulary (DCAT) Version 3", "organization": "World Wide Web Consortium", "url": "https://www.w3.org/TR/vocab-dcat-3/", "version_or_date": "W3C Recommendation 22 August 2024", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Defines catalog, dataset, series, version and distribution metadata." }, { "id": "SRC-007", "title": "PROV-O: The PROV Ontology", "organization": "World Wide Web Consortium", "url": "https://www.w3.org/TR/prov-o/", "version_or_date": "W3C Recommendation 30 April 2013", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Defines entity, activity, agent, derivation, revision, attribution and time for lineage." }, { "id": "SRC-008", "title": "Croissant ML-ready Dataset Metadata Format", "organization": "MLCommons", "url": "https://docs.mlcommons.org/croissant/", "version_or_date": "Croissant 1.0 documentation current at access", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Combines dataset metadata, resource descriptions, record structure and default ML semantics." }, { "id": "SRC-009", "title": "MLCommons Benchmarks", "organization": "MLCommons", "url": "https://mlcommons.org/benchmarks/", "version_or_date": "First-party benchmark programme current at access", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "States fairness, usefulness, reproducibility and expert working-group governance goals." }, { "id": "SRC-010", "title": "MLPerf Inference Rules", "organization": "MLCommons", "url": "https://github.com/mlcommons/inference_policies/blob/master/inference_rules.adoc", "version_or_date": "Current master at access; release or commit pin required", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Distinguishes benchmark, reference implementation, accuracy, performance and calibration datasets, targets, procedures and verification." }, { "id": "SRC-011", "title": "Dataset Cards", "organization": "Hugging Face", "url": "https://huggingface.co/docs/hub/datasets-cards", "version_or_date": "Hub documentation current at access", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Defines discoverable dataset metadata and responsible-use documentation." }, { "id": "SRC-012", "title": "Data Cards: Purposeful and Transparent Dataset Documentation for Responsible AI", "organization": "Google Research", "url": "https://research.google/pubs/data-cards-purposeful-and-transparent-dataset-documentation-for-responsible-ai/", "version_or_date": "2022 publication", "source_type": "scientific", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Provides lifecycle-oriented dataset documentation for origins, methods, intent, ethics and evolution." }, { "id": "SRC-013", "title": "SPDX Specification Dataset Profile", "organization": "SPDX", "url": "https://spdx.github.io/spdx-spec/v3.0.1/model/Dataset/Dataset/", "version_or_date": "SPDX 3.0.1", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Defines dataset package metadata, preparation, characteristics, access and licensing relationships." }, { "id": "SRC-014", "title": "SKOS Simple Knowledge Organization System Reference", "organization": "World Wide Web Consortium", "url": "https://www.w3.org/TR/skos-reference/", "version_or_date": "W3C Recommendation 18 August 2009", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Defines controlled concepts, labels, schemes and mapping relations for classifications." }, { "id": "SRC-015", "title": "Date and Time on the Internet: Timestamps", "organization": "Internet Engineering Task Force", "url": "https://www.rfc-editor.org/info/rfc3339/", "version_or_date": "RFC 3339, July 2002", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-09-06T12:01:01Z", "relevance": "Defines interoperable timestamps with seconds and an explicit UTC relationship." } ], "structure": { "bundles": [ { "id": "benchmark-identity-scope-version-and-definition", "name": "Benchmark identity, scope, version and definition", "description": "Groups governed context for benchmark identity, scope, version and definition.", "rationale": "The governed benchmark definition is distinct from datasets, executions and result claims.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-006", "SRC-009", "SRC-010", "SRC-011", "SRC-012", "SRC-014" ], "layers": [ { "id": "benchmark-root-family-suite-task-and-release", "name": "Benchmark root, family, suite, task and release", "description": "Groups benchmark context for benchmark root, family, suite, task and release.", "source_refs": [ "SRC-006", "SRC-009", "SRC-010", "SRC-014" ], "findings": [ { "id": "benchmark-identity-namespace-owner-revision-and-current-head", "name": "Benchmark identity, namespace, owner, revision and current head", "description": "Records benchmark identity, namespace, owner, revision and current head as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-006", "SRC-009", "SRC-010", "SRC-014" ], "questions": [ { "id": "benchmark-identity-namespace-owner-revision-and-current-head-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish benchmark identity, namespace, owner, revision and current head?", "kind": "identity", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "benchmark-identity-namespace-owner-revision-and-current-head-q02", "text": "Who may create, assert, review, approve, correct or disclose benchmark identity, namespace, owner, revision and current head, under which purpose and authority?", "kind": "composition", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "benchmark-identity-namespace-owner-revision-and-current-head-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify benchmark identity, namespace, owner, revision and current head?", "kind": "privacy", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "benchmark-identity-namespace-owner-revision-and-current-head-data", "name": "Benchmark identity, namespace, owner, revision and current head data", "description": "Typed benchmark data for benchmark identity, namespace, owner, revision and current head, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-006", "SRC-009", "SRC-010", "SRC-014" ] } ], "artifacts": [ { "id": "benchmark-identity-namespace-owner-revision-and-current-head-record", "name": "Benchmark identity, namespace, owner, revision and current head record", "description": "Immutable or successor-versioned evidence for benchmark identity, namespace, owner, revision and current head.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for benchmark-identity-namespace-owner-revision-and-current-head; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-006", "SRC-009", "SRC-010", "SRC-014" ] } ], "inline_only_rationale": null }, { "id": "family-suite-task-profile-version-alias-and-equivalence", "name": "Family, suite, task, profile, version, alias and equivalence", "description": "Records family, suite, task, profile, version, alias and equivalence as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-006", "SRC-009", "SRC-010", "SRC-014" ], "questions": [ { "id": "family-suite-task-profile-version-alias-and-equivalence-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish family, suite, task, profile, version, alias and equivalence?", "kind": "classification", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "family-suite-task-profile-version-alias-and-equivalence-q02", "text": "Who may create, assert, review, approve, correct or disclose family, suite, task, profile, version, alias and equivalence, under which purpose and authority?", "kind": "evidence", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "family-suite-task-profile-version-alias-and-equivalence-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify family, suite, task, profile, version, alias and equivalence?", "kind": "lifecycle", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "family-suite-task-profile-version-alias-and-equivalence-data", "name": "Family, suite, task, profile, version, alias and equivalence data", "description": "Typed benchmark data for family, suite, task, profile, version, alias and equivalence, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-006", "SRC-009", "SRC-010", "SRC-014" ] } ], "artifacts": [ { "id": "family-suite-task-profile-version-alias-and-equivalence-record", "name": "Family, suite, task, profile, version, alias and equivalence record", "description": "Immutable or successor-versioned evidence for family, suite, task, profile, version, alias and equivalence.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for family-suite-task-profile-version-alias-and-equivalence; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-006", "SRC-009", "SRC-010", "SRC-014" ] } ], "inline_only_rationale": null } ] }, { "id": "purpose-population-domain-and-coverage", "name": "Purpose, population, domain and coverage", "description": "Groups benchmark context for purpose, population, domain and coverage.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-011", "SRC-012" ], "findings": [ { "id": "intended-use-question-domain-modality-language-and-jurisdiction", "name": "Intended use, question, domain, modality, language and jurisdiction", "description": "Records intended use, question, domain, modality, language and jurisdiction as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-011", "SRC-012" ], "questions": [ { "id": "intended-use-question-domain-modality-language-and-jurisdiction-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish intended use, question, domain, modality, language and jurisdiction?", "kind": "requirement", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "intended-use-question-domain-modality-language-and-jurisdiction-q02", "text": "Who may create, assert, review, approve, correct or disclose intended use, question, domain, modality, language and jurisdiction, under which purpose and authority?", "kind": "ownership", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "intended-use-question-domain-modality-language-and-jurisdiction-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify intended use, question, domain, modality, language and jurisdiction?", "kind": "quality", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "intended-use-question-domain-modality-language-and-jurisdiction-data", "name": "Intended use, question, domain, modality, language and jurisdiction data", "description": "Typed benchmark data for intended use, question, domain, modality, language and jurisdiction, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-011", "SRC-012" ] } ], "artifacts": [ { "id": "intended-use-question-domain-modality-language-and-jurisdiction-record", "name": "Intended use, question, domain, modality, language and jurisdiction record", "description": "Immutable or successor-versioned evidence for intended use, question, domain, modality, language and jurisdiction.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for intended-use-question-domain-modality-language-and-jurisdiction; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-011", "SRC-012" ] } ], "inline_only_rationale": null }, { "id": "target-population-sampling-frame-inclusion-exclusion-and-coverage", "name": "Target population, sampling frame, inclusion, exclusion and coverage", "description": "Records target population, sampling frame, inclusion, exclusion and coverage as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-011", "SRC-012" ], "questions": [ { "id": "target-population-sampling-frame-inclusion-exclusion-and-coverage-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish target population, sampling frame, inclusion, exclusion and coverage?", "kind": "measurement", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "target-population-sampling-frame-inclusion-exclusion-and-coverage-q02", "text": "Who may create, assert, review, approve, correct or disclose target population, sampling frame, inclusion, exclusion and coverage, under which purpose and authority?", "kind": "measurement", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "target-population-sampling-frame-inclusion-exclusion-and-coverage-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify target population, sampling frame, inclusion, exclusion and coverage?", "kind": "security", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "target-population-sampling-frame-inclusion-exclusion-and-coverage-data", "name": "Target population, sampling frame, inclusion, exclusion and coverage data", "description": "Typed benchmark data for target population, sampling frame, inclusion, exclusion and coverage, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-011", "SRC-012" ] } ], "artifacts": [ { "id": "target-population-sampling-frame-inclusion-exclusion-and-coverage-record", "name": "Target population, sampling frame, inclusion, exclusion and coverage record", "description": "Immutable or successor-versioned evidence for target population, sampling frame, inclusion, exclusion and coverage.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for target-population-sampling-frame-inclusion-exclusion-and-coverage; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-011", "SRC-012" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "content-items-labels-splits-and-distributions", "name": "Content, items, labels, splits and distributions", "description": "Groups governed context for content, items, labels, splits and distributions.", "rationale": "Versioned items and split roles preserve evaluative meaning and leakage controls.", "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010", "SRC-011", "SRC-012" ], "layers": [ { "id": "item-schema-input-reference-and-annotation", "name": "Item schema, input, reference and annotation", "description": "Groups benchmark context for item schema, input, reference and annotation.", "source_refs": [ "SRC-008", "SRC-010", "SRC-011", "SRC-012" ], "findings": [ { "id": "item-case-identity-input-prompt-context-response-and-schema", "name": "Item or case identity, input, prompt, context, response and schema", "description": "Records item or case identity, input, prompt, context, response and schema as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-008", "SRC-010", "SRC-011", "SRC-012" ], "questions": [ { "id": "item-case-identity-input-prompt-context-response-and-schema-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish item or case identity, input, prompt, context, response and schema?", "kind": "identity", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "item-case-identity-input-prompt-context-response-and-schema-q02", "text": "Who may create, assert, review, approve, correct or disclose item or case identity, input, prompt, context, response and schema, under which purpose and authority?", "kind": "exception", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "item-case-identity-input-prompt-context-response-and-schema-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify item or case identity, input, prompt, context, response and schema?", "kind": "retention", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "item-case-identity-input-prompt-context-response-and-schema-data", "name": "Item or case identity, input, prompt, context, response and schema data", "description": "Typed benchmark data for item or case identity, input, prompt, context, response and schema, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-008", "SRC-010", "SRC-011", "SRC-012" ] } ], "artifacts": [ { "id": "item-case-identity-input-prompt-context-response-and-schema-record", "name": "Item or case identity, input, prompt, context, response and schema record", "description": "Immutable or successor-versioned evidence for item or case identity, input, prompt, context, response and schema.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for item-case-identity-input-prompt-context-response-and-schema; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-008", "SRC-010", "SRC-011", "SRC-012" ] } ], "inline_only_rationale": null }, { "id": "label-reference-answer-rubric-annotation-adjudication-and-uncertainty", "name": "Label, reference answer, rubric, annotation, adjudication and uncertainty", "description": "Records label, reference answer, rubric, annotation, adjudication and uncertainty as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-008", "SRC-010", "SRC-011", "SRC-012" ], "questions": [ { "id": "label-reference-answer-rubric-annotation-adjudication-and-uncertainty-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish label, reference answer, rubric, annotation, adjudication and uncertainty?", "kind": "quality", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "label-reference-answer-rubric-annotation-adjudication-and-uncertainty-q02", "text": "Who may create, assert, review, approve, correct or disclose label, reference answer, rubric, annotation, adjudication and uncertainty, under which purpose and authority?", "kind": "provenance", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "label-reference-answer-rubric-annotation-adjudication-and-uncertainty-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify label, reference answer, rubric, annotation, adjudication and uncertainty?", "kind": "interoperability", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "label-reference-answer-rubric-annotation-adjudication-and-uncertainty-data", "name": "Label, reference answer, rubric, annotation, adjudication and uncertainty data", "description": "Typed benchmark data for label, reference answer, rubric, annotation, adjudication and uncertainty, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-008", "SRC-010", "SRC-011", "SRC-012" ] } ], "artifacts": [ { "id": "label-reference-answer-rubric-annotation-adjudication-and-uncertainty-record", "name": "Label, reference answer, rubric, annotation, adjudication and uncertainty record", "description": "Immutable or successor-versioned evidence for label, reference answer, rubric, annotation, adjudication and uncertainty.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for label-reference-answer-rubric-annotation-adjudication-and-uncertainty; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-008", "SRC-010", "SRC-011", "SRC-012" ] } ], "inline_only_rationale": null } ] }, { "id": "splits-sampling-hidden-tests-and-distributions", "name": "Splits, sampling, hidden tests and distributions", "description": "Groups benchmark context for splits, sampling, hidden tests and distributions.", "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ], "findings": [ { "id": "train-validation-calibration-evaluation-split-role-and-leakage-control", "name": "Train, validation, calibration and evaluation split role and leakage control", "description": "Records train, validation, calibration and evaluation split role and leakage control as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ], "questions": [ { "id": "train-validation-calibration-evaluation-split-role-and-leakage-control-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish train, validation, calibration and evaluation split role and leakage control?", "kind": "validation", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "train-validation-calibration-evaluation-split-role-and-leakage-control-q02", "text": "Who may create, assert, review, approve, correct or disclose train, validation, calibration and evaluation split role and leakage control, under which purpose and authority?", "kind": "process", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "train-validation-calibration-evaluation-split-role-and-leakage-control-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify train, validation, calibration and evaluation split role and leakage control?", "kind": "decision", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "train-validation-calibration-evaluation-split-role-and-leakage-control-data", "name": "Train, validation, calibration and evaluation split role and leakage control data", "description": "Typed benchmark data for train, validation, calibration and evaluation split role and leakage control, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ] } ], "artifacts": [ { "id": "train-validation-calibration-evaluation-split-role-and-leakage-control-record", "name": "Train, validation, calibration and evaluation split role and leakage control record", "description": "Immutable or successor-versioned evidence for train, validation, calibration and evaluation split role and leakage control.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for train-validation-calibration-evaluation-split-role-and-leakage-control; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ] } ], "inline_only_rationale": null }, { "id": "sampling-stratification-weight-hidden-test-distribution-and-digest", "name": "Sampling, stratification, weight, hidden test, distribution and digest", "description": "Records sampling, stratification, weight, hidden test, distribution and digest as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ], "questions": [ { "id": "sampling-stratification-weight-hidden-test-distribution-and-digest-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish sampling, stratification, weight, hidden test, distribution and digest?", "kind": "security", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "sampling-stratification-weight-hidden-test-distribution-and-digest-q02", "text": "Who may create, assert, review, approve, correct or disclose sampling, stratification, weight, hidden test, distribution and digest, under which purpose and authority?", "kind": "validation", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "sampling-stratification-weight-hidden-test-distribution-and-digest-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify sampling, stratification, weight, hidden test, distribution and digest?", "kind": "state", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "sampling-stratification-weight-hidden-test-distribution-and-digest-data", "name": "Sampling, stratification, weight, hidden test, distribution and digest data", "description": "Typed benchmark data for sampling, stratification, weight, hidden test, distribution and digest, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ] } ], "artifacts": [ { "id": "sampling-stratification-weight-hidden-test-distribution-and-digest-record", "name": "Sampling, stratification, weight, hidden test, distribution and digest record", "description": "Immutable or successor-versioned evidence for sampling, stratification, weight, hidden test, distribution and digest.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for sampling-stratification-weight-hidden-test-distribution-and-digest; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "provenance-rights-privacy-security-and-quality", "name": "Provenance, rights, privacy, security and quality", "description": "Groups governed context for provenance, rights, privacy, security and quality.", "rationale": "Source lineage and permission boundaries qualify every benchmark release.", "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-012", "SRC-013" ], "layers": [ { "id": "source-collection-transformation-and-lineage", "name": "Source, collection, transformation and lineage", "description": "Groups benchmark context for source, collection, transformation and lineage.", "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-008", "SRC-012" ], "findings": [ { "id": "source-acquisition-collection-generation-transform-and-release-lineage", "name": "Source, acquisition, collection, generation, transform and release lineage", "description": "Records source, acquisition, collection, generation, transform and release lineage as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-008", "SRC-012" ], "questions": [ { "id": "source-acquisition-collection-generation-transform-and-release-lineage-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish source, acquisition, collection, generation, transform and release lineage?", "kind": "provenance", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "source-acquisition-collection-generation-transform-and-release-lineage-q02", "text": "Who may create, assert, review, approve, correct or disclose source, acquisition, collection, generation, transform and release lineage, under which purpose and authority?", "kind": "privacy", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "source-acquisition-collection-generation-transform-and-release-lineage-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify source, acquisition, collection, generation, transform and release lineage?", "kind": "identity", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "source-acquisition-collection-generation-transform-and-release-lineage-data", "name": "Source, acquisition, collection, generation, transform and release lineage data", "description": "Typed benchmark data for source, acquisition, collection, generation, transform and release lineage, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-008", "SRC-012" ] } ], "artifacts": [ { "id": "source-acquisition-collection-generation-transform-and-release-lineage-record", "name": "Source, acquisition, collection, generation, transform and release lineage record", "description": "Immutable or successor-versioned evidence for source, acquisition, collection, generation, transform and release lineage.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for source-acquisition-collection-generation-transform-and-release-lineage; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-008", "SRC-012" ] } ], "inline_only_rationale": null }, { "id": "contributor-tool-pipeline-derivation-version-and-current-release", "name": "Contributor, tool, pipeline, derivation, version and current release", "description": "Records contributor, tool, pipeline, derivation, version and current release as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-008", "SRC-012" ], "questions": [ { "id": "contributor-tool-pipeline-derivation-version-and-current-release-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish contributor, tool, pipeline, derivation, version and current release?", "kind": "ownership", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "contributor-tool-pipeline-derivation-version-and-current-release-q02", "text": "Who may create, assert, review, approve, correct or disclose contributor, tool, pipeline, derivation, version and current release, under which purpose and authority?", "kind": "lifecycle", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "contributor-tool-pipeline-derivation-version-and-current-release-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify contributor, tool, pipeline, derivation, version and current release?", "kind": "classification", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "contributor-tool-pipeline-derivation-version-and-current-release-data", "name": "Contributor, tool, pipeline, derivation, version and current release data", "description": "Typed benchmark data for contributor, tool, pipeline, derivation, version and current release, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-008", "SRC-012" ] } ], "artifacts": [ { "id": "contributor-tool-pipeline-derivation-version-and-current-release-record", "name": "Contributor, tool, pipeline, derivation, version and current release record", "description": "Immutable or successor-versioned evidence for contributor, tool, pipeline, derivation, version and current release.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for contributor-tool-pipeline-derivation-version-and-current-release; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-008", "SRC-012" ] } ], "inline_only_rationale": null } ] }, { "id": "rights-consent-privacy-security-and-access", "name": "Rights, consent, privacy, security and access", "description": "Groups benchmark context for rights, consent, privacy, security and access.", "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ], "findings": [ { "id": "license-ip-data-right-consent-purpose-export-and-use-term", "name": "License, IP, data right, consent, purpose, export and use term", "description": "Records license, ip, data right, consent, purpose, export and use term as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ], "questions": [ { "id": "license-ip-data-right-consent-purpose-export-and-use-term-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish license, ip, data right, consent, purpose, export and use term?", "kind": "authority", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "license-ip-data-right-consent-purpose-export-and-use-term-q02", "text": "Who may create, assert, review, approve, correct or disclose license, ip, data right, consent, purpose, export and use term, under which purpose and authority?", "kind": "quality", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "license-ip-data-right-consent-purpose-export-and-use-term-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify license, ip, data right, consent, purpose, export and use term?", "kind": "relationship", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "license-ip-data-right-consent-purpose-export-and-use-term-data", "name": "License, IP, data right, consent, purpose, export and use term data", "description": "Typed benchmark data for license, ip, data right, consent, purpose, export and use term, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ] } ], "artifacts": [ { "id": "license-ip-data-right-consent-purpose-export-and-use-term-record", "name": "License, IP, data right, consent, purpose, export and use term record", "description": "Immutable or successor-versioned evidence for license, ip, data right, consent, purpose, export and use term.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for license-ip-data-right-consent-purpose-export-and-use-term; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ] } ], "inline_only_rationale": null }, { "id": "personal-sensitive-data-deidentification-security-access-and-disclosure", "name": "Personal or sensitive data, deidentification, security, access and disclosure", "description": "Records personal or sensitive data, deidentification, security, access and disclosure as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ], "questions": [ { "id": "personal-sensitive-data-deidentification-security-access-and-disclosure-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish personal or sensitive data, deidentification, security, access and disclosure?", "kind": "privacy", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "personal-sensitive-data-deidentification-security-access-and-disclosure-q02", "text": "Who may create, assert, review, approve, correct or disclose personal or sensitive data, deidentification, security, access and disclosure, under which purpose and authority?", "kind": "security", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "personal-sensitive-data-deidentification-security-access-and-disclosure-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify personal or sensitive data, deidentification, security, access and disclosure?", "kind": "authority", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "personal-sensitive-data-deidentification-security-access-and-disclosure-data", "name": "Personal or sensitive data, deidentification, security, access and disclosure data", "description": "Typed benchmark data for personal or sensitive data, deidentification, security, access and disclosure, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ] } ], "artifacts": [ { "id": "personal-sensitive-data-deidentification-security-access-and-disclosure-record", "name": "Personal or sensitive data, deidentification, security, access and disclosure record", "description": "Immutable or successor-versioned evidence for personal or sensitive data, deidentification, security, access and disclosure.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for personal-sensitive-data-deidentification-security-access-and-disclosure; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-002", "SRC-003", "SRC-004", "SRC-005", "SRC-013" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "protocol-metrics-scoring-and-statistics", "name": "Protocol, metrics, scoring and statistics", "description": "Groups governed context for protocol, metrics, scoring and statistics.", "rationale": "Comparable evaluation depends on an explicit execution contract and statistical interpretation.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ], "layers": [ { "id": "execution-protocol-interface-and-environment", "name": "Execution protocol, interface and environment", "description": "Groups benchmark context for execution protocol, interface and environment.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ], "findings": [ { "id": "evaluated-interface-task-contract-prompt-template-and-tool-access", "name": "Evaluated interface, task contract, prompt template and tool access", "description": "Records evaluated interface, task contract, prompt template and tool access as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ], "questions": [ { "id": "evaluated-interface-task-contract-prompt-template-and-tool-access-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish evaluated interface, task contract, prompt template and tool access?", "kind": "requirement", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "evaluated-interface-task-contract-prompt-template-and-tool-access-q02", "text": "Who may create, assert, review, approve, correct or disclose evaluated interface, task contract, prompt template and tool access, under which purpose and authority?", "kind": "retention", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "evaluated-interface-task-contract-prompt-template-and-tool-access-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify evaluated interface, task contract, prompt template and tool access?", "kind": "requirement", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "evaluated-interface-task-contract-prompt-template-and-tool-access-data", "name": "Evaluated interface, task contract, prompt template and tool access data", "description": "Typed benchmark data for evaluated interface, task contract, prompt template and tool access, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ] } ], "artifacts": [ { "id": "evaluated-interface-task-contract-prompt-template-and-tool-access-record", "name": "Evaluated interface, task contract, prompt template and tool access record", "description": "Immutable or successor-versioned evidence for evaluated interface, task contract, prompt template and tool access.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for evaluated-interface-task-contract-prompt-template-and-tool-access; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ] } ], "inline_only_rationale": null }, { "id": "inference-setting-environment-repetition-randomness-resource-and-timing", "name": "Inference setting, environment, repetition, randomness, resource and timing", "description": "Records inference setting, environment, repetition, randomness, resource and timing as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ], "questions": [ { "id": "inference-setting-environment-repetition-randomness-resource-and-timing-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish inference setting, environment, repetition, randomness, resource and timing?", "kind": "temporal", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "inference-setting-environment-repetition-randomness-resource-and-timing-q02", "text": "Who may create, assert, review, approve, correct or disclose inference setting, environment, repetition, randomness, resource and timing, under which purpose and authority?", "kind": "interoperability", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "inference-setting-environment-repetition-randomness-resource-and-timing-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify inference setting, environment, repetition, randomness, resource and timing?", "kind": "constraint", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "inference-setting-environment-repetition-randomness-resource-and-timing-data", "name": "Inference setting, environment, repetition, randomness, resource and timing data", "description": "Typed benchmark data for inference setting, environment, repetition, randomness, resource and timing, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ] } ], "artifacts": [ { "id": "inference-setting-environment-repetition-randomness-resource-and-timing-record", "name": "Inference setting, environment, repetition, randomness, resource and timing record", "description": "Immutable or successor-versioned evidence for inference setting, environment, repetition, randomness, resource and timing.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for inference-setting-environment-repetition-randomness-resource-and-timing; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ] } ], "inline_only_rationale": null } ] }, { "id": "metrics-thresholds-scoring-aggregation-and-uncertainty", "name": "Metrics, thresholds, scoring, aggregation and uncertainty", "description": "Groups benchmark context for metrics, thresholds, scoring, aggregation and uncertainty.", "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ], "findings": [ { "id": "metric-definition-direction-unit-threshold-rubric-and-valid-range", "name": "Metric definition, direction, unit, threshold, rubric and valid range", "description": "Records metric definition, direction, unit, threshold, rubric and valid range as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ], "questions": [ { "id": "metric-definition-direction-unit-threshold-rubric-and-valid-range-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish metric definition, direction, unit, threshold, rubric and valid range?", "kind": "measurement", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "metric-definition-direction-unit-threshold-rubric-and-valid-range-q02", "text": "Who may create, assert, review, approve, correct or disclose metric definition, direction, unit, threshold, rubric and valid range, under which purpose and authority?", "kind": "decision", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "metric-definition-direction-unit-threshold-rubric-and-valid-range-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify metric definition, direction, unit, threshold, rubric and valid range?", "kind": "event", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "metric-definition-direction-unit-threshold-rubric-and-valid-range-data", "name": "Metric definition, direction, unit, threshold, rubric and valid range data", "description": "Typed benchmark data for metric definition, direction, unit, threshold, rubric and valid range, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ] } ], "artifacts": [ { "id": "metric-definition-direction-unit-threshold-rubric-and-valid-range-record", "name": "Metric definition, direction, unit, threshold, rubric and valid range record", "description": "Immutable or successor-versioned evidence for metric definition, direction, unit, threshold, rubric and valid range.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for metric-definition-direction-unit-threshold-rubric-and-valid-range; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ] } ], "inline_only_rationale": null }, { "id": "item-score-aggregation-weight-statistic-interval-significance-and-uncertainty", "name": "Item score, aggregation, weight, statistic, interval, significance and uncertainty", "description": "Records item score, aggregation, weight, statistic, interval, significance and uncertainty as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ], "questions": [ { "id": "item-score-aggregation-weight-statistic-interval-significance-and-uncertainty-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish item score, aggregation, weight, statistic, interval, significance and uncertainty?", "kind": "quality", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "item-score-aggregation-weight-statistic-interval-significance-and-uncertainty-q02", "text": "Who may create, assert, review, approve, correct or disclose item score, aggregation, weight, statistic, interval, significance and uncertainty, under which purpose and authority?", "kind": "state", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "item-score-aggregation-weight-statistic-interval-significance-and-uncertainty-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify item score, aggregation, weight, statistic, interval, significance and uncertainty?", "kind": "temporal", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "item-score-aggregation-weight-statistic-interval-significance-and-uncertainty-data", "name": "Item score, aggregation, weight, statistic, interval, significance and uncertainty data", "description": "Typed benchmark data for item score, aggregation, weight, statistic, interval, significance and uncertainty, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ] } ], "artifacts": [ { "id": "item-score-aggregation-weight-statistic-interval-significance-and-uncertainty-record", "name": "Item score, aggregation, weight, statistic, interval, significance and uncertainty record", "description": "Immutable or successor-versioned evidence for item score, aggregation, weight, statistic, interval, significance and uncertainty.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for item-score-aggregation-weight-statistic-interval-significance-and-uncertainty; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "validity-contamination-robustness-fairness-and-limitations", "name": "Validity, contamination, robustness, fairness and limitations", "description": "Groups governed context for validity, contamination, robustness, fairness and limitations.", "rationale": "Benchmark fitness requires explicit failure, subgroup and adversarial evidence.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ], "layers": [ { "id": "quality-validity-leakage-and-contamination", "name": "Quality, validity, leakage and contamination", "description": "Groups benchmark context for quality, validity, leakage and contamination.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ], "findings": [ { "id": "completeness-label-accuracy-schema-conformance-duplicate-and-quality", "name": "Completeness, label accuracy, schema conformance, duplicate and quality", "description": "Records completeness, label accuracy, schema conformance, duplicate and quality as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ], "questions": [ { "id": "completeness-label-accuracy-schema-conformance-duplicate-and-quality-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish completeness, label accuracy, schema conformance, duplicate and quality?", "kind": "validation", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "completeness-label-accuracy-schema-conformance-duplicate-and-quality-q02", "text": "Who may create, assert, review, approve, correct or disclose completeness, label accuracy, schema conformance, duplicate and quality, under which purpose and authority?", "kind": "identity", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "completeness-label-accuracy-schema-conformance-duplicate-and-quality-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify completeness, label accuracy, schema conformance, duplicate and quality?", "kind": "composition", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "completeness-label-accuracy-schema-conformance-duplicate-and-quality-data", "name": "Completeness, label accuracy, schema conformance, duplicate and quality data", "description": "Typed benchmark data for completeness, label accuracy, schema conformance, duplicate and quality, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ] } ], "artifacts": [ { "id": "completeness-label-accuracy-schema-conformance-duplicate-and-quality-record", "name": "Completeness, label accuracy, schema conformance, duplicate and quality record", "description": "Immutable or successor-versioned evidence for completeness, label accuracy, schema conformance, duplicate and quality.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for completeness-label-accuracy-schema-conformance-duplicate-and-quality; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ] } ], "inline_only_rationale": null }, { "id": "overlap-leakage-contamination-memorization-gaming-and-detection", "name": "Overlap, leakage, contamination, memorization, gaming and detection", "description": "Records overlap, leakage, contamination, memorization, gaming and detection as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ], "questions": [ { "id": "overlap-leakage-contamination-memorization-gaming-and-detection-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish overlap, leakage, contamination, memorization, gaming and detection?", "kind": "security", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "overlap-leakage-contamination-memorization-gaming-and-detection-q02", "text": "Who may create, assert, review, approve, correct or disclose overlap, leakage, contamination, memorization, gaming and detection, under which purpose and authority?", "kind": "classification", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "overlap-leakage-contamination-memorization-gaming-and-detection-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify overlap, leakage, contamination, memorization, gaming and detection?", "kind": "evidence", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "overlap-leakage-contamination-memorization-gaming-and-detection-data", "name": "Overlap, leakage, contamination, memorization, gaming and detection data", "description": "Typed benchmark data for overlap, leakage, contamination, memorization, gaming and detection, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ] } ], "artifacts": [ { "id": "overlap-leakage-contamination-memorization-gaming-and-detection-record", "name": "Overlap, leakage, contamination, memorization, gaming and detection record", "description": "Immutable or successor-versioned evidence for overlap, leakage, contamination, memorization, gaming and detection.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for overlap-leakage-contamination-memorization-gaming-and-detection; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ] } ], "inline_only_rationale": null } ] }, { "id": "subgroups-robustness-safety-and-limitations", "name": "Subgroups, robustness, safety and limitations", "description": "Groups benchmark context for subgroups, robustness, safety and limitations.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ], "findings": [ { "id": "subgroup-coverage-performance-fairness-representativeness-and-disparity", "name": "Subgroup coverage, performance, fairness, representativeness and disparity", "description": "Records subgroup coverage, performance, fairness, representativeness and disparity as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ], "questions": [ { "id": "subgroup-coverage-performance-fairness-representativeness-and-disparity-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish subgroup coverage, performance, fairness, representativeness and disparity?", "kind": "measurement", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "subgroup-coverage-performance-fairness-representativeness-and-disparity-q02", "text": "Who may create, assert, review, approve, correct or disclose subgroup coverage, performance, fairness, representativeness and disparity, under which purpose and authority?", "kind": "relationship", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "subgroup-coverage-performance-fairness-representativeness-and-disparity-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify subgroup coverage, performance, fairness, representativeness and disparity?", "kind": "ownership", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "subgroup-coverage-performance-fairness-representativeness-and-disparity-data", "name": "Subgroup coverage, performance, fairness, representativeness and disparity data", "description": "Typed benchmark data for subgroup coverage, performance, fairness, representativeness and disparity, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ] } ], "artifacts": [ { "id": "subgroup-coverage-performance-fairness-representativeness-and-disparity-record", "name": "Subgroup coverage, performance, fairness, representativeness and disparity record", "description": "Immutable or successor-versioned evidence for subgroup coverage, performance, fairness, representativeness and disparity.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for subgroup-coverage-performance-fairness-representativeness-and-disparity; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ] } ], "inline_only_rationale": null }, { "id": "adversarial-robustness-privacy-security-safety-limit-and-misuse", "name": "Adversarial robustness, privacy, security, safety, limit and misuse", "description": "Records adversarial robustness, privacy, security, safety, limit and misuse as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ], "questions": [ { "id": "adversarial-robustness-privacy-security-safety-limit-and-misuse-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish adversarial robustness, privacy, security, safety, limit and misuse?", "kind": "constraint", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "adversarial-robustness-privacy-security-safety-limit-and-misuse-q02", "text": "Who may create, assert, review, approve, correct or disclose adversarial robustness, privacy, security, safety, limit and misuse, under which purpose and authority?", "kind": "authority", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "adversarial-robustness-privacy-security-safety-limit-and-misuse-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify adversarial robustness, privacy, security, safety, limit and misuse?", "kind": "measurement", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "adversarial-robustness-privacy-security-safety-limit-and-misuse-data", "name": "Adversarial robustness, privacy, security, safety, limit and misuse data", "description": "Typed benchmark data for adversarial robustness, privacy, security, safety, limit and misuse, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ] } ], "artifacts": [ { "id": "adversarial-robustness-privacy-security-safety-limit-and-misuse-record", "name": "Adversarial robustness, privacy, security, safety, limit and misuse record", "description": "Immutable or successor-versioned evidence for adversarial robustness, privacy, security, safety, limit and misuse.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for adversarial-robustness-privacy-security-safety-limit-and-misuse; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-011", "SRC-012" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "lifecycle-results-governance-and-projections", "name": "Lifecycle, results, governance and projections", "description": "Groups governed context for lifecycle, results, governance and projections.", "rationale": "Lifecycle and result bindings remain authority-qualified and auditable.", "source_refs": [ "SRC-001", "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-009", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ], "layers": [ { "id": "release-validation-correction-and-result-bindings", "name": "Release, validation, correction and result bindings", "description": "Groups benchmark context for release, validation, correction and result bindings.", "source_refs": [ "SRC-001", "SRC-004", "SRC-006", "SRC-007", "SRC-009", "SRC-010" ], "findings": [ { "id": "draft-candidate-validated-released-active-deprecated-withdrawn-and-superseded", "name": "Draft, candidate, validated, released, active, deprecated, withdrawn and superseded", "description": "Records draft, candidate, validated, released, active, deprecated, withdrawn and superseded as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-004", "SRC-006", "SRC-007", "SRC-009", "SRC-010" ], "questions": [ { "id": "draft-candidate-validated-released-active-deprecated-withdrawn-and-superseded-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish draft, candidate, validated, released, active, deprecated, withdrawn and superseded?", "kind": "lifecycle", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "draft-candidate-validated-released-active-deprecated-withdrawn-and-superseded-q02", "text": "Who may create, assert, review, approve, correct or disclose draft, candidate, validated, released, active, deprecated, withdrawn and superseded, under which purpose and authority?", "kind": "requirement", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "draft-candidate-validated-released-active-deprecated-withdrawn-and-superseded-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify draft, candidate, validated, released, active, deprecated, withdrawn and superseded?", "kind": "exception", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "draft-candidate-validated-released-active-deprecated-withdrawn-and-superseded-data", "name": "Draft, candidate, validated, released, active, deprecated, withdrawn and superseded data", "description": "Typed benchmark data for draft, candidate, validated, released, active, deprecated, withdrawn and superseded, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-004", "SRC-006", "SRC-007", "SRC-009", "SRC-010" ] } ], "artifacts": [ { "id": "draft-candidate-validated-released-active-deprecated-withdrawn-and-superseded-record", "name": "Draft, candidate, validated, released, active, deprecated, withdrawn and superseded record", "description": "Immutable or successor-versioned evidence for draft, candidate, validated, released, active, deprecated, withdrawn and superseded.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for draft-candidate-validated-released-active-deprecated-withdrawn-and-superseded; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-004", "SRC-006", "SRC-007", "SRC-009", "SRC-010" ] } ], "inline_only_rationale": null }, { "id": "evaluation-run-result-leaderboard-baseline-claim-correction-and-binding", "name": "Evaluation run, result, leaderboard, baseline, claim, correction and binding", "description": "Records evaluation run, result, leaderboard, baseline, claim, correction and binding as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-001", "SRC-004", "SRC-006", "SRC-007", "SRC-009", "SRC-010" ], "questions": [ { "id": "evaluation-run-result-leaderboard-baseline-claim-correction-and-binding-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish evaluation run, result, leaderboard, baseline, claim, correction and binding?", "kind": "relationship", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "evaluation-run-result-leaderboard-baseline-claim-correction-and-binding-q02", "text": "Who may create, assert, review, approve, correct or disclose evaluation run, result, leaderboard, baseline, claim, correction and binding, under which purpose and authority?", "kind": "constraint", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "evaluation-run-result-leaderboard-baseline-claim-correction-and-binding-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify evaluation run, result, leaderboard, baseline, claim, correction and binding?", "kind": "provenance", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "evaluation-run-result-leaderboard-baseline-claim-correction-and-binding-data", "name": "Evaluation run, result, leaderboard, baseline, claim, correction and binding data", "description": "Typed benchmark data for evaluation run, result, leaderboard, baseline, claim, correction and binding, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-004", "SRC-006", "SRC-007", "SRC-009", "SRC-010" ] } ], "artifacts": [ { "id": "evaluation-run-result-leaderboard-baseline-claim-correction-and-binding-record", "name": "Evaluation run, result, leaderboard, baseline, claim, correction and binding record", "description": "Immutable or successor-versioned evidence for evaluation run, result, leaderboard, baseline, claim, correction and binding.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for evaluation-run-result-leaderboard-baseline-claim-correction-and-binding; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-001", "SRC-004", "SRC-006", "SRC-007", "SRC-009", "SRC-010" ] } ], "inline_only_rationale": null } ] }, { "id": "access-retention-audit-and-interoperability", "name": "Access, retention, audit and interoperability", "description": "Groups benchmark context for access, retention, audit and interoperability.", "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ], "findings": [ { "id": "role-purpose-access-hidden-test-audit-retention-hold-and-tombstone", "name": "Role, purpose, access, hidden test, audit, retention, hold and tombstone", "description": "Records role, purpose, access, hidden test, audit, retention, hold and tombstone as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ], "questions": [ { "id": "role-purpose-access-hidden-test-audit-retention-hold-and-tombstone-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish role, purpose, access, hidden test, audit, retention, hold and tombstone?", "kind": "retention", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "role-purpose-access-hidden-test-audit-retention-hold-and-tombstone-q02", "text": "Who may create, assert, review, approve, correct or disclose role, purpose, access, hidden test, audit, retention, hold and tombstone, under which purpose and authority?", "kind": "event", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "role-purpose-access-hidden-test-audit-retention-hold-and-tombstone-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify role, purpose, access, hidden test, audit, retention, hold and tombstone?", "kind": "process", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "role-purpose-access-hidden-test-audit-retention-hold-and-tombstone-data", "name": "Role, purpose, access, hidden test, audit, retention, hold and tombstone data", "description": "Typed benchmark data for role, purpose, access, hidden test, audit, retention, hold and tombstone, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ] } ], "artifacts": [ { "id": "role-purpose-access-hidden-test-audit-retention-hold-and-tombstone-record", "name": "Role, purpose, access, hidden test, audit, retention, hold and tombstone record", "description": "Immutable or successor-versioned evidence for role, purpose, access, hidden test, audit, retention, hold and tombstone.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for role-purpose-access-hidden-test-audit-retention-hold-and-tombstone; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ] } ], "inline_only_rationale": null }, { "id": "dcat-croissant-spdx-huggingface-mlperf-prov-skos-projection-and-loss", "name": "DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS projection and loss", "description": "Records dcat, croissant, spdx, hugging face, mlperf, prov and skos projection and loss as versioned benchmark-definition context while evaluation runs, models, results, leaderboards, policies and records masters remain external.", "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ], "questions": [ { "id": "dcat-croissant-spdx-huggingface-mlperf-prov-skos-projection-and-loss-q01", "text": "What stable identity, benchmark release, declared value and explicit unknown establish dcat, croissant, spdx, hugging face, mlperf, prov and skos projection and loss?", "kind": "interoperability", "answer_data": [ "benchmark and release identifiers", "typed value and applicable scope", "unknown, withheld and not-applicable states" ] }, { "id": "dcat-croissant-spdx-huggingface-mlperf-prov-skos-projection-and-loss-q02", "text": "Who may create, assert, review, approve, correct or disclose dcat, croissant, spdx, hugging face, mlperf, prov and skos projection and loss, under which purpose and authority?", "kind": "temporal", "answer_data": [ "owner, steward, contributor and reviewer", "purpose, authority, policy and access", "exception, contest and escalation path" ] }, { "id": "dcat-croissant-spdx-huggingface-mlperf-prov-skos-projection-and-loss-q03", "text": "Which effective, observed, recorded, ingested and knowledge times, evidence and uncertainty qualify dcat, croissant, spdx, hugging face, mlperf, prov and skos projection and loss?", "kind": "validation", "answer_data": [ "distinct event and knowledge times", "source, method, evidence and uncertainty", "successor, correction, retention and audit" ] } ], "data_elements": [ { "id": "dcat-croissant-spdx-huggingface-mlperf-prov-skos-projection-and-loss-data", "name": "DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS projection and loss data", "description": "Typed benchmark data for dcat, croissant, spdx, hugging face, mlperf, prov and skos projection and loss, qualified by release, scope, source, authority, time, evidence and provenance.", "value_kind": "collection", "cardinality": "1", "required": true, "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ] } ], "artifacts": [ { "id": "dcat-croissant-spdx-huggingface-mlperf-prov-skos-projection-and-loss-record", "name": "DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS projection and loss record", "description": "Immutable or successor-versioned evidence for dcat, croissant, spdx, hugging face, mlperf, prov and skos projection and loss.", "media_or_form": [ "logical benchmark assertion", "dataset, protocol, metric, validation, lifecycle or projection record" ], "serial": true, "identity_strategy": "Benchmark release ID plus independent assertion, item, decision or artifact ID for dcat-croissant-spdx-huggingface-mlperf-prov-skos-projection-and-loss; name, split label, URL, digest and timestamp never identify it alone.", "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ] } ], "inline_only_rationale": null } ] } ] } ] }, "functions": [ { "id": "register-benchmark", "name": "Register a benchmark definition", "description": "Governed operation to register a benchmark definition without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "benchmark mandate", "owner", "scope" ], "outputs": [ "stable benchmark root" ], "preconditions": [ "namespace, owner, purpose and duplicate checks pass" ], "effects": [ "a new definition head exists without creating an evaluation run" ], "source_refs": [ "SRC-006", "SRC-007", "SRC-009" ] }, { "id": "compose-versioned-items", "name": "Compose or import versioned items", "description": "Governed operation to compose or import versioned items without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "benchmark release", "source items", "schema" ], "outputs": [ "versioned item set" ], "preconditions": [ "rights, lineage, schema, digest and duplicate checks pass" ], "effects": [ "items are bound to a release without obscuring their sources" ], "source_refs": [ "SRC-006", "SRC-007", "SRC-008", "SRC-011", "SRC-012", "SRC-013" ] }, { "id": "define-splits-and-sampling", "name": "Define split roles and sampling", "description": "Governed operation to define split roles and sampling without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "item population", "sampling plan", "split roles" ], "outputs": [ "versioned splits and distributions" ], "preconditions": [ "population, exclusion, stratification, leakage and hidden-test checks pass" ], "effects": [ "split membership and weights become reproducible release assertions" ], "source_refs": [ "SRC-002", "SRC-003", "SRC-008", "SRC-010" ] }, { "id": "annotate-and-adjudicate", "name": "Annotate and adjudicate reference data", "description": "Governed operation to annotate and adjudicate reference data without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "items", "annotation guide", "annotators" ], "outputs": [ "labels, references, rubrics and uncertainty" ], "preconditions": [ "qualification, independence, conflict and quality checks pass" ], "effects": [ "reference data retains attribution, disagreement and adjudication" ], "source_refs": [ "SRC-001", "SRC-004", "SRC-011", "SRC-012" ] }, { "id": "define-protocol", "name": "Define the evaluation protocol", "description": "Governed operation to define the evaluation protocol without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "task", "interface", "environment controls" ], "outputs": [ "executable protocol definition" ], "preconditions": [ "input, output, prompt, tools, repetition, randomness and timing checks pass" ], "effects": [ "execution conditions become reproducible without executing a model" ], "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ] }, { "id": "define-scoring", "name": "Define metrics, scoring and aggregation", "description": "Governed operation to define metrics, scoring and aggregation without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "protocol", "metric definitions", "statistical plan" ], "outputs": [ "versioned scoring contract" ], "preconditions": [ "direction, unit, threshold, weighting, uncertainty and validity checks pass" ], "effects": [ "scores can be interpreted without owning evaluation results" ], "source_refs": [ "SRC-001", "SRC-004", "SRC-009", "SRC-010" ] }, { "id": "validate-benchmark", "name": "Validate quality, leakage and contamination", "description": "Governed operation to validate quality, leakage and contamination without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "candidate release", "validation plan", "comparison evidence" ], "outputs": [ "validation findings and disposition" ], "preconditions": [ "quality, overlap, contamination, subgroup, robustness and gaming checks pass" ], "effects": [ "limitations and unresolved risks remain visible" ], "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004", "SRC-012" ] }, { "id": "transition-release", "name": "Release, deprecate, withdraw or supersede", "description": "Governed operation to release, deprecate, withdraw or supersede without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "benchmark head", "authorized decision", "reason and successor" ], "outputs": [ "successor lifecycle assertion" ], "preconditions": [ "authority, review, notification, access and retention checks pass" ], "effects": [ "discoverability changes while prior releases remain reconstructable" ], "source_refs": [ "SRC-004", "SRC-006", "SRC-007", "SRC-009" ] }, { "id": "bind-results", "name": "Bind evaluation results and leaderboard observations", "description": "Governed operation to bind evaluation results and leaderboard observations without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "benchmark release", "external run or result", "source-qualified observation" ], "outputs": [ "non-owning result binding" ], "preconditions": [ "identity, version, metric, source, review and freshness checks pass" ], "effects": [ "the benchmark references results without absorbing execution or claim ownership" ], "source_refs": [ "SRC-007", "SRC-009", "SRC-010" ] }, { "id": "correct-project-and-audit", "name": "Correct, project, disclose, retain and audit", "description": "Governed operation to correct, project, disclose, retain and audit without autonomous release, hidden-test disclosure, rights waiver, access expansion, benchmark manipulation or destructive cleanup.", "inputs": [ "benchmark head", "target profile", "governance policy" ], "outputs": [ "successor, projection, disclosure or audit event" ], "preconditions": [ "purpose, authority, access, mapping version, loss, hold and idempotency checks pass" ], "effects": [ "context remains reconstructable and explicit about loss and current head" ], "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014", "SRC-015" ] } ], "composition": [ { "target": "WM-DAT-001 Dataset", "relation": "REFERENCE", "purpose": "Represent the unfrozen parent boundary as a non-owning dataset reference until an extension edge is approved.", "required": true, "source_refs": [ "SRC-006", "SRC-007", "SRC-008", "SRC-013" ] }, { "target": "WM-AI-003 Evaluation Run, AI Model, Result and Leaderboard", "relation": "REFERENCE", "purpose": "Resolve evaluated subjects, executions and outcomes without absorbing their masters.", "required": false, "source_refs": [ "SRC-001", "SRC-002", "SRC-004", "SRC-009", "SRC-010" ] }, { "target": "Policy, credential, provenance, audit and records models", "relation": "REFERENCE", "purpose": "Resolve authoritative control and evidence records without transferring ownership.", "required": false, "source_refs": [ "SRC-001", "SRC-004", "SRC-005", "SRC-007" ] }, { "target": "DCAT 3, Croissant 1.0, SPDX 3.0.1 Dataset, Hugging Face Dataset Cards, MLPerf policies, PROV-O and SKOS", "relation": "ALIGN", "purpose": "Project release-pinned catalog, dataset, supply-chain, card, benchmark, provenance and classification views with declared loss.", "required": false, "source_refs": [ "SRC-006", "SRC-007", "SRC-008", "SRC-010", "SRC-011", "SRC-013", "SRC-014" ] } ], "serviceLayers": { "dimension": { "owner_package_requirements": [ "Dimension owner, benchmark mandate and accountable AI system owner", "Authoritative dataset, model, evaluation-run, result, leaderboard, policy, identity, provenance, audit and records registries", "Approved task, domain, jurisdiction, rights, privacy, security, safety, quality, release, access, retention and interoperability profiles", "Role, delegation, review, publication, disclosure, incident and agent-operation policies" ], "namespace_guidance": "Mint benchmark, release, item, split, annotation, protocol, metric, validation, lifecycle, result-binding, disclosure and projection IDs; preserve all external master identifiers.", "registry_links": [ "https://ver.cy/models/", "https://ver.cy/model-agent-protocol.md" ] }, "canon_and_patch": { "canonicalization_rules": [ "Canonicalize one benchmark by authoritative benchmark identifier and namespace, never by display name, task, split label, metric, URL or score alone.", "Keep benchmark definition, release, item, dataset, model, evaluation run, result, leaderboard, claim and policy independently identifiable." ], "patch_rules": [ "Extensions declare task, modality, population, jurisdiction, rights, privacy, security, safety, scoring, release and interoperability effects.", "Released items, splits, labels, protocols, metrics and decisions are immutable; corrections create linked successors.", "Never silently change benchmark scope, content, split roles, references, protocol, metrics, thresholds, rights, access or provenance." ], "compatibility_rules": [ "Ignore additive fields only when benchmark and release identity, content, split role, protocol, metric, source, authority, access and provenance survive.", "Every projection pins standard, implementation, schema, vocabulary, policy and mapping versions and declares information loss." ] }, "artifact_rules": { "identity_priority": [ "Authoritative master-system identifier for each benchmark, release, item, assertion, decision or projection, qualified by issuer, namespace and record kind.", "Governed globally resolvable benchmark-release IRI.", "Dimension UUID or ULID when neither preceding identifier exists." ], "timestamp_rule": "Use RFC 3339 timestamps with seconds and explicit offset or Z; distinguish created, collected, annotated, validated, released, deprecated, withdrawn, observed, recorded, ingested and knowledge times whenever they differ.", "serial_naming_rule": "Use {benchmark-id}--{release-id}--{item-assertion-decision-or-artifact-id}--{artifact-kind}--{revision-id}.", "integrity_rule": "Store digest, media type, record kind, benchmark and release scope, content and schema version, actor, event and knowledge times, access marking and provenance." }, "policies": [ "The benchmark owns its governed definition and release context but not Dataset, AI Model, Evaluation Run, Result, Leaderboard, Policy, Credential, Provenance, Audit or Records masters.", "Train, validation, calibration and evaluation are explicit roles; one split label cannot prove absence of leakage or contamination.", "A high score, checksum, signature or published leaderboard does not by itself prove validity, safety, fairness, ownership, authorization or fitness.", "Agents cannot release, disclose hidden tests, waive rights, widen access, manipulate a benchmark or destroy records outside explicit delegated authority." ], "crud": { "read": [ "Resolve purpose, current release, population, items, labels, splits, lineage, rights, protocol, metrics, validity, limitations, lifecycle, result bindings, access, retention and projection loss under the permitted view." ], "create": [ "Bind stable benchmark identity, namespace, owner, purpose, task, population, source, authority and initial state before adding content." ], "update": [ "Append successor content, annotation, split, protocol, metric, validation, lifecycle, correction, disclosure and projection assertions with reason, authority, expected revision, event time and knowledge time." ], "delete": [ "Apply rights, privacy, security, legal-hold and adopting-Dimension records policy; withdraw or tombstone only the benchmark release, never cascade to dataset, model, run, result or records masters, and let authoritative systems execute physical disposition." ] }, "roles": [ { "name": "Benchmark owner and accountable AI system owner", "responsibilities": [ "Own purpose, scope, risk acceptance, release boundaries and accountable use." ] }, { "name": "Dataset and item steward", "responsibilities": [ "Supply source-qualified items, schemas, splits, lineage, rights and quality evidence." ] }, { "name": "Annotator and adjudicator", "responsibilities": [ "Create reference answers, labels and rubrics while preserving disagreement and uncertainty." ] }, { "name": "Evaluation methodologist and statistician", "responsibilities": [ "Define reproducible protocol, metrics, aggregation, uncertainty and validity tests." ] }, { "name": "Independent safety, security and fairness reviewer", "responsibilities": [ "Review contamination, gaming, subgroups, robustness, privacy, security, safety and limitations." ] }, { "name": "Release and disclosure authority", "responsibilities": [ "Make attributable release, exception, correction, withdrawal and disclosure decisions." ] }, { "name": "Legal, privacy and records steward", "responsibilities": [ "Own rights, consent, export, privacy, access, hold, retention and disposition profiles." ] } ], "access": { "default_rule": "Deny hidden tests, personal or sensitive data, restricted source material, secrets, security findings and unpublished validation evidence unless a purpose-bound policy permits the minimum necessary view.", "scopes": [ "bundle", "layer", "finding", "artifact" ], "exceptions": [ "Declared validation, audit, incident, legal, subject-rights, safety or emergency access must cite authority, scope, purpose and time limit and must be logged." ], "audit_requirements": [ "Log actor, agent, role, purpose, benchmark and release, operation, authority, policy, RFC 3339 time, affected fields, source revision and outcome without unnecessary restricted-content duplication." ] }, "agents_bootstrap": { "filename": "AGENTS.md", "required_fields": [ "Name", "Type", "Specification URL", "Storage type URL", "Interface URL", "Processes URL" ], "read_order": [ "Read Dimension AI, benchmark, dataset, evaluation, data-rights, privacy, security, safety, access, retention and agent policies.", "Read this benchmark and linked dataset, model, evaluation-run, result, leaderboard, provenance and records models before mutation." ] } }, "coverage": { "claim": "WM-AI-009 covers one governed evaluation dataset or benchmark definition from identity and versioned content through items, labels, splits, lineage, rights, protocol, metrics, validity, contamination, subgroups, lifecycle, result bindings, access, retention and projections. Task-specific validity, jurisdiction, release-pinned mappings and independent external review remain deferred.", "confidence": "medium", "checklist": [ { "dimension": "identity", "status": "covered", "notes": "Identity is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "classification and direct properties", "status": "covered", "notes": "Classification and direct properties is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "recognition and observation", "status": "covered", "notes": "Recognition and observation is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "capabilities and possible actions", "status": "covered", "notes": "Capabilities and possible actions is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "composition", "status": "covered", "notes": "Composition is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "lifecycle", "status": "covered", "notes": "Lifecycle is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "relationships", "status": "covered", "notes": "Relationships is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "temporal", "status": "covered", "notes": "Temporal is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "spatial", "status": "covered", "notes": "Spatial is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "provenance", "status": "covered", "notes": "Provenance is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "ownership and stewardship", "status": "covered", "notes": "Ownership and stewardship is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "validation and quality", "status": "covered", "notes": "Validation and quality is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "access and privacy", "status": "covered", "notes": "Access and privacy is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "retention and deletion", "status": "covered", "notes": "Retention and deletion is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "interoperability", "status": "covered", "notes": "Interoperability is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." }, { "dimension": "authority and ethics", "status": "covered", "notes": "Authority and ethics is explicit; task-specific validity, jurisdiction, rights, release-pinned mappings and independent external review remain held where applicable." } ], "known_omissions": [ "Claude and Grok each timed out on one bounded attempt; no independent external result was admitted.", "The unified parent_ids WM-DAT-001 has no approved frozen extension edge; this result records only a non-owning reference.", "The incoming candidate WM-AI-003 reference grants no ownership of evaluation runs, results, leaderboards or claims.", "Task-specific psychometric, scientific, cultural, accessibility and domain-validity profiles require separate research." ], "conflicts": [ "Benchmark definition, dataset release, evaluation run, evaluated model, result, leaderboard and claim are not interchangeable identities.", "Train, validation, calibration and evaluation split labels do not by themselves prove non-overlap, non-contamination or fitness.", "Reference answer, rubric, metric, threshold and aggregate score encode different judgments and must remain versioned." ], "regional_assumptions": [ "Rights, consent, privacy, security, export, testing, disclosure, retention and high-risk AI obligations depend on jurisdiction, sector and use case.", "The EU AI Act and GDPR are European Union profiles; NIST publications are voluntary United States guidance unless adopted by policy or contract.", "DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS are versioned profiles and are not assumed lossless or universally applicable." ], "adversarial_checks": [ "Reject a benchmark without stable identity, release, owner, purpose, task, population, source, authority and current head.", "Reject hidden-test disclosure or access expansion without explicit delegated authority and audit.", "Reject claims of validity or safety inferred only from score, popularity, checksum, signature or publication.", "Reject evaluation results that omit model version, benchmark release, protocol, metric, environment, sample scope and uncertainty.", "Reject a projection that collapses benchmark, dataset, run, result and leaderboard identities or hides information loss." ] }, "researchAdjudication": { "providerMode": "single-provider-waiver", "activeProviders": [ "codex" ], "waivedProviders": [ "claude", "grok" ], "providerPolicy": { "contract_version": "1.0.0", "mode": "single-provider-waiver", "effective_at": "2026-09-06T00:00:00Z", "scope": "Canonical single-stream subject-model research after the six-workstream consolidation", "active_providers": [ "codex" ], "waived_providers": [ { "provider": "claude", "authorized_by": "repository owner", "authorized_at": "2026-09-06T00:00:00Z", "reason": "Claude produced no result on prior 1800-second and 900-second attempts and again timed out on bounded 600-second Sonnet and 300-second Haiku passes. The owner prioritized completion over provider availability." }, { "provider": "grok", "authorized_by": "repository owner", "authorized_at": "2026-09-06T00:00:00Z", "reason": "The repository owner authorized completion without Grok when Grok is unavailable, slow or schema-invalid. Grok may still be attempted as a bounded supplemental reviewer, but its failure never blocks a valid Claude plus no-tools result." } ], "review_rule": "Codex may complete source-grounded fallback research after bounded Claude and Grok attempts fail. It requires a separate no-tools adversarial audit and remains reviewable-draft with a visible absence-of-external-review hold.", "supplemental_provider_attempts": [ { "provider": "claude", "required": false, "maximum_attempts": 1, "failure_policy": "record-and-continue", "admission_rule": "Use only a locally schema-valid result whose sources and boundaries survive adjudication." }, { "provider": "grok", "required": false, "maximum_attempts": 1, "failure_policy": "record-and-continue", "admission_rule": "Use only a locally schema-valid result whose sources and boundaries survive adjudication." } ] }, "boundaryDecision": { "entry_kind": "aggregate", "status": "accepted", "rationale": "The root combines a versioned evaluation dataset and benchmark definition with multiple governed component identities, release semantics and protocol contracts. Aggregate is more accurate than a generic dataset, execution event, result, leaderboard or document. The frozen catalogue entry_kind extension is a separate axis." }, "decisions": [ { "concept": "Benchmark definition versus evaluation execution", "disposition": "accepted", "rationale": "The benchmark owns versioned definition, content, split roles, protocol and scoring. Evaluation runs own execution inputs, environment, outputs and measurements." }, { "concept": "Unified parent signal WM-DAT-001", "disposition": "accepted-as-unapproved-reference", "rationale": "The unified row identifies Dataset as the parent boundary but the frozen relation ledger has no approved edge. Only a non-owning reference is permitted and no inheritance, mutation, release or cascade authority follows." }, { "concept": "Incoming WM-AI-003 reference", "disposition": "accepted-as-use-direction", "rationale": "The candidate Evaluation Run reference expresses reproducibility use. It grants WM-AI-009 no ownership of runs, evaluated models, results, leaderboards or claims." }, { "concept": "Split roles and leakage", "disposition": "accepted", "rationale": "Train, validation, calibration and evaluation are explicit versioned roles. Labels alone do not prove non-overlap, non-contamination or fitness." }, { "concept": "Reference answers, rubrics and annotation", "disposition": "accepted", "rationale": "Reference data preserves creator, method, qualification, disagreement, adjudication, uncertainty, version and applicable scope." }, { "concept": "Metrics, statistics and comparability", "disposition": "accepted-with-profile-hold", "rationale": "Metric definition, direction, unit, threshold, per-item scoring, weighting, aggregation, interval, significance and uncertainty remain separate and task-profiled." }, { "concept": "Rights, privacy, security and hidden tests", "disposition": "accepted-with-mandatory-hold", "rationale": "Rights, consent, sensitive data, deidentification, security, hidden-test access, disclosure and retention require purpose-bound adopting-Dimension policies." }, { "concept": "Validity, contamination and gaming", "disposition": "accepted", "rationale": "Quality, subgroup, representativeness, leakage, contamination, memorization, robustness, fairness, safety and gaming checks are separately evidenced and never inferred from a score." }, { "concept": "Machine-readable projections", "disposition": "accepted-with-validation-hold", "rationale": "DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS have distinct scopes; every mapping must pin release or commit and declare information loss." }, { "concept": "Single-provider waiver and no-tools audit", "disposition": "accepted-with-mandatory-hold", "rationale": "One bounded Claude Sonnet and one bounded Grok attempt each timed out after 120 seconds. Codex separately audits the frozen validated result and comparison locally without tools or new research facts; assurance remains reviewable-draft." } ], "publicationHolds": [ "Absence-of-external-review hold: one Claude Sonnet and one Grok attempt for WM-AI-009 each timed out after 120 seconds; no external research result was admitted.", "Relation hold: unified parent_ids WM-DAT-001 has no frozen approved edge and grants no inheritance, composition, ownership, mutation, release or cascade behavior.", "Evaluation-boundary hold: the incoming candidate WM-AI-003 reference grants no ownership of evaluated models, runs, results, leaderboards or claims.", "Validity-profile hold: task, modality, domain, population, language, culture, accessibility, psychometric and scientific validity require benchmark-specific profiles.", "Rights and jurisdiction hold: source rights, license, consent, IP, export, privacy, security, hidden-test disclosure, safety and retention require adopting-Dimension policies.", "Interoperability hold: all DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS projections require release or commit pins, conformance tests and loss validation.", "Source-version hold: NIST AI RMF 1.0 is under revision and the mutable MLPerf rules source requires a release or commit pin for implementation.", "Independent external review was explicitly waived by the repository owner; this codex-only result remains a reviewable draft." ], "deferredResearch": [ "Approve or reject the WM-DAT-001 parent boundary and register evaluation-run, model, result, leaderboard, policy, provenance and records relations.", "Create task, modality, domain, population, language, culture, accessibility, psychometric and scientific validity profiles.", "Validate jurisdiction and organization-specific source rights, license, consent, IP, export, privacy, security, hidden-test disclosure, safety and retention policies.", "Test release-pinned DCAT, Croissant, SPDX, Hugging Face, MLPerf, PROV and SKOS mappings with conformance, round-trip and information-loss evidence.", "Refresh NIST AI RMF and mutable MLPerf policy mappings after normative changes and obtain supplemental independent external review before canonical promotion." ] }, "statistics": { "sources": 15, "bundles": 6, "layers": 12, "findings": 24, "questions": 72, "artifacts": 24, "functions": 10 } }