# Vercy AI instruction - YAML 1.2 (JSON-compatible) { "vercy": "1.0-draft", "publication": { "status": "published", "adjudicationStatus": "reviewable-draft", "publishableCanonical": false, "generatedAt": "2026-08-26T14:41:29Z", "synthesisSha256": "8dc70d9d4534bca936ce010678ecea271c880c5fc6d924e87215a6ad766a3a8a", "providerMode": "dual-provider", "providers": [ "Claude", "Grok" ], "waivedProviders": [] }, "metaModel": { "id": "WM-AI-004", "registryId": "vr.wm-ai-004", "name": "AI Inference / Agent Run", "version": "0.3.0-research.1", "previousVersions": [], "entryKind": "event", "family": "World Models", "category": "Information and virtual systems", "industry": [ "Cross-industry" ], "domain": [ "INF.AI.RUN" ], "tags": [ "ai", "inference", "agent", "run", "inf.ai.run" ], "status": "published" }, "canonicalUrl": "https://ver.cy/models/wm-ai-004-ai-inference-agent-run/", "sourceUrl": "https://github.com/ver-cy/world-models/tree/feat/mega-model-registry/research/runs/wm-ai-004", "model": { "registry_id": "vr.wm-ai-004", "model_id": "WM-AI-004", "name": "AI Inference / Agent Run", "entry_kind": "event", "purpose": "Provide the format-neutral context structure an AI agent needs to open, execute, observe, close, audit, cost and retain a single AI inference or agent run, so that what was asked, what was configured, what was invoked, what was produced, at whose authority, at what cost and with what evidence can be reconstructed after the fact.", "scope_statement": "One bounded execution occurrence of an AI system: from the moment an invocation is accepted until the run reaches a terminal state, plus the durable record of that occurrence. The run is an occurrent (a PROV Activity) that references, but does not define, the agent that performed it, the configuration it was bound to, or the AI system it belongs to. It covers single model calls, agentic loops, orchestrated multi-agent workflows and asynchronous or batched executions whose result is retrieved later. It is storage- and interface-neutral: OTLP spans, JSON log records, Git-tracked files, MongoDB documents and MCP resources are projections of the same semantics.", "in_scope": [ "Run identity, correlation to traces, conversations and parent runs", "Invocation inputs, system instructions and their trust provenance", "Binding to a specific model, provider, endpoint and immutable configuration snapshot", "Capability surface offered to the run (tools, resources, data sources) and its authorization scopes", "Ordered step trajectory including model calls, planning steps, tool calls and delegated sub-runs", "Run state machine, termination reason, error classification and partial completion", "Outputs, structured results and provenance manifests attached to generated assets", "Principals, on-behalf-of delegation chains and human oversight interventions", "Token, resource and monetary consumption attributable to the run", "Evaluation results, guardrail decisions and other quality evidence bound to the run", "Statutory record-keeping, retention, legal hold, deletion, access scoping and record integrity", "Export profiles and lossy-mapping declarations to external telemetry and provenance schemas" ], "out_of_scope": [ "Definition, capabilities, persona and lifecycle of the agent itself (WM-AI-002)", "Prompt templates, parameter defaults, weights and versioned configuration content (WM-AI-005)", "AI system registration, model cards, training data and conformity assessment (WM-AI-001)", "Conversation, session or thread as a container entity that groups many runs", "Evaluation campaign design, benchmark definition and dataset curation", "Incident management, post-market monitoring cases and regulatory reporting workflows", "Pricing schedules, rate cards, invoices and contractual billing terms", "Person and organisation master data for users, approvers and deployers", "Compute infrastructure inventory, capacity planning and rate-limit configuration", "Physical actuation, robot motion planning and safety envelopes" ], "boundary_notes": [ { "neighbor": "WM-AI-002 AI Agent", "distinction": "The agent is a continuant that bears responsibility; the run is the occurrence it performs. PROV-O separates prov:Agent from prov:Activity and links them with prov:wasAssociatedWith. The run stores only a reference plus the agent identity attributes observed at execution time (gen_ai.agent.id, gen_ai.agent.name); it never redefines the agent.", "source_refs": [ "SRC-008", "SRC-002" ] }, { "neighbor": "WM-AI-005 AI Configuration", "distinction": "The run references an immutable, content-addressed configuration snapshot. Parameter semantics, defaults and version history belong to the configuration model; the run records only which snapshot was in force and any per-call overrides actually sent.", "source_refs": [ "SRC-001", "SRC-013" ] }, { "neighbor": "Distributed trace and span", "distinction": "A trace is a correlation mechanism, not the run. OpenTelemetry trace-ids are 16-byte values that may be unsampled, absent, shared across many runs, or restarted at a process boundary; the run keeps trace-id and span-id as correlation keys, never as identity.", "source_refs": [ "SRC-007", "SRC-001" ] }, { "neighbor": "Conversation, session or thread", "distinction": "gen_ai.conversation.id groups runs that share a history; it is Conditionally Required telemetry, not run identity. A conversation container entity, if modelled, is a sibling aggregate that references runs; embeddings and classification runs have no conversation at all.", "source_refs": [ "SRC-001", "SRC-002" ] }, { "neighbor": "Evaluation result", "distinction": "OpenTelemetry defines gen_ai.evaluation.result as a separate event that may be emitted asynchronously after the run. The run holds references to evaluation results and their verdicts; evaluator definitions, rubrics and campaign scope stay outside.", "source_refs": [ "SRC-004" ] }, { "neighbor": "Content provenance manifest (C2PA)", "distinction": "A C2PA manifest asserts provenance of an asset, not of an execution. It is attached to outputs of the run as a downstream artifact; there is no normative C2PA field that carries a run identifier, so the binding is an adopting-Dimension convention and must be recorded as such.", "source_refs": [ "SRC-016" ] }, { "neighbor": "Regulatory logging obligation", "distinction": "EU AI Act Articles 12 and 19 impose automatic-logging and retention duties on providers of high-risk AI systems. The run record is the technical substrate that can satisfy those duties, but the obligation, its addressee and its assessment belong to the AI system governance model, not to this model.", "source_refs": [ "SRC-005", "SRC-006" ] }, { "neighbor": "MCP server and tool definition", "distinction": "MCP defines the wire contract for tools/list and tools/call. The run records which tool was called with which arguments and what came back; tool schemas, annotations and server capabilities are external declarations that the run snapshots by reference and digest.", "source_refs": [ "SRC-009" ] } ] }, "sources": [ { "id": "SRC-001", "title": "Semantic conventions for generative AI spans", "organization": "OpenTelemetry Authors (Cloud Native Computing Foundation)", "url": "https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-spans.md", "version_or_date": "main branch as published 2026-08-26; gen_ai.* attributes marked Development status", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:05:00Z", "relevance": "Defines gen_ai.operation.name, gen_ai.provider.name, request/response model, response id, finish reasons, token usage, conversation id, error.type and server.address with requirement levels." }, { "id": "SRC-002", "title": "Semantic conventions for generative AI agent spans", "organization": "OpenTelemetry Authors (Cloud Native Computing Foundation)", "url": "https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-agent-spans.md", "version_or_date": "main branch as published 2026-08-26; Development status", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:06:00Z", "relevance": "Defines create_agent, invoke_agent, invoke_workflow, plan and execute_tool spans plus agent id/name/description, tool name, tool call id, tool type and data source id." }, { "id": "SRC-003", "title": "Semantic conventions for generative AI metrics", "organization": "OpenTelemetry Authors (Cloud Native Computing Foundation)", "url": "https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-metrics.md", "version_or_date": "main branch as published 2026-08-26; Development status", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:07:00Z", "relevance": "Defines gen_ai.client.token.usage with gen_ai.token.type input/output, operation duration, time to first chunk, invoke_agent duration, inference_calls, tool_calls and execute_tool duration; contains no cost metric." }, { "id": "SRC-004", "title": "Semantic conventions for generative AI events", "organization": "OpenTelemetry Authors (Cloud Native Computing Foundation)", "url": "https://github.com/open-telemetry/semantic-conventions-genai/blob/main/docs/gen-ai/gen-ai-events.md", "version_or_date": "main branch as published 2026-08-26; Development status", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:08:00Z", "relevance": "Defines gen_ai.client.inference.operation.details and gen_ai.evaluation.result events, and the opt-in gen_ai.input.messages, gen_ai.output.messages and gen_ai.system_instructions content-capture attributes with explicit privacy warnings." }, { "id": "SRC-005", "title": "Article 12: Record-keeping, Regulation (EU) 2024/1689 (Artificial Intelligence Act)", "organization": "European Commission - AI Act Service Desk", "url": "https://ai-act-service-desk.ec.europa.eu/en/ai-act/article-12", "version_or_date": "Regulation (EU) 2024/1689, OJ L of 12 July 2024, official version dated 13 June 2024", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:20:00Z", "relevance": "Requires high-risk AI systems to technically allow automatic recording of events over the system lifetime, and sets the Annex III point 1(a) minimum log set: period of each use with start and end date and time, reference database, matching input data and identity of the verifying natural persons." }, { "id": "SRC-006", "title": "Article 19: Automatically Generated Logs, EU Artificial Intelligence Act", "organization": "Future of Life Institute (EU AI Act Explorer)", "url": "https://artificialintelligenceact.eu/article/19/", "version_or_date": "Reproduction of Regulation (EU) 2024/1689 text, accessed 2026-08-26", "source_type": "secondary", "primary_source": false, "authority_tier": 3, "accessed_at": "2026-08-26T09:22:00Z", "relevance": "States the provider duty to keep automatically generated logs under their control for a period appropriate to the intended purpose and at least six months, with a sectoral carve-out for financial institutions. Used as a secondary reproduction because the EUR-Lex ELI endpoint returned no retrievable body." }, { "id": "SRC-007", "title": "W3C Trace Context, Level 1", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/trace-context/", "version_or_date": "W3C Recommendation, 23 November 2021", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:12:00Z", "relevance": "Defines the traceparent header as version-trace-id-parent-id-trace-flags with a 16-byte trace-id and 8-byte parent-id, and the tracestate vendor key/value list, establishing how runs correlate across process and vendor boundaries." }, { "id": "SRC-008", "title": "PROV-O: The PROV Ontology", "organization": "World Wide Web Consortium (W3C)", "url": "https://www.w3.org/TR/prov-o/", "version_or_date": "W3C Recommendation, 30 April 2013; namespace http://www.w3.org/ns/prov#", "source_type": "ontology", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:13:00Z", "relevance": "Supplies the Entity/Activity/Agent split and the wasGeneratedBy, used, wasDerivedFrom, wasAttributedTo, wasAssociatedWith, actedOnBehalfOf, startedAtTime and endedAtTime properties used to type a run as an activity with attributable inputs and outputs." }, { "id": "SRC-009", "title": "Model Context Protocol specification 2026-07-28: Tools", "organization": "Model Context Protocol maintainers", "url": "https://modelcontextprotocol.io/specification/2026-07-28/server/tools", "version_or_date": "Protocol revision 2026-07-28 (current revision per the MCP versioning page)", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:30:00Z", "relevance": "Normative structure of tools/call: name, arguments, content blocks, structuredContent, outputSchema, isError, the protocol-error versus tool-execution-error distinction, untrusted-annotation warning, human-in-the-loop SHOULD, and the explicit requirement to log tool usage for audit purposes." }, { "id": "SRC-010", "title": "Model Context Protocol specification 2026-07-28: Authorization", "organization": "Model Context Protocol maintainers", "url": "https://modelcontextprotocol.io/specification/2026-07-28/basic/authorization", "version_or_date": "Protocol revision 2026-07-28", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:34:00Z", "relevance": "Requires OAuth 2.1, RFC 8707 resource indicators on every authorization and token request, audience-bound token validation, refusal to accept or transit foreign tokens, and per-operation scope challenges; supplies the authority model for what a run was permitted to do." }, { "id": "SRC-011", "title": "RFC 3339: Date and Time on the Internet: Timestamps", "organization": "Internet Engineering Task Force (IETF)", "url": "https://www.rfc-editor.org/rfc/rfc3339", "version_or_date": "July 2002", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:15:00Z", "relevance": "Defines the full-date and full-time grammar, mandatory numeric UTC offset or Z, optional fractional seconds, and the -00:00 convention for an unknown local offset; governs every instant recorded on a run." }, { "id": "SRC-012", "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1", "organization": "National Institute of Standards and Technology (NIST), U.S. Department of Commerce", "url": "https://www.nist.gov/itl/ai-risk-management-framework", "version_or_date": "NIST AI 100-1, released 26 January 2023; Generative AI Profile NIST AI 600-1 released 26 July 2024", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T09:26:00Z", "relevance": "Voluntary framework whose GOVERN, MAP, MEASURE and MANAGE functions justify measurement, evidence and accountability layers over runs, and whose Generative AI Profile motivates content provenance and human-AI configuration oversight." }, { "id": "SRC-013", "title": "Claude Messages API reference", "organization": "Anthropic PBC", "url": "https://platform.claude.com/docs/en/api/messages", "version_or_date": "API version 2023-06-01, documentation accessed 2026-08-26", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:40:00Z", "relevance": "Concrete response shape for a model call: message id, content blocks including thinking and tool_use, model, stop_reason values end_turn/max_tokens/stop_sequence, stop_sequence, and usage with input_tokens, output_tokens, cache_creation_input_tokens and cache_read_input_tokens; also marks temperature, top_p and top_k as deprecated for newer models." }, { "id": "SRC-014", "title": "Usage and Cost API", "organization": "Anthropic PBC", "url": "https://platform.claude.com/docs/en/manage-claude/usage-cost-api", "version_or_date": "Documentation accessed 2026-08-26", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:44:00Z", "relevance": "Shows authoritative usage and cost reporting as time-bucketed aggregates (1m/1h/1d) grouped by model, workspace, api_key_id, service_tier, context_window, inference_geo and speed, with costs in USD and a stated data-freshness lag; establishes that per-run cost is derived, not billed." }, { "id": "SRC-015", "title": "Monitor model invocation using CloudWatch Logs and Amazon S3", "organization": "Amazon Web Services", "url": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-invocation-logging.html", "version_or_date": "Amazon Bedrock User Guide, accessed 2026-08-26", "source_type": "first-party-doc", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:47:00Z", "relevance": "A concrete invocation-log record schema: schemaType, schemaVersion, timestamp, accountId, region, requestId, operation, modelId, identity.arn, caller-supplied requestMetadata, and input/output bodies with token counts, including the 100 KB inline limit above which payloads are externalised to object storage." }, { "id": "SRC-016", "title": "C2PA Specification, version 2.2", "organization": "Coalition for Content Provenance and Authenticity (C2PA)", "url": "https://spec.c2pa.org/specifications/specifications/2.2/specs/C2PA_Specification.html", "version_or_date": "Version 2.2, May 2025", "source_type": "standard", "primary_source": true, "authority_tier": 2, "accessed_at": "2026-08-26T09:52:00Z", "relevance": "Defines manifest, claim, assertions, ingredients, hard binding and claim signature, and the c2pa.actions assertion with digitalSourceType values used to disclose that an asset was produced with generative AI." }, { "id": "SRC-017", "title": "Regulation (EU) 2024/1689 of the European Parliament and of the Council of 13 June 2024 (Artificial Intelligence Act)", "organization": "European Union", "url": "https://eur-lex.europa.eu/legal-content/EN/TXT/?uri=CELEX:32024R1689", "version_or_date": "OJ L 2024/1689, 12.7.2024; ELI http://data.europa.eu/eli/reg/2024/1689/oj", "source_type": "legislation", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Articles 12, 14, 19 and 26 require automatic event logs over the lifetime of high-risk systems, start and end of each use, human oversight, and provider/deployer retention of logs under their control for at least six months unless other law applies." }, { "id": "SRC-018", "title": "Artificial Intelligence Risk Management Framework (AI RMF 1.0), NIST AI 100-1", "organization": "National Institute of Standards and Technology", "url": "https://nvlpubs.nist.gov/nistpubs/ai/NIST.AI.100-1.pdf", "version_or_date": "NIST AI 100-1, 26 January 2023", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Govern, Map, Measure and Manage functions require documented monitoring of deployed AI, TEVV, accountability, third-party resource monitoring and post-deployment capture of operation." }, { "id": "SRC-019", "title": "PROV-DM: The PROV Data Model", "organization": "World Wide Web Consortium", "url": "https://www.w3.org/TR/2013/REC-prov-dm-20130430/", "version_or_date": "W3C Recommendation 30 April 2013", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Defines Activity, Entity, Agent, usage, generation, association, attribution, delegation, start and end times, and bundles for provenance of provenance, which map directly onto an inference or agent run." }, { "id": "SRC-020", "title": "OpenInference Semantic Conventions", "organization": "Arize AI", "url": "https://github.com/Arize-ai/openinference/blob/main/spec/semantic_conventions.md", "version_or_date": "OpenInference spec main, accessed 2026-08-26", "source_type": "schema", "primary_source": true, "authority_tier": 3, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Complementary span-kind model (LLM, AGENT, TOOL, RETRIEVER, GUARDRAIL, EVALUATOR) plus token, USD cost, session, user, graph-node and annotation attributes used widely in agent tracing backends." }, { "id": "SRC-021", "title": "ISO/IEC 22989:2022 Information technology — Artificial intelligence — Artificial intelligence concepts and terminology", "organization": "ISO/IEC JTC 1/SC 42", "url": "https://www.iso.org/standard/74296.html", "version_or_date": "ISO/IEC 22989:2022, published 2022-07, Edition 1", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Defines inference as reasoning by which conclusions are derived from known premises, covering both the process and the result, and defines agent and machine-learning model concepts that bound this event model." }, { "id": "SRC-022", "title": "Artificial Intelligence Risk Management Framework: Generative Artificial Intelligence Profile, NIST AI 600-1", "organization": "National Institute of Standards and Technology", "url": "https://www.nist.gov/publications/artificial-intelligence-risk-management-framework-generative-artificial-intelligence", "version_or_date": "NIST AI 600-1, 26 July 2024", "source_type": "public-authority", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Generative-AI profile of the AI RMF; requires monitoring, incident handling and documented retention treatment of GAI inputs and outputs after deployment." }, { "id": "SRC-023", "title": "Trace Context", "organization": "World Wide Web Consortium", "url": "https://www.w3.org/TR/2021/REC-trace-context-1-20211123/", "version_or_date": "W3C Recommendation 23 November 2021 (editorial update of 6 February 2020)", "source_type": "standard", "primary_source": true, "authority_tier": 1, "accessed_at": "2026-08-26T18:00:00Z", "relevance": "Defines trace-id and parent/span-id propagation for correlating a run across services; all-zero identifiers are invalid; trace-id should be globally unique and random." } ], "structure": { "bundles": [ { "id": "bun-run-identity-and-class", "name": "Run identity and classification", "description": "What this run is, how it is named, how it is correlated to surrounding executions, and which operational and regulatory class it falls into.", "rationale": "Every downstream concern - provenance, cost, retention, audit - hangs off a stable run identity and a correct class. Vendors each mint their own identifier (response id, message id, requestId) and none of them is globally governed, so identity must be decided explicitly before anything else. Classification decides which obligations even apply.", "source_refs": [ "SRC-001", "SRC-005", "SRC-007", "SRC-013", "SRC-015" ], "layers": [ { "id": "lay-identity-and-correlation", "name": "Identity and correlation", "description": "The authoritative identifier for the run and the correlation keys that place it inside a trace, a conversation and a chain of related runs.", "source_refs": [ "SRC-001", "SRC-007", "SRC-013", "SRC-015" ], "findings": [ { "id": "find-run-identifier", "name": "Authoritative run identifier", "description": "Which identifier authoritatively names this run, which system issued it, how it behaves under retries and asynchronous retrieval, and which vendor-native identifiers are kept alongside it. OpenTelemetry marks gen_ai.response.id as Required only for fetch_response operations, Anthropic returns a msg_ prefixed message id, and Amazon Bedrock records a requestId; none of these is a governed global identifier, so the adopting Dimension must state its priority order.", "source_refs": [ "SRC-001", "SRC-013", "SRC-015" ], "questions": [ { "id": "q-runid-master", "text": "Which system is the master of record for this run's identifier, and what is the exact literal form of that identifier?", "kind": "identity", "answer_data": [ "Name and URI of the issuing master system", "Identifier literal", "Identifier scheme or namespace" ] }, { "id": "q-runid-mint", "text": "When the executing provider returns no run identifier, what identifier does the adopting Dimension mint, and how is uniqueness guaranteed?", "kind": "identity", "answer_data": [ "Minting rule (UUID or ULID)", "Minting authority", "Uniqueness scope" ] }, { "id": "q-runid-retry", "text": "Does the identifier stay stable across retries, streaming resumption and later asynchronous result retrieval, or does each attempt receive a new one?", "kind": "constraint", "answer_data": [ "Stability rule", "Attempt number", "Identifier of the superseded attempt" ] }, { "id": "q-runid-vendor", "text": "Which vendor-native identifiers are retained beside the authoritative identifier, and what is each one good for?", "kind": "interoperability", "answer_data": [ "Provider response identifier", "Provider request identifier", "Mapping note per identifier" ] } ], "data_elements": [ { "id": "de-run-id", "name": "Run identifier", "description": "Authoritative identifier naming this single execution occurrence.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-015" ] }, { "id": "de-run-id-scheme", "name": "Run identifier scheme", "description": "Namespace or scheme under which the run identifier is issued.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-008" ] }, { "id": "de-provider-response-id", "name": "Provider response identifier", "description": "Identifier returned by the model provider for the completion or message.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001", "SRC-013" ] }, { "id": "de-provider-request-id", "name": "Provider request identifier", "description": "Identifier assigned by the serving platform to the inbound request.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-015" ] }, { "id": "de-attempt-number", "name": "Attempt number", "description": "Ordinal of this attempt when the same logical invocation was retried.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009" ] } ], "artifacts": [ { "id": "art-run-record", "name": "Run record", "description": "The canonical durable record of one run, carrying its identifier, class, bindings, state, timings and references to all externalised payloads. Amazon Bedrock's ModelInvocationLog entry is one concrete realisation of this artifact.", "media_or_form": [ "structured record (JSON, BSON or equivalent)", "append-only log entry", "telemetry span with attributes" ], "serial": false, "identity_strategy": "Named by the authoritative run identifier; the record is the identity anchor for every other artifact in this model.", "source_refs": [ "SRC-015", "SRC-001" ] } ], "inline_only_rationale": null }, { "id": "find-correlation-keys", "name": "Trace, conversation and parent-run correlation", "description": "The keys that place the run inside a wider execution graph: W3C trace-id and span-id, gen_ai.conversation.id for session grouping, and the parent-run edge for delegated executions. These are correlation, not identity: trace-ids may be unsampled or absent, and a conversation groups many runs.", "source_refs": [ "SRC-007", "SRC-001", "SRC-002" ], "questions": [ { "id": "q-corr-trace", "text": "Which trace identifier and span identifier bind this run into the surrounding distributed trace, and in which propagation format?", "kind": "relationship", "answer_data": [ "16-byte trace identifier", "8-byte span identifier", "traceparent header value", "Sampled flag" ] }, { "id": "q-corr-conversation", "text": "Which conversation, session or thread identifier groups this run with the runs that preceded and followed it?", "kind": "relationship", "answer_data": [ "Conversation identifier", "Position within the conversation", "Conversation-owning system" ] }, { "id": "q-corr-boundary", "text": "How are correlation keys preserved when the run crosses a process, vendor or network boundary?", "kind": "interoperability", "answer_data": [ "Propagation mechanism", "Vendor tracestate entries", "Boundary crossing points" ] }, { "id": "q-corr-absent", "text": "What is recorded when the trace is not sampled, telemetry is dropped, or no conversation exists for this operation type?", "kind": "exception", "answer_data": [ "Absence reason code", "Fallback correlation key", "Completeness flag" ] } ], "data_elements": [ { "id": "de-trace-id", "name": "Trace identifier", "description": "W3C Trace Context 16-byte trace-id correlating the run to a distributed trace.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007" ] }, { "id": "de-span-id", "name": "Span identifier", "description": "8-byte span identifier of the span representing this run.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007" ] }, { "id": "de-conversation-id", "name": "Conversation identifier", "description": "Session or thread identifier grouping runs that share history.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-parent-run-ref", "name": "Parent run reference", "description": "Reference to the run that invoked this run, when it is a sub-run.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-trace-sampled-flag", "name": "Trace sampled flag", "description": "Whether the enclosing trace was sampled, affecting telemetry completeness.", "value_kind": "boolean", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-007" ] } ], "artifacts": [ { "id": "art-trace-export-bundle", "name": "Trace export bundle", "description": "Exported telemetry covering the run and its child spans, used to reconstruct the execution graph independently of the run record.", "media_or_form": [ "OTLP span export", "telemetry backend query result" ], "serial": false, "identity_strategy": "Identified by trace identifier plus the root span identifier of the run; never used as the run's own identity.", "source_refs": [ "SRC-007", "SRC-002" ] } ], "inline_only_rationale": null } ] }, { "id": "lay-operation-type-and-risk", "name": "Operation typing and risk classification", "description": "The operational shape of the run and the regulatory or risk class that determines which obligations attach to it.", "source_refs": [ "SRC-001", "SRC-002", "SRC-005", "SRC-012" ], "findings": [ { "id": "find-operation-and-invocation-type", "name": "Operation type and invocation mode", "description": "Which operation the run represents and how it was invoked. OpenTelemetry enumerates chat, embeddings, retrieval, fetch_response, create_agent, invoke_agent, invoke_workflow, plan and execute_tool; the run must also record whether execution was synchronous, streaming, batched or asynchronously retrieved, because that changes where the run's temporal boundaries lie.", "source_refs": [ "SRC-001", "SRC-002", "SRC-014" ], "questions": [ { "id": "q-optype-name", "text": "Which operation type does this run represent, drawn from a closed and versioned code list?", "kind": "classification", "answer_data": [ "Operation name code", "Code list version", "Provider discriminator" ] }, { "id": "q-optype-mode", "text": "Was the run synchronous, streaming, batched or asynchronously retrieved, and where exactly do its temporal boundaries fall?", "kind": "process", "answer_data": [ "Invocation mode code", "Submission instant", "Result retrieval instant" ] }, { "id": "q-optype-loop", "text": "Does this run represent a single model call or an agentic loop that may contain many model and tool calls?", "kind": "composition", "answer_data": [ "Agentic loop flag", "Maximum iteration bound", "Observed iteration count" ] } ], "data_elements": [ { "id": "de-operation-name", "name": "Operation name", "description": "Coded operation type such as chat, embeddings, invoke_agent, plan or execute_tool.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-002" ] }, { "id": "de-invocation-mode", "name": "Invocation mode", "description": "Synchronous, streaming, batch or deferred-retrieval execution mode.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001", "SRC-014" ] }, { "id": "de-agentic-loop-flag", "name": "Agentic loop flag", "description": "Whether the run contains an iterative reason-act loop rather than one call.", "value_kind": "boolean", "cardinality": "1", "required": true, "source_refs": [ "SRC-002" ] } ], "artifacts": [], "inline_only_rationale": "Operation typing is a small set of coded scalar attributes carried on the run record declared under find-run-identifier. It produces no separate deliverable, and minting a distinct artifact would create a second, competing identity anchor for the same execution." }, { "id": "find-risk-and-oversight-class", "name": "Regulatory and oversight classification", "description": "Whether the AI system executing this run falls into a regulated class, which minimum log content that triggers, and what level of human oversight is mandated. EU AI Act Article 12 obligations attach only to high-risk systems, and the Article 12(3) minimum log set applies only to Annex III point 1(a) remote biometric identification, so over-application is a real modelling error.", "source_refs": [ "SRC-005", "SRC-006", "SRC-012" ], "questions": [ { "id": "q-riskcls-regime", "text": "Under which regulatory regime and which risk category is the AI system performing this run classified, and by whose determination?", "kind": "classification", "answer_data": [ "Regime identifier", "Risk category code", "Determining party", "Determination reference" ] }, { "id": "q-riskcls-minimum", "text": "Does the applicable regime impose a minimum log content set for runs of this class, and is every mandated element actually present?", "kind": "requirement", "answer_data": [ "Mandated element list", "Presence check per element", "Non-conformity register entry" ] }, { "id": "q-riskcls-oversight", "text": "What degree of human oversight must be exercised over runs of this class before their effects take hold?", "kind": "authority", "answer_data": [ "Oversight level code", "Oversight role", "Blocking or advisory indicator" ] }, { "id": "q-riskcls-jurisdiction", "text": "Which jurisdictions' obligations apply to this run, given where it executed and where its effects land?", "kind": "spatial", "answer_data": [ "Execution jurisdiction", "Effect jurisdiction", "Applicable obligation set" ] } ], "data_elements": [ { "id": "de-regulatory-class", "name": "Regulatory class", "description": "Coded risk classification of the executing AI system under an applicable regime.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "de-mandated-log-elements", "name": "Mandated log element set", "description": "The minimum log elements the applicable regime requires for this run class.", "value_kind": "collection", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "de-oversight-level", "name": "Required oversight level", "description": "Level of human oversight mandated before the run's effects take place.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005", "SRC-009" ] }, { "id": "de-applicable-jurisdictions", "name": "Applicable jurisdictions", "description": "Jurisdictions whose obligations attach to this run.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-006" ] } ], "artifacts": [], "inline_only_rationale": "The classification is inherited from the AI system registration held in the parent model and is carried on the run only as coded references plus the determination reference. Materialising a classification artifact here would duplicate a governed record owned elsewhere and risk it drifting out of step with the authoritative determination." } ] } ] }, { "id": "bun-invocation-context", "name": "Invocation context and execution binding", "description": "Everything that was fixed at the moment of invocation: what was sent in, where it came from and how far it was trusted, which model and provider served the run, which decoding parameters applied, and which capability surface was exposed.", "rationale": "Reproducibility and audit both require the exact inputs and the exact bindings. OpenTelemetry deliberately makes content capture opt-in for privacy reasons while EU record-keeping pushes the other way, so the capture rule must be an explicit, recorded decision rather than an implementation accident.", "source_refs": [ "SRC-001", "SRC-004", "SRC-005", "SRC-009", "SRC-013", "SRC-015" ], "layers": [ { "id": "lay-request-and-input-capture", "name": "Request content and input trust", "description": "The content actually presented to the model, how much of it is captured verbatim, and what is known about where each segment came from.", "source_refs": [ "SRC-004", "SRC-009", "SRC-013", "SRC-015" ], "findings": [ { "id": "find-input-messages-and-instructions", "name": "Input messages, system instructions and payload handling", "description": "What was sent to the model - system instructions, chat history, prompt variables and attachments - and under what capture rule. OpenTelemetry marks gen_ai.input.messages and gen_ai.system_instructions as opt-in with explicit sensitive-data warnings, and Bedrock externalises any body over 100 KB to object storage, so the run record must distinguish inline content from a reference plus digest.", "source_refs": [ "SRC-004", "SRC-013", "SRC-015" ], "questions": [ { "id": "q-input-content", "text": "What exactly was presented to the model as system instructions, chat history, prompt variables and attachments?", "kind": "composition", "answer_data": [ "System instruction text or reference", "Ordered input message list", "Prompt variable bindings", "Attachment references and media types" ] }, { "id": "q-input-capture-rule", "text": "Is verbatim input content captured, and under which opt-in, truncation, filtering or sampling rule was that decided?", "kind": "privacy", "answer_data": [ "Capture mode code", "Truncation or filter rule identifier", "Decision owner" ] }, { "id": "q-input-externalisation", "text": "Where content is too large or is binary, how is it externalised and referenced back from the run record?", "kind": "constraint", "answer_data": [ "Inline size threshold", "External payload reference", "Content digest and algorithm" ] }, { "id": "q-input-reconstruct", "text": "How can the exact input be reconstructed later if the referenced payload store has since changed?", "kind": "validation", "answer_data": [ "Immutability guarantee of the payload store", "Digest verification procedure", "Reconstruction failure handling" ] } ], "data_elements": [ { "id": "de-system-instructions", "name": "System instructions", "description": "Instructions supplied outside the chat history that shaped model behaviour.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-input-messages", "name": "Input messages", "description": "Ordered chat history presented to the model for this run.", "value_kind": "collection", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-input-content-ref", "name": "Input payload reference", "description": "Pointer to externally stored input content above the inline threshold.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-015" ] }, { "id": "de-input-content-digest", "name": "Input content digest", "description": "Cryptographic digest binding the run record to the exact input bytes.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-capture-mode", "name": "Content capture mode", "description": "Whether inputs are stored verbatim, truncated, redacted or omitted.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-004" ] } ], "artifacts": [ { "id": "art-input-payload-object", "name": "Input payload object", "description": "Externalised, immutable copy of the run's input content when it exceeds the inline threshold or is binary, referenced from the run record by URI and digest.", "media_or_form": [ "object-store blob", "content-addressed file", "embedded resource" ], "serial": false, "identity_strategy": "Content-addressed by digest, with the run identifier and role (system, message, attachment) recorded as metadata rather than as the name.", "source_refs": [ "SRC-015", "SRC-004" ] } ], "inline_only_rationale": null }, { "id": "find-input-provenance-and-trust", "name": "Input segment provenance and trust level", "description": "For each segment of the model's context, where it came from and how far it may be trusted. MCP states that tool annotations and tool results must be treated as untrusted unless the server is trusted, and directs clients to validate tool results before passing them to the model; a run record that flattens user text, retrieved documents and tool output into one undifferentiated context loses the evidence needed to investigate an injection.", "source_refs": [ "SRC-009", "SRC-012", "SRC-004" ], "questions": [ { "id": "q-inputtrust-origin", "text": "For each input segment, what is its origin - end user, retrieved document, tool result, upstream agent or system operator?", "kind": "provenance", "answer_data": [ "Segment boundary", "Origin class", "Originating system reference" ] }, { "id": "q-inputtrust-label", "text": "Which input segments were treated as untrusted data that must never be interpreted as instructions?", "kind": "security", "answer_data": [ "Trust label per segment", "Isolation mechanism applied", "Trusted-server determination" ] }, { "id": "q-inputtrust-injection", "text": "How is a suspected prompt-injection or untrusted-instruction attempt recorded against the run?", "kind": "event", "answer_data": [ "Detection event record", "Detector identity and version", "Action taken" ] } ], "data_elements": [ { "id": "de-input-segment-origin", "name": "Input segment origin", "description": "Origin class of one delimited region of the model context.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-input-trust-label", "name": "Input trust label", "description": "Trust level assigned to an input segment at assembly time.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-injection-detection-outcome", "name": "Injection detection outcome", "description": "Result of any untrusted-instruction detection applied to the context.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] } ], "artifacts": [], "inline_only_rationale": "Trust labels are per-segment annotations over content already externalised as the input payload object; storing them as a separate artifact would duplicate segment boundaries and invite the two copies to disagree. They are therefore modelled as inline annotations carrying offsets into the referenced payload." } ] }, { "id": "lay-execution-binding", "name": "Model, parameter and capability binding", "description": "The exact execution substrate the run was bound to: model and provider, decoding parameters and configuration snapshot, and the tools and data sources made available.", "source_refs": [ "SRC-001", "SRC-002", "SRC-009", "SRC-010", "SRC-013" ], "findings": [ { "id": "find-model-and-provider-binding", "name": "Model, provider and endpoint binding", "description": "Which model was requested, which model actually answered, and through which provider, endpoint and region. OpenTelemetry keeps gen_ai.request.model and gen_ai.response.model separate precisely because a gateway or alias may resolve to a different model, and Bedrock records both a modelId and the serving region.", "source_refs": [ "SRC-001", "SRC-015", "SRC-014" ], "questions": [ { "id": "q-binding-requested-vs-served", "text": "Which model was requested and which model identifier actually produced the response?", "kind": "identity", "answer_data": [ "Requested model identifier", "Responding model identifier", "Alias resolution note" ] }, { "id": "q-binding-endpoint", "text": "Which provider, endpoint address and serving region handled the run?", "kind": "spatial", "answer_data": [ "Provider name code", "Server address and port", "Serving region or inference geography" ] }, { "id": "q-binding-version", "text": "Which model version, snapshot or inference profile is recorded so a later reader knows which weights answered?", "kind": "provenance", "answer_data": [ "Model version or snapshot label", "Inference profile identifier", "Deprecation or retirement status" ] } ], "data_elements": [ { "id": "de-request-model", "name": "Requested model", "description": "Model name or alias named in the request.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-response-model", "name": "Responding model", "description": "Model that actually generated the response, which may differ from the request.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001", "SRC-013" ] }, { "id": "de-provider-name", "name": "Provider name", "description": "Coded generative AI provider acting as telemetry-format discriminator.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "de-server-address", "name": "Server address", "description": "Endpoint host serving the inference request.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-serving-region", "name": "Serving region", "description": "Geographic region or inference geography where the run executed.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-015", "SRC-014" ] } ], "artifacts": [], "inline_only_rationale": "Model and endpoint binding is a set of scalar attributes on the run record; the immutable artifact that describes what those identifiers mean - the versioned configuration and weight snapshot - is owned by the sibling configuration model and is referenced rather than copied, so that a single snapshot serves every run bound to it." }, { "id": "find-decoding-params-and-reproducibility", "name": "Decoding parameters and reproducibility limits", "description": "Which sampling and decoding parameters were in force and what actually prevents an identical re-execution. Anthropic marks temperature, top_p and top_k as deprecated for newer models, so a model that assumes these are always present and always determinative will mis-record reproducibility. The run references an immutable configuration snapshot rather than restating its content.", "source_refs": [ "SRC-001", "SRC-013" ], "questions": [ { "id": "q-repro-params", "text": "Which sampling and decoding parameters were actually transmitted and honoured for this run?", "kind": "constraint", "answer_data": [ "Maximum output token limit", "Temperature, top_p and top_k values where applicable", "Stop sequences", "Parameters ignored by the model" ] }, { "id": "q-repro-feasibility", "text": "Can this run be re-executed to produce an identical result, and what specifically prevents bit-identical reproduction?", "kind": "validation", "answer_data": [ "Reproducibility class", "Non-determinism sources", "Seed value if supported" ] }, { "id": "q-repro-deprecated", "text": "How is a parameter recorded when the provider has deprecated, capped or silently ignored it for the selected model?", "kind": "exception", "answer_data": [ "Parameter disposition code", "Requested value", "Effective value" ] }, { "id": "q-repro-snapshot", "text": "Which configuration snapshot does the run reference, and is that snapshot immutable and content-addressed?", "kind": "provenance", "answer_data": [ "Configuration snapshot reference", "Snapshot digest", "Per-call override set" ] } ], "data_elements": [ { "id": "de-max-output-tokens", "name": "Maximum output tokens", "description": "Upper bound on generated tokens requested for the run.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] }, { "id": "de-sampling-params", "name": "Sampling parameters", "description": "Temperature, top_p, top_k and related decoding controls as transmitted.", "value_kind": "object", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001", "SRC-013" ] }, { "id": "de-reproducibility-class", "name": "Reproducibility class", "description": "Whether the run is bit-reproducible, statistically reproducible or not reproducible.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-013" ] }, { "id": "de-config-snapshot-ref", "name": "Configuration snapshot reference", "description": "Reference to the immutable configuration the run was bound to.", "value_kind": "reference", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "de-config-digest", "name": "Configuration digest", "description": "Digest proving which configuration content was in force.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016" ] } ], "artifacts": [ { "id": "art-config-snapshot-manifest", "name": "Configuration snapshot manifest", "description": "Manifest listing the exact configuration content, parameters and digests the run was bound to, produced by the configuration model and referenced immutably by the run.", "media_or_form": [ "content-addressed manifest", "structured record" ], "serial": false, "identity_strategy": "Content-addressed by digest of the canonicalised configuration; the run stores the reference and digest, never a mutable copy.", "source_refs": [ "SRC-001", "SRC-013" ] } ], "inline_only_rationale": null }, { "id": "find-capability-surface", "name": "Capability surface and granted scopes", "description": "Which tools, resources and data sources the model could reach at each point in the run, and under which authorization scopes. MCP requires tool sets to be deterministic per authorization but permits them to change over time via list-changed notifications, and requires RFC 8707 resource indicators binding tokens to a specific server, so the surface is a time-varying, scope-bounded fact rather than a static list.", "source_refs": [ "SRC-009", "SRC-010", "SRC-002" ], "questions": [ { "id": "q-capsurface-offered", "text": "Which tools, resources and data sources were offered to the model at run start, and in which order were they presented?", "kind": "composition", "answer_data": [ "Tool name list with ordering", "Tool schema digests", "Data source identifiers" ] }, { "id": "q-capsurface-scopes", "text": "Which authorization scopes and resource indicators bounded what the run could reach?", "kind": "authority", "answer_data": [ "Granted scope set", "Canonical resource URI per server", "Token audience" ] }, { "id": "q-capsurface-drift", "text": "Did the available capability set change during the run, and how is each change point recorded?", "kind": "state", "answer_data": [ "Capability change event", "Change instant", "Before and after tool set digests" ] } ], "data_elements": [ { "id": "de-tool-catalogue-ref", "name": "Tool catalogue snapshot reference", "description": "Reference to the tool and resource list exposed to the run.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-granted-scopes", "name": "Granted scopes", "description": "Authorization scopes in force for the run's outbound calls.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-010" ] }, { "id": "de-resource-indicator", "name": "Resource indicator", "description": "Canonical URI identifying the protected resource a token was issued for.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-010" ] }, { "id": "de-capability-change-event", "name": "Capability change event", "description": "Recorded point where the available tool or resource set changed mid-run.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] } ], "artifacts": [ { "id": "art-tool-catalogue-snapshot", "name": "Tool catalogue snapshot", "description": "Frozen copy of the tool and resource definitions - names, titles, input and output schemas, annotations - as they were presented to the model during this run.", "media_or_form": [ "structured record", "schema document set" ], "serial": false, "identity_strategy": "Content-addressed by digest of the canonicalised catalogue, with server canonical URI and capture instant as metadata.", "source_refs": [ "SRC-009" ] } ], "inline_only_rationale": null }, { "id": "find-executing-agent-binding", "name": "Executing agent and exact configuration references", "description": "Invoke and create agent spans record gen_ai.agent.id, name, description and version. For hosted agents the id should be the provider-assigned stable identifier (Bedrock ARN, GCP registry id, OpenAI assistant id). In-memory instance ids are not recommended. PROV associates the activity with an agent and optionally a plan. OpenInference records agent.name. The run also records the configuration identifiers needed for reproducibility (prompt version, tool definitions version, policy version, decoding profile) as references to WM-AI-005 rather than embedding the configuration catalogue.", "source_refs": [ "SRC-002", "SRC-019", "SRC-020" ], "questions": [ { "id": "find-executing-agent-binding-q01", "text": "Which stable executing agent identifier, name and version performed this run?", "kind": "relationship", "answer_data": [ "agent-id", "agent-name", "agent-version", "agent-description", "wm-ai-002-ref" ] }, { "id": "find-executing-agent-binding-q02", "text": "Which PROV association, attribution or delegation links this activity to the responsible agent, organisation or human overseer?", "kind": "provenance", "answer_data": [ "prov-agent-id", "prov-role", "plan-id", "acted-on-behalf-of" ] }, { "id": "find-executing-agent-binding-q03", "text": "Which exact configuration identifiers were in force for prompts, tools, decoding, routing and safety policy?", "kind": "relationship", "answer_data": [ "configuration-id", "prompt-version", "tool-definitions-version", "policy-version", "wm-ai-005-ref" ] }, { "id": "find-executing-agent-binding-q04", "text": "Which organisation, deployer or system owner is accountable for this execution?", "kind": "ownership", "answer_data": [ "accountable-deployer", "system-owner", "provider-id", "tenant-id" ] } ], "data_elements": [ { "id": "find-executing-agent-binding-data01", "name": "Stable agent identifier", "description": "Provider-assigned stable identifier of the invoked or created agent.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "find-executing-agent-binding-data02", "name": "Agent name", "description": "Human-readable name of the invoked agent.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002", "SRC-020" ] }, { "id": "find-executing-agent-binding-data03", "name": "Agent version", "description": "Version of the invoked or created agent.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "find-executing-agent-binding-data04", "name": "Configuration reference", "description": "Identifier of the exact configuration object used for this run.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001", "SRC-020" ] } ], "artifacts": [ { "id": "find-executing-agent-binding-artifact01", "name": "Run binding record", "description": "Typed edges from the run to the executing agent and the exact configuration.", "media_or_form": [ "structured-record", "typed-edges" ], "serial": false, "identity_strategy": "Run master identifier plus target identifiers; agent instance ids are forbidden as agent identity.", "source_refs": [ "SRC-002", "SRC-019" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "bun-execution-trajectory", "name": "Execution trajectory", "description": "What the run actually did, step by step: its internal decomposition and deliberation, its interactions with tools and retrieved data, and any work it delegated to sub-runs.", "rationale": "An agent run's accountability rests on the trajectory, not just the endpoints. OpenTelemetry models this as a span tree with plan, chat and execute_tool children, MCP defines the tool-call wire contract and explicitly tells clients to log tool usage for audit, and multi-agent delegation creates authority questions that only a recorded chain can answer.", "source_refs": [ "SRC-002", "SRC-003", "SRC-009", "SRC-010" ], "layers": [ { "id": "lay-step-trajectory", "name": "Step trajectory and deliberation", "description": "How the run decomposes into ordered, nested steps, and how internal reasoning and planning output is treated.", "source_refs": [ "SRC-002", "SRC-003", "SRC-013" ], "findings": [ { "id": "find-step-and-deliberation-record", "name": "Step decomposition and deliberation capture", "description": "The ordered, nested set of steps making up the run, and the treatment of intermediate reasoning. OpenTelemetry defines plan spans for task decomposition and counts inference_calls and tool_calls per agent invocation; Anthropic returns thinking blocks as first-class content. Reasoning content is high-sensitivity and often subject to a shorter retention rule than the final output, so its capture policy must be recorded rather than assumed.", "source_refs": [ "SRC-002", "SRC-003", "SRC-013" ], "questions": [ { "id": "q-step-decompose", "text": "How is the run decomposed into ordered steps, and what identifies each step uniquely within the run?", "kind": "composition", "answer_data": [ "Step identifier scheme", "Step kind code", "Sequence index" ] }, { "id": "q-step-nesting", "text": "What parent-child relationships hold between orchestration steps, model-call steps and tool-call steps?", "kind": "relationship", "answer_data": [ "Parent step reference", "Nesting depth", "Span kind" ] }, { "id": "q-step-reasoning", "text": "Are intermediate reasoning or planning outputs captured, and under which disclosure and retention rule?", "kind": "retention", "answer_data": [ "Reasoning capture mode", "Reasoning content reference", "Separate retention period" ] }, { "id": "q-step-counts", "text": "How many model calls and tool calls did the run make, and were any iteration limits reached?", "kind": "measurement", "answer_data": [ "Inference call count", "Tool call count", "Iteration limit and whether it bound" ] } ], "data_elements": [ { "id": "de-step-id", "name": "Step identifier", "description": "Identifier of one step within the run's trajectory.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-step-kind", "name": "Step kind", "description": "Coded step type such as plan, chat, execute_tool or invoke_workflow.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-parent-step-id", "name": "Parent step identifier", "description": "Reference to the enclosing step, forming the trajectory tree.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-reasoning-content-ref", "name": "Reasoning content reference", "description": "Pointer to captured intermediate reasoning or planning output.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-013" ] }, { "id": "de-inference-call-count", "name": "Inference call count", "description": "Number of model calls made within this run.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003" ] }, { "id": "de-tool-call-count", "name": "Tool call count", "description": "Number of tool executions made within this run.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003" ] } ], "artifacts": [ { "id": "art-run-step-sequence", "name": "Run step sequence", "description": "The ordered, append-only series of step records that together constitute the run's trajectory, each carrying its kind, parent, timings and payload references.", "media_or_form": [ "append-only event series", "span tree export", "structured record collection" ], "serial": true, "identity_strategy": "Each member is named by run identifier plus a zero-padded, gapless sequence index; the index is an ordinal within the run and is never treated as a timestamp.", "source_refs": [ "SRC-002", "SRC-003" ] } ], "inline_only_rationale": null } ] }, { "id": "lay-tool-and-environment", "name": "Tool and environment interaction", "description": "Calls the run made out to tools, servers and knowledge sources, and the evidence retained about each.", "source_refs": [ "SRC-009", "SRC-002", "SRC-012" ], "findings": [ { "id": "find-tool-call-record", "name": "Tool call record and effect character", "description": "For each tool invocation: what was called, with what arguments, what came back, whether it failed, and what side-effect character the tool declared. MCP separates protocol errors (JSON-RPC error) from tool execution errors (isError true in the result) and warns that annotations are untrusted unless the server is trusted, so the record must keep the declaration and the trust judgement apart. Retries, timeouts and multi-round-trip input-required flows must be linked to one logical call.", "source_refs": [ "SRC-009", "SRC-002", "SRC-003" ], "questions": [ { "id": "q-toolcall-what", "text": "For each tool call, which tool on which server was invoked, with which arguments, and what did it return?", "kind": "process", "answer_data": [ "Tool call identifier", "Tool name and server canonical URI", "Argument object", "Result content and structured content" ] }, { "id": "q-toolcall-failure", "text": "Did the call fail at protocol level or return an execution error, and how is that distinction preserved in the record?", "kind": "exception", "answer_data": [ "Protocol error code and message", "Execution error flag", "Low-cardinality error type" ] }, { "id": "q-toolcall-effects", "text": "What side-effect character did the tool declare - read-only, destructive, idempotent, open-world - and was that declaration treated as trusted?", "kind": "security", "answer_data": [ "Declared annotation set", "Server trust determination", "Observed effect classification" ] }, { "id": "q-toolcall-retry", "text": "How are retries, timeouts, cancellations and input-required round trips of the same logical call linked together?", "kind": "state", "answer_data": [ "Logical call identifier", "Attempt sequence", "Timeout or cancellation reason" ] } ], "data_elements": [ { "id": "de-tool-call-id", "name": "Tool call identifier", "description": "Identifier of a single tool invocation instance within the run.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-tool-name", "name": "Tool name", "description": "Name of the invoked tool, scoped to its serving server.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-tool-arguments", "name": "Tool arguments", "description": "Argument object supplied to the tool call.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-tool-result-content", "name": "Tool result content", "description": "Unstructured content blocks returned by the tool.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-tool-structured-content", "name": "Tool structured content", "description": "Structured result conforming to the tool's declared output schema.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-tool-is-error", "name": "Tool execution error flag", "description": "Whether the tool reported an execution error in its result.", "value_kind": "boolean", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-tool-annotations", "name": "Tool annotations", "description": "Declared behaviour hints such as read-only, destructive or idempotent.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-tool-duration", "name": "Tool execution duration", "description": "Elapsed time for one tool execution, in seconds.", "value_kind": "duration", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-003" ] } ], "artifacts": [ { "id": "art-tool-call-log-entry", "name": "Tool call log entries", "description": "Per-call audit entries capturing request, result, error disposition and timing for every tool invocation, satisfying the MCP guidance that clients log tool usage for audit purposes.", "media_or_form": [ "append-only log entry", "structured record", "telemetry span" ], "serial": true, "identity_strategy": "Named by run identifier plus tool call identifier; where a provider supplies no call identifier, a zero-padded ordinal within the run is minted and marked as locally assigned.", "source_refs": [ "SRC-009", "SRC-002" ] } ], "inline_only_rationale": null }, { "id": "find-retrieval-and-grounding", "name": "Retrieval and grounding evidence", "description": "Which knowledge sources were consulted and which retrieved items entered the model context. OpenTelemetry defines gen_ai.data_source.id for retrieval and grounding, but nothing normative pins the retrieved item to a resolvable version, so the run must record item references with digests and retrieval instants if the grounding is to be re-examined later.", "source_refs": [ "SRC-002", "SRC-001", "SRC-012" ], "questions": [ { "id": "q-retrieval-sources", "text": "Which data sources or knowledge bases were queried to ground this run's output?", "kind": "relationship", "answer_data": [ "Data source identifier", "Query issued", "Result count" ] }, { "id": "q-retrieval-items", "text": "Which specific retrieved items entered the model context, and can each be re-resolved at the version that was used?", "kind": "evidence", "answer_data": [ "Item reference", "Item version or digest", "Rank or relevance score" ] }, { "id": "q-retrieval-freshness", "text": "How fresh and how authoritative was the retrieved content at the instant it was used?", "kind": "quality", "answer_data": [ "Item last-modified instant", "Retrieval instant", "Source authority tier" ] } ], "data_elements": [ { "id": "de-data-source-id", "name": "Data source identifier", "description": "Identifier of a knowledge base or corpus consulted during the run.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-retrieved-item-ref", "name": "Retrieved item reference", "description": "Reference to one retrieved item placed into the model context.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-retrieved-item-digest", "name": "Retrieved item digest", "description": "Digest fixing the exact version of a retrieved item.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-retrieval-instant", "name": "Retrieval instant", "description": "Time at which the retrieval was performed.", "value_kind": "timestamp", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-011" ] } ], "artifacts": [ { "id": "art-retrieval-evidence-set", "name": "Retrieval evidence set", "description": "The set of retrieved items, queries and relevance scores that grounded the run, retained so a reviewer can judge whether the output was supported by its sources.", "media_or_form": [ "structured record collection", "embedded resource set" ], "serial": false, "identity_strategy": "Named by run identifier plus retrieval step identifier; each member item is content-addressed by its digest.", "source_refs": [ "SRC-002", "SRC-012" ] } ], "inline_only_rationale": null } ] }, { "id": "lay-delegation-and-subruns", "name": "Delegation and sub-runs", "description": "Work this run handed to other agents or nested runs, and how authority, cost and outcome flow across the delegation boundary.", "source_refs": [ "SRC-002", "SRC-010", "SRC-008" ], "findings": [ { "id": "find-subrun-and-delegation", "name": "Sub-run and multi-agent delegation", "description": "Which sub-runs the run spawned, on whose authority each acted, and how their results roll up. OpenTelemetry provides invoke_workflow for coordinated multi-agent processes but defines no rollup semantics; PROV-O supplies actedOnBehalfOf for the responsibility chain. Unbounded recursion and authority widening across hops are the two failure modes the record has to make visible.", "source_refs": [ "SRC-002", "SRC-008", "SRC-010" ], "questions": [ { "id": "q-deleg-children", "text": "Which sub-runs or delegated agent invocations did this run spawn, and in which order?", "kind": "composition", "answer_data": [ "Sub-run references", "Spawn instants", "Delegating step reference" ] }, { "id": "q-deleg-authority", "text": "On whose authority did each sub-agent act, and was the granted authority narrowed at every delegation hop?", "kind": "authority", "answer_data": [ "Acted-on-behalf-of chain", "Scope set per hop", "Narrowing or widening determination" ] }, { "id": "q-deleg-rollup", "text": "How are the cost, failure and output of a sub-run rolled up into the parent without double counting?", "kind": "measurement", "answer_data": [ "Rollup mode code", "Attributed totals", "Double-count guard rule" ] }, { "id": "q-deleg-bound", "text": "What terminates a delegation chain, and what prevents unbounded recursion or fan-out?", "kind": "constraint", "answer_data": [ "Maximum delegation depth", "Fan-out limit", "Termination reason on limit breach" ] } ], "data_elements": [ { "id": "de-sub-run-ref", "name": "Sub-run reference", "description": "Reference to a run spawned by this run.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-delegation-depth", "name": "Delegation depth", "description": "Number of delegation hops between the originating run and this one.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-delegated-scope", "name": "Delegated scope", "description": "Authority actually conferred on a sub-agent for its sub-run.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-010" ] }, { "id": "de-rollup-mode", "name": "Rollup mode", "description": "Rule by which sub-run measures are aggregated into the parent run.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003" ] } ], "artifacts": [], "inline_only_rationale": "Delegation is a typed edge between run records, not a document. Each sub-run is itself an instance of this model and owns its own artifacts; materialising a separate delegation artifact would create a third copy of facts already held on both endpoints and would go stale whenever a sub-run is corrected or redacted." } ] } ] }, { "id": "bun-outcome-state-and-time", "name": "Outcome, state and time", "description": "How the run progressed through its states, why it ended, when everything happened, and what it produced.", "rationale": "Terminal state and termination reason drive downstream handling; EU AI Act Article 12(3) explicitly requires start and end date and time for each use of certain systems; and outputs need durable, digest-bound evidence plus, for media, a signed provenance manifest.", "source_refs": [ "SRC-001", "SRC-005", "SRC-011", "SRC-013", "SRC-016" ], "layers": [ { "id": "lay-lifecycle-state-and-time", "name": "Lifecycle state and temporal semantics", "description": "The run's state machine, its termination outcome, and the time model that governs every instant recorded on it.", "source_refs": [ "SRC-001", "SRC-005", "SRC-011", "SRC-013" ], "findings": [ { "id": "find-run-state-and-termination", "name": "Run state machine and termination outcome", "description": "The legal states of a run and how it ended. OpenTelemetry records gen_ai.response.finish_reasons as an array and a Stable error.type when an operation ends in error; Anthropic returns stop_reason values end_turn, max_tokens and stop_sequence. A truncated run is neither a clean success nor a failure, and conflating the two destroys audit value.", "source_refs": [ "SRC-001", "SRC-013", "SRC-014" ], "questions": [ { "id": "q-state-machine", "text": "What is the complete set of run states, and which transitions between them are legal?", "kind": "lifecycle", "answer_data": [ "State code list", "Permitted transition matrix", "Terminal state set" ] }, { "id": "q-state-termination", "text": "In which state did the run terminate, and for what stated reason?", "kind": "state", "answer_data": [ "Terminal state", "Finish or stop reason", "Matched stop sequence if any" ] }, { "id": "q-state-partial", "text": "How is a partially completed or truncated run distinguished from an outright failure for audit purposes?", "kind": "classification", "answer_data": [ "Completeness classification", "Incomplete details", "Usable-output indicator" ] }, { "id": "q-state-error", "text": "Which low-cardinality error type is recorded when the run ends in error, and where does the full diagnostic live?", "kind": "exception", "answer_data": [ "Error type code", "Diagnostic reference", "Retryability flag" ] } ], "data_elements": [ { "id": "de-run-status", "name": "Run status", "description": "Current or terminal state of the run within its state machine.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-001" ] }, { "id": "de-stop-reason", "name": "Finish or stop reason", "description": "Reason generation terminated, such as natural end, token limit or stop sequence.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001", "SRC-013" ] }, { "id": "de-error-type", "name": "Error type", "description": "Low-cardinality classification of the error that ended the run.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-completeness-class", "name": "Completeness classification", "description": "Whether the run completed fully, partially or not at all.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-013" ] } ], "artifacts": [], "inline_only_rationale": "State and termination are coded scalar fields whose durable home is the run record artifact already declared under find-run-identifier. Emitting a separate state artifact would split the run's authoritative state across two objects and create an unresolvable question about which one wins after a correction." }, { "id": "find-temporal-semantics", "name": "Temporal semantics and latency measures", "description": "Every instant on the run, in a single governed format, with event time kept apart from observation, ingestion and billing-bucket time. RFC 3339 requires an explicit numeric offset or Z and reserves -00:00 for an unknown local offset. Anthropic's usage reporting is bucketed with a stated freshness lag, so cost timestamps are demonstrably not the same clock as execution timestamps.", "source_refs": [ "SRC-011", "SRC-005", "SRC-003", "SRC-014" ], "questions": [ { "id": "q-time-boundaries", "text": "What are the run's start and end instants, and in which format and offset are they expressed?", "kind": "temporal", "answer_data": [ "Start instant", "End instant", "Offset or Z designator", "Fractional-second precision" ] }, { "id": "q-time-layers", "text": "How is the instant at which the run occurred distinguished from the instants at which it was observed, ingested and billed?", "kind": "provenance", "answer_data": [ "Event time", "Observation time", "Ingestion time", "Billing bucket boundaries" ] }, { "id": "q-time-latency", "text": "Which latency measures are recorded, and from whose vantage point are they measured?", "kind": "measurement", "answer_data": [ "Total duration in seconds", "Time to first chunk or token", "Client versus server vantage point" ] }, { "id": "q-time-clock", "text": "Which clock source produced these instants, and what skew is tolerated before ordering becomes unreliable?", "kind": "constraint", "answer_data": [ "Clock source identifier", "Skew tolerance", "Ordering fallback rule" ] } ], "data_elements": [ { "id": "de-run-start-time", "name": "Run start time", "description": "Instant the run began, per the model timestamp rule.", "value_kind": "timestamp", "cardinality": "1", "required": true, "source_refs": [ "SRC-011", "SRC-005" ] }, { "id": "de-run-end-time", "name": "Run end time", "description": "Instant the run reached a terminal state.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-011", "SRC-005" ] }, { "id": "de-observed-at", "name": "Observation time", "description": "Instant a telemetry collector observed the run event.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-011" ] }, { "id": "de-ingested-at", "name": "Ingestion time", "description": "Instant the record entered the durable store.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-run-duration", "name": "Run duration", "description": "Elapsed operation time in seconds.", "value_kind": "duration", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003" ] }, { "id": "de-time-to-first-token", "name": "Time to first token", "description": "Latency to the first streamed token or chunk.", "value_kind": "duration", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003" ] } ], "artifacts": [], "inline_only_rationale": "Instants and durations are inline scalar values on the run and its step records, governed centrally by the model timestamp rule. A dedicated temporal artifact would either duplicate those scalars or become a second, competing source of truth about when the run happened." }, { "id": "find-rare-endings-and-degenerate-runs", "name": "Retries, cancellation, compaction and fetch-without-inference", "description": "Rare but specified endings: automatic retries folded into one logical span; cancellation or timeout; gen_ai.conversation.compacted indicating the effective context is a compacted view of a prior conversation and must not be set to false; and fetch_response, which loads a previously generated response by identifier without performing inference and must not report token usage. These cases are easy to omit from naive chat-only schemas.", "source_refs": [ "SRC-001", "SRC-002" ], "questions": [ { "id": "find-rare-endings-and-degenerate-runs-q01", "text": "Were transient automatic retries folded into this logical run, and how many attempts occurred?", "kind": "process", "answer_data": [ "retry-count", "retry-folded-flag", "attempt-error-types" ] }, { "id": "find-rare-endings-and-degenerate-runs-q02", "text": "Was the run cancelled or terminated by timeout before a model finish reason was produced?", "kind": "lifecycle", "answer_data": [ "cancelled-flag", "timeout-flag", "cancel-actor", "cancel-at" ] }, { "id": "find-rare-endings-and-degenerate-runs-q03", "text": "Was the effective conversation context a compacted view of a prior conversation, and was the compacted attribute left unset rather than set to false when unknown?", "kind": "constraint", "answer_data": [ "conversation-compacted", "compaction-evidence", "attribute-unset-when-unknown" ] }, { "id": "find-rare-endings-and-degenerate-runs-q04", "text": "Is this a fetch of a stored response without inference, and were token usage attributes and metrics omitted as required?", "kind": "classification", "answer_data": [ "operation-name", "fetched-response-id", "token-usage-omitted-flag" ] } ], "data_elements": [ { "id": "find-rare-endings-and-degenerate-runs-data01", "name": "Automatic retry count", "description": "Number of automatic retries folded into the logical span.", "value_kind": "number", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "find-rare-endings-and-degenerate-runs-data02", "name": "Conversation compacted indicator", "description": "Positive indicator that compacted context was applied; must not be recorded as false.", "value_kind": "boolean", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "find-rare-endings-and-degenerate-runs-data03", "name": "Cancelled flag", "description": "True when the caller or runtime cancelled the logical operation.", "value_kind": "boolean", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-001" ] } ], "artifacts": [ { "id": "find-rare-endings-and-degenerate-runs-artifact01", "name": "Rare ending note", "description": "Retry, cancel, compaction or fetch-without-inference annotation on the run.", "media_or_form": [ "structured-record" ], "serial": true, "identity_strategy": "Keyed by run master identifier plus ending-kind code.", "source_refs": [ "SRC-001" ] } ], "inline_only_rationale": null } ] }, { "id": "lay-outputs-and-assets", "name": "Outputs and generated assets", "description": "What the run returned, how that return is evidenced, and what provenance travels with any media it produced.", "source_refs": [ "SRC-004", "SRC-009", "SRC-013", "SRC-016" ], "findings": [ { "id": "find-output-content-and-structure", "name": "Output content, structure and evidence", "description": "The content blocks the run produced, whether they validated against a declared schema, and how they are digested and referenced. MCP requires servers to make structured results conform to a declared outputSchema and clients to validate them; Bedrock externalises outputs above 100 KB. Which parts a human actually saw is a distinct and audit-relevant fact.", "source_refs": [ "SRC-004", "SRC-009", "SRC-013", "SRC-015" ], "questions": [ { "id": "q-output-blocks", "text": "Which content blocks did the run produce, in which modalities and in which order?", "kind": "composition", "answer_data": [ "Output message list", "Modality per block", "Choice or candidate index" ] }, { "id": "q-output-schema", "text": "Was the output schema-constrained, and did it validate against the declared output schema?", "kind": "validation", "answer_data": [ "Output schema reference", "Validation verdict", "Validation error detail" ] }, { "id": "q-output-evidence", "text": "How is the output stored, digested and referenced so a later reader can prove what was actually returned?", "kind": "evidence", "answer_data": [ "Output payload reference", "Output digest and algorithm", "Storage immutability guarantee" ] }, { "id": "q-output-exposure", "text": "Which parts of the output were presented to a human, and which were consumed only by downstream machinery?", "kind": "access", "answer_data": [ "Audience annotation per block", "Presentation surface", "Suppressed block references" ] } ], "data_elements": [ { "id": "de-output-messages", "name": "Output messages", "description": "Content blocks generated by the run, one set per candidate.", "value_kind": "collection", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004", "SRC-013" ] }, { "id": "de-output-content-ref", "name": "Output payload reference", "description": "Pointer to externally stored output content above the inline threshold.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-015" ] }, { "id": "de-output-digest", "name": "Output digest", "description": "Digest binding the run record to the exact output bytes.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-output-schema-ref", "name": "Output schema reference", "description": "Schema the structured output was required to satisfy.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-output-audience", "name": "Output audience annotation", "description": "Declared audience for a content block, such as user or assistant.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] } ], "artifacts": [ { "id": "art-output-payload-object", "name": "Output payload object", "description": "Externalised, immutable copy of the run's output content, referenced from the run record by URI and digest so the returned bytes remain provable after the run closes.", "media_or_form": [ "object-store blob", "content-addressed file", "structured record" ], "serial": false, "identity_strategy": "Content-addressed by digest; run identifier and candidate index are carried as metadata rather than embedded in the name.", "source_refs": [ "SRC-015", "SRC-009" ] } ], "inline_only_rationale": null }, { "id": "find-generated-asset-provenance", "name": "Generated asset provenance and AI disclosure", "description": "For media assets the run produced, whether a signed content-provenance manifest is attached and what it asserts. C2PA 2.2 binds assertions and a claim through a claim signature with a hard binding to the content, and the c2pa.actions assertion carries a digitalSourceType disclosing generative-AI involvement. No normative C2PA field carries a run identifier, so the binding back to this model is an adopting-Dimension convention.", "source_refs": [ "SRC-016", "SRC-012" ], "questions": [ { "id": "q-asset-manifest", "text": "For each media asset produced, is a signed provenance manifest attached, and which assertions does it carry?", "kind": "provenance", "answer_data": [ "Manifest reference", "Assertion labels present", "Claim signature and signer" ] }, { "id": "q-asset-disclosure", "text": "Is the asset marked as AI-generated using a recognised digital source type, and does that marking survive downstream processing?", "kind": "interoperability", "answer_data": [ "Digital source type value", "Durability class", "Hard binding hash" ] }, { "id": "q-asset-binding", "text": "How is the manifest bound back to this run and to the ingredients that were consumed to create the asset?", "kind": "relationship", "answer_data": [ "Run reference convention used", "Ingredient references", "Binding verification method" ] } ], "data_elements": [ { "id": "de-manifest-ref", "name": "Provenance manifest reference", "description": "Reference to the content-provenance manifest attached to a generated asset.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-digital-source-type", "name": "Digital source type", "description": "Coded disclosure of how the asset was produced, including AI generation.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-ingredient-refs", "name": "Ingredient references", "description": "Assets consumed in producing the generated asset, carrying their own provenance.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-hard-binding-hash", "name": "Hard binding hash", "description": "Cryptographic hash binding the manifest to the asset bytes.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-016" ] } ], "artifacts": [ { "id": "art-content-provenance-manifest", "name": "Content provenance manifest", "description": "Signed manifest of assertions, claim and claim signature travelling with a generated media asset, disclosing AI involvement and preserving the ingredient chain.", "media_or_form": [ "embedded manifest store", "detached sidecar manifest" ], "serial": false, "identity_strategy": "Identified by the manifest's own claim identifier and hard binding hash; the run identifier is recorded as a custom assertion because no normative C2PA field carries it.", "source_refs": [ "SRC-016" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "bun-accountability", "name": "Accountability, authority and provenance", "description": "Who authorised the run, who intervened in it, and to whom its activity, entities and effects are attributed.", "rationale": "An agent run acts with borrowed authority. MCP requires audience-bound tokens and forbids token passthrough; PROV-O supplies the vocabulary for attribution and delegation; EU AI Act Article 12(3) requires identifying the natural persons involved in verifying results for the most sensitive class. Without a recorded authority chain, effects cannot be traced to a responsible party.", "source_refs": [ "SRC-005", "SRC-008", "SRC-009", "SRC-010", "SRC-012" ], "layers": [ { "id": "lay-actors-and-authority", "name": "Actors, authority and oversight", "description": "The principals behind the run, the scope they conferred, and the human decisions taken during execution.", "source_refs": [ "SRC-005", "SRC-009", "SRC-010", "SRC-015" ], "findings": [ { "id": "find-principals-and-authority", "name": "Principals, credentials and delegation of rights", "description": "Which principal initiated the run and whose credentials were presented to each downstream resource. MCP requires tokens to be validated as issued specifically for the receiving server and forbids servers accepting or transiting foreign tokens; Bedrock records identity.arn for the calling principal. The record captures who and with what scope, never the credential material itself.", "source_refs": [ "SRC-010", "SRC-015", "SRC-008", "SRC-013" ], "questions": [ { "id": "q-auth-initiator", "text": "Which principal initiated this run, and whose credentials were presented against each downstream resource?", "kind": "authority", "answer_data": [ "Initiating principal identifier", "Per-resource principal", "Credential type" ] }, { "id": "q-auth-binding", "text": "Which scopes were granted, and was each access token bound to the specific resource it was used against?", "kind": "security", "answer_data": [ "Granted scope set per resource", "Token audience", "Audience validation outcome" ] }, { "id": "q-auth-chain", "text": "Where the agent acted on behalf of a person or organisation, how is that on-behalf-of chain recorded?", "kind": "provenance", "answer_data": [ "Acted-on-behalf-of edges", "Delegating party identity", "Delegation evidence reference" ] }, { "id": "q-auth-pseudonym", "text": "How is an end user identified in the run record without persisting direct personal identifiers?", "kind": "privacy", "answer_data": [ "Pseudonymous user identifier", "Derivation or hashing rule", "Re-identification control" ] } ], "data_elements": [ { "id": "de-initiating-principal", "name": "Initiating principal", "description": "Identity of the party whose action caused the run to start.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-015" ] }, { "id": "de-acted-on-behalf-of", "name": "Acted on behalf of", "description": "Party on whose behalf the executing agent acted.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "de-token-audience", "name": "Token audience", "description": "Resource the presented access token was issued for.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-010" ] }, { "id": "de-end-user-pseudonym", "name": "End user pseudonym", "description": "Non-identifying reference to the end user associated with the run.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] } ], "artifacts": [], "inline_only_rationale": "Authority facts are attributes and typed edges of the run record, and the underlying credential material must never be persisted anywhere in this model. Creating an authority artifact would produce an attractive target holding scope and audience data next to identifiers, with no offsetting audit benefit over the inline edges." }, { "id": "find-human-oversight-and-approval", "name": "Human oversight, approval and intervention", "description": "Human decisions taken inside the run. MCP states there SHOULD always be a human able to deny tool invocations and that clients should show tool inputs before calling; EU AI Act Article 12(3)(d) requires identifying the natural persons involved in verifying results for Annex III point 1(a) systems. The record must capture what the decider was shown, not only what they decided.", "source_refs": [ "SRC-009", "SRC-005", "SRC-012" ], "questions": [ { "id": "q-oversight-gates", "text": "Which actions in this run required explicit human approval before they could execute?", "kind": "decision", "answer_data": [ "Gated action references", "Gating rule identifier", "Gate outcome" ] }, { "id": "q-oversight-decider", "text": "Who approved or denied each gated action, at what instant, and what information were they shown at the time?", "kind": "evidence", "answer_data": [ "Decider identity", "Decision instant", "Presented context reference" ] }, { "id": "q-oversight-intervention", "text": "Was the run paused, stopped or overridden by a human, and what effect did that have on the trajectory?", "kind": "event", "answer_data": [ "Intervention type", "Intervention instant", "Resulting state transition" ] }, { "id": "q-oversight-missing", "text": "How is a missing but required approval detected, escalated and recorded as a non-conformity?", "kind": "exception", "answer_data": [ "Detection rule", "Escalation path", "Non-conformity record reference" ] } ], "data_elements": [ { "id": "de-approval-request-id", "name": "Approval request identifier", "description": "Identifier of a human-approval request raised during the run.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-approval-decision", "name": "Approval decision", "description": "Outcome of a human approval gate: approved, denied or timed out.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-decider-identity", "name": "Decider identity", "description": "Identity of the natural person who approved, denied or verified.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "de-presented-context-ref", "name": "Presented context reference", "description": "Reference to exactly what was shown to the human before deciding.", "value_kind": "reference", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-intervention-type", "name": "Intervention type", "description": "Kind of human intervention such as pause, stop, override or edit.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] } ], "artifacts": [ { "id": "art-oversight-decision-record", "name": "Oversight decision record", "description": "Evidence of each human approval, denial, verification or override during the run, including the decider, the instant and a reference to the exact information presented.", "media_or_form": [ "structured record", "append-only log entry", "signed attestation" ], "serial": false, "identity_strategy": "Named by run identifier plus approval request identifier; the decider is referenced by a governed person identifier held in the identity master system, not copied.", "source_refs": [ "SRC-009", "SRC-005" ] } ], "inline_only_rationale": null } ] }, { "id": "lay-provenance-and-accountability", "name": "Provenance and accountable party", "description": "Formal provenance typing of the run and the identification of who owns the record and who answers for its effects.", "source_refs": [ "SRC-008", "SRC-006", "SRC-014" ], "findings": [ { "id": "find-attribution-and-accountability", "name": "Provenance attribution, ownership and accountability", "description": "How the run maps onto a formal provenance model and who is answerable. PROV-O types the run as an Activity with used and wasGeneratedBy edges to its inputs and outputs and wasAssociatedWith to its agent. Separately, EU AI Act Article 19 puts the log-keeping duty on the provider for logs under their control, which differs from the deployer who is answerable for effects, so record owner and accountable party are two distinct fields.", "source_refs": [ "SRC-008", "SRC-006", "SRC-014" ], "questions": [ { "id": "q-prov-mapping", "text": "Which provenance classes and properties does this run map to, and what are the resolvable IRIs for its activity and agents?", "kind": "provenance", "answer_data": [ "Activity IRI", "Associated agent IRIs", "Property mapping table" ] }, { "id": "q-prov-ownership", "text": "Who owns the run record, and who is the accountable deployer answerable for the run's effects?", "kind": "ownership", "answer_data": [ "Record owner", "Accountable deployer", "Basis of the distinction" ] }, { "id": "q-prov-derivation", "text": "Which outputs were derived from which inputs and retrieved items, and is that derivation edge explicit rather than inferred?", "kind": "relationship", "answer_data": [ "Derivation edges", "Generating activity reference", "Edge assertion confidence" ] }, { "id": "q-prov-tenancy", "text": "Under which tenant, workspace, project or cost centre is this run attributed?", "kind": "classification", "answer_data": [ "Tenant identifier", "Workspace identifier", "Cost centre code" ] } ], "data_elements": [ { "id": "de-prov-activity-iri", "name": "Provenance activity IRI", "description": "Resolvable IRI typing the run as a provenance activity.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "de-prov-agent-iri", "name": "Provenance agent IRI", "description": "Resolvable IRI for an agent associated with the run.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "de-derivation-edges", "name": "Derivation edges", "description": "Explicit was-derived-from links between outputs and their sources.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-008" ] }, { "id": "de-record-owner", "name": "Record owner", "description": "Party holding and controlling the run record.", "value_kind": "identifier", "cardinality": "1", "required": true, "source_refs": [ "SRC-006" ] }, { "id": "de-accountable-deployer", "name": "Accountable deployer", "description": "Party answerable for the run's effects under the applicable regime.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-006" ] }, { "id": "de-tenant-id", "name": "Tenant identifier", "description": "Tenant, workspace or project the run is attributed to.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-014" ] } ], "artifacts": [ { "id": "art-provenance-graph-fragment", "name": "Provenance graph fragment", "description": "Graph projection of the run as a provenance activity with its used, generated, associated-with and acted-on-behalf-of edges, exportable for cross-system lineage queries.", "media_or_form": [ "RDF graph serialisation", "typed edge set", "structured record" ], "serial": false, "identity_strategy": "Named by the activity IRI minted in the adopting Dimension's governed namespace, resolvable back to the run identifier.", "source_refs": [ "SRC-008" ] } ], "inline_only_rationale": null } ] } ] }, { "id": "bun-measurement-and-evidence", "name": "Measurement, cost and quality evidence", "description": "What the run consumed, what it cost, how good its output was judged to be, and which policy checks it passed or failed.", "rationale": "The registry purpose names cost explicitly, and cost is the one dimension where no standard exists: OpenTelemetry's GenAI metrics define token usage and durations but no cost metric, while authoritative cost arrives later as a time-bucketed aggregate. Quality evidence is likewise a separate, often asynchronous event that must be bound back to the run.", "source_refs": [ "SRC-003", "SRC-004", "SRC-013", "SRC-014" ], "layers": [ { "id": "lay-consumption-and-cost", "name": "Consumption and cost", "description": "Resource consumption attributable to the run and the monetary cost derived from it.", "source_refs": [ "SRC-003", "SRC-013", "SRC-014", "SRC-015" ], "findings": [ { "id": "find-token-and-resource-usage", "name": "Token and resource consumption", "description": "What the run consumed. OpenTelemetry defines gen_ai.client.token.usage with token types input and output only, while providers report a richer set including cache-creation, cache-read and reasoning tokens, plus server-side tool use. Provider-reported counts and locally computed counts routinely disagree, so the reconciliation rule must be explicit.", "source_refs": [ "SRC-003", "SRC-013", "SRC-015", "SRC-014" ], "questions": [ { "id": "q-usage-tokens", "text": "How many input, output, cache-read, cache-creation and reasoning tokens did the run consume?", "kind": "measurement", "answer_data": [ "Token counts by type", "Token type code list", "Counting authority" ] }, { "id": "q-usage-nontoken", "text": "Which non-token resources were consumed, and how is that set enumerated for this operation type?", "kind": "composition", "answer_data": [ "Server tool use counts", "Code execution units", "Retrieved byte volume" ] }, { "id": "q-usage-reconcile", "text": "How are provider-reported counts reconciled with locally computed counts when the two disagree?", "kind": "validation", "answer_data": [ "Authoritative source rule", "Variance tolerance", "Discrepancy record" ] }, { "id": "q-usage-bucket", "text": "Into which measurement time bucket is usage assigned when the run spans a bucket boundary?", "kind": "temporal", "answer_data": [ "Bucket assignment rule", "Bucket width", "Bucket start and end instants" ] } ], "data_elements": [ { "id": "de-input-tokens", "name": "Input tokens", "description": "Count of input tokens consumed, including cached tokens where reported.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003", "SRC-013" ] }, { "id": "de-output-tokens", "name": "Output tokens", "description": "Count of tokens generated by the run.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-003", "SRC-013" ] }, { "id": "de-cache-read-tokens", "name": "Cache read tokens", "description": "Input tokens served from a prompt cache.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] }, { "id": "de-cache-creation-tokens", "name": "Cache creation tokens", "description": "Input tokens written into a prompt cache for later reuse.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-013" ] }, { "id": "de-server-tool-use-count", "name": "Server tool use count", "description": "Count of provider-side tool invocations such as web search.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-usage-authority", "name": "Usage counting authority", "description": "Which party's count is treated as authoritative for this run.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-014" ] } ], "artifacts": [], "inline_only_rationale": "Per-run consumption counters are scalar measures on the run record. The durable artifact in this area is the reconciled usage and cost statement declared on find-cost-attribution, which is produced later from authoritative bucketed reporting; duplicating counters into a second artifact here would guarantee divergence once reconciliation adjusts them." }, { "id": "find-cost-attribution", "name": "Cost attribution and reconciliation", "description": "What the run cost and how confident that figure is. No GenAI telemetry standard defines a cost metric, and authoritative cost is published as daily aggregates in a single currency with a data-freshness lag and known exclusions such as priority-tier pricing. Per-run cost is therefore an estimate until reconciled, and the record must say which it is.", "source_refs": [ "SRC-014", "SRC-003", "SRC-015" ], "questions": [ { "id": "q-cost-amount", "text": "What monetary cost is attributed to this run, in which currency and against which price basis?", "kind": "measurement", "answer_data": [ "Cost amount and minor-unit convention", "Currency code", "Price basis reference" ] }, { "id": "q-cost-confidence", "text": "Is the recorded cost a local estimate or an authoritative billed amount, and how is that status flagged and later upgraded?", "kind": "provenance", "answer_data": [ "Cost basis code", "Reconciliation status", "Reconciliation instant" ] }, { "id": "q-cost-charge", "text": "To which API key, workspace, service tier and cost centre is this run's cost charged?", "kind": "ownership", "answer_data": [ "API key identifier", "Workspace identifier", "Service tier", "Cost centre" ] }, { "id": "q-cost-allocation", "text": "How are sub-run costs and cached-token savings allocated so that nothing is counted twice?", "kind": "constraint", "answer_data": [ "Allocation method", "Sub-run inclusion rule", "Double-count guard" ] } ], "data_elements": [ { "id": "de-cost-amount", "name": "Cost amount", "description": "Monetary cost attributed to the run, in the smallest currency unit.", "value_kind": "quantity", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-cost-currency", "name": "Cost currency", "description": "Currency in which the cost is expressed.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-cost-basis", "name": "Cost basis", "description": "Whether the figure is a local estimate or an authoritative billed amount.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-service-tier", "name": "Service tier", "description": "Service tier under which the run was served, affecting pricing.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-api-key-id", "name": "API key identifier", "description": "Identifier of the credential the usage is charged to, where applicable.", "value_kind": "identifier", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-allocation-method", "name": "Allocation method", "description": "Rule for distributing shared and sub-run costs to this run.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] } ], "artifacts": [ { "id": "art-usage-and-cost-statement", "name": "Usage and cost statement", "description": "Reconciled statement joining the run's consumption counters to authoritative bucketed cost reporting, carrying the cost basis, reconciliation status and charge dimensions.", "media_or_form": [ "structured record", "tabular report row", "aggregate report extract" ], "serial": false, "identity_strategy": "Named by run identifier plus the reporting bucket boundaries; superseded by a later statement rather than edited, with the supersession edge retained.", "source_refs": [ "SRC-014" ] } ], "inline_only_rationale": null } ] }, { "id": "lay-quality-and-safety", "name": "Quality and safety evidence", "description": "Evaluations, feedback and policy decisions that qualify how good and how safe this run was.", "source_refs": [ "SRC-004", "SRC-012", "SRC-009" ], "findings": [ { "id": "find-evaluation-and-quality", "name": "Evaluation results and outcome quality", "description": "Assessments bound to the run. OpenTelemetry defines gen_ai.evaluation.result as its own event, capable of being emitted independently of the trace, which means evaluation is frequently asynchronous and arrives after the run has closed. Human feedback is a distinct signal from automated evaluation and must not be merged into the same score field.", "source_refs": [ "SRC-004", "SRC-012" ], "questions": [ { "id": "q-eval-which", "text": "Which evaluations were performed against this run's output, by which evaluator and at which evaluator version?", "kind": "quality", "answer_data": [ "Evaluation name", "Evaluator identifier", "Evaluator version" ] }, { "id": "q-eval-verdict", "text": "What score, label or verdict did each evaluation produce, and on what scale is it interpretable?", "kind": "measurement", "answer_data": [ "Score value", "Label", "Scale definition", "Explanation text" ] }, { "id": "q-eval-timing", "text": "Was the evaluation performed inline during the run or asynchronously afterwards, and how late may it still arrive?", "kind": "temporal", "answer_data": [ "Evaluation instant", "Inline or asynchronous flag", "Acceptance window" ] }, { "id": "q-eval-feedback", "text": "How is human feedback on the run captured and kept distinguishable from automated evaluation?", "kind": "evidence", "answer_data": [ "Feedback source class", "Feedback value", "Submitter reference" ] } ], "data_elements": [ { "id": "de-evaluation-name", "name": "Evaluation name", "description": "Name of an evaluation applied to the run's output.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-evaluator-id", "name": "Evaluator identifier", "description": "Identity and version of the evaluating system or rubric.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-evaluation-score", "name": "Evaluation score", "description": "Numeric result produced by an evaluation.", "value_kind": "number", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-evaluation-label", "name": "Evaluation label", "description": "Categorical verdict produced by an evaluation.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-feedback-source", "name": "Feedback source", "description": "Whether a quality signal came from a human or an automated evaluator.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] } ], "artifacts": [ { "id": "art-evaluation-result-record", "name": "Evaluation result record", "description": "Standalone record of one evaluation applied to the run, emitted independently of the trace and referencing the run by identifier.", "media_or_form": [ "structured event record", "log record", "structured record" ], "serial": false, "identity_strategy": "Named by evaluation identifier assigned by the evaluating system, with the run identifier and evaluator version carried as references.", "source_refs": [ "SRC-004" ] } ], "inline_only_rationale": null }, { "id": "find-guardrail-and-policy-decisions", "name": "Guardrail and policy decisions", "description": "Policy checks applied to inputs and outputs, their verdicts, and the policy version in force. A blocked or refused run creates a genuine tension: the record must prove the block occurred without persisting the prohibited content, so the model records decision metadata and digests rather than the content itself.", "source_refs": [ "SRC-012", "SRC-009", "SRC-004" ], "questions": [ { "id": "q-guard-checks", "text": "Which policy or guardrail checks were applied to this run's inputs, intermediate steps and outputs?", "kind": "security", "answer_data": [ "Guardrail names", "Application points", "Check ordering" ] }, { "id": "q-guard-decision", "text": "What did each check decide - allow, transform or block - and what exactly was modified?", "kind": "decision", "answer_data": [ "Decision code", "Modified region references", "Replacement content indicator" ] }, { "id": "q-guard-version", "text": "Which policy version was in force at the instant of each decision?", "kind": "provenance", "answer_data": [ "Policy identifier", "Policy version", "Effective-from instant" ] }, { "id": "q-guard-blocked", "text": "How is a blocked or refused run evidenced without persisting the prohibited content itself?", "kind": "privacy", "answer_data": [ "Refusal code", "Content digest without content", "Retention exception applied" ] } ], "data_elements": [ { "id": "de-guardrail-name", "name": "Guardrail name", "description": "Name of a policy or safety check applied during the run.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] }, { "id": "de-guardrail-version", "name": "Guardrail version", "description": "Version of the policy in force when the decision was made.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] }, { "id": "de-guardrail-decision", "name": "Guardrail decision", "description": "Verdict returned by a guardrail: allow, transform or block.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-012" ] }, { "id": "de-refusal-code", "name": "Refusal code", "description": "Coded reason a run or step was refused, recorded without the content.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-004" ] } ], "artifacts": [], "inline_only_rationale": "Guardrail outcomes are decision annotations attached to the run and its steps. Deliberately, the prohibited or transformed content is not persisted, so there is no payload to externalise; producing an artifact here would either be empty metadata duplicated from the run record or would reintroduce the very content the block existed to prevent retaining." } ] } ] }, { "id": "bun-record-governance", "name": "Governance of the run record", "description": "How the record of the run is retained, protected, exposed and exported once the run itself is over.", "rationale": "The run ends in seconds; the record lives for years. EU AI Act Article 19 sets a minimum retention of at least six months for provider-held logs, content-capture attributes carry explicit sensitive-data warnings, and no cross-vendor run schema exists, so retention, access, integrity and export all need governed answers.", "source_refs": [ "SRC-004", "SRC-005", "SRC-006", "SRC-009", "SRC-016" ], "layers": [ { "id": "lay-retention-and-integrity", "name": "Retention and record integrity", "description": "How long the record is kept, under what basis, how it is disposed of, and how alteration is made detectable.", "source_refs": [ "SRC-005", "SRC-006", "SRC-016", "SRC-015" ], "findings": [ { "id": "find-statutory-log-and-retention", "name": "Statutory log content, retention and disposal", "description": "The regulated content and lifespan of the record. Article 19 requires providers to keep automatically generated logs under their control for a period appropriate to the intended purpose and at least six months unless other Union or national law provides otherwise, with financial institutions keeping them under sectoral rules. That floor collides with data-minimisation and erasure duties, so the reconciliation rule must be recorded rather than left to the operator.", "source_refs": [ "SRC-005", "SRC-006", "SRC-004" ], "questions": [ { "id": "q-ret-content", "text": "Which regulated minimum log elements must this run's record contain, and is each one demonstrably present?", "kind": "requirement", "answer_data": [ "Required element list", "Presence verdict per element", "Gap register reference" ] }, { "id": "q-ret-period", "text": "For how long must the record be kept, counted from which start point, and on which legal basis?", "kind": "retention", "answer_data": [ "Retention period", "Retention clock start", "Legal basis citation" ] }, { "id": "q-ret-holder", "text": "Who holds the logs when provider and deployer differ, and how is the phrase under their control demonstrated?", "kind": "ownership", "answer_data": [ "Holder role", "Control evidence", "Hand-over arrangement" ] }, { "id": "q-ret-conflict", "text": "How are erasure requests, legal holds and statutory minimum-retention duties reconciled when they conflict?", "kind": "exception", "answer_data": [ "Conflict resolution rule", "Legal hold flag", "Partial-redaction option" ] }, { "id": "q-ret-disposal", "text": "What exactly is destroyed at end of life, and what tombstone remains to prove the record once existed?", "kind": "lifecycle", "answer_data": [ "Disposition action", "Disposal instant", "Tombstone content" ] } ], "data_elements": [ { "id": "de-retention-class", "name": "Retention class", "description": "Coded retention rule governing this run record.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-006" ] }, { "id": "de-retention-until", "name": "Retention until", "description": "Earliest instant at which the record may lawfully be disposed of.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-006" ] }, { "id": "de-legal-basis", "name": "Legal basis", "description": "Cited basis for retaining or erasing the record.", "value_kind": "text", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-006" ] }, { "id": "de-legal-hold-flag", "name": "Legal hold flag", "description": "Whether disposal is suspended by an active hold.", "value_kind": "boolean", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-006" ] }, { "id": "de-disposition-action", "name": "Disposition action", "description": "Action taken at end of retention: destroy, redact or transfer.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-005" ] }, { "id": "de-tombstone-ref", "name": "Tombstone reference", "description": "Minimal residue proving a disposed record previously existed.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-006" ] } ], "artifacts": [ { "id": "art-retention-and-hold-register", "name": "Retention and hold register", "description": "Register recording the retention class, computed disposal date, active holds, erasure requests and disposition outcomes for run records, enabling proof of both retention and lawful destruction.", "media_or_form": [ "structured register", "append-only log", "tabular report" ], "serial": false, "identity_strategy": "Register entries are named by run identifier plus retention-decision ordinal; the register itself is identified by its governing retention schedule reference.", "source_refs": [ "SRC-006", "SRC-005" ] } ], "inline_only_rationale": null }, { "id": "find-record-integrity", "name": "Record integrity and tamper evidence", "description": "How a closed run record is made alteration-evident. C2PA demonstrates the pattern that matters here: assertions gathered into a claim, hashed, and covered by a claim signature verifiable against a trust anchor. A run record without a seal cannot support an accountability claim, and corrections must be recorded as new assertions rather than overwrites.", "source_refs": [ "SRC-016", "SRC-006", "SRC-015" ], "questions": [ { "id": "q-integ-seal", "text": "How is the run record sealed at close so that any later alteration becomes detectable?", "kind": "security", "answer_data": [ "Digest algorithm", "Seal signature", "Signer identity" ] }, { "id": "q-integ-immutable", "text": "Which fields become immutable when the run closes, and which may still legitimately be appended afterwards?", "kind": "constraint", "answer_data": [ "Immutable field set", "Append-permitted field set", "Late-arrival window" ] }, { "id": "q-integ-correction", "text": "How is a correction recorded without overwriting the original assertion?", "kind": "lifecycle", "answer_data": [ "Correction record reference", "Superseded assertion reference", "Correction rationale" ] }, { "id": "q-integ-verify", "text": "Who can verify the seal, against which trust anchor, and what happens when verification fails?", "kind": "validation", "answer_data": [ "Verifier role", "Trust anchor reference", "Verification failure procedure" ] } ], "data_elements": [ { "id": "de-record-digest", "name": "Record digest", "description": "Digest over the canonicalised run record at seal time.", "value_kind": "text", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-digest-algorithm", "name": "Digest algorithm", "description": "Algorithm used to compute the record digest.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-seal-signature", "name": "Seal signature", "description": "Signature covering the record digest and its metadata.", "value_kind": "binary", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016" ] }, { "id": "de-sealed-at", "name": "Sealed at", "description": "Instant at which the record was sealed.", "value_kind": "timestamp", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-011" ] }, { "id": "de-correction-of-ref", "name": "Correction of", "description": "Reference from a correcting record to the assertion it supersedes.", "value_kind": "reference", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-016" ] } ], "artifacts": [ { "id": "art-integrity-seal", "name": "Integrity seal", "description": "Signed digest covering the closed run record and its referenced payload digests, giving an independent verifier the means to detect any later alteration.", "media_or_form": [ "detached signature", "hash-chain checkpoint entry", "signed attestation" ], "serial": false, "identity_strategy": "Identified by the record digest it covers, with the run identifier, signer identity and seal instant carried as metadata.", "source_refs": [ "SRC-016" ] } ], "inline_only_rationale": null } ] }, { "id": "lay-access-and-interop", "name": "Access control and interoperability", "description": "Who may see which parts of the record, and how the record moves into and out of external schemas.", "source_refs": [ "SRC-004", "SRC-009", "SRC-001", "SRC-014" ], "findings": [ { "id": "find-access-scoping", "name": "Access scoping and confidentiality of run content", "description": "Separating access to run metadata from access to run content. OpenTelemetry warns explicitly that input messages, output messages, system instructions, prompt variables and tool definitions are likely to contain sensitive and personal data, which makes field-level rather than record-level access control the default position. Residency constraints on the record may differ from those on the inference itself.", "source_refs": [ "SRC-004", "SRC-009", "SRC-014" ], "questions": [ { "id": "q-acc-tiers", "text": "Who may read run inputs and outputs, as opposed to run metadata and measures only?", "kind": "access", "answer_data": [ "Role to field-set mapping", "Content access approval path", "Default deny scope" ] }, { "id": "q-acc-sensitivity", "text": "Which fields are classified as sensitive, and what masking or truncation applies by default?", "kind": "privacy", "answer_data": [ "Sensitivity class per field", "Default masking rule", "Unmasking authority" ] }, { "id": "q-acc-audit", "text": "Which accesses to the run record must themselves be logged, and what does that access log capture?", "kind": "security", "answer_data": [ "Auditable access types", "Access log fields", "Access log retention" ] }, { "id": "q-acc-residency", "text": "Which residency or cross-border constraints bind the record, and how do they differ from the constraints on the inference itself?", "kind": "spatial", "answer_data": [ "Record residency region", "Inference residency region", "Transfer control mechanism" ] } ], "data_elements": [ { "id": "de-sensitivity-class", "name": "Sensitivity classification", "description": "Confidentiality class assigned to a field or payload of the run.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004" ] }, { "id": "de-field-level-acl", "name": "Field-level access rule", "description": "Rule granting or denying a role access to a specific field set.", "value_kind": "object", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-009" ] }, { "id": "de-record-residency-region", "name": "Record residency region", "description": "Region in which the run record must be stored.", "value_kind": "code", "cardinality": "0..1", "required": false, "source_refs": [ "SRC-014" ] }, { "id": "de-masking-rule", "name": "Masking rule", "description": "Default transformation applied to sensitive content on read.", "value_kind": "code", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-004" ] } ], "artifacts": [ { "id": "art-access-audit-trail", "name": "Access audit trail", "description": "Append-only trail of reads, exports and unmasking events against run records, capturing who accessed which fields, when and under what authorisation.", "media_or_form": [ "append-only log", "structured record collection" ], "serial": true, "identity_strategy": "Entries are named by run identifier plus a zero-padded, gapless access ordinal; the accessing principal is referenced by governed identity, never copied.", "source_refs": [ "SRC-009", "SRC-004" ] } ], "inline_only_rationale": null }, { "id": "find-export-and-interoperability", "name": "Export profiles and schema mapping", "description": "Projecting the run into external schemas and importing foreign run records without silently mis-stating their origin. Because gen_ai.* attributes are still in Development status, and providers differ in identifiers and usage fields, every mapping is provisional and lossy in known ways; the model records the loss rather than claiming conformance.", "source_refs": [ "SRC-001", "SRC-002", "SRC-015", "SRC-008" ], "questions": [ { "id": "q-interop-targets", "text": "Into which external schemas can a run record be projected, and which profile governs each projection?", "kind": "interoperability", "answer_data": [ "Export profile identifier", "Target schema URI and version", "Profile stability status" ] }, { "id": "q-interop-loss", "text": "Which fields have no target in a given external schema, and how is that loss declared to the consumer?", "kind": "constraint", "answer_data": [ "Unmapped field list", "Loss declaration mechanism", "Round-trip fidelity class" ] }, { "id": "q-interop-extensions", "text": "How are vendor-specific attributes carried without polluting the canonical model?", "kind": "definition", "answer_data": [ "Extension namespace", "Extension registration rule", "Promotion criteria to canonical" ] }, { "id": "q-interop-imported", "text": "How is an imported third-party run record marked as externally asserted rather than locally observed?", "kind": "provenance", "answer_data": [ "Assertion origin code", "Importing system reference", "Import instant" ] } ], "data_elements": [ { "id": "de-export-profile-id", "name": "Export profile identifier", "description": "Identifier of the governed profile controlling one projection.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-target-schema-uri", "name": "Target schema URI", "description": "Versioned URI of the external schema a projection targets.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-unmapped-fields", "name": "Unmapped fields", "description": "Canonical fields with no counterpart in the target schema.", "value_kind": "collection", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-002" ] }, { "id": "de-extension-namespace", "name": "Extension namespace", "description": "Namespace carrying vendor-specific attributes outside the canonical set.", "value_kind": "identifier", "cardinality": "0..n", "required": false, "source_refs": [ "SRC-001" ] }, { "id": "de-assertion-origin", "name": "Assertion origin", "description": "Whether a record was locally observed or externally asserted and imported.", "value_kind": "code", "cardinality": "1", "required": true, "source_refs": [ "SRC-008" ] } ], "artifacts": [ { "id": "art-run-export-package", "name": "Run export package", "description": "Self-contained projection of one run into a named external profile, bundling the mapped record, the field mapping table and an explicit declaration of unmapped fields.", "media_or_form": [ "export bundle", "structured record with mapping table", "telemetry export" ], "serial": false, "identity_strategy": "Named by run identifier plus export profile identifier and profile version; regenerated rather than edited when the profile changes.", "source_refs": [ "SRC-001", "SRC-002" ] } ], "inline_only_rationale": null } ] } ] } ] }, "functions": [ { "id": "fn-open-run", "name": "Open run", "description": "Accept an invocation, assign the authoritative run identifier, bind the run to a configuration snapshot, capability surface and principal, and record the start instant.", "inputs": [ "Invocation request with operation type and inputs", "Configuration snapshot reference and digest", "Initiating principal and granted scopes", "Correlation context (traceparent, conversation identifier, parent run)" ], "outputs": [ "Run record in active state with authoritative identifier", "Recorded start instant and correlation keys", "Capability surface snapshot reference" ], "preconditions": [ "The configuration snapshot is immutable and resolvable by digest", "The presented token is audience-bound to each intended resource", "The applicable content-capture mode has been determined for this run class" ], "effects": [ "Run enters the active state and becomes referenceable by other records", "Start event time is fixed and cannot later be edited", "Retention clock start point is established" ], "source_refs": [ "SRC-001", "SRC-010", "SRC-011", "SRC-015" ] }, { "id": "fn-record-step", "name": "Record step", "description": "Append one ordered step to the run trajectory: a planning step, a model call, a tool call or a delegated sub-run reference, with its payload references, timings and error disposition.", "inputs": [ "Run identifier and parent step reference", "Step kind and payload or payload reference", "Step start and end instants", "Error disposition where applicable" ], "outputs": [ "Appended step record with gapless sequence index", "Updated inference-call and tool-call counters" ], "preconditions": [ "The run is in an active state", "Payloads above the inline threshold have been externalised and digested" ], "effects": [ "Trajectory grows by exactly one immutable entry", "Protocol errors and tool execution errors are kept distinguishable" ], "source_refs": [ "SRC-002", "SRC-003", "SRC-009" ] }, { "id": "fn-request-oversight-decision", "name": "Request oversight decision", "description": "Suspend a gated action, present the exact proposed inputs to an authorised human, and record the approval, denial or timeout together with what was shown.", "inputs": [ "Proposed action and its arguments", "Gating rule identifier and required oversight level", "Candidate decider role" ], "outputs": [ "Oversight decision record", "Resume or abort instruction for the run" ], "preconditions": [ "The action is classified as requiring human approval for this run class", "The decider can be shown the actual arguments before the call is made" ], "effects": [ "Run pauses at the gate until a decision or timeout", "Decider identity and presented context become part of the audit evidence" ], "source_refs": [ "SRC-009", "SRC-005", "SRC-012" ] }, { "id": "fn-close-run", "name": "Close run", "description": "Move the run to a terminal state, record the finish reason or error type, fix the end instant, and finalise consumption counters and completeness classification.", "inputs": [ "Run identifier", "Terminal outcome and finish or stop reason", "Provider-reported usage counters" ], "outputs": [ "Run record in a terminal state", "Completeness classification and end instant", "Provisional cost figure flagged as an estimate" ], "preconditions": [ "All in-flight steps have completed, failed or been cancelled", "The run is not already in a terminal state" ], "effects": [ "Immutable field set becomes read-only", "Late-arriving evaluations and reconciled costs may still be appended within the declared window" ], "source_refs": [ "SRC-001", "SRC-013", "SRC-014" ] }, { "id": "fn-seal-run-record", "name": "Seal run record", "description": "Canonicalise the closed run record together with its payload digests, compute a digest and apply a verifiable signature so that later alteration is detectable.", "inputs": [ "Closed run record", "Referenced payload digests", "Signing key and trust anchor reference" ], "outputs": [ "Integrity seal covering the record digest", "Seal instant and signer identity" ], "preconditions": [ "The run is in a terminal state", "Canonicalisation rules are pinned to a stated version" ], "effects": [ "Any subsequent change requires a correction record rather than an overwrite", "Independent parties can verify the record against the trust anchor" ], "source_refs": [ "SRC-016", "SRC-006" ] }, { "id": "fn-attach-evaluation", "name": "Attach evaluation", "description": "Bind an evaluation result or human feedback signal to a closed run by reference, without mutating the sealed run record.", "inputs": [ "Run identifier", "Evaluator identity and version", "Score, label and explanation, or human feedback value" ], "outputs": [ "Evaluation result record referencing the run", "Updated quality index for the run" ], "preconditions": [ "The run identifier resolves", "The evaluation arrives inside the declared acceptance window" ], "effects": [ "Quality evidence accumulates as separate records rather than edits", "Automated and human signals remain distinguishable" ], "source_refs": [ "SRC-004", "SRC-012" ] }, { "id": "fn-reconstruct-run", "name": "Reconstruct run", "description": "Rebuild the full trajectory, inputs, bindings and outputs of a past run for audit, incident investigation or reproduction, reporting explicitly on any evidence that can no longer be resolved.", "inputs": [ "Run identifier", "Requester identity and access scope", "Reconstruction depth including sub-runs" ], "outputs": [ "Reconstructed trajectory with resolved payloads", "Evidence-gap report listing unresolvable or redacted references", "Seal verification verdict" ], "preconditions": [ "Requester holds access to the requested field tiers", "Referenced payload stores are reachable or their absence is recordable" ], "effects": [ "Reconstruction is itself logged in the access audit trail", "Missing evidence is reported rather than silently omitted" ], "source_refs": [ "SRC-015", "SRC-016", "SRC-009" ] }, { "id": "fn-redact-run-content", "name": "Redact run content", "description": "Remove or mask sensitive content from a run's payloads while preserving the record skeleton, digests and audit structure required by retention duties.", "inputs": [ "Run identifier and target field or payload set", "Redaction basis such as an erasure request", "Approving authority" ], "outputs": [ "Redacted payload references with preserved digests of the removed content", "Redaction record with basis and approver" ], "preconditions": [ "No active legal hold blocks the redaction", "Statutory minimum log elements survive the redaction or an exception is recorded" ], "effects": [ "Content becomes unrecoverable while the fact of its prior existence stays provable", "The prior integrity seal is superseded by a correction and a new seal" ], "source_refs": [ "SRC-006", "SRC-004", "SRC-016" ] }, { "id": "fn-apply-retention-decision", "name": "Apply retention decision", "description": "Evaluate a run record against its retention class, active holds and erasure requests, then hold, redact, transfer or destroy it and register the disposition.", "inputs": [ "Run identifier and retention class", "Current holds and erasure requests", "Applicable legal basis" ], "outputs": [ "Disposition action and instant", "Tombstone entry where the record is destroyed", "Register entry evidencing the decision" ], "preconditions": [ "The statutory minimum retention period has elapsed or an overriding basis applies", "Conflicts between erasure duties and retention duties have been resolved by the recorded rule" ], "effects": [ "Record is retained, redacted, transferred or destroyed", "Proof of lawful disposal survives the record itself" ], "source_refs": [ "SRC-006", "SRC-005" ] }, { "id": "fn-export-run", "name": "Export run", "description": "Project a run record into a named external profile, emitting the mapped payload together with its field mapping table and an explicit declaration of unmapped fields.", "inputs": [ "Run identifier", "Export profile identifier and version", "Requested field tier" ], "outputs": [ "Run export package", "Unmapped field declaration and round-trip fidelity class" ], "preconditions": [ "The export profile is registered and versioned", "The requester is authorised for the requested field tier" ], "effects": [ "Downstream consumers receive an explicit statement of lossiness instead of an implied conformance claim", "Export is recorded in the access audit trail" ], "source_refs": [ "SRC-001", "SRC-002", "SRC-008" ] }, { "id": "apply-content-controls", "name": "Apply content capture and redaction", "description": "Decide whether to store payloads inline, truncated, hashed, redacted or externally, and record the manifest.", "inputs": [ "run-id", "raw-payloads", "capture-policy", "data-class-findings" ], "outputs": [ "redaction-and-capture-manifest", "content-hash", "external-content-uri-if-any" ], "preconditions": [ "A capture policy exists for the environment (for example production versus pre-production)." ], "effects": [ "Unredacted sensitive content is stored only according to policy.", "Integrity hashes remain even when payloads are offloaded or truncated." ], "source_refs": [ "SRC-001", "SRC-022" ] } ], "composition": [ { "target": "WM-AI-001", "relation": "CHILD", "purpose": "The run belongs to the AI system whose registration, intended purpose and risk classification determine which record-keeping obligations attach to each execution. The system model owns the classification; the run inherits it as a coded reference.", "required": true, "source_refs": [ "SRC-005", "SRC-012" ] }, { "target": "WM-AI-002", "relation": "REFERENCE", "purpose": "Every run references the agent that performed it, mirroring the PROV-O separation of Activity from Agent and carrying the observed agent identifier, name and description without redefining the agent.", "required": true, "source_refs": [ "SRC-008", "SRC-002" ] }, { "target": "WM-AI-005", "relation": "REFERENCE", "purpose": "Every run references the immutable configuration snapshot it was bound to, by reference and digest, so reproducibility questions resolve against a single governed snapshot rather than a copy embedded in each run.", "required": true, "source_refs": [ "SRC-001", "SRC-013" ] }, { "target": "OpenTelemetry GenAI semantic conventions (spans, metrics, events)", "relation": "ALIGN", "purpose": "Field-level alignment for operation naming, provider discrimination, model binding, token usage, agent and tool attributes, and content-capture events. Alignment only: the gen_ai.* attribute set is in Development status, so no conformance is claimed.", "required": false, "source_refs": [ "SRC-001", "SRC-002", "SRC-003", "SRC-004" ] }, { "target": "W3C PROV-O (http://www.w3.org/ns/prov#)", "relation": "ALIGN", "purpose": "Typing alignment: the run maps to prov:Activity with used, wasGeneratedBy, wasAssociatedWith and actedOnBehalfOf edges, giving a vendor-neutral lineage projection for outputs and delegation chains.", "required": false, "source_refs": [ "SRC-008" ] }, { "target": "W3C Trace Context Level 1", "relation": "ALIGN", "purpose": "Correlation alignment: traceparent and tracestate supply the trace and span identifiers that place a run inside a distributed execution, without ever being treated as the run's identity.", "required": false, "source_refs": [ "SRC-007" ] }, { "target": "Regulation (EU) 2024/1689 Articles 12 and 19", "relation": "ALIGN", "purpose": "Obligation alignment for the subset of runs performed by high-risk AI systems: automatic event recording over the system lifetime, the Annex III point 1(a) minimum log set, and provider retention of at least six months. Applies conditionally, not universally.", "required": false, "source_refs": [ "SRC-005", "SRC-006" ] }, { "target": "Model Context Protocol 2026-07-28 (tools and authorization)", "relation": "ALIGN", "purpose": "Interaction alignment for tool call structure, the protocol-error versus execution-error distinction, untrusted annotation handling, human-in-the-loop denial, and audience-bound authorization scopes recorded on the run.", "required": false, "source_refs": [ "SRC-009", "SRC-010" ] }, { "target": "C2PA Specification 2.2", "relation": "ALIGN", "purpose": "Downstream provenance alignment for media assets a run produces: signed manifests, ingredient chains and digitalSourceType disclosure. The run-to-manifest binding is an adopting-Dimension convention because C2PA defines no run identifier field.", "required": false, "source_refs": [ "SRC-016" ] } ], "serviceLayers": { "dimension": { "owner_package_requirements": [ "A Dimension adopting WM-AI-004 must name an accountable owner for run records who is distinct from the operator that generates them, and must state whether that owner acts as provider, deployer or both for each AI system in scope.", "The owner package must declare the content-capture policy per run class - verbatim, truncated, redacted or metadata-only - together with the authority that approved it, before any run is opened under that class.", "The owner package must publish the retention schedule, the minimum-retention floor applied per jurisdiction, and the conflict rule that resolves erasure requests against statutory retention duties.", "The owner package must register at least one export profile with a stated round-trip fidelity class, and must not publish conformance claims against Development-status external conventions.", "The owner package must nominate a verifier role able to check integrity seals against a declared trust anchor independently of the operator." ], "namespace_guidance": "Run identifiers minted by the adopting Dimension live in a governed namespace under the Dimension's own authority, for example a path segment reserved for AI run activities, and are resolvable to the run record. Provider-native identifiers are retained under vendor-scoped keys and are never promoted into the governed namespace. Provenance activity IRIs are minted in the same governed namespace so that lineage graphs resolve without a vendor dependency. Vendor-specific telemetry attributes stay inside an explicitly declared extension namespace and are promoted into canonical fields only after the corresponding external convention reaches stable status.", "registry_links": [ "Registry entry vr.wm-ai-004 in the world-model record plane, navigation path NAV.INF.AI.RUN, domain tags INF.AI.RUN", "Parent registry entry WM-AI-001 supplying AI system registration and risk classification", "Sibling registry entries WM-AI-002 (agent) and WM-AI-005 (configuration) referenced through typed edges", "Relations file planning/VERCY-MODEL-RELATIONS.csv holding the REFERENCE edges from WM-AI-004" ] }, "canon_and_patch": { "canonicalization_rules": [ "The canonical form of a run record is a deterministic serialisation with lexicographically ordered keys, no insignificant whitespace, and all instants normalised to the model timestamp rule; storage projections (JSON, YAML, Markdown, BSON) must round-trip to this canonical form before any digest is computed.", "Ordered collections - the step sequence, input messages, output content blocks - preserve their original order as significant; unordered collections such as granted scopes are sorted before canonicalisation so that equal sets produce equal digests.", "Payloads above the declared inline threshold are canonicalised as a reference plus digest plus media type, never as embedded bytes, so that record digests remain stable regardless of storage backend.", "Absent values and explicitly-null values are distinguished: an omitted field means not recorded, a null means recorded as absent, and a redacted field carries a redaction marker plus the digest of what was removed." ], "patch_rules": [ "Before close, a run record may be patched only by appending steps, measures and correlation keys; identity, configuration binding and start instant are write-once from the moment the run opens.", "After close and seal, no field is edited in place. A change is expressed as a correction record that references the superseded assertion, carries its own rationale and approver, and triggers a new seal while the prior seal is retained.", "Redaction is the sole mechanism that removes content after seal. It preserves the record skeleton and the digest of the removed content, and it must record its legal basis and approver.", "Late-arriving facts - reconciled costs, asynchronous evaluations - are appended as separate referencing records inside a declared acceptance window and never rewrite the closed run." ], "compatibility_rules": [ "Adding an optional field, a new coded value or a new export profile is a compatible change. Removing a field, narrowing cardinality, tightening a requirement level or changing the meaning of an existing code is breaking and requires a new model version.", "The identity priority order is treated as frozen: changing which identifier is authoritative is always a breaking change because it invalidates every existing cross-reference.", "Alignments to external conventions are versioned independently of this model. When an aligned convention changes, the export profile version increments and previously emitted packages remain valid under their stated profile version.", "Consumers must tolerate unknown fields inside declared extension namespaces and must not infer meaning from an absent optional field." ] }, "artifact_rules": { "identity_priority": [ "Authoritative master-system identifier: the run identifier assigned by the system of record that actually executed the run - for example the provider response or invocation identifier - recorded together with the identity of the issuing system.", "Governed global identifier or IRI: a resolvable IRI minted in the adopting Dimension's governed namespace, used when several master identifiers compete or when a cross-system lineage anchor is required.", "UUID or ULID assigned by the adopting Dimension, used only when neither an authoritative master identifier nor a governed IRI exists, and always marked as locally assigned.", "Never identity: a date or timestamp, a trace identifier, a conversation identifier, a sequence ordinal, a filename or a content digest. These are correlation, ordering or integrity values and are recorded as such." ], "timestamp_rule": "Every instant is recorded per RFC 3339 with explicit seconds and an explicit numeric UTC offset or the Z designator; -00:00 is reserved for the case where UTC is known but the local offset is not. Event time - when the run, step or tool call actually occurred - is recorded separately from observation time (when a collector saw it) and ingestion time (when the durable store accepted it) whenever those differ, and a run must never be dated by its ingestion time. Billing-bucket boundaries are a fourth, derived time that is reported alongside but never substituted for event time. Durations are recorded in seconds and are derived from event-time instants, not from observation instants.", "serial_naming_rule": "Artifacts marked serial - the run step sequence, tool call log entries and the access audit trail - are named by the authoritative run identifier plus a zero-padded, gapless ordinal that expresses position within the run, not elapsed time. Ordinals start at one, are never reused after a correction, and any gap must be explained by an explicit gap marker so that a missing entry is distinguishable from an entry never created.", "integrity_rule": "Each externalised payload is content-addressed by a named digest algorithm and referenced from the run record by URI plus digest. At close, the canonicalised run record together with its referenced payload digests is digested and signed, producing an integrity seal verifiable against a declared trust anchor. Verification failure is a reportable condition, not a silent fallback; a record whose seal cannot be verified may be read but must not be used as accountability evidence without that fact being stated." }, "policies": [ "Content capture is a governed decision, not an implementation default. Because run inputs, outputs, system instructions and tool definitions are likely to contain personal or sensitive data, verbatim capture requires a recorded basis, a declared retention period and a masking rule, and the chosen capture mode is stored on every run so a later reader knows what was deliberately not kept.", "Alignment is never conformance. External conventions whose attributes are marked Development status may be mapped and cited, but the adopting Dimension must not publish a conformance claim against them and must record the stability status of each alignment at the time of mapping.", "Regulated obligations apply conditionally. Record-keeping duties under the EU AI Act attach to high-risk systems, and the enumerated minimum log set attaches to a single Annex III category; applying those obligations universally is treated as a modelling defect, and so is failing to apply them where they do bind.", "Credentials are never persisted. The run records principals, scopes, token audiences and validation outcomes, but no access token, refresh token, API key value or other credential material is written into any artifact of this model.", "Cost figures carry their confidence. A per-run cost is an estimate until reconciled against authoritative bucketed reporting, and the cost basis field must state which it is; estimates and billed amounts are never summed without that distinction being preserved.", "Treat run execution bodies as immutable audit records; allow only annotated, redaction, retention and integrity patches.", "Minimise payload capture in production; default to hash, truncate or external store with access exceptions for unredacted content.", "Report billed token counts when both billed and consumed counts exist; do not invent monetary cost from unofficial rate cards.", "Do not fabricate conversation identifiers from trace ids, content hashes or fresh UUIDs.", "Apply the stricter of high-risk log retention and personal-data law; record holds and skeleton-versus-payload deletion explicitly.", "Declare export profile and convention version; do not claim EU, OTel or OpenInference conformance without evidence." ], "crud": { "read": [ "Metadata-tier reads - identity, class, state, timings, measures - are available to any role holding the model's baseline read scope.", "Content-tier reads - inputs, outputs, reasoning and tool payloads - require an explicit content scope and are logged in the access audit trail with the accessing principal and field set.", "Reconstruction of a full trajectory, including sub-runs, is a distinct privileged read that must return an evidence-gap report naming every reference it could not resolve.", "Exports are reads: every export package emitted is recorded with its profile, version and requesting principal." ], "create": [ "A run record is created only by opening a run, which fixes its identifier, configuration binding, principal and start instant in a single atomic step.", "Step records, tool call entries and access audit entries are created by append only, each receiving the next gapless ordinal within its serial.", "Evaluation results, oversight decisions and usage-and-cost statements are created as independent records that reference the run rather than mutate it.", "Imported third-party run records are created with assertion origin set to externally asserted and with the importing system and import instant recorded." ], "update": [ "Before close, updates are restricted to appending trajectory entries and measures; identity, configuration binding and start instant are write-once.", "After close and seal, in-place update is prohibited. Change is expressed as a correction record referencing the superseded assertion, with rationale and approver, followed by a new seal.", "Redaction is the only operation that removes content post-seal; it preserves the skeleton and the digest of the removed content and records its legal basis.", "Cost reconciliation supersedes an earlier statement with a new one and retains the supersession edge rather than overwriting the estimate." ], "delete": [ "Deletion is permitted only through an applied retention decision that has evaluated the retention class, any active legal hold and any pending erasure request.", "Destruction leaves a tombstone recording the run identifier, its class, its retention basis and the disposal instant, so that the prior existence of the record stays provable.", "Where a statutory minimum retention period has not elapsed, an erasure request is satisfied by content redaction rather than record destruction, and the exception is registered.", "Cascade rules are explicit: destroying a parent run does not implicitly destroy referenced sub-runs, evaluation records or cost statements, each of which carries its own retention class." ] }, "roles": [ { "name": "Model steward", "responsibilities": [ "Maintain the canonical structure, code lists and export profiles of WM-AI-004 and version them under the compatibility rules", "Adjudicate boundary disputes with WM-AI-001, WM-AI-002 and WM-AI-005 and record the outcome as a boundary note", "Re-assess external alignments whenever an aligned convention changes stability status" ] }, { "name": "Run record custodian", "responsibilities": [ "Operate the stores holding run records and externalised payloads, and guarantee their immutability and reachability", "Compute and apply integrity seals at close and preserve superseded seals across corrections", "Execute retention decisions, redactions and disposals, and keep the retention and hold register current" ] }, { "name": "Accountable deployer", "responsibilities": [ "Answer for the effects of runs performed under their control and confirm which regulatory class applies", "Ensure required human oversight is genuinely available and that gated actions cannot execute unapproved", "Approve the content-capture policy and the retention schedule for each run class in scope" ] }, { "name": "Auditor or conformity assessor", "responsibilities": [ "Reconstruct past runs and verify that mandated minimum log elements are present", "Verify integrity seals against the declared trust anchor and report verification failures", "Test that access to content-tier fields is scoped, logged and justified" ] }, { "name": "Privacy and data protection officer", "responsibilities": [ "Set and review masking, pseudonymisation and truncation rules for sensitive run content", "Adjudicate conflicts between erasure requests and statutory minimum retention duties and record the resolution", "Approve cross-border transfers and residency placement of run records" ] }, { "name": "Cost and usage controller", "responsibilities": [ "Reconcile estimated per-run costs against authoritative bucketed reporting and publish superseding statements", "Maintain allocation rules for sub-runs, cached tokens and shared capacity so nothing is double counted", "Attribute consumption to tenants, workspaces and cost centres consistently with the identity rules" ] } ], "access": { "default_rule": "Deny by default, then grant at the narrowest scope that satisfies the stated purpose. Run metadata and measures are separable from run content: a principal granted metadata access sees identity, class, state, timings, counters and references, but not inputs, outputs, reasoning or tool payloads, which require an additional explicit content scope tied to a recorded purpose.", "scopes": [ "bundle", "layer", "finding", "artifact" ], "exceptions": [ "An auditor or conformity assessor acting under a recorded mandate may read content-tier fields for named runs without the ordinary purpose approval, provided every such read is logged and time-bounded.", "A break-glass grant may be issued during an active incident investigation; it expires automatically, requires two-party authorisation, and produces a mandatory post-hoc review entry.", "A data subject exercising access rights may receive the content they themselves supplied and the output returned to them, without receiving system instructions, tool credentials or another party's content.", "Content that a guardrail blocked is not retained, so no access scope can retrieve it; only the refusal code, decision metadata and a digest are available.", "Records under an active legal hold remain readable to the holding authority even where a retention decision would otherwise have disposed of them.", "Competent-authority or market-surveillance request for Article 12 logs.", "Documented incident investigation or legal hold.", "Human overseer viewing the payload they must verify.", "Pre-production environments where capture policy explicitly permits full content." ], "audit_requirements": [ "Every content-tier read, every unmasking action and every export must be written to the access audit trail with the accessing principal, the field set, the purpose and the instant.", "Break-glass grants must record the two authorising parties, the incident reference, the expiry and the post-hoc review outcome.", "Redactions and disposals must record the basis, the approver, the instant and the digest of the removed content, and must be reflected in the retention and hold register.", "Failed access attempts against content-tier fields are logged with the same fidelity as successful ones, since they are the primary signal of scope misconfiguration.", "The access audit trail is itself append-only and sealed on the same schedule as the run records it covers.", "Every read of unredacted payloads, every export, every retention override and every deletion must itself produce an auditable event with actor, event time and ingestion time.", "Sampling decisions that drop a high-risk-required log are policy violations and must be recorded as exceptions." ] }, "agents_bootstrap": { "filename": "AGENTS.md", "required_fields": [ "Name", "Type", "Specification URL", "Storage type URL", "Interface URL", "Processes URL", "Model ID", "Registry ID", "Owner and accountable deployer", "Content capture policy reference", "Retention schedule reference" ], "read_order": [ "AGENTS.md - identify the model, its type and where its specification, storage, interface and processes are documented", "Specification URL - read the scope statement, boundary notes and the bundle, layer and finding structure before touching any record", "Storage type URL - learn the concrete projection in use (structured store, log stream, object store, version control) and the canonicalisation rules that apply to it", "Interface URL - learn the read, create, update and delete operations actually exposed and the scopes each one requires", "Processes URL - follow the governed procedures for opening, closing, sealing, redacting, retaining and exporting runs", "Composition links - resolve WM-AI-001, WM-AI-002 and WM-AI-005 before asserting anything about the agent, its configuration or the system's risk class" ] } }, "coverage": { "claim": "Base is Claude's structure (7 bundles / 15 layers / 28 findings / 10 functions over 16 sources, 15 primary, 9 organisations), extended by two grok findings (executing-agent identity binding; rare endings covering folded retries, cancellation/timeout, context compaction and fetch-without-inference) and one grok function (content capture policy at capture time). Together these cover run identity and correlation, invocation and execution binding, step trajectory and tool/retrieval/delegation, outcome, state and temporal semantics, accountability and authority, consumption, cost and quality evidence, and record governance for one bounded AI inference or agent run. This is not a claim of universal completeness: agent memory-step operations, OpenTelemetry MCP conventions, the billed-versus-consumed token rule, encrypted-reasoning replay signatures and proxy-versus-upstream provider discrimination are explicitly deferred, and cross-vendor run identity plus per-run cost remain adopting-Dimension decisions rather than settled standards.", "confidence": "medium", "checklist": [ { "dimension": "identity", "status": "covered", "notes": "Covered by find-run-identifier with an explicit priority order placing the authoritative master-system identifier first. Confidence is limited by a real gap: no governed global identifier for an AI run exists. OpenTelemetry makes gen_ai.response.id Required only for fetch_response, Anthropic returns a msg_ identifier and Bedrock a requestId, so cross-vendor identity is a Dimension decision, not a standards fact." }, { "dimension": "lifecycle", "status": "covered", "notes": "find-run-state-and-termination defines the state machine and terminal outcomes, drawing on OpenTelemetry finish_reasons plus the Stable error.type and Anthropic stop_reason values. The state code list itself is synthesised: no cited source publishes a normative run state machine, so it is presented as a required Dimension declaration rather than an external standard." }, { "dimension": "relationships", "status": "covered", "notes": "Correlation to traces and conversations (find-correlation-keys), parent-child step nesting (find-step-and-deliberation-record), delegation to sub-runs (find-subrun-and-delegation) and derivation edges (find-attribution-and-accountability) are each separately sourced. Rollup semantics across a delegation boundary have no standard basis and are flagged in the finding." }, { "dimension": "temporal", "status": "covered", "notes": "find-temporal-semantics fixes RFC 3339 with seconds and an explicit offset or Z, separates event, observation, ingestion and billing-bucket time, and records clock source and skew tolerance. EU AI Act Article 12(3)(a) independently requires start and end date and time for the regulated class." }, { "dimension": "provenance", "status": "covered", "notes": "Three distinct layers of provenance are separated: input-segment origin and trust (find-input-provenance-and-trust), formal PROV-O activity typing (find-attribution-and-accountability) and asset-level C2PA manifests (find-generated-asset-provenance). The weak link is stated: C2PA defines no run identifier field, so that binding is a local convention." }, { "dimension": "ownership", "status": "covered", "notes": "Record owner and accountable deployer are modelled as distinct fields because EU AI Act Article 19 puts the keeping duty on the provider for logs under their control while effects are answerable elsewhere. Tenant, workspace and cost-centre attribution follow the provider dimensions evidenced in the Anthropic Usage and Cost API." }, { "dimension": "validation", "status": "covered", "notes": "Output schema validation (MCP outputSchema, clients SHOULD validate), usage reconciliation between provider-reported and locally computed counts, reproducibility classification, and seal verification against a trust anchor are each modelled with an explicit failure path rather than an assumed success." }, { "dimension": "access", "status": "covered", "notes": "Metadata-tier and content-tier access are separated by default, justified by the explicit sensitive-data warnings OpenTelemetry attaches to input messages, output messages, system instructions, prompt variables and tool definitions. Break-glass, data-subject and auditor exceptions are enumerated with their audit obligations." }, { "dimension": "retention and deletion", "status": "covered", "notes": "find-statutory-log-and-retention carries the Article 19 floor of at least six months for provider-held logs, the financial-services carve-out, the redaction-instead-of-destruction path where erasure and retention conflict, tombstones on disposal, and explicit non-cascading rules for sub-runs and derived records." }, { "dimension": "interoperability", "status": "covered", "notes": "find-export-and-interoperability requires named, versioned export profiles that declare unmapped fields and a round-trip fidelity class, and requires imported records to be marked as externally asserted. Conformance claims are prohibited while aligned conventions remain in Development status." }, { "dimension": "authority and delegation", "status": "covered", "notes": "Grounded in MCP authorization: OAuth 2.1, RFC 8707 resource indicators on every request, audience-bound validation, and the prohibition on accepting or transiting foreign tokens. PROV-O actedOnBehalfOf supplies the responsibility chain across delegation hops." }, { "dimension": "cost and metering", "status": "gap", "notes": "Structurally covered by find-token-and-resource-usage and find-cost-attribution, but marked a gap because no cited telemetry standard defines a cost metric at all - the OpenTelemetry GenAI metrics document contains none - and authoritative cost arrives only as delayed daily aggregates with known exclusions. Per-run cost is therefore always derived, and the model records that rather than hiding it." }, { "dimension": "human oversight", "status": "covered", "notes": "MCP states there SHOULD always be a human able to deny tool invocations and that inputs should be shown before the call; EU AI Act Article 12(3)(d) requires identifying the verifying natural persons for one Annex III category. The model captures what the decider was shown, not only the verdict." }, { "dimension": "reproducibility", "status": "covered", "notes": "find-decoding-params-and-reproducibility ties the run to an immutable configuration snapshot and forces an explicit reproducibility class. It also handles the counterexample that sampling parameters are deprecated for some current models, so their presence cannot be assumed." }, { "dimension": "security of the record", "status": "covered", "notes": "Integrity sealing follows the C2PA pattern of assertions gathered into a signed claim with a hard binding. Credentials are prohibited from every artifact. Untrusted-content handling follows the MCP requirement to treat tool annotations and results as untrusted absent a trusted server." }, { "dimension": "measurement", "status": "covered", "notes": "Token types, non-token resources, agent-level inference and tool call counts, operation duration and time-to-first-token map onto the OpenTelemetry GenAI metrics set; the model adds an explicit counting-authority field because provider and local counts routinely differ." }, { "dimension": "spatial", "status": "covered", "notes": "Serving region and inference geography are recorded on the execution binding, and record residency is modelled separately from inference residency, since the two are governed by different rules and can legitimately differ." } ], "known_omissions": [ "Training-time provenance, dataset lineage and model card content, which belong to the parent AI system model.", "Conversation, session or thread as a container entity: the run references a conversation identifier but does not define what a conversation is or how it is bounded.", "Streaming chunk-level telemetry below the step granularity; time-per-output-chunk is referenced as a measure but individual chunk records are out of scope.", "Rate limiting, quota and capacity management, which shape whether a run may start but are properties of the serving platform.", "Incident management, post-market monitoring case files and regulatory notification workflows that consume run records downstream.", "Pricing schedules, rate cards and invoices; only the derived cost attributable to a run is in scope.", "Federated or fully on-device runs where no network telemetry exists and neither server address nor provider-side usage counters are obtainable.", "Physical actuation and robotics safety envelopes, consistent with the registry's zero robotics factor for this model.", "Person and organisation master data for initiators, approvers and deployers, which is referenced by governed identifier only.", "OpenTelemetry MCP semantic conventions were listed by the GenAI repository but not ingested in this research budget, so MCP-flavoured tool spans are a gap pending that document.", "ISO/IEC 42001 AI management-system controls and ISO/IEC 23894 risk management were not fetched and are not claimed.", "IETF SCITT AI-agent action receipts are an emerging draft, not an adopted alignment.", "Provider-specific cost APIs, invoices, reservations and spot GPU economics are omitted beyond run-level quantities.", "Multi-agent A2A directories, long-running unsupervised agent loops without a closable span, and hardware/accelerator traces are omitted.", "C2PA/watermark packaging of generated media is omitted.", "NIST AI 800-4 post-deployment monitoring (March 2026) appeared during search but was not fully ingested; treat as a likely future alignment.", "Amendment 1 to ISO/IEC 22989 for generative AI is still a draft and not used as terminology beyond the published 2022 edition." ], "conflicts": [ "Nearly all gen_ai.* attributes in the OpenTelemetry GenAI semantic conventions are marked Development status; only error.type, server.address and server.port are Stable. Any claim of conformance to these conventions would be unsupportable today, so this model records alignment only.", "OpenTelemetry treats capture of input messages, output messages and system instructions as opt-in and attaches explicit sensitive-data warnings, while EU AI Act Article 12 requires automatic recording of events over the system lifetime. These pull in opposite directions and the model resolves it by making content-capture mode an explicit, recorded and approved decision per run class rather than a default.", "There is no cross-vendor run identifier. OpenAI returns a response id, Anthropic a msg_ prefixed message id, Amazon Bedrock a requestId, and OpenTelemetry requires gen_ai.response.id only for fetch_response operations. Identity is therefore a Dimension decision under a stated priority order, not an inherited standard.", "Anthropic documents temperature, top_p and top_k as deprecated for newer models. A reproducibility model that assumes decoding parameters are always present and always determinative would be wrong for a growing share of runs.", "No cited telemetry standard defines a cost metric. The OpenTelemetry GenAI metrics document defines token usage and durations but nothing monetary, while authoritative cost is published as daily aggregates in USD with a stated freshness lag and exclusions such as priority-tier pricing. Per-run cost cannot be both timely and authoritative.", "C2PA binds provenance to assets, not to executions, and defines no field carrying a run identifier. Any link from a manifest back to the run that produced the asset is a local convention and must not be presented as C2PA conformance.", "MCP has no protocol-level session and its stateful-handle guidance is explicitly non-normative, so run-scoped tool state has no standardised representation and cannot be assumed reconstructible from the protocol alone.", "The EUR-Lex ELI endpoint for Regulation (EU) 2024/1689 returned no retrievable body during research. Article 12 was verified against the European Commission AI Act Service Desk, but Article 19 rests on a secondary reproduction and should be re-verified against the Official Journal text before any compliance assertion is made.", "OpenTelemetry defines gen_ai.evaluation.result as an event that may be emitted independently of the trace. Evaluation therefore frequently arrives after the run has closed and been sealed, which is why the model appends it as a referencing record rather than as a field on the run.", "OpenTelemetry GenAI conventions are Development, not stable; existing instrumentations on v1.36.0 or prior are told not to change default emission.", "OpenTelemetry operation names and span kinds conflict in naming with OpenInference openinference.span.kind (especially AGENT, TOOL, LLM versus invoke_agent, execute_tool, chat).", "OpenTelemetry uses gen_ai.conversation.id and forbids UUID/trace/hash fallbacks; OpenInference uses session.id and user.id.", "Monetary cost is specified by OpenInference in USD and is absent from OpenTelemetry GenAI metrics.", "EU Articles 12/19/26 apply to high-risk systems with a six-month log floor; claiming that duty for every GenAI observability span would over-assert the law.", "Personal-data erasure can conflict with log retention; the model records the conflict rather than picking a universal winner.", "Requested model versus served model, and billed tokens versus consumed tokens, are specified as distinct and must not be collapsed.", "Secondary reporting describes a Digital Omnibus delay of some high-risk application dates; this research treats Regulation (EU) 2024/1689 as in force on its published terms and flags the delay as a regional legal-watch item rather than rewriting Article 12." ], "regional_assumptions": [ "EU AI Act Articles 12 and 19 bind only providers and deployers of high-risk AI systems placed on the Union market; they are not a global baseline and must not be applied to every run.", "The enumerated minimum log set - period of use, reference database, matching input data, identity of verifying natural persons - applies specifically to Annex III point 1(a) remote biometric identification systems, not to high-risk systems generally.", "Providers that are financial institutions subject to Union financial services law keep these logs as part of their sectoral documentation, which can override the general regime.", "NIST AI RMF 1.0 and the Generative AI Profile are voluntary United States guidance with no legal force, and they inform structure here rather than impose obligations.", "Data-residency and inference-geography dimensions are provider-specific vocabularies with values such as global or a country code; they are not standardised across vendors and cannot be assumed comparable.", "Currency and cost reporting in the cited first-party source is USD-only with costs expressed in minor units; multi-currency operation requires a conversion policy this model does not supply.", "Retention floors, erasure rights and legal-hold mechanics vary by jurisdiction; the six-month figure cited is an EU provider minimum and is not a global default.", "EU logging, oversight and six-month retention are mandatory only where Regulation (EU) 2024/1689 classifies the system as high-risk and the actor is in scope (including certain third-country providers whose output is used in the Union).", "NIST AI RMF 1.0 and NIST AI 600-1 are voluntary United States frameworks used here as monitoring and documentation alignments, not as legal duties.", "ISO/IEC 22989 is a paid international terminology standard; only publicly visible definitions (inference as process and result, agent, ML model) are relied upon.", "OpenInference cost examples assume USD; other currencies require an explicit code and are not specified by OTel.", "Server address does not by itself prove data-residency region." ], "adversarial_checks": [ "Counterexample - asynchronous and batched runs. OpenTelemetry defines a fetch_response operation and providers offer batch service tiers, so submission and result retrieval can be separated by hours. The model therefore records invocation mode and both a submission and a retrieval instant, and does not assume a run is a single synchronous interval.", "Counterexample - runs with no tool calls, no conversation and no finish reason. An embeddings or classification run has none of these. Every one of conversation identifier, finish reason, tool call and data source is modelled as optional, and the step decomposition finding does not require both model and tool calls to be present.", "Counterexample - fully local or on-device execution. There is no provider endpoint, no server address and no provider-side usage counter. The model keeps server address optional, requires an explicit usage-counting authority field, and treats absence as recordable rather than as an error.", "Counterexample - a run that is blocked before any model call. A guardrail refusal produces a run with a terminal state, a refusal code and no output. The model records the block without persisting the prohibited content, which is why find-guardrail-and-policy-decisions deliberately produces no artifact.", "Rejected structure - a prompt template layer. Templates, their variables and their version history belong to WM-AI-005; the run records only the resolved content actually sent plus a snapshot reference. Including templates here would duplicate a governed asset and let the two copies drift.", "Rejected structure - trace identifier as run identity. W3C trace-ids are 16-byte values that may be absent when unsampled, may cover many runs, and may restart at a process boundary. They are modelled as correlation keys and explicitly excluded from the identity priority order.", "Rejected structure - a separate state artifact. Materialising run state outside the run record would split authoritative state across two objects with no rule for which wins after a correction, so state stays inline with a substantive rationale recorded.", "Rejected structure - claiming an OpenTelemetry conformance profile. With the gen_ai.* attribute set in Development status, a conformance claim would be unfalsifiable today and would break on the next convention revision; the model publishes versioned export profiles with declared lossiness instead.", "Stress test - does the model survive a vendor that returns no identifier, no token counts and no finish reason? Yes: identity falls through to a locally minted ULID marked as such, usage-counting authority records that counts are locally computed, and completeness classification captures that the terminal reason is unknown. Every one of those degradations is visible in the record rather than silently absent.", "Reject any run identity that uses a date, a freshly minted conversation UUID, a trace id, or a prompt hash as the master identifier when the rules require otherwise.", "Reject a claim that OpenTelemetry GenAI is a stable specification or that OpenTelemetry defines monetary cost.", "Reject silent equivalence of OpenInference span.kind and OpenTelemetry span kind or operation name.", "Reject applying Article 12/19/26 six-month retention and verifying-person fields to systems not evidenced as high-risk.", "Reject collapsing requested versus served model or billed versus consumed tokens.", "Reject treating this event as the executing agent (WM-AI-002) or as the configuration catalogue (WM-AI-005).", "Reject capturing full prompts in production without a capture-mode and redaction manifest while still claiming privacy alignment.", "Reject fetch_response operations that report token usage, and compacted attributes set to false rather than unset." ] }, "researchAdjudication": { "providerMode": "dual-provider", "activeProviders": [ "claude", "grok" ], "waivedProviders": [], "providerPolicy": {}, "boundaryDecision": { "entry_kind": "event", "status": "accepted", "rationale": "Both providers independently reached entry_kind 'event' and the same boundary: the run is an occurrent (PROV Activity / OpenTelemetry span forest) that references but never redefines the executing agent (WM-AI-002, a continuant that bears responsibility), the configuration snapshot (WM-AI-005) or the registered AI system (WM-AI-001). Claude's eight sourced boundary notes additionally fix the four non-model neighbours that most often collapse into a run model - distributed trace/span as correlation not identity, conversation/session as a sibling container, asynchronous evaluation results, and C2PA asset manifests - so the boundary is accepted as stated with no reclassification or split required." }, "decisions": [ { "concept": "Base provider selection", "disposition": "accepted - claude as base", "rationale": "Claude carries eight sourced boundary notes against four sibling models and four non-model neighbours, an explicit occurrent/continuant framing, and a substantive inline_only_rationale on every artifact-free finding, which is direct evidence of boundary discipline. Grok is coherent but treats MCP tool spans and C2PA media provenance as out of scope or gaps, leaving two live surfaces of an agent run unmodelled. Size was not decisive; boundary completeness was." }, { "concept": "Entry kind and model boundary", "disposition": "accepted as event", "rationale": "Independent agreement between providers on entry_kind, on the PROV Activity framing, and on excluding the agent, the configuration catalogue, the AI system registration and the conversation container. No reclassification or split is warranted before structure is accepted." }, { "concept": "Executing agent identity binding (grok executing-agent-and-configuration)", "disposition": "accepted into lay-execution-binding", "rationale": "The base names the agent reference only in a boundary note; no finding asks which stable agent identifier performed the run, nor forbids minting a transient instance id as agent identity. This is the single clearest material gap in the base and it fits an existing layer without inventing structure." }, { "concept": "Rare endings: folded retries, cancellation, compaction, fetch-without-inference (grok retry-cancel-compaction-fetch)", "disposition": "accepted into lay-lifecycle-state-and-time", "rationale": "Normative negative rules (compacted must not be set false when unknown; fetch_response must not report token usage; retries fold into one logical span) that the base state finding does not carry. These are precisely the cases a chat-shaped schema gets wrong, and they are evidence-backed in the source provider." }, { "concept": "Content capture policy as an operating function (grok apply-content-controls)", "disposition": "accepted as an added function", "rationale": "Distinct in time and intent from the base fn-redact-run-content: capture mode is decided when the payload is first observed, redaction happens afterwards. The base structure already requires the decision but provides no function that makes and records it." }, { "concept": "Token usage breakdown (grok token-usage-breakdown)", "disposition": "rejected as duplicative; one rule deferred", "rationale": "Claude's find-token-and-resource-usage already covers input, output, cache-read, cache-creation and reasoning tokens plus a reconciliation rule. Importing a second usage finding would split the counting authority across two nodes. The one genuinely novel element - the OpenTelemetry rule that billed counts MUST be reported when billed and consumed both exist - is recorded as deferred research to fold into the existing finding rather than as new structure." }, { "concept": "Distributed trace and span context (grok distributed-trace-context)", "disposition": "rejected as duplicative", "rationale": "Base find-correlation-keys plus the dedicated trace-versus-identity boundary note already hold trace-id and span-id as correlation keys and explicitly exclude them from the identity priority order. Grok's added detail (all-zero id rejection, graph node ids) is a validation refinement to an existing finding, not a missing finding." }, { "concept": "Requested versus served model (grok requested-versus-served-model)", "disposition": "rejected as duplicative", "rationale": "Base find-model-and-provider-binding asks the same question directly (q-binding-requested-vs-served) and is grounded in the same separation of gen_ai.request.model from gen_ai.response.model. A second node would create two competing homes for the same scalar pair." }, { "concept": "Operation name and provider discriminator (grok operation-and-provider)", "disposition": "rejected; proxy discriminator deferred", "rationale": "Operation typing is already a base finding, deliberately inline-only with a recorded rationale against minting a second identity anchor. Adding grok's node would duplicate it. The sharp element worth keeping - whether the recorded provider name identifies a gateway or proxy rather than the upstream model vendor - is deferred as a refinement to find-model-and-provider-binding." }, { "concept": "GenAI server address and port (grok server-address-and-port)", "disposition": "rejected as duplicative", "rationale": "Base q-binding-endpoint already asks which provider, endpoint address and serving region handled the run, and the base separately models record residency against inference residency. Grok itself gives the node no artifact, confirming it is a scalar attribute rather than independent structure." }, { "concept": "Retrieval, memory, plan and nested agent spans (grok retrieval-memory-plan-nested)", "disposition": "rejected as node; memory operations deferred", "rationale": "Roughly three quarters of the node duplicates existing base findings - find-retrieval-and-grounding, find-step-and-deliberation-record and find-subrun-and-delegation. Importing it would triple-model retrieval, planning and delegation. Agent memory operations (create, search, update, delete, upsert) are genuinely absent from the base and are deferred for a scoped memory-step finding in the next pass." }, { "concept": "Output messages, choices and hidden reasoning (grok output-messages-and-modalities)", "disposition": "rejected as node; replay signatures deferred", "rationale": "Base find-output-content-and-structure covers output blocks, modalities, schema validation, digesting and human exposure, and q-step-reasoning covers intermediate reasoning capture and its retention rule. The novel element - preserving vendor signature, encrypted_content and redacted_thinking blocks so a run can be replayed without persisting plaintext chain-of-thought - is deferred as a refinement rather than a competing output node." }, { "concept": "Monetary cost of the run (grok monetary-cost)", "disposition": "rejected as node; evidence note retained", "rationale": "Base find-cost-attribution is stronger because it flags per-run cost as an estimate until reconciled against authoritative bucketed reporting. Grok's contribution is evidentiary, not structural: an OpenInference-class vendor schema does define USD cost fields, which refines but does not overturn the base statement that no cited telemetry standard defines a cost metric. Recorded as a coverage nuance and a source-verification hold." }, { "concept": "Automatic event logging duties and retention holds (grok automatic-event-logging-duties, retention-deletion-and-holds)", "disposition": "rejected as duplicative; citation hold raised", "rationale": "Base find-statutory-log-and-retention already carries the Article 12 minimum set, the six-month floor, provider-versus-deployer control, the erasure/hold conflict and tombstones. Grok's Article 26(6) deployer log-keeping citation strengthens the same finding's legal basis and is raised as a publication hold to verify and fold in, not as new structure." }, { "concept": "Access, privacy exceptions and export profiles (grok access-privacy-and-exceptions)", "disposition": "rejected as duplicative; sampling question deferred", "rationale": "Base find-access-scoping and find-export-and-interoperability cover metadata-versus-content tiers, break-glass, audit of access, named export profiles and declared lossiness. The one open question - whether a run sampled out of telemetry still satisfies a statutory automatic-recording duty - is deferred because neither provider resolved it against primary text." }, { "concept": "C2PA asset provenance in scope (base) versus out of scope (grok)", "disposition": "retained from base", "rationale": "Grok pushes media provenance packaging entirely out of scope; the base keeps a narrow, honestly-qualified finding stating that C2PA defines no run identifier field, so the manifest-to-run link is an adopting-Dimension convention and must never be presented as C2PA conformance. That is the more useful and more falsifiable position for a generated-asset run." }, { "concept": "MCP tool and authorization evidence", "disposition": "retained from base", "rationale": "Grok records MCP telemetry conventions as an un-ingested gap; the base cites the MCP specification directly for tool call semantics, protocol-versus-execution error separation, untrusted-annotation handling and RFC 8707 resource-indicator token binding. Retaining the base preserves evidence grok did not have." } ], "publicationHolds": [ "Source verification of EU AI Act text: the base provider's Article 19 evidence rests on a secondary reproduction because the EUR-Lex ELI endpoint returned no retrievable body, while the non-base provider cites EUR-Lex CELEX 32024R1689 directly. Re-verify Articles 12, 19 and 26(6) against the Official Journal text and pin an ELI-versioned citation before publishing any compliance-adjacent statement.", "Live-version verification of every accepted source: the OpenTelemetry GenAI conventions are cited from a moving main branch at Development status with no release tag, and MCP revision 2026-07-28, C2PA 2.2 and the Anthropic API and usage/cost pages are cited as accessed on a single date. Re-fetch and pin commit hashes or revision identifiers before publication.", "Verification of the two grok-sourced additions and their evidence: the executing-agent binding and rare-endings findings must resolve against OpenTelemetry SRC-001/SRC-002 and PROV-DM in the merged source list, and any residual OpenInference (authority tier 3) or paywalled ISO/IEC 22989 dependency must be either replaced with a primary citation or explicitly marked as vendor-level alignment.", "Multi-profile validation: the model has been exercised mainly against a telemetry profile and an EU high-risk profile. Validate against at least one non-EU, non-telemetry adopting profile - a fully on-device or federated run with no server address and no provider-side counters, a financial-services sectoral-documentation deployer, and a US NIST-only voluntary programme - before claiming the structure holds outside those two profiles.", "Regional legal-watch item: secondary reporting of a Digital Omnibus delay to certain high-risk application dates is unverified. Confirm the current application timetable before publishing any statement that ties the run record to an in-force compliance date.", "Cost evidence reconciliation: the base states that no cited telemetry standard defines a cost metric, while the non-base provider cites a vendor schema (OpenInference) that does define USD cost fields. Reconcile the wording so the published draft says OpenTelemetry GenAI defines no cost instrument and that vendor-level cost schemas exist, without implying either is authoritative billing." ], "deferredResearch": [ "Agent memory-step operations (gen_ai memory create, search, update, delete, upsert): add a scoped memory finding under the tool and environment layer. Deliberately not imported this pass because the carrying grok node also duplicated retrieval, planning and nested-agent structure already in the base.", "OpenTelemetry MCP semantic conventions: listed by the GenAI repository but ingested by neither provider. Ingest and reconcile MCP-flavoured tool spans against the base tool-call finding, which currently relies on the MCP protocol specification rather than the telemetry convention.", "Billed-versus-consumed token counting: fold the OpenTelemetry rule that billed counts must be reported when both billed and consumed counts exist into the existing token and resource usage finding, alongside the current provider-reported versus locally-computed reconciliation rule.", "Encrypted and redacted reasoning-block preservation for stateless replay (message content signatures, encrypted_content, redacted_thinking, tool-call reasoning signatures): evaluate as a refinement to the step deliberation and output content findings, given the tension with the base rule that reasoning content carries a shorter retention rule.", "Proxy versus upstream vendor provider discrimination: extend the model and provider binding finding so a recorded provider name that identifies a gateway or relay is distinguishable from the upstream model vendor.", "Sampling versus statutory logging duty: determine, against primary regulatory text, whether a run sampled out of full telemetry can still satisfy an automatic-recording obligation, and where the resulting rule belongs.", "Emerging alignments neither provider ingested: NIST AI 800-4 post-deployment monitoring (March 2026), ISO/IEC 42001 and ISO/IEC 23894, IETF SCITT AI agent action receipts, and the draft ISO/IEC 22989 generative-AI amendment.", "Cross-vendor run identity: no governed global identifier exists and both providers fall through to an adopting-Dimension minted identifier. Track whether any standards body publishes a run identifier before the identity priority order is frozen." ] }, "statistics": { "sources": 23, "bundles": 7, "layers": 15, "findings": 30, "questions": 115, "artifacts": 20, "functions": 11 } }