{"schema":"https://ver.cy/schemas/card/1.0.0","id":"EM-AI-03","code":"em-ai-03","url":"https://ver.cy/models/em-ai-03/","name":"AI evaluation and safety evidence","alternateNames":[],"kind":"enterprise-contour","status":"research-draft","version":"research-checkpoint","language":"en","classifiers":{"family":"Enterprise profiles","category":"Enterprise subject","entryKind":"subject","plane":"","domain":["Enterprise","AI"],"industry":["Cross-industry"],"navPath":"","tags":["EM-AI-03","W3","subject","EvaluationPlan","EvaluationRun","BenchmarkSpecification","SafetyAssessment","DeploymentDecision"],"facets":{}},"whatItIs":"Evaluation plan, benchmark, run, results, risk assessment and decision on permissible use. Metric, dataset and evidence are reused.","purpose":"Evaluation plan, benchmark, run, results, risk assessment and decision on permissible use. Metric, dataset and evidence are reused.","scope":{"in":[],"out":[],"boundaries":[]},"distinguishingFeatures":["A result pins the model and dataset versions","An evaluation is not universal outside its context","A decision has an accountable owner and conditions"],"structure":{"bundles":[]},"agentConduct":{"may":[],"mustNot":["Negative case: One high benchmark score declares the system safe for all tasks."],"requiresHuman":[]},"ethics":{"considerations":[],"affectedParties":[]},"owners":{"steward":"Руководитель AI / исследований","roles":[],"masterSystems":["Model registry","experiment tracker","eval store"]},"relations":[{"target":"EM-AI-01","type":"neighbor"},{"target":"EM-AI-02","type":"neighbor"},{"target":"WM-AI-003","type":"references","note":"conceptual-candidate"},{"target":"WM-AI-009","type":"references","note":"conceptual-candidate"},{"target":"WM-AI-008","type":"references","note":"conceptual-candidate"},{"target":"WM-AI-010","type":"references","note":"conceptual-candidate"}],"interaction":{"identity":{"applicability":"required","items":["EvaluationPlan","EvaluationRun","BenchmarkSpecification","SafetyAssessment","DeploymentDecision"]},"properties":{"applicability":"not-applicable","items":[]},"recognition":{"applicability":"required","items":[]},"capabilities":{"applicability":"required","items":[]},"hazards":{"applicability":"required","items":[]},"interfaces":{"applicability":"required","items":[]},"context":{"applicability":"required","items":[]}},"sources":[{"title":"NIST AI RMF: context, evaluation and risk management"},{"title":"PROV/SPDX: provenance of data, artifacts and actions"},{"title":"Model registry/evaluation tooling practice; map system, artifact, run and endpoint"}],"openQuestions":["How to detect benchmark leakage into training?","How to separate capability measurement from permissibility of use?","How to account for uncertainty and a context change?","Установить границу и решение reuse/extend/new по действующим спецификациям.","Подтвердить semantic crosswalk, права и source mastership.","Выбрать immutable refs; провести проверки fixtures до заявления о публикационной готовности."],"resources":{"source":"https://ver.cy/enterprise/models/em-ai-03/"},"provenance":{"origin":"enterprise research programme","builtFrom":["enterprise/models/em-ai-03/brief.json"],"providers":[],"researchStatus":"published-partial","generatedAt":"","builder":"tools/build_cards.py@1.0.0"},"completeness":{"sections":{"classifiers":"filled","whatItIs":"filled","purpose":"filled","distinguishingFeatures":"derived","structure":"missing","agentConduct":"derived","ethics":"missing","owners":"filled","relations":"filled","interaction.identity":"derived","interaction.properties":"not-applicable","interaction.recognition":"missing","interaction.capabilities":"missing","interaction.hazards":"missing","interaction.interfaces":"missing","interaction.context":"missing","sources":"filled"},"notes":{"agentConduct":"Negative case of the research brief, not yet a rule for agents.","interaction.properties":"Enterprise record contour: physical properties belong to referenced world models.","structure":"Research contour: bundles not designed yet; questions are listed as open questions.","language":"Still in Russian: suggested owner, blocking decisions, vercy candidates. Translate in research/enterprise/i18n/units.en.json."},"score":0.469}}