{
  "acceptanceEvidence": [
    "reproducible evaluations",
    "frozen artifacts",
    "independent challenge",
    "bounded failure behavior"
  ],
  "acceptanceMatrix": [
    {
      "acceptanceState": "PASS, FAIL, PARTIAL, UNAVAILABLE, or NOT APPLICABLE with rationale",
      "assessmentMethod": "hash and lineage replay",
      "criterion": "Provenance",
      "requiredResult": "Model, data, code and evaluation artifacts are versioned and linked"
    },
    {
      "acceptanceState": "PASS, FAIL, PARTIAL, UNAVAILABLE, or NOT APPLICABLE with rationale",
      "assessmentMethod": "policy enforcement test",
      "criterion": "Bounded authority",
      "requiredResult": "Tool and actuator permissions are explicit and testable"
    },
    {
      "acceptanceState": "PASS, FAIL, PARTIAL, UNAVAILABLE, or NOT APPLICABLE with rationale",
      "assessmentMethod": "fault-injection exercise",
      "criterion": "Failure behavior",
      "requiredResult": "Fallback and rollback operate under declared uncertainty"
    }
  ],
  "bidEvaluationCriteria": [
    "model-system assurance depth",
    "adversarial evaluation method",
    "secure evidence handling",
    "vendor-independent test design",
    "price realism and schedule credibility",
    "data-rights and evidence-delivery terms",
    "subcontractor and supply-chain transparency"
  ],
  "canonicalUrl": "https://xn--mwe.com/procurement/work-packages/model-assurance/",
  "dataRequirements": [
    "model inventory and hashes",
    "training/evaluation data lineage",
    "tool interfaces",
    "system prompts and policies under controlled access",
    "runtime telemetry and incident history"
  ],
  "deliverables": [
    "model assurance case",
    "adversarial evaluation summary",
    "tool-authority matrix",
    "drift and rollback plan",
    "unresolved model risks"
  ],
  "dependencies": [
    "WP-K06-04",
    "WP-K06-05"
  ],
  "evidenceRights": [
    "The acquiring authority receives perpetual access to final reports, schemas, manifests, acceptance evidence, defects, and correction history within the negotiated data-rights regime.",
    "Contractor proprietary methods may remain protected only when they do not prevent independent replay of required results.",
    "Source artifacts, hashes, versions, tool outputs, and negative findings required for acceptance cannot be withheld merely because they are unfavorable.",
    "No clause transfers authority, licensing status, or ownership beyond the signed contract and governing law."
  ],
  "exclusions": [
    "no claim that benchmark success proves field judgment",
    "no unrestricted tool use"
  ],
  "id": "WP-K06-06",
  "inputs": [
    "model inventory",
    "training and evaluation records",
    "tool policy",
    "agent architecture",
    "safety constraints"
  ],
  "k07Id": "WP-K07-06",
  "name": "Model and autonomous-agent assurance",
  "negativeResultClauses": [
    "A failed, partial, stale, disputed, unavailable, or superseded result must be delivered and may not be converted into a pass.",
    "Discovery of a safety, authority, evidence, or common-cause defect triggers prompt notice and preserves stop authority.",
    "Acceptance of one deliverable does not waive latent defects, falsified evidence, or later-discovered nonconformance.",
    "The final package must distinguish work completed, work not performed, evidence unavailable, and owner decisions pending."
  ],
  "objective": "Evaluate model provenance, data lineage, adversarial robustness, tool authority, memory boundaries, drift, fallback, and runtime constraints.",
  "releaseId": "K12-2026-08-16",
  "slug": "model-assurance",
  "statementOfWork": {
    "contractorDeliverables": [
      "model assurance case",
      "adversarial evaluation summary",
      "tool-authority matrix",
      "drift and rollback plan",
      "unresolved model risks"
    ],
    "dependencies": [
      "WP-K06-04",
      "WP-K06-05"
    ],
    "exclusions": [
      "no claim that benchmark success proves field judgment",
      "no unrestricted tool use"
    ],
    "governmentOrOwnerFurnishedInformation": [
      "model inventory and hashes",
      "training/evaluation data lineage",
      "tool interfaces",
      "system prompts and policies under controlled access",
      "runtime telemetry and incident history"
    ],
    "performanceOutcomes": [
      "model and data provenance baseline",
      "tool-authority and memory-boundary matrix",
      "adversarial robustness evidence",
      "drift, rollback and fallback plan"
    ],
    "performanceStandards": [
      "Model, data, code and evaluation artifacts are versioned and linked",
      "Tool and actuator permissions are explicit and testable",
      "Fallback and rollback operate under declared uncertainty"
    ],
    "periodOfPerformance": "To be defined by the acquiring authority; no duration is inferred by the public pattern.",
    "purpose": "Evaluate model provenance, data lineage, adversarial robustness, tool authority, memory boundaries, drift, fallback, and runtime constraints.",
    "qualitySurveillance": [
      "hash and lineage replay",
      "policy enforcement test",
      "fault-injection exercise"
    ]
  }
}
