{
  "schemaVersion": "eve.engineering-outcome-cohort.v1",
  "documentType": "protocol",
  "contractVersion": "1.0.0",
  "taskId": "11.6",
  "frozenAt": "2026-09-13T02:00:00.000Z",
  "status": "preparation-frozen",
  "admission": {
    "status": "blocked-by-dependencies",
    "dependencyTasks": ["11.4", "11.5", "11.12", "12.2"],
    "taskCheckbox": "open",
    "closureManifestPresent": false,
    "promotionClaimed": false
  },
  "sourceBindings": [
    {
      "path": "EVE_SOTA_GAP_CLOSURE_TODOS_2026-09-01.md",
      "selector": "task-clause:11.6",
      "sha256": "c6fa86eae432080afc08652e40387d51868f067cd877c2e5b5d01079ada57cbd"
    },
    {
      "path": "tools/eve-everywhere/gap-task-dependencies.mjs",
      "sha256": "6c07dba6169aeceea7b93c887b95eb4c0232e7a102c508b63ea9eb687d90e14a"
    },
    {
      "path": "docs/audits/EVE_SOTA_OUTCOME_SCORECARD_2026-09.md",
      "sha256": "490bcffb93d2b34a1af46dd846a16ee052460c91ef9199e75a51f1ef3c3d4030"
    },
    {
      "path": "docs/audits/eve-sota-outcome-scorecard/2026-09-05.v2.json",
      "sha256": "f8bd820e7bdceefc151ef619a33669ed87fd0e22620a6caec783886c7bc27a5d"
    },
    {
      "path": "docs/audits/EVE_ENGINEERING_BENCHMARK_2026-09.md",
      "sha256": "7b39d764a87d44c82edf08d313d78d191d9f89841d8bcc69669b84dceb4b18ed"
    },
    {
      "path": "docs/audits/eve-engineering-benchmark.schema.json",
      "sha256": "f07d0cd5cfc6c4a0d8f3d86a234278ba298bd1deca2fcc6a329d52c45acee2e7"
    },
    {
      "path": "docs/audits/eve-engineering-benchmark/v1.0.0/release.json",
      "sha256": "597cf31209754769d2ad1212bf1790c84308454e2c392b17b72a3f3871061367"
    },
    {
      "path": "docs/audits/eve-engineering-benchmark/v1.0.0/public/catalog.json",
      "sha256": "5b56b695634ce3c4d7dcbff9bbf88215c78d2aa4f0919c02010d5ff126d13047"
    },
    {
      "path": "docs/audits/eve-engineering-benchmark/v1.0.0/evaluator/acceptance.json",
      "sha256": "49db506a778658871a135b06ca3f774089756e4155d9ba8a830b1944049606cd"
    },
    {
      "path": "docs/audits/eve-engineering-review/v1.0.0/preparation-evidence.json",
      "sha256": "9bdca35b33eb110707f9b688e5829e13612d7a1802e0fbf2b2474b8f3bc336fd"
    },
    {
      "path": "docs/audits/eve-delivery-fault-game/v1.0.0/preparation-evidence.json",
      "sha256": "95ad2bec34877ce584ad72d1757db08340dae3451fbded4538e4e45d0c645cb0"
    },
    {
      "path": "tools/eve-task-ledger-binding.mjs",
      "sha256": "edd5237f9b112807a81c38c5ba75f80bcb1370ac25bf0888156633db332ee4d0"
    },
    {
      "path": "tools/eve-task-ledger-binding.test.mjs",
      "sha256": "b8e30052c9da6d501fcfb18bfa1a3be1b13e44b7ea4be1c993b6a977e23f0b28"
    },
    {
      "path": "tools/eve-engineering-outcome-cohort-lib.mjs",
      "sha256": "0631a5b8390b22853500f0676b550ea822f51517392f85b8b607a32eb0b7b51b"
    },
    {
      "path": "tools/eve-engineering-outcome-cohort.test.mjs",
      "sha256": "2f2654b105f1fc0353d7884b0e791be911811ea7678560575c03422e0ed99b2a"
    },
    {
      "path": "tools/eve-everywhere/seal-engineering-outcome-cohort.mjs",
      "sha256": "184445e8fb9b189ea42d3aa0d440c4e43fee0952b08918fac9db0e1db75da6a3"
    },
    {
      "path": "tools/eve-everywhere/verify-engineering-outcome-cohort.mjs",
      "sha256": "b66a8771542a2a7783f233ff8715195e9d65cdfab775203007fc827e706223f4"
    },
    {
      "path": "tools/eve-everywhere/verify-engineering-outcome-cohort.test.mjs",
      "sha256": "4433388490bbea64e9999bbcb34be4cc048b242dbd557ff5b752a239318ae48d"
    },
    {
      "path": "tools/eve-everywhere/run-engineering-outcome-cohort-preparation-evidence.mjs",
      "sha256": "66a3966a24ccc5938121833819ce22e31d89af02d2ab67fcb127ad22b7c87973"
    },
    {
      "path": "tools/eve-everywhere/verify-engineering-outcome-cohort-preparation-evidence.mjs",
      "sha256": "68c7305e6edbc7abb1982559f3abd6c62aff515b2dfdcd93a0a132755047c52e"
    },
    {
      "path": "tools/eve-everywhere/verify-engineering-outcome-cohort-preparation-evidence.test.mjs",
      "sha256": "ecd11d4949b1e6f18181f9b2fa45294bc078728fa9a3332f13edba8f1c4c4d61"
    },
    {
      "path": "docs/audits/eve-engineering-outcome-cohort.schema.json",
      "sha256": "37dc792197fc9fe472b7e18ca4f0d43f7bc70f88f689fd28a96a27dbc219f967"
    },
    {
      "path": "docs/audits/eve-engineering-outcome-cohort/v1.0.0/readiness-2026-09-13.json",
      "sha256": "8ab7cb6805217da119e1a0f6dbfdec0269ac47143f80e02c125d7bbfd2d03389"
    },
    {
      "path": "docs/audits/EVE_ENGINEERING_OUTCOME_COHORT_PREPARATION_2026-09.md",
      "sha256": "3618434340a7df2317664d9b2a8e392ef7771832a628410afb26dea26d037738"
    }
  ],
  "benchmark": {
    "version": "1.0.0",
    "releasePath": "docs/audits/eve-engineering-benchmark/v1.0.0/release.json",
    "catalogPath": "docs/audits/eve-engineering-benchmark/v1.0.0/public/catalog.json",
    "evaluatorPath": "docs/audits/eve-engineering-benchmark/v1.0.0/evaluator/acceptance.json",
    "caseCount": 11,
    "split": "held-out",
    "caseReuseAcrossLanes": "paired-same-case-only",
    "caseReuseAcrossIndependentUnits": "forbidden",
    "knownOrAnswerAwareDisposition": "retire-and-replace"
  },
  "scorecard": {
    "version": 2,
    "status": "preregistered-unmeasured",
    "path": "docs/audits/eve-sota-outcome-scorecard/2026-09-05.v2.json",
    "requiredOutcomeIds": [
      "verified-task-success",
      "human-intervention-rate",
      "rollback-rate",
      "false-success-rate",
      "time-per-verified-result",
      "cost-per-verified-result",
      "operator-labor-per-verified-result"
    ]
  },
  "cohortDesign": {
    "lanes": ["eve", "manual"],
    "pairingKey": "benchmarkCaseId",
    "baseSnapshotPolicy": "identical-within-pair",
    "acceptancePolicy": "same-hidden-qa-oracle-within-pair",
    "laneIsolation": "no-artifact-or-outcome-cross-exposure",
    "laneOrder": "preregistered-balanced-randomization",
    "manualBaselineLocked": "before-eve-outcome-inspection",
    "independentUnit": "one previously unseen independently authored goal trajectory",
    "seedPolicy": "fixed nested repeats never increase independent n",
    "failedAttemptsIncluded": true,
    "successfulRetryUnitInflationAllowed": false,
    "sampleDesignAuthority": "task-12.2",
    "operatorLaborAuthority": "task-11.12",
    "independentVerificationAuthority": "task-11.4",
    "faultRecoveryAuthority": "task-11.5",
    "executionIsolationAuthority": "task-11.3",
    "candidateContext": "single-case archive without git history network or evaluator material",
    "resultRetention": "sanitized immutable run receipts plus 14-day observation state"
  },
  "metrics": [
    {
      "id": "verified-success",
      "status": "preregistered-unmeasured",
      "definition": "A trajectory succeeds only when every hidden acceptance criterion passes and an actor distinct from the implementer verifies the exact artifact revision.",
      "primaryUnit": "one independent goal trajectory counted once across all attempts",
      "target": {
        "operator": ">=",
        "value": 0.9,
        "statistic": "point-rate"
      },
      "floor": {
        "operator": ">=",
        "value": 0.8,
        "statistic": "wilson-95-lower"
      },
      "requiredEvidence": [
        "exact artifact revision",
        "hidden criterion verdicts",
        "independent verifier identity",
        "failed and retried attempts"
      ],
      "manualComparison": "Report paired success difference and each lane separately; a failed lane stays in its denominator."
    },
    {
      "id": "human-interventions",
      "status": "preregistered-unmeasured",
      "definition": "An unplanned intervention is any human action needed to redirect, repair, unblock, or correct Eve beyond preregistered mandatory governance decisions.",
      "primaryUnit": "one eligible goal trajectory counted once even after several interventions",
      "target": {
        "operator": "<=",
        "value": 0.1,
        "statistic": "point-rate"
      },
      "floor": {
        "operator": "<=",
        "value": 0.2,
        "statistic": "wilson-95-upper"
      },
      "requiredEvidence": [
        "attributed intervention reason",
        "hands-on interval receipt",
        "mandatory-versus-unplanned classification",
        "all retries and failed work"
      ],
      "manualComparison": "Report intervention events and hands-on minutes by lane; Task 11.12 owns complete labor accounting."
    },
    {
      "id": "escaped-defects",
      "status": "preregistered-unmeasured",
      "definition": "Any material defect discovered after an independently verified terminal outcome is both an escaped defect and a false-success event.",
      "primaryUnit": "one independently verified result after its full persistence window",
      "target": {
        "operator": "=",
        "value": 0,
        "statistic": "event-count",
        "hardLock": true
      },
      "floor": {
        "operator": "=",
        "value": 0,
        "statistic": "event-count",
        "hardLock": true
      },
      "requiredEvidence": [
        "14-day state observations",
        "defect and incident reconciliation",
        "operator correction and undo records",
        "false-success cross-check"
      ],
      "manualComparison": "Report events independently by lane; no aggregate advantage can offset one escaped false-success event.",
      "observationWindowDays": 14
    },
    {
      "id": "revert-rate",
      "status": "preregistered-unmeasured",
      "definition": "A result counts as reverted when any substantive part of the verified delivery is rolled back during the complete observation window.",
      "primaryUnit": "one independently verified persistent result after 14 complete days",
      "target": {
        "operator": "<=",
        "value": 0.01,
        "statistic": "point-rate"
      },
      "floor": {
        "operator": "<=",
        "value": 0.05,
        "statistic": "wilson-95-upper"
      },
      "requiredEvidence": [
        "delivery revision",
        "revert ancestry or equivalent state receipt",
        "complete observation timestamps",
        "revert reason"
      ],
      "manualComparison": "Report paired and lane-specific rates; pending or shortened windows are incomplete, never successes.",
      "observationWindowDays": 14,
      "rightCensoredDisposition": "incomplete-not-success"
    },
    {
      "id": "changed-lines-quality",
      "status": "preregistered-unmeasured",
      "definition": "A QA-owned hidden rubric scores necessity, correctness, maintainability, security/privacy, compatibility, and test strength over substantive changed lines, clustered by trajectory.",
      "primaryUnit": "one independent trajectory-level changed-line review, not each line",
      "target": {
        "operator": ">=",
        "value": 0.9,
        "statistic": "weighted-rubric-score"
      },
      "floor": {
        "operator": ">=",
        "value": 0.8,
        "statistic": "cluster-bootstrap-95-lower"
      },
      "requiredEvidence": [
        "exact diff and tree digest",
        "QA-owned rubric version",
        "criterion-level independent scores",
        "substantive correction record"
      ],
      "manualComparison": "Use the same blinded rubric and evaluator separation for both lanes and publish paired score deltas.",
      "hardLocks": {
        "criticalDefects": 0,
        "securityPrivacyRegressions": 0
      }
    },
    {
      "id": "elapsed-time",
      "status": "preregistered-unmeasured",
      "definition": "Wall time includes every attempt from authorized start to verified terminal result; decision-wait time is retained and reported separately, not deleted.",
      "primaryUnit": "one paired hidden benchmark goal with both lane clocks",
      "target": {
        "operator": "<=",
        "value": 0.75,
        "statistic": "stratified-paired-ratio"
      },
      "floor": {
        "operator": "<=",
        "value": 1,
        "statistic": "stratified-bootstrap-95-upper"
      },
      "requiredEvidence": [
        "monotonic start and stop receipts",
        "all attempt intervals",
        "decision-wait intervals",
        "verification completion timestamp"
      ],
      "manualComparison": "The numerator is Eve-lane time per verified result and the denominator is its preregistered paired manual lane.",
      "comparison": "paired-manual-lane"
    },
    {
      "id": "tokens-and-cost",
      "status": "preregistered-unmeasured",
      "definition": "Retain input, output, cached-input, and reasoning tokens plus every resolved model, endpoint, tool, compute, retry, cache, and billed USD leg, including failed attempts.",
      "primaryUnit": "one goal trajectory with all attempts and priced execution legs",
      "target": {
        "operator": "<=",
        "value": 0.8,
        "statistic": "budget-ratio"
      },
      "floor": {
        "operator": "<=",
        "value": 1,
        "statistic": "stratified-bootstrap-95-upper"
      },
      "requiredEvidence": [
        "resolved model and endpoint",
        "token and cache counters",
        "tool and compute duration",
        "versioned pricing and budget",
        "failed-attempt charges"
      ],
      "manualComparison": "Publish Eve variable cost and tokens beside manual labor cost without pretending the manual lane has model tokens; compare total economic cost separately when Task 11.12 receipts permit.",
      "hardLocks": {
        "unpricedLegs": 0
      },
      "tokenFields": ["input", "output", "cached-input", "reasoning"]
    },
    {
      "id": "test-selection",
      "status": "preregistered-unmeasured",
      "definition": "Selection recall is the share of QA-preregistered required checks actually run against the final revision; optional check precision and cost are reported but cannot excuse a required miss.",
      "primaryUnit": "one goal's QA-owned hidden verification plan",
      "target": {
        "operator": "=",
        "value": 1,
        "statistic": "required-check-recall"
      },
      "floor": {
        "operator": "=",
        "value": 1,
        "statistic": "required-check-recall",
        "hardLock": true
      },
      "requiredEvidence": [
        "hidden required-check IDs",
        "selected command receipts",
        "exact final revision under test",
        "omission and extra-check classification"
      ],
      "manualComparison": "Apply the same hidden required-check set to both lanes; publish recall, extra checks, elapsed verification time, and missed-defect linkage.",
      "requiredTestMissesAllowed": 0,
      "oracleSource": "qa-owned-hidden-verification-plan"
    },
    {
      "id": "explanation-fidelity",
      "status": "preregistered-unmeasured",
      "definition": "Independent graders reconcile every material terminal claim about changes, checks, limitations, delivery, and rollback against retained artifacts.",
      "primaryUnit": "one material claim clustered within one terminal explanation",
      "target": {
        "operator": ">=",
        "value": 0.98,
        "statistic": "supported-claim-precision"
      },
      "floor": {
        "operator": ">=",
        "value": 0.95,
        "statistic": "cluster-bootstrap-95-lower"
      },
      "requiredEvidence": [
        "claim-to-artifact joins",
        "command and delivery receipts",
        "declared limitations",
        "independent claim labels"
      ],
      "manualComparison": "Blindly grade both lane explanations with the same claim taxonomy and publish corrections and omissions.",
      "hardLocks": {
        "fabricatedMaterialClaims": 0,
        "falseSuccessOutcomes": 0
      }
    }
  ],
  "promotion": {
    "decisionWhenBlocked": "HOLD",
    "decisionOnMissingMetric": "HOLD",
    "decisionOnFloorBreach": "HOLD",
    "allMetricsRequired": true,
    "partialPoolingAllowed": false,
    "postResultExpansionAllowed": false,
    "realRunEvidenceRequired": true,
    "targetsRequiredForTaskCompletion": true,
    "freshCohortAfterRepair": true
  },
  "limitations": [
    "This protocol preregisters a paired measurement boundary; it contains no agent or manual benchmark outcomes.",
    "The eleven frozen benchmark cases cannot be duplicated to satisfy the scorecard's minimum independent sample size.",
    "Tasks 11.4, 11.5, 11.12, and 12.2 must supply independent review, admitted recovery, complete operator labor, and a statistically feasible sample design before any cohort is admitted.",
    "Task 11.6 remains open and no closure manifest, gap-matrix evidence mapping, promotion, or verified claim is present."
  ]
}
