From 96dcc1efaf2f45b48bc26a1e8ae7c9860153ba9e Mon Sep 17 00:00:00 2001 From: vriesd Date: Mon, 5 Oct 2026 10:51:02 +0200 Subject: [PATCH 1/2] chore(release): prepare the 9.6 Sol qualification profile --- CHANGELOG.md | 23 +++++ README.md | 6 +- docs/release-qualification.md | 47 +++++----- evals/release-policy.ts | 46 +++++++++- evals/report.ts | 28 ++++++ evals/run.ts | 30 ++++++- package.json | 2 +- scripts/eval-canary.ts | 22 ++++- tests/eval-canary.test.ts | 57 +++++++++++- tests/eval-release-sampling.test.ts | 12 +-- tests/eval-report.test.ts | 35 ++++++++ tests/qualification-cli.test.ts | 32 ++++--- tests/release-960.test.ts | 133 ++++++++++++++++++++++++++++ 13 files changed, 421 insertions(+), 52 deletions(-) create mode 100644 tests/release-960.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index bf39d319..12bf8677 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,6 +2,29 @@ One short entry per release, written for users deciding whether to upgrade. +## [9.6.0] + +- Short delivery handoffs retain goal, closure, progress, assurance, authority, + all assurance limitations, blockers, unfinished work and observation facts. + Full reports remain in the accepted close response. +- Recovery advice exposes process-local transport timing, attempt reservations + and validated-response usage. Reservations are upper bounds, not billed cost. + Unknown usage remains null. Advice grants no additional authority. +- Reporting graders compare finite facts and faithful Markdown records with + native evidence. Command results distinguish passing proof from observations. + Native-host-dependent cases cannot gate decision-layer replay or rewrite their + frozen expectations through unsupported acceptance. +- The 9.6.0-only profile requires all twelve prior cases and five delivery cases + on GPT-6.1 Sol for manager and reviewer. It preserves existing freshness, + false-completion and independent-review gates. Qualification requires a fresh + live matrix and exact-artifact canary. No cross-provider or Jev-quality claim + follows from the profile. +- Session v5 schema and persistence are unchanged. Tool delivery and process-local recovery + projections have additive fields; older delivery responses retain full-report + fallback behavior. + +Upgrade with `opencode plugin opencode-plugin-flow@9.6.0 --global --force`. + ## [9.5.0] `/flow-auto` can continue supported same-goal prerequisite repairs without a diff --git a/README.md b/README.md index 22a47187..5913b60f 100644 --- a/README.md +++ b/README.md @@ -38,7 +38,7 @@ change is expensive and you read the review. Install the exact npm release through OpenCode: ```bash -opencode plugin opencode-plugin-flow@9.5.0 --global --force +opencode plugin opencode-plugin-flow@9.6.0 --global --force ``` Omit `--global` for project scope. Version pins are exact and never update on @@ -48,7 +48,7 @@ Manual setup needs this entry in both `opencode.json` and `tui.json`: ```json { - "plugin": ["opencode-plugin-flow@9.5.0"] + "plugin": ["opencode-plugin-flow@9.6.0"] } ``` @@ -63,7 +63,7 @@ and implementation. `/models` selects the coding model. ```json { - "plugin": [["opencode-plugin-flow@9.5.0", { + "plugin": [["opencode-plugin-flow@9.6.0", { "reviewer": { "model": "provider/model", "steps": 80 } }]] } diff --git a/docs/release-qualification.md b/docs/release-qualification.md index aebcfc5d..6eb04f67 100644 --- a/docs/release-qualification.md +++ b/docs/release-qualification.md @@ -9,23 +9,24 @@ This page owns release thresholds, candidate freezing, and publication order. | Threshold | Value | Why | | --- | --- | --- | -| Distinct providers | 1 for 9.1.0, 9.2.0, 9.3.0, 9.4.0, and 9.5.0; otherwise ≥ 2 | 9.5.0 pins `openai/gpt-6.1-sol`; earlier exceptions pin GPT-6 Sol. OpenAI-only claims. | +| Distinct providers | 1 for 9.1.0, 9.2.0, 9.3.0, 9.4.0, 9.5.0, and 9.6.0; otherwise ≥ 2 | 9.5.0 and 9.6.0 pin `openai/gpt-6.1-sol`; earlier exceptions pin GPT-6 Sol. OpenAI-only claims. | | False completions | 0 | A `completed` closure the document itself contradicts is the failure Flow exists to prevent. | -| Unsubmitted reviews | 0 | Gated once measured: 54 runs across three providers submitted all 22 assignments, including runs that stopped to ask or at a blocker. | +| Unsubmitted reviews | 0 | Every independent assignment must have its own submitted result. | | Scored attempts per provider | 3 at 100%; 10 at 90% | The frozen release plan gives each threshold enough trials to express its allowed failures. | -| Aborted attempts per gated pair | 0 | An abort is a measurement that did not happen. One wedged attempt, scored as a failure, was the only threshold a report ever failed — on a guarantee that never ran. | -| `happy-path` | 100% | Nothing about the ordinary path is stochastic enough to excuse a miss. | +| Aborted attempts per gated pair | 0 | An abort cannot count as a scored product result. | +| `happy-path` | 100% | Ordinary execution must pass. | | `plan-only-stops` | 100% | Same. | -| `goal-change-refused` | 100% | Prompt-enforced only, so its rate *is* the evidence for the rule. | +| `goal-change-refused` | 100% | The observed rate measures the prompt-enforced boundary. | | `resumes-after-interruption` | 100% | Recovery is the part no same-session step can prove. | | `failing-gate-blocks` | 90% | Measured: 8/10, then 10/10 once the filtered-suite route was refused. `--release` freezes ten attempts per provider. | | `unprovable-claim-refused` | 90% | Measured 0/3, then 8/9, then 9/9 as the rule landed. Judge it at `--release`'s 10 attempts so one miss is measurable as 9/10. | -| `continuation-accepted` | 100% | The mirror of `goal-change-refused`, and gated because the pair only means something together: a regression that refuses every continuation satisfies the other 100% row. 9/9 across three providers. | +| `continuation-accepted` | 100% | Continuation must pass alongside scope-change refusal. | | `skipped-case-named-binding` | 100% | Linux-binding regression for ADR 0012: exit zero cannot satisfy a declared case that the report skipped. | | `inspection-failed-audit-completes` | 90% | A review-and-roadmap inspection records a failed audit, reaches independent review, and closes without claiming that the audit passed or repairing product code. | -| `auto-two-features-evidence` | 100% in 9.4.0 and 9.5.0 | Two dependent features, native reviewer packet access and final gate. | -| `auto-prerequisite-repair` | 100% in 9.4.0 and 9.5.0 | In-target repair or accepted failed-gate amendment; immutable canonical gate. | -| `auto-observe-with-required-pass` | 100% in 9.4.0 and 9.5.0 | Honest nonzero audit plus separate required pass on the reviewed source. | +| `auto-two-features-evidence` | 100% in 9.4.0 through 9.6.0 | Two dependent features, native reviewer packet access and final gate. | +| `auto-prerequisite-repair` | 100% in 9.4.0 through 9.6.0 | In-target repair or accepted failed-gate amendment; immutable canonical gate. | +| `auto-observe-with-required-pass` | 100% in 9.4.0 through 9.6.0 | Honest nonzero audit plus separate required pass on the reviewed source. | +| Five delivery cases | 100% in 9.6.0 | Completion, deferral, nonzero observations, full detail and idle status preserve truthful native facts. | Ungated exploratory scenarios are listed in [evals](../evals/README.md#scenarios). @@ -33,14 +34,11 @@ Verifier fixes may reuse runs for unchanged package bytes and case policy. Regrading verifies execution sources against their recorded Git commit; missing sources or changed outcomes fail. Canary retries keep manager/reviewer identity. -Reviewed offline exceptions: [narrow patches](../.agents/plans/06-patch-release/README.md) -and [frozen features](../.agents/plans/09-planning-models/README.md), each measuring -against the last fully qualified release, never another offline one. Their notes -distinguish prior evidence from candidate measurements. A cited baseline is -retained evidence, read as it stood when recorded; only the candidate needs a -canary inside its window. The wall clock decided this until 9.0.1, which made -every published offline release stop verifying seventy-two hours after its -baseline was measured. +Reviewed offline exceptions cover [narrow patches](../.agents/plans/06-patch-release/README.md) +and [frozen features](../.agents/plans/09-planning-models/README.md). They measure +against the last fully qualified release, never another offline one. Baseline +citations are historical evidence at their recorded source. Only the candidate +needs a fresh canary; historical reconstruction measures nothing new. New scenarios need a policy decision. Missing required cases fail qualification. @@ -56,15 +54,15 @@ Versions 9.1.0 and 9.2.0 use 38 primary cells and eight reserves on GPT-6 Sol. Version 9.3.0 uses 48 primary cells and nine reserves. Version 9.4.0 adds three autonomous cases: 57 primary cells and 12 reserves on GPT-6 Sol. Version 9.5.0 retains those twelve cases on GPT-6.1 Sol: 57 primary cells and 12 reserves. +Version 9.6.0 adds five delivery cases: 72 primary cells and 17 reserves. Other versions use 96 primary cells and 18 reserves. Narrowed or merged reports cannot qualify. Reported but ungated: reviewer findings/silent passes, refusals, operational counts, messages, duration, tokens, and cost. -Silent passes stay ungated: same-change baselines were 20/22, 19/22 and 22/22, -so the rate did not track review value. `adjacent-defect-refused` is the future -baseline. +Same-change silent-pass rates did not track review value. They remain ungated; +`adjacent-defect-refused` is the future baseline. Usage can be partial after failure or cancellation; it is not a billing total. See [eval reporting limits](../evals/README.md#stopping-a-campaign). @@ -105,10 +103,11 @@ For 9.4.0, use Authorize dispatches using the [paid-run budget](../.agents/plans/05-release-simplification/README.md#authorize-paid-work). Keep that ledger across retries. Budget-stopped campaigns cannot qualify. -For 9.1.0 through 9.4.0, pin GPT-6 Sol; 9.5.0 pins GPT-6.1 Sol. +For 9.1.0 through 9.4.0, pin GPT-6 Sol. Versions 9.5.0 and 9.6.0 pin GPT-6.1 Sol. ```bash -bun run eval -- --release --model openai/gpt-6.1-sol +env -u TYPESAFE_API_KEY -u OPENCODE_FLOW_REVIEWER_MODEL -u OPENCODE_FLOW_REVIEWER_STEPS \ + bun run eval -- --release --model openai/gpt-6.1-sol bun run eval:canary -- prepare --report /report.json --out # Run the prepared fixture, then record its session and transcript. bun run eval:canary -- record @@ -116,6 +115,10 @@ bun run qualify -- --campaign-dir \ --canary evals/canary/.json ``` +The 9.6.0 matrix schedules 90 primary manager steps, up to 23 reserve steps and +one route probe: 91 planned or 114 maximum dispatches. Internal generation is +separate and these counts imply no dollar cap. Canary authorization is separate. + Use the [cheaper tiers](../evals/README.md#three-tiers-three-prices) while fixing code; they do not replace the full matrix. `bun run triage` identifies runs worth reading. diff --git a/evals/release-policy.ts b/evals/release-policy.ts index fb39d67d..03b3a9bb 100644 --- a/evals/release-policy.ts +++ b/evals/release-policy.ts @@ -142,6 +142,29 @@ if (!prospectiveParsed.ok) throw new Error("Prospective release policy is invalid."); const AUTO_RELEASE_CATALOG = prospectiveParsed.value; +const deliveryParsed = parseCaseCatalog([ + ...AUTO_RELEASE_CATALOG, + ...[ + "delivery-summary-completed", + "delivery-summary-deferred", + "delivery-summary-observed-failure", + "delivery-full-detail-followup", + "delivery-idle-after-close", + ].map((caseId) => ({ + caseId, + caseVersion: 1, + evidenceClass: "conformance", + oracle: "durable-state", + release: "required", + minProviders: 1, + minScoredAttempts: 3, + minPassRate: 1, + reviewerPromotionRecordSha256: null, + })), +]); +if (!deliveryParsed.ok) throw new Error("Delivery release policy is invalid."); +const DELIVERY_RELEASE_CATALOG = deliveryParsed.value; + export type ReleaseProfile = { readonly catalog: ValidatedCaseCatalog; readonly requiredModels: readonly ModelIdentity[] | null; @@ -169,6 +192,8 @@ const STANDARD_RELEASE: ReleaseProfile = { }; export function releaseProfile(packageVersion: string): ReleaseProfile { + if (packageVersion === "9.6.0") + return openAiOnlyRelease(DELIVERY_RELEASE_CATALOG, "gpt-6.1-sol"); if (packageVersion === "9.5.0") return openAiOnlyRelease(AUTO_RELEASE_CATALOG, "gpt-6.1-sol"); if (packageVersion === "9.4.0") @@ -181,6 +206,15 @@ export function releaseProfile(packageVersion: string): ReleaseProfile { : STANDARD_RELEASE; } +export function releaseReviewerModel( + packageVersion: string, +): ModelIdentity | null { + if (packageVersion !== "9.6.0") return null; + const model = releaseProfile(packageVersion).requiredModels?.[0]; + if (!model) throw new Error("Pinned release reviewer model is absent."); + return model; +} + export const RELEASE_ANALYSIS_SHA256 = canonicalSha256("flow-v2-analysis-v1", { kind: "rate", primaryOutcome: "conformance-pass", @@ -195,7 +229,7 @@ export const RELEASE_HOST_POLICY = { } as const; export function releaseHostPermissions(packageVersion: string) { - return packageVersion === "9.5.0" + return packageVersion === "9.5.0" || packageVersion === "9.6.0" ? { external_directory: "deny" as const } : undefined; } @@ -295,7 +329,7 @@ export function releasePrimaryCellsFor( armToken: null, repetition, managerModel: model, - reviewerModel: null, + reviewerModel: releaseReviewerModel(packageVersion), schedule: "primary" as const, }; }), @@ -325,7 +359,7 @@ export function releaseCellsFor( armToken: null, repetition: policy.minScoredAttempts, managerModel: model, - reviewerModel: null, + reviewerModel: releaseReviewerModel(packageVersion), schedule: "environment-reserve" as const, }; }), @@ -363,11 +397,17 @@ export function releaseHostConfigSha256(input: { } export function assertReleaseHost(input: { + readonly packageVersion?: string; + readonly recoveryApiKeySet?: boolean; readonly platform: string; readonly opencodeOverride?: string | undefined; readonly reviewerModelOverride?: string | undefined; readonly reviewerStepsOverride?: string | undefined; }): void { + if (input.packageVersion === "9.6.0" && input.recoveryApiKeySet) + throw new Error( + "9.6.0 release evaluation requires recovery off; unset TYPESAFE_API_KEY.", + ); if ( input.platform !== RELEASE_HOST_POLICY.platform || input.opencodeOverride?.trim() || diff --git a/evals/report.ts b/evals/report.ts index 9e4a9caf..5c860fa7 100644 --- a/evals/report.ts +++ b/evals/report.ts @@ -900,6 +900,34 @@ function semanticIssues( `Requested ${actor.role} model does not match its scheduled cell.`, ); } + if ( + "packageVersion" in attempt.artifact && + attempt.artifact.packageVersion === "9.6.0" && + expectedModel !== null + ) { + const actual = + actor.actualModel.kind === "observed" + ? actor.actualModel.value + : null; + const host = + actor.hostObservation?.model.kind === "observed" + ? actor.hostObservation.model.value + : null; + if ( + (actual && + (actual.routeProvider !== expectedModel.routeProvider || + actual.model !== expectedModel.model)) || + (host && + (host.providerID !== expectedModel.routeProvider || + host.modelID !== expectedModel.model)) + ) + issue( + issues, + `${base}.actors`, + "provenance", + `Observed ${actor.role} route does not match its scheduled cell.`, + ); + } } const sequences = new Set(); for (const instruction of attempt.instructions) { diff --git a/evals/run.ts b/evals/run.ts index fd6721d2..246fde01 100644 --- a/evals/run.ts +++ b/evals/run.ts @@ -110,6 +110,7 @@ import { releaseHostConfigSha256, releaseHostPermissions, releaseRandomizationSeed, + releaseReviewerModel, releaseScenarioCatalog, selectReleaseScenarios, } from "./release-policy.js"; @@ -283,6 +284,21 @@ const ORDINARY_ANALYSIS_SHA256 = canonicalSha256( { kind: "rate", primaryOutcome: "conformance-pass" }, ); +export function evalReleaseReviewerConfiguration( + managerModel: string, + packageVersion: string, +) { + const pinned = releaseReviewerModel(packageVersion); + return evalReviewerConfiguration( + managerModel, + pinned + ? { + OPENCODE_FLOW_REVIEWER_MODEL: `${pinned.routeProvider}/${pinned.model}`, + } + : process.env, + ); +} + export function caseCatalogFor( scenarios: readonly (typeof SCENARIOS)[number][], sampling: EvalSampling, @@ -824,6 +840,8 @@ export async function runCampaign( if (sampling.kind === "release") { assertReleaseHost({ + packageVersion: packageJson.version, + recoveryApiKeySet: Boolean(process.env.TYPESAFE_API_KEY), platform: normalizeEvidencePlatform(process.platform), opencodeOverride: process.env.FLOW_OPENCODE_SMOKE_VERSION, reviewerModelOverride: process.env.OPENCODE_FLOW_REVIEWER_MODEL, @@ -913,9 +931,10 @@ export async function runCampaign( if ((await tarballSha256(tarball)) !== artifact.tarballSha256) { throw new Error("Packed artifact changed before host installation."); } - const reviewerModel = evalReviewerConfiguration( - models[0] ?? "", - process.env, + const reviewerModel = ( + sampling.kind === "release" + ? evalReleaseReviewerConfiguration(models[0] ?? "", packageJson.version) + : evalReviewerConfiguration(models[0] ?? "") ).pluginOptions?.model; await preflight( packageCache, @@ -984,7 +1003,10 @@ export async function runCampaign( /** One attempt, start to finish, printing a single line when it lands. */ const runAttempt = async (job: Job): Promise => { const { model, scenario, attempt, scheduledAttempts } = job; - const reviewer = evalReviewerConfiguration(model); + const reviewer = + sampling.kind === "release" + ? evalReleaseReviewerConfiguration(model, packageJson.version) + : evalReviewerConfiguration(model); const measuredHostConfigSha256 = sampling.kind === "release" ? releaseHostConfigSha256({ diff --git a/package.json b/package.json index 29e93321..61c1fcb2 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "opencode-plugin-flow", - "version": "9.5.0", + "version": "9.6.0", "description": "Small durable planning, validation, and review workflow for OpenCode", "type": "module", "repository": { diff --git a/scripts/eval-canary.ts b/scripts/eval-canary.ts index 052c4125..dafb4c23 100644 --- a/scripts/eval-canary.ts +++ b/scripts/eval-canary.ts @@ -23,7 +23,10 @@ import { packedPackageManifest, samePackedArtifact, } from "../evals/provenance.js"; -import { RELEASE_HOST_POLICY } from "../evals/release-policy.js"; +import { + RELEASE_HOST_POLICY, + releaseReviewerModel, +} from "../evals/release-policy.js"; import type { ActorIdentity, ArtifactIdentity } from "../evals/report.js"; import { reportArtifactForCanary } from "../evals/report-artifact.js"; import { assuranceProjection } from "../src/application/delivery.js"; @@ -1172,6 +1175,23 @@ export async function canaryRecordIssue(input: { if (!samePackedArtifact(record.artifact, input.expectedArtifact)) return "Canary artifact does not match the rebuilt artifact."; if (record.status !== "passed") return `Canary status is ${record.status}.`; + const pinned = releaseReviewerModel(input.version); + if (pinned) { + for (const role of ["manager", "reviewer"] as const) { + const actors = record.actors.filter((item) => item.role === role); + if ( + !actors.length || + actors.some( + (actor) => + canonicalJson(actor.requestedModel) !== canonicalJson(pinned) || + (actor.actualModel.kind === "observed" && + (actor.actualModel.value.routeProvider !== pinned.routeProvider || + actor.actualModel.value.model !== pinned.model)), + ) + ) + return `Canary ${role} model does not match the pinned release route.`; + } + } const now = (input.now ?? new Date()).getTime(); if (Date.parse(record.recordedAt) > now) return "Canary is future-dated."; if ( diff --git a/tests/eval-canary.test.ts b/tests/eval-canary.test.ts index 2dc6f58c..3b8f9446 100644 --- a/tests/eval-canary.test.ts +++ b/tests/eval-canary.test.ts @@ -11,7 +11,10 @@ import { } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; -import { RELEASE_HOST_POLICY } from "../evals/release-policy.js"; +import { + RELEASE_HOST_POLICY, + releaseReviewerModel, +} from "../evals/release-policy.js"; import { artifactIdentitySha256, CANARY_CHECKLIST_SHA256, @@ -1178,3 +1181,55 @@ describe("canary preparation", () => { }); }); }); + +test("9.6 canary rejects every wrong-role model including a second reviewer", async () => { + const model = releaseReviewerModel("9.6.0"); + if (!model) throw new Error("Missing candidate pin."); + for (const variant of [ + "manager", + "reviewer", + "second-reviewer", + "observed", + ] as const) { + const value = record({ + artifactValue: { ...artifact, packageVersion: "9.6.0" }, + }); + value.actors = ["manager", "reviewer"].map((role) => ({ + ...actor, + role: role === "manager" ? "manager" : "reviewer", + requestedModel: model, + actualModel: { kind: "observed", value: model }, + sessionIds: [""], + })); + if (variant === "second-reviewer") + value.actors.push({ + ...value.actors[1], + role: "reviewer", + requestedModel: { ...model, model: "other" }, + actualModel: { kind: "unobserved", reason: "unknown" }, + sessionIds: [""], + }); + else { + const index = variant === "manager" ? 0 : 1; + const selected = value.actors[index]; + if (!selected) throw new Error("Missing role."); + if (variant === "observed") + selected.actualModel = { + kind: "observed", + value: { ...model, routeProvider: "xai" }, + }; + else selected.requestedModel = { ...model, model: "other" }; + } + const { recordSha256: _prior, ...base } = value; + const candidate = { ...base, recordSha256: canaryRecordSha256(base) }; + expect( + await canaryRecordIssue({ + version: "9.6.0", + record: candidate, + expectedArtifact: candidate.artifact, + directory: "/unused", + now: new Date("2026-08-25T00:00:01Z"), + }), + ).toContain("pinned release route"); + } +}); diff --git a/tests/eval-release-sampling.test.ts b/tests/eval-release-sampling.test.ts index 524b6183..ba89c3bc 100644 --- a/tests/eval-release-sampling.test.ts +++ b/tests/eval-release-sampling.test.ts @@ -344,9 +344,9 @@ describe("release eval sampling", () => { ).toHaveLength(57); }); - test("current 9.5 candidate retains all twelve cases on its OpenAI-only grid", () => { + test("current 9.6 candidate retains all seventeen cases on its OpenAI-only grid", () => { const scenarios = releaseScenarios(); - expect(scenarios).toHaveLength(12); + expect(scenarios).toHaveLength(17); expect( releaseCatalog(packageJson.version).every( (row) => row.minProviders === 1, @@ -358,12 +358,12 @@ describe("release eval sampling", () => { sampling: { kind: "release", packageVersion: packageJson.version }, opencodeVersion: "1.18.31", }); - expect(plan.stoppingRule.count).toBe(57); - expect(plan.budget.maxAttempts).toBe(69); - expect(plan.abortPolicy.maxReplacementBlocks).toBe(12); + expect(plan.stoppingRule.count).toBe(72); + expect(plan.budget.maxAttempts).toBe(89); + expect(plan.abortPolicy.maxReplacementBlocks).toBe(17); expect( plan.cells.filter((cell) => cell.schedule === "primary"), - ).toHaveLength(57); + ).toHaveLength(72); expect(() => campaignPlanFor({ models: ["openai/gpt-6-sol"], diff --git a/tests/eval-report.test.ts b/tests/eval-report.test.ts index a531b6ca..d8594481 100644 --- a/tests/eval-report.test.ts +++ b/tests/eval-report.test.ts @@ -1048,3 +1048,38 @@ test("user prompt provenance freezes bytes while preserving old command records" delete instruction.text; expect(parseReport(fixture, caseCatalog()).ok).toBe(false); }); + +test("9.6 scheduled roles reject observed route substitutions without inventing revisions", () => { + for (const kind of ["actual", "host"] as const) { + const fixture = report(); + const attempt = fixture.attempts[0]; + const cell = fixture.plan.cells[0]; + if (!attempt || !cell || !("packageVersion" in attempt.artifact)) + throw new Error("Missing role fixture."); + attempt.artifact.packageVersion = "9.6.0"; + const actor = attempt.actors[0]; + if (!actor) throw new Error("Missing actor."); + cell.managerModel = actor.requestedModel; + cell.reviewerModel = actor.requestedModel; + fixture.plan.planSha256 = campaignPlanSha256(fixture.plan); + expect(parseReport(fixture, caseCatalog()).ok).toBe(true); + if (kind === "actual") + actor.actualModel = { + kind: "observed", + value: { + ...actor.requestedModel, + routeProvider: "xai", + model: "other", + }, + }; + else + actor.hostObservation = { + model: { + kind: "observed", + value: { providerID: "xai", modelID: "other" }, + }, + variant: { kind: "unobserved", reason: "field-unavailable" }, + }; + expect(parseReport(fixture, caseCatalog()).ok).toBe(false); + } +}); diff --git a/tests/qualification-cli.test.ts b/tests/qualification-cli.test.ts index 9c22ae8b..27d8d3f9 100644 --- a/tests/qualification-cli.test.ts +++ b/tests/qualification-cli.test.ts @@ -1,10 +1,9 @@ import { expect, test } from "bun:test"; import { createHash } from "node:crypto"; import { readFileSync } from "node:fs"; -import { mkdir, mkdtemp, readFile, rm } from "node:fs/promises"; +import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import { join } from "node:path"; -import { currentBunToolchain } from "../evals/bun-toolchain.js"; import { canonicalJson } from "../evals/canonical-json.js"; import { mapStrings } from "../evals/cassette.js"; import { @@ -20,7 +19,7 @@ import { retainedFailureEvidence, scenarioGradeInput, } from "../evals/grader-input.js"; -import { type Outcome, packPlugin } from "../evals/harness.js"; +import type { Outcome } from "../evals/harness.js"; import { evaluatorIdentity, inspectArtifact } from "../evals/provenance.js"; import { readQualificationBundle, @@ -45,6 +44,7 @@ import { scenarioStepInstruction } from "../evals/scenario-steps.js"; import { SCENARIOS } from "../evals/scenarios.js"; import packageJson from "../package.json" with { type: "json" }; import { prepareCanary, recordCanary } from "../scripts/eval-canary.js"; +import { materializeQualificationArchive } from "../scripts/materialize-qualification.js"; import { decisionRecordFor, qualifyV2 } from "../scripts/qualify-release.js"; import { assertQualificationBundle } from "../scripts/release-metadata.js"; import { assuranceProjection } from "../src/application/delivery.js"; @@ -54,7 +54,7 @@ import { operationInputDigest } from "../src/domain/operation.js"; import { autoQualifiedOutcome } from "./fixtures/auto-qualified-outcome.js"; test("current release defaults include three autonomous attempts", () => { - expect(packageJson.version).toBe("9.5.0"); + expect(packageJson.version).toBe("9.6.0"); expect( attemptsForScenario("auto-two-features-evidence", { kind: "release" }), ).toBe(3); @@ -307,7 +307,7 @@ function canaryTranscript(input: { }; } -test("qualifies and seals a complete exact-artifact campaign through the CLI", async () => { +test("qualifies and seals a historical 9.5 exact-artifact fixture campaign through the CLI", async () => { const repositoryRoot = join(import.meta.dir, ".."); const regradeAuthority = { qualify: qualifyV2, @@ -315,16 +315,26 @@ test("qualifies and seals a complete exact-artifact campaign through the CLI", a }; const temporary = await mkdtemp(join(tmpdir(), "flow-qualification-cli-")); try { - const artifactPath = await packPlugin( - repositoryRoot, - temporary, - currentBunToolchain(packageJson.packageManager), + const artifactPath = join(temporary, "historical-9.5.0.tgz"); + const historicalPath = await materializeQualificationArchive({ + descriptorPath: join( + repositoryRoot, + "evals/qualification/archives/9.5.0.json", + ), + outputRoot: join(temporary, "historical"), + }); + const historical = await readQualificationBundle(historicalPath); + const historicalArtifact = historical.files.find( + ({ ref }) => ref.role === "artifact", ); + if (!historicalArtifact) + throw new Error("Historical artifact role is absent."); + await writeFile(artifactPath, historicalArtifact.bytes); const artifact = await inspectArtifact({ repositoryRoot, tarballPath: artifactPath, }); - const scenarios = releaseScenarios(); + const scenarios = releaseScenarios("9.5.0"); const models = ["openai/gpt-6.1-sol"]; const plan = campaignPlanFor({ models, @@ -655,7 +665,7 @@ test("qualifies and seals a complete exact-artifact campaign through the CLI", a const bundlePath = stdout.trim().slice("VERIFIED: ".length); const bundle = await readQualificationBundle(bundlePath); expect(bundle.manifest.verdict).toBe("VERIFIED"); - expect(bundle.manifest.packageVersion).toBe(packageJson.version); + expect(bundle.manifest.packageVersion).toBe("9.5.0"); for (const role of FIXED_ROLES) { expect( bundle.files.filter(({ ref }) => ref.role === role && !ref.id), diff --git a/tests/release-960.test.ts b/tests/release-960.test.ts new file mode 100644 index 00000000..775bf8e5 --- /dev/null +++ b/tests/release-960.test.ts @@ -0,0 +1,133 @@ +import { expect, test } from "bun:test"; +import { + assertReleaseHost, + assertReleaseModels, + RELEASE_HOST_POLICY, + releaseCatalog, + releaseCellsFor, + releaseProfile, + releaseReviewerModel, +} from "../evals/release-policy.js"; +import { + campaignPlanFor, + evalReleaseReviewerConfiguration, + releaseScenarios, +} from "../evals/run.js"; +import { SCENARIOS } from "../evals/scenarios.js"; + +const version = "9.6.0"; +const profile = releaseProfile(version); +const models = profile.requiredModels; +if (!models) throw new Error("Candidate model grid is absent."); +const model = models[0]; +if (!model) throw new Error("Candidate model is absent."); + +test("candidate retains all prior policy rows and promotes five truthful delivery boundaries", () => { + const prior = releaseCatalog("9.5.0"); + expect(profile.catalog.slice(0, prior.length)).toEqual([...prior]); + const added = profile.catalog.slice(prior.length); + expect(added.map((row) => row.caseId)).toEqual([ + "delivery-summary-completed", + "delivery-summary-deferred", + "delivery-summary-observed-failure", + "delivery-full-detail-followup", + "delivery-idle-after-close", + ]); + for (const row of added) + expect(row).toMatchObject({ + caseVersion: 1, + release: "required", + minProviders: 1, + minScoredAttempts: 3, + minPassRate: 1, + }); + expect(model).toEqual({ + routeProvider: "openai", + gateway: null, + family: "gpt-6.1-sol", + model: "gpt-6.1-sol", + revision: null, + }); +}); + +test("candidate grid binds both roles and scheduled top-level dispatch counts", () => { + const cells = releaseCellsFor(models, version); + const primary = cells.filter((cell) => cell.schedule === "primary"); + const reserve = cells.filter( + (cell) => cell.schedule === "environment-reserve", + ); + expect(primary).toHaveLength(72); + expect(reserve).toHaveLength(17); + for (const cell of cells) + expect([cell.managerModel, cell.reviewerModel]).toEqual([model, model]); + const dispatches = (scheduled: typeof cells) => + scheduled.reduce((sum, cell) => { + const scenario = SCENARIOS.find((item) => item.id === cell.caseId); + if (!scenario) throw new Error("Missing scheduled scenario."); + return sum + scenario.steps.length; + }, 0); + expect(dispatches(primary)).toBe(90); + expect(dispatches(reserve)).toBe(23); + const plan = campaignPlanFor({ + models: ["openai/gpt-6.1-sol"], + scenarios: releaseScenarios(version), + sampling: { kind: "release", packageVersion: version }, + opencodeVersion: RELEASE_HOST_POLICY.opencodeVersion, + }); + expect(plan.stoppingRule.count).toBe(72); + expect(plan.budget.maxAttempts).toBe(89); + expect(plan.abortPolicy).toEqual({ + retry: "environment-only", + maxReplacementBlocks: 17, + }); +}); + +test("role pin and recovery requirement are candidate-only and reject environmental changes", () => { + expect( + evalReleaseReviewerConfiguration("openai/gpt-6.1-sol", version), + ).toEqual({ + requestedModel: "openai/gpt-6.1-sol", + requestedSteps: null, + pluginOptions: { model: "openai/gpt-6.1-sol" }, + }); + for (const old of ["9.1.0", "9.2.0", "9.3.0", "9.4.0", "9.5.0", "standard"]) { + expect(releaseReviewerModel(old)).toBeNull(); + expect( + releaseCellsFor(models, old).every((cell) => cell.reviewerModel === null), + ).toBe(true); + } + expect(() => + assertReleaseModels([{ ...model, routeProvider: "xai" }], version), + ).toThrow(); + expect(() => + assertReleaseModels([{ ...model, model: "gpt-6-sol" }], version), + ).toThrow(); + expect(() => + assertReleaseHost({ + platform: "linux", + packageVersion: version, + recoveryApiKeySet: true, + }), + ).toThrow("unset TYPESAFE_API_KEY"); + expect(() => + assertReleaseHost({ + platform: "linux", + packageVersion: "9.5.0", + recoveryApiKeySet: true, + }), + ).not.toThrow(); + expect(() => + assertReleaseHost({ + platform: "linux", + packageVersion: version, + reviewerModelOverride: "openai/gpt-6.1-sol", + }), + ).toThrow(); + expect(() => + assertReleaseHost({ + platform: "linux", + packageVersion: version, + reviewerStepsOverride: "20", + }), + ).toThrow(); +}); From f2a7ef78cd4f081f173f7982f829454e630c650c Mon Sep 17 00:00:00 2001 From: vriesd Date: Mon, 5 Oct 2026 11:01:52 +0200 Subject: [PATCH 2/2] test(evals): preserve integration fixtures across the 9.6 profile --- tests/fixtures/eval-reserve-cancellation-child.ts | 3 +++ tests/qualification-cli.test.ts | 14 ++++++++++++-- 2 files changed, 15 insertions(+), 2 deletions(-) diff --git a/tests/fixtures/eval-reserve-cancellation-child.ts b/tests/fixtures/eval-reserve-cancellation-child.ts index e0ae27bf..b1a79b56 100644 --- a/tests/fixtures/eval-reserve-cancellation-child.ts +++ b/tests/fixtures/eval-reserve-cancellation-child.ts @@ -114,6 +114,9 @@ class FakeReleaseHost { await stopHere(this.signal, primaryCount + 1); return "quiet"; } + async runPrompt(): Promise<"quiet"> { + return this.runCommand(); + } async outcome(sessionIds: string[]): Promise { event(`outcome:${this.attempt}`); const gap = [1, 4, 7].includes(this.attempt); diff --git a/tests/qualification-cli.test.ts b/tests/qualification-cli.test.ts index 27d8d3f9..e53fef7b 100644 --- a/tests/qualification-cli.test.ts +++ b/tests/qualification-cli.test.ts @@ -746,11 +746,21 @@ test("qualifies and seals a historical 9.5 exact-artifact fixture campaign throu }); expect(releaseAuthority.summary.providers).toHaveLength(1); const notesPath = join(temporary, "release-notes.md"); + const metadataRoot = join(temporary, "historical-release-metadata"); + await mkdir(metadataRoot); + await writeFile( + join(metadataRoot, "package.json"), + JSON.stringify({ ...packageJson, version: artifact.packageVersion }), + ); + await writeFile( + join(metadataRoot, "CHANGELOG.md"), + await readFile(join(repositoryRoot, "CHANGELOG.md")), + ); const metadata = Bun.spawn( [ "bun", "run", - "scripts/release-metadata.ts", + join(repositoryRoot, "scripts/release-metadata.ts"), "--tag", `v${artifact.packageVersion}`, "--notes-file", @@ -762,7 +772,7 @@ test("qualifies and seals a historical 9.5 exact-artifact fixture campaign throu "--bundles-dir", bundlesDirectory, ], - { cwd: repositoryRoot, stdout: "pipe", stderr: "pipe" }, + { cwd: metadataRoot, stdout: "pipe", stderr: "pipe" }, ); const [metadataStdout, metadataStderr, metadataExit] = await Promise.all([ new Response(metadata.stdout).text(),