diff --git a/evals/cassette.ts b/evals/cassette.ts index 4b47b97d..64d28ec7 100644 --- a/evals/cassette.ts +++ b/evals/cassette.ts @@ -16,6 +16,7 @@ // nothing that could be a credential is ever written into one, and the recording // host's absolute paths are replaced by a token rather than baked in. +import { z } from "zod"; import { commandUsesManagedJUnitPath, MANAGED_JUNIT_PATH, @@ -277,6 +278,35 @@ export function capturedValidationIdentity( return { id: marker.id, revision: marker.recordedRevision }; } +const CapturedResult = z + .object({ + id: z.string().min(1).max(256), + scope: z.enum(["focused", "broad"]), + intent: z.literal("pass"), + passed: z.literal(true), + observed: z.literal(false), + recordedRevision: z.number().int().safe().positive(), + assertions: z + .array( + z + .object({ name: z.string().min(1), status: z.literal("passed") }) + .strict(), + ) + .optional(), + fullOutputDigest: z + .string() + .regex(/^sha256:[a-f0-9]{64}$/) + .optional(), + }) + .strict(); +export function capturedValidationResult( + output: string, +): z.infer | null { + if (output.split("[flow-validation]").length !== 2) return null; + const parsed = CapturedResult.safeParse(validationMarker(output)); + return parsed.success ? parsed.data : null; +} + function validationReport(output: string): string | null { const assertions = validationMarker(output)?.assertions; if (!Array.isArray(assertions)) return null; diff --git a/evals/delivery-presentation.ts b/evals/delivery-presentation.ts index c6c661e8..f9fe9797 100644 --- a/evals/delivery-presentation.ts +++ b/evals/delivery-presentation.ts @@ -48,6 +48,16 @@ type CurrentHandoffFacts = { auxiliaryCounts: AuxiliaryCount[]; assuranceCheckClaims: { count: number; status: "satisfied" }[]; unavailableProofPlatforms: string[]; + unavailableCommands: { + command: string; + targetPlatform: string; + hostPlatform: string; + }[]; + independentReview: ( + | { kind: "passed"; findings: "none" | "not-claimed" } + | { kind: "not-performed" } + | null + )[]; observations: CommandObservation[]; unsupported: string[]; }; @@ -273,7 +283,7 @@ function closureStatement( value: string, ): { closure: Closure; unavailablePlatform: string | null } | null { const match = - /^(completed|complete|deferred|abandoned)(?: and archived(?: the Flow session)?)?(?: because (macOS|darwin|Linux|Windows) validation is unavailable)?$/i.exec( + /^(completed|complete|deferred|abandoned)(?: and archived(?: (?:the |this )?(?:current )?(?:Flow )?(?:session|workflow))?)?(?: because (macOS|darwin|Linux|Windows) validation is unavailable)?$/i.exec( value, ); if (!match) return null; @@ -295,6 +305,10 @@ function commandStatusAssertion(text: string): boolean { text, ); } +function platformValue(value: string) { + const lower = value.toLowerCase(); + return lower === "macos" ? "darwin" : lower === "windows" ? "win32" : lower; +} function sentenceBoundary(text: string, start = 0) { let quote: string | null = null; for (let index = start; index < text.length; index++) { @@ -334,6 +348,21 @@ function commandResultValue(line: string, commands: readonly string[]) { .find((value) => value !== null); if (!matched) continue; const rawBody = matched[1] ?? ""; + const availability = + /^ on (macOS|darwin|Linux|Windows); unavailable on this (macOS|darwin|Linux|Windows) host\.?$/i.exec( + rawBody, + ); + if (availability) + return { + observation: null, + unavailable: { + command, + targetPlatform: platformValue(availability[1] ?? ""), + hostPlatform: platformValue(availability[2] ?? ""), + }, + remainder: "", + source: line, + }; let boundary = sentenceBoundary(rawBody); while ( boundary.end < rawBody.length && @@ -473,6 +502,8 @@ export function currentHandoffFacts( auxiliaryCounts: [], assuranceCheckClaims: [], unavailableProofPlatforms: [], + unavailableCommands: [], + independentReview: [], observations: [], unsupported: [], }; @@ -486,6 +517,8 @@ export function currentHandoffFacts( } const commandRecord = commandResultValue(line, observationCommands); if (commandRecord) { + if ("unavailable" in commandRecord && commandRecord.unavailable) + facts.unavailableCommands.push(commandRecord.unavailable); if (commandRecord.observation) { facts.observations.push(commandRecord.observation); if (commandRecord.observation.qualification === null) @@ -507,9 +540,30 @@ export function currentHandoffFacts( ) { facts.unsupported.push(line); } + if (/^(?:[^:]+:\s*)?(?:node|bun) \S+.*\bunavailable\b/i.test(line)) + facts.unsupported.push(line); for (const segment of line.split(/;|\.\s+(?=[A-Z])/)) { const claim = segment.trim().replace(/\.$/, ""); if (!claim) continue; + const review = /^Independent review(?::|\s)\s*(.*)$/i.exec(claim); + if (review) { + const value = review[1] ?? ""; + facts.independentReview.push( + /^(?:was )?not performed$/i.test(value) + ? { kind: "not-performed" } + : /^(?:has |was )?(?:passed|passed with no findings)$/i.test(value) + ? { + kind: "passed", + findings: /with no findings$/i.test(value) + ? "none" + : "not-claimed", + } + : null, + ); + if (facts.independentReview.at(-1) === null) + facts.unsupported.push(claim); + continue; + } const handoff = closureProgressValue(claim); if (handoff) { if ("closure" in handoff) facts.closure.push(handoff.closure ?? null); diff --git a/evals/delivery-scenario-checks.ts b/evals/delivery-scenario-checks.ts index 1fb5ae30..10dd3eb7 100644 --- a/evals/delivery-scenario-checks.ts +++ b/evals/delivery-scenario-checks.ts @@ -2,6 +2,7 @@ import { posix } from "node:path"; import { z } from "zod"; import { isArtifactPath } from "../src/domain/artifact.js"; import { canonicalJson } from "./canonical-json.js"; +import { capturedValidationResult } from "./cassette.js"; import { currentHandoffFacts, fullReportMatches, @@ -245,6 +246,295 @@ function primary( ); } +const Digest = z.string().regex(/^sha256:[a-f0-9]{64}$/); +const CapturedValidation = z.object({ + id: Id, + featureId: Id, + runId: Id, + command: z.string(), + scope: z.enum(["focused", "broad"]), + intent: z.literal("pass"), + exitCode: z.literal(0), + outputComplete: z.literal(true), + sourceDigest: Digest, + outputDigest: Digest, + recordedRevision: z.number().int().safe().positive(), + hostPlatform: z.enum(["linux", "darwin", "win32", "other"]), + ineligibleReason: z.never().optional(), + resultsPath: z.string().optional(), + observedAssertions: z + .array(z.object({ name: Id, status: z.literal("passed") }).strict()) + .optional(), +}); +const CaptureArm = z.object({ + status: z.literal("ok"), + workflowData: z.object({ + capture: z.object({ + captureId: Id, + expiresInMs: z.number().finite().positive(), + }), + command: z.string(), + scope: z.string(), + intent: z.literal("pass"), + }), +}); +const ArmRequest = z.object({ + featureId: Id, + command: z.string(), + scope: z.string(), + expectedRevision: z.number().int().safe().nonnegative(), + sessionId: Id.optional(), + intent: z.literal("pass").optional(), +}); +function nativeOutputComplete( + metadata: Record, +): boolean | null { + if (metadata.truncated === true || metadata.complete === false) return false; + if (metadata.truncated === false || metadata.complete === true) return true; + return null; +} +function capturedDeferredPass( + input: ScenarioGradeInput, + archive: z.infer, + close: ScenarioGradeInput["allCalls"][number], + command: string, +): { hostPlatform: string; status: Record } | null { + if (archive.closure.kind !== "deferred") return null; + const candidates = archive.runs.flatMap((run) => + run.validations + .filter((value) => value.command === command) + .map((value) => ({ run, value })), + ); + for (const candidate of candidates) { + const parsed = CapturedValidation.safeParse(candidate.value); + if (!parsed.success) continue; + const proof = attestedCapture( + input, + archive, + close, + candidate.run, + parsed.data, + ); + if (proof) return proof; + } + return null; +} +function attestedCapture( + input: ScenarioGradeInput, + archive: z.infer, + close: ScenarioGradeInput["allCalls"][number], + run: z.infer["runs"][number], + validation: z.infer, +): { hostPlatform: string; status: Record } | null { + const command = validation.command; + if ( + run.id !== validation.runId || + run.featureId !== validation.featureId || + validation.recordedRevision > archive.closure.recordedRevision + ) + return null; + if ( + archive.runs + .flatMap((run) => run.validations) + .filter((value) => value.id === validation.id).length !== 1 + ) + return null; + const arms = input.allCalls.filter( + (call) => + call.tool === "flow_validation_start" && + CaptureArm.safeParse(call.output).success && + CaptureArm.parse(call.output).workflowData.capture.captureId === + validation.id, + ); + if (arms.length !== 1) return null; + const arm = arms[0]; + if (arm?.status !== "completed" || !primary(arm, input)) return null; + const request = ArmRequest.safeParse(arm.input.request); + const data = CaptureArm.parse(arm.output).workflowData; + if ( + !request.success || + request.data.featureId !== validation.featureId || + request.data.command !== command || + request.data.scope !== validation.scope || + (request.data.sessionId !== undefined && + request.data.sessionId !== archive.id) || + request.data.expectedRevision + 1 !== validation.recordedRevision || + data.command !== command || + data.scope !== validation.scope + ) + return null; + const armIndex = input.allCalls.indexOf(arm); + const next = input.allCalls + .slice(armIndex + 1) + .find( + (call) => + call.native?.sessionId === arm.native?.sessionId && + ["bash", "flow_validation_start"].includes(call.tool), + ); + if ( + next?.tool !== "bash" || + next.status !== "completed" || + !primary(next, input) || + next.input.command !== command || + next.metadata.exit !== 0 + ) + return null; + const marker = capturedValidationResult(next.rawOutput); + const complete = nativeOutputComplete(next.metadata); + if ( + !marker || + complete === null || + (complete === false && + marker.fullOutputDigest !== validation.outputDigest) || + marker.id !== validation.id || + marker.recordedRevision !== validation.recordedRevision || + marker.scope !== validation.scope || + (marker.fullOutputDigest !== undefined && + marker.fullOutputDigest !== validation.outputDigest) || + canonicalJson(marker.assertions ?? []) !== + canonicalJson(validation.observedAssertions ?? []) + ) + return null; + const armedAt = arm.native?.completedAt; + const beganAt = next.native?.startedAt; + const finishedAt = next.native?.completedAt; + const closedAt = close.native?.startedAt; + if ( + armedAt == null || + beganAt == null || + finishedAt == null || + closedAt == null || + armedAt > beganAt || + beganAt >= armedAt + data.capture.expiresInMs || + finishedAt >= closedAt + ) + return null; + const nativeIds = input.allCalls.flatMap((call) => + call.native?.callId ? [call.native.callId] : [], + ); + if (new Set(nativeIds).size !== nativeIds.length) return null; + const declared = (archive.plan.evidence ?? []).filter( + (entry) => entry.command === command, + ); + const feature = archive.plan.features.find( + (value) => value.id === validation.featureId, + ); + const checks = z + .array( + z + .object({ + command: z.string(), + intent: z.string(), + platform: z.string().optional(), + }) + .passthrough(), + ) + .safeParse(feature?.checks ?? []); + if ( + !checks.success || + checks.data.some( + (check) => + check.command === command && + (check.intent !== "pass" || + (check.platform !== undefined && + check.platform !== validation.hostPlatform)), + ) + ) + return null; + if ( + declared.some( + (entry) => + entry.platform !== undefined && + entry.platform !== validation.hostPlatform, + ) + ) + return null; + if ( + [ + ...declared, + ...checks.data.filter((check) => check.command === command), + ].some( + (entry) => + Array.isArray(entry.assertions) && + entry.assertions.some( + (name) => + typeof name !== "string" || + !validation.observedAssertions?.some( + (assertion) => assertion.name === name, + ), + ), + ) + ) + return null; + let attestedStatus: Record | null = null; + for (const call of input.allCalls.slice( + input.allCalls.indexOf(next) + 1, + input.allCalls.indexOf(close), + )) { + if ( + call.tool !== "flow_status" || + call.status !== "completed" || + !primary(call, input) || + !call.native || + call.native.sessionId !== next.native?.sessionId || + call.native.startedAt == null || + call.native.completedAt == null || + call.native.startedAt < finishedAt || + call.native.completedAt >= closedAt + ) + continue; + const status = z + .object({ + status: z.literal("ok"), + workflowData: z.object({ + projection: z + .object({ + sessionId: Id, + runs: z.array( + z + .object({ + id: Id, + featureId: Id, + validations: z.array(z.unknown()), + }) + .passthrough(), + ), + }) + .passthrough(), + }), + }) + .safeParse(call.output); + if ( + !status.success || + status.data.workflowData.projection.sessionId !== archive.id + ) + continue; + const projection = status.data.workflowData.projection; + const observations = projection.runs.flatMap((run) => + run.validations + .filter( + (value) => + z.object({ id: Id }).safeParse(value).data?.id === validation.id, + ) + .map((value) => ({ run, value })), + ); + if (observations.length !== 1) return null; + const observation = observations[0]; + const attested = CapturedValidation.safeParse(observation?.value); + if ( + !attested.success || + observation?.run.id !== validation.runId || + observation.run.featureId !== validation.featureId || + canonicalJson(attested.data) !== canonicalJson(validation) + ) + return null; + attestedStatus = projection; + } + return attestedStatus + ? { hostPlatform: validation.hostPlatform, status: attestedStatus } + : null; +} + export function deliveryIssues( input: ScenarioGradeInput, expected: DeliveryExpectation, @@ -449,6 +739,97 @@ export function deliveryIssues( ]), ].sort((a, b) => b.length - a.length); const facts = currentHandoffFacts(input.finalText, commands); + const findings = + close.data.workflowData.delivery.findingsDigest?.filter( + (finding) => finding.live, + ) ?? + archive.plan.features.flatMap( + (feature) => + archive.runs + .filter( + (run) => run.featureId === feature.id && run.state !== "superseded", + ) + .at(-1) + ?.reviews.at(-1)?.result?.findings ?? [], + ); + const acceptedReviews = archive.runs.flatMap((run) => + run.reviews.filter( + (review) => + review.result?.verdict === "passed" && + run.validations.some( + (validation) => + review.validationIds.includes(validation.id) && + validation.outputComplete && + validation.intent !== "observe" && + validation.sourceDigest === review.sourceDigest, + ), + ), + ); + const hasAcceptedReview = acceptedReviews.length > 0; + if ( + facts.independentReview.some( + (value) => + value === null || + (value.kind === "passed" + ? !hasAcceptedReview || + checkReviewerEvidenceAccess(input, input.archives[0]).length > 0 || + (value.findings === "none" && findings.length !== 0) + : hasAcceptedReview), + ) + ) + issues.push( + "Independent review claim contradicts accepted native review evidence.", + ); + for (const claim of facts.unavailableCommands) { + const proof = + conclusion === "completion-not-claimed" + ? capturedDeferredPass(input, archive, accepted, expected.gate) + : null; + const declared = (archive.plan.evidence ?? []).some( + (entry) => + entry.command === claim.command && + entry.scope === "extra" && + entry.platform === claim.targetPlatform, + ); + const nativePlan = z + .object({ + evidence: z.array( + z + .object({ + command: z.string(), + scope: z.string(), + platform: z.string().optional(), + }) + .passthrough(), + ), + }) + .safeParse(proof?.status.plan); + const unmet = + nativePlan.success && + nativePlan.data.evidence.some( + (entry) => + entry.command === claim.command && + entry.scope === "extra" && + entry.platform === claim.targetPlatform, + ); + if ( + !proof || + claim.command !== expected.missingEvidenceCommand || + claim.targetPlatform === claim.hostPlatform || + claim.hostPlatform !== proof.hostPlatform || + !declared || + !unmet || + archive.runs.some((run) => + run.validations.some( + (validation) => + validation.command === claim.command && validation.exitCode === 0, + ), + ) + ) + issues.push( + "Unavailable command claim contradicts declared unmet native platform proof.", + ); + } const nativeChecks = close.data.workflowData.delivery.assurance.checks ?? close.data.workflowData.delivery.report.flatMap((line) => { @@ -525,7 +906,12 @@ export function deliveryIssues( ), ), ); - if (!acceptedPass) + const capturedPass = + expected.closure === "deferred" && + conclusion === "completion-not-claimed" && + result.exitCode === 0 && + capturedDeferredPass(input, archive, accepted, result.command) !== null; + if (!acceptedPass && !capturedPass) issues.push( "Claimed command pass lacks matching accepted complete source evidence.", ); @@ -582,19 +968,6 @@ export function deliveryIssues( issues.push( "Recorded feature progress was omitted, changed, or contradicted.", ); - const findings = - close.data.workflowData.delivery.findingsDigest?.filter( - (finding) => finding.live, - ) ?? - archive.plan.features.flatMap( - (feature) => - archive.runs - .filter( - (run) => run.featureId === feature.id && run.state !== "superseded", - ) - .at(-1) - ?.reviews.at(-1)?.result?.findings ?? [], - ); const blockedFeatures = archive.plan.features.filter( (feature) => archive.runs.findLast( diff --git a/tests/delivery-claim-composition.test.ts b/tests/delivery-claim-composition.test.ts index bb8a4b0f..73ed454b 100644 --- a/tests/delivery-claim-composition.test.ts +++ b/tests/delivery-claim-composition.test.ts @@ -336,7 +336,8 @@ test("superseded blocked run does not invalidate current zero blocked features", expect(deliveryIssues(input, expected)).toEqual([]); }); test("zero blocked feature claim is independent from a live blocking review finding", () => { - const input = fixture(actual); + const reviewPassed = actual.replace("passed with no findings", "passed"); + const input = fixture(reviewPassed); const close = input.allCalls.find( (call) => call.tool === "flow_session_close", ); @@ -347,7 +348,7 @@ test("zero blocked feature claim is independent from a live blocking review find expect(deliveryIssues(input, expected)).toEqual([]); expect( deliveryIssues( - { ...input, finalText: `${actual}\nNo blockers.` }, + { ...input, finalText: `${reviewPassed}\nNo blockers.` }, expected, ), ).toContain(countIssue); diff --git a/tests/delivery-deferred-capture.test.ts b/tests/delivery-deferred-capture.test.ts new file mode 100644 index 00000000..00da8d7d --- /dev/null +++ b/tests/delivery-deferred-capture.test.ts @@ -0,0 +1,756 @@ +import { expect, test } from "bun:test"; +import { deliveryIssues } from "../evals/delivery-scenario-checks.js"; +import type { ScenarioGradeInput } from "../evals/grader-input.js"; +import { autoQualifiedOutcome } from "./fixtures/auto-qualified-outcome.js"; +import { + deferredAnswer, + deferredCaptureOutcome, + externalCommand, + localCommand, + nativeTrace, + record, + validation, +} from "./fixtures/deferred-capture-outcome.js"; +import completed from "./fixtures/delivery-flow-zero-count-answer.json" with { + type: "json", +}; + +const expected = { + closure: "deferred" as const, + presentation: "summary" as const, + gate: localCommand, + missingEvidenceCommand: externalCommand, + allowedPaths: ["src/parser.mjs"], +}; +const noOptionalClaims = deferredAnswer + .replace("the current Flow session", "the Flow session") + .replace(/^- \*\*Linux validation:\*\*.*\n/m, "") + .replace(/^- \*\*Outstanding proof:\*\*.*\n/m, ""); +const capturedOnly = noOptionalClaims.replace( + "- **Progress:**", + `- ${localCommand} passed with exit code 0.\n- **Progress:**`, +); +function call(input: ScenarioGradeInput, tool: string) { + const result = input.allCalls.find((value) => value.tool === tool); + if (!result) throw new Error(`Missing fixture ${tool}.`); + return result; +} +function statusValidation(input: ScenarioGradeInput) { + const projection = record( + record(record(call(input, "flow_status").output).workflowData).projection, + ); + return record( + (record((projection.runs as unknown[])[0]).validations as unknown[])[0], + ); +} +test("synthetic deferred native closure is valid without optional command claims", () => { + expect( + deliveryIssues(deferredCaptureOutcome(noOptionalClaims), expected), + ).toEqual([]); +}); +test("truthful retained deferred wording accepts local captured pass, missing macOS proof and current archive object", () => { + expect(deliveryIssues(deferredCaptureOutcome(), expected)).toEqual([]); +}); +test("local captured pass does not require a reviewer when completion is explicitly not claimed", () => { + expect( + deliveryIssues(deferredCaptureOutcome(capturedOnly), expected), + ).toEqual([]); +}); +test("native status digest need not hash portable redacted Bash output", () => { + const input = deferredCaptureOutcome(capturedOnly); + Object.assign(call(input, "bash"), { + rawOutput: call(input, "bash").rawOutput.replace( + "(no output)", + "portable [REDACTED] /workspace", + ), + }); + expect(deliveryIssues(input, expected)).toEqual([]); +}); +for (const matching of [true, false]) { + test(`optional terminal full output digest must match native attested capture ${matching}`, () => { + const input = deferredCaptureOutcome(capturedOnly); + const digest = matching + ? validation(input).outputDigest + : `sha256:${"b".repeat(64)}`; + const output = call(input, "bash").rawOutput.replace( + '"recordedRevision":4', + `"fullOutputDigest":${JSON.stringify(digest)},"recordedRevision":4`, + ); + Object.assign(call(input, "bash"), { rawOutput: output, output }); + const issues = deliveryIssues(input, expected); + if (matching) expect(issues).toEqual([]); + else + expect(issues).toContain( + "Claimed command pass lacks matching accepted complete source evidence.", + ); + }); +} +for (const [name, flags, completeness] of [ + ["explicit complete", { complete: true }, true], + ["explicit untruncated", { truncated: false }, true], + ["both complete signals", { truncated: false, complete: true }, true], + ["truncated overrides complete", { truncated: true, complete: true }, false], + [ + "incomplete overrides untruncated", + { truncated: false, complete: false }, + false, + ], + ["explicit truncated", { truncated: true }, false], + ["explicit incomplete", { complete: false }, false], + ["unknown signals", {}, null], +] as const) { + for (const markerDigest of ["absent", "matching", "wrong"] as const) { + test(`captured completeness ${name} with ${markerDigest} full output digest`, () => { + const input = deferredCaptureOutcome(capturedOnly); + const bash = call(input, "bash"); + const metadata = { exit: 0, ...flags }; + const digest = + markerDigest === "matching" + ? validation(input).outputDigest + : `sha256:${"b".repeat(64)}`; + const output = + markerDigest === "absent" + ? bash.rawOutput + : bash.rawOutput.replace( + '"recordedRevision":4', + `"fullOutputDigest":${JSON.stringify(digest)},"recordedRevision":4`, + ); + Object.assign(bash, { metadata, rawOutput: output, output }); + const supported = + markerDigest !== "wrong" && + (completeness === true || + (completeness === false && markerDigest === "matching")); + const issues = deliveryIssues(input, expected); + if (supported) expect(issues).toEqual([]); + else + expect(issues).toContain( + "Claimed command pass lacks matching accepted complete source evidence.", + ); + }); + } +} +for (const [name, mutate] of [ + [ + "absent arm", + (g: ScenarioGradeInput) => { + Object.assign(g, { + allCalls: g.allCalls.filter((c) => c.tool !== "flow_validation_start"), + }); + }, + ], + [ + "absent Bash", + (g: ScenarioGradeInput) => { + Object.assign(g, { + allCalls: g.allCalls.filter((c) => c.tool !== "bash"), + }); + }, + ], + [ + "absent status", + (g: ScenarioGradeInput) => { + Object.assign(g, { + allCalls: g.allCalls.filter((c) => c.tool !== "flow_status"), + }); + }, + ], + [ + "counterfeit native call", + (g: ScenarioGradeInput) => { + Object.assign(call(g, "bash"), { native: null }); + }, + ], + [ + "child Bash", + (g: ScenarioGradeInput) => { + const c = call(g, "bash"); + if (c.native) c.native.sessionId = "ses_child"; + }, + ], + [ + "wrong next command", + (g: ScenarioGradeInput) => { + call(g, "bash").input.command = "node scripts/other.mjs"; + }, + ], + [ + "wrong arm capture", + (g: ScenarioGradeInput) => { + record( + record(record(call(g, "flow_validation_start").output).workflowData) + .capture, + ).captureId = "foreign-capture"; + }, + ], + [ + "wrong arm feature", + (g: ScenarioGradeInput) => { + record(call(g, "flow_validation_start").input.request).featureId = + "foreign-feature"; + }, + ], + [ + "expired arm", + (g: ScenarioGradeInput) => { + record( + record(record(call(g, "flow_validation_start").output).workflowData) + .capture, + ).expiresInMs = 1; + }, + ], + [ + "nonterminal marker", + (g: ScenarioGradeInput) => { + Object.assign(call(g, "bash"), { + rawOutput: `${call(g, "bash").rawOutput}\nadditional command output`, + }); + }, + ], + [ + "duplicate marker", + (g: ScenarioGradeInput) => { + const c = call(g, "bash"); + Object.assign(c, { + rawOutput: `${c.rawOutput}\n${c.rawOutput.split("\n").at(-1)}`, + }); + }, + ], + [ + "missing marker", + (g: ScenarioGradeInput) => { + Object.assign(call(g, "bash"), { rawOutput: "(no output)" }); + }, + ], + [ + "failed marker", + (g: ScenarioGradeInput) => { + Object.assign(call(g, "bash"), { + rawOutput: call(g, "bash").rawOutput.replace( + '"passed":true', + '"passed":false', + ), + }); + }, + ], + [ + "observed marker", + (g: ScenarioGradeInput) => { + Object.assign(call(g, "bash"), { + rawOutput: call(g, "bash").rawOutput.replace( + '"observed":false', + '"observed":true', + ), + }); + }, + ], + [ + "drift marker", + (g: ScenarioGradeInput) => { + Object.assign(call(g, "bash"), { + rawOutput: call(g, "bash").rawOutput.replace( + '"recordedRevision":4', + '"ineligibleReason":"source-drift","recordedRevision":4', + ), + }); + }, + ], + [ + "nonzero native exit", + (g: ScenarioGradeInput) => { + call(g, "bash").metadata.exit = 1; + }, + ], + [ + "truncated native output", + (g: ScenarioGradeInput) => { + call(g, "bash").metadata.truncated = true; + }, + ], + [ + "archive source mismatch", + (g: ScenarioGradeInput) => { + validation(g).sourceDigest = `sha256:${"b".repeat(64)}`; + }, + ], + [ + "status source mismatch", + (g: ScenarioGradeInput) => { + statusValidation(g).sourceDigest = `sha256:${"b".repeat(64)}`; + }, + ], + [ + "archive output mismatch", + (g: ScenarioGradeInput) => { + validation(g).outputDigest = `sha256:${"b".repeat(64)}`; + }, + ], + [ + "observe intent", + (g: ScenarioGradeInput) => { + validation(g).intent = "observe"; + }, + ], + [ + "incomplete archive output", + (g: ScenarioGradeInput) => { + validation(g).outputComplete = false; + }, + ], + [ + "archive drift", + (g: ScenarioGradeInput) => { + validation(g).ineligibleReason = "source-drift"; + }, + ], + [ + "foreign status session", + (g: ScenarioGradeInput) => { + record( + record(record(call(g, "flow_status").output).workflowData).projection, + ).sessionId = "foreign-session"; + }, + ], + [ + "duplicate archived capture", + (g: ScenarioGradeInput) => { + const run = record((record(g.archives[0]).runs as unknown[])[0]); + (run.validations as unknown[]).push(structuredClone(validation(g))); + }, + ], + [ + "duplicate native arm", + (g: ScenarioGradeInput) => { + const arm = structuredClone(call(g, "flow_validation_start")); + if (arm.native) + Object.assign(arm.native, { + messageId: "msg_rearm", + partId: "prt_rearm", + callId: "call_rearm", + startedAt: 21, + completedAt: 22, + }); + Object.assign(g, { + allCalls: [...g.allCalls.slice(0, 4), arm, ...g.allCalls.slice(4)], + }); + }, + ], + [ + "intervening mismatched Bash", + (g: ScenarioGradeInput) => { + const shell = structuredClone(call(g, "bash")); + shell.input.command = "node scripts/other.mjs"; + if (shell.native) + Object.assign(shell.native, { + messageId: "msg_intervening", + partId: "prt_intervening", + callId: "call_intervening", + startedAt: 21, + completedAt: 22, + }); + const at = g.allCalls.indexOf(call(g, "bash")); + Object.assign(g, { + allCalls: [...g.allCalls.slice(0, at), shell, ...g.allCalls.slice(at)], + }); + }, + ], + [ + "status before Bash", + (g: ScenarioGradeInput) => { + const c = call(g, "flow_status"); + if (c.native) Object.assign(c.native, { startedAt: 18, completedAt: 19 }); + }, + ], + [ + "status after close", + (g: ScenarioGradeInput) => { + const c = call(g, "flow_status"); + if (c.native) + Object.assign(c.native, { startedAt: 102, completedAt: 103 }); + }, + ], + [ + "unknown source digest", + (g: ScenarioGradeInput) => { + validation(g).sourceDigest = "model-asserted-source"; + statusValidation(g).sourceDigest = "model-asserted-source"; + }, + ], +] as const) { + test(`deferred captured pass refuses ${name}`, () => { + const input = deferredCaptureOutcome(capturedOnly); + mutate(input); + Object.assign(input, { hostTrace: nativeTrace(input.allCalls) }); + expect(deliveryIssues(input, expected)).not.toEqual([]); + }); +} +test("captured pass cannot supply completion without accepted independent review", () => { + const input = deferredCaptureOutcome(capturedOnly); + record(record(input.archives[0]).closure).kind = "completed"; + record(call(input, "flow_session_close").input.request).kind = "completed"; + expect( + deliveryIssues(input, { ...expected, closure: "completed" }), + ).toContain("Required gate lacks source-bound passing reviewed evidence."); +}); +test("deferred native result cannot justify invented independent review acceptance", () => { + const text = noOptionalClaims.replace( + "Independent review was not performed.", + "Independent review passed.", + ); + expect(deliveryIssues(deferredCaptureOutcome(text), expected)).not.toEqual( + [], + ); +}); +test("availability refuses an obligation already recorded as passing", () => { + const input = deferredCaptureOutcome(); + const run = record((record(input.archives[0]).runs as unknown[])[0]); + (run.validations as unknown[]).push({ + ...validation(input), + id: "external-validation", + command: externalCommand, + hostPlatform: "darwin", + }); + expect(deliveryIssues(input, expected)).toContain( + "Unavailable external proof was falsely recorded as passing.", + ); +}); +test("command-scoped unavailable proof is a distinct truthful statement", () => { + const text = noOptionalClaims.replace( + "- **Progress:**", + `- Outstanding proof: ${externalCommand} on macOS; unavailable on this Linux host.\n- **Progress:**`, + ); + expect(deliveryIssues(deferredCaptureOutcome(text), expected)).toEqual([]); +}); +test("unregistered unavailable command cannot borrow the real missing platform obligation", () => { + const text = noOptionalClaims.replace( + "- **Progress:**", + "- Outstanding proof: node scripts/foreign.mjs on macOS; unavailable on this Linux host.\n- **Progress:**", + ); + expect(deliveryIssues(deferredCaptureOutcome(text), expected)).toContain( + "Unsupported or conflicting current handoff assertions.", + ); +}); +test("typed feature check cannot label a Linux capture as passing Darwin proof", () => { + const input = deferredCaptureOutcome(capturedOnly); + const feature = record( + (record(record(input.archives[0]).plan).features as unknown[])[0], + ); + feature.checks = [ + { command: localCommand, intent: "pass", platform: "darwin" }, + ]; + expect(deliveryIssues(input, expected)).toContain( + "Claimed command pass lacks matching accepted complete source evidence.", + ); +}); +test("declared named assertions need passed native recorded assertion evidence", () => { + const input = deferredCaptureOutcome(capturedOnly); + const evidence = record( + (record(record(input.archives[0]).plan).evidence as unknown[])[0], + ); + evidence.assertions = ["null-handling"]; + expect(deliveryIssues(input, expected)).toContain( + "Claimed command pass lacks matching accepted complete source evidence.", + ); +}); +test("completed native independent review cannot be denied in the final summary", () => { + const input = autoQualifiedOutcome("single", { + goal: completed.goal, + featureId: completed.featureId, + }); + record( + record(call(input, "flow_session_close").output).workflowData, + ).delivery = structuredClone(completed.delivery); + const finalText = completed.answer.replace( + "passed with no findings", + "was not performed", + ); + expect( + deliveryIssues( + { ...input, finalText }, + { + closure: "completed", + presentation: "summary", + gate: localCommand, + allowedPaths: ["src/parser.mjs"], + }, + ), + ).toContain( + "Independent review claim contradicts accepted native review evidence.", + ); +}); +test("declared named assertion passes with matching native marker, status and archived assertion", () => { + const input = deferredCaptureOutcome(capturedOnly); + record( + (record(record(input.archives[0]).plan).evidence as unknown[])[0], + ).assertions = ["null-handling"]; + const assertions = [{ name: "null-handling", status: "passed" }]; + validation(input).observedAssertions = assertions; + statusValidation(input).observedAssertions = structuredClone(assertions); + Object.assign(call(input, "bash"), { + rawOutput: call(input, "bash").rawOutput.replace( + '"recordedRevision":4', + `"assertions":${JSON.stringify(assertions)},"recordedRevision":4`, + ), + }); + expect(deliveryIssues(input, expected)).toEqual([]); +}); +test("contradictory later native status cannot be rescued by an earlier matching snapshot", () => { + const input = deferredCaptureOutcome(capturedOnly); + const contradictory = structuredClone(call(input, "flow_status")); + if (contradictory.native) + Object.assign(contradictory.native, { + messageId: "msg_conflicting_status", + partId: "prt_conflicting_status", + callId: "call_conflicting_status", + startedAt: 26, + completedAt: 27, + }); + const projection = record( + record(record(contradictory.output).workflowData).projection, + ); + record( + (record((projection.runs as unknown[])[0]).validations as unknown[])[0], + ).sourceDigest = `sha256:${"b".repeat(64)}`; + const at = input.allCalls.indexOf(call(input, "flow_session_close")); + const allCalls = [ + ...input.allCalls.slice(0, at), + contradictory, + ...input.allCalls.slice(at), + ]; + expect( + deliveryIssues( + { ...input, allCalls, hostTrace: nativeTrace(allCalls) }, + expected, + ), + ).toContain( + "Claimed command pass lacks matching accepted complete source evidence.", + ); +}); +test("prior failed same-command validation does not erase a distinct attested passing capture", () => { + const input = deferredCaptureOutcome(capturedOnly); + const archiveRun = record((record(input.archives[0]).runs as unknown[])[0]); + const earlier = { + ...validation(input), + id: "earlier-failed-capture", + exitCode: 1, + recordedRevision: 3, + }; + archiveRun.startedRevision = 2; + (archiveRun.validations as unknown[]).push(earlier); + const projection = record( + record(record(call(input, "flow_status").output).workflowData).projection, + ); + const nativeRun = record((projection.runs as unknown[])[0]); + nativeRun.startedRevision = 2; + (nativeRun.validations as unknown[]).push(structuredClone(earlier)); + expect(deliveryIssues(input, expected)).toEqual([]); +}); +test("unmet platform proof does not depend on the native route hint", () => { + const input = deferredCaptureOutcome(); + record( + record(record(call(input, "flow_status").output).workflowData).projection, + ).nextAction = "provide-required-evidence"; + expect(deliveryIssues(input, expected)).toEqual([]); +}); +function reviewedAdvisoryOutcome() { + const input = autoQualifiedOutcome("single", { + goal: completed.goal, + featureId: completed.featureId, + }); + const run = record((record(input.archives[0]).runs as unknown[])[0]); + const findings = [ + { + findingId: "parser-null.R5-01", + severity: "advisory", + summary: "Consider broader parser documentation.", + }, + ]; + record(record((run.reviews as unknown[])[0]).result).findings = findings; + record( + record(call(input, "flow_feature_complete").input.request).result, + ).findings = structuredClone(findings); + const delivery = structuredClone(completed.delivery); + Object.assign(delivery, { findingsDigest: [{ ...findings[0], live: true }] }); + record( + record(call(input, "flow_session_close").output).workflowData, + ).delivery = delivery; + return input; +} +const completedExpected = { + closure: "completed" as const, + presentation: "summary" as const, + gate: localCommand, + allowedPaths: ["src/parser.mjs"], +}; +test("accepted independent review may pass with a nonblocking advisory", () => { + expect( + deliveryIssues( + { + ...reviewedAdvisoryOutcome(), + finalText: completed.answer.replace( + "passed with no findings", + "passed", + ), + }, + completedExpected, + ), + ).toEqual([]); +}); +test("review no-findings qualifier cannot conceal a native accepted advisory", () => { + expect( + deliveryIssues( + { ...reviewedAdvisoryOutcome(), finalText: completed.answer }, + completedExpected, + ), + ).not.toEqual([]); +}); +test("cleared historical review findings do not contradict a current review with no findings", () => { + const input = autoQualifiedOutcome("single", { + goal: completed.goal, + featureId: completed.featureId, + }); + const run = record((record(input.archives[0]).runs as unknown[])[0]); + const prior = structuredClone(run); + Object.assign(prior, { id: "prior-run", state: "superseded" }); + const finding = { + findingId: "parser-null.prior-advisory", + severity: "advisory", + summary: "Earlier parser documentation suggestion.", + }; + const priorReview = record((prior.reviews as unknown[])[0]); + Object.assign(priorReview, { id: "prior-review", runId: "prior-run" }); + record(priorReview.result).findings = [finding]; + const priorValidation = record((prior.validations as unknown[])[0]); + Object.assign(priorValidation, { + id: "prior-validation", + runId: "prior-run", + }); + priorReview.validationIds = ["prior-validation"]; + (record(input.archives[0]).runs as unknown[]).unshift(prior); + const delivery = structuredClone(completed.delivery); + Object.assign(delivery, { findingsDigest: [{ ...finding, live: false }] }); + record( + record(call(input, "flow_session_close").output).workflowData, + ).delivery = delivery; + expect( + deliveryIssues( + { ...input, finalText: completed.answer }, + completedExpected, + ), + ).toEqual([]); +}); +test("two distinct attested passing captures support the same past-tense command outcome", () => { + const input = deferredCaptureOutcome(capturedOnly); + const second = { + ...validation(input), + id: "second-passing-capture", + recordedRevision: 5, + }; + const run = record((record(input.archives[0]).runs as unknown[])[0]); + (run.validations as unknown[]).unshift(second); + const arm = structuredClone(call(input, "flow_validation_start")); + record(arm.input.request).expectedRevision = 4; + record(record(record(arm.output).workflowData).capture).captureId = second.id; + Object.assign(arm, { rawOutput: JSON.stringify(arm.output) }); + const bash = structuredClone(call(input, "bash")); + const output = bash.rawOutput + .replace("validation-0", second.id) + .replace('"recordedRevision":4', '"recordedRevision":5'); + Object.assign(bash, { output, rawOutput: output }); + const status = structuredClone(call(input, "flow_status")); + const projection = record( + record(record(status.output).workflowData).projection, + ); + Object.assign(projection, { + revision: 5, + runs: [{ ...structuredClone(run), state: "active" }], + }); + Object.assign(status, { rawOutput: JSON.stringify(status.output) }); + for (const [index, next] of [arm, bash, status].entries()) { + if (next.native) + Object.assign(next.native, { + messageId: `msg_second${index}`, + partId: `prt_second${index}`, + callId: `call_second${index}`, + startedAt: 30 + index * 2, + completedAt: 31 + index * 2, + }); + } + const close = call(input, "flow_session_close"); + record(input.archives[0]).revision = 6; + record(record(input.archives[0]).closure).recordedRevision = 6; + record(close.input.request).expectedRevision = 5; + record(record(record(close.output).workflowData).operation).revision = 6; + const at = input.allCalls.indexOf(close); + const allCalls = [ + ...input.allCalls.slice(0, at), + arm, + bash, + status, + ...input.allCalls.slice(at), + ]; + expect( + deliveryIssues( + { ...input, allCalls, hostTrace: nativeTrace(allCalls) }, + expected, + ), + ).toEqual([]); +}); +for (const prefix of ["", "Required external evidence: "]) { + test(`qualified unavailable command is independent of heading ${prefix}`, () => { + const text = noOptionalClaims.replace( + "- **Progress:**", + `${prefix}${externalCommand} on macOS; unavailable on this Linux host.\n- **Progress:**`, + ); + expect(deliveryIssues(deferredCaptureOutcome(text), expected)).toEqual([]); + }); +} +for (const suffix of [ + "on Windows; unavailable on this Linux host", + "on macOS; unavailable on this Windows host", + "on Linux; unavailable on this Linux host", + "on macOS; unavailable on this Linux host if it passes", + "on macOS; unavailable on this Linux host and passed with exit code 0", +]) { + test(`command availability refuses ${suffix}`, () => { + const text = noOptionalClaims.replace( + "- **Progress:**", + `- Outstanding proof: ${externalCommand} ${suffix}.\n- **Progress:**`, + ); + expect(deliveryIssues(deferredCaptureOutcome(text), expected)).not.toEqual( + [], + ); + }); +} +for (const object of [ + "the current Flow session", + "the Flow session", + "the current workflow", +]) { + test(`archived lifecycle object composes ${object}`, () => { + expect( + deliveryIssues( + deferredCaptureOutcome( + noOptionalClaims.replace("the Flow session", object), + ), + expected, + ), + ).toEqual([]); + }); +} +for (const object of [ + "another Flow session", + "the previous Flow session", + "the current Flow session if validation passes", + "not the current Flow session", +]) { + test(`archived lifecycle object refuses ${object}`, () => { + expect( + deliveryIssues( + deferredCaptureOutcome( + noOptionalClaims.replace("the Flow session", object), + ), + expected, + ), + ).not.toEqual([]); + }); +} diff --git a/tests/fixtures/deferred-capture-outcome.ts b/tests/fixtures/deferred-capture-outcome.ts new file mode 100644 index 00000000..01dd9b5d --- /dev/null +++ b/tests/fixtures/deferred-capture-outcome.ts @@ -0,0 +1,273 @@ +import type { ScenarioGradeInput } from "../../evals/grader-input.js"; +import { collectHostTrace } from "../../evals/host-trace.js"; +import { autoQualifiedOutcome } from "./auto-qualified-outcome.js"; + +export const localCommand = "node scripts/verify.mjs"; +export const externalCommand = "node scripts/platform-check.mjs"; +export const deferredGoal = + "Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs with Linux validation and required macOS proof."; +export const deferredAnswer = `**Deferred and archived the current Flow session.** The parser change is retained, and required macOS evidence remains **unfulfilled**. + +- **Linux validation:** \`${localCommand}\` passed with exit code 0. +- **Outstanding proof:** \`${externalCommand}\` on macOS; unavailable on this Linux host. +- **Progress:** 0 of 1 features complete; unfinished feature: \`parser-null\`. +- **Assurance:** Completion is not claimed. Independent review was not performed. +- **Findings:** No live advisory or historical findings. +- **External action authority:** Not granted. + +Goal: ${deferredGoal} + +Flow’s archived handoff retains these assurance limitations: +- Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness. +- Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments. +- Freshness holds when review is accepted; an archive does not attest the current workspace. + +Flow recorded no artifact declarations; that is not an exhaustive Git-change inventory.`; + +export function record(value: unknown): Record { + if (!value || typeof value !== "object" || Array.isArray(value)) + throw new Error("Missing fixture record."); + return value as Record; +} +export function nativeTrace(calls: ScenarioGradeInput["allCalls"]) { + return collectHostTrace({ + runnerRootSessionIds: ["ses_root"], + directory: "/workspace", + childrenComplete: true, + sessionMetadata: [{ id: "ses_root", directory: "/workspace" }], + sessionMessages: [ + { + sessionId: "ses_root", + messages: [ + { + info: { + id: "msg_user", + sessionID: "ses_root", + role: "user", + time: { created: 1 }, + }, + parts: [ + { + id: "prt_user", + sessionID: "ses_root", + messageID: "msg_user", + type: "text", + text: "Complete the approved task", + }, + ], + }, + ...calls.map((call) => ({ + info: { + id: call.native?.messageId, + sessionID: "ses_root", + role: "assistant", + parentID: "msg_user", + agent: call.agent, + time: { created: call.native?.startedAt }, + }, + parts: [ + { + id: call.native?.partId, + sessionID: "ses_root", + messageID: call.native?.messageId, + callID: call.native?.callId, + type: "tool", + tool: call.tool, + state: { + status: call.status, + time: { + start: call.native?.startedAt, + end: call.native?.completedAt, + }, + }, + }, + ], + })), + ], + }, + ], + }); +} +export function validation(input: ScenarioGradeInput) { + const run = record((record(input.archives[0]).runs as unknown[])[0]); + return record((run.validations as unknown[])[0]); +} +export function deferredCaptureOutcome( + finalText = deferredAnswer, +): ScenarioGradeInput { + const base = autoQualifiedOutcome("single", { + goal: deferredGoal, + featureId: "parser-null", + }); + const archive = record(base.archives[0]); + const run = record((archive.runs as unknown[])[0]); + const capture = validation(base); + Object.assign(capture, { recordedRevision: 4 }); + Object.assign(run, { + startedRevision: 3, + state: "superseded", + reviews: [], + artifactsChanged: [], + }); + Object.assign(archive, { + revision: 5, + closure: { + kind: "deferred", + operationId: "close-task", + recordedRevision: 5, + summary: "Required macOS proof remains unavailable.", + }, + }); + const plan = record(archive.plan); + plan.evidence = [ + { scope: "gate", command: localCommand, platform: "linux", assertions: [] }, + { + scope: "extra", + command: externalCommand, + platform: "darwin", + assertions: [], + }, + ]; + const close = base.allCalls.find( + (call) => call.tool === "flow_session_close", + ); + if (!close) throw new Error("Missing fixture close."); + Object.assign(record(close.input.request), { + kind: "deferred", + expectedRevision: 4, + }); + const closeData = record(record(close.output).workflowData); + Object.assign(record(closeData.operation), { + revision: 5, + entity: archive.closure, + }); + closeData.delivery = { + findingsDigest: [], + assurance: { + conclusion: "completion-not-claimed", + checks: Array.from({ length: 4 }, () => ({ status: "not-applicable" })), + }, + report: [ + "Handoff format: 1", + "External action authority: not-granted", + `Goal: ${deferredGoal}`, + "Closure: deferred", + "Progress: 0 of 1 features complete", + "Unfinished features: parser-null", + "Findings digest: none", + "Reported artifacts: 0 latest, 0 superseded", + "Assurance: completion not claimed", + "Features:", + "- parser-null: Safely parse null and trim strings", + " attempts: 1; latest state: superseded", + " outcome: none recorded", + " terminal findings: none", + "Assurance checks:", + "- not-applicable [TS-enforced] Recorded completion: deferred closure makes no completion claim.", + "- not-applicable [host-attested] Accepted validation: deferred closure makes no completion claim.", + "- not-applicable [host-attested] Canonical gate: deferred closure makes no completion claim.", + "- not-applicable [host-attested] Declared evidence: deferred closure makes no completion claim.", + "Assurance limitations:", + "- Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.", + "- Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.", + "- Freshness holds when review is accepted; an archive does not attest the current workspace.", + "Artifacts as reported by Flow from caller declarations, not an exact or exhaustive Git delta:", + "- latest attempts: none reported", + "- superseded attempts only: none reported", + `Recorded local validation: ${JSON.stringify(capture)}`, + `Required external evidence: ${JSON.stringify(plan.evidence)}`, + ], + }; + const manager = base.allCalls.filter( + (call) => call.agent === "build" && call.tool !== "flow_review_start", + ); + const append = ( + tool: string, + input: Record, + output: unknown, + startedAt: number, + metadata: Record = {}, + ) => ({ + tool, + status: "completed" as const, + agent: "build", + sessionIndex: 0, + input, + output, + rawOutput: typeof output === "string" ? output : JSON.stringify(output), + metadata, + native: { + sessionId: "ses_root", + messageId: `msg_capture${startedAt}`, + partId: `prt_capture${startedAt}`, + partIndex: 0, + callId: `call_capture${startedAt}`, + startedAt, + completedAt: startedAt + 1, + }, + }); + const marker = `[flow-validation] ${JSON.stringify({ id: capture.id, scope: "broad", intent: "pass", passed: true, observed: false, recordedRevision: 4 })}`; + const arm = append( + "flow_validation_start", + { + request: { + featureId: "parser-null", + scope: "broad", + command: localCommand, + expectedRevision: 3, + }, + }, + { + status: "ok", + workflowData: { + capture: { captureId: capture.id, expiresInMs: 900000 }, + command: localCommand, + scope: "broad", + intent: "pass", + }, + }, + 20, + ); + const bash = append( + "bash", + { command: localCommand }, + `(no output)\n\n${marker}`, + 22, + { exit: 0, truncated: false, output: "(no output)" }, + ); + const status = append( + "flow_status", + { request: { view: "detail" } }, + { + status: "ok", + workflowData: { + projection: { + sessionId: archive.id, + view: "detail", + revision: 4, + status: "running", + plan: structuredClone(plan), + runs: [{ ...structuredClone(run), state: "active" }], + progress: { completed: 0, total: 1, remaining: 1 }, + nextAction: "await-user-direction", + }, + }, + }, + 24, + ); + const calls = [ + ...manager.filter((call) => call !== close), + arm, + bash, + status, + close, + ]; + return { + ...base, + allCalls: calls, + flowCalls: calls.filter((call) => call.tool.startsWith("flow_")), + packetBytes: [], + hostTrace: nativeTrace(calls), + finalText, + }; +}