Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions evals/delivery-presentation.ts
Original file line number Diff line number Diff line change
Expand Up @@ -339,6 +339,22 @@ export function currentHandoffFacts(
for (const segment of line.split(/;|\.\s+(?=[A-Z])/)) {
const claim = segment.trim().replace(/\.$/, "");
if (!claim) continue;
const compound = observation
? null
: /^(.+?)\s+[—–]\s+(\d+\s*(?:of|\/)\s*\d+\s+features\b.*)$/i.exec(
claim,
);
if (compound) {
const closure = closureStatement(compound[1] ?? "");
const progress = progressValue(compound[2] ?? "");
facts.closure.push(closure?.closure ?? null);
facts.progress.push(progress?.progress ?? null);
if (!closure || !progress) facts.unsupported.push(claim);
if (closure?.unavailablePlatform)
facts.unavailableProofPlatforms.push(closure.unavailablePlatform);
if (progress) facts.auxiliaryCounts.push(...progress.counts);
continue;
}
const field =
/^(?:(?:current|Flow)\s+)?(closure|assurance|external[- ]action authority|progress):\s*(.*)$/i.exec(
claim,
Expand Down
141 changes: 141 additions & 0 deletions tests/delivery-compound-heading.test.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,141 @@
import { expect, test } from "bun:test";
import { currentHandoffFacts } from "../evals/delivery-presentation.js";
import { deliveryIssues } from "../evals/delivery-scenario-checks.js";
import { checkReviewerEvidenceAccess } from "../evals/reviewer-access.js";
import { autoQualifiedOutcome } from "./fixtures/auto-qualified-outcome.js";
import saved from "./fixtures/delivery-combined-heading-answer.json" with {
type: "json",
};

const goal = saved.goal;
const actualAnswer = saved.answer;
const heading = "**Completed and archived — 1 of 1 features complete.**";
const canonical = actualAnswer.replace(
heading,
"Closure: completed.\nProgress: 1 of 1 features complete.",
);
const expectation = {
closure: "completed" as const,
presentation: "summary" as const,
gate: "node scripts/verify.mjs",
allowedPaths: ["src/parser.mjs"],
};

function coherentNativeFixture() {
const input = autoQualifiedOutcome("single", {
goal,
featureId: saved.featureId,
});
const close = input.allCalls.find(
(call) => call.tool === "flow_session_close",
);
if (
!close ||
close.output === null ||
typeof close.output !== "object" ||
!("workflowData" in close.output)
)
throw new Error("Missing close fixture.");
const data = close.output.workflowData;
if (data === null || typeof data !== "object")
throw new Error("Missing close workflow data.");
Object.assign(data, { delivery: structuredClone(saved.delivery) });
expect(checkReviewerEvidenceAccess(input, input.archives[0])).toEqual([]);
expect(
deliveryIssues({ ...input, finalText: canonical }, expectation),
).toEqual([]);
return input;
}

test("actual measured answer preserves closure and progress in one bold heading", () => {
const input = coherentNativeFixture();
expect(
deliveryIssues({ ...input, finalText: actualAnswer }, expectation),
).toEqual([]);
});

test("compound heading consumes both complete scalar clauses without requiring separate labels", () => {
for (const value of [
heading,
"### Completed and archived – 1 / 1 features complete.",
]) {
const facts = currentHandoffFacts(value);
expect(facts.closure).toEqual(["completed"]);
expect(facts.progress).toEqual([{ completed: 1, total: 1 }]);
expect(facts.unsupported).toEqual([]);
}
});

test("compound headings cannot hide negation, contradictory facts or unsupported tails behind canonical facts", () => {
const correct = "Closure: completed.\nProgress: 1 of 1 features complete.";
for (const mutation of [
"Not completed and archived — 1 of 1 features complete.",
"Deferred and archived — 1 of 1 features complete.",
"Completed and archived — 0 of 1 features complete.",
"Completed and archived — 1 of 1 features not complete.",
"Completed and archived and deployed — 1 of 1 features complete.",
"Completed and archived — 1 of 1 features complete and deployed.",
"Unknown current state — 1 of 1 features complete.",
]) {
const facts = currentHandoffFacts(`${correct}\n**${mutation}**`);
expect(
facts.closure.every((value) => value === "completed") &&
facts.progress.every(
(value) => value?.completed === 1 && value.total === 1,
) &&
facts.unsupported.length === 0,
).toBe(false);
}
});

test("historical compound headings cannot supply current closure or progress", () => {
const historical = currentHandoffFacts(`## Historical handoff\n${heading}`);
expect(historical.closure).toEqual([]);
expect(historical.progress).toEqual([]);
const current = currentHandoffFacts(
`## Historical handoff\n${heading}\n## Current handoff\nClosure: deferred.\nProgress: 0 of 1 features complete.`,
);
expect(current.closure).toEqual(["deferred"]);
expect(current.progress).toEqual([{ completed: 0, total: 1 }]);
});

test("combined presentation does not supply a missing scalar fact", () => {
const input = coherentNativeFixture();
for (const incomplete of [
"**Completed and archived.**",
"**1 of 1 features complete.**",
]) {
expect(
deliveryIssues(
{ ...input, finalText: actualAnswer.replace(heading, incomplete) },
expectation,
),
).not.toEqual([]);
}
});

test("compound-looking bytes in goals and registered commands remain their original records", () => {
const goal =
'Keep "Completed and archived — 1 of 1 features complete" as a label.';
const goalFacts = currentHandoffFacts(`Goal: ${goal}`);
expect(goalFacts.goal).toEqual([goal]);
expect(goalFacts.closure).toEqual([]);
expect(goalFacts.progress).toEqual([]);
const command =
'node scripts/verify.mjs --label "Completed and archived — 1 of 1 features complete"';
const commandFacts = currentHandoffFacts(
`${command} passed with exit code 0.`,
[command],
);
expect(commandFacts.observations).toEqual([
{
command,
exitCode: 0,
unchangedInvocation: false,
qualification: "claimed-pass",
},
]);
expect(commandFacts.closure).toEqual([]);
expect(commandFacts.progress).toEqual([]);
expect(commandFacts.unsupported).toEqual([]);
});
72 changes: 72 additions & 0 deletions tests/fixtures/delivery-combined-heading-answer.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,72 @@
{
"answer": "Goal: Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.\n\n**Completed and archived \u2014 1 of 1 features complete.**\n\nUpdated `src/parser.mjs`:\n```js\nexport function parse(input) { return input === null ? '' : input.trim(); }\n```\n\n- **Validation:** `node scripts/verify.mjs` passed with exit code 0.\n- **Independent review:** passed with no findings.\n- **Scope:** only `src/parser.mjs` changed; the verifier remains unchanged.\n- **Unfinished features or blockers:** none.\n- **Assurance:** completion supported; all four Flow assurance checks satisfied.\n- **External action authority:** not granted.\n\nFlow assurance limitations:\n- Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.\n- Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.\n- Freshness holds when review is accepted; an archive does not attest the current workspace.",
"goal": "Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.",
"featureId": "parser-null",
"delivery": {
"report": [
"Handoff format: 1",
"External action authority: not-granted",
"Goal: Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.",
"Closure: completed \u2014 Implemented null-specific empty-string handling in src/parser.mjs, preserved string trimming, passed unchanged repository gate, and obtained independent review with no findings.",
"Progress: 1 of 1 features complete",
"Features:",
"- parser-null \u2014 Null-safe string parser",
" attempts: 1; latest state: completed",
" outcome: Verified parser-null and all approved requirements: null returns an empty string, strings retain input.trim(), and only src/parser.mjs changed with verifier and canonical command unchanged. Inspected current parser, verifier, README, package surface, and complete base diff; file mode/type and preexisting work preserved. Current-source broad canonical gate passed with complete output and matching source binding. No prior findings or remaining gaps.",
" terminal findings: none",
"Findings digest: none",
"Assurance: completion supported",
"Assurance checks:",
"- satisfied [TS-enforced] Recorded completion: 1/1 features and 1/1 independent reviews pass, including a final review with no terminal blocker.",
"- satisfied [host-attested] Accepted validation: 1/1 terminal runs have eligible host evidence accepted by review.",
"- satisfied [host-attested] Canonical gate: \"node scripts/verify.mjs\" must have passing broad evidence accepted by review.",
"- satisfied [host-attested] Declared evidence: 1/1 declared obligations have accepted evidence on their declared host; observed gates do not claim a pass.",
"Assurance limitations:",
"- Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.",
"- Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.",
"- Freshness holds when review is accepted; an archive does not attest the current workspace.",
"Artifacts as reported by Flow from caller declarations, not an exact or exhaustive Git delta:",
"- latest attempts: src/parser.mjs",
"- superseded attempts only: none reported"
],
"assurance": {
"checks": [
{
"explanation": "1/1 features and 1/1 independent reviews pass, including a final review with no terminal blocker.",
"id": "recorded-completion",
"label": "Recorded completion",
"status": "satisfied",
"tier": "ts-enforced"
},
{
"explanation": "1/1 terminal runs have eligible host evidence accepted by review.",
"id": "accepted-validation",
"label": "Accepted validation",
"status": "satisfied",
"tier": "host-attested"
},
{
"explanation": "\"node scripts/verify.mjs\" must have passing broad evidence accepted by review.",
"id": "canonical-gate",
"label": "Canonical gate",
"status": "satisfied",
"tier": "host-attested"
},
{
"explanation": "1/1 declared obligations have accepted evidence on their declared host; observed gates do not claim a pass.",
"id": "declared-evidence",
"label": "Declared evidence",
"status": "satisfied",
"tier": "host-attested"
}
],
"conclusion": "completion-supported",
"limitations": [
"Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.",
"Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.",
"Freshness holds when review is accepted; an archive does not attest the current workspace."
]
},
"findingsDigest": []
}
}
Loading