diff --git a/evals/delivery-presentation.ts b/evals/delivery-presentation.ts index c5e97022..bd8d1e04 100644 --- a/evals/delivery-presentation.ts +++ b/evals/delivery-presentation.ts @@ -339,6 +339,7 @@ function sentenceBoundary(text: string, start = 0) { return { end: text.length, unterminatedQuote: quote !== null }; } function commandResultValue(line: string, commands: readonly string[]) { + if (/^Example:/i.test(line)) return null; for (const command of [...commands].sort((a, b) => b.length - a.length)) { const escaped = command.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); const prefix = new RegExp( @@ -464,8 +465,9 @@ function parseCommandResult( /^(?:(passed)(?:,\s*| with )|(recorded as an observation),\s*)?(?:exited|exit(?: code)?)\s+(-?\d+|unavailable)(.*)$/i.exec( status, ); - if (!value) return invalid; - const rawExit = value[3] ?? ""; + const barePass = /^passed$/i.test(status); + if (!value && !barePass) return invalid; + const rawExit = barePass ? "0" : (value?.[3] ?? ""); const exitCode = rawExit.toLowerCase() === "unavailable" ? null : Number(rawExit); if (exitCode !== null && !Number.isSafeInteger(exitCode)) return invalid; @@ -475,7 +477,7 @@ function parseCommandResult( qualification: null, integrity: "not-claimed", }; - const metadata = value[4] ?? ""; + const metadata = value?.[4] ?? ""; if ( metadata && !/^(?:, (?:host|source|output|report) .+|, reporting \d+ [A-Za-z ]+)$/i.test( @@ -489,7 +491,8 @@ function parseCommandResult( | "observation" | "does-not-claim-pass" | "claimed-pass" - | null = value[1] ? "claimed-pass" : value[2] ? "observation" : null; + | null = + barePass || value?.[1] ? "claimed-pass" : value?.[2] ? "observation" : null; let integrity: CommandIntegrity = "not-claimed"; for (const qualifier of parts) { if ( @@ -584,9 +587,11 @@ export function currentHandoffFacts( line = line.slice(0, boundary.end); } if ( - /^(?:[^:]+:\s*)?(?:node|bun) \S+[^;]*\bpassed(?:,\s*| with )exit(?: code)? -?\d+\b/i.test( + !/^Example:/i.test(line) && + (/^(?:[^:]+:\s*)?(?:node|bun) \S+[^;]*\bpassed(?:,\s*| with )exit(?: code)? -?\d+\b/i.test( line, - ) + ) || + /^(?:[^:]+:\s*)?(?:node|bun) \S+[^;]*\s+passed(?:[.;]|$)/i.test(line)) ) { facts.unsupported.push(line); } diff --git a/tests/delivery-bare-pass.test.ts b/tests/delivery-bare-pass.test.ts new file mode 100644 index 00000000..6f2c5b4c --- /dev/null +++ b/tests/delivery-bare-pass.test.ts @@ -0,0 +1,168 @@ +import { expect, test } from "bun:test"; +import { currentHandoffFacts } from "../evals/delivery-presentation.js"; +import { DELIVERY_SCENARIOS } from "../evals/delivery-scenarios.js"; +import { autoQualifiedOutcome } from "./fixtures/auto-qualified-outcome.js"; +import saved from "./fixtures/delivery-bare-pass-answer.json" with { + type: "json", +}; + +const gate = "node scripts/verify.mjs"; +const scenario = DELIVERY_SCENARIOS.find( + (item) => item.id === "delivery-summary-completed", +); +if (!scenario) throw new Error("Missing completed delivery scenario."); +function object(value: unknown): Record { + if (!value || typeof value !== "object" || Array.isArray(value)) + throw new Error("Missing native fixture object."); + return value as Record; +} +function fixture() { + const input = autoQualifiedOutcome("single", { + goal: saved.goal, + featureId: saved.featureId, + }); + const close = input.allCalls.find( + (call) => call.tool === "flow_session_close", + ); + object(object(close?.output).workflowData).delivery = structuredClone( + saved.delivery, + ); + return { ...input, finalText: saved.answer }; +} +test("native 4c7c answer with bare command pass succeeds through the whole completed scenario", () => { + expect(scenario.check(fixture())).toEqual([]); +}); +for (const change of [ + "nonzero", + "unknown-exit", + "incomplete", + "source", + "unaccepted", + "observe", +]) { + test(`bare pass cannot rescue ${change} native proof`, () => { + const input = fixture(); + const archive = object(input.archives[0]); + const runs = archive.runs; + if (!Array.isArray(runs)) throw new Error("Missing native runs."); + const run = object(runs[0]); + if (!Array.isArray(run.validations) || !Array.isArray(run.reviews)) + throw new Error("Missing native evidence."); + const validation = run.validations + .map(object) + .find((value) => value.command === gate); + if (!validation) throw new Error("Missing native gate."); + const review = object(run.reviews[0]); + if (change === "nonzero") validation.exitCode = 1; + if (change === "unknown-exit") validation.exitCode = null; + if (change === "incomplete") validation.outputComplete = false; + if (change === "source") + validation.sourceDigest = `sha256:${"b".repeat(64)}`; + if (change === "unaccepted") review.validationIds = []; + if (change === "observe") validation.intent = "observe"; + expect(scenario.check(input)).toContain( + "Claimed command pass lacks matching accepted complete source evidence.", + ); + }); +} +test("bare pass does not substitute for an omitted assurance limitation", () => { + const input = fixture(); + input.finalText = input.finalText.replace( + "Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.", + "", + ); + expect(scenario.check(input).length).toBeGreaterThan(0); +}); +for (const status of [ + "passed if it ran", + "passed and deployed", + "passed; this observation does not claim a pass", + "passed; exit unavailable", + "passed; exited 1", + "passed; recorded as an observation", + "passed; its script changed", +]) { + test(`bare pass refuses conflicting or unknown qualification ${status}`, () => { + const input = fixture(); + input.finalText = input.finalText.replace( + "`node scripts/verify.mjs` passed.", + `\`${gate}\` ${status}.`, + ); + expect(scenario.check(input)).toContain( + "Unsupported or conflicting current handoff assertions.", + ); + }); +} +test("unknown command cannot borrow the registered gate's bare pass proof", () => { + const input = fixture(); + input.finalText = input.finalText.replace( + "`node scripts/verify.mjs` passed.", + "`node scripts/other.mjs` passed.", + ); + expect(scenario.check(input)).toContain( + "Unsupported or conflicting current handoff assertions.", + ); +}); +for (const text of [ + `Goal: Preserve '${gate} passed.'`, + `Historical handoff\n${gate} passed.`, + `Example: ${gate} passed.`, + `Example: ${gate} passed with exit code 0.`, +]) { + test(`scoped text cannot supply a bare current command pass ${text}`, () => { + expect(currentHandoffFacts(text, [gate]).observations).toEqual([]); + }); +} +for (const qualifier of [ + "its script remained unchanged", + "its script and invocation are unchanged", + "it passed", +]) { + test(`bare pass retains the closed allowed qualifier ${qualifier}`, () => { + const input = fixture(); + input.finalText = input.finalText.replace( + "`node scripts/verify.mjs` passed.", + `\`${gate}\` passed; ${qualifier}.`, + ); + expect(scenario.check(input)).toEqual([]); + }); +} +test("bare command pass preserves quoted arguments as opaque command identity", () => { + const command = `${gate} --label "retained, and the Linux gate bun fake.mjs passed with exit code 9"`; + const facts = currentHandoffFacts(`${command} passed.`, [command]); + expect(facts.observations).toEqual([ + { + command, + exitCode: 0, + qualification: "claimed-pass", + integrity: "not-claimed", + }, + ]); + expect(facts.unsupported).toEqual([]); +}); +for (const runtime of ["Node version 24", "Bun version 1.4"]) { + test(`runtime compatibility prose remains ordinary context for ${runtime}`, () => { + expect( + currentHandoffFacts(`${runtime} passed our compatibility tests.`, [gate]) + .unsupported, + ).toEqual([]); + const input = fixture(); + input.finalText += `\n${runtime} passed our compatibility tests.`; + expect(scenario.check(input)).toEqual([]); + }); +} +for (const suffix of [ + "passed.", + "passed; its script remained unchanged.", + "passed; exited 1.", + "passed with exit code 0.", + "passed with exit code 0 and deployed.", +]) { + test(`unregistered command still rejects status ${suffix}`, () => { + const input = fixture(); + input.finalText += `\nValidation: node scripts/other.mjs ${suffix}`; + expect(scenario.check(input)).toContain( + "Unsupported or conflicting current handoff assertions.", + ); + }); +} diff --git a/tests/fixtures/delivery-bare-pass-answer.json b/tests/fixtures/delivery-bare-pass-answer.json new file mode 100644 index 00000000..15345f3c --- /dev/null +++ b/tests/fixtures/delivery-bare-pass-answer.json @@ -0,0 +1,118 @@ +{ + "answer": "Updated `src/parser.mjs` with a minimal null guard:\n\n```js\nexport function parse(input) { return input === null ? '' : input.trim(); }\n```\n\nValidation: `node scripts/verify.mjs` passed. Independent review passed with no findings.\n\n```text\nHandoff format: 1\nExternal action authority: not-granted\nGoal: Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.\nClosure: completed\nProgress: 1 of 1 features complete\nUnfinished features: none\nNonblocking live findings: advisory 0. Historical findings: 0.\nAssurance: completion supported\nAssurance checks: 4 satisfied, 0 not applicable, 0 unsatisfied.\n- Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.\n- Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.\n- Freshness holds when review is accepted; an archive does not attest the current workspace.\nReported artifacts: 1 latest, 0 superseded only. Caller declarations, not an exact or exhaustive Git delta.\nFull report is included in this close response.\n```", + "goal": "Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.", + "featureId": "parser-null", + "delivery": { + "assurance": { + "checks": [ + { + "explanation": "1/1 features and 1/1 independent reviews pass, including a final review with no terminal blocker.", + "id": "recorded-completion", + "label": "Recorded completion", + "status": "satisfied", + "tier": "ts-enforced" + }, + { + "explanation": "1/1 terminal runs have eligible host evidence accepted by review.", + "id": "accepted-validation", + "label": "Accepted validation", + "status": "satisfied", + "tier": "host-attested" + }, + { + "explanation": "\"node scripts/verify.mjs\" must have passing broad evidence accepted by review.", + "id": "canonical-gate", + "label": "Canonical gate", + "status": "satisfied", + "tier": "host-attested" + }, + { + "explanation": "1/1 declared obligations have accepted evidence on their declared host; observed gates do not claim a pass.", + "id": "declared-evidence", + "label": "Declared evidence", + "status": "satisfied", + "tier": "host-attested" + } + ], + "conclusion": "completion-supported", + "limitations": [ + "Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.", + "Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.", + "Freshness holds when review is accepted; an archive does not attest the current workspace." + ] + }, + "closure": { + "kind": "completed", + "summary": "Implemented minimal null guard in src/parser.mjs; unchanged canonical gate passed and independent review passed without findings." + }, + "features": [ + { + "attempts": 1, + "id": "parser-null", + "latestState": "completed", + "outcomeSummary": "Verified parser-null and all three approved requirements: null returns ''; strings retain trim behavior; only src/parser.mjs changed, with the verification script and command unchanged. Inspected current parser, gate, README and package surface, all bounded evidence pages, and complete base diff; no deletion, rename, type/mode change or loss of pre-existing work. Source-bound node scripts/verify.mjs validation exited 0 with complete output and tests both outcomes. Other input behavior remains unchanged. No prior findings or remaining gaps.", + "terminalFindings": [], + "title": "Handle null in parse" + } + ], + "findingsDigest": [], + "goal": "Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.", + "handoff": { + "externalActionAuthority": "not-granted", + "formatVersion": 1 + }, + "progress": { + "completed": 1, + "total": 1 + }, + "report": [ + "Handoff format: 1", + "External action authority: not-granted", + "Goal: Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.", + "Closure: completed \u2014 Implemented minimal null guard in src/parser.mjs; unchanged canonical gate passed and independent review passed without findings.", + "Progress: 1 of 1 features complete", + "Features:", + "- parser-null \u2014 Handle null in parse", + " attempts: 1; latest state: completed", + " outcome: Verified parser-null and all three approved requirements: null returns ''; strings retain trim behavior; only src/parser.mjs changed, with the verification script and command unchanged. Inspected current parser, gate, README and package surface, all bounded evidence pages, and complete base diff; no deletion, rename, type/mode change or loss of pre-existing work. Source-bound node scripts/verify.mjs validation exited 0 with complete output and tests both outcomes. Other input behavior remains unchanged. No prior findings or remaining gaps.", + " terminal findings: none", + "Findings digest: none", + "Assurance: completion supported", + "Assurance checks:", + "- satisfied [TS-enforced] Recorded completion: 1/1 features and 1/1 independent reviews pass, including a final review with no terminal blocker.", + "- satisfied [host-attested] Accepted validation: 1/1 terminal runs have eligible host evidence accepted by review.", + "- satisfied [host-attested] Canonical gate: \"node scripts/verify.mjs\" must have passing broad evidence accepted by review.", + "- satisfied [host-attested] Declared evidence: 1/1 declared obligations have accepted evidence on their declared host; observed gates do not claim a pass.", + "Assurance limitations:", + "- Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.", + "- Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.", + "- Freshness holds when review is accepted; an archive does not attest the current workspace.", + "Artifacts as reported by Flow from caller declarations, not an exact or exhaustive Git delta:", + "- latest attempts: src/parser.mjs", + "- superseded attempts only: none reported" + ], + "reportedArtifacts": { + "latestAttempts": ["src/parser.mjs"], + "supersededAttemptsOnly": [] + }, + "summary": { + "fullReportAvailable": true, + "lines": [ + "Handoff format: 1", + "External action authority: not-granted", + "Goal: Make parse(input) safely handle null and preserve trimmed strings, changing only src/parser.mjs and keeping node scripts/verify.mjs and its script unchanged.", + "Closure: completed", + "Progress: 1 of 1 features complete", + "Unfinished features: none", + "Nonblocking live findings: advisory 0. Historical findings: 0.", + "Assurance: completion supported", + "Assurance checks: 4 satisfied, 0 not applicable, 0 unsatisfied.", + "- Artifact paths and the canonical gate are caller declarations; Flow validates binding, not completeness or fitness.", + "- Goal alignment, scope discipline, evidence completeness, requirement coverage, test adequacy, and review substance remain model judgments.", + "- Freshness holds when review is accepted; an archive does not attest the current workspace.", + "Reported artifacts: 1 latest, 0 superseded only. Caller declarations, not an exact or exhaustive Git delta.", + "Full report is included in this close response." + ] + } + } +}