From a11bd236e33833dc081cc3702baa3d3f98d8d12f Mon Sep 17 00:00:00 2001 From: Dotta Date: Mon, 21 Sep 2026 16:51:28 -0500 Subject: [PATCH] fix: keep quarantined cleanup out of automatic recovery projection Also record the successful two-turn grounding qualification and semantic review. Co-Authored-By: Paperclip --- .../src/services/execution-projection.test.ts | 3 +++ server/src/services/execution-projection.ts | 1 + tests/runner-e2e/QUALIFICATION-2026-09-21.md | 27 +++++++++++++++++++ 3 files changed, 31 insertions(+) diff --git a/server/src/services/execution-projection.test.ts b/server/src/services/execution-projection.test.ts index a8f8d75c49..df55aa800c 100644 --- a/server/src/services/execution-projection.test.ts +++ b/server/src/services/execution-projection.test.ts @@ -44,6 +44,9 @@ describe("execution truth projection", () => { permittedActions: ["inspect_run", "inspect_recovery"], }); expect(projectExecution(stopped, coordinator({ phase: "retryable_failure" }), [], undefined, now).phase).toBe("recovery_needed"); + expect(projectExecution({ ...stopped, finishedAt: now }, coordinator({ + phase: "terminal_failure", failureCode: "native_provider_terminal_failed", + }), [], undefined, now)).toMatchObject({ phase: "recovery_needed", recoveryOwner: "board" }); expect(projectExecution({ ...stopped, errorCode: "provider_frame_too_large" }, undefined, [], undefined, now).phase).toBe("failed"); expect(projectExecution(stopped, undefined, [], { status: "resolved", cause: "native_session_cleanup_quarantined", nextAction: "Verify prior execution", evidence: { executionReconciliation: { providerStopped: true }, continuationRunId: "successor" } }, now)) diff --git a/server/src/services/execution-projection.ts b/server/src/services/execution-projection.ts index 74fa6ff717..cab39d73e3 100644 --- a/server/src/services/execution-projection.ts +++ b/server/src/services/execution-projection.ts @@ -260,6 +260,7 @@ export function projectExecution( } if (coordinator?.phase === "terminal_failure" || recoveryAction || cleanupQuarantined) { if ( + !cleanupQuarantined && coordinator?.failureCode === "native_provider_terminal_failed" && !detail.replacementDenied && run.finishedAt && diff --git a/tests/runner-e2e/QUALIFICATION-2026-09-21.md b/tests/runner-e2e/QUALIFICATION-2026-09-21.md index 7ed4da30e4..ba83b4b706 100644 --- a/tests/runner-e2e/QUALIFICATION-2026-09-21.md +++ b/tests/runner-e2e/QUALIFICATION-2026-09-21.md @@ -88,3 +88,30 @@ Two user-facing limitations remain worth deciding: Production instructions were preserved. These observations must not be hidden by changing prompts solely to make the benchmark green. + +## Grounded status answers: corrected live proof + +[Campaign 35658262695](https://github.com/paperclipai/paperclip/actions/runs/35658262695) +on `4a26f10be7dd6aeabf2ca7b44b3d6b2ece7817e0`: **2/2 passed**, both +cleanup passes. Each provider answered both turns, preserved both source tasks, +and started no execution on either task. The original failed attempts above are +retained; this is a new campaign with an explicit output contract. + +Semantic review of all four retained replies against the task descriptions and +chronological comments found: + +- Both identified venue confirmation as the current blocker, printing as deferred, + no task execution, and unknown attendance. Both gave the useful next step of + confirming the venue before printing. +- Both rejected the obsolete budget claim and unsupported printing claim on the + follow-up, and distinguished a planned Friday from a guaranteed calendar date. + Neither invented a venue, date, or attendance count. +- Codex's explanations were compact and clear. Claude's second explanation was + longer but readable and grounded in named task records. Its phrase “no venue has + been confirmed” is slightly stronger than “no confirmation is recorded”; its + immediately following quotation and null fact make the evidence limitation clear. + +This qualifies these two-turn grounding stories, not general answer quality, +statistical reliability, multilingual behavior, or arbitrary long conversations. +The five review dimensions remain factual grounding, stale-premise correction, +honest uncertainty, useful next step, and clear prose. No production prompt changed.