From 147e42f7e19e560797d0c8241704ae921eaaafab Mon Sep 17 00:00:00 2001 From: Dotta Date: Mon, 21 Sep 2026 17:07:57 -0500 Subject: [PATCH] fix: show inspection instead of retry for cleanup quarantine Exercise the recovery banner and wait for scheduled recovery to settle before testing user retry. Retain previous evidence and completed story results. Co-Authored-By: Paperclip --- tests/runner-e2e/QUALIFICATION-2026-09-21.md | 63 +++++++++++++++++++ tests/runner-e2e/catalog.ts | 2 +- tests/runner-e2e/chat-qualification.ts | 18 +++++- .../ExecutionBlockerNotice.test.tsx | 4 +- ui/src/components/ExecutionBlockerNotice.tsx | 3 +- 5 files changed, 84 insertions(+), 6 deletions(-) diff --git a/tests/runner-e2e/QUALIFICATION-2026-09-21.md b/tests/runner-e2e/QUALIFICATION-2026-09-21.md index ba83b4b706..154ff4f289 100644 --- a/tests/runner-e2e/QUALIFICATION-2026-09-21.md +++ b/tests/runner-e2e/QUALIFICATION-2026-09-21.md @@ -115,3 +115,66 @@ This qualifies these two-turn grounding stories, not general answer quality, statistical reliability, multilingual behavior, or arbitrary long conversations. The five review dimensions remain factual grounding, stale-premise correction, honest uncertainty, useful next step, and clear prose. No production prompt changed. + +## Active reassignment: corrected live proof + +[Campaign 35659014397](https://github.com/paperclipai/paperclip/actions/runs/35659014397) +on `cf6d4ae3a8576d822861bc916e7f536d0a0b7bc7`: **2/2 passed**, both +cleanup passes, using `gpt-5.6-sol` and `claude-sonnet-5`. + +The boundary snapshot proves the original worker was running with a saved draft. +Chat then reassigns the existing task to a second agent. The original stops with +`issue_reassigned` before the successor starts; exactly one successor completes +the same task. The plan, scope, and original draft revision survive. The successor +may revise the canonical document, and its final contribution must be attributed +to that successor. The audit records the native reassignment tool. + +The preceding [campaign 35657945095](https://github.com/paperclipai/paperclip/actions/runs/35657945095) +on `6039b02ed` remains failed: both handoffs actually completed, but the draft +oracle incorrectly required the latest document to remain frozen. Both successors +legitimately revised that document. The new oracle requires the exact original +revision to remain retrievable through the history API while allowing progress. +Negative calibration rejects deletion or alteration of that revision. Both crash +cases in the preceding campaign independently reproduced cleanup quarantine. + +## Measurement limits + +All selected environments here are local Linux GitHub Actions workers, not +Daytona-hosted execution or the user's staging company. The result JSON retains +source SHA, suite definition hash, profile/model, attempt, timing, token usage, +and billing coverage; initial failures are never regraded as passes. + +Onboarding reports complete billing for only 4/26 results; active handoff reports +incomplete billing for both interrupted workers; the two answer-quality results +report complete coverage. Provider-reported monetary cost is zero while billing +type is unknown, and local runtime is not metered. This does **not** establish +zero actual spend. Recorded total tokens (including cached input) are 9,812,045 +for onboarding, 825,131 for final handoffs, and 1,273,098 for grounded answers. + +Public campaign viewers use the standard +[Product E2E history](https://d1p6rlowie26tp.cloudfront.net/runner-e2e/) +with the `gha--` campaign identifier. These are bounded +story qualifications, not a claim that all native-runner reliability is solved. + +## Crash guard follow-up + +[Campaign 35658772755](https://github.com/paperclipai/paperclip/actions/runs/35658772755) +did not reach provider cases: GitHub artifact finalization returned HTTP 403. +A replacement [campaign 35659580100](https://github.com/paperclipai/paperclip/actions/runs/35659580100) +on `a11bd236e33833dc081cc3702baa3d3f98d8d12f` retained two failed recovery +attempts, both with successful disposable cleanup. The API correctly refused +Retry with 409 and created no second run, but the eval then waited for a run +that had never been admitted. + +Retained network evidence explains the transition: `provider_transport_failed` +first schedules a same-run retry; that attempt then reaches cleanup quarantine. +The fixture now waits for this recovery classification before requesting a user +retry. Screenshots also exposed a separate Retry button in the task-recovery +banner. That banner previously recognized only native continuation reconciliation; +it now also recognizes cleanup quarantine and links to Inspect run. Its regression +test failed before the fix and passes afterward. The live guard must verify this +link is rendered, no Retry remains, the API refuses retry, and saved work survives. + +These corrections do not supply a successful post-crash continuation. Worker-crash +recovery remains unqualified until an agreed recovery policy is implemented and +its successful outcome passes the eval. diff --git a/tests/runner-e2e/catalog.ts b/tests/runner-e2e/catalog.ts index f53649fae2..4f39bec921 100644 --- a/tests/runner-e2e/catalog.ts +++ b/tests/runner-e2e/catalog.ts @@ -975,7 +975,7 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [ profiles: runnerProfiles.filter(profile => ["runner-codex", "runner-acpx-claude"].includes(profile.id)) .map(profile => productionStoryProfile(defaultPermissionProfile(profile))), environments: [localEnvironment], tasks: chatQualificationTasks, expectedMatrixSize: 6, - definitionMetadata: { version: 5, permissions: "production-defaults", instructions: "production", crashBoundary: "verified-native-worker-pid-at-file-wait", recovery: "user-visible-retry", answerGrading: "exact-grounded-propositions-plus-separate-semantic-review", scheduling: "explicit-only" }, + definitionMetadata: { version: 6, permissions: "production-defaults", instructions: "production", crashBoundary: "verified-native-worker-pid-at-file-wait", recovery: "user-visible-retry", answerGrading: "exact-grounded-propositions-plus-separate-semantic-review", scheduling: "explicit-only" }, }, ...(process.env.PAPERCLIP_RUNNER_E2E_CONNECTION_REVIEWS === "1" ? [connectionReviewSuite] : []), { diff --git a/tests/runner-e2e/chat-qualification.ts b/tests/runner-e2e/chat-qualification.ts index 7bf5610c6b..102016ce82 100644 --- a/tests/runner-e2e/chat-qualification.ts +++ b/tests/runner-e2e/chat-qualification.ts @@ -151,7 +151,14 @@ export async function runWorkerCrash(context: Context) { await input.evidence("chat-worker-fault-delivered.json", fault); context.expectedStops.set(boundary.id, "failed"); await expect.poll(async () => (await input.api.get(`/api/heartbeat-runs/${boundary.id}`)).status, { timeout: 120_000 }).toBe("failed"); - const failed = await input.api.get(`/api/heartbeat-runs/${boundary.id}`); + // A failed transport may still have a scheduled same-run retry. Wait for + // recovery classification before treating it as available for a user Retry. + let failed: Row = {}; + await expect.poll(async () => { + failed = await input.api.get(`/api/heartbeat-runs/${boundary.id}`); + return ["failed", "recovery_needed"].includes(failed.execution?.phase); + }, { timeout: 120_000 }).toBe(true); + await input.evidence("chat-worker-settled-failure.json", failed); await input.capture("worker-failed", "Worker loss before user Retry", "worker-failed.png"); await writeFile(wait.gate, reference); if (failed.errorCode === "native_session_cleanup_quarantined") { @@ -159,6 +166,7 @@ export async function runWorkerCrash(context: Context) { // the UI/API must not offer an attempt which cannot pass cleanup admission. await input.page.reload({ waitUntil: "domcontentloaded" }); await expect(input.page.getByTestId("task-chat-composer-input")).toBeVisible(); + await expect(input.page.getByRole("status", { name: "Task recovery" }).getByRole("link", { name: "Inspect run" })).toBeVisible(); await expect(input.page.getByRole("button", { name: "Retry", exact: true })).toHaveCount(0); const refused = await input.api.request.post(`/api/agents/${input.fixtures.agent.id}/wakeup`, { data: { failedRunId: failed.id, reason: "retry_failed_run" }, @@ -173,7 +181,13 @@ export async function runWorkerCrash(context: Context) { } const retry = input.page.getByRole("button", { name: "Retry", exact: true }); await expect(retry).toHaveCount(1, { timeout: 60_000 }); - await retry.click(); + const [retryResponse] = await Promise.all([ + input.page.waitForResponse(response => response.request().method() === "POST" && + new URL(response.url()).pathname === `/api/agents/${input.fixtures.agent.id}/wakeup`), + retry.click(), + ]); + await input.evidence("chat-worker-retry-response.json", { status: retryResponse.status(), body: await retryResponse.json() }); + expect(retryResponse.ok(), "The visible Retry must admit an attempt before waiting for its result").toBe(true); await context.idle(2); const e = { boundary, failed, runs: await context.allRuns(), issueId: context.issue().id, prompt, comments: await context.comments(), reference, marker, planBefore, diff --git a/ui/src/components/ExecutionBlockerNotice.test.tsx b/ui/src/components/ExecutionBlockerNotice.test.tsx index 45c79bf0b4..3e8d42b420 100644 --- a/ui/src/components/ExecutionBlockerNotice.test.tsx +++ b/ui/src/components/ExecutionBlockerNotice.test.tsx @@ -49,11 +49,11 @@ describe("stopped task recovery notice", () => { expect(container.textContent).toContain("Verify the external action outcome before continuing."); expect(container.textContent).not.toContain("Automatic recovery of this task stopped."); }); - it("links to the source run instead of offering a retry rejected by native reconciliation", async () => { + it.each(["native_continuation_requires_reconciliation", "native_session_cleanup_quarantined"])("links to the source run instead of offering a retry rejected by %s", async (cause) => { await act(async () => root.render( )); diff --git a/ui/src/components/ExecutionBlockerNotice.tsx b/ui/src/components/ExecutionBlockerNotice.tsx index dd8356059e..527ef3c6c8 100644 --- a/ui/src/components/ExecutionBlockerNotice.tsx +++ b/ui/src/components/ExecutionBlockerNotice.tsx @@ -19,7 +19,8 @@ export function ExecutionBlockerNotice({ companyId, issueId, blocker, onRetried }); const failedRun = runs?.find(run => run.runId === blocker.runId && ["failed", "timed_out"].includes(run.status)); - const requiresInspection = blocker.cause === "native_continuation_requires_reconciliation"; + const requiresInspection = blocker.cause === "native_continuation_requires_reconciliation" || + blocker.cause === "native_session_cleanup_quarantined"; const retry = useMutation({ mutationFn: () => agentsApi.retryFailedRun(failedRun!.agentId, failedRun!.runId, companyId), onSuccess: () => {