diff --git a/tests/runner-e2e/QUALIFICATION-2026-09-21.md b/tests/runner-e2e/QUALIFICATION-2026-09-21.md index ba83b4b706..154ff4f289 100644 --- a/tests/runner-e2e/QUALIFICATION-2026-09-21.md +++ b/tests/runner-e2e/QUALIFICATION-2026-09-21.md @@ -115,3 +115,66 @@ This qualifies these two-turn grounding stories, not general answer quality, statistical reliability, multilingual behavior, or arbitrary long conversations. The five review dimensions remain factual grounding, stale-premise correction, honest uncertainty, useful next step, and clear prose. No production prompt changed. + +## Active reassignment: corrected live proof + +[Campaign 35659014397](https://github.com/paperclipai/paperclip/actions/runs/35659014397) +on `cf6d4ae3a8576d822861bc916e7f536d0a0b7bc7`: **2/2 passed**, both +cleanup passes, using `gpt-5.6-sol` and `claude-sonnet-5`. + +The boundary snapshot proves the original worker was running with a saved draft. +Chat then reassigns the existing task to a second agent. The original stops with +`issue_reassigned` before the successor starts; exactly one successor completes +the same task. The plan, scope, and original draft revision survive. The successor +may revise the canonical document, and its final contribution must be attributed +to that successor. The audit records the native reassignment tool. + +The preceding [campaign 35657945095](https://github.com/paperclipai/paperclip/actions/runs/35657945095) +on `6039b02ed` remains failed: both handoffs actually completed, but the draft +oracle incorrectly required the latest document to remain frozen. Both successors +legitimately revised that document. The new oracle requires the exact original +revision to remain retrievable through the history API while allowing progress. +Negative calibration rejects deletion or alteration of that revision. Both crash +cases in the preceding campaign independently reproduced cleanup quarantine. + +## Measurement limits + +All selected environments here are local Linux GitHub Actions workers, not +Daytona-hosted execution or the user's staging company. The result JSON retains +source SHA, suite definition hash, profile/model, attempt, timing, token usage, +and billing coverage; initial failures are never regraded as passes. + +Onboarding reports complete billing for only 4/26 results; active handoff reports +incomplete billing for both interrupted workers; the two answer-quality results +report complete coverage. Provider-reported monetary cost is zero while billing +type is unknown, and local runtime is not metered. This does **not** establish +zero actual spend. Recorded total tokens (including cached input) are 9,812,045 +for onboarding, 825,131 for final handoffs, and 1,273,098 for grounded answers. + +Public campaign viewers use the standard +[Product E2E history](https://d1p6rlowie26tp.cloudfront.net/runner-e2e/) +with the `gha--` campaign identifier. These are bounded +story qualifications, not a claim that all native-runner reliability is solved. + +## Crash guard follow-up + +[Campaign 35658772755](https://github.com/paperclipai/paperclip/actions/runs/35658772755) +did not reach provider cases: GitHub artifact finalization returned HTTP 403. +A replacement [campaign 35659580100](https://github.com/paperclipai/paperclip/actions/runs/35659580100) +on `a11bd236e33833dc081cc3702baa3d3f98d8d12f` retained two failed recovery +attempts, both with successful disposable cleanup. The API correctly refused +Retry with 409 and created no second run, but the eval then waited for a run +that had never been admitted. + +Retained network evidence explains the transition: `provider_transport_failed` +first schedules a same-run retry; that attempt then reaches cleanup quarantine. +The fixture now waits for this recovery classification before requesting a user +retry. Screenshots also exposed a separate Retry button in the task-recovery +banner. That banner previously recognized only native continuation reconciliation; +it now also recognizes cleanup quarantine and links to Inspect run. Its regression +test failed before the fix and passes afterward. The live guard must verify this +link is rendered, no Retry remains, the API refuses retry, and saved work survives. + +These corrections do not supply a successful post-crash continuation. Worker-crash +recovery remains unqualified until an agreed recovery policy is implemented and +its successful outcome passes the eval. diff --git a/tests/runner-e2e/catalog.ts b/tests/runner-e2e/catalog.ts index f53649fae2..4f39bec921 100644 --- a/tests/runner-e2e/catalog.ts +++ b/tests/runner-e2e/catalog.ts @@ -975,7 +975,7 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [ profiles: runnerProfiles.filter(profile => ["runner-codex", "runner-acpx-claude"].includes(profile.id)) .map(profile => productionStoryProfile(defaultPermissionProfile(profile))), environments: [localEnvironment], tasks: chatQualificationTasks, expectedMatrixSize: 6, - definitionMetadata: { version: 5, permissions: "production-defaults", instructions: "production", crashBoundary: "verified-native-worker-pid-at-file-wait", recovery: "user-visible-retry", answerGrading: "exact-grounded-propositions-plus-separate-semantic-review", scheduling: "explicit-only" }, + definitionMetadata: { version: 6, permissions: "production-defaults", instructions: "production", crashBoundary: "verified-native-worker-pid-at-file-wait", recovery: "user-visible-retry", answerGrading: "exact-grounded-propositions-plus-separate-semantic-review", scheduling: "explicit-only" }, }, ...(process.env.PAPERCLIP_RUNNER_E2E_CONNECTION_REVIEWS === "1" ? [connectionReviewSuite] : []), { diff --git a/tests/runner-e2e/chat-qualification.ts b/tests/runner-e2e/chat-qualification.ts index 7bf5610c6b..102016ce82 100644 --- a/tests/runner-e2e/chat-qualification.ts +++ b/tests/runner-e2e/chat-qualification.ts @@ -151,7 +151,14 @@ export async function runWorkerCrash(context: Context) { await input.evidence("chat-worker-fault-delivered.json", fault); context.expectedStops.set(boundary.id, "failed"); await expect.poll(async () => (await input.api.get(`/api/heartbeat-runs/${boundary.id}`)).status, { timeout: 120_000 }).toBe("failed"); - const failed = await input.api.get(`/api/heartbeat-runs/${boundary.id}`); + // A failed transport may still have a scheduled same-run retry. Wait for + // recovery classification before treating it as available for a user Retry. + let failed: Row = {}; + await expect.poll(async () => { + failed = await input.api.get(`/api/heartbeat-runs/${boundary.id}`); + return ["failed", "recovery_needed"].includes(failed.execution?.phase); + }, { timeout: 120_000 }).toBe(true); + await input.evidence("chat-worker-settled-failure.json", failed); await input.capture("worker-failed", "Worker loss before user Retry", "worker-failed.png"); await writeFile(wait.gate, reference); if (failed.errorCode === "native_session_cleanup_quarantined") { @@ -159,6 +166,7 @@ export async function runWorkerCrash(context: Context) { // the UI/API must not offer an attempt which cannot pass cleanup admission. await input.page.reload({ waitUntil: "domcontentloaded" }); await expect(input.page.getByTestId("task-chat-composer-input")).toBeVisible(); + await expect(input.page.getByRole("status", { name: "Task recovery" }).getByRole("link", { name: "Inspect run" })).toBeVisible(); await expect(input.page.getByRole("button", { name: "Retry", exact: true })).toHaveCount(0); const refused = await input.api.request.post(`/api/agents/${input.fixtures.agent.id}/wakeup`, { data: { failedRunId: failed.id, reason: "retry_failed_run" }, @@ -173,7 +181,13 @@ export async function runWorkerCrash(context: Context) { } const retry = input.page.getByRole("button", { name: "Retry", exact: true }); await expect(retry).toHaveCount(1, { timeout: 60_000 }); - await retry.click(); + const [retryResponse] = await Promise.all([ + input.page.waitForResponse(response => response.request().method() === "POST" && + new URL(response.url()).pathname === `/api/agents/${input.fixtures.agent.id}/wakeup`), + retry.click(), + ]); + await input.evidence("chat-worker-retry-response.json", { status: retryResponse.status(), body: await retryResponse.json() }); + expect(retryResponse.ok(), "The visible Retry must admit an attempt before waiting for its result").toBe(true); await context.idle(2); const e = { boundary, failed, runs: await context.allRuns(), issueId: context.issue().id, prompt, comments: await context.comments(), reference, marker, planBefore, diff --git a/ui/src/components/ExecutionBlockerNotice.test.tsx b/ui/src/components/ExecutionBlockerNotice.test.tsx index 45c79bf0b4..3e8d42b420 100644 --- a/ui/src/components/ExecutionBlockerNotice.test.tsx +++ b/ui/src/components/ExecutionBlockerNotice.test.tsx @@ -49,11 +49,11 @@ describe("stopped task recovery notice", () => { expect(container.textContent).toContain("Verify the external action outcome before continuing."); expect(container.textContent).not.toContain("Automatic recovery of this task stopped."); }); - it("links to the source run instead of offering a retry rejected by native reconciliation", async () => { + it.each(["native_continuation_requires_reconciliation", "native_session_cleanup_quarantined"])("links to the source run instead of offering a retry rejected by %s", async (cause) => { await act(async () => root.render( )); diff --git a/ui/src/components/ExecutionBlockerNotice.tsx b/ui/src/components/ExecutionBlockerNotice.tsx index dd8356059e..527ef3c6c8 100644 --- a/ui/src/components/ExecutionBlockerNotice.tsx +++ b/ui/src/components/ExecutionBlockerNotice.tsx @@ -19,7 +19,8 @@ export function ExecutionBlockerNotice({ companyId, issueId, blocker, onRetried }); const failedRun = runs?.find(run => run.runId === blocker.runId && ["failed", "timed_out"].includes(run.status)); - const requiresInspection = blocker.cause === "native_continuation_requires_reconciliation"; + const requiresInspection = blocker.cause === "native_continuation_requires_reconciliation" || + blocker.cause === "native_session_cleanup_quarantined"; const retry = useMutation({ mutationFn: () => agentsApi.retryFailedRun(failedRun!.agentId, failedRun!.runId, companyId), onSuccess: () => {