From d0f69670db2aba8ddae992594edacc5504e39fbf Mon Sep 17 00:00:00 2001 From: Dotta <34892728+cryppadotta@users.noreply.github.com> Date: Thu, 8 Oct 2026 05:57:20 -0500 Subject: [PATCH] fix(runner): recover saved execution prompts after upgrades (#15518) ## Thinking Path > - Paperclip is the open source app people use to manage AI agents for work. > - Native runs save an immutable execution context for restart recovery. > - The context includes the prompt text, revision, and content hashes. > - The parser required that saved prompt to match the current release. > - A server upgrade could reject a valid saved run before provider recovery. > - This pull request validates and preserves the saved prompt snapshot. > - Routine prompt changes no longer need a catalog of past strings. ## Linked Issues or Issue Description **What happened?** A hot restart selected a dead native runner for same-run recovery. Reading its saved v5 execution input failed with `input.runtimeContext.prompt must match the fixed Paperclip prompt revision`. The new controller accepted only v6. **Expected behavior** Recovery uses the saved prompt and validates its content hashes. It preserves the same run and provider session without starting a duplicate turn. **Steps to reproduce** 1. Start a native run and save its execution input and provider checkpoint. 2. Change the fixed execution prompt in the server release. 3. Stop the runner and recover the saved run with the new controller. 4. Observe that the old parser rejects the saved prompt before provider recovery. **Paperclip version or commit** The v5-to-v6 prompt change was introduced in #15446. The defect also reproduces on current master before this fix. **Deployment mode** Source-built server with the native runner. Related work: #15446 added task-monitor guidance. The held prompt-size experiment in #15489 changes prompt wording but does not add recovery compatibility. ## What Changed - Read the prompt text and revision from the saved execution snapshot. - Treat the revision as non-empty metadata and preserve the exact saved bytes. - Validate the prompt SHA-256 and the aggregate context digest. - Keep fresh-run builders on the current prompt constants. - Test arbitrary saved prompts, malformed fields, altered text, stale hashes, and aggregate drift. - Test recovery parsing for Codex input versions v3-v5 and OpenCode, ACPX Pi, and Dot v6 inputs. - Extend the real-process restart suite with both the incident's v5 wire fixture and a prompt unknown to this release. - Run the restart recovery suite in the existing Rust-equipped PR lane, where its runner and fake-provider binaries are built. Verify complete, non-overlapping test coverage for PR, release, and local callers. - Document recovery from saved snapshots without a historical prompt catalog. ## Verification - Red: the new contract regressions fail against the catalog-based parser with the original prompt-validation error. - Green: 53 focused contract and materialization tests pass. - Red: the real-process unknown-prompt regression fails with master's original parser at the saved-input recovery read after process loss. - Green: all 15 real-process restart tests pass locally on the final branch. The saved-prompt cases keep the run and provider session, replace the PID, and record one `turn/start`. - Local repository `pnpm -r typecheck` and `pnpm build` passed after rebase on `89f09dad723766e5351953f0731b9aa5daada28d`. The 53 focused tests also passed on that head. - Red: the new test-roster checks fail against the old CI placement. - Green: all 26 test-scheduling checks pass after moving the restart suite. - CI ran all 15 restart recovery tests with no skips on final head `6719fc2bb7a31a0f72ea04c7e525a63dcc6f9105`. [Runner test job](https://github.com/paperclipai/paperclip/actions/runs/37764782334/job/113271464966). - Greptile scored 5/5 on that exact head with no actionable findings. - The complete CI matrix passed on final head `6719fc2bb7a31a0f72ea04c7e525a63dcc6f9105`: general/workspace tests, serialized server suites, both runner Vitest lanes, Rust and static checks, all browser shards, typecheck, build, and the canary dry run. [CI run](https://github.com/paperclipai/paperclip/actions/runs/37764782334). - There are no unresolved review threads or merge conflicts. - Reproduce focused tests with `pnpm --filter @paperclipai/paperclip-runner exec vitest run src/contracts/runtime-context.test.ts src/contracts/native-execution.test.ts src/drivers/runtime-context-materializer.test.ts`. - Reproduce restart tests with `pnpm --filter @paperclipai/paperclip-runner build:rust` followed by `pnpm exec vitest run server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts`. They use temporary PostgreSQL, real runner processes, and a fake Codex provider. They do not use paid inference. ## Risks - The parser now accepts internally consistent saved prompt text that is absent from the current source. Inputs must come from trusted server persistence. Content hashes verify consistency; they do not authenticate authorship. - Existing execution-schema, ownership, checkpoint, provider, permission, and session-compatibility checks still apply. - This change validates the saved base prompt. It does not make all additional code-generated instruction strings versioned. - The process-level recovery proof uses Codex. Other provider coverage verifies the shared input parser and retained provider configuration. ## Model Used OpenAI Codex, GPT-6 family, with repository inspection, code editing, and test tools. The exact serving model ID and context-window size are not exposed in this session. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [x] All Paperclip CI gates are green - [x] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip --- doc/DEVELOPING.md | 8 ++ .../src/contracts/native-execution.test.ts | 65 ++++++++++- .../src/contracts/runtime-context.test.ts | 107 ++++++++++++++++++ .../src/contracts/runtime-context.ts | 14 ++- .../run-vitest-stable-shard.test.mjs | 9 +- scripts/run-vitest-stable.mjs | 4 +- ...unner-restart-recovery.integration.test.ts | 56 ++++++++- 7 files changed, 250 insertions(+), 13 deletions(-) create mode 100644 packages/paperclip-runner/src/contracts/runtime-context.test.ts diff --git a/doc/DEVELOPING.md b/doc/DEVELOPING.md index 43d0b7e67c..4cd8de8d3b 100644 --- a/doc/DEVELOPING.md +++ b/doc/DEVELOPING.md @@ -1374,6 +1374,14 @@ registers a correlated recovery request before it signals the dev supervisor. An uncoordinated server restart uses the same durable recovery classifier without trusting a handoff marker. +Recovery retains each run's saved execution prompt, revision, and context +digest across server upgrades. The shared parser validates the saved prompt's +SHA-256 and aggregate context digest. The revision is non-empty metadata; it +does not need to match the current release or a catalog of past prompts. New +runs use the current prompt. Do not rewrite saved execution inputs to the latest +prompt. Existing execution-schema, ownership, checkpoint, and permission checks +still determine whether recovery can proceed. + Startup binds the HTTP and PRP listener before it classifies native runs. Public health reports a startup state until every candidate is reattached, dispatched for same-run resume, finalized from durable evidence, or held for explicit diff --git a/packages/paperclip-runner/src/contracts/native-execution.test.ts b/packages/paperclip-runner/src/contracts/native-execution.test.ts index 111e705b5d..0a28f977e4 100644 --- a/packages/paperclip-runner/src/contracts/native-execution.test.ts +++ b/packages/paperclip-runner/src/contracts/native-execution.test.ts @@ -1,7 +1,7 @@ import { createHash } from "node:crypto"; import { describe, expect, it } from "vitest"; -import { buildNativeModelEnvelope, parseNativeExecutionInput, NATIVE_EXECUTION_INPUT_SCHEMA, type NativeExecutionInputV1 } from "./native-execution.js"; +import { buildNativeModelEnvelope, parseNativeExecutionInput, NATIVE_EXECUTION_INPUT_SCHEMA, NATIVE_EXECUTION_INPUT_SCHEMA_V6, type NativeExecutionInputV1 } from "./native-execution.js"; import { NATIVE_RUNTIME_ASSET_SCHEMA, PAPERCLIP_EXECUTION_PROMPT, @@ -56,6 +56,34 @@ const input: NativeExecutionInputV1 = { }; describe("NativeExecutionInputV1", () => { + it.each(["paperclip.native-execution-input.v3", "paperclip.native-execution-input.v4", NATIVE_EXECUTION_INPUT_SCHEMA])( + "preserves a saved %s execution through recovery parsing", + (schema) => { + const digest = "0".repeat(64); + const text = "Saved system instructions absent from the current release."; + const context = { + prompt: { revision: "saved-prompt-before-upgrade", text, digest: createHash("sha256").update(text).digest("hex") }, + instructions: { + entryPath: "AGENTS.md", + bundle: { schema: NATIVE_RUNTIME_ASSET_SCHEMA, digest, manifestDigest: digest, rootPath: "/runtime/instructions", fileCount: 1, totalBytes: 42 }, + }, + skills: [], + mcp: { assignmentSetId: "none", digest, bindingId: null }, + }; + const persisted = JSON.parse(JSON.stringify({ + ...input, + schema, + provider: schema === "paperclip.native-execution-input.v3" ? input.provider : { ...input.provider, approvalPolicy: "never" }, + executionMode: "default", + planningContext: null, + runtimeContext: { ...context, aggregateDigest: canonicalNativeRuntimeContextDigest(context) }, + })); + const recovered = parseNativeExecutionInput(persisted); + expect(recovered.runtimeContext).toEqual(persisted.runtimeContext); + expect(parseNativeExecutionInput(recovered)).toEqual(recovered); + }, + ); + it("parses v3 immutable runtime context without changing the model task envelope", () => { const digest = "0".repeat(64); const context = { @@ -484,6 +512,41 @@ describe("native task context ownership", () => { }); } + it.each([ + { driverKind: "opencode_server", provider: { kind: "opencode", model: "openrouter/deepseek/deepseek-v4-flash-0731", permissionMode: "deny" } }, + { driverKind: "acpx_runtime", provider: { + kind: "acpx", agent: "pi", model: "openrouter/deepseek/deepseek-v4-flash-0731", permissionMode: "deny-all", + profile: { driverKind: "acpx_runtime", protocolVersion: 1, acpxVersion: "0.13.1", agent: "pi", agentProfileVersion: 1, + agentServerPackage: "pi-acp", agentServerVersion: "0.0.33", agentRuntimePackage: "@earendil-works/pi-coding-agent", + agentRuntimeVersion: "0.84.2", commandDigest: `sha256:${"a".repeat(64)}` }, + } }, + { driverKind: "openai_dot_mcp", provider: { + kind: "openai_dot", model: null, + binding: { bindingId: "dot-binding", bindingGeneration: 1, companyId: input.binding.companyId, agentId: input.binding.agentId, + acceptByUnixMs: 1_000, expiresAtUnixMs: 2_000 }, + } }, + ])("preserves an unregistered saved prompt for $provider.kind", ({ driverKind, provider }) => { + const current = currentInput(); + if (!("runtimeContext" in current)) throw new Error("Expected a runtime context"); + const text = "Saved instructions absent from the current release."; + const context = { ...current.runtimeContext, + prompt: { revision: "saved-prompt-before-upgrade", text, digest: createHash("sha256").update(text).digest("hex") } }; + context.aggregateDigest = canonicalNativeRuntimeContextDigest(context); + const persisted = JSON.parse(JSON.stringify({ + ...current, provider, session: { ...current.session, driverKind }, runtimeContext: context, + ...(provider.kind === "openai_dot" ? { + schema: NATIVE_EXECUTION_INPUT_SCHEMA_V6, + workspace: { access: "none", cwd: null, repoUrl: null, repoRef: null, branchName: null }, + credentialBindings: [], + } : {}), + })); + const recovered = parseNativeExecutionInput(persisted); + expect(recovered.runtimeContext).toEqual(context); + expect(recovered.provider).toEqual(provider); + expect(recovered.session.driverKind).toBe(driverKind); + expect(parseNativeExecutionInput(recovered)).toEqual(recovered); + }); + it("carries an opaque provider mode without a vendor restriction and fences obsolete field names", () => { const current = currentInput(); const provider = { diff --git a/packages/paperclip-runner/src/contracts/runtime-context.test.ts b/packages/paperclip-runner/src/contracts/runtime-context.test.ts new file mode 100644 index 0000000000..09a931317c --- /dev/null +++ b/packages/paperclip-runner/src/contracts/runtime-context.test.ts @@ -0,0 +1,107 @@ +import { createHash } from "node:crypto"; +import { describe, expect, it } from "vitest"; +import { + NATIVE_RUNTIME_ASSET_SCHEMA, + PAPERCLIP_EXECUTION_PROMPT, + PAPERCLIP_EXECUTION_PROMPT_REVISION, + canonicalNativeRuntimeContextDigest, + composeNativeSystemInstructions, + parseNativeRuntimeContext, +} from "./runtime-context.js"; + +const hash = (text: string) => createHash("sha256").update(text).digest("hex"); + +function savedContext(revision: string, text: string) { + const digest = "0".repeat(64); + const context = { + prompt: { revision, text, digest: hash(text) }, + instructions: { + entryPath: "AGENTS.md", + bundle: { schema: NATIVE_RUNTIME_ASSET_SCHEMA, digest, manifestDigest: digest, rootPath: "/runtime/instructions", fileCount: 1, totalBytes: 42 }, + }, + skills: [], + mcp: { assignmentSetId: "none", digest, bindingId: "native-mcp:run-1" }, + connectionInstructions: { text: "Saved connection instructions", digest: hash("Saved connection instructions") }, + }; + // This is persisted wire data, deliberately created independently of the parser. + return { + ...context, + aggregateDigest: hash(JSON.stringify({ + prompt: context.prompt, + instructions: { entryPath: context.instructions.entryPath, bundleDigest: digest }, + skills: [], + mcp: { assignmentSetId: "none", digest }, + connectionInstructions: context.connectionInstructions, + })), + }; +} + +describe("persisted native execution prompts", () => { + it.each([ + ["snapshot-before-upgrade", "Saved instructions from before the upgrade."], + ["future-prompt-fixture", " Saved instructions absent from this release.\n"], + [PAPERCLIP_EXECUTION_PROMPT_REVISION, "Saved text under a revision reused by a later release."], + ["__proto__", "The revision is opaque metadata."], + ])( + "recovers %s without a prompt catalog or rewriting its saved instructions", + (revision, text) => { + const persisted = JSON.parse(JSON.stringify(savedContext(revision, text))); + const parsed = parseNativeRuntimeContext(persisted); + expect(parsed).toEqual(persisted); + expect(canonicalNativeRuntimeContextDigest(parsed)).toBe(persisted.aggregateDigest); + expect(parseNativeRuntimeContext(parsed)).toEqual(persisted); + expect(composeNativeSystemInstructions(parsed, "Agent instructions")).toBe( + `${text}\n\nAgent instructions\n\nSaved connection instructions\n\nRead-only instruction sibling root: /runtime/instructions`, + ); + }, + ); + + it("keeps the current prompt valid for new executions", () => { + const current = savedContext(PAPERCLIP_EXECUTION_PROMPT_REVISION, PAPERCLIP_EXECUTION_PROMPT); + expect(parseNativeRuntimeContext(current)).toEqual(current); + }); + + it.each(["revision", "text"])("requires a non-empty prompt %s", (field) => { + const persisted = savedContext("saved-prompt", "Saved instructions."); + for (const value of [undefined, null, 42, {}, "", " \n"]) { + expect(() => parseNativeRuntimeContext({ + ...persisted, + prompt: { ...persisted.prompt, [field]: value }, + })).toThrow(`input.runtimeContext.prompt.${field} must be a non-empty string`); + } + }); + + it("rejects altered text and malformed or mismatched prompt hashes", () => { + const persisted = savedContext("saved-prompt", "Saved instructions."); + expect(() => parseNativeRuntimeContext({ + ...persisted, prompt: { ...persisted.prompt, text: "Changed instructions." }, + })).toThrow("prompt.digest does not match prompt text"); + for (const digest of [undefined, null, "", "not-a-hash", "f".repeat(64)]) { + expect(() => parseNativeRuntimeContext({ + ...persisted, prompt: { ...persisted.prompt, digest }, + })).toThrow("input.runtimeContext.prompt.digest"); + } + }); + + it("rejects a prompt change with a matching text hash but a stale aggregate digest", () => { + const persisted = savedContext("saved-prompt", "Saved instructions."); + const changed = savedContext("changed-prompt", "Changed instructions."); + expect(() => parseNativeRuntimeContext({ + ...persisted, prompt: changed.prompt, + })).toThrow("aggregateDigest does not match the canonical context"); + }); + + it("includes the saved revision in the aggregate digest", () => { + const persisted = savedContext("saved-prompt", "Saved instructions."); + expect(() => parseNativeRuntimeContext({ + ...persisted, prompt: { ...persisted.prompt, revision: "changed-revision" }, + })).toThrow("aggregateDigest does not match the canonical context"); + }); + + it("rejects aggregate context drift", () => { + const persisted = savedContext("saved-prompt", "Saved instructions."); + expect(() => parseNativeRuntimeContext({ + ...persisted, aggregateDigest: "f".repeat(64), + })).toThrow("aggregateDigest does not match the canonical context"); + }); +}); diff --git a/packages/paperclip-runner/src/contracts/runtime-context.ts b/packages/paperclip-runner/src/contracts/runtime-context.ts index fd17acf09f..1d0127e50d 100644 --- a/packages/paperclip-runner/src/contracts/runtime-context.ts +++ b/packages/paperclip-runner/src/contracts/runtime-context.ts @@ -14,7 +14,7 @@ export interface NativeRuntimeAssetReference { } export interface NativeRuntimeContextSnapshot { - prompt: { revision: typeof PAPERCLIP_EXECUTION_PROMPT_REVISION; text: typeof PAPERCLIP_EXECUTION_PROMPT; digest: string }; + prompt: { revision: string; text: string; digest: string }; instructions: { entryPath: string; bundle: NativeRuntimeAssetReference; @@ -103,10 +103,12 @@ export function parseNativeRuntimeContext(value: unknown): NativeRuntimeContextS exact(context, ["prompt", "instructions", "skills", "mcp", "connectionInstructions", "aggregateDigest"], "input.runtimeContext"); const prompt = object(context.prompt, "input.runtimeContext.prompt"); exact(prompt, ["revision", "text", "digest"], "input.runtimeContext.prompt"); - if (prompt.revision !== PAPERCLIP_EXECUTION_PROMPT_REVISION || prompt.text !== PAPERCLIP_EXECUTION_PROMPT) { - throw new NativeRuntimeContextError("input.runtimeContext.prompt must match the fixed Paperclip prompt revision"); - } - if (digest(prompt.digest, "input.runtimeContext.prompt.digest") !== nativeRuntimePromptDigest()) { + // New runs use the current constants. Recovery uses the immutable saved + // snapshot; its revision is metadata, not a lookup in this release's source. + const promptRevision = text(prompt.revision, "input.runtimeContext.prompt.revision"); + const promptText = text(prompt.text, "input.runtimeContext.prompt.text"); + const promptDigest = digest(prompt.digest, "input.runtimeContext.prompt.digest"); + if (sha256(promptText) !== promptDigest) { throw new NativeRuntimeContextError("input.runtimeContext.prompt.digest does not match prompt text"); } const instructions = object(context.instructions, "input.runtimeContext.instructions"); @@ -140,7 +142,7 @@ export function parseNativeRuntimeContext(value: unknown): NativeRuntimeContextS connectionInstructions = { text: content, digest: contentDigest }; } const parsed = { - prompt: { revision: PAPERCLIP_EXECUTION_PROMPT_REVISION, text: PAPERCLIP_EXECUTION_PROMPT, digest: nativeRuntimePromptDigest() }, + prompt: { revision: promptRevision, text: promptText, digest: promptDigest }, instructions: { entryPath: safeRelativePath(instructions.entryPath, "input.runtimeContext.instructions.entryPath"), bundle: parseAsset(instructions.bundle, "input.runtimeContext.instructions.bundle"), diff --git a/scripts/__tests__/run-vitest-stable-shard.test.mjs b/scripts/__tests__/run-vitest-stable-shard.test.mjs index 04b8f19f64..d4fcb7a0d5 100644 --- a/scripts/__tests__/run-vitest-stable-shard.test.mjs +++ b/scripts/__tests__/run-vitest-stable-shard.test.mjs @@ -294,6 +294,8 @@ const chatSuitePath = "server/src/__tests__/chat-channels.integration.test.ts"; const nativeRunnerSuitePath = "server/src/services/native-runtime/native-codex-runner.integration.test.ts"; const dotRunnerSuitePath = "server/src/__tests__/dot-runner.test.ts"; +const restartRecoverySuitePath = + "server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts"; // Mirrors pr-trusted.yml (12 shards, called by pr.yml so GITHUB_WORKFLOW is // "PR"): the chat suite runs in its dedicated lanes and the cargo-dependent @@ -310,7 +312,8 @@ test("12 PR without-chat shards plus the dedicated chat and native-runner lanes assert.ok(!files.includes(chatSuitePath)); assert.ok(!files.includes(nativeRunnerSuitePath)); assert.ok(!files.includes(dotRunnerSuitePath)); - assert.deepEqual([...files, chatSuitePath, nativeRunnerSuitePath, dotRunnerSuitePath].sort(), full.selectedGeneralServerSuites.sort()); + assert.ok(!files.includes(restartRecoverySuitePath)); + assert.deepEqual([...files, chatSuitePath, nativeRunnerSuitePath, dotRunnerSuitePath, restartRecoverySuitePath].sort(), full.selectedGeneralServerSuites.sort()); assert.equal(new Set(files).size, files.length); const defaultRun = dryRunJson([], prEnv); assert.ok(defaultRun.generalServerSuiteCount === full.generalServerSuiteCount); @@ -330,16 +333,18 @@ for (const [caller, envOverrides] of [["Release", { GITHUB_WORKFLOW: "Release" } assert.ok(!files.includes(chatSuitePath)); assert.ok(files.includes(nativeRunnerSuitePath)); assert.ok(files.includes(dotRunnerSuitePath)); + assert.ok(files.includes(restartRecoverySuitePath)); assert.deepEqual([...files, chatSuitePath].sort(), full.selectedGeneralServerSuites.sort()); assert.equal(new Set(files).size, files.length); }); } -test("the native-runner lane runs exactly the cargo-dependent vertical-slice suite", () => { +test("the native-runner lane runs exactly the cargo-dependent suites", () => { const lane = dryRunJson(["--mode", "general", "--group", "general-server-native-runner"]); assert.deepEqual(lane.selectedGeneralServerSuites, [ "server/src/services/native-runtime/native-codex-runner.integration.test.ts", dotRunnerSuitePath, + restartRecoverySuitePath, ]); }); diff --git a/scripts/run-vitest-stable.mjs b/scripts/run-vitest-stable.mjs index 2580cd6384..1d98b848af 100644 --- a/scripts/run-vitest-stable.mjs +++ b/scripts/run-vitest-stable.mjs @@ -79,13 +79,15 @@ const generalServerWithoutChatGroupName = "general-server-without-chat"; const generalChatGroupName = "general-chat"; const generalServerNativeRunnerGroupName = "general-server-native-runner"; const chatSuite = "server/src/__tests__/chat-channels.integration.test.ts"; -// This suite rebuilds the Runner release binaries with cargo in beforeAll. +// The first suite rebuilds the Runner release binaries with cargo in beforeAll. // Inside the PR workflow's plain server shards, which carry no Rust cache, // that build was a ~4m30s cold compile of every third-party crate on each run // (277s of a 291s shard vitest step, actions run 35246999382, 2026-09-17). const nativeRunnerSuites = [ "server/src/services/native-runtime/native-codex-runner.integration.test.ts", "server/src/__tests__/dot-runner.test.ts", + // These process cases otherwise skip in server shards without Runner binaries. + "server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts", ]; // In the PR workflow (pr.yml, the caller of pr-trusted.yml — reusable // workflows inherit the caller's GITHUB_WORKFLOW), the last Verify Paperclip diff --git a/server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts b/server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts index c73db0b848..6e5a37f437 100644 --- a/server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts +++ b/server/src/services/native-runtime/native-runner-restart-recovery.integration.test.ts @@ -21,8 +21,10 @@ import { import { afterAll, beforeAll, describe, expect, it, vi } from "vitest"; import { + canonicalNativeRuntimeContextDigest, createRunnerdCodexTransport, defaultCapabilityRunnerdBinary, + parseNativeExecutionInput, } from "../../vendor/paperclip-runner/index.js"; import { getEmbeddedPostgresTestSupport, @@ -35,6 +37,8 @@ import { } from "../../realtime/runner-prp-ws.js"; import { readProcessStartedAt } from "../hot-restart.js"; import { prepareNativeHeartbeatRun } from "./prepare-native-run.js"; +import { buildNativeExecutionInput } from "./native-execution-input.js"; +import { nativeRuntimeContextFixture } from "./runtime-context.test-fixture.js"; const mockCaptureRunFailure = vi.hoisted(() => vi.fn()); vi.mock("../../sentry.js", async () => { @@ -50,6 +54,20 @@ import { type NativeControllerIdentity, } from "./native-restart-recovery.js"; +// Exact v5 wire prompt from before the task-monitor upgrade (#15446). +const historicalPrompt = { + revision: "paperclip-execution.v5", + text: "You are running as a Paperclip agent. Complete the assigned task in the provided execution environment. Follow the attached agent instructions and use assigned skills and tools when relevant. Use Paperclip tools for coordination. To hire or reuse a persistent teammate, use list_agents, then search_api for agent-hires and call_api if a hire is needed. Provider helper threads do not create Paperclip agents. When the user assigns work or a revision to a teammate, use create_task with that agent's ID; review their result rather than doing their assigned work yourself. When remaining work depends on a child task, use set_dependencies to add its ID while preserving existing blocker IDs. Complete independent work, then call paperclip_block with the child agent as owner and child completion as the unblock action. End the turn so the child can use the workspace. Do not sleep or poll for child results while holding the workspace. Paperclip resumes the parent when the dependency completes. When the user asks to connect a service, call connections_search before any service tool, even when that tool is already installed. Follow the returned instruction and wait for any required user choice before executing. For other tasks needing a service, use installed tools if available; otherwise use connections_search and follow its instruction. The request appears as a card in the task. Finish independent work before yielding for access; do not poll or request the same connection repeatedly. Paperclip will continue automatically with updated tools after resolution. After a decline, pursue alternatives unless the user explicitly asks to retry. Finish exactly once with `paperclip_finish` or `paperclip_block`.", + digest: "bec3e633d8d828103ce50b3a8b8dc9991c8ef65e07538bdab1a7663957c2ef9b", +} as const; + +// A saved prompt unknown to this release must recover without adding source data. +const unregisteredPrompt = { + revision: "saved-prompt-before-upgrade", + text: "Complete the assigned task using the saved instructions.", + digest: "b0cab3306694028624bbccbf40984a7aca7c809f77b29bfd959325a2af3f7a66", +} as const; + const embeddedPostgresSupport = await getEmbeddedPostgresTestSupport(); const describeEmbeddedPostgres = embeddedPostgresSupport.supported ? describe @@ -564,8 +582,29 @@ describeEmbeddedPostgres("native runner restart recovery with real processes", ( } }, 45_000); - realProcessIt("hard-restarts a dead runner on the same run and provider session", async () => { - const fixture = await seedRun("DEAD"); + realProcessIt.each([ + ["hard", "current"], + ["hot", "historical"], + ["hot", "unregistered"], + ] as const)("%s-restarts a dead runner with a %s prompt on the same run and provider session", async (restartKind, promptKind) => { + const fixture = await seedRun(`DEAD-${promptKind}`); + const context = nativeRuntimeContextFixture(); + const runtimeContext = { ...context, prompt: promptKind === "current" + ? context.prompt + : promptKind === "historical" ? historicalPrompt : unregisteredPrompt }; + runtimeContext.aggregateDigest = canonicalNativeRuntimeContextDigest(runtimeContext); + const execution = { ...buildNativeExecutionInput({ + companyId, + runId: fixture.runId, + issue: { id: fixture.issueId, identifier: `NRR-DEAD-${promptKind}`, title: "Recover a dead runner", description: null, workMode: "standard" }, + taskPrompt: "Resume this active turn after process loss.", + agentId, + workspace: { id: fixture.runId, cwd: tmpdir(), repoUrl: null, repoRef: null, branchName: null }, + normalizedSessionId: fixture.native.normalizedSessionId, + provider: "codex", + completionContract: { id: randomUUID(), sha256: `sha256:${"a".repeat(64)}`, schemaVersion: "paperclip.run-result.v1", contract: { revision: "1", objective: "Recover the same turn", criteria: [{ id: "objective", requirement: "Resume without a duplicate turn" }] } }, + runtimeContext: context, + }), runtimeContext }; const stateDirectory = resolve(runtimeRoot, fixture.runId); const baseOptions = transportOptions(fixture, stateDirectory); const options = { @@ -580,6 +619,7 @@ describeEmbeddedPostgres("native runner restart recovery with real processes", ( const started = await first.transport.request("thread/start", { cwd: tmpdir(), dynamicTools: [], + baseInstructions: runtimeContext.prompt.text, }); const thread = started.thread as Record; const turnStarted = await first.transport.request("turn/start", { @@ -598,6 +638,10 @@ describeEmbeddedPostgres("native runner restart recovery with real processes", ( providerPid, providerSessionId: String(thread.id), }); + const [persistedRun] = await fixture.db.select().from(heartbeatRuns).where(eq(heartbeatRuns.id, fixture.runId)); + await fixture.db.update(heartbeatRuns).set({ + runnerProfileJson: { ...persistedRun.runnerProfileJson, nativeExecutionInput: execution }, + }).where(eq(heartbeatRuns.id, fixture.runId)); await first.detachControllerForRestart(); await stopOwnedProcessGroup(firstRunnerPid, stateDirectory); @@ -609,7 +653,7 @@ describeEmbeddedPostgres("native runner restart recovery with real processes", ( const [claim] = await claimNativeRestartRecoveries({ db: fixture.db, controller: successor, - restartKind: "hard", + restartKind, now: new Date(), runIds: [fixture.runId], }); @@ -626,6 +670,12 @@ describeEmbeddedPostgres("native runner restart recovery with real processes", ( const [stoppedRun] = await fixture.db.select().from(heartbeatRuns).where(eq(heartbeatRuns.id, fixture.runId)); expect(stoppedRun.processPid).toBeNull(); expect(stoppedRun.processGroupId).toBeNull(); + // executeRun reads this immutable input before resuming after a restart. + const recoveredExecution = parseNativeExecutionInput(stoppedRun.runnerProfileJson?.nativeExecutionInput); + if (!("runtimeContext" in recoveredExecution)) throw new Error("Expected a pinned runtime context"); + expect(recoveredExecution.binding).toEqual(execution.binding); + expect(recoveredExecution.session).toEqual(execution.session); + expect(recoveredExecution.runtimeContext).toEqual(runtimeContext); restored = createRunnerdCodexTransport({ ...options,