fix(sandbox): reset step store for long-lived bridge work

A run-time `sandbox.exec` span attached to a dead startup step. Two bridge
startup steps start long-lived work inside their measured step body: a poll
timer, socket handlers, and a callback-bridge worker loop. Node snapshots the
active-step store on each async resource at creation time, so this work kept
the ended step store. Each later run-time exec then read that store and opened
its span under the ended step, and it copied a wrong `criticalPath: false` flag.

Add an exported `runWithoutActiveStep` helper in startup-timing and wrap the
long-lived poll timer, the socket handlers, and the callback-bridge worker loop
of both bridge lanes. Each continuation now reads an empty store, so each
run-time exec opens an unparented span with no stale `criticalPath` flag.
Control flow is unchanged and the change adds no new dependency.

Co-authored-by: Paperclip <noreply@paperclip.ing>
This commit is contained in:
Priya RamanandPaperclip committed 2026-08-04 06:22:17 +00:00
1 parent bd7a13eb9c
commit dd6742f7d0
4 files changed
+241 -9

No files matched your search

@@ -10,6 +10,7 @@ vi.mock("../services/environment-config.js", () => ({
import {
measureStartupStep,
runWithoutActiveStep,
SANDBOX_STARTUP_SPAN_ATTRS,
} from "@paperclipai/adapter-utils/acpx-engine/startup-timing";
import {
@@ -847,4 +848,116 @@ describe("resolveEnvironmentExecutionTarget", () => {
// The seam reached the log callback exactly once (the stdout delivery).
expect(onLog).toHaveBeenCalledTimes(1);
});
// Fire one run-time exec from a bridge continuation that runs after the step
// span ended. Each bridge step (`bridge.paperclip`, `bridge.process-session`)
// starts long-lived work with `criticalPath: false`. The bridge boundary wraps
// that long-lived work in `runWithoutActiveStep`, exactly as modeled here, so
// the continuation reads an empty active step. Return the recorded exec span.
async function runContinuationExec(step: string, options: { wrap: boolean }) {
const { tracer, contextWithSpan, spans } = createRecordingTrace();
const runner = await runnerFor({
provider: "daytona",
execResult: { exitCode: 0, signal: null, timedOut: false, stdout: "", stderr: "" },
tracer,
});
let resolveExec!: () => void;
const execDone = new Promise<void>((resolve) => {
resolveExec = resolve;
});
// Schedule the exec from a timer inside the step body, so it fires after the
// step span ends. The `wrap` flag models the fix: when true, the boundary
// wraps the long-lived work in `runWithoutActiveStep`; when false, it models
// the pre-fix leak.
const scheduleContinuation = () => {
setTimeout(() => {
void runner.execute({ command: "echo" }).then(() => resolveExec());
}, 0);
};
await measureStartupStep(
{},
() => 0,
step,
async () => {
if (options.wrap) {
runWithoutActiveStep(scheduleContinuation);
} else {
scheduleContinuation();
}
return "started";
},
{ tracer, contextWithSpan, criticalPath: false },
);
await execDone;
return spans.find((span) => span.name === "sandbox.exec");
}
it("opens an unparented exec span for a process-session bridge continuation", async () => {
const execSpan = await runContinuationExec("bridge.process-session", { wrap: true });
expect(execSpan).toBeTruthy();
// The step span ended and the boundary emptied the store, so the continuation
// exec opens a root span, not one under the dead bridge step.
expect(execSpan!.parent).toBeNull();
});
it("opens an unparented exec span for a paperclip bridge continuation", async () => {
const execSpan = await runContinuationExec("bridge.paperclip", { wrap: true });
expect(execSpan).toBeTruthy();
expect(execSpan!.parent).toBeNull();
});
it("does not copy the stale criticalPath = false flag onto a continuation exec", async () => {
const execSpan = await runContinuationExec("bridge.process-session", { wrap: true });
expect(execSpan).toBeTruthy();
// The bridge step set `criticalPath: false`. The continuation reads an empty
// store, so the exec span records the default `true`, never the stale `false`.
expect(execSpan!.attributes[A.execCriticalPath]).toBe(true);
expect(execSpan!.attributes[A.execCriticalPath]).not.toBe(false);
});
it("leaks the ended step onto a continuation exec without the boundary wrap", async () => {
// The mechanism guard: an unwrapped continuation keeps the ended bridge step
// store, so the exec span parents to the dead step and copies its
// `criticalPath: false`. The boundary wrap in the two tests above removes both
// defects, so this suite fails if a future edit drops the wrap.
const { tracer, contextWithSpan, spans } = createRecordingTrace();
const runner = await runnerFor({
provider: "daytona",
execResult: { exitCode: 0, signal: null, timedOut: false, stdout: "", stderr: "" },
tracer,
});
let resolveExec!: () => void;
const execDone = new Promise<void>((resolve) => {
resolveExec = resolve;
});
await measureStartupStep(
{},
() => 0,
"bridge.process-session",
async () => {
setTimeout(() => {
void runner.execute({ command: "echo" }).then(() => resolveExec());
}, 0);
return "started";
},
{ tracer, contextWithSpan, criticalPath: false },
);
await execDone;
const stepSpan = spans.find((span) => span.name === "bridge.process-session");
const execSpan = spans.find((span) => span.name === "sandbox.exec");
expect(stepSpan).toBeTruthy();
expect(execSpan).toBeTruthy();
// The unwrapped continuation parents the exec span to the ended step and
// copies the stale flag.
expect(execSpan!.parent).toBe(stepSpan);
expect(execSpan!.attributes[A.execCriticalPath]).toBe(false);
});
});