Files
PaperClipAI/tests/runner-e2e/selectors.ts
T
Dotta bcc6fe7a44 fix(runner): restore multi-turn remote sessions (#12840)
## Thinking Path

> - Paperclip manages AI agents and their work.
> - The runner executes agent turns on local and remote providers.
> - A remote per-turn session must save its state before Paperclip
releases its sandbox.
> - The session runtime returned after 100 milliseconds while the remote
checkpoint still ran.
> - The next turn also checked the local state path instead of the
verified remote backup.
> - This pull request waits for the bounded remote close and accepts
only a verified suspended backup.
> - The benefit is reliable multi-turn execution without weaker identity
checks.

## Linked Issues or Issue Description

**What happened?**

A successful remote agent turn released its sandbox before the runner
saved the verified continuation backup. The next turn failed with
`runner_state_identity_mismatch`.

**Expected behavior**

Paperclip must finish the bounded remote checkpoint before it releases
the sandbox. A later turn must validate and restore the digest-matched
suspended backup.

**Steps to reproduce**

1. Run a native ACPX Claude Plan test in a non-reusable Daytona sandbox.
2. Reject the first plan to start a second turn.
3. Observe that the second turn fails before provider execution.

**Paperclip version or commit**

The failure reproduced at `13775a90b078ff64872f50961ea1b83d575e7bc6`.

**Deployment mode**

GitHub Actions with a Daytona sandbox.

## What Changed

- Wait for the internally bounded remote runner close and checkpoint
before the host returns.
- Preserve the existing short cleanup bound for other providers.
- Validate remote continuation lifecycle from a complete digest-verified
backup when local runner state is absent.
- Keep corrupt, non-suspended, mismatched, and unverified state
fail-closed.
- Make native Plan completion and accepted-Plan wake prompts
deterministic.

## Verification

- A prior 45-cell local campaign passed 44 cells. The only failure was
the OpenCode Plan prompt variance fixed here.
- A focused OpenCode local Plan rerun passed.
- ACPX Claude Daytona message and question cells passed.
- Focused regressions cover delayed checkpoint close and verified remote
backup lifecycle.
- GitHub Build and the focused ACPX Claude Daytona Plan cell will
validate this exact head.

## Risks

Remote runnerd sessions now wait for their internally bounded
close/checkpoint path before returning; generic provider cleanup retains
the existing 100 millisecond bound. Durable run success still cannot be
reversed. The environment release guard still blocks sandbox destruction
when no verified backup stamp exists.

## Model Used

OpenAI Codex, GPT-5.6, extended reasoning, with code execution and
GitHub Actions inspection.

## Checklist

- [x] I have included a thinking path that traces from project context
to this change
- [x] I have specified the model used (with version and capability
details)
- [x] I have checked ROADMAP.md and confirmed this PR does not duplicate
planned core work
- [x] I have searched GitHub for duplicate or related PRs and linked
them above
- [x] I have either (a) linked existing issues with `Fixes: #` / `Closes
#` / `Refs #` OR (b) described the issue in-PR following the relevant
issue template
- [x] I have not referenced internal/instance-local Paperclip issues or
links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip`
URLs)
- [x] My branch name describes the change and contains no internal task
id
- [ ] I have run tests locally and they pass
- [x] I have added or updated tests where applicable
- [ ] I have updated relevant documentation to reflect my changes
- [x] I have considered and documented any risks above
- [ ] All Paperclip CI gates are green
- [ ] Greptile is 5/5 with no open findings
- [ ] I will address all Greptile and reviewer comments before
requesting merge
2026-09-05 06:25:06 -05:00

213 lines
6.2 KiB
TypeScript

import { runnerMatrix } from "./catalog.js";
import type { MatrixExecution, MatrixJob } from "./types.js";
export interface RunnerSelectorOptions {
all: boolean;
list: boolean;
matrixJson: boolean;
ids: string[];
suites: string[];
groups: string[];
profiles: string[];
environments: string[];
cases: string[];
headed: boolean;
ui: boolean;
debug: boolean;
maxParallel: number;
}
export class RunnerSelectorError extends Error {}
function valueFor(args: string[], index: number, flag: string): string {
const value = args[index + 1];
if (!value || value.startsWith("--"))
throw new RunnerSelectorError(`${flag} requires a value`);
return value;
}
export function parseRunnerSelectors(
rawArgs: readonly string[],
): RunnerSelectorOptions {
const args = rawArgs.filter((value) => value !== "--");
const options: RunnerSelectorOptions = {
all: false,
list: false,
matrixJson: false,
ids: [],
suites: [],
groups: [],
profiles: [],
environments: [],
cases: [],
headed: false,
ui: false,
debug: false,
maxParallel: Number(process.env.PAPERCLIP_E2E_MAX_PARALLEL ?? "1"),
};
for (let index = 0; index < args.length; index += 1) {
const flag = args[index];
if (flag === "--all") options.all = true;
else if (flag === "--list") options.list = true;
else if (flag === "--matrix-json") options.matrixJson = true;
else if (flag === "--headed") options.headed = true;
else if (flag === "--ui") options.ui = true;
else if (flag === "--debug") options.debug = true;
else if (flag === "--max-parallel") {
const value = valueFor(args, index, flag);
index += 1;
options.maxParallel = Number(value);
} else if (
[
"--id",
"--suite",
"--group",
"--profile",
"--environment",
"--case",
].includes(flag)
) {
const value = valueFor(args, index, flag);
index += 1;
if (flag === "--id") options.ids.push(value);
else if (flag === "--suite") options.suites.push(value);
else if (flag === "--group") options.groups.push(value);
else if (flag === "--profile") options.profiles.push(value);
else if (flag === "--environment") options.environments.push(value);
else options.cases.push(value);
} else {
throw new RunnerSelectorError(`Unknown runner E2E argument: ${flag}`);
}
}
if (!Number.isInteger(options.maxParallel) || options.maxParallel < 1) {
throw new RunnerSelectorError("--max-parallel must be a positive integer");
}
const hasDimensions =
options.suites.length +
options.groups.length +
options.profiles.length +
options.environments.length +
options.cases.length >
0;
if (options.ids.length > 0 && (hasDimensions || options.all)) {
throw new RunnerSelectorError(
"--id is exclusive with --all and dimension filters",
);
}
if (options.all && hasDimensions) {
throw new RunnerSelectorError("--all is exclusive with dimension filters");
}
if (
!options.list &&
!options.matrixJson &&
!options.all &&
options.ids.length === 0 &&
!hasDimensions
) {
throw new RunnerSelectorError(
"Billable runner E2E runs require --all or an explicit selector",
);
}
return options;
}
function assertKnown(
label: string,
selected: readonly string[],
known: Set<string>,
) {
const unknown = selected.filter((value) => !known.has(value));
if (unknown.length > 0)
throw new RunnerSelectorError(`Unknown ${label}: ${unknown.join(", ")}`);
}
export function selectRunnerExecutions(
options: RunnerSelectorOptions,
matrix: readonly MatrixExecution[] = runnerMatrix,
): MatrixExecution[] {
const knownGroups = new Set(matrix.flatMap((execution) => execution.groups));
assertKnown(
"suite",
options.suites,
new Set(matrix.map((execution) => execution.suite.id)),
);
assertKnown("group", options.groups, knownGroups);
assertKnown(
"profile",
options.profiles,
new Set(matrix.map((execution) => execution.profile.id)),
);
assertKnown(
"environment",
options.environments,
new Set(matrix.map((execution) => execution.environment.id)),
);
assertKnown(
"case",
options.cases,
new Set(matrix.map((execution) => execution.task.id)),
);
assertKnown(
"execution id",
options.ids,
new Set(matrix.map((execution) => execution.id)),
);
const selected = matrix.filter((execution) => {
if (options.ids.length > 0) return options.ids.includes(execution.id);
if (
options.all ||
(options.list &&
options.groups.length === 0 &&
options.suites.length === 0 &&
options.profiles.length === 0 &&
options.environments.length === 0 &&
options.cases.length === 0)
)
return true;
return (
(options.suites.length === 0 ||
options.suites.includes(execution.suite.id)) &&
options.groups.every((group) => execution.groups.includes(group)) &&
(options.profiles.length === 0 ||
options.profiles.includes(execution.profile.id)) &&
(options.environments.length === 0 ||
options.environments.includes(execution.environment.id)) &&
(options.cases.length === 0 || options.cases.includes(execution.task.id))
);
});
if (selected.length === 0)
throw new RunnerSelectorError(
"Runner E2E selectors matched zero executions",
);
return selected;
}
export function buildMatrixJobs(
executions: readonly MatrixExecution[],
): MatrixJob[] {
return executions
.map((execution) => ({
executionId: execution.id,
suiteId: execution.suite.id,
profileId: execution.profile.id,
credentialName: execution.profile.credential,
environmentId: execution.environment.id,
caseId: execution.task.id,
timeoutMinutes: Math.max(
execution.environment.id === "daytona" ? 40 : 25,
Math.ceil(
(2 *
(execution.task.attemptTimeoutMs[execution.environment.id] +
90_000) +
5 * 60_000) /
60_000,
),
),
needsDaytona: execution.environment.id === "daytona",
}))
.sort((left, right) => left.executionId.localeCompare(right.executionId));
}