fix(apps): document Railway SSH setup and sync with master

Register the board-only SSH setup operation with its shared request schema and error responses. Resolve current catalog and branding metadata conflicts.

Co-Authored-By: Paperclip <noreply@paperclip.ing>
This commit is contained in:
Devin FoleyandPaperclip committed 2026-09-15 17:16:15 -07:00
commit 303340f190
488 files changed
+116005 -2175

No files matched your search

+1 -1
View File
@@ -156,7 +156,7 @@ Guidelines:
- Use `info` for successful checks and context.
Severity policy is product-critical: warnings are not save blockers.
Example: for `claude_local`, detected `ANTHROPIC_API_KEY` must be a `warn`, not an `error`, because Claude can still run (it just uses API-key auth instead of subscription auth).
Example: for `claude_local`, an explicitly configured `ANTHROPIC_API_KEY` or selected managed API connection is `info`: the user chose that authentication. An ambient server key overriding subscription login remains `warn`, not `error`.
---
@@ -0,0 +1,14 @@
{
"Sid": "AllowCloudFrontReadCloudMigrators",
"Effect": "Allow",
"Principal": {
"Service": "cloudfront.amazonaws.com"
},
"Action": "s3:GetObject",
"Resource": "arn:aws:s3:::paperclipai-runner-e2e-history-078455283791-us-east-1/cloud-migrators/v1/*",
"Condition": {
"StringEquals": {
"AWS:SourceArn": "arn:aws:cloudfront::078455283791:distribution/E3GTU28BBO2SFR"
}
}
}
@@ -0,0 +1,18 @@
{
"Version": "2012-10-17",
"Statement": [
{
"Effect": "Allow",
"Principal": {
"Federated": "arn:aws:iam::078455283791:oidc-provider/token.actions.githubusercontent.com"
},
"Action": "sts:AssumeRoleWithWebIdentity",
"Condition": {
"StringEquals": {
"token.actions.githubusercontent.com:aud": "sts.amazonaws.com",
"token.actions.githubusercontent.com:sub": "repo:paperclipai/paperclip:ref:refs/heads/master"
}
}
}
]
}
@@ -0,0 +1,25 @@
{
"Version": "2012-10-17",
"Statement": [
{
"Effect": "Allow",
"Action": "s3:PutObject",
"Resource": "arn:aws:s3:::paperclipai-runner-e2e-history-078455283791-us-east-1/cloud-migrators/v1/*",
"Condition": {
"StringEquals": {
"s3:if-none-match": "*"
}
}
},
{
"Effect": "Allow",
"Action": "s3:ListBucket",
"Resource": "arn:aws:s3:::paperclipai-runner-e2e-history-078455283791-us-east-1",
"Condition": {
"StringLike": {
"s3:prefix": "cloud-migrators/v1/*"
}
}
}
]
}
+3
View File
@@ -37,3 +37,6 @@ updates:
day: monday
time: "06:00"
open-pull-requests-limit: 5
ignore:
# This first-party workflow follows CODEOWNERS-protected master.
- dependency-name: "paperclipai/paperclip/.github/workflows/pr-trusted.yml"
+102 -49
View File
@@ -1,20 +1,47 @@
import test from "node:test";
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
import { waitForCloudArtifacts } from "../../../scripts/cloud-readiness.mjs";
import { existsSync, readFileSync } from "node:fs";
import { gzipSync } from "node:zlib";
import { waitForCloudArtifacts, verifyManifestProvenance, migratorPublished } from "../../../scripts/cloud-readiness.mjs";
import { artifactBase, descriptor } from "../../../scripts/cloud-migrator-artifacts.mjs";
import { previewManifest } from "../../../scripts/preview-artifacts.mjs";
const sha = "a".repeat(40);
const version = `0.0.0-preview.g${sha}`;
const digest = `sha256:${"b".repeat(64)}`;
const json = (body, status = 200) => new Response(JSON.stringify(body), { status });
function registry({ missing = new Set(), failure, wrongImage = false, wrongPackage = false } = {}) {
return async (url) => {
const producer = { id: 123, head_sha: sha, head_branch: "master", path: ".github/workflows/cloud-migrator-artifacts.yml",
head_repository: { id: 1170821064, full_name: "paperclipai/paperclip" }, event: "push", status: "completed", conclusion: "success" };
function bundle() {
const packages = {}; const files = new Map();
const entries = { "": { dependencies: { "@paperclipai/db": version } } };
for (const name of ["db", "shared"]) {
const metadata = previewManifest({ name: `@paperclipai/${name}`, dependencies: {} }, sha);
const bytes = Buffer.from(JSON.stringify(metadata));
const header = Buffer.alloc(512); header.write("package/package.json"); header.write(bytes.length.toString(8).padStart(11, "0"), 124, 11); header[156] = 48;
const padded = Buffer.alloc(Math.ceil(bytes.length / 512) * 512); bytes.copy(padded);
const archive = gzipSync(Buffer.concat([header, padded, Buffer.alloc(1024)]));
const pin = descriptor(archive, "tgz"); packages[name] = pin; files.set(pin.url, archive);
entries[`node_modules/@paperclipai/${name}`] = { version, resolved: pin.url, integrity: pin.integrity, dependencies: metadata.dependencies };
}
const lock = Buffer.from(JSON.stringify({ lockfileVersion: 3, packages: entries }));
const manifest = { version: 1, sourceSha: sha, packageVersion: version, packages, lockfile: descriptor(lock, "json") };
files.set(manifest.lockfile.url, lock);
const bytes = Buffer.from(JSON.stringify(manifest) + "\n");
files.set(`${artifactBase}/${sha}/manifest.json`, bytes);
return { manifest, bytes, files };
}
function registry({ missing = new Set(), failure, wrongImage = false, run = producer, objects = bundle() } = {}) {
return async (url, options) => {
assert.ok(!url.startsWith("https://registry.npmjs.org/"), "readiness must never wait for npm");
if (failure) return json({}, failure);
if (url.startsWith("https://registry.npmjs.org/")) {
const name = decodeURIComponent(new URL(url).pathname.split("/")[1]);
if (missing.has(name.split("/")[1])) return json({}, 404);
const pkg = previewManifest({ name, version: "0.0.0" }, sha);
return json({ ...pkg, ...(wrongPackage ? { gitHead: "c".repeat(40) } : {}), dist: { integrity: "sha512-fixture", tarball: "https://registry.npmjs.org/fixture.tgz" } });
if (url.startsWith("https://api.github.com/")) {
assert.match(url, new RegExp(`head_sha=${sha}&per_page=100&page=1$`));
return json({ total_count: missing.has("migrator") ? 0 : 1, workflow_runs: missing.has("migrator") ? [] : [run] });
}
if (url.startsWith(artifactBase)) {
assert.equal(options.headers?.Authorization, undefined, "GitHub credentials stay off the artifact origin");
return objects.files.has(url) ? new Response(objects.files.get(url)) : json({}, 403);
}
if (url.includes("/token?")) return json({ token: "fixture" });
if (url.includes("/manifests/")) return missing.has("image") ? json({}, 404) : json({ config: { digest } });
@@ -22,70 +49,96 @@ function registry({ missing = new Set(), failure, wrongImage = false, wrongPacka
throw new Error(`Unexpected request: ${url}`);
};
}
const noSignature = async () => {}; // Signature enforcement is exercised separately below.
test("readiness requires the image and both exact-source packages on the successful poll", async () => {
const missing = new Set(["image", "shared", "db"]);
let clock = 0;
const states = [];
test("readiness rechecks image and publisher, then verifies the exact signed bundle with no npm requests", async () => {
const missing = new Set(["image", "migrator"]); const objects = bundle(); let clock = 0; let signatures = 0;
const result = await waitForCloudArtifacts(sha, {
fetchImpl: registry({ missing }), now: () => clock, intervalMs: 10, timeoutMs: 100, log: (message) => states.push(message),
fetchImpl: registry({ missing, objects }), token: "fixture", now: () => clock, intervalMs: 10, timeoutMs: 100, log: () => {},
verifyProvenance: async (bytes, source) => { assert.deepEqual(bytes, objects.bytes); assert.equal(source, sha); signatures++; },
sleep: async (ms) => {
clock += ms;
if (clock === 10) missing.delete("image");
if (clock === 20) missing.delete("shared");
if (clock === 30) { missing.delete("db"); missing.add("image"); }
if (clock === 40) missing.delete("image");
if (clock === 20) { missing.delete("migrator"); missing.add("image"); }
if (clock === 30) missing.delete("image");
},
});
assert.equal(clock, 40, "an artifact disappearing before the final poll must prevent readiness");
assert.deepEqual(result, { version: 1, sha, packageVersion: `0.0.0-preview.g${sha}` });
assert.match(states.at(-1), /Cloud artifacts available/);
assert.equal(clock, 30); assert.equal(signatures, 1);
assert.deepEqual(result, { version: 1, sha, packageVersion: version });
});
test("missing artifacts time out with a precise inventory and bounded sleep", async () => {
let clock = 0;
const sleeps = [];
await assert.rejects(waitForCloudArtifacts(sha, {
fetchImpl: registry({ missing: new Set(["db"]) }), now: () => clock, timeoutMs: 25, intervalMs: 20, log: () => {},
sleep: async (ms) => { sleeps.push(ms); clock += ms; },
}), /timed out.*missing: db/);
assert.deepEqual(sleeps, [20, 5]);
});
for (const fixture of [{ failure: 403 }, { failure: 503 }, { wrongImage: true }, { wrongPackage: true }]) {
test(`registry errors and identity mismatches fail without waiting: ${JSON.stringify(fixture)}`, async () => {
test("missing or in-progress publishers time out with a precise inventory and bounded sleep", async () => {
for (const fixture of [{ missing: new Set(["migrator"]) }, { run: { ...producer, status: "in_progress", conclusion: null } }]) {
let clock = 0; const sleeps = [];
await assert.rejects(waitForCloudArtifacts(sha, {
fetchImpl: registry(fixture), sleep: async () => assert.fail("must not retry an invalid artifact or upstream error"), log: () => {},
}));
fetchImpl: registry(fixture), now: () => clock, timeoutMs: 25, intervalMs: 20, log: () => {}, verifyProvenance: noSignature,
sleep: async (ms) => { sleeps.push(ms); clock += ms; },
}), /timed out.*missing: migrator/);
assert.deepEqual(sleeps, [20, 5]);
}
});
for (const fixture of [{ failure: 403 }, { failure: 503 }, { wrongImage: true },
...["failure", "cancelled", "skipped"].map((conclusion) => ({ run: { ...producer, conclusion } })),
...[{ head_sha: "b".repeat(40) }, { head_branch: "feature" }, { path: ".github/workflows/evil.yml" },
{ head_repository: { id: 123, full_name: "someone/paperclip" } }, { event: "pull_request" }].map((wrong) => ({ run: { ...producer, ...wrong } }))]) {
test(`upstream errors, failed publication and identity mismatches fail immediately: ${JSON.stringify(fixture)}`, async () => {
await assert.rejects(waitForCloudArtifacts(sha, { fetchImpl: registry(fixture), verifyProvenance: noSignature,
sleep: async () => assert.fail("must not retry an invalid artifact or upstream error"), log: () => {} }));
});
}
test("successful publication cannot hide inaccessible or corrupt archives or an invalid signature", async () => {
for (const corrupt of [false, true]) {
const objects = bundle();
if (corrupt) objects.files.set(objects.manifest.packages.db.url, Buffer.from("corrupt"));
else objects.files.delete(objects.manifest.packages.db.url);
await assert.rejects(waitForCloudArtifacts(sha, { fetchImpl: registry({ objects }), verifyProvenance: noSignature, log: () => {} }), /download failed|immutable pin/);
}
await assert.rejects(waitForCloudArtifacts(sha, { fetchImpl: registry(), verifyProvenance: async () => { throw new Error("invalid signature"); }, log: () => {} }), /invalid signature/);
});
test("CLI verifies the exact bytes, source, master workflow and hosted runner and cleans up on failure", () => {
let temporary;
assert.throws(() => verifyManifestProvenance(Buffer.from("exact manifest\n"), sha, { exec: (cmd, args) => {
assert.equal(cmd, "gh"); assert.deepEqual(args.slice(0, 2), ["attestation", "verify"]); temporary = args[2];
assert.equal(readFileSync(temporary, "utf8"), "exact manifest\n");
for (const [flag, value] of [["--repo", "paperclipai/paperclip"], ["--source-digest", sha], ["--source-ref", "refs/heads/master"],
["--cert-identity", "https://github.com/paperclipai/paperclip/.github/workflows/cloud-migrator-artifacts.yml@refs/heads/master"]]) assert.equal(args[args.indexOf(flag) + 1], value);
assert.ok(args.includes("--deny-self-hosted-runners")); throw new Error("verification rejected");
} }), /verification rejected/);
assert.equal(existsSync(temporary), false);
});
test("invalid source and timing configuration are rejected before registry access", async () => {
const fetchImpl = async () => assert.fail("invalid inputs must not reach a registry");
await assert.rejects(waitForCloudArtifacts("master", { fetchImpl }), /full immutable commit SHA/);
for (const options of [{ timeoutMs: 0 }, { intervalMs: -1 }, { timeoutMs: Infinity }]) {
await assert.rejects(waitForCloudArtifacts(sha, { ...options, fetchImpl }), /positive finite/);
}
for (const options of [{ timeoutMs: 0 }, { intervalMs: -1 }, { timeoutMs: Infinity }]) await assert.rejects(waitForCloudArtifacts(sha, { ...options, fetchImpl }), /positive finite/);
});
test("the versioned readiness job requires successful source, image and artifact jobs", () => {
test("versioned readiness retains every source gate and removes duplicate automatic npm publication", () => {
const workflow = readFileSync(new URL("../../workflows/cloud-readiness.yml", import.meta.url), "utf8");
assert.match(workflow, /push:\s*\n\s*branches: \[master\]/);
assert.match(workflow, /group: cloud-readiness-\$\{\{ github.sha \}\}/);
assert.match(workflow, /uses: \.\/\.github\/workflows\/release-verify.yml\s+with:\s+ref: \$\{\{ github.sha \}\}/);
assert.match(workflow, /uses: \.\/\.github\/workflows\/docker-cloud.yml/);
assert.match(workflow, /attestations: read/); assert.match(workflow, /GH_TOKEN: \$\{\{ github.token \}\}/);
const ready = workflow.split(" ready:")[1];
assert.match(ready, /name: Cloud deployable v1/);
assert.match(ready, /needs: \[verify, image, artifacts\]/);
assert.match(ready, /name: Cloud deployable v1/); assert.match(ready, /needs: \[verify, image, artifacts\]/);
assert.match(ready, /if: github.repository == 'paperclipai\/paperclip' && github.ref == 'refs\/heads\/master'/);
assert.doesNotMatch(ready, /^\s*(?:if:.*always\(|continue-on-error:)/m);
assert.doesNotMatch(workflow, /secrets: inherit|id-token: write|actions: write|checks: write|uses: .*@v\d\b/);
const cloud = readFileSync(new URL("../../workflows/docker-cloud.yml", import.meta.url), "utf8");
assert.doesNotMatch(cloud, /^ push:/m, "the master image must build only once");
const migrator = readFileSync(new URL("../../workflows/cloud-artifacts.yml", import.meta.url), "utf8");
assert.match(migrator, /push:\s*\n\s*branches: \[master\]/);
assert.match(migrator, /SOURCE_SHA: \$\{\{ github.sha \}\}/);
assert.match(migrator, /gh workflow run release.yml .*--ref master/);
assert.match(migrator, /--field channel=cloud-migrator/);
assert.match(migrator, /--field source_ref="\$SOURCE_SHA"/);
assert.equal(existsSync(new URL("../../workflows/cloud-artifacts.yml", import.meta.url)), false);
});
test("later manual failures or pending retries cannot hide an earlier successful immutable publication", async () => {
for (const latest of [{ status: "completed", conclusion: "failure" }, { status: "in_progress", conclusion: null }]) {
let calls = 0;
assert.equal(await migratorPublished(sha, async (url) => {
calls++;
if (url.endsWith("page=1")) return json({ total_count: 101, workflow_runs: Array.from({ length: 100 }, (_, i) => ({ ...producer, ...latest, id: 200 + i, event: "workflow_dispatch" })) });
assert.ok(url.endsWith("page=2")); return json({ total_count: 101, workflow_runs: [producer] });
}), true);
assert.equal(calls, 2);
}
});
@@ -11,7 +11,7 @@ const base = {
};
const expectedJobs = {
"cloud-readiness.yml": [],
"cloud-artifacts.yml": ["dispatch_migrator"],
"cloud-migrator-artifacts.yml": [],
"release-verify.yml": ["typecheck", "general_tests", "serialized_tests", "runner_workflow_evals", "verify_paperclip_runner", "build"],
"runner-chaos-evals.yml": ["chaos_and_recovery"],
"release.yml": ["plan_preview", "package_preview"],
@@ -0,0 +1,55 @@
import test from "node:test";
import assert from "node:assert/strict";
import { readFileSync } from "node:fs";
const read = (name) => readFileSync(new URL(`../../workflows/${name}`, import.meta.url), "utf8");
const prWorkflow = read("pr-trusted.yml");
const releaseWorkflow = read("release-verify.yml");
const pr = prWorkflow.split(" verify_paperclip_runner:")[1].split(" build:")[0];
const release = releaseWorkflow.split(" verify_paperclip_runner:")[1].split(" build:")[0];
// The key is computed from these inputs. A pull request that disagrees with
// the master writer on any of them misses every time and silently recompiles
// all 313 third-party crates in both profiles, which is exactly the cost this
// restore exists to remove.
const keyInputs = [
/uses: Swatinem\/rust-cache@([0-9a-f]{40}) # v[0-9.]+/,
/workspaces: (packages\/paperclip-runner\/runner -> target)/,
/shared-key: (release-runner-v1)/,
/cache-workspace-crates: (false)/,
/cache-bin: (false)/,
];
test("the PR lane restores the Rust cache under the same key the master push writes", () => {
for (const pattern of keyInputs) {
const mine = pr.match(pattern);
const theirs = release.match(pattern);
assert.ok(mine, `PR lane is missing ${pattern}`);
assert.ok(theirs, `master writer is missing ${pattern}`);
assert.equal(mine[1], theirs[1], `key input drifted from the master writer: ${pattern}`);
}
assert.doesNotMatch(pr, /prefix-key:|cache-on-failure: true|cache-all-crates: true/);
});
test("the PR lane pins the compiler before the key is computed", () => {
const select = pr.indexOf(" - name: Select the pinned Runner Rust toolchain");
const cache = pr.indexOf(" - name: Restore Runner Rust dependencies (read only)");
const verify = pr.indexOf(" - name: Verify Paperclip Runner\n");
assert.ok(select >= 0 && cache > select && verify > cache);
const setup = pr.slice(select, cache);
assert.match(setup, /working-directory: packages\/paperclip-runner/);
assert.match(setup, /rustup show active-toolchain/);
assert.match(setup, /echo "RUSTUP_TOOLCHAIN=\$toolchain" >> "\$GITHUB_ENV"/);
// The gate routes to either ubuntu-latest or the public PR fleet, so a
// missing rustup must cost the cache, never the pull request.
assert.match(setup, /command -v rustup/);
assert.doesNotMatch(setup, /set -euo pipefail/);
});
test("a pull request never writes to or evicts the master cache entry", () => {
const step = pr.split(" - name: Restore Runner Rust dependencies (read only)")[1]
.split(" - name: Verify Paperclip Runner\n")[0];
assert.equal(step.match(/^\s*save-if: (.+)$/m)?.[1], "false");
assert.doesNotMatch(step, /^\s*if:/m, "the restore must not be conditional; a miss is already free");
assert.doesNotMatch(prWorkflow, /uses: Swatinem\/rust-cache@[0-9a-f]{40}[\s\S]*?save-if: (?!false)/);
});
-34
View File
@@ -1,34 +0,0 @@
name: Cloud artifacts
on:
push:
branches: [master]
workflow_dispatch:
permissions: {}
jobs:
dispatch_migrator:
name: Start exact-source cloud migrator publication
if: github.repository == 'paperclipai/paperclip' && github.ref == 'refs/heads/master'
runs-on: ${{ vars.AWS_POST_MERGE_CI_ENABLED == 'true' && github.repository == 'paperclipai/paperclip' && github.repository_id == '1170821064' && github.ref == 'refs/heads/master' && (github.event_name == 'push' || github.event_name == 'workflow_dispatch') && 'runs-on/fleet=paperclip-post-merge-x64/env=public-ci' || 'ubuntu-latest' }}
timeout-minutes: 5
permissions:
actions: write
steps:
# This separate workflow starts at merge, outside the full npm release's
# concurrency group. Publication stays in release.yml so npm recognizes
# the established trusted-publisher identity and npm-canary environment.
# No source checkout or package code runs with the dispatch credential.
- name: Dispatch the migrator-only release
env:
GH_TOKEN: ${{ github.token }}
SOURCE_SHA: ${{ github.sha }}
run: |
set -euo pipefail
request_id="$(cat /proc/sys/kernel/random/uuid)"
gh workflow run release.yml --repo "$GITHUB_REPOSITORY" --ref master \
--field channel=cloud-migrator \
--field source_ref="$SOURCE_SHA" \
--field request_id="$request_id"
echo "Started Cloud migrator $SOURCE_SHA in release.yml (request $request_id)." >> "$GITHUB_STEP_SUMMARY"
@@ -0,0 +1,90 @@
name: Cloud migrator artifacts
run-name: Cloud migrator artifacts ${{ github.sha }}
on:
push:
branches: [master]
workflow_dispatch:
permissions: {}
concurrency:
group: cloud-migrator-artifacts-${{ github.sha }}
cancel-in-progress: false
jobs:
build:
if: github.repository == 'paperclipai/paperclip' && github.repository_id == '1170821064' && github.ref == 'refs/heads/master'
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
ref: ${{ github.sha }}
persist-credentials: false
- uses: pnpm/action-setup@0977fd99725f1db4007ccb2928dbb4e90d06cc86 # v6
with:
version: 9.15.4
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
with:
node-version: 24
- name: Install migrator build dependencies
run: pnpm install --ignore-scripts --no-frozen-lockfile --filter @paperclipai/db... --filter @paperclipai/shared...
- name: Build exact-source packages and dependency lockfile
env:
SOURCE_SHA: ${{ github.sha }}
run: |
node scripts/preview-artifacts.mjs pack . migrator-artifacts "$SOURCE_SHA"
node scripts/cloud-migrator-artifacts.mjs build migrator-artifacts "$SOURCE_SHA"
node scripts/cloud-migrator-artifacts.mjs verify-install migrator-artifacts "$SOURCE_SHA"
- uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: cloud-migrator-bundle
path: |
migrator-artifacts/db.tgz
migrator-artifacts/shared.tgz
migrator-artifacts/package-lock.json
migrator-artifacts/manifest.json
if-no-files-found: error
retention-days: 3
publish:
needs: build
if: github.repository == 'paperclipai/paperclip' && github.repository_id == '1170821064' && github.ref == 'refs/heads/master'
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
id-token: write
attestations: write
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
ref: ${{ github.sha }}
persist-credentials: false
sparse-checkout: scripts
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
with:
node-version: 24
- uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4
with:
name: cloud-migrator-bundle
path: migrator-artifacts
- name: Validate the complete bundle before attestation
env:
SOURCE_SHA: ${{ github.sha }}
run: node scripts/cloud-migrator-artifacts.mjs validate migrator-artifacts "$SOURCE_SHA"
- name: Attest the manifest and every content hash it pins
uses: actions/attest@1e69f48acb82d1966a394da916b4c1698aa569d6 # v4
with:
subject-path: migrator-artifacts/manifest.json
- uses: aws-actions/configure-aws-credentials@e6de054238d6b7531b4efff3b6587d9aade6a06c # v6
with:
role-to-assume: arn:aws:iam::078455283791:role/paperclip-cloud-migrator-github
aws-region: us-east-1
role-duration-seconds: 900
- name: Publish immutable migrator and verify public downloads
env:
SOURCE_SHA: ${{ github.sha }}
run: node scripts/cloud-migrator-artifacts.mjs publish migrator-artifacts "$SOURCE_SHA"
+8 -3
View File
@@ -37,6 +37,8 @@ jobs:
timeout-minutes: 35
permissions:
contents: read
actions: read
attestations: read
steps:
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
@@ -46,6 +48,7 @@ jobs:
node-version: 24
- name: Wait for verified image and exact-source migrator
env:
GH_TOKEN: ${{ github.token }}
SOURCE_SHA: ${{ github.sha }}
run: node scripts/cloud-readiness.mjs "$SOURCE_SHA"
@@ -91,6 +94,8 @@ jobs:
env:
SOURCE_SHA: ${{ github.sha }}
run: |
echo "Cloud deployable v1: $SOURCE_SHA" >> "$GITHUB_STEP_SUMMARY"
echo "Source verification passed; the full-SHA image and exact-source migrator are available." >> "$GITHUB_STEP_SUMMARY"
echo "Deployment tooling must still resolve and pin the image and migrator and validate migration compatibility." >> "$GITHUB_STEP_SUMMARY"
{
echo "Cloud deployable v1: $SOURCE_SHA"
echo "Source verification passed; the full-SHA image and exact-source migrator are available."
echo "Deployment tooling must still resolve and pin the image and migrator and validate migration compatibility."
} >> "$GITHUB_STEP_SUMMARY"
+36
View File
@@ -658,6 +658,42 @@ jobs:
- name: Install dependencies
run: pnpm install --frozen-lockfile
# Same restore-only contract as the pnpm store above, for the Rust
# dependency tree: master's post-merge verification is the sole writer
# of release-runner-v1, and PR merge refs must not save branch-scoped
# copies of a ~680MB target directory. Every key input below has to
# match that writer in release-verify.yml exactly or each PR misses and
# recompiles all 313 third-party crates in both profiles. A miss is a
# slow run, never a wrong one.
- name: Select the pinned Runner Rust toolchain
working-directory: packages/paperclip-runner
run: |
set -uo pipefail
# release-verify.yml runs on a single post-merge fleet image; the
# gate here can route to ubuntu-latest or the public PR fleet, so
# this tolerates an image without rustup instead of failing every
# pull request. Without the pin the cache key simply will not match.
if ! command -v rustup >/dev/null 2>&1; then
echo '::notice title=Runner Rust cache::rustup is unavailable; building with the image default toolchain'
exit 0
fi
rustup show
toolchain="$(rustup show active-toolchain | awk '{print $1}')"
echo "RUSTUP_TOOLCHAIN=$toolchain" >> "$GITHUB_ENV"
- name: Restore Runner Rust dependencies (read only)
uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2
with:
workspaces: packages/paperclip-runner/runner -> target
shared-key: release-runner-v1
# Mirror the master writer: these also feed the cache key.
cache-workspace-crates: false
cache-bin: false
# Restore only. Never let a pull request evict master's entry.
save-if: false
- name: Verify Paperclip Runner
run: pnpm --filter @paperclipai/paperclip-runner check:all
+3 -2
View File
@@ -10,5 +10,6 @@ permissions:
jobs:
ci:
# Pin: #13300 merge — restore-only dependency caches and parallel native verification.
uses: paperclipai/paperclip/.github/workflows/pr-trusted.yml@44dde2dec42a22746a2f36b595acacc9ccfa1df6
# Master requires CODEOWNERS review for .github/**, including this workflow.
# The AWS runner group permits pr-trusted.yml@refs/heads/master.
uses: paperclipai/paperclip/.github/workflows/pr-trusted.yml@master
+19 -5
View File
@@ -71,7 +71,21 @@ env:
# 07:07:40.) The publish jobs' timeout-minutes are sized for several
# laggard packages; if most of a batch lags the full budget, npm is having
# a real incident and the job failing is correct.
NPM_PUBLISH_VERIFY_ATTEMPTS: "60"
#
# Raised to 30 minutes on 2026-09-14, when 10 was not enough twice in a
# row: @paperclipai/server was accepted at 17:49:09 and became visible at
# 18:04:29 — five minutes after the poll gave up. Each timeout aborts the
# release mid-batch, leaving that version half-published, and because the
# next version number is derived from what is already on npm the retry
# moves to a new number and meets the same lag. Waiting is cheap; a
# half-published release is not.
#
# Packages are published and verified one at a time, so the publish jobs'
# timeout-minutes went to 150 alongside this: a 30-minute build plus four
# packages each lagging the full budget still finishes inside the job,
# instead of the job timing out mid-batch and leaving the same
# half-published state this budget exists to avoid.
NPM_PUBLISH_VERIFY_ATTEMPTS: "180"
NPM_PUBLISH_VERIFY_DELAY_SECONDS: "10"
jobs:
@@ -326,7 +340,7 @@ jobs:
if: github.event_name == 'push'
needs: verify_canary
runs-on: ubuntu-latest
timeout-minutes: 90
timeout-minutes: 150
environment: npm-canary
outputs:
canary_version: ${{ steps.canary_tag.outputs.version }}
@@ -582,7 +596,7 @@ jobs:
(needs.smoke_nightly.result == 'success' ||
(needs.smoke_nightly.result == 'skipped' && github.event_name == 'workflow_dispatch' && inputs.dry_run))
runs-on: ubuntu-latest
timeout-minutes: 90
timeout-minutes: 150
environment: npm-canary
# The workflow-level concurrency group is per event, so a forced dispatch
# nightly could otherwise overlap the scheduled one and race it to the
@@ -848,7 +862,7 @@ jobs:
(needs.verify_beta_candidate.result == 'success' ||
(needs.verify_beta_candidate.result == 'skipped' && needs.select_beta.outputs.mode == 'promote'))
runs-on: ubuntu-latest
timeout-minutes: 90
timeout-minutes: 150
environment: npm-beta
# Serialize beta publishes so two dispatches cannot race to the same next
# -beta.N version; release.sh additionally refuses to double-publish a
@@ -1271,7 +1285,7 @@ jobs:
if: github.event_name == 'workflow_dispatch' && inputs.channel == 'stable' && !inputs.dry_run
needs: [preflight_stable, verify_stable]
runs-on: ubuntu-latest
timeout-minutes: 90
timeout-minutes: 150
environment: npm-stable
permissions:
contents: write
+5
View File
@@ -74,6 +74,11 @@ pnpm dev
1. Keep changes company-scoped.
Every domain entity should be scoped to a company and company boundaries must be enforced in routes/services.
Explicit exception: announcement dismissals are instance-wide user preferences,
keyed by user and announcement so they persist across companies. Their audit
context must still validate company membership. The announcement publication-ID
registry is instance-level feed metadata; it contains no company or user data.
2. Keep contracts synchronized.
If you change schema/API behavior, update all impacted layers:
- `packages/db` schema and exports
+4
View File
@@ -0,0 +1,4 @@
{
"schemaVersion": 1,
"announcement": null
}
Binary file not shown.

After

Width:  |  Height:  |  Size: 25 KiB

@@ -0,0 +1,12 @@
<!doctype html>
<html><head><style>
*{box-sizing:border-box}html,body{margin:0;width:100%;height:100%;overflow:hidden}body{font-family:system-ui,sans-serif;background:#14171c;color:#e9edf2;display:grid;place-items:center}
.scene{position:relative;width:100%;height:100%;min-height:90px;background:radial-gradient(ellipse at 50% 120%,#403762 0%,transparent 68%),linear-gradient(110deg,#181c22,#20242c)}
.grid{position:absolute;inset:0;background-image:linear-gradient(#ffffff06 1px,transparent 1px),linear-gradient(90deg,#ffffff06 1px,transparent 1px);background-size:20px 20px}
.label{position:absolute;top:12%;left:7%;font-size:9px;letter-spacing:2px;color:#a9b0bf}.spark{display:inline-block;width:5px;height:5px;border-radius:50%;background:#ac9af7;margin-right:6px;box-shadow:0 0 10px #ac9af7}
.flow{position:absolute;inset:38% 7% 16%;display:flex;align-items:center;justify-content:space-between;gap:12px}.track{position:absolute;left:10%;right:10%;top:50%;height:1px;background:#505167}.packet{position:absolute;top:-2px;width:5px;height:5px;border-radius:50%;background:#c4b5fd;box-shadow:0 0 9px #b7a0fc;animation:travel 4.8s linear infinite}
.node{position:relative;display:flex;align-items:center;justify-content:center;gap:7px;width:29%;height:42px;border:1px solid #4b4b60;border-radius:10px;background:#262834;box-shadow:0 4px 12px #0003;font-size:11px;font-weight:500;animation:breathe 4.8s ease-in-out infinite}.node:nth-of-type(3){animation-delay:1.6s}.node:nth-of-type(4){animation-delay:3.2s}.icon{width:15px;height:15px;fill:none;stroke:#c4b5fd;stroke-width:1.5;stroke-linecap:round;stroke-linejoin:round}
@keyframes travel{0%{left:0;opacity:0}8%{opacity:1}92%{opacity:1}100%{left:100%;opacity:0}}
@keyframes breathe{0%,65%,100%{transform:translateY(0);border-color:#4b4b60}15%,35%{transform:translateY(-3px);border-color:#a79ade;box-shadow:0 4px 20px #9580d52b}}
@media(prefers-reduced-motion:reduce){*,*::before,*::after{animation:none!important}}
</style></head><body><div class="scene"><div class="grid"></div><div class="label"><span class="spark"></span>ONE IDEA. A WHOLE TEAM.</div><div class="flow"><div class="track"><div class="packet"></div></div><div class="node"><svg class="icon" viewBox="0 0 20 20"><path d="M6 8a4 4 0 1 1 8 0c0 2-2 2-2 5H8c0-3-2-3-2-5M8 16h4" /></svg>Plan</div><div class="node"><svg class="icon" viewBox="0 0 20 20"><path d="m6 6-4 4 4 4m8-8 4 4-4 4m-3-10-2 12" /></svg>Build</div><div class="node"><svg class="icon" viewBox="0 0 20 20"><circle cx="10" cy="10" r="7"/><path d="m6 10 3 3 5-6"/></svg>Review</div></div></div></body></html>
@@ -0,0 +1,27 @@
{
"schemaVersion": 1,
"announcement": {
"id": "preview-animated-team",
"eyebrow": "Staging preview",
"title": "From one idea to a working team",
"description": "Set a goal, bring in your agents, and follow the work as it moves forward.",
"image": {
"path": "assets/6ac073cb3447e26f856b15ec058bb9d4d036b0babff60af88ee0dd49cabfa124.png",
"alt": "Paperclip. Ideas become work."
},
"secondaryLink": {
"kind": "external",
"label": "Learn more",
"url": "https://paperclip.ing"
},
"primaryAction": {
"kind": "route",
"label": "Explore your projects",
"path": "/projects"
},
"animation": {
"path": "assets/78bafb6adbfd9da899cdbb5d934b4c0b9df5d419d6f0a5104a87a7b25dcc6bd8.html",
"alt": "Agents plan, build and review work together."
}
}
}
+4
View File
@@ -0,0 +1,4 @@
{
"schemaVersion": 1,
"announcement": null
}
Binary file not shown.

After

Width:  |  Height:  |  Size: 25 KiB

@@ -0,0 +1,23 @@
{
"schemaVersion": 1,
"announcement": {
"id": "staging-announcement-2026-09-14",
"eyebrow": "Staging preview",
"title": "Your next idea starts here",
"description": "Bring your agents and work together in one place. Explore your projects, or learn more about Paperclip.",
"image": {
"path": "assets/6ac073cb3447e26f856b15ec058bb9d4d036b0babff60af88ee0dd49cabfa124.png",
"alt": "Paperclip. Ideas become work."
},
"secondaryLink": {
"kind": "external",
"label": "Learn more",
"url": "https://paperclip.ing"
},
"primaryAction": {
"kind": "route",
"label": "Explore your projects",
"path": "/projects"
}
}
}
+54 -5
View File
@@ -5,10 +5,28 @@ must be attached to the Paperclip issue before the agent chooses a final
disposition. A local workspace path is not enough, because cloud users and
reviewers often cannot access the agent's disk.
Use the helper bundled with the Paperclip skill from the repo root:
## Native runner
When `register_deliverable` is available, use it for files in the bound local or
remote workspace. Supply a workspace-relative `contentRef`, basename `filename`,
`contentType`, exact `byteSize` and SHA-256, `title`, and a stable `idempotencyKey`.
The tool verifies the file, stores an attachment and artifact work product, and
binds it to the response. Generic API tools and a legacy API key are unnecessary.
Wait for the receipt. It includes `attachmentId`, `contentPath`, and
`downloadPath`, along with the existing command, revision, entity references,
and disposition. Reuse the original key after an ambiguous result. A receipt
confirms storage and response binding in Paperclip; it does not confirm delivery
to an external chat provider. If registration fails, use the returned error to
resolve the failure or explain the limitation; do not describe a workspace path
as an uploaded file.
## Legacy adapters
Use Bash to run the helper bundled with the Paperclip skill from the repo root; installed skill files may not retain executable permissions:
```sh
skills/paperclip/scripts/paperclip-upload-artifact.sh path/to/output.webm \
bash skills/paperclip/scripts/paperclip-upload-artifact.sh path/to/output.webm \
--title "Walkthrough render" \
--summary "Rendered walkthrough for review"
```
@@ -27,6 +45,16 @@ It uploads the file to
artifact work product on `POST /api/issues/{issueId}/work-products` by default.
The command prints issue-safe markdown links for the final task comment.
## Task artifact presentation
While a task is open, a new agent attachment, work product, or document opens
the task's Artifacts tab and reveals the side panel (or mobile drawer). This
uses stored object IDs, so it works with either runner. Uploading a file and
registering its work product counts as one arrival. Existing history, revisions,
and repeated query refreshes preserve the user's tab selection. Plans retain
their existing Plan-tab behavior; unregistered user input attachments remain in
the conversation.
## Uploaded Artifacts vs Workspace Files
Use uploaded artifacts for deliverables: videos, PDFs, screenshots, archives,
@@ -96,7 +124,7 @@ available, not the preferred way to deliver files to users.
Upload an `.mp4` render:
```sh
skills/paperclip/scripts/paperclip-upload-artifact.sh dist/demo.mp4 \
bash skills/paperclip/scripts/paperclip-upload-artifact.sh dist/demo.mp4 \
--title "Demo video render" \
--summary "MP4 render for board review"
```
@@ -104,7 +132,7 @@ skills/paperclip/scripts/paperclip-upload-artifact.sh dist/demo.mp4 \
Upload a `.webm` render:
```sh
skills/paperclip/scripts/paperclip-upload-artifact.sh out/walkthrough.webm \
bash skills/paperclip/scripts/paperclip-upload-artifact.sh out/walkthrough.webm \
--title "Walkthrough video" \
--summary "WebM walkthrough render"
```
@@ -113,7 +141,7 @@ The helper detects `.mp4`, `.webm`, and `.mov` content types. If a renderer uses
an unusual extension, pass the MIME type explicitly:
```sh
skills/paperclip/scripts/paperclip-upload-artifact.sh render.bin \
bash skills/paperclip/scripts/paperclip-upload-artifact.sh render.bin \
--title "Demo video render" \
--content-type video/mp4
```
@@ -144,3 +172,24 @@ curl -sS -X POST \
Use `type: "artifact"`, `provider: "paperclip"`, and metadata containing the
uploaded `attachmentId`. The server canonicalizes `contentType`, `byteSize`,
`contentPath`, `openPath`, `downloadPath`, and `originalFilename`.
## Verification
The file-delivery integration suite runs the real helper through queue and
HTTP/2 gateways against a disposable API, database, and storage. It also tests
native registration with generic API tools disabled, duplicate retries, Unicode
filenames, company isolation, and downloads after deleting the workspace.
```sh
pnpm exec vitest run server/src/__tests__/file-delivery-bridges.test.ts
```
To run the same suite on disposable Daytona sandboxes, install the standalone
Daytona plugin's dependencies and set `DAYTONA_API_KEY` in the test process:
```sh
PAPERCLIP_FILE_DELIVERY_DAYTONA=1 pnpm exec vitest run server/src/__tests__/file-delivery-bridges.test.ts
```
The live fixture deletes each sandbox before checking that its attachments
remain downloadable from Paperclip. It does not run unless explicitly enabled.
+290
View File
@@ -0,0 +1,290 @@
# In-app announcements
Paperclip displays one optional announcement card in the board UI. Its feed is
`https://pages.paperclip.ing/announcements/v1/current.json`. The instance fetches
JSON on demand and renders it with native components.
## Operator configuration
- `PAPERCLIP_ANNOUNCEMENTS_ENABLED=false` disables fetching and display.
- `PAPERCLIP_ANNOUNCEMENTS_FEED_URL` overrides the public HTTPS manifest URL.
Credentials, query strings, private destinations and redirects are rejected.
Announcements are independent of telemetry. Feed/media requests originate from
the instance without account IDs, company data, cookies or event tracking. The
host sees ordinary server network request metadata. The browser requests only
its own Paperclip API.
## Authoring and publishing
The shared `announcementManifestSchema` defines the format:
```json
{
"schemaVersion": 1,
"announcement": {
"id": "2026-09-projects",
"eyebrow": "New in Paperclip",
"title": "Your next idea starts here",
"description": "Bring your agents and work together in a project.",
"secondaryLink": { "kind": "external", "label": "Learn more", "url": "https://paperclip.ing" },
"primaryAction": { "kind": "route", "label": "Open projects", "path": "/projects" }
}
}
```
Content is plain text. Every manifest object rejects unknown fields, including
misspellings in actions and media. Optional fields: `image: { path, alt }`,
`animation: { path, alt }`, `expiresAt` (ISO
timestamp), and `minimumPaperclipVersion` (stable `major.minor.patch`). Internal
actions accept stable pages in `ANNOUNCEMENT_APP_ROUTES` and use the selected
company. External HTTPS links open a new tab. Actions only navigate.
Images are `assets/<sha256>.png`, `.jpg` or `.webp`, at most 2 MiB, relative to
the manifest directory. Use an approximately 2.6:1 banner with important content
near the center; mobile crops it shorter. The manifest is limited to 64 KiB.
Run `shasum -a 256 hero.png` to get the image digest, copy the file to
`announcements/assets/<digest>.png`, and use `assets/<digest>.png` in the
manifest. An image correction changes this asset filename while retaining the
announcement ID.
Edit `announcements/current.json`, put its image under `announcements/assets/`,
then run:
```sh
node cli/node_modules/tsx/dist/cli.mjs scripts/publish-announcements.ts announcements --dry-run
```
Set `PAPERCLIP_PAGE_BUCKET`, optionally `PAPERCLIP_PAGE_BASE_URL`, and the page
uploader's namespaced `PAPERCLIP_PAGE_AWS_ACCESS_KEY_ID` and
`PAPERCLIP_PAGE_AWS_SECRET_ACCESS_KEY` (optional `PAPERCLIP_PAGE_AWS_SESSION_TOKEN`),
or `PAPERCLIP_PAGE_AWS_PROFILE`. Ambient AWS credentials also work.
For a host serving a subdirectory, `PAPERCLIP_PAGE_DEFAULT_PREFIX` prepends a
validated path to both S3 keys and public URLs. Use lowercase letters, numbers
and hyphens in each segment, without leading/trailing slashes.
```sh
node cli/node_modules/tsx/dist/cli.mjs scripts/publish-announcements.ts announcements --publish
```
The helper rejects symlinks, validates asset digests and animated HTML, uploads assets first and
the manifest last, and verifies the public manifest and asset headers. It writes
only the resolved announcement prefix; no remote objects are deleted or
infrastructure changed. Allow up
to six minutes for CDN propagation. Manifest caching is five minutes; immutable
assets use one year. Before first publication verify the distribution's active
cache policy has minimum TTL <= 300 and maximum TTL >= 300 for the manifest,
and maximum TTL >= 31536000 for assets. Check the behavior matching each path,
including any referenced cache policy. Public response headers alone cannot
prove the effective cache lifetime or override a higher minimum. See
[AWS cache expiration](https://docs.aws.amazon.com/AmazonCloudFront/latest/DeveloperGuide/Expiration.html).
Retain the ID when fixing copy/images/animations. Use a new ID to announce something new.
ETags improve fetching but never determine redisplay.
## Animated hero media
An announcement can show a self-contained **HTML/CSS animation** in its hero
area. The headline, description, close button and actions remain native
Paperclip controls. Add an `animation` alongside the required static `image`:
```json
"image": { "path": "assets/<image-sha256>.png", "alt": "A team working together" },
"animation": { "path": "assets/<html-sha256>.html", "alt": "Agents plan, build and review work together." }
```
Replace the placeholders with the files' actual 64-character SHA-256 digests.
HTML is UTF-8, limited to 128 KiB, and uses a responsive document with zero body
margin. The hero is about 352 × 136 on desktop and shorter on phones. Use CSS
keyframes, inline styles, system fonts, and visual HTML (`div`, `span`, `p`,
`br`, `strong`, `em`, `b`, `i`) or inline SVG shapes/text (`svg`, `g`, `path`,
`circle`, `ellipse`, `rect`, `line`, `polyline`, `polygon`, `text`, `tspan`,
`title`, `desc`). No scripts, external libraries, links, forms, iframes, images,
SVG SMIL/foreignObject, meta refresh or other embedded resources. CSS URL
requests and imports are blocked by CSP; keep all styling self-contained.
The publisher and server use the same strict DOMPurify allowlist and reject
unsupported markup rather than publishing a silently changed animation.
Paperclip verifies the digest, validates the HTML, and renders the result in an
opaque sandboxed iframe with no permissions. A Content Security Policy blocks
scripts and network resources both inside the card and on direct API visits.
The browser fetches HTML from its own authenticated instance; it never loads
the publisher's page in an unsandboxed frame. The frame cannot receive pointer
or keyboard focus; its accessible description is supplied by `animation.alt`.
See [iframe sandboxing](https://developer.mozilla.org/en-US/docs/Web/HTML/Reference/Elements/iframe).
The animation plays automatically without playback controls. The static image
stays visible while loading and on failure. With reduced motion enabled,
Paperclip does not request or play the animation. Also include a
`prefers-reduced-motion` CSS rule in authored documents for standalone previews.
Animations share the feed's constrained host, three-second server timeout,
bounded cache, request deduplication and fifteen-minute failure cooldown.
Dismissal and ID reuse rules are identical for animated and static cards.
Older Paperclip builds that do not recognize `animation` treat that feed as
unsupported and quietly show no card.
The complete authoring example is `announcements/examples/animated/`. Preview
it with the same staging/test-drive workflow below:
```sh
cp -R announcements/examples/animated .paperclip/announcement-animation-preview
# Edit HTML; recompute its digest and rename it; update current.json.
node cli/node_modules/tsx/dist/cli.mjs scripts/publish-announcements.ts .paperclip/announcement-animation-preview --staging animated-preview --dry-run
node cli/node_modules/tsx/dist/cli.mjs scripts/publish-announcements.ts .paperclip/announcement-animation-preview --staging animated-preview --publish
```
Point the isolated instance at the printed URL and restart it. Verify movement,
reduced motion, mobile sizing, and dismissal across reloads. Try a
missing animation asset: the poster and native controls must remain usable.
Storybook's Animated, AnimatedDark, AnimatedMobile and MissingAnimation stories,
and the design guide, provide local examples without changing the remote feed.
## Preview an announcement before publishing
Use a named staging feed. `--staging <name>` writes
`announcements/staging/<name>/v1/` instead of `announcements/v1/`, so a preview
cannot overwrite the production manifest. With no source directory it uses
`announcements/examples/staging/`, including a sample banner. Commands default
to dry-run unless `--publish` is supplied.
For Paperclip's existing preview host, use the branch preview area that
CloudFront already has permission to read:
```sh
aws sso login --profile paperclip-dev
export PAPERCLIP_PAGE_AWS_PROFILE=paperclip-dev
export PAPERCLIP_PAGE_BUCKET=paperclipai-runner-e2e-history-078455283791-us-east-1
export PAPERCLIP_PAGE_BASE_URL=https://d1p6rlowie26tp.cloudfront.net
export PAPERCLIP_PAGE_DEFAULT_PREFIX=storybook/branches/codex-announcements
# Copy the public fixture into an ignored directory and edit current.json there.
mkdir -p .paperclip
cp -R announcements/examples/staging .paperclip/announcement-preview
node cli/node_modules/tsx/dist/cli.mjs scripts/publish-announcements.ts .paperclip/announcement-preview --staging my-preview --dry-run
node cli/node_modules/tsx/dist/cli.mjs scripts/publish-announcements.ts .paperclip/announcement-preview --staging my-preview --publish
```
Choose a unique staging name for your test and use the printed manifest URL.
The preview host currently uses CloudFront's `Managed-CachingDisabled` policy
for this branch area: edge TTL is zero even though public responses preserve
the five-minute manifest and one-year asset cache headers. This is useful for
preview iteration; it does not verify a production distribution's effective
cache lifetime. For another host, configure its bucket, base URL and optional
prefix, then verify its matching cache behavior as described above.
Create a test-drive configuration in this worktree. Put the feed override in
the **selected instance's `.env`**, not just the invoking shell: test-drive
deliberately clears inherited `PAPERCLIP_*` variables.
```sh
mkdir -p .paperclip/announcement-test-drive/instances/default
# On a new test directory, create this file. On reuse, update these entries
# while preserving the file's existing keys.
cat > .paperclip/announcement-test-drive/instances/default/.env <<'EOF'
PAPERCLIP_ANNOUNCEMENTS_FEED_URL=https://d1p6rlowie26tp.cloudfront.net/storybook/branches/codex-announcements/announcements/staging/my-preview/v1/current.json
PAPERCLIP_ANNOUNCEMENTS_ENABLED=true
PAPERCLIP_DB_BACKUP_ENABLED=false
HEARTBEAT_SCHEDULER_ENABLED=false
EOF
# A fresh test-drive needs a provider key for its initial CEO. Use your usual
# provider environment variable; never put a real key into a manifest or commit.
# Reusing an initialized data directory does not require a bootstrap key.
node cli/node_modules/tsx/dist/cli.mjs cli/src/index.ts test-drive --data-dir .paperclip/announcement-test-drive --no-browser
```
See [test-drive setup](DEVELOPING.md#one-command-isolated-manual-test-drive) for harness/key options.
Open the printed local URL. The app chooses a free port starting at 3100 and
keeps its database under the supplied directory. It creates no tasks or initial
agent run. After onboarding/company selection, allow three seconds for the card.
Before promoting content, verify:
- The image, copy and both actions fit desktop/mobile and both themes.
- A dialog or bottom-left toast temporarily hides the card, then restores it.
- Close or follow a link; reload, switch companies and open another tab/browser.
The same account should keep that ID dismissed.
- Edit copy with the same ID: it stays dismissed. Publish a new ID: it appears
on the next eligible visit. Clearing browser storage alone does not reset
database dismissals; use a new ID or a fresh isolated data directory.
- Try the empty fixture and a URL that returns 404. The dashboard remains usable
with no announcement and no announcement error popup.
Stop and restart test-drive after changing its feed URL or republishing content
when you need immediate results. This clears the server's one-hour feed cache
while retaining dismissal records in the same data directory. Reload or return
to the app after restart; an uninterrupted active tab does not discover cards.
For promotion, validate the reviewed content again, configure the production
host/prefix, and publish without `--staging`. Production's checked-in manifest
remains empty until a real announcement is ready.
## No announcement and withdrawal
The explicit **none** value is JSON `null`, not the string `"none"`:
```json
{
"schemaVersion": 1,
"announcement": null
}
```
Publish this manifest to withdraw an announcement while retaining every user's
dismissed IDs. Restoring an old announcement cannot resurrect it for people who
dismissed it. The ready-to-publish empty fixture is
`announcements/examples/none/current.json`:
```sh
node cli/node_modules/tsx/dist/cli.mjs scripts/publish-announcements.ts announcements/examples/none --staging my-empty-preview --publish
```
A remote **404** is also a normal empty feed: the board API returns HTTP 200
with `null`, clears previous content/ETag, and waits fifteen minutes before
checking upstream again. It produces no announcement warning in server logs or
popup in the UI. Other unavailable or invalid feeds likewise produce no card
or UI error popup; unexpected upstream failures can be logged for operators.
Withdrawal follows the cache/return timing below. Explicit expiration also
removes a visible card when its deadline arrives.
## Timing and persistence
Show after three seconds when opening or returning to Paperclip, after company
selection and onboarding. Dialogs and toasts take priority. Phones show it above
bottom navigation. No automatic timeout, outside-click dismissal or carousel.
Tab visibility controls the return check: moving focus to the address bar or
an adjacent app pane leaves the card visible and does not restart its settling
period. A hidden tab clears the card; becoming visible fetches fresh dismissal
state before showing anything, even if that lookup takes longer than three
seconds.
The instance caches the feed for an hour, deduplicates concurrent fetches, and
uses conditional requests. Failed requests have a fifteen-minute cooldown; no
card appears for unavailable/invalid/incompatible content. Each request has a
three-second deadline. Active tabs do not poll for announcements. Publication
and withdrawal are discovered on a return after cache expiry (normally within
about 65 minutes for returning users).
Closing or following either link saves a unique `(userId, announcementId)`
record in the instance DB, shared across browsers and companies. Its first
write and audit entry commit together; the active company is audit context.
Viewers can dismiss their own card. No-login instances share `local-board`.
Separate installations do not share state.
The browser hides immediately, stores pending writes per account, and retries
on reconnect/return. Failed saves explain that cross-device sync has not
completed. If browser storage is unavailable, state lasts for this visit. Other
tabs close through BroadcastChannel/storage events; another browser refreshes
state on return. Logout clears displayed state and aborts account-bound work.
A failed state lookup never shows a card.
Board-only APIs: `GET /api/announcements/current`,
`GET /api/announcements/:id/image`, `GET /api/announcements/:id/animation`, and `POST /api/announcements/:id/dismiss`
with `{ "companyId": "..." }`. Responses use `private, no-store`. Repeated POSTs
return 204 without duplicate audits. Pending dismissals remain valid after the
feed moves to another ID. The instance retains only the IDs of validated
announcements in a publication registry, so offline retries survive withdrawal
and restarts. A caller-invented ID returns 404 without creating dismissal or
audit rows. This registry is not an archive and records no interaction events.
Production ships with an empty manifest. Design guide / Storybook fixtures are
never used as a production fallback.
+30
View File
@@ -23,6 +23,24 @@ GitHub Actions owns `pnpm-lock.yaml`.
- Pull request CI validates dependency resolution when manifests change.
- Pushes to `master` regenerate `pnpm-lock.yaml` with `pnpm install --lockfile-only --no-frozen-lockfile`, commit it back if needed, and then run verification with `--frozen-lockfile`.
## Trusted PR Workflow
The PR caller uses `paperclipai/paperclip/.github/workflows/pr-trusted.yml@master`.
The AWS runner group `paperclip-public-pr` must allow
`paperclipai/paperclip/.github/workflows/pr-trusted.yml@refs/heads/master`.
New workflow versions merged into master then receive runner access without a
separate SHA allowlist update. Dependabot leaves this first-party reference on
master.
Keep the `.github/**` rule in `.github/CODEOWNERS` and the active master ruleset's
code-owner review requirement enabled. This covers the caller, the trusted
workflow, and CODEOWNERS itself. Existing administrator pull-request bypasses
remain governed by the repository ruleset.
When changing the workflow path or branch, authorize the new reference before
updating the caller. Retain older authorized SHA references while queued runs or
supported reruns still use them.
## Start Dev
From repo root:
@@ -594,6 +612,16 @@ If the `codex` CLI is not installed or not on `PATH`, `codex_local` agent runs f
Local adapters require their corresponding CLI/session setup on the machine running Paperclip. External adapters are installed through the adapter/plugin flow and should not require hardcoded imports in `server/` or `ui/`.
## Project Repository Checkouts
Tasks use every distinct repository attached to their project, including repository-only sources with no local folder. Paperclip creates a managed checkout when no local folder is configured. The selected repository remains at the task workspace root. Other project repositories have editable, independent Git checkouts under `.paperclip-repositories/<name>-<key>`. Workspace hints expose each checkout path to the agent.
When an additional repository has a configured local checkout, Paperclip seeds the task copy from its current commit and uncommitted files. Git-ignored files stay out of that copy. Subsequent task edits stay in the task copy. They do not overwrite the configured source folder. Existing task copies retain their work across runs.
Sandbox staging, including Daytona, transfers each repository's Git history and working files. Restore merges files and commits back into each local task checkout independently. Durable sandbox recovery keeps the same repository snapshots. Normal ignore and workspace exclusion rules still apply. A clone failure stops task preparation with an error so the agent does not start with only part of the project.
If a repository is detached or its source configuration changes, its previous task copy is retained under `.paperclip-runtime/detached-repositories/` and excluded from future sandbox transfers. Referenced projects continue to use the separate read-only multi-project workspace behavior.
## Config Freshness
Agent, project, environment, secret, skill, and workspace config edits are sampled at the next run boundary. A heartbeat that is already running finishes with the config it started with.
@@ -604,6 +632,8 @@ When effective run config changes, Paperclip may intentionally skip a saved adap
Paperclip applies one process-wide scheduler to expensive host-side workspace Git enumeration, including changed-file browsing, runtime/finalization cleanliness guards, and adapter sandbox-sync snapshots. The scheduler defaults to two active scans and a bounded queue of 32. Identical scans of the same canonical worktree share one subprocess, while successful changed-file listings are cached for 10 seconds. Correctness-sensitive runtime guards bypass the result cache.
Workspace snapshots list ignored paths with `git ls-files --others --ignored --exclude-standard --directory -z` so ignored directory contents do not require a full status walk. Snapshot failures retain their typed cause instead of becoming a non-Git-folder result. During pre-provider setup, scan timeouts and queue saturation use the existing two automatic failure retries with a 30-second delay. Cancellation, output limits, and other Git errors stop with specific recovery guidance. See `doc/execution-semantics.md` for the ownership and retry-budget contract.
The cache intentionally trades up to a few seconds of changed-file freshness for stable server latency. The file browser retains an explicit refresh action, does not start its query while the panel or browser tab is hidden, and presents overloads as retryable failures rather than an empty workspace. A full queue returns `503` with code `workspace_git_scan_saturated`; a scan exceeding its wall-clock limit returns `504` with code `workspace_git_scan_timeout`. Both responses include `Retry-After: 1`.
Sandbox Git sync treats only the selected repository root as a clone source. A selected subfolder uses directory sync within that folder, applies the enclosing repository's ignore rules, and does not transfer parent files or Git history.
+17
View File
@@ -192,3 +192,20 @@ Chat instructions require selecting a suitable project, reusing an existing one
The `create_project` runtime tool uses the normal project API with durable idempotency. `list_projects` and `list_project_repositories` support selection. Multiple `repositoryIds` select authorized catalog entries; multiple HTTPS GitHub `repositoryUrls` register existing repositories absent from the catalog. IDs and URLs may be combined, but cannot accompany an explicit `workspace`. URLs do not create repositories on GitHub or grant credentials. Execution uses normal repository access rules. Repository IDs are revalidated against the authenticated run's responsible user and connection grants. Agents should consider proper available repositories, clarify material ambiguity, and use repository-free projects when appropriate for non-code work.
Confirmed project creation appears as a durable card in the shared task transcript, including selected repository links. Tasks are linked inline. Failed creation never produces a success card. Tool evals cover planning/handoff, project/repository selection, retries, permission and mode denials, and ordinary delegation regressions using the production chat directive.
### In-app announcements
An optional announcement card shares product news with board users on opening
or returning to Paperclip. Dismissals persist per user across companies and
browsers within an instance. Operators can disable fetching independently of
telemetry. See [Announcements](ANNOUNCEMENTS.md).
### Agent chat discovery
With Agent Chat enabled, the Chats sidebar always includes the company's
earliest-created agent, plus personal starred agents and up to four other recent
conversations. First use has the same compact rows as returning use. The compose
icon shares a column with stars and appears on hover or keyboard focus (always on
touch). It opens a company-wide name/role search, independent of sidebar membership.
Selecting an agent opens their persistent conversation; it does not reset history
or create a task until the existing first-write flow requires one.
+3 -1
View File
@@ -41,7 +41,9 @@ This script:
3. bundles the CLI entrypoint with esbuild into `cli/dist/index.js`
4. verifies the bundled entrypoint with `node --check`
5. rewrites `cli/package.json` into a publishable npm manifest and stores the dev copy as `cli/package.dev.json`
6. copies the repo `README.md` into `cli/README.md` for npm metadata
6. copies the repo `README.md` into `cli/README.md` for npm metadata, rewriting
repository-relative image assets to raw GitHub URLs pinned to the source
commit
After the release script exits, the dev manifest and temporary files are restored automatically.
+14
View File
@@ -1664,3 +1664,17 @@ with bounded continuation and visible recovery. Preserve explicit approvals,
current task ownership, cancellation, dependencies, and newer task state. See
`doc/architecture/native-status-arbitration.md` for finish feedback and the
provenance-checked cleanup of historical automatic completion reviews.
### In-app announcements
A versioned remote JSON manifest supplies one optional board announcement.
The instance validates/caches content, proxies its raster image, and stores
user-scoped dismissals. Closing or following an action dismisses the ID; copy
edits retain it. Writes are board-only, idempotent and transactionally audited
using an authorized company's context. Viewers may dismiss their own card.
This is an explicit exception to company-scoped business entities: the
preference follows one account across companies on the instance. A separate
instance-level registry retains validated publication IDs, allowing offline
dismissal retries after withdrawal while rejecting caller-invented IDs. It
stores no announcement content, account data or interaction events.
See [Announcements](ANNOUNCEMENTS.md) for API and publishing details.
+26 -14
View File
@@ -9,11 +9,12 @@ The `Cloud readiness` workflow starts for every master push. Its versioned
image, including Sentry resolution and orphan reaping, then publishes the
full-SHA cloud tag. Cloud readiness owns the master trigger so there is one
cloud build per push. Release tags and manual Docker runs retain their callers.
- The full-SHA image and both exact-source npm packages are visible. The
packages are `@paperclipai/shared` and `@paperclipai/db` at
`0.0.0-preview.g<FULL_SHA>`, published through the migrator-only release lane.
Registry metadata must match the full commit, and the database package must
pin the matching shared package.
- The full-SHA image is visible and the exact-source `Cloud migrator artifacts`
workflow has succeeded. Readiness verifies the manifest's GitHub attestation
against the full SHA, canonical master workflow, and GitHub-hosted runner,
then downloads and validates both package archives and the prepared dependency
lockfile. The database package pins the matching shared package. New-version
npm metadata and tarball propagation are outside this path.
The Cloud workflow builds the image with `USER_UID=1001` and `USER_GID=1001`,
matching the managed runtime. This avoids a startup user remap, which can walk
@@ -56,16 +57,19 @@ before that bot's PR merges. Verification must install and test that commit
without waiting for another merge. The generated lockfile stays in the job's
workspace; these checks do not commit it back to the repository.
The artifact wait runs for up to 30 minutes and reports what is missing. Only
an HTTP 404 means publication is pending; authorization errors, upstream outages,
and identity mismatches fail the job. A failed, cancelled, or skipped prerequisite
The artifact wait runs for up to 30 minutes and reports what is missing. A
missing image or an exact-source publisher with no successful run yet means publication
is pending. An earlier successful push or manual run remains valid after a failed
retry because publication is immutable. If all matching runs failed, readiness
fails. An invalid signature, inaccessible or corrupt
bundle, authorization error, or identity mismatch fails the job. A failed, cancelled, or skipped prerequisite
cannot produce a successful readiness job. Retry the failed publication or build,
then rerun the failed readiness workflow jobs to check the same commit again.
## Consumer contract
`Cloud deployable v1` is a source-and-artifact readiness signal. A deployment
consumer must still resolve and pin the image digest and npm integrity/lockfile,
consumer must still resolve and pin the image digest and migrator integrity/lockfile,
validate migration contents and compatibility, and apply its target health gates.
The check creates no release record and deploys no instance. A full-SHA tag by
itself, or a successful migrator dispatch, is not this readiness signal.
@@ -78,9 +82,16 @@ Do not trust a similarly named check from another workflow or a manual branch ru
Order candidates by master ancestry, not job completion time: an older commit
finishing late must not roll a fleet backward. Fail closed on API errors.
Existing npm canary discovery is unchanged by this producer workflow. Consumers
can adopt the versioned signal separately after the workflow has landed and
successfully verified a real master commit.
Cloud consumers must enable `CLOUD_HARNESS_DIRECT_MIGRATOR_ARTIFACTS` before
this gate is adopted: readiness no longer promises preview npm availability.
The automatic npm-only migrator dispatcher has been removed. Manual
`release.yml` runs with `channel=cloud-migrator`, branch previews, and stable
releases retain their npm publisher for legacy consumers and rollback.
For rollback, restore the npm dispatcher and gate together before disabling the
cloud direct-artifact switch. Already-created releases retain their immutable
archive URLs and lockfiles; keep those objects available. The master producer
can be retried independently without republishing or overwriting a valid bundle.
## Timing and rollout
@@ -151,9 +162,10 @@ its trusted-publisher identity.
Before enabling the switch, deploy the separate Fleet and restrict its GitHub
runner group to repository ID `1170821064` and these workflows at
`refs/heads/master`: `cloud-readiness.yml`, `cloud-artifacts.yml`,
`refs/heads/master`: `cloud-readiness.yml`,
`release-verify.yml`, `runner-chaos-evals.yml`, and `release.yml`. Do not authorize
PR-controlled workflow versions. PR placement retains its independent pinned
PR-controlled workflow versions. The direct migrator producer always uses
GitHub-hosted runners and needs no AWS runner-group authorization. PR placement retains its independent pinned
workflow and six-account author/actor allowlist.
Disable the switch and rerun the whole workflow to restore GitHub-hosted
+22
View File
@@ -29,6 +29,28 @@ included: clearing the plain variable to blank disables injection even while
a base64 value is still deployed. Everything else about the snippet is
unchanged.
Base64 does not defeat every firewall. Some decode the value before matching,
so they reject a base64 snippet whose decoded bytes still contain script
markup. Deliver a bare script body (below) through one of these.
## Bare script body
Set the value to the script body alone — the JavaScript with no surrounding
`<script>` element:
```sh
PAPERCLIP_CLOUD_UI_SNIPPET_B64="$(base64 < snippet.body.js)"
```
The server wraps a bare body in a `<script>` element before it inserts it. The
first non-whitespace character decides the form: a value that starts with `<`
is treated as markup and injected unchanged; any other value is treated as a
body and wrapped. This applies to both the plain and the base64 variant.
The body must be safe to embed inline. It must not contain a literal
`</script>`, which would close the wrapper early. Because the value carries no
`<script` marker, a firewall that decodes base64 before matching passes it.
## Plain closed beta
Set the value to this standard embed, replacing `YOUR_CHAT_APP_ID` with the
+65 -13
View File
@@ -93,6 +93,9 @@ subsequent agent creation fails or is cancelled.
## Runtime isolation
`prepareManagedAiRuntime` is shared by runs, environment tests, and adoption.
Claude ACP validates working directories on the selected execution target. A
sandbox directory does not need to exist on the Paperclip server. When the agent
has no configured directory, the test uses the remote target's working directory.
It checks responsible identity, membership, compatibility, connection health,
human audience and agent installation before reading credentials.
Missing credentials produce an actionable configuration failure; responsible-user
@@ -105,17 +108,31 @@ grant's credentials. Inherited credential variables are cleared. Conflicting
project authentication and provider-routing overrides are rejected. Managed
failure cannot reactivate host or legacy credentials.
Subscription invocations take a grant-scoped transaction advisory lease. The
reserved database client keeps one transaction open until cleanup, including on
transaction-pooling proxies such as PgBouncer. Session-level advisory locks must
not be used here: a pooled connection can return to a different backend for
cleanup and leave the original lock behind. The lease transaction disables its
idle timeout and contains no application data writes; cleanup rolls it back.
Two
different users' grants can run concurrently; a second invocation of the same
subscription receives a retryable busy response while it is in use. Refreshes
are merged only into the originating active grant, with reconnect/revocation
version checks. Temporary homes are removed on normal completion or failure.
A subscription invocation takes no lease. Two invocations of one grant, from
the same or a different provider account, run at the same time. At cleanup,
each invocation re-reads the credential stored at that moment under a row
lock on the grant, then compares it against its own refreshed copy using the
provider's own freshness field: Codex compares `last_refresh` and bounds it
against the host clock; Grok compares `expires_at`. The newer credential
persists; a tie or an unparseable freshness value keeps the stored
credential, so a spent single-use refresh token never overwrites a good one.
Refreshes are merged only into the originating active grant, with a
revocation check. Temporary homes are removed on normal completion or
failure.
A fresh task execution cannot enter subscription contention. The freshest-write
rule above resolves the conflict instead. A run that already entered this wait
keeps a durable scheduled retry, checked every 60–120 seconds. The task shows
“Waiting for AI subscription”. It does not request a reconnect, and it does not
consume its provider-failure retry allowance.
Each attempt rechecks task eligibility, ownership, budget, and current credential
access. Revocation and other configuration failures still require user action.
Authorized comment wakes that started as non-assignee runs can resume without
claiming the assignee’s execution lock. Admission records this authority while
holding the task and run locks. A reassignment during preflight cannot grant it.
Assignee retries must still own that lock.
Already-started native sessions retain their existing same-run recovery path;
they must not be replaced by a fresh execution with a pre-provider receipt.
Session reuse includes grant identity, responsible user, and credential
generation. A changed identity starts a fresh provider session. Managed native
@@ -195,8 +212,8 @@ unmanaged legacy agents retain their existing authentication paths.
`server/src/__tests__/ai-connections.test.ts` exercises storage, isolation,
defaults, human audiences, agent access, reconnect races, refresh ownership,
subscription locking, migration replay, and redacted API failures against a real
embedded database. Existing login, adapter, tool, and channel suites cover their
concurrent subscription write-backs, migration replay, and redacted API
failures against a real embedded database. Existing login, adapter, tool, and channel suites cover their
shared integration paths. The onboarding tests cover managed reuse and keeping a
successfully connected account after failed agent creation.
@@ -251,3 +268,38 @@ revoking credentials or submitting work. Delete the disposable instance and revo
its provider key after the test; failed tests may leave a paused task for inspection.
Authenticated public deployments must configure a trusted runtime host (`PAPERCLIP_TRUSTED_MCP_RUNTIME_HOST` or `PAPERCLIP_TOOL_RUNTIME_TRUSTED_HOST`) before offering server-host subscription login, matching the local stdio runtime boundary. Health reports this capability so setup can offer a supported environment or API key instead of an unusable terminal command. Private authenticated self-hosted instances support isolated local login without that extra setting. Isolated Claude credential files must be private, owned by the server user, bounded, and free of symlinks.
### Hiring and delegated work
When a managed agent creates or hires another agent without an explicit AI binding
or adapter auth setting, the server inherits its compatible managed connection choice.
Explicit credentials, blank overrides, credential directories, and provider routing
settings for the child provider take precedence. Unrelated provider keys do not
suppress the default. Unmanaged parents keep their existing authentication path. The new agent resolves
the responsible user's account at execution time; it never copies the parent's
credentials or identity. Same-provider hires preserve subscription/API-key choice.
A different provider selects the responsible user's default for that provider.
Native Codex and ACPX/Claude provider selections follow the same compatibility rules.
Hiring may succeed before that personal account exists or while it needs repair,
including hires awaiting board approval. The first assigned task then shows an AI
connection card. First-time setup presents the provider's subscription/API controls
inside the task. Connecting installs access for that agent and resumes the pending
work automatically. Explicit incompatible bindings and shared-account permission
denials still fail; hiring never expands a restricted shared account's audience.
Concurrent runs of one subscription do not wait for each other. No credential
lease exists to hold them, so a fresh task execution cannot enter a contention
wait. A run that already entered this wait keeps its scheduled retries. It does
not request new credentials, and it does not consume the provider-failure retry
allowance. Each retry revalidates the account, and existing run-dispatch rules
still suppress cancelled, reassigned, or otherwise ineligible work. An assignee
retry must still keep execution-lock ownership at scheduling, promotion, and
dispatch.
`server/src/__tests__/agent-hire-ai-connections.test.ts` covers both creation routes,
both providers and methods, approval gates, native provider mapping, shared access
boundaries, and concurrent runs of one subscription for both providers. The opt-in
[`tests/hiring-ai-connections/README.md`](../../tests/hiring-ai-connections/README.md)
describes real browser hiring, subtask, connection, and automatic-resume checks on
local and Daytona environments, plus the production component Storybook checks.
+29
View File
@@ -0,0 +1,29 @@
# Connector icons
Use the shared `AppLogo` component and bundled artwork in `ui/public/brands/apps`.
Keep the existing gray rounded frame, caller size and border, and contained image
padding. Preserve vendor shapes and colors; use a separate dark asset only when
the light artwork is unsuitable on the dark frame. Do not invert or stretch marks.
The public manifest contains only identity, catalog visibility, artwork paths,
and optional aliases. Omit `darkAsset` when both themes use the same file. Keep
matching logo paths in the app definition. Brand-library membership does not
enable a connector. Google People and Workspace Search share the Google mark.
Use reviewed vendor or supplied source files. Keep source research and review
records outside the browser-served manifest. Never add credentials or private
review links to public assets. Render SVGs as images, not inline HTML. The
structural safety check rejects common active SVG features; it is not a general
sanitizer for untrusted uploads.
Before submitting artwork, run:
```sh
node scripts/check-app-brand-assets.mjs
node --test scripts/app-brand-validation.test.mjs
pnpm exec vitest run ui/src/lib/app-brand-assets.test.ts ui/src/pages/apps/AppLogo.brand-assets.test.tsx packages/shared/src/app-definitions.test.ts
```
Run the structural artwork check locally. Review the Storybook canonical icon registry
in light and dark themes at 24–48px, then inspect affected product surfaces. Check
contrast, optical size, native details, and the existing image-error fallback.
+8 -6
View File
@@ -5,6 +5,8 @@ and shipping Paperclip app connections.
Status: canonical end-to-end authoring guide for Apps v2 catalog connections.
For connector artwork, follow [Connector icons](./CONNECTOR-ICONS.md): fixed gray Paperclip frames, authentic vendor artwork, explicit theme variants, optical fit and exact provenance. Brand-library additions do not activate connectors. Use the shared registry/resolver and branding generator; do not introduce per-screen logos or outer-surface overrides.
This runbook is the repeatable, agent-executable procedure for adding a vendor
to the Apps catalog as data, not as a plugin. It follows the accepted
connections framework in [PAP-13211](/PAP/issues/PAP-13211), the first-30
@@ -87,7 +89,7 @@ scripts/ingest-app-definitions.mjs # human-authored definition so
packages/shared/src/app-definitions/<slug>.json # generated definition
packages/shared/src/app-definitions.generated.ts # generated registry
ui/public/brands/apps/<slug>.svg # official, sanitized mark
ui/public/brands/apps/manifest.json # branding provenance
ui/public/brands/apps/manifest.json # runtime branding paths
packages/shared/src/app-definitions.test.ts # manifest/provider assertions
```
@@ -571,15 +573,15 @@ only a runtime image-failure fallback.
external executable content, or unsafe references.
5. Save assets under `ui/public/brands/apps/`. Add a `-dark` variant only when
the normal mark loses contrast in dark mode.
6. Add the provider to `ui/public/brands/apps/manifest.json` with slug, local
asset, optional dark asset, official source URL, exact upstream asset URL,
asset type, visibility, and dark-variant requirement.
6. Add the provider to `ui/public/brands/apps/manifest.json` with slug, name,
local asset, optional dark asset, visibility, and optional aliases. Keep source
URLs and verification notes in the review record, outside the public manifest.
7. Let the ingestion script derive `branding.logoUrl` and `darkLogoUrl` from the
provenance manifest.
runtime manifest.
The manifest test decodes PNG headers, requires at least 128 by 128 pixels,
sanity-checks SVG markup, verifies files exist, and requires store-visible
definitions and visible provenance entries to match exactly.
definitions and visible manifest entries to match exactly.
### Phase 5: Author the definition at the durable source
+16 -3
View File
@@ -50,6 +50,18 @@ step through review, access and install. Reads are enabled for review;
state-changing actions start off; newly discovered actions are quarantined until
reviewed. That is the same treatment a curated connection gets.
For **Just me**, the first probe of a new URL with no supplied credentials runs
before a personal authorization exists. If the server requires OAuth, Paperclip
creates the personal grant only after sign-in succeeds. If the server is public,
Paperclip creates a personal grant with no credentials after the probe succeeds.
That successful public probe saves the draft identity and its grant. A later
catalog-refresh failure leaves them available for retry instead of undoing a
grant that another setup attempt may already be using.
Catalog and default-profile writes after that probe are atomic: a failure rolls
back that step while retaining the established draft identity.
Later health checks still require the user's authorization and return an
actionable `422` error when it is missing.
### Advanced authentication
Collapsed by default. Open it when the server's docs are specific:
@@ -177,9 +189,10 @@ through this page with either a key or browser sign-in.
## Verifying
Deterministic coverage lives in
`server/src/__tests__/generic-mcp-connection.test.ts`, which stands up an
in-process MCP server plus authorization server. It needs no network and no
vendor credentials, and every case connects by URL without naming a gallery app.
`server/src/__tests__/generic-mcp-connection.test.ts` uses a simulated MCP/OAuth
provider and a real loopback HTTP server for the personal public-URL regression.
It needs no vendor credentials. Tests create an isolated PostgreSQL database
and close their servers after use.
A credentialed vendor smoke (for example live PostHog OAuth) may be recorded by
QA but is not required for deterministic verification.
+5 -2
View File
@@ -144,8 +144,11 @@ Sources: [hosted MCP](https://docs.railway.com/ai/mcp-server),
[GraphQL](https://docs.railway.com/integrations/api),
[SSH](https://docs.railway.com/cli/ssh), and
[official CLI GraphQL schema and commands](https://github.com/railwayapp/cli/tree/ac4f16e5f3db047b941bf0b9ac3be388e7c73697).
Brand marks are sanitized from Railway's own homepage inline SVG; provenance is
in `ui/public/brands/apps/manifest.json`.
Brand marks were sanitized from the inline SVG at
[Railway's official homepage](https://railway.com) on 2026-09-13. The original mark
contains the Railway train silhouette; the dark variant changes only its fill
for contrast. The public
manifest contains runtime artwork paths; this record retains source provenance.
## Recovery
+6
View File
@@ -418,3 +418,9 @@ Per-component rationale:
| Setup completion | `ConnectionSetupCompletionScreen` in the shared setup module | Page and dialog; identity, granted agent access and enabled actions |
Independently addressable examples live under `Connections/In-task connections` in Storybook. The task composer remains available while a card is pending. These components use the existing token and primitive layers.
## Announcements
- `AnnouncementCard`: image, eyebrow, headline, description, navigation links and dismissal; accepts an announcement and `onDismiss`.
- `AnnouncementWell`: one app-shell placement that owns eligibility, dismissal sync, modal deferral and toast priority. Use only once in Layout.
- Preview variants live in `/design-guide` and Storybook under `Announcements/AnnouncementCard`.
+54
View File
@@ -376,6 +376,17 @@ If the committed update assigns the issue to a user, clears the agent assignee,
Plain text is not assignment. Writing an agent's name, role, or team label in a comment does not change ownership and does not create an agent wake. Agent routing from comment text requires a structured agent mention that resolves inside the company, an explicit `assigneeAgentId` mutation, or an existing current agent assignee receiving normal issue-thread feedback.
A delegation comment from the current assignee's run on this parent must not start competing parent work when the named worker already owns the referenced child. This applies to issue updates with a comment and standalone comments. Verify the source run's company, agent, and parent-task context. Then verify that the comment references the child's identifier, the child's `parentId` names this parent, and the child belongs to the same company and is assigned to the mentioned worker. Apply child-aware routing only in either of these states:
- The parent is `blocked` and the child is `in_progress`. The child must have a blocker edge to the parent. The child's execution or checkout run must still be `running`, belong to that worker and company, and name that child in its run context. Verify comment and mutation access to the child, then retain the parent comment and append a linked copy on the child. Preserve the full comment, author, source run, responsible-user attribution, and source trust. Target the normal mention wake at the child and its new comment ID, with explicit resume and follow-up intent. This keeps new feedback available to the worker and lets the existing queue serialize a child continuation behind its current execution.
- The parent and child are both `done`. The assignee's closing comment must not start another worker run for the completed delegation. A blocker edge is not required after completion: a fast child can finish before the lead needs to record a wait. New agent work must use explicit `resume: true`, a status change, or a new assigned task. Explicit resume moves the parent out of `done` before this rule runs; the comment's prose alone does not restart completed work.
The completed-delegation comment remains on the parent without a worker wake. Neither path changes ownership. Board-user comments and unrelated mentions retain their normal wake behavior. If multiple referenced children qualify for the same worker, the child or run no longer meets these conditions, child comment or mutation access is denied, or a lookup or copy fails, use the normal parent mention path. Do not parse mentions again while copying a comment, which would create another routing loop. Completion of the child still uses the existing blocker-resolution wake for the parent's assignee.
The parent may receive a closing comment before its assignee changes the status to `done`. Recheck the completed-delegation rule when releasing that parent execution, before promoting a deferred mention. On the same transaction, verify the final parent state, finishing run, and every original queued or deferred comment ID. Each comment must belong to this parent and company, come from its assignee's finishing run, and reference exactly one completed direct child assigned to the mentioned worker. A link to the parent itself is allowed; any other extra issue reference keeps the normal mention path, including an unknown or foreign reference. Mixed human, other-run, unrelated, or ambiguous input retains its normal wake path. Explicit continuation and interaction requests also retain their normal path.
Accepted agent feedback must survive a child changing to `done` before its active run exits. Deferred wake promotion may reopen that completed child only for its current assignee, with explicit agent resume intent and live tracked comments from another author. Claim promotion before reopening. Cancelled tasks, deleted comments, self-authored comments, empty continuations, and agent continuations without explicit intent keep their existing suppression rules. Normal pause, ownership, authorization, and budget gates still apply.
Pause and tree-control previews should make the same distinction visible. They should report whether the affected subtree contains live running work, queued wakes, agent-owned work, or only human-owned/static issues, so a pause after a handoff does not look like it interrupted agent execution when no agent execution path existed.
### Adapter-backed workspace coherence
@@ -400,6 +411,26 @@ Workspace incoherence feeds into the same non-terminal liveness and stranded ass
For runtime-created `git_worktree` execution workspaces, branch coherence is part of workspace coherence. The persisted execution workspace branch is the recorded branch for future dispatch. Reusing that workspace must verify that the worktree is still registered and that `HEAD` is on the recorded branch. Successful run finalization must perform the same check before recording `workspace_finalize=succeeded`. If the run switched to a publishing/PR branch without updating the execution workspace record, finalization may auto-restore the recorded branch only when the worktree is clean, still registered, and the recorded branch points at the current `HEAD`; the repair is recorded as a workspace operation before the successful finalize row. If that safe repair cannot be proven, finalization records a failed workspace finalize and the run fails with bounded evidence for the expected and actual branch. A branch change is sanctioned when a control-plane path updates the execution workspace record before finalization, when publishing work happens in a separate worktree and the managed issue worktree remains on its recorded branch, or when the finalizer performs this clean same-commit restoration.
### Workspace scan failures before provider startup
Repository discovery distinguishes an ordinary folder from a failed Git read.
A timeout, full scan queue, cancellation, output limit, or Git failure must keep
its typed cause through workspace preparation and run persistence. It must not
be reported as a missing repository or fall back to an unfiltered directory copy.
When workspace preparation fails before provider work starts, scan timeouts and
queue saturation use the existing durable failure budget: two automatic retries,
30 seconds apart. The scheduled successor is persisted before execution is
released. Restart and duplicate wake handling reuse that successor. Normal task,
ownership, pause, dependency, approval, and budget gates still apply. Existing
workspace content is retained, and incomplete temporary clones are not published.
Cancelled scans, output-limit failures, and other Git failures do not authorize
an automatic setup retry. Exhaustion or an unsafe retry opens the source-scoped
recovery path with the specific scan cause and an operator action. Generic
stranded-work recovery must not grant another budget for these errors. This
does not automatically replay historical generic `setup_failed` runs.
### ACP startup handshake bound
An adapter-backed live path also requires that the ACP startup handshake itself cannot hang forever. The engine bounds the handshake with a fixed startup deadline and a poll of the duplex control-channel disposition. Either condition ends the handshake and reports a closed, typed code, so the issue can reach a settled disposition instead of staying `in_progress` with no observable next action.
@@ -410,6 +441,8 @@ The handshake failure code is distinct from a session-identity mismatch. A timeo
An explicit recovery action is a typed liveness repair path for a source issue. It is the recovery primitive; the action can be rendered directly on the source issue or backed by a separate recovery issue when the repair needs its own work item.
A new user message can continue a terminal native run whose process fields were cleared before local stop receipts existed. Admission must verify the exact run, runner, workspace, and provider session in the retained suspended state, with no active provider turn, pending tool call, or undelivered output. Missing or mismatched state keeps the hold. A later recorded process launch also keeps the hold until its stop is verified. Normal assignment, decision, controller, environment cleanup, and active-run gates still apply. The message starts one fresh conversation turn; it does not replay the failed run, reset its recovery budget, or certify unknown action outcomes.
The task thread exposes the existing guarded Retry action for failed or timed-out legacy conversation runs. Where the server supports an explicit new attempt after a stopped legacy conversation, the thread must not hide that action solely because the old run still has a recovery-needed projection. Native and process recovery holds, pending decisions, active execution, and other retry gates remain in force. When a gate hides Retry, the thread says the message is preserved instead of promising an unavailable action. This presentation change does not rewrite historical outcomes or certify prior actions.
A valid recovery action must name:
@@ -1237,3 +1270,24 @@ awaits it. That background invocation observes rejection immediately, including
when a remote sandbox has already disappeared. The owner's awaited close still
receives the original failure; containment never fabricates a successful close
or permission to reuse an unverified execution.
### Assigned connections in native ACPX sessions
Native ACPX sessions register the assigned Paperclip MCP gateway alongside the
task tool bridge. Gateway calls retain the existing connection grants and action
approvals. Missing assigned bindings and names that collide with the task bridge
stop admission. Upstream credentials remain with the gateway; providers receive
its scoped access binding. The qualified ACPX sidecar receives the gateway name,
URL, and token together through the launch allowlist; unrelated environment
secrets remain excluded. This does not restrict arbitrary network access to a
public service outside the gateway.
### Use real connection requests (2026-09-14)
When a user asks to connect a known service, the agent searches for that service
and uses `connection_request` if setup is needed. The agent must not ask the same
permission again or copy Connect / Not now into a generic question. A generic
question does not start setup. The real connection card keeps user identity,
access grants, the decision, and continuation together. This guidance does not
approve a connection or bypass its normal user decision.
+13
View File
@@ -375,11 +375,24 @@ sends, so an operator can read what the feature does before turning it on.
Each Sentry integration name below is verified against the default
integration list of `@sentry/node@10.71.0` and `@sentry/browser@10.71.0`.
**Server attribute this feature sets**
- `server_name` — every server event carries the host name of the process.
The `@sentry/node` client already sets this value by default when the
operator does not pass a `serverName` option; this feature passes the
value directly, so the server keeps sending it even if a later SDK
version changes its default. To send a different value in place of the
host name, set the environment variable `SENTRY_NAME` to that value.
**Server events this feature adds**
- An Express `HttpError` with `status >= 500`.
- Any unknown throw that is not a `ZodError`. It always answers 500.
- A server startup failure.
- A run that ends with the status `failed` or the status `timed_out`. The
event carries five context fields: `taskId`, `runId`, `errorMessage`,
`errorCode`, and `agentAdapter`. The server redacts the error message and
the error code before it sends the event.
**Server events the default integrations add**
+11
View File
@@ -84,6 +84,17 @@ pnpm build
## Supported alpha surface
### CreateOS sandbox provider
The in-repo [`@paperclipai/plugin-createos`](../../packages/plugins/sandbox-providers/createos/README.md)
package implements environment lifecycle hooks and incremental managed-process
output and binary workspace transfers using CreateOS's public HTTP API. It does
not advertise interactive login or template capture. Install
the built package by local path; its optional managed-image catalog key is
`createos`. The package README describes configuration and the opt-in live smoke.
### Worker APIs
Worker:
- config
+67
View File
@@ -37,6 +37,12 @@ or existing package identity mismatches fail the workflow. Retries reuse matchin
published artifacts, including a shared package published before a DB publish
failure. Allow npm's visibility polling to finish before retrying.
The publisher validates both package archives first, then submits each missing
package without waiting for the other to become visible. One visibility poll
checks both accepted packages, so their registry propagation delays overlap.
The publisher succeeds only after both packages pass the identity and
distribution-pin checks. A visibility timeout names the package still missing.
The final `stack-deploy-result` artifact contains `result.json` with contract
version 1, request ID, SHA, stage `build`, and status `ready`. It expires after
30 days. This confirms artifact availability; it does not certify a tenant deploy.
@@ -116,3 +122,64 @@ node scripts/preview-artifacts.mjs pack /path/to/source /path/to/output FULL_SHA
This executes source build scripts. Keep output outside the repository and use an
environment without publishing or cloud-admin credentials.
## Direct cloud migrator artifacts
`cloud-migrator-artifacts.yml` builds the DB and shared preview packages on each
canonical `master` push. A manual run also requires `master` and uses its exact
commit. This workflow runs on GitHub-hosted runners. It has no PR trigger.
The build resolves a complete npm lockfile from the two local archives. It then
pins their download URLs to immutable, content-addressed objects. The cloud
migration runner can use `npm ci` with this lockfile before either new package
version is available on npm. Existing external dependencies still come from npm
and carry SHA-512 integrity pins. Package lifecycle scripts remain disabled.
Before upload, the build job smoke-installs the real archives and their complete
external and bundled dependency graph with an empty npm cache. It imports both
installed packages. This check uses local archive URLs because public objects
do not exist yet; all versions and integrity pins remain unchanged.
Artifacts use the existing runner-history S3 bucket and CloudFront distribution,
under the separate `cloud-migrators/v1/` prefix. The manifest at
`https://d1p6rlowie26tp.cloudfront.net/cloud-migrators/v1/<full-sha>/manifest.json`
records the full source SHA, exact preview version, and the size, URL, and SHA-512
hash of each archive and the lockfile. Blob URLs include the content hash.
The publisher validates the complete bundle before any write, writes all blobs
before the manifest, and verifies downloads through the public endpoint.
A retry reuses a complete existing manifest after verification. The publisher
also creates a GitHub/Sigstore build-provenance attestation for the manifest
before upload. This independently binds all package and lockfile content hashes
to the canonical workflow, master ref, repository identity, and source commit.
Cloud must verify this signature and its certificate claims before accepting
the executable archives; hashes served by the artifact store alone are not
sufficient provenance.
The build job has no AWS credential. The publish job downloads only the four
fixed files, validates them, and uploads them without executing their code.
The dedicated `paperclip-cloud-migrator-github` OIDC role trusts only
`repo:paperclipai/paperclip:ref:refs/heads/master`. Its policy permits prefix
listing and conditional `PutObject` calls in this one prefix. It permits no
object deletion or overwrite. PRs, including allowlisted PRs, cannot assume it.
The deploy policies are checked in under `.github/cloud-migrator-deploy/`:
- `trust-policy.json`: the role trust policy.
- `upload-policy.json`: the role's inline permission policy.
- `cloudfront-read-statement.json`: append this statement to the existing bucket
policy, preserving its other statements and public-access blocks.
There is no lifecycle expiry on this prefix. Keep referenced artifacts for
rollback; deleting them can prevent a fresh migrator install for an old release.
Cloud readiness consumes the signed direct bundle after its exact-source
publisher succeeds. It retains source verification, image identity, and the
cloud runner's integrity and migration compatibility checks. The cloud direct
artifact switch must be enabled before adopting this gate. Automatic npm-only
migrator dispatch is removed; explicit npm previews and manual migrator runs
remain available. See `doc/cloud-build-readiness.md` for coordinated rollback.
Local verification:
```sh
node --test scripts/cloud-migrator-artifacts.test.mjs
node scripts/cloud-migrator-artifacts.mjs verify <full-sha>
```
+7 -3
View File
@@ -28,9 +28,13 @@ Old Overview URLs and saved Overview preferences redirect to Configuration.
No schema migration is required. Selected repositories are normal project workspaces
with `metadata.githubRepositoryId`. Existing manual `repoUrl` workspaces remain
editable through Configuration and the workspace API. One workspace remains primary;
additional repositories do not change the existing runtime workspace-selection or
responsible-user credential rules. A repository selection never delegates credentials.
editable through Configuration and the workspace API. One workspace remains primary.
Tasks materialize the other distinct repositories as editable checkouts inside their
workspace, including when no local folders are configured. Local execution and sandbox
staging use the same layout; sandbox restore preserves each repository's Git history.
See [Project Repository Checkouts](DEVELOPING.md#project-repository-checkouts) for paths,
ignore rules, and reuse behavior. Responsible-user credential rules still apply.
A repository selection never delegates credentials.
## Discovery and setup
+19
View File
@@ -159,6 +159,14 @@ Provider identity diagnostics remain in the local run log. They record the notif
Recovery lifecycle events retain the original structured failure code, retry attempt, next retry time, and predecessor/successor identifiers. Durable status delivery uses an idempotency marker; delivery grants no provider authority. Failed publication is retried without repeating provider work. These records are not first-party Telemetry.
Bounded retry exhaustion writes one lifecycle receipt per run, retry reason,
scheduled attempt, and retry limit. Repeated or concurrent recovery checks reuse
that receipt, including receipts from earlier builds, without advancing the event
sequence or publishing another live event. Attention reads select the latest
matching receipt in PostgreSQL and project only the run's issue/task identifiers
from its context, so historical duplicate receipts cannot multiply run contexts
in server memory. Existing duplicate events do not require deletion or migration.
## Codex resume usage snapshot
The native runner retains a bounded local `harness.diagnostic` event with code
@@ -168,3 +176,14 @@ and records cumulative usage counters. It does not include provider credentials
or message content. The event establishes the accounting baseline; it is not a
new billable usage receipt or a user-facing provider warning. Other provider
identity checks remain in force.
## AI subscription contention
A fresh task execution cannot enter this wait. A run that already entered this
wait writes an informational `lifecycle` event to the local run log. Its
payload contains only `retryScheduled`, a boolean that reports
whether the scheduler created a retry.
The message distinguishes an automatic retry from work that is no longer eligible.
This pre-provider wait records `ai_connection_busy` on the cancelled run and does
not consume the provider-failure retry allowance. The event contains no credentials
and creates no Telemetry or OpenTelemetry export.
+4
View File
@@ -53,6 +53,10 @@ can be supplied. Routes still validate payloads and enforce permissions.
Requests have a 30-second HTTP timeout, 16 KiB URL limit and 10 MiB payload/response
transfer limit. Responses above 24 KiB and binary responses become company-owned
assets with retrievable references; text previews are limited to 2,000 bytes.
Tool responses identify the HTTP route with `apiOperationId`. The native protocol
reserves `operationId` and `callId` for semantic tool-call identity; API metadata
must not masquerade as that envelope. Saved mutation receipts are normalized at
the tool boundary as well, without repeating their HTTP request.
All redirects are refused. Oversized or interrupted mutation responses have an
unknown outcome, requiring inspection before another mutation.
Mutation responses with HTTP 5xx, HTTP 408, redirects, or malformed JSON also
+2 -1
View File
@@ -59,7 +59,7 @@
"smoke:posthog-live": "node scripts/smoke/posthog-live.mjs",
"smoke:pipelines-tutorial": "./scripts/smoke/pipelines-tutorial-smoke.sh",
"smoke:terminal-bench-loop-skill": "node scripts/smoke/terminal-bench-loop-skill-smoke.mjs",
"test:release-registry": "node --test scripts/verify-release-registry-state.test.mjs scripts/release-package-map.test.mjs scripts/check-release-package-bootstrap.test.mjs scripts/check-no-git-push.test.mjs scripts/release-lib.test.mjs scripts/release-registry-versions.test.mjs scripts/link-plugin-dev-sdk.test.js scripts/acpx-patch-packaging.test.mjs scripts/service-onboard-smoke.test.mjs scripts/docker-onboard-smoke.test.mjs scripts/preview-artifacts.test.mjs scripts/select-cloud-cache.test.mjs",
"test:release-registry": "node --test scripts/verify-release-registry-state.test.mjs scripts/release-package-map.test.mjs scripts/check-release-package-bootstrap.test.mjs scripts/check-no-git-push.test.mjs scripts/release-lib.test.mjs scripts/release-registry-versions.test.mjs scripts/link-plugin-dev-sdk.test.js scripts/acpx-patch-packaging.test.mjs scripts/service-onboard-smoke.test.mjs scripts/docker-onboard-smoke.test.mjs scripts/preview-artifacts.test.mjs scripts/cloud-migrator-artifacts.test.mjs scripts/select-cloud-cache.test.mjs",
"storybook-visual:baseline": "node scripts/storybook-visual-baseline.mjs",
"test:storybook-visual": "node scripts/storybook-visual-baseline.mjs download && node scripts/storybook-visual-baseline.mjs verify && pnpm build-storybook && npx playwright test --config tests/storybook-visual/playwright.config.ts",
"test:storybook-visual:update": "node scripts/storybook-visual-baseline.mjs download && pnpm build-storybook && npx playwright test --config tests/storybook-visual/playwright.config.ts --update-snapshots && node scripts/storybook-visual-baseline.mjs pack",
@@ -69,6 +69,7 @@
"test:e2e:runner:dashboard": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/dashboard-regenerate.ts",
"test:e2e:runner:models:update": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/openrouter-models-update.ts",
"test:e2e:runner:history:publish": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/history-publish.ts",
"test:runner-recovery": "vitest run server/src/services/native-runtime/native-replacement-evidence.test.ts server/src/services/native-runtime/stopped-codex-turn.test.ts server/src/services/native-runtime/native-safe-replacement.test.ts",
"test:e2e:runner:unit": "vitest run --config tests/runner-e2e/vitest.config.ts",
"test:e2e:runner:typecheck": "tsc -p tests/runner-e2e/tsconfig.json",
"test:e2e:runner:report": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/report.ts",
@@ -1619,6 +1619,35 @@ describe("shared ACPX engine runtime behavior", () => {
},
);
it.each(["OPENAI_API_KEY", "CODEX_API_KEY"] as const)(
"selects Codex ACP API-key authentication when only the host process provides %s",
async (apiKeyName) => {
const root = await makeTempRoot();
const codexHome = path.join(root, "codex-home");
await fs.mkdir(codexHome, { recursive: true });
// Simulate a local launch that inherits a provider key from the host
// process environment. No adapter config sets the key directly, so the
// launched env only receives it through host projection.
vi.stubEnv(apiKeyName, "sk-host-inherited-key");
try {
const { sessionInputs } = await runExecutor({
agent: "codex",
stateDir: path.join(root, "state"),
env: { CODEX_HOME: codexHome },
paperclipRuntimeSkills: [],
paperclipSkillSync: { desiredSkills: [] },
});
const env = (sessionInputs[0]!.sessionOptions as { env: Record<string, string> }).env;
expect(env[apiKeyName]).toBe("sk-host-inherited-key");
expect(env.DEFAULT_AUTH_REQUEST).toBe(JSON.stringify({ methodId: "api-key" }));
} finally {
vi.unstubAllEnvs();
}
},
);
it("busts the session fingerprint when resolved adapter env changes but not across wakes", async () => {
const root = await makeTempRoot();
const stateDir = path.join(root, "state");
@@ -1969,17 +1969,6 @@ async function buildRuntime(input: {
// are absent from tempKeysApplied and keep their compatibility protection.
if (!scratchKeys.has(key) || value !== scratch.dir) resolvedAdapterEnv[key] = value;
}
// codex-acp supports both key names, but ACP clients must select its
// api-key authentication method during session creation. Without this
// request, the server advertises authentication and rejects session/new even
// though the credential is present in the launched process environment.
if (
acpxAgent === "codex" &&
(env.OPENAI_API_KEY || env.CODEX_API_KEY) &&
!env.DEFAULT_AUTH_REQUEST
) {
env.DEFAULT_AUTH_REQUEST = JSON.stringify({ methodId: "api-key" });
}
if (authToken) env.PAPERCLIP_API_KEY = authToken;
// For the claude agent, set model via ANTHROPIC_MODEL at startup rather than
// via session/set_config_option — the ACP server's set_config_option handler
@@ -2636,11 +2625,25 @@ function resolveRuntimeEnv(
env,
(options.platform ?? process.platform) === "win32",
);
return Object.fromEntries(
const finalEnv = Object.fromEntries(
Object.entries(mergedEnv).filter(
(entry): entry is [string, string] => typeof entry[1] === "string",
),
);
// codex-acp supports both key names, but ACP clients must select its
// api-key authentication method during session creation. Without this
// request, the server advertises authentication and rejects session/new even
// though the credential is present in the launched process environment. Check
// the final merged environment, not just the explicit run config, so a host
// key the local launch inherits still selects this default.
if (
acpxAgent === "codex" &&
(finalEnv.OPENAI_API_KEY || finalEnv.CODEX_API_KEY) &&
!finalEnv.DEFAULT_AUTH_REQUEST
) {
finalEnv.DEFAULT_AUTH_REQUEST = JSON.stringify({ methodId: "api-key" });
}
return finalEnv;
}
function mergeRuntimeEnvironment(
+16 -18
View File
@@ -4219,10 +4219,16 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: {
const queueDir = path.posix.join(bridgeRuntimeDir, "queue");
const assetRemoteDir = path.posix.join(bridgeRuntimeDir, "server");
const bridgeToken = createSandboxCallbackBridgeToken();
const configuredAttachmentBytes = Number(process.env.PAPERCLIP_ATTACHMENT_MAX_BYTES);
// A larger upload limit needs multipart headroom. A smaller attachment limit
// remains enforced by the API and must not shrink unrelated JSON responses.
const defaultBodyBytes = Number.isSafeInteger(configuredAttachmentBytes) && configuredAttachmentBytes > 0
? Math.max(DEFAULT_SANDBOX_CALLBACK_BRIDGE_MAX_BODY_BYTES, configuredAttachmentBytes + 64 * 1024)
: DEFAULT_SANDBOX_CALLBACK_BRIDGE_MAX_BODY_BYTES;
const maxBodyBytes =
typeof input.maxBodyBytes === "number" && Number.isFinite(input.maxBodyBytes) && input.maxBodyBytes > 0
? Math.trunc(input.maxBodyBytes)
: DEFAULT_SANDBOX_CALLBACK_BRIDGE_MAX_BODY_BYTES;
: defaultBodyBytes;
// The bridge worker runs inside the same process that serves the Paperclip
// API, so forwarded sandbox calls must target the LOCAL listen origin. The
// PAPERCLIP_RUNTIME_API_URL / PAPERCLIP_API_URL exports now prefer a
@@ -4282,20 +4288,15 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: {
path: string;
query: string;
headers: Record<string, string>;
/** The file bridge passes the whole request body here as one string.
* The HTTP/2 bridge passes it as the raw `Buffer` it read off the wire. */
/** Legacy text envelopes remain strings; binary uploads are raw bytes. */
body?: string | Buffer;
},
signal?: AbortSignal,
options?: {
suppressDebugLog?: boolean;
/**
* The caller's stream reservation owner, if it has one. The HTTP/2
* bridge passes the stream's own owner here, so the response body copy
* reserves against the same ceiling the request body copy already
* reserved against. The queue transport passes no owner, so its
* response-body read enforces only the per-request size ceiling, exactly
* as it did before this option existed.
* Both transports pass the request's reservation owner, so response
* buffers count toward the same host process ceiling as request buffers.
*/
reservation?: BridgeBodyReservation;
},
@@ -4324,8 +4325,8 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: {
const timeoutSignal = AbortSignal.timeout(forwardTimeoutMs);
const forwardSignal = signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
// Build the request-body init. A GET or a HEAD carries no body. The file
// bridge passes the whole body as one string; the HTTP/2 bridge passes it
// as a raw `Buffer`. Undici accepts a `Buffer` request body directly (a
// bridge passes legacy JSON as a string and binary data as a `Buffer`;
// HTTP/2 passes raw `Buffer` bodies. Undici accepts a `Buffer` request body directly (a
// `Buffer` is an `ArrayBufferView`), so neither shape needs a conversion.
// The cast below only bridges a `BodyInit` typing gap: the DOM library
// type this project's ambient `RequestInit` resolves to excludes a
@@ -4736,13 +4737,10 @@ export async function startAdapterExecutionTargetPaperclipBridge(input: {
maxBodyBytes,
getRuntimeParentContext: input.getRuntimeParentContext,
runtimeSpan: input.runtimeSpan,
// The queue transport writes the response body to a text file, so this
// is the one place the forward path decodes the response `Buffer` to a
// UTF-8 string. The queue's own on-wire behavior does not change.
handleRequest: async (request, options) => {
const result = await forwardBridgeRequest(request, options?.signal);
return { status: result.status, headers: result.headers, body: result.body.toString("utf8") };
},
// The worker encodes binary bodies only at the queue boundary.
handleRequest: (request, options) => forwardBridgeRequest(request, options?.signal, {
reservation: options?.reservation,
}),
});
server = await startSandboxCallbackBridgeServer({
runner,
@@ -105,6 +105,35 @@ describe("git workspace sync", () => {
expect(snapshot?.ignoredPaths).toContain(ignoredName);
});
it.each(["workspace_git_scan_timeout", "workspace_git_scan_saturated", "workspace_git_scan_output_limit", "workspace_git_scan_cancelled", "workspace_git_scan_failed"])("preserves %s instead of reporting a non-Git folder", async (code) => {
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-scan-failure-"));
cleanupDirs.push(rootDir);
const repo = await createRepo(rootDir);
const failure = Object.assign(new Error("Git enumeration failed"), { code });
setExpensiveWorkspaceGitExecutor(async (input) => {
if (input.operation === "adapter_sync.ignored_files") throw failure;
return runLocalGit(input.localDir, [...input.args]);
});
await expect(readGitWorkspaceSnapshot(repo, false)).rejects.toBe(failure);
});
it("lists ignored paths without traversing ignored directory contents", async () => {
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-git-ignored-scan-"));
cleanupDirs.push(rootDir);
const repo = await createRepo(rootDir);
await writeFile(path.join(repo, ".gitignore"), "dependencies/\n*.secret\n");
await mkdir(path.join(repo, "dependencies", "nested"), { recursive: true });
await writeFile(path.join(repo, "dependencies", "nested", "private"), "private");
await writeFile(path.join(repo, "token.secret"), "private");
let ignoredArgs: readonly string[] = [];
setExpensiveWorkspaceGitExecutor(async (input) => {
if (input.operation === "adapter_sync.ignored_files") ignoredArgs = input.args;
return runLocalGit(input.localDir, [...input.args]);
});
expect((await readGitWorkspaceSnapshot(repo))?.ignoredPaths).toEqual(["dependencies", "token.secret"]);
expect(ignoredArgs).toEqual(["ls-files", "--others", "--ignored", "--exclude-standard", "--directory", "-z"]);
});
async function createRepo(rootDir: string): Promise<string> {
const repo = path.join(rootDir, "repo");
await mkdir(repo, { recursive: true });
+112 -70
View File
@@ -15,8 +15,12 @@ export interface GitWorkspaceSnapshot {
overlayPaths: string[];
deletedPaths: string[];
ignoredPaths: string[];
/** Managed, editable repositories inside the task workspace. */
repositories?: Array<{ path: string; snapshot: GitWorkspaceSnapshot }>;
}
export const PROJECT_REPOSITORIES_DIR = ".paperclip-repositories";
export interface ExpensiveWorkspaceGitInput {
localDir: string;
args: readonly string[];
@@ -136,83 +140,110 @@ async function runExpensiveWorkspaceGit(
return await runLocalGit(localDir, args, options);
}
export async function readGitWorkspaceSnapshot(localDir: string): Promise<GitWorkspaceSnapshot | null> {
try {
const insideWorkTree = await runLocalGit(localDir, ["rev-parse", "--is-inside-work-tree"], {
timeout: 10_000,
maxBuffer: 16 * 1024,
export async function readGitWorkspaceSnapshot(localDir: string, includeRepositories = true): Promise<GitWorkspaceSnapshot | null> {
const repositories: NonNullable<GitWorkspaceSnapshot["repositories"]> = [];
if (includeRepositories) {
const root = path.join(localDir, PROJECT_REPOSITORIES_DIR);
const rootStat = await fs.lstat(root).catch((error: NodeJS.ErrnoException) => {
if (error.code === "ENOENT") return null;
throw error;
});
if (insideWorkTree.stdout.trim() !== "true") {
return null;
if (rootStat) {
if (!rootStat.isDirectory() || rootStat.isSymbolicLink()) throw new Error("Invalid project repositories directory");
for (const entry of (await fs.readdir(root, { withFileTypes: true })).sort((a, b) => a.name.localeCompare(b.name))) {
if (!entry.isDirectory() || !/^[a-zA-Z0-9_-]+$/.test(entry.name)) throw new Error("Invalid project repository directory");
const relative = `${PROJECT_REPOSITORIES_DIR}/${entry.name}`;
const snapshot = await readGitWorkspaceSnapshot(path.join(localDir, relative), false);
if (!snapshot) throw new Error(`Project repository is not a Git checkout: ${relative}`);
repositories.push({ path: relative, snapshot });
}
}
const toplevelResult = await runLocalGit(localDir, ["rev-parse", "--show-toplevel"], {
}
// Only repository discovery may report an ordinary directory. A failed
// snapshot of a confirmed repository must never fall back to directory sync.
let insideWorkTree: GitCommandResult;
try {
insideWorkTree = await runLocalGit(localDir, ["rev-parse", "--is-inside-work-tree"], {
timeout: 10_000,
maxBuffer: 16 * 1024,
});
// Git discovers a parent repository from a nested project directory, but
// that directory is not a fetch source. Keep the selected workspace
// boundary: subfolders use directory sync instead of importing the parent.
const [workspacePath, repositoryPath] = await Promise.all([
fs.realpath(localDir),
fs.realpath(toplevelResult.stdout.trim()),
]);
if (workspacePath !== repositoryPath) return null;
const [headCommitResult, branchResult, overlayDiffResult, untrackedResult, deletedResult, ignoredResult] = await Promise.all([
runLocalGit(localDir, ["rev-parse", "HEAD"], {
timeout: 10_000,
maxBuffer: 16 * 1024,
}),
runLocalGit(localDir, ["rev-parse", "--abbrev-ref", "HEAD"], {
timeout: 10_000,
maxBuffer: 16 * 1024,
}),
runExpensiveWorkspaceGit(localDir, ["diff", "--name-only", "-z", "--diff-filter=ACMRTUXB", "HEAD", "--"], "adapter_sync.overlay_diff", {
timeout: 10_000,
maxBuffer: 1024 * 1024,
}),
runExpensiveWorkspaceGit(localDir, ["ls-files", "--others", "--exclude-standard", "-z"], "adapter_sync.untracked_files", {
timeout: 10_000,
maxBuffer: 1024 * 1024,
}),
runExpensiveWorkspaceGit(localDir, ["diff", "--name-only", "-z", "--diff-filter=D", "HEAD", "--"], "adapter_sync.deleted_files", {
timeout: 10_000,
maxBuffer: 256 * 1024,
}),
runExpensiveWorkspaceGit(localDir, ["status", "--ignored", "--porcelain=v1", "-z", "--untracked-files=normal"], "adapter_sync.ignored_files", {
timeout: 10_000,
maxBuffer: 1024 * 1024,
}),
]);
const branchName = branchResult.stdout.trim();
// `-z` already delimits each record with a NUL byte, so a leading or
// trailing space in a record is part of the path itself, not padding to
// remove — trimming it would resolve to a path that does not exist. A
// length check finds the one genuinely empty record `-z` appends after
// the last NUL, without eating a real path's own leading or trailing
// whitespace. This applies to all four NUL-delimited outputs below (the
// overlay diff, the untracked list, the deleted list, and the ignored
// list); `branchName` and `headCommit` come from non-`-z` commands and
// keep their own `.trim()` above and below, which is safe.
const splitNul = (value: string) => value.split("\0").filter((entry) => entry.length > 0);
return {
headCommit: headCommitResult.stdout.trim(),
branchName: branchName && branchName !== "HEAD" ? branchName : null,
overlayPaths: [...new Set([...splitNul(overlayDiffResult.stdout), ...splitNul(untrackedResult.stdout)])]
.sort((left, right) => left.localeCompare(right)),
deletedPaths: [...new Set(splitNul(deletedResult.stdout))]
.sort((left, right) => left.localeCompare(right)),
ignoredPaths: splitNul(ignoredResult.stdout)
.filter((entry) => entry.startsWith("!! "))
.map((entry) => entry.slice(3).replace(/\/+$/, ""))
.filter(Boolean)
.sort((left, right) => left.localeCompare(right)),
};
} catch {
} catch (error) {
if (repositories.length === 0 && isNotAGitRepositoryError(error)) return null;
throw error;
}
if (insideWorkTree.stdout.trim() !== "true") {
return null;
}
const toplevelResult = await runLocalGit(localDir, ["rev-parse", "--show-toplevel"], {
timeout: 10_000,
maxBuffer: 16 * 1024,
});
// Git discovers a parent repository from a nested project directory, but
// that directory is not a fetch source. Keep the selected workspace
// boundary: subfolders use directory sync instead of importing the parent.
const [workspacePath, repositoryPath] = await Promise.all([
fs.realpath(localDir),
fs.realpath(toplevelResult.stdout.trim()),
]);
if (workspacePath !== repositoryPath) return null;
const [headCommitResult, branchResult, overlayDiffResult, untrackedResult, deletedResult, ignoredResult] = await Promise.all([
runLocalGit(localDir, ["rev-parse", "HEAD"], {
timeout: 10_000,
maxBuffer: 16 * 1024,
}),
runLocalGit(localDir, ["rev-parse", "--abbrev-ref", "HEAD"], {
timeout: 10_000,
maxBuffer: 16 * 1024,
}),
runExpensiveWorkspaceGit(localDir, ["diff", "--name-only", "-z", "--diff-filter=ACMRTUXB", "HEAD", "--"], "adapter_sync.overlay_diff", {
timeout: 10_000,
maxBuffer: 1024 * 1024,
}),
runExpensiveWorkspaceGit(localDir, ["ls-files", "--others", "--exclude-standard", "-z"], "adapter_sync.untracked_files", {
timeout: 10_000,
maxBuffer: 1024 * 1024,
}),
runExpensiveWorkspaceGit(localDir, ["diff", "--name-only", "-z", "--diff-filter=D", "HEAD", "--"], "adapter_sync.deleted_files", {
timeout: 10_000,
maxBuffer: 256 * 1024,
}),
// Collapse ignored directories instead of walking their contents, and
// avoid producing unrelated tracked/untracked status records.
runExpensiveWorkspaceGit(localDir, ["ls-files", "--others", "--ignored", "--exclude-standard", "--directory", "-z"], "adapter_sync.ignored_files", {
timeout: 10_000,
maxBuffer: 1024 * 1024,
}),
]);
const branchName = branchResult.stdout.trim();
// `-z` already delimits each record with a NUL byte, so a leading or
// trailing space in a record is part of the path itself, not padding to
// remove — trimming it would resolve to a path that does not exist. A
// length check finds the one genuinely empty record `-z` appends after
// the last NUL, without eating a real path's own leading or trailing
// whitespace. This applies to all four NUL-delimited outputs below (the
// overlay diff, the untracked list, the deleted list, and the ignored
// list); `branchName` and `headCommit` come from non-`-z` commands and
// keep their own `.trim()` above and below, which is safe.
const splitNul = (value: string) => value.split("\0").filter((entry) => entry.length > 0);
return {
headCommit: headCommitResult.stdout.trim(),
branchName: branchName && branchName !== "HEAD" ? branchName : null,
overlayPaths: [...new Set([...splitNul(overlayDiffResult.stdout), ...splitNul(untrackedResult.stdout),
...repositories.flatMap((repo) => repo.snapshot.overlayPaths.map((entry) => `${repo.path}/${entry}`))])]
.sort((left, right) => left.localeCompare(right)),
deletedPaths: [...new Set([...splitNul(deletedResult.stdout),
...repositories.flatMap((repo) => repo.snapshot.deletedPaths.map((entry) => `${repo.path}/${entry}`))])]
.sort((left, right) => left.localeCompare(right)),
ignoredPaths: [...splitNul(ignoredResult.stdout)
.map((entry) => entry.replace(/\/+$/, ""))
.filter((entry) => Boolean(entry) && !(repositories.length > 0 && entry === PROJECT_REPOSITORIES_DIR)),
...repositories.flatMap((repo) => repo.snapshot.ignoredPaths.map((entry) => `${repo.path}/${entry}`))]
.sort((left, right) => left.localeCompare(right)),
...(repositories.length > 0 ? { repositories } : {}),
};
}
/** The `git ls-files --others --ignored` output for one directory, read by {@link readReferencedSourceGitIgnoredPaths}. */
@@ -557,6 +588,17 @@ export async function withShallowGitWorkspaceClone<T>(
timeout: 60_000,
maxBuffer: 1024 * 1024,
});
for (const repository of input.snapshot.repositories ?? []) {
await withShallowGitWorkspaceClone({
localDir: path.join(input.localDir, repository.path),
snapshot: repository.snapshot,
}, async (nestedClone) => {
await fs.cp(nestedClone, path.join(cloneDir, repository.path), { recursive: true });
});
}
if (input.snapshot.repositories?.length) {
await fs.appendFile(path.join(cloneDir, ".git/info/exclude"), `\n/${PROJECT_REPOSITORIES_DIR}/\n`);
}
return await fn(cloneDir);
} finally {
await runLocalGit(input.localDir, ["update-ref", "-d", tempRef], {
@@ -117,9 +117,9 @@ export const PAPERCLIP_RUNNER_PERMISSION_CAPABILITIES = {
},
{
value: "approve-reads",
label: "Conservative (fail closed)",
label: "Allow Paperclip reads",
description:
"Delegate ACPX permission requests and fail closed until a verified interactive approval bridge is available.",
"Automatically allow assigned Paperclip read tools. Other operations stop with an approval-required message because this runner has no interactive approval handler.",
},
{
value: "deny-all",
@@ -0,0 +1,34 @@
import { runInNewContext } from "node:vm";
import { describe, expect, it } from "vitest";
import { decodeSandboxBridgeBody, encodeSandboxBridgeBody, sandboxBridgeBodyCodecSource, sandboxBridgeEnvelopeLimit } from "./sandbox-callback-bridge-body.js";
const embedded = runInNewContext(`${sandboxBridgeBodyCodecSource()}; ({ encodeSandboxBridgeBody, decodeSandboxBridgeBody })`, { Buffer });
describe.each([
{ name: "host", encode: encodeSandboxBridgeBody, decode: decodeSandboxBridgeBody },
{ name: "generated gateway", encode: embedded.encodeSandboxBridgeBody as typeof encodeSandboxBridgeBody, decode: embedded.decodeSandboxBridgeBody as typeof decodeSandboxBridgeBody },
])("queue body codec: $name", ({ encode, decode }) => {
it("preserves old text envelopes and arbitrary binary bytes at the limit", () => {
const text = "猫\u0000\n";
expect(encode(text, 5)).toEqual({ body: text });
expect(decode({ body: text }, 5)).toEqual(Buffer.from(text));
expect(decode({ body: text, bodyEncoding: "utf8" }, 5)).toEqual(Buffer.from(text));
const bytes = Buffer.from(Array.from({ length: 256 }, (_, index) => index));
expect(decode(encode(bytes, bytes.length), bytes.length)).toEqual(bytes);
expect(encode(bytes, bytes.length).bodyEncoding).toBe("base64");
expect(() => encode(bytes, bytes.length - 1)).toThrow(/size limit/);
expect(() => decode(encode(bytes, bytes.length), bytes.length - 1)).toThrow(/size limit/);
expect(() => decode({ body: text }, 4)).toThrow(/size limit/);
});
it.each(["!AAA", "YQ", "YQ=", "YQ===", "YQ==\n", "YR==", "=AAA", "____", "猫"])('rejects malformed or noncanonical base64 "%s"', body => {
expect(() => decode({ body, bodyEncoding: "base64" }, 1024)).toThrow(/base64/);
});
it("rejects unknown encodings and bounds JSON escaping overhead", () => {
expect(() => decode({ body: "abc", bodyEncoding: "hex" as "utf8" }, 1024)).toThrow(/encoding/);
expect(() => decode({ body: null as unknown as string }, 1024)).toThrow(/Invalid/);
const envelope = { id: "request", ...encode("\u0000".repeat(1024), 1024) };
expect(Buffer.byteLength(JSON.stringify(envelope))).toBeLessThan(sandboxBridgeEnvelopeLimit(1024));
});
});
@@ -0,0 +1,40 @@
export interface SandboxCallbackBridgeBody {
body: string;
/** Omitted by older queue peers, whose bodies are UTF-8 text. */
bodyEncoding?: "utf8" | "base64";
}
/** JSON can escape each input byte as six characters. Metadata is bounded too. */
export function sandboxBridgeEnvelopeLimit(maxBodyBytes: number): number {
return 6 * maxBodyBytes + 64 * 1024;
}
export function encodeSandboxBridgeBody(body: string | Buffer, maxBodyBytes: number): SandboxCallbackBridgeBody {
if (Buffer.byteLength(body) > maxBodyBytes) throw new Error("Bridge body exceeded the configured size limit.");
return Buffer.isBuffer(body) ? { body: body.toString("base64"), bodyEncoding: "base64" } : { body };
}
/** Self-contained so the same decoder can be embedded in the remote gateway. */
export function decodeSandboxBridgeBody(envelope: SandboxCallbackBridgeBody, maxBodyBytes: number): Buffer {
if (!envelope || typeof envelope.body !== "string") throw new Error("Invalid bridge body.");
if (envelope.bodyEncoding === undefined || envelope.bodyEncoding === "utf8") {
if (Buffer.byteLength(envelope.body, "utf8") > maxBodyBytes) throw new Error("Bridge body exceeded the configured size limit.");
return Buffer.from(envelope.body, "utf8");
}
if (envelope.bodyEncoding !== "base64") throw new Error("Unsupported bridge body encoding.");
const value = envelope.body;
if (value.length > 4 * Math.ceil(maxBodyBytes / 3)) throw new Error("Bridge body exceeded the configured size limit.");
// Buffer.from is permissive; reject malformed input before allocating bytes.
if (value.length % 4 !== 0 || /[^A-Za-z0-9+/=]/.test(value) || !/^[A-Za-z0-9+/]*={0,2}$/.test(value)) {
throw new Error("Invalid bridge base64 body.");
}
const bytes = Buffer.from(value, "base64");
if (bytes.length > maxBodyBytes) throw new Error("Bridge body exceeded the configured size limit.");
if (bytes.toString("base64") !== value) throw new Error("Invalid bridge base64 body.");
return bytes;
}
export function sandboxBridgeBodyCodecSource(): string {
return [sandboxBridgeEnvelopeLimit, encodeSandboxBridgeBody, decodeSandboxBridgeBody]
.map(fn => `const ${fn.name} = ${fn.toString()};`).join("\n");
}
@@ -164,7 +164,7 @@ describe("sandbox callback bridge", () => {
path: string;
query: string;
headers: Record<string, string>;
body: string;
body: string | Buffer;
}> = [];
const worker = await startSandboxCallbackBridgeWorker({
@@ -1495,9 +1495,9 @@ describe("sandbox callback bridge", () => {
);
}
// The HTTP/2 route list adds the two binary attachment routes on top of
// the documented heartbeat surface.
// Both transports admit the full attachment workflow.
const http2Allowed: Array<{ method: string; path: string }> = [
{ method: "GET", path: "/api/issues/issue-1/attachments" },
{ method: "POST", path: "/api/companies/co-1/issues/issue-1/attachments" },
{ method: "GET", path: "/api/attachments/att-1/content" },
];
@@ -1522,15 +1522,14 @@ describe("sandbox callback bridge", () => {
}
});
it("denies both attachment routes on the default (queue) route list", () => {
it("admits listing, uploads and downloads on the default queue route list", () => {
const attachmentRequests: Array<{ method: string; path: string }> = [
{ method: "GET", path: "/api/issues/issue-1/attachments" },
{ method: "POST", path: "/api/companies/co-1/issues/issue-1/attachments" },
{ method: "GET", path: "/api/attachments/att-1/content" },
];
for (const request of attachmentRequests) {
expect(authorizeSandboxCallbackBridgeRequestWithRoutes(request)).toBe(
`Route not allowed: ${request.method} ${request.path}`,
);
expect(authorizeSandboxCallbackBridgeRequestWithRoutes(request)).toBeNull();
}
});
@@ -3576,6 +3575,83 @@ describe("sandbox callback bridge", () => {
expect(response.headers.get("x-paperclip-bridge-outcome")).toBeNull();
}, 15_000);
async function startQueueGatewayForFileTest(options: {
maxBodyBytes: number;
client?: SandboxCallbackBridgeQueueClient;
handleRequest: Parameters<typeof startSandboxCallbackBridgeWorker>[0]["handleRequest"];
}) {
const root = await mkdtemp(path.join(os.tmpdir(), "paperclip-file-codec-"));
cleanupDirs.push(root);
const asset = await createSandboxCallbackBridgeAsset();
cleanupFns.push(asset.cleanup);
const queueDir = path.join(root, "queue");
const bridgeToken = createSandboxCallbackBridgeToken();
const worker = await startSandboxCallbackBridgeWorker({
client: options.client ?? createFileSystemSandboxCallbackBridgeQueueClient(), queueDir,
maxBodyBytes: options.maxBodyBytes, handleRequest: options.handleRequest, pollIntervalMs: 10,
});
cleanupFns.push(() => worker.stop());
const gateway = await startSandboxCallbackBridgeServer({
runner: createExecRunner(), remoteCwd: root, assetRemoteDir: asset.localDir,
queueDir, bridgeToken, maxBodyBytes: options.maxBodyBytes, pollIntervalMs: 10,
});
cleanupFns.push(() => gateway.stop());
return { ...gateway, bridgeToken, queueDir };
}
it.each(["queue", "http2"])("preserves multipart bytes at the configured limit on %s and rejects overflow before forwarding", async mode => {
const maxBodyBytes = 1024;
const bytes = Buffer.alloc(maxBodyBytes, 0xff);
const handled = vi.fn(async (request: { body?: string | Buffer }) => ({
status: 200, headers: { "content-type": "application/octet-stream" }, body: Buffer.from(request.body ?? ""),
}));
const bridgeToken = createSandboxCallbackBridgeToken();
const gateway = mode === "queue"
? await startQueueGatewayForFileTest({ maxBodyBytes, handleRequest: handled })
: { ...await startHttp2GatewayForTest({ bridgeToken, maxBodyBytes, forwardRequest: handled }), bridgeToken };
const post = (body: Buffer) => fetch(`${gateway.baseUrl}/api/companies/c/issues/i/attachments`, {
method: "POST", headers: { authorization: `Bearer ${gateway.bridgeToken}`, "content-type": "multipart/form-data; boundary=exact-boundary" }, body: new Uint8Array(body),
});
const response = await post(bytes);
expect(response.status).toBe(200);
expect(Buffer.from(await response.arrayBuffer())).toEqual(bytes);
expect(handled).toHaveBeenCalledTimes(1);
expect(handled.mock.calls[0][0]).toMatchObject({ headers: { "content-type": "multipart/form-data; boundary=exact-boundary" } });
const overflow = await post(Buffer.alloc(maxBodyBytes + 1));
expect(overflow.status).toBeGreaterThanOrEqual(400);
expect(handled).toHaveBeenCalledTimes(1);
});
it("rejects malformed queue encodings without forwarding a mutation", async () => {
const handled = vi.fn(async () => ({ status: 200, body: "ok" }));
const gateway = await startQueueGatewayForFileTest({ maxBodyBytes: 1024, handleRequest: handled });
const directories = sandboxCallbackBridgeDirectories(gateway.queueDir);
for (const [id, bodyEncoding, body] of [["bad-base64", "base64", "YR=="], ["bad-encoding", "hex", "00"], ["too-big", "base64", Buffer.alloc(1025).toString("base64")]]) {
await createFileSystemSandboxCallbackBridgeQueueClient().writeTextFile(path.join(directories.requestsDir, `${id}.json`), JSON.stringify({
id, method: "POST", path: "/api/companies/c/issues/i/attachments", query: "", headers: {}, bodyEncoding, body,
}));
const responsePath = path.join(directories.responsesDir, `${id}.json`);
await vi.waitFor(async () => expect(JSON.parse(await readFile(responsePath, "utf8")).status).toBe(400));
}
expect(handled).not.toHaveBeenCalled();
});
it("reports a corrupted upload response as indeterminate and never forwards twice", async () => {
const client = createFileSystemSandboxCallbackBridgeQueueClient();
const writeResponse = client.writeResponseFile!.bind(client);
client.writeResponseFile = (destination, body, options) => writeResponse(destination,
JSON.stringify({ ...JSON.parse(body), bodyEncoding: "base64", body: "invalid" }), options);
const handled = vi.fn(async () => ({ status: 201, body: Buffer.from("saved") }));
const gateway = await startQueueGatewayForFileTest({ client, maxBodyBytes: 1024, handleRequest: handled });
const response = await fetch(`${gateway.baseUrl}/api/companies/c/issues/i/attachments`, {
method: "POST", headers: { authorization: `Bearer ${gateway.bridgeToken}`, "content-type": "multipart/form-data; boundary=boundary" }, body: Buffer.from([0xff]),
});
expect(response.status).toBe(409);
expect(response.headers.get("x-paperclip-bridge-outcome")).toBe("indeterminate");
await expect(response.json()).resolves.toMatchObject({ retryable: false, outcome: "indeterminate" });
expect(handled).toHaveBeenCalledTimes(1);
});
it("rejects a request body over maxBodyBytes on the queue path before it writes the queue file", async () => {
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-bridge-queue-maxbody-"));
cleanupDirs.push(rootDir);
@@ -3656,7 +3732,7 @@ describe("sandbox callback bridge", () => {
const bridgeToken = createSandboxCallbackBridgeToken();
const requestBodyText = JSON.stringify({ note: "café" });
const seenRequests: Array<{ body: string }> = [];
const seenRequests: Array<{ body: string | Buffer }> = [];
const worker = await startSandboxCallbackBridgeWorker({
client: createFileSystemSandboxCallbackBridgeQueueClient(),
queueDir,
@@ -4,6 +4,14 @@ import http2 from "node:http2";
import os from "node:os";
import path from "node:path";
import type { Duplex } from "node:stream";
import {
decodeSandboxBridgeBody,
encodeSandboxBridgeBody,
sandboxBridgeBodyCodecSource,
sandboxBridgeEnvelopeLimit,
type SandboxCallbackBridgeBody,
} from "./sandbox-callback-bridge-body.js";
import type { BridgeBodyReservation } from "./http2-bridge-server.js";
import {
runWithoutActiveStep,
@@ -162,6 +170,11 @@ export const DEFAULT_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST: readonly SandboxCa
{ method: "PATCH", path: /^\/api\/issues\/[^/]+$/ },
{ method: "GET", path: /^\/api\/issues\/[^/]+\/approvals$/ },
// Files: the queue encodes binary bodies; HTTP/2 carries the same bytes directly.
{ method: "GET", path: /^\/api\/issues\/[^/]+\/attachments$/ },
{ method: "POST", path: /^\/api\/companies\/[^/]+\/issues\/[^/]+\/attachments$/ },
{ method: "GET", path: /^\/api\/attachments\/[^/]+\/content$/ },
// Work products: publish branch/commit/artifact metadata for completed work.
{ method: "GET", path: /^\/api\/issues\/[^/]+\/work-products$/ },
{ method: "POST", path: /^\/api\/issues\/[^/]+\/work-products$/ },
@@ -209,15 +222,8 @@ export const DEFAULT_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST: readonly SandboxCa
{ method: "DELETE", path: /^\/api\/routine-triggers\/[^/]+$/ },
] as const;
// The HTTP/2 bridge carries the request body as raw bytes, so it can admit
// the two binary attachment routes. The queue transport keeps
// `DEFAULT_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST`, because its envelope
// carries a string body only.
export const HTTP2_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST: readonly SandboxCallbackBridgeRouteRule[] = [
...DEFAULT_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST,
{ method: "POST", path: /^\/api\/companies\/[^/]+\/issues\/[^/]+\/attachments$/ },
{ method: "GET", path: /^\/api\/attachments\/[^/]+\/content$/ },
] as const;
// Keep the public alias for callers selecting the HTTP/2 transport.
export const HTTP2_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST = DEFAULT_SANDBOX_CALLBACK_BRIDGE_ROUTE_ALLOWLIST;
export const DEFAULT_SANDBOX_CALLBACK_BRIDGE_HEADER_ALLOWLIST = [
"accept",
@@ -227,21 +233,16 @@ export const DEFAULT_SANDBOX_CALLBACK_BRIDGE_HEADER_ALLOWLIST = [
"x-paperclip-github-capability",
] as const;
export interface SandboxCallbackBridgeRequest {
export interface SandboxCallbackBridgeRequest extends SandboxCallbackBridgeBody {
id: string;
method: string;
path: string;
query: string;
headers: Record<string, string>;
/**
* UTF-8 body contents. The bridge rejects non-JSON request bodies; binary
* payloads are intentionally out of scope for this queue protocol.
*/
body: string;
createdAt: string;
}
export interface SandboxCallbackBridgeResponse {
export interface SandboxCallbackBridgeResponse extends SandboxCallbackBridgeBody {
id: string;
status: number;
headers: Record<string, string>;
@@ -273,7 +274,8 @@ export interface SandboxCallbackBridgeQueueClient {
// external implementation stays compatible without a change.
makeDirs?(remotePaths: string[]): Promise<void>;
listJsonFiles(remotePath: string): Promise<string[]>;
readTextFile(remotePath: string): Promise<string>;
fileSize?(remotePath: string): Promise<number>;
readTextFile(remotePath: string, maxBytes?: number): Promise<string>;
writeTextFile(remotePath: string, body: string): Promise<void>;
writeResponseFile?(
responsePath: string,
@@ -522,7 +524,24 @@ export function createFileSystemSandboxCallbackBridgeQueueClient(): SandboxCallb
.map((entry) => entry.name)
.sort((left, right) => left.localeCompare(right));
},
readTextFile: async (remotePath) => await fs.readFile(remotePath, "utf8"),
fileSize: async (remotePath) => (await fs.stat(remotePath)).size,
readTextFile: async (remotePath, maxBytes) => {
if (maxBytes === undefined) return fs.readFile(remotePath, "utf8");
const file = await fs.open(remotePath, "r");
try {
const stat = await file.stat();
if (stat.size > maxBytes) throw new Error("Bridge envelope exceeded the configured size limit.");
const bytes = Buffer.alloc(Math.min(stat.size, maxBytes) + 1);
let length = 0;
while (length < bytes.length) {
const read = await file.read(bytes, length, bytes.length - length, length);
if (!read.bytesRead) break;
length += read.bytesRead;
}
if (length > stat.size) throw new Error("Bridge envelope changed while reading.");
return bytes.subarray(0, length).toString("utf8");
} finally { await file.close(); }
},
writeTextFile: async (remotePath, body) => {
await fs.mkdir(path.posix.dirname(remotePath), { recursive: true });
// Write to a temporary path that does NOT end in `.json`, then rename it
@@ -661,9 +680,18 @@ export function createCommandManagedSandboxCallbackBridgeQueueClient(input: {
.filter((line) => line.length > 0)
.sort((left, right) => left.localeCompare(right));
},
readTextFile: async (remotePath) => {
const result = await runChecked(`read ${remotePath}`, `base64 < ${shellQuote(remotePath)}`);
return Buffer.from(result.stdout.replace(/\s+/g, ""), "base64").toString("utf8");
fileSize: async (remotePath) => {
const result = await runChecked(`size ${remotePath}`, `wc -c < ${shellQuote(remotePath)}`);
return Number(result.stdout.trim());
},
readTextFile: async (remotePath, maxBytes) => {
const command = maxBytes === undefined
? `base64 < ${shellQuote(remotePath)}`
: `head -c ${Math.trunc(maxBytes) + 1} ${shellQuote(remotePath)} | base64`;
const result = await runChecked(`read ${remotePath}`, command);
const bytes = Buffer.from(result.stdout.replace(/\s+/g, ""), "base64");
if (maxBytes !== undefined && bytes.length > maxBytes) throw new Error("Bridge envelope exceeded the configured size limit.");
return bytes.toString("utf8");
},
writeTextFile: async (remotePath, body) => {
const remoteDir = path.posix.dirname(remotePath);
@@ -792,12 +820,12 @@ export async function startSandboxCallbackBridgeWorker(input: {
// not strand with no response. A handler that ignores the signal keeps its
// earlier behavior.
handleRequest: (
request: SandboxCallbackBridgeRequest,
options?: { signal: AbortSignal },
request: Omit<SandboxCallbackBridgeRequest, "body" | "bodyEncoding"> & { body: string | Buffer },
options?: { signal: AbortSignal; reservation: BridgeBodyReservation },
) => Promise<{
status: number;
headers?: Record<string, string>;
body?: string;
body?: string | Buffer;
}>;
maxBodyBytes?: number | null;
// Return the current-run parent-context token. The worker reads it per request
@@ -821,6 +849,10 @@ export async function startSandboxCallbackBridgeWorker(input: {
DEFAULT_BRIDGE_ABORTED_HANDLER_GRACE_MS,
);
const maxBodyBytes = normalizeTimeoutMs(input.maxBodyBytes, DEFAULT_BRIDGE_MAX_BODY_BYTES);
const maxEnvelopeBytes = sandboxBridgeEnvelopeLimit(maxBodyBytes);
// Load lazily to avoid the HTTP/2 module's constants depending on this module
// during initialization. Both transports share the host process memory ceiling.
const { createBridgeBodyReservation } = await import("./http2-bridge-server.js");
const directories = sandboxCallbackBridgeDirectories(input.queueDir);
const queueDirectories = [
directories.rootDir,
@@ -924,6 +956,8 @@ export async function startSandboxCallbackBridgeWorker(input: {
finalized: false,
};
inFlightRequestGuards.set(fileName, guard);
const reservation = createBridgeBodyReservation();
const envelopeReadReservation = createBridgeBodyReservation();
// Claim the request for the handler. Return `false` when the recovery path
// already claimed it; the caller must then not run the mutation and must not
// write a response, because the recovery path writes a 503 and the caller
@@ -993,9 +1027,33 @@ export async function startSandboxCallbackBridgeWorker(input: {
await writeAbortedHandlerBackstop(fileName, guard, lastWriteError);
};
try {
// Bound the read and its encoded/parsed copies before allocating. Built-in
// clients stat first so a high configured file limit does not reserve that
// whole limit for a tiny request. Older clients use the conservative cap.
const envelopeBytes = input.client.fileSize
? await input.client.fileSize(requestPath).catch(() => maxEnvelopeBytes)
: maxEnvelopeBytes;
if (envelopeBytes > maxEnvelopeBytes) {
await finalize({
id: fileName.replace(/\.json$/i, ""), status: 413,
headers: { "content-type": "application/json" },
body: JSON.stringify({ error: "Bridge request envelope exceeded the configured size limit." }),
completedAt: new Date().toISOString(),
});
return;
}
const readLimit = Number.isSafeInteger(envelopeBytes) && envelopeBytes >= 0
? Math.min(envelopeBytes, maxEnvelopeBytes) : maxEnvelopeBytes;
if (!envelopeReadReservation.reserve(6 * readLimit)) {
await finalize({ id: fileName.replace(/\.json$/i, ""), status: 503,
headers: { "content-type": "application/json" },
body: JSON.stringify({ error: "Bridge host body capacity is busy. Retry later." }),
completedAt: new Date().toISOString() });
return;
}
let raw: string;
try {
raw = await input.client.readTextFile(requestPath);
raw = await input.client.readTextFile(requestPath, readLimit);
} catch (error) {
// The gateway deletes a request file when its caller stops waiting
// (client-side timeout cleanup). A read that fails because the file is
@@ -1011,7 +1069,9 @@ export async function startSandboxCallbackBridgeWorker(input: {
}
let request: SandboxCallbackBridgeRequest;
try {
if (Buffer.byteLength(raw) > maxEnvelopeBytes) throw new Error("Bridge envelope too large");
request = JSON.parse(raw) as SandboxCallbackBridgeRequest;
decodeSandboxBridgeBody(request, maxBodyBytes);
} catch {
const requestId = fileName.replace(/\.json$/i, "") || randomUUID();
await finalize({
@@ -1024,6 +1084,16 @@ export async function startSandboxCallbackBridgeWorker(input: {
return;
}
// Keep only the actual request allocation reserved while forwarding;
// an abandoned small request must not retain a maximum-sized reservation.
envelopeReadReservation.release();
if (!reservation.reserve(4 * Buffer.byteLength(raw) + 2 * Buffer.byteLength(request.body))) {
await finalize({ id: request.id, status: 503,
headers: { "content-type": "application/json" },
body: JSON.stringify({ error: "Bridge host body capacity is busy. Retry later." }),
completedAt: new Date().toISOString() });
return;
}
const denialReason = await authorizeRequest(request);
if (denialReason) {
await finalize({
@@ -1048,17 +1118,24 @@ export async function startSandboxCallbackBridgeWorker(input: {
// Build the response, then finalize once. The handler already holds the
// claim, so `finalize` writes the real response.
let response: SandboxCallbackBridgeResponse;
let handlerReturned = false;
try {
const result = await input.handleRequest(request, { signal: guard.controller.signal });
const responseBody = result.body ?? "";
if (Buffer.byteLength(responseBody, "utf8") > maxBodyBytes) {
throw new Error(`Bridge response body exceeded the configured size limit of ${maxBodyBytes} bytes.`);
const { bodyEncoding, ...forwardRequest } = request;
const result = await input.handleRequest({ ...forwardRequest,
body: bodyEncoding === "base64" ? decodeSandboxBridgeBody(request, maxBodyBytes) : request.body,
}, { signal: guard.controller.signal, reservation });
handlerReturned = true;
const responseBytes = Buffer.byteLength(result.body ?? "");
if (responseBytes > maxBodyBytes) throw new Error("Bridge response body exceeded the configured size limit.");
if (!reservation.reserve(4 * sandboxBridgeEnvelopeLimit(responseBytes))) {
throw new Error("Bridge host response body capacity is busy.");
}
const responseBody = encodeSandboxBridgeBody(result.body ?? "", maxBodyBytes);
response = {
id: request.id,
status: result.status,
headers: result.headers ?? {},
body: responseBody,
...responseBody,
completedAt: new Date().toISOString(),
};
} catch (error) {
@@ -1075,7 +1152,7 @@ export async function startSandboxCallbackBridgeWorker(input: {
// the outcome indeterminate. The caller must not retry a 504 from the
// bridge, unlike the retry-safe 503 that the recovery path writes only
// before the host operation starts.
if (guard.controller.signal.aborted) {
if (guard.controller.signal.aborted || (handlerReturned && !["GET", "HEAD", "OPTIONS", "TRACE"].includes(request.method))) {
response = {
id: request.id,
status: 504,
@@ -1104,6 +1181,8 @@ export async function startSandboxCallbackBridgeWorker(input: {
}
await finalize(response);
} finally {
envelopeReadReservation.release();
reservation.release();
// Drop the guard only when it still points to this attempt. A retry can
// register a new attempt under the same file name; that new guard must
// stay in the map. Keep the guard when a backstop is still pending: a
@@ -2201,8 +2280,9 @@ ${DUPLEX_GATEWAY_CODEC_SOURCE}
// use it: readBodyBytes reserves against it, and each mode's request
// handler releases what it reserved once the body is no longer needed.
${BRIDGE_PROCESS_BODY_LEDGER_SOURCE}
${sandboxBridgeBodyCodecSource()}
// The multiplier matches HTTP2_BRIDGE_MAX_CONCURRENT_STREAMS (4) in
// HTTP/2's multiplier matches HTTP2_BRIDGE_MAX_CONCURRENT_STREAMS (4) in
// http2-bridge-server.ts, doubled because readBodyBytes reserves a body's
// bytes twice: once for the retained chunk array, once for the concatenated
// copy, since both are live buffers at once. This gives the gateway process
@@ -2210,8 +2290,11 @@ ${BRIDGE_PROCESS_BODY_LEDGER_SOURCE}
// concurrent requests cannot grow this process's memory without limit, even
// though every individual body already passes the maxBodyBytes check below.
// This ceiling is independent of, and separate from, the ceiling the host
// enforces on its own side of the bridge connection.
const maxProcessBodyBytes = maxBodyBytes * 8;
// enforces on its own side of the bridge connection. Queue mode also counts
// JSON/base64 envelopes; its aggregate ceiling never exceeds 1 GiB.
const maxProcessBodyBytes = bridgeMode === "${SANDBOX_CALLBACK_BRIDGE_FILE_MODE}"
? Math.min(1024 * 1024 * 1024, 4 * (4 * sandboxBridgeEnvelopeLimit(maxBodyBytes) + 4 * maxBodyBytes))
: maxBodyBytes * 8;
const processBodyLedger = createBridgeProcessBodyLedger(maxProcessBodyBytes);
// A denied process-ledger reservation answers 503: the sandbox client should
@@ -2280,11 +2363,6 @@ async function readBodyBytes(req) {
}
}
async function readBody(req) {
const { body, release } = await readBodyBytes(req);
return { body: body.toString("utf8"), release };
}
function tokensMatch(received) {
const expected = Buffer.from(bridgeToken, "utf8");
const actual = Buffer.from(typeof received === "string" ? received : "", "utf8");
@@ -2327,14 +2405,32 @@ async function runFileGateway() {
}
}
async function waitForResponse(requestId) {
async function waitForResponse(requestId, reserveResponse) {
const responsePath = path.posix.join(responsesDir, \`\${requestId}.json\`);
const deadline = Date.now() + responseTimeoutMs;
while (Date.now() < deadline) {
const body = await fs.readFile(responsePath, "utf8").catch(() => null);
if (body != null) {
await fs.rm(responsePath, { force: true }).catch(() => undefined);
return JSON.parse(body);
const handle = await fs.open(responsePath, "r").catch(error => {
if (error.code === "ENOENT") return null;
throw error;
});
if (handle) {
try {
const stat = await handle.stat();
if (stat.size > sandboxBridgeEnvelopeLimit(maxBodyBytes)) throw new Error("Bridge response envelope exceeded the configured size limit.");
reserveResponse(6 * stat.size + 1);
const bytes = Buffer.alloc(stat.size + 1);
let length = 0;
while (length < bytes.length) {
const read = await handle.read(bytes, length, bytes.length - length, length);
if (!read.bytesRead) break;
length += read.bytesRead;
}
if (length !== stat.size) throw new Error("Bridge response envelope changed while reading.");
return JSON.parse(bytes.subarray(0, length).toString("utf8"));
} finally {
await handle.close();
await fs.rm(responsePath, { force: true }).catch(() => undefined);
}
}
await sleep(pollIntervalMs);
}
@@ -2342,12 +2438,14 @@ async function runFileGateway() {
}
const server = createServer(async (req, res) => {
// readBody reserves the body's bytes against the process ledger and
// readBodyBytes reserves the body's bytes against the process ledger and
// hands back a release function; this holds it so the finally below
// releases those bytes exactly once no matter how this handler ends —
// its normal completion, a thrown error, a client abort, or a deadline
// timeout all reach the same finally.
let releaseBodyReservation = null;
let envelopeReservation = 0;
let dispatched = false;
try {
const auth = req.headers.authorization || "";
const receivedToken = auth.startsWith("Bearer ") ? auth.slice("Bearer ".length) : "";
@@ -2368,30 +2466,42 @@ async function runFileGateway() {
const url = new URL(req.url || "/", "http://127.0.0.1");
const contentType = typeof req.headers["content-type"] === "string" ? req.headers["content-type"] : "";
if (req.method && req.method !== "GET" && req.method !== "HEAD" && !/json/i.test(contentType)) {
const multipartAttachment = req.method === "POST"
&& /^\\/api\\/companies\\/[^/]+\\/issues\\/[^/]+\\/attachments$/.test(url.pathname)
&& /^multipart\\/form-data(?:;|$)/i.test(contentType);
if (req.method && req.method !== "GET" && req.method !== "HEAD" && !/json/i.test(contentType) && !multipartAttachment) {
writeJsonResponse(res, 415, { error: "Bridge only accepts JSON request bodies." });
return;
}
const requestId = randomUUID();
const { body: requestBody, release } = await readBody(req);
// Reserve encoded envelopes and response decoding separately from the
// incoming byte buffers. Capacity rejection precedes dispatch, hence 503.
const { body: requestBody, release } = await readBodyBytes(req);
releaseBodyReservation = release;
const reserveBytes = 4 * sandboxBridgeEnvelopeLimit(requestBody.length);
if (!processBodyLedger.reserve(reserveBytes)) throw new BridgeProcessCapacityError();
envelopeReservation = reserveBytes;
const payload = {
id: requestId,
method: req.method || "GET",
path: url.pathname,
query: url.search,
headers: normalizeHeaders(req.headers),
body: requestBody,
...encodeSandboxBridgeBody(multipartAttachment ? requestBody : requestBody.toString("utf8"), maxBodyBytes),
createdAt: new Date().toISOString(),
};
const requestPath = path.posix.join(requestsDir, \`\${requestId}.json\`);
const tempPath = \`\${requestPath}.tmp\`;
await fs.writeFile(tempPath, \`\${JSON.stringify(payload)}\\n\`, "utf8");
dispatched = true;
await fs.rename(tempPath, requestPath);
let response;
try {
response = await waitForResponse(requestId);
response = await waitForResponse(requestId, bytes => {
if (!processBodyLedger.reserve(bytes)) throw new BridgeProcessCapacityError();
envelopeReservation += bytes;
});
} catch (error) {
// The host never delivered a response inside the deadline. Remove this
// request's file so it cannot pile up toward the queue-depth cap. The
@@ -2420,15 +2530,20 @@ async function runFileGateway() {
if (typeof value !== "string" || key.toLowerCase() === "content-length") continue;
res.setHeader(key, value);
}
res.end(typeof response.body === "string" ? response.body : "");
res.end(decodeSandboxBridgeBody(response, maxBodyBytes));
} catch (error) {
// A denied process-ledger reservation is retryable: the caller should
// try again once other in-flight bodies release their bytes. Every
// other body-read or handling fault stays a generic 502.
const status = error instanceof BridgeProcessCapacityError ? 503 : 502;
writeJsonResponse(res, status, { error: error instanceof Error ? error.message : String(error) });
// Capacity rejection before dispatch is retryable. After a mutation was
// dispatched, a missing or corrupt receipt cannot prove that it failed.
const uncertainWrite = dispatched && !["GET", "HEAD", "OPTIONS", "TRACE"].includes(req.method || "GET");
const status = uncertainWrite ? 409 : error instanceof BridgeProcessCapacityError ? 503 : 502;
if (uncertainWrite) res.setHeader("x-paperclip-bridge-outcome", "indeterminate");
writeJsonResponse(res, status, {
error: error instanceof Error ? error.message : String(error),
...(uncertainWrite ? { outcome: "indeterminate", retryable: false } : {}),
});
} finally {
releaseBodyReservation?.();
processBodyLedger.release(envelopeReservation);
}
});
@@ -126,6 +126,45 @@ describe("sandbox native file sync", () => {
expect(await readFile(path.join(repo, "outside.txt"), "utf8")).toBe("outside boundary\n");
});
it("prepares a runtime whose workspace directory does not exist without failing the ignore scan", async () => {
// An env test staging only credential assets can hand the runtime a
// workspace path that never existed on this host. A directory with no
// files has nothing for ignore rules to govern, so the scan is skipped
// rather than failed (git-ignore-scan-failed took down the whole
// preparation in production).
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-native-absent-workspace-"));
cleanupDirs.push(rootDir);
const remoteDir = path.join(rootDir, "remote");
const { client } = makeNativeClient();
const prepared = await prepareSandboxManagedRuntime({
spec: { transport: "sandbox", provider: "test", sandboxId: "s1", remoteCwd: remoteDir, timeoutMs: 30_000, apiKey: null },
adapterKey: "test-adapter",
client,
workspaceLocalDir: path.join(rootDir, "never-created"),
});
await prepared.restoreWorkspace();
});
it("fails preparation when the workspace cannot be read for a reason other than absence", async () => {
// Only absence means "nothing to sync". A workspace that is there but
// unreadable must not quietly become an empty remote workspace, so any
// other access error still fails the preparation. A path whose parent is
// a file gives a deterministic non-ENOENT error on every platform.
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-native-unreadable-workspace-"));
cleanupDirs.push(rootDir);
const notADirectory = path.join(rootDir, "a-file");
await writeFile(notADirectory, "not a directory\n");
const { client } = makeNativeClient();
await expect(prepareSandboxManagedRuntime({
spec: { transport: "sandbox", provider: "test", sandboxId: "s1", remoteCwd: path.join(rootDir, "remote"), timeoutMs: 30_000, apiKey: null },
adapterKey: "test-adapter",
client,
workspaceLocalDir: path.join(notADirectory, "workspace"),
})).rejects.toMatchObject({ code: "ENOTDIR" });
});
it("prefers the native path for default-provision asset inbound and workspace outbound", async () => {
const rootDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-native-sync-"));
cleanupDirs.push(rootDir);
@@ -340,6 +340,64 @@ function createRecordingTraceContext(): {
describe("sandbox managed runtime", () => {
const cleanupDirs: string[] = [];
it.each(["host_current", "adopt_remote", "durable_seed"] as const)("stages and restores both project repositories with independent Git histories (%s)", async (mode) => {
const root = await mkdtemp(path.join(os.tmpdir(), "paperclip-multi-repo-"));
cleanupDirs.push(root);
const local = path.join(root, "local");
const remote = path.join(root, "remote");
const secondPath = ".paperclip-repositories/backend";
for (const [relative, contents] of [["", "frontend"], [secondPath, "backend"]]) {
const cwd = path.join(local, relative!);
await mkdir(cwd, { recursive: true });
await git(cwd, ["init", "-b", "main"]);
await git(cwd, ["config", "user.name", "Test"]);
await git(cwd, ["config", "user.email", "test@example.com"]);
await writeFile(path.join(cwd, "README.md"), contents!);
await writeFile(path.join(cwd, ".gitignore"), "secret.txt\n");
await git(cwd, ["add", "."]);
await git(cwd, ["commit", "-m", contents!]);
await writeFile(path.join(cwd, "secret.txt"), "must stay local");
}
await writeFile(path.join(local, ".git/info/exclude"), ".paperclip-repositories/\n");
await writeFile(path.join(local, secondPath, "dirty.txt"), "local edit");
await writeFile(path.join(local, secondPath, "host-config.txt"), "excluded by operator");
const seed = { workspaceArchivePath: path.join(root, "workspace.tar"), gitArchivePath: path.join(root, "git.tar") };
const input = {
spec: { transport: "sandbox", provider: "test", sandboxId: "two-repos", remoteCwd: remote, timeoutMs: 30_000, apiKey: null },
client: makeFilesystemClient(), adapterKey: "test", workspaceLocalDir: local,
workspaceDurableSeed: seed,
workspaceExclude: [`${secondPath}/host-config.txt`],
} satisfies Parameters<typeof prepareSandboxManagedRuntime>[0];
let prepared = await prepareSandboxManagedRuntime(input);
if (mode !== "host_current") {
if (mode === "durable_seed") await rm(remote, { recursive: true, force: true });
await writeFile(path.join(local, secondPath, "host-only.txt"), "concurrent host work");
prepared = await prepareSandboxManagedRuntime({
...input, workspaceInboundMode: mode,
workspaceBaseline: prepared.workspaceSyncSnapshot!.baseline,
workspaceGitSnapshot: prepared.workspaceSyncSnapshot!.gitSnapshot,
});
}
for (const relative of ["", secondPath]) {
const cwd = path.join(remote, relative);
expect((await lstat(path.join(cwd, ".git"))).isDirectory()).toBe(true);
await expect(stat(path.join(cwd, "secret.txt"))).rejects.toMatchObject({ code: "ENOENT" });
await writeFile(path.join(cwd, "README.md"), `updated ${relative}`);
await git(cwd, ["-c", "user.name=Test", "-c", "user.email=test@example.com", "commit", "-am", "remote change"]);
}
expect(await readFile(path.join(remote, secondPath, "dirty.txt"), "utf8")).toBe("local edit");
await expect(stat(path.join(remote, secondPath, "host-config.txt"))).rejects.toMatchObject({ code: "ENOENT" });
await writeFile(path.join(remote, secondPath, "host-config.txt"), "remote must not replace host config");
await prepared.restoreWorkspace();
expect(await readFile(path.join(local, secondPath, "host-config.txt"), "utf8")).toBe("excluded by operator");
if (mode !== "host_current") expect(await readFile(path.join(local, secondPath, "host-only.txt"), "utf8")).toBe("concurrent host work");
for (const relative of ["", secondPath]) {
expect(await readFile(path.join(local, relative, "README.md"), "utf8")).toBe(`updated ${relative}`);
expect(await git(path.join(local, relative), ["log", "-1", "--format=%s"])).toBe("remote change");
expect(await readFile(path.join(local, relative, "secret.txt"), "utf8")).toBe("must stay local");
}
}, 30_000);
afterEach(async () => {
while (cleanupDirs.length > 0) {
const dir = cleanupDirs.pop();
@@ -1084,7 +1084,21 @@ export async function prepareSandboxManagedRuntime(input: {
}): Promise<PreparedSandboxManagedRuntime> {
const workspaceRemoteDir = input.workspaceRemoteDir ?? input.spec.remoteCwd;
const runtimeRootDir = path.posix.join(workspaceRemoteDir, ".paperclip-runtime", input.adapterKey);
const syncWorkspace = input.syncWorkspace !== false;
// A workspace directory that does not exist on this host has nothing to
// stage, no files for ignore rules to govern, and nothing to restore into —
// callers that only stage credential assets (the adapter env tests) hand
// the runtime a fresh path. Treat that one case as "do not sync" rather
// than letting the ignore scan or the staging walk die on ENOENT. Any other
// access failure still fails the preparation: a workspace that exists but
// cannot be read must not silently become an empty remote workspace.
const syncWorkspace = input.syncWorkspace !== false &&
(await fs.access(input.workspaceLocalDir).then(
() => true,
(error: NodeJS.ErrnoException) => {
if (error.code === "ENOENT") return false;
throw error;
},
));
const workspaceInboundMode = input.workspaceInboundMode ?? "host_current";
const stageWorkspace =
syncWorkspace && workspaceInboundMode !== "adopt_remote";
@@ -1152,6 +1166,8 @@ export async function prepareSandboxManagedRuntime(input: {
input.workspaceExclude,
gitIgnoredExcludes,
);
const repositories = gitSnapshot?.repositories ?? [];
const workspaceRestoreExclude = mergeExcludes(restoreExclude, repositories.map((repo) => repo.path));
// The baseline "before" snapshot: a recursive walk of the whole workspace that
// `lstat`s every entry and SHA-256-hashes every file's bytes. This is the
// dominant cost in the pre-`pack` window — it reads the content of every
@@ -1659,6 +1675,43 @@ export async function prepareSandboxManagedRuntime(input: {
if (syncWorkspace) {
outboundTasks.push(() =>
runStepSpan("restore.workspace", async () => {
// Each repository owns its Git history and merge. The parent baseline also
// records child files so restart recovery has their original merge inputs.
for (const repository of repositories) {
const prefix = `${repository.path}/`;
const localDir = path.join(input.workspaceLocalDir, repository.path);
if (await fs.realpath(localDir) !== path.join(await fs.realpath(input.workspaceLocalDir), repository.path)) {
throw new Error("Project repository escaped its workspace");
}
const nestedExclude = mergeExcludes(
repository.snapshot.ignoredPaths,
baselineSnapshot!.exclude.flatMap((entry) =>
entry.startsWith(prefix) ? [entry.slice(prefix.length)]
: entry.startsWith("*/") ? [entry] : []),
);
const nested = await prepareSandboxManagedRuntime({
spec: { ...input.spec, remoteCwd: path.posix.join(workspaceRemoteDir, repository.path) },
client: input.client,
adapterKey: input.adapterKey,
workspaceLocalDir: localDir,
workspaceInboundMode: "adopt_remote",
workspaceGitSnapshot: repository.snapshot,
workspaceExclude: nestedExclude,
workspaceBaseline: {
exclude: mergeExcludes(
SANDBOX_WORKSPACE_HEAVY_DIR_EXCLUDES,
[...GIT_ARCHIVE_EXCLUDES],
[".paperclip-runtime"],
nestedExclude,
),
entries: new Map([...baselineSnapshot!.entries]
.filter(([entry]) => entry.startsWith(prefix))
.map(([entry, value]) => [entry.slice(prefix.length), value])),
},
onRuntimeProgress: input.onRuntimeProgress,
});
await nested.restoreWorkspace(restoreSink);
}
await withTempDir("paperclip-sandbox-restore-", async (tempDir) => {
let importedRef: string | null = null;
let importedHead: string | null = null;
@@ -1768,7 +1821,7 @@ export async function prepareSandboxManagedRuntime(input: {
sourcePath: workspaceRemoteDir,
targetPath: extractedDir,
kind: "directory",
exclude: restoreExclude,
exclude: workspaceRestoreExclude,
}],
}];
assertSyncOperationsConfined(operations, {
@@ -1792,7 +1845,7 @@ export async function prepareSandboxManagedRuntime(input: {
`sh -c ${shellQuote(createRemoteTarballFromDirectoryCommand({
remoteDir: workspaceRemoteDir,
archivePath: remoteWorkspaceTar,
exclude: restoreExclude,
exclude: workspaceRestoreExclude,
}))}`,
{ timeoutMs: input.spec.timeoutMs },
);
@@ -1816,7 +1869,11 @@ export async function prepareSandboxManagedRuntime(input: {
}
const gitHeadToIntegrate = importedHead;
await mergeDirectoryWithBaseline({
baseline: baselineSnapshot!,
baseline: repositories.length === 0 ? baselineSnapshot! : {
exclude: workspaceRestoreExclude,
entries: new Map([...baselineSnapshot!.entries].filter(([entry]) =>
!repositories.some((repo) => entry === repo.path || entry.startsWith(`${repo.path}/`)))),
},
sourceDir: extractedDir,
targetDir: input.workspaceLocalDir,
beforeApply: gitHeadToIntegrate
@@ -3406,6 +3406,13 @@ describe("applyPaperclipWorkspaceEnv", () => {
});
describe("shapePaperclipWorkspaceEnvForExecution", () => {
it("maps editable project repositories inside the remote workspace", () => {
const result = shapePaperclipWorkspaceEnvForExecution({
workspaceCwd: "/host/task", executionCwd: "/sandbox/task", executionTargetIsRemote: true,
workspaceHints: [{ workspaceId: "backend", cwd: "/host/task/.paperclip-repositories/backend" }],
});
expect(result.workspaceHints).toEqual([{ workspaceId: "backend", cwd: "/sandbox/task/.paperclip-repositories/backend" }]);
});
it("rewrites workspace env paths for remote execution", () => {
const shaped = shapePaperclipWorkspaceEnvForExecution({
workspaceCwd: "/tmp/workspace",
@@ -3234,6 +3234,11 @@ export function shapePaperclipWorkspaceEnvForExecution(input: {
}
return nextHint;
}
const relative = localWorkspaceCwd ? path.relative(localWorkspaceCwd, hintCwd).split(path.sep).join("/") : "";
if (realizedWorkspaceCwd && /^\.paperclip-repositories\/[a-zA-Z0-9_-]+$/.test(relative)) {
nextHint.cwd = path.posix.join(realizedWorkspaceCwd, relative);
return nextHint;
}
// A referenced (mentioned) project hint carries its `projectId`. When the transport staged that
// project into the sandbox, repoint the hint at its staged `project-<projectId>` directory so
@@ -505,6 +505,25 @@ describe("Claude ACP hello probe on local and SSH targets", () => {
expect(JSON.stringify(spawnedEnv)).not.toContain("caller-proxy");
});
it("reports an explicitly selected API key as normal authentication on the ACP lane", async () => {
const result = await testClaudeAcpEnvironment({
companyId: "company-1",
adapterType: "claude_local",
config: { engine: "acp", agentCommand: process.execPath, env: { ANTHROPIC_API_KEY: "selected-test-key" } },
executionTarget: null,
environmentName: null,
});
expect(result.status).toBe("pass");
expect(result.checks).toContainEqual(expect.objectContaining({
code: "claude_acp_anthropic_api_key_detected",
level: "info",
message: "Using the selected Claude API connection.",
hint: undefined,
}));
expect(JSON.stringify(result.checks)).not.toContain("selected-test-key");
});
it("runs the host login probe with the host ANTHROPIC_API_KEY on a local target", async () => {
// A local ACP run inherits the host environment, so a host ANTHROPIC_API_KEY
// authenticates the real run. The Test lane runs the login probe with the
@@ -530,7 +549,9 @@ describe("Claude ACP hello probe on local and SSH targets", () => {
expect(result.checks.some((check) => check.code === "claude_hello_probe_auth_required")).toBe(false);
expect(result.checks.some((check) => check.code === "claude_acp_login_probe_unavailable")).toBe(false);
// The lane still reports that API-key auth is in use.
expect(result.checks.some((check) => check.code === "claude_acp_anthropic_api_key_detected")).toBe(true);
expect(result.checks).toContainEqual(expect.objectContaining({
code: "claude_acp_anthropic_api_key_detected", level: "warn",
}));
// The host key value never enters a check.
expect(JSON.stringify(result.checks)).not.toContain("sk-ant-host-key");
});
@@ -449,6 +449,28 @@ describe("claude_local ACP lane", () => {
});
});
it.each([undefined, "/sandbox/configured-workspace"])("checks sandbox directories on the sandbox (configured cwd=%s)", async (configuredCwd) => {
const remoteCwd = "/sandbox/workspace";
const mkdir = vi.spyOn(fs, "mkdir").mockRejectedValue(new Error("Host filesystem must not be used"));
const execute = vi.fn(async () => ({
exitCode: 0, signal: null, timedOut: false, stdout: "", stderr: "",
pid: null, startedAt: new Date().toISOString(),
}));
try {
const result = await testClaudeAcpEnvironment({
companyId: "company-1", adapterType: "claude_local",
config: { cwd: configuredCwd, agentCommand: "claude-agent-acp", env: { ANTHROPIC_API_KEY: "fixture" } },
executionTarget: { kind: "remote", transport: "sandbox", remoteCwd, runner: { execute } },
});
expect(result.status, JSON.stringify(result.checks)).toBe("pass");
expect(result.checks).toContainEqual(expect.objectContaining({
code: "claude_acp_cwd_valid", message: `Working directory is valid: ${configuredCwd ?? remoteCwd}`,
}));
expect(mkdir).not.toHaveBeenCalled();
expect(JSON.stringify(execute.mock.calls)).toContain(`mkdir -p '${configuredCwd ?? remoteCwd}'`);
} finally { mkdir.mockRestore(); }
});
it("reports ACP prerequisites for the ACP lane", async () => {
const root = await makeTempRoot("paperclip-claude-acp-env-");
const commandPath = path.join(root, "bin", "claude-agent-acp");
@@ -15,6 +15,7 @@ import {
} from "@paperclipai/adapter-utils/local-process-sandbox";
import {
ensureAdapterExecutionTargetCommandResolvable,
ensureAdapterExecutionTargetDirectory,
readAdapterExecutionTarget,
resolveAdapterExecutionTargetCwd,
runAdapterExecutionTargetProcess,
@@ -718,9 +719,13 @@ export async function testClaudeAcpEnvironment(
});
}
const cwd = asString(config.cwd, process.cwd());
const cwd = resolveAdapterExecutionTargetCwd(target, asString(config.cwd, ""), process.cwd());
try {
await fs.mkdir(cwd, { recursive: true });
await ensureAdapterExecutionTargetDirectory(`claude-acp-envtest-${Date.now()}`, target, cwd, {
cwd,
env: {},
createIfMissing: true,
});
checks.push({
code: "claude_acp_cwd_valid",
level: "info",
@@ -785,12 +790,13 @@ export async function testClaudeAcpEnvironment(
});
} else if (isNonEmpty(configApiKey) || isNonEmpty(hostApiKey)) {
const source = isNonEmpty(configApiKey) ? "adapter config env" : "server environment";
const selectedApiKey = Boolean(config.managedAiConnection) || isNonEmpty(configApiKey);
checks.push({
code: "claude_acp_anthropic_api_key_detected",
level: config.managedAiConnection ? "info" : "warn",
message: config.managedAiConnection ? "Using the selected Claude API connection." : "ANTHROPIC_API_KEY is set. Claude ACP will use API-key auth instead of subscription credentials.",
level: selectedApiKey ? "info" : "warn",
message: selectedApiKey ? "Using the selected Claude API connection." : "ANTHROPIC_API_KEY is set. Claude ACP will use API-key auth instead of subscription credentials.",
detail: `Detected in ${source}.`,
hint: config.managedAiConnection ? undefined : "Unset ANTHROPIC_API_KEY if you want subscription-based Claude login behavior.",
hint: selectedApiKey ? undefined : "Unset ANTHROPIC_API_KEY if you want subscription-based Claude login behavior.",
});
} else if (
isNonEmpty(envConfig.CLAUDE_CODE_OAUTH_TOKEN) ||
@@ -25,6 +25,26 @@ describe("explicit Claude Keychain import", () => {
await expect(readClaudeToken({ allowKeychain: true })).resolves.toBeNull();
expect(mocks.exec).not.toHaveBeenCalled();
});
it("skips an expired credentials file and falls through to Keychain", async () => {
vi.spyOn(process, "platform", "get").mockReturnValue("darwin");
vi.stubEnv("CLAUDE_CONFIG_DIR", "");
mocks.read.mockResolvedValue(JSON.stringify({ claudeAiOauth: { accessToken: "stale", expiresAt: Date.now() - 60_000 } }));
mocks.exec.mockResolvedValue({ stdout: JSON.stringify({ claudeAiOauth: { accessToken: "fresh", expiresAt: Date.now() + 60_000 } }) });
await expect(readClaudeToken({ allowKeychain: true })).resolves.toBe("fresh");
expect(mocks.exec).toHaveBeenCalledTimes(1);
});
it("returns null for an expired credentials file without Keychain access", async () => {
mocks.read.mockResolvedValue(JSON.stringify({ claudeAiOauth: { accessToken: "stale", expiresAt: Date.now() - 60_000 } }));
await expect(readClaudeToken()).resolves.toBeNull();
expect(mocks.exec).not.toHaveBeenCalled();
});
it("still accepts a credentials file that records no expiry", async () => {
vi.spyOn(process, "platform", "get").mockReturnValue("darwin");
vi.stubEnv("CLAUDE_CONFIG_DIR", "");
mocks.read.mockResolvedValue(JSON.stringify({ claudeAiOauth: { accessToken: "file" } }));
await expect(readClaudeToken({ allowKeychain: true })).resolves.toBe("file");
expect(mocks.exec).not.toHaveBeenCalled();
});
it("does not surface a credential-bearing subprocess error", async () => {
vi.spyOn(process, "platform", "get").mockReturnValue("darwin");
vi.stubEnv("CLAUDE_CONFIG_DIR", "");
@@ -92,10 +92,22 @@ async function readClaudeTokenFromFile(credPath: string): Promise<string | null>
} catch {
return null;
}
return parseClaudeCredentialToken(raw);
const credential = parseClaudeCredential(raw);
if (!credential) return null;
// On macOS the CLI refreshes the Keychain item, not this file, so a file
// whose token has expired is a stale leftover. Skip it so the caller can
// fall through to a live credential instead of failing with a dead token.
if (credential.expiresAt != null && credential.expiresAt <= Date.now()) return null;
return credential.token;
}
function parseClaudeCredentialToken(raw: string): string | null {
interface ClaudeCredential {
token: string;
/** Epoch milliseconds, when the credential file records one. */
expiresAt: number | null;
}
function parseClaudeCredential(raw: string): ClaudeCredential | null {
let parsed: unknown;
try {
parsed = JSON.parse(raw);
@@ -107,7 +119,13 @@ function parseClaudeCredentialToken(raw: string): string | null {
const oauth = obj["claudeAiOauth"];
if (typeof oauth !== "object" || oauth === null) return null;
const token = (oauth as Record<string, unknown>)["accessToken"];
return typeof token === "string" && token.length > 0 ? token : null;
if (typeof token !== "string" || token.length === 0) return null;
const expiresAt = (oauth as Record<string, unknown>)["expiresAt"];
return { token, expiresAt: typeof expiresAt === "number" && Number.isFinite(expiresAt) ? expiresAt : null };
}
function parseClaudeCredentialToken(raw: string): string | null {
return parseClaudeCredential(raw)?.token ?? null;
}
interface ClaudeAuthStatus {
@@ -205,13 +205,14 @@ export async function testEnvironment(
});
} else if (isNonEmpty(configApiKey) || isNonEmpty(hostApiKey)) {
const source = isNonEmpty(configApiKey) ? "adapter config env" : "server environment";
const selectedApiKey = Boolean(config.managedAiConnection) || isNonEmpty(configApiKey);
checks.push({
code: "claude_anthropic_api_key_overrides_subscription",
level: config.managedAiConnection ? "info" : "warn",
level: selectedApiKey ? "info" : "warn",
message:
config.managedAiConnection ? "Using the selected Claude API connection." : "ANTHROPIC_API_KEY is set. Claude will use API-key auth instead of subscription credentials.",
selectedApiKey ? "Using the selected Claude API connection." : "ANTHROPIC_API_KEY is set. Claude will use API-key auth instead of subscription credentials.",
detail: `Detected in ${source}.`,
hint: config.managedAiConnection ? undefined : "Unset ANTHROPIC_API_KEY if you want subscription-based Claude login behavior.",
hint: selectedApiKey ? undefined : "Unset ANTHROPIC_API_KEY if you want subscription-based Claude login behavior.",
});
} else if (
isNonEmpty(env.CLAUDE_CODE_OAUTH_TOKEN) ||
@@ -7,7 +7,11 @@ import {
readSubscriptionAccountId,
writeCodexAuthCacheEntry,
} from "./codex-auth-cache.js";
import { USE_SOURCE_EXIT, decideCodexAuthMerge } from "./codex-auth-merge-decision.js";
import {
IMPLAUSIBLE_LAST_REFRESH_EXIT,
USE_SOURCE_EXIT,
decideCodexAuthMerge,
} from "./codex-auth-merge-decision.js";
// The outbound copy-back reuses the exact same direction-agnostic decision
// predicate the inbound restore runs, through the shared `decideCodexAuthMerge`
@@ -122,6 +126,17 @@ export async function copyBackCodexAuth(input: CopyBackCodexAuthInput): Promise<
return "copied";
}
// Make a clock-bound rejection visible. A host with a wrong clock would
// otherwise discard an honest refresh with no signal. Log only the exit
// code and this caller's own label — never a timestamp read from either
// file, and never credential bytes.
if (decision === IMPLAUSIBLE_LAST_REFRESH_EXIT) {
await log(
`[paperclip] Codex auth copy-back: WARNING host credential kept (decision exit ${IMPLAUSIBLE_LAST_REFRESH_EXIT}, codex auth copy-back) — the sandbox copy's last_refresh sat further ahead of the host clock than the plausible skew allowance. Check the host clock if this is unexpected.`,
);
return "kept-host";
}
await log(
"[paperclip] Codex auth copy-back: host credential kept (sandbox copy is not a strictly-newer same-identity subscription credential).",
);
@@ -64,54 +64,107 @@ function parseAuth(filePath) {
// A leading positional flag (not an environment variable) keeps the mode
// explicit per call, so a host-default two-path call can never enter seed mode.
//
// Exit 10 = use source; exit 20 = keep destination. The predicate only ever
// reads the two files and exits with a code — it never prints token bytes.
// Exit contract. Exit 10 = use source; exit 20 = keep destination; exit 22 =
// keep destination, the source `last_refresh` sat further ahead of the host
// clock than the plausible skew allowance. The predicate only ever reads the
// two files and exits with a code — it never prints token bytes.
const USE_SOURCE = 10;
const KEEP_DESTINATION = 20;
const IMPLAUSIBLE_LAST_REFRESH = 22;
const SEED_IF_DEST_ABSENT_FLAG = "--seed-if-dest-absent";
const rawArgs = process.argv.slice(2);
const seedIfDestAbsent = rawArgs[0] === SEED_IF_DEST_ABSENT_FLAG;
const [sourceAuthPath, destinationAuthPath] = seedIfDestAbsent ? rawArgs.slice(1) : rawArgs;
const sourceAuth = parseAuth(sourceAuthPath);
const destinationAuth = parseAuth(destinationAuthPath);
// A `last_refresh` records an event that already happened, so an honest value
// always sits at or before the host clock. Only clock skew between a sandbox
// and the host explains a small future value. Five minutes is larger than
// real skew on a time-synchronised host, and it is short enough that a source
// which claims a `last_refresh` further ahead than this cannot be trusted.
const MAX_FUTURE_LAST_REFRESH_SKEW_MS = 5 * 60 * 1000;
// Seed mode only: fill an ABSENT (unusable) destination slot from a usable
// subscription source. A subscription-kind source is guaranteed usable and to
// carry a real account_id (parseAuth returns "subscription" only then), so this
// is never a random pick. This branch changes ONLY the destination-unusable
// case; the api-key and unusable-source guards below still keep the destination.
if (
seedIfDestAbsent &&
destinationAuth.kind === "unusable" &&
sourceAuth.kind === "subscription"
) {
process.exit(USE_SOURCE);
// `decide` takes already-parsed `{ kind, accountId, lastRefresh }` shapes plus
// a caller-supplied `nowMs` and the `seedIfDestAbsent` flag, so a test can
// drive the exact skew bound without spawning a process and racing wall-clock
// drift. Guard order, first match wins:
// 1. Seed mode, the destination is unusable, and the source is a usable
// subscription credential -> USE_SOURCE, unless the source fails the
// skew bound in step 3 below, in which case the absent slot stays
// absent (IMPLAUSIBLE_LAST_REFRESH). A poisoned seed would out-live
// every honest refresh, since nothing could ever compare as "strictly
// newer" than an unbounded future value, so the bound applies here too.
// 2. Either side unusable, a kind mismatch, the destination is an api-key
// credential, or the two sides carry a different account_id ->
// KEEP_DESTINATION.
// 3. The source `last_refresh` sits further ahead of `nowMs` than
// MAX_FUTURE_LAST_REFRESH_SKEW_MS -> IMPLAUSIBLE_LAST_REFRESH. Only the
// source is bounded, and only against the caller's clock: the source is
// the sandbox-supplied side, so it must never supply the reference time.
// 4. The source `last_refresh` is strictly greater than the destination's
// -> USE_SOURCE.
// 5. Otherwise (a tie, a null value on either side, or an older source) ->
// KEEP_DESTINATION.
function decide(source, destination, nowMs, seedIfDestAbsent) {
const sourceIsImplausible =
source.lastRefresh !== null &&
source.lastRefresh - nowMs > MAX_FUTURE_LAST_REFRESH_SKEW_MS;
// Seed mode only: fill an ABSENT (unusable) destination slot from a usable
// subscription source. A subscription-kind source is guaranteed usable and to
// carry a real account_id (parseAuth returns "subscription" only then), so this
// is never a random pick. This branch changes ONLY the destination-unusable
// case; the api-key and unusable-source guards below still keep the destination.
if (
seedIfDestAbsent &&
destination.kind === "unusable" &&
source.kind === "subscription"
) {
return sourceIsImplausible ? IMPLAUSIBLE_LAST_REFRESH : USE_SOURCE;
}
// Fail closed to the destination unless both sides are the same usable,
// subscription-kind identity — an unusable side, an api-key credential, a kind
// mismatch, or a different account_id all keep the destination copy.
if (
destination.kind === "unusable" ||
source.kind === "unusable" ||
source.kind !== destination.kind ||
destination.kind === "apikey" ||
source.accountId !== destination.accountId
) {
return KEEP_DESTINATION;
}
if (sourceIsImplausible) {
return IMPLAUSIBLE_LAST_REFRESH;
}
// Use the source credential only when it is strictly fresher: both sides must
// carry a parseable last_refresh and the source one must be strictly greater.
// Ties and null/unparseable freshness keep the destination copy so a spent
// single-use refresh token is never written over a good one.
if (
source.lastRefresh !== null &&
destination.lastRefresh !== null &&
source.lastRefresh > destination.lastRefresh
) {
return USE_SOURCE;
}
return KEEP_DESTINATION;
}
// Fail closed to the destination unless both sides are the same usable,
// subscription-kind identity — an unusable side, an api-key credential, a kind
// mismatch, or a different account_id all keep the destination copy.
if (
destinationAuth.kind === "unusable" ||
sourceAuth.kind === "unusable" ||
sourceAuth.kind !== destinationAuth.kind ||
destinationAuth.kind === "apikey" ||
sourceAuth.accountId !== destinationAuth.accountId
) {
process.exit(KEEP_DESTINATION);
if (require.main === module) {
const rawArgs = process.argv.slice(2);
const seedIfDestAbsent = rawArgs[0] === SEED_IF_DEST_ABSENT_FLAG;
const [sourceAuthPath, destinationAuthPath] = seedIfDestAbsent ? rawArgs.slice(1) : rawArgs;
const sourceAuth = parseAuth(sourceAuthPath);
const destinationAuth = parseAuth(destinationAuthPath);
process.exit(decide(sourceAuth, destinationAuth, Date.now(), seedIfDestAbsent));
}
// Use the source credential only when it is strictly fresher: both sides must
// carry a parseable last_refresh and the source one must be strictly greater.
// Ties and null/unparseable freshness keep the destination copy so a spent
// single-use refresh token is never written over a good one.
if (
sourceAuth.lastRefresh !== null &&
destinationAuth.lastRefresh !== null &&
sourceAuth.lastRefresh > destinationAuth.lastRefresh
) {
process.exit(USE_SOURCE);
}
process.exit(KEEP_DESTINATION);
module.exports = {
decide,
parseAuth,
USE_SOURCE,
KEEP_DESTINATION,
IMPLAUSIBLE_LAST_REFRESH,
MAX_FUTURE_LAST_REFRESH_SKEW_MS,
};
@@ -1,5 +1,6 @@
import { execFile as execFileCallback } from "node:child_process";
import { mkdtemp, rm, writeFile } from "node:fs/promises";
import { createRequire } from "node:module";
import os from "node:os";
import path from "node:path";
import { fileURLToPath } from "node:url";
@@ -8,6 +9,28 @@ import { afterEach, describe, expect, it } from "vitest";
const execFile = promisify(execFileCallback);
// The exact-boundary cases below drive the predicate's pure `decide` function
// directly (via `require`, never spawning a process), so the skew bound can be
// tested to the exact millisecond without racing subprocess-spawn wall-clock
// drift. Every other case in this file drives the REAL `.cjs` through a
// spawned `node` process (no stub), matching how the wrapper invokes it in
// production.
const decisionModule = createRequire(import.meta.url)(
fileURLToPath(new URL("./codex-auth-merge-decision.cjs", import.meta.url)),
) as {
decide: (
source: { kind: "subscription" | "apikey" | "unusable"; accountId?: string; lastRefresh: number | null },
destination: { kind: "subscription" | "apikey" | "unusable"; accountId?: string; lastRefresh: number | null },
nowMs: number,
seedIfDestAbsent?: boolean,
) => number;
USE_SOURCE: number;
KEEP_DESTINATION: number;
IMPLAUSIBLE_LAST_REFRESH: number;
MAX_FUTURE_LAST_REFRESH_SKEW_MS: number;
};
const { decide, IMPLAUSIBLE_LAST_REFRESH, MAX_FUTURE_LAST_REFRESH_SKEW_MS } = decisionModule;
// This suite pins the opt-in seed mode of the single decision predicate. The
// default (no-flag) call keeps the fail-closed host-default contract unchanged.
// The leading positional `--seed-if-dest-absent` flag adds one behaviour: fill
@@ -165,3 +188,94 @@ describe("codex-auth-merge-decision predicate seed mode", () => {
expect(seedCode).toBe(USE_SOURCE);
});
});
// This suite pins the host-clock bound on the source `last_refresh`. A
// `last_refresh` records an event that already happened, so an honest value
// never sits far ahead of the host clock; only clock skew explains a small
// future value. A source that claims a value further ahead than
// `MAX_FUTURE_LAST_REFRESH_SKEW_MS` cannot be trusted, since a same-account
// sandbox that controls its own `auth.json` could otherwise pin an unbounded
// future timestamp that every later honest refresh compares as older than.
// These cases drive `decide` directly with an explicit `nowMs`, so the bound
// is proven to the exact millisecond and never depends on the real clock.
describe("codex-auth-merge-decision predicate: host-clock bound on last_refresh", () => {
const USE_SOURCE = 10;
const KEEP_DESTINATION = 20;
function subscription(accountId: string, lastRefresh: number | null) {
return { kind: "subscription" as const, accountId, lastRefresh };
}
it("keeps the destination when the source last_refresh sits one millisecond beyond the bound", () => {
const nowMs = Date.now();
const source = subscription("acct", nowMs + MAX_FUTURE_LAST_REFRESH_SKEW_MS + 1);
const destination = subscription("acct", nowMs - 60_000);
expect(decide(source, destination, nowMs, false)).toBe(IMPLAUSIBLE_LAST_REFRESH);
});
it("uses the source when the source last_refresh sits exactly at the bound", () => {
const nowMs = Date.now();
const source = subscription("acct", nowMs + MAX_FUTURE_LAST_REFRESH_SKEW_MS);
const destination = subscription("acct", nowMs - 60_000);
expect(decide(source, destination, nowMs, false)).toBe(USE_SOURCE);
});
it("still uses the source for an ordinary same-account, strictly-newer, past timestamp", () => {
const nowMs = Date.now();
const source = subscription("acct", nowMs - 60_000);
const destination = subscription("acct", nowMs - 120_000);
expect(decide(source, destination, nowMs, false)).toBe(USE_SOURCE);
});
it("measures the bound against the host clock and not against a value embedded in either payload", () => {
// Fixed instants; only the caller-supplied `nowMs` moves between the two
// assertions, so a passing/failing bound can only be explained by the
// caller's clock, never by a value embedded in either payload.
const sourceLastRefresh = 2_000_000_000_000;
const destinationLastRefresh = 1_000_000_000_000;
const source = subscription("acct", sourceLastRefresh);
const destination = subscription("acct", destinationLastRefresh);
expect(decide(source, destination, sourceLastRefresh - MAX_FUTURE_LAST_REFRESH_SKEW_MS, false)).toBe(USE_SOURCE);
expect(decide(source, destination, sourceLastRefresh - MAX_FUTURE_LAST_REFRESH_SKEW_MS - 1, false)).toBe(
IMPLAUSIBLE_LAST_REFRESH,
);
});
it("keeps the destination for a same-identity source whose kind or account guard would otherwise fire, regardless of the bound", () => {
// A kind mismatch or a different account_id already keeps the destination
// before the bound runs, so an implausible source never surfaces as
// IMPLAUSIBLE_LAST_REFRESH when a different guard already applies.
const nowMs = Date.now();
const implausibleLastRefresh = nowMs + MAX_FUTURE_LAST_REFRESH_SKEW_MS + 1;
const differentAccount = decide(
subscription("acct-x", implausibleLastRefresh),
subscription("acct-y", nowMs - 60_000),
nowMs,
false,
);
expect(differentAccount).toBe(KEEP_DESTINATION);
});
// Seed mode fills an ABSENT destination slot from a usable subscription
// source. The bound applies there too: an absent destination that got
// seeded with an implausible future last_refresh would never be
// refreshable again, since no later honest refresh could ever compare as
// "strictly newer" than an unbounded future value. The seed-mode source in
// production is the sandbox's own `auth.json`
// (see `writeCodexAuthCacheEntry` in codex-auth-cache.ts), so this is the
// same untrusted input the write-back guard defends against.
it("seed mode keeps an absent destination slot absent when the source last_refresh is implausibly far in the future", () => {
const nowMs = Date.now();
const source = subscription("acct", nowMs + MAX_FUTURE_LAST_REFRESH_SKEW_MS + 1);
const destination = { kind: "unusable" as const, lastRefresh: null };
expect(decide(source, destination, nowMs, true)).toBe(IMPLAUSIBLE_LAST_REFRESH);
});
it("seed mode still fills an absent destination slot when the source last_refresh sits at or before the bound", () => {
const nowMs = Date.now();
const source = subscription("acct", nowMs + MAX_FUTURE_LAST_REFRESH_SKEW_MS);
const destination = { kind: "unusable" as const, lastRefresh: null };
expect(decide(source, destination, nowMs, true)).toBe(USE_SOURCE);
});
});
@@ -11,10 +11,12 @@ const execFile = promisify(execFileCallback);
// `codex-auth-merge-decision.cjs`. It reads only the two files, and it exits with
// a code; it never prints token bytes. Argument order sets source and
// destination (first = source, second = destination), so the caller frames the
// direction. Exit 10 = use source; exit 20 = keep destination. The leading
// `--seed-if-dest-absent` flag adds one behavior: fill an absent destination slot
// from a usable subscription source. This module gives every caller one shared
// entry point, so the predicate contract can never drift between callers.
// direction. Exit 10 = use source; exit 20 = keep destination; exit 22 = keep
// destination, the source `last_refresh` sat further ahead of the host clock
// than the plausible skew allowance. The leading `--seed-if-dest-absent` flag
// adds one behavior: fill an absent destination slot from a usable
// subscription source. This module gives every caller one shared entry point,
// so the predicate contract can never drift between callers.
const DECISION_SCRIPT_PATH = fileURLToPath(
new URL("./codex-auth-merge-decision.cjs", import.meta.url),
@@ -25,6 +27,10 @@ const SEED_IF_DEST_ABSENT_FLAG = "--seed-if-dest-absent";
export const USE_SOURCE_EXIT = 10;
/** Exit code: keep the destination credential. */
export const KEEP_DESTINATION_EXIT = 20;
/** Exit code: keep the destination; the source `last_refresh` was implausibly far in the future. */
export const IMPLAUSIBLE_LAST_REFRESH_EXIT = 22;
const KNOWN_EXIT_CODES = new Set([USE_SOURCE_EXIT, KEEP_DESTINATION_EXIT, IMPLAUSIBLE_LAST_REFRESH_EXIT]);
export interface DecideCodexAuthMergeOptions {
/** Opt in to the cache-slot seed mode: fill an absent destination from a
@@ -36,9 +42,10 @@ export interface DecideCodexAuthMergeOptions {
}
/**
* Runs the shared decision predicate and returns its exit code (10 or 20). A
* non-10/20 exit, or a failure to run `node`, is a hard failure: this throws so a
* broken predicate is never mistaken for a "keep destination" decision.
* Runs the shared decision predicate and returns its exit code (10, 20, or
* 22). Any other exit, or a failure to run `node`, is a hard failure: this
* throws so a broken predicate is never mistaken for a "keep destination"
* decision.
*/
export async function decideCodexAuthMerge(
sourcePath: string,
@@ -52,7 +59,7 @@ export async function decideCodexAuthMerge(
await execFile("node", args);
} catch (error) {
const code = (error as { code?: unknown }).code;
if (code === USE_SOURCE_EXIT || code === KEEP_DESTINATION_EXIT) {
if (typeof code === "number" && KNOWN_EXIT_CODES.has(code)) {
return code;
}
const detail =
@@ -65,9 +72,9 @@ export async function decideCodexAuthMerge(
: String(error);
throw new Error(`${options.errorLabel} decision predicate failed: ${detail}`);
}
// `execFile` resolved, so the predicate exited 0. The predicate always exits 10
// or 20, so a clean exit 0 is unexpected; fail loud.
// `execFile` resolved, so the predicate exited 0. The predicate always exits
// 10, 20, or 22, so a clean exit 0 is unexpected; fail loud.
throw new Error(
`${options.errorLabel} decision predicate exited 0 (expected 10 or 20)`,
`${options.errorLabel} decision predicate exited 0 (expected 10, 20, or 22)`,
);
}
@@ -596,6 +596,7 @@ describe("codex-auth-merge-decision predicate (source/destination)", () => {
const KEEP_DESTINATION = 20;
const USE_SOURCE = 10;
const IMPLAUSIBLE_LAST_REFRESH = 22;
function subscriptionAuth(input: {
accountId: string;
@@ -737,6 +738,18 @@ describe("codex-auth-merge-decision predicate (source/destination)", () => {
destinationAuth: "{not valid json",
expected: KEEP_DESTINATION,
},
{
// 400 days ahead is comfortably beyond the 5-minute skew allowance and
// any subprocess-spawn scheduling delay, so this integration-level
// check never depends on millisecond timing.
name: "source last_refresh implausibly far in the future → keep destination",
sourceAuth: subscriptionAuth({
accountId: "acct",
lastRefresh: new Date(Date.now() + 400 * 24 * 60 * 60 * 1000).toISOString(),
}),
destinationAuth: subscriptionAuth({ accountId: "acct", lastRefresh: OLDER }),
expected: IMPLAUSIBLE_LAST_REFRESH,
},
];
for (const entry of cases) {
@@ -761,4 +774,17 @@ describe("codex-auth-merge-decision predicate (source/destination)", () => {
expect(result.code).toBe(USE_SOURCE);
expect(result.output).not.toContain("SENTINEL");
});
it("never emits source token bytes for an implausibly-future last_refresh", async () => {
const result = await runDecision({
sourceAuth: subscriptionAuth({
accountId: "acct",
lastRefresh: new Date(Date.now() + 400 * 24 * 60 * 60 * 1000).toISOString(),
marker: "SECRET-SENTINEL",
}),
destinationAuth: subscriptionAuth({ accountId: "acct", lastRefresh: OLDER }),
});
expect(result.code).toBe(IMPLAUSIBLE_LAST_REFRESH);
expect(result.output).not.toContain("SENTINEL");
});
});
@@ -73,6 +73,8 @@ function summarizeProbeDetail(stdout: string, stderr: string, parsedError: strin
const CODEX_AUTH_REQUIRED_RE =
/(?:not\s+logged\s+in|login\s+required|authentication\s+required|unauthorized|invalid(?:\s+or\s+missing)?\s+api(?:[_\s-]?key)?|openai[_\s-]?api[_\s-]?key|api[_\s-]?key.*required|please\s+run\s+`?codex\s+login`?)/i;
const PROBE_CLEANUP_WARNING = "[paperclip] Codex probe cleanup incomplete";
async function prepareCodexHelloProbe(input: {
runId: string;
companyId: string;
@@ -214,11 +216,13 @@ async function prepareCodexHelloProbe(input: {
const probeHome = input.targetIsRemote
? path.posix.join(input.cwd, ".paperclip-runtime", "codex", `probe-home-${input.runId}`)
: path.join(os.tmpdir(), `paperclip-codex-probe-${input.runId}`);
// The local finally path retries cleanup independently of the model result.
if (!input.targetIsRemote) probeHomeLocalDir = probeHome;
return {
command: "sh",
args: [
"-c",
'set -e; mkdir -p "$CODEX_HOME"; umask 077; printf "%s" "$_PAPERCLIP_CODEX_AUTH_JSON" > "$CODEX_HOME/auth.json"; unset _PAPERCLIP_CODEX_AUTH_JSON; trap \'rm -rf "$CODEX_HOME"\' EXIT INT TERM; "$0" "$@"',
`set -e; umask 077; mkdir -p "$CODEX_HOME"; printf "%s" "$_PAPERCLIP_CODEX_AUTH_JSON" > "$CODEX_HOME/auth.json"; unset _PAPERCLIP_CODEX_AUTH_JSON; cleanup() { result=$?; trap - EXIT; rm -f "$CODEX_HOME/auth.json" || true; rm -rf "$CODEX_HOME" || printf '%s\\n' '${PROBE_CLEANUP_WARNING}' >&2; exit "$result"; }; trap cleanup EXIT; trap 'exit 130' INT; trap 'exit 143' TERM; "$0" "$@"`,
input.command,
...input.args,
],
@@ -384,7 +388,16 @@ export async function testEnvironment(
{ ...config, fastMode: false },
{ skipGitRepoCheck: targetIsSandbox },
);
const args = execArgs.args;
// A connection test needs one small response, not plugin catalog sync,
// repository instructions, or a durable session. Keep provider/model
// configuration intact while removing unrelated startup work.
const args = [...execArgs.args];
args.splice(args.length - 1, 0,
"-c", "features.plugins=false",
"-c", "features.remote_plugin=false",
"-c", "project_doc_max_bytes=0",
...(args.includes("--ephemeral") ? [] : ["--ephemeral"]),
);
if (execArgs.fastModeIgnoredReason) {
checks.push({
code: "codex_fast_mode_unsupported_model",
@@ -440,8 +453,20 @@ export async function testEnvironment(
},
);
const parsed = parseCodexJsonl(probe.stdout);
const detail = summarizeProbeDetail(probe.stdout, probe.stderr, parsed.errorMessage);
const authEvidence = `${parsed.errorMessage ?? ""}\n${probe.stdout}\n${probe.stderr}`.trim();
// Plugin-catalog login is separate from model authentication. Its
// warnings must not explain an unrelated provider/process failure.
const providerStderr = probe.stderr.split(/\r?\n/)
.filter((line) => !/\bcodex_core_plugins(?:::|:)/.test(line) && !line.includes(PROBE_CLEANUP_WARNING))
.join("\n");
const detail = summarizeProbeDetail(probe.stdout, providerStderr, parsed.errorMessage);
const authEvidence = parsed.errorMessage?.trim() || providerStderr;
if (probe.stderr.includes(PROBE_CLEANUP_WARNING)) {
checks.push({
code: "codex_probe_cleanup_incomplete",
level: "info",
message: "Temporary probe files could not be fully removed; this does not change the connection result.",
});
}
if (probe.timedOut) {
checks.push({
@@ -3,16 +3,23 @@ import { describe, expect, it, vi, beforeEach } from "vitest";
const ensureDirectoryMock = vi.hoisted(() => vi.fn(async () => {}));
const ensureCommandMock = vi.hoisted(() => vi.fn(async () => {}));
const runProcessMock = vi.hoisted(() => vi.fn());
const prepareRuntimeMock = vi.hoisted(() => vi.fn());
const stageGrokHomeMock = vi.hoisted(() => vi.fn(async () => "/tmp/staged-grok-home"));
vi.mock("@paperclipai/adapter-utils/execution-target", () => ({
describeAdapterExecutionTarget: () => "local",
ensureAdapterExecutionTargetCommandResolvable: ensureCommandMock,
ensureAdapterExecutionTargetDirectory: ensureDirectoryMock,
prepareAdapterExecutionTargetRuntime: prepareRuntimeMock,
resolveAdapterExecutionTargetCwd: (_target: unknown, configuredCwd: string, fallbackCwd: string) =>
configuredCwd || fallbackCwd,
runAdapterExecutionTargetProcess: runProcessMock,
}));
vi.mock("./grok-home.js", () => ({ stageGrokHomeForSync: stageGrokHomeMock }));
vi.mock("./grok-auth-copyback.js", () => ({ copyBackGrokAuth: vi.fn(async () => {}) }));
import { existsSync } from "node:fs";
import { parseGrokModelsOutput, testEnvironment } from "./test.js";
describe("parseGrokModelsOutput", () => {
@@ -38,6 +45,8 @@ describe("grok_local testEnvironment", () => {
ensureDirectoryMock.mockClear();
ensureCommandMock.mockClear();
runProcessMock.mockReset();
prepareRuntimeMock.mockReset();
stageGrokHomeMock.mockClear();
});
it("reports a healthy authenticated host with a working hello probe", async () => {
@@ -249,6 +258,58 @@ describe("grok_local testEnvironment", () => {
);
});
it("stages the managed account through a host temp workspace, never the remote cwd", async () => {
// The remote cwd is a path inside the sandbox. Handing it to the runtime
// as `workspaceLocalDir` made the host-side ignore scan fail on a
// directory that does not exist (git-ignore-scan-failed in production).
let workspaceLocalDirAtCall: string | undefined;
let workspaceExistedAtCall: boolean | undefined;
prepareRuntimeMock.mockImplementationOnce(async (input: { workspaceLocalDir: string }) => {
workspaceLocalDirAtCall = input.workspaceLocalDir;
workspaceExistedAtCall = existsSync(input.workspaceLocalDir);
return { assetDirs: { home: "/sandbox/assets/home" }, restoreWorkspace: async () => {} };
});
runProcessMock
.mockResolvedValueOnce({ exitCode: 0, signal: null, timedOut: false, stdout: "You are logged in with grok.com.\n\nDefault model: grok-4\n\nAvailable models:\n * grok-4", stderr: "" })
.mockResolvedValueOnce({ exitCode: 0, signal: null, timedOut: false, stdout: [JSON.stringify({ type: "text", data: "hello" }), JSON.stringify({ type: "end", stopReason: "EndTurn", sessionId: "s", requestId: "r" })].join("\n"), stderr: "" });
const result = await testEnvironment({
companyId: "company-1",
adapterType: "grok_local",
config: { command: "grok", cwd: "/sandbox/workspace", managedAiConnection: { provider: "xai" } },
executionTarget: { kind: "remote", transport: "sandbox" } as never,
});
expect(prepareRuntimeMock).toHaveBeenCalledOnce();
expect(prepareRuntimeMock).toHaveBeenCalledWith(
expect.objectContaining({ workspaceRemoteDir: "/sandbox/workspace" }),
);
expect(workspaceLocalDirAtCall).not.toBe("/sandbox/workspace");
expect(workspaceExistedAtCall).toBe(true);
// The staged home dir the runtime uploaded is what the probe must use.
expect(runProcessMock.mock.calls[0]?.[4]).toMatchObject({ env: expect.objectContaining({ GROK_HOME: "/sandbox/assets/home" }) });
expect(result.status).toBe("pass");
});
it("reports a failed runtime preparation as a check instead of crashing", async () => {
prepareRuntimeMock.mockRejectedValueOnce(
new Error("Workspace ignore scan failed: git-ignore-scan-failed"),
);
const result = await testEnvironment({
companyId: "company-1",
adapterType: "grok_local",
config: { command: "grok", cwd: "/sandbox/workspace", managedAiConnection: { provider: "xai" } },
executionTarget: { kind: "remote", transport: "sandbox" } as never,
});
expect(result.status).toBe("fail");
const unprepared = result.checks.find((check: { code: string }) => check.code === "grok_environment_unprepared");
expect(unprepared).toMatchObject({ level: "error" });
// A failed staging means there is no environment worth probing.
expect(runProcessMock).not.toHaveBeenCalled();
});
it("emits no adapter_auth_missing check for a local target with missing authentication", async () => {
// The canonical check gates sandbox login eligibility only. A local target
// has no sandbox login to offer, so the check must not appear.
+52 -16
View File
@@ -18,7 +18,8 @@ import {
resolveAdapterExecutionTargetCwd,
runAdapterExecutionTargetProcess,
} from "@paperclipai/adapter-utils/execution-target";
import { rm } from "node:fs/promises";
import { mkdtemp, rm } from "node:fs/promises";
import os from "node:os";
import path from "node:path";
import { stageGrokHomeForSync } from "./grok-home.js";
import { copyBackGrokAuth } from "./grok-auth-copyback.js";
@@ -151,22 +152,43 @@ export async function testEnvironment(
const env = normalizeEnv(config.env);
let stagedHome: string | undefined;
let runtimeWorkspaceLocalDir: string | undefined;
let restore: (() => Promise<void>) | undefined;
try {
if (config.managedAiConnection && targetIsRemote) {
const hostHome = env.GROK_HOME;
stagedHome = await stageGrokHomeForSync(hostHome, { runId });
const prepared = await prepareAdapterExecutionTargetRuntime({
runId, target, adapterKey: "grok", workspaceLocalDir: cwd,
assets: [{ key: "home", localDir: stagedHome, followSymlinks: true,
restore: async ({ assetDir, readFile }) => { await copyBackGrokAuth({
readSandboxAuth: () => readFile(path.posix.join(assetDir, "auth.json")),
hostHomeDir: hostHome, log: () => {},
}); },
}],
});
env.GROK_HOME = prepared.assetDirs.home;
restore = () => prepared.restoreWorkspace(() => {});
try {
const hostHome = env.GROK_HOME;
stagedHome = await stageGrokHomeForSync(hostHome, { runId });
// `cwd` is the remote target path here, never a host directory, so the
// runtime gets an empty host workspace to stage from and the remote
// path separately — the same split codex-local and opencode-local
// draw. Handing it the remote path as `workspaceLocalDir` made the
// host-side ignore scan fail on a directory that does not exist.
runtimeWorkspaceLocalDir = await mkdtemp(
path.join(os.tmpdir(), `paperclip-grok-envtest-${runId}-`),
);
const prepared = await prepareAdapterExecutionTargetRuntime({
runId, target, adapterKey: "grok",
workspaceLocalDir: runtimeWorkspaceLocalDir,
workspaceRemoteDir: cwd,
assets: [{ key: "home", localDir: stagedHome, followSymlinks: true,
restore: async ({ assetDir, readFile }) => { await copyBackGrokAuth({
readSandboxAuth: () => readFile(path.posix.join(assetDir, "auth.json")),
hostHomeDir: hostHome, log: () => {},
}); },
}],
});
env.GROK_HOME = prepared.assetDirs.home;
restore = () => prepared.restoreWorkspace(() => {});
} catch (err) {
// The environment test reports what is wrong with the environment; a
// credential-staging failure is a finding, not a crash.
checks.push({
code: "grok_environment_unprepared",
level: "error",
message: err instanceof Error ? err.message : "Could not stage the managed account into the environment",
});
}
}
const runtimeEnv = ensurePathInEnv({ ...process.env, ...env });
@@ -187,7 +209,10 @@ export async function testEnvironment(
}
const canRunProbe =
checks.every((check) => check.code !== "grok_cwd_invalid" && check.code !== "grok_command_unresolvable");
checks.every((check) =>
check.code !== "grok_cwd_invalid" &&
check.code !== "grok_command_unresolvable" &&
check.code !== "grok_environment_unprepared");
const configuredModel = asString(config.model, DEFAULT_GROK_LOCAL_MODEL).trim();
@@ -366,5 +391,16 @@ export async function testEnvironment(
checks,
testedAt: new Date().toISOString(),
};
} finally { try { await restore?.(); } finally { if (stagedHome) await rm(stagedHome, { recursive: true, force: true }); } }
} finally {
try { await restore?.(); } finally {
// Both temporary directories are removed even when one removal fails,
// and neither failure turns an answered environment test into a thrown
// error: these are best-effort temp directories, and the caller asked
// for checks.
await Promise.allSettled([
stagedHome ? rm(stagedHome, { recursive: true, force: true }) : undefined,
runtimeWorkspaceLocalDir ? rm(runtimeWorkspaceLocalDir, { recursive: true, force: true }) : undefined,
]);
}
}
}
+5 -1
View File
@@ -5,6 +5,7 @@ import { readFile, readdir } from "node:fs/promises";
import { fileURLToPath } from "node:url";
import postgres from "postgres";
import * as schema from "./schema/index.js";
import { withTransientWriteRetry } from "./transient-write-retry.js";
const MIGRATIONS_FOLDER = fileURLToPath(new URL("./migrations", import.meta.url));
const DRIZZLE_MIGRATIONS_TABLE = "__drizzle_migrations";
@@ -259,7 +260,10 @@ export function createDb(url: string, options?: DatabaseClientOptions) {
const sql = postgres(url, postgresJsOptions(resolved));
const key = hostPortKeyOrNull(url);
if (key) registerClient(key, sql);
return drizzlePg(sql, { schema });
// The registry keeps the real client (teardown must end the actual pool);
// drizzle gets the retrying face so a pooler-recycled socket replays the
// query instead of failing the request that happened to draw it.
return drizzlePg(withTransientWriteRetry(sql), { schema });
}
export async function getPostgresDataDirectory(url: string): Promise<string | null> {
@@ -0,0 +1,6 @@
CREATE TABLE IF NOT EXISTS "announcement_dismissals" (
"user_id" text NOT NULL,
"announcement_id" text NOT NULL,
"dismissed_at" timestamp with time zone DEFAULT now() NOT NULL,
CONSTRAINT "announcement_dismissals_user_id_announcement_id_pk" PRIMARY KEY("user_id","announcement_id")
);
@@ -0,0 +1,3 @@
CREATE TABLE IF NOT EXISTS "announcement_publications" (
"announcement_id" text PRIMARY KEY NOT NULL
);
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+14
View File
@@ -1933,6 +1933,20 @@
"when": 1789260211664,
"tag": "0277_uneven_lady_deathstrike",
"breakpoints": true
},
{
"idx": 278,
"version": "7",
"when": 1789308547419,
"tag": "0278_nappy_colonel_america",
"breakpoints": true
},
{
"idx": 279,
"version": "7",
"when": 1789390281190,
"tag": "0279_tired_deathstrike",
"breakpoints": true
}
]
}
@@ -0,0 +1,16 @@
import { pgTable, text, timestamp, primaryKey } from "drizzle-orm/pg-core";
// Only IDs from a validated feed are registered. This durable allowlist keeps
// offline dismissal retries working without accepting caller-invented IDs.
// It contains no announcement content, account data or interaction events.
export const announcementPublications = pgTable("announcement_publications", {
announcementId: text("announcement_id").primaryKey(),
});
// Instance-wide personal preference, like user_sidebar_preferences. No auth
// foreign key: local_trusted uses the synthetic local-board principal.
export const announcementDismissals = pgTable("announcement_dismissals", {
userId: text("user_id").notNull(),
announcementId: text("announcement_id").notNull(),
dismissedAt: timestamp("dismissed_at", { withTimezone: true }).notNull().defaultNow(),
}, (table) => ({ pk: primaryKey({ columns: [table.userId, table.announcementId] }) }));
+1
View File
@@ -208,3 +208,4 @@ export { chatTelegramDraftIds } from "./chat_telegram_draft_ids.js";
export { aiConnectionDefaults } from "./ai_connection_defaults.js";
export { aiProviderDefaults } from "./ai_provider_defaults.js";
export * from "./email.js";
export { announcementDismissals, announcementPublications } from "./announcement_dismissals.js";
@@ -0,0 +1,191 @@
import net from "node:net";
import { sql as drizzleSql } from "drizzle-orm";
import { afterEach, describe, expect, it } from "vitest";
import type { Sql } from "postgres";
import { closeRegisteredClients, createDb } from "./client.js";
import { isTransientWritePhaseError, withTransientWriteRetry } from "./transient-write-retry.js";
/** The exact shape postgres.js raises when the socket write fails. */
function writeClosedError(): Error {
const error = new Error("write CONNECTION_CLOSED db.example.internal:5432");
(error as Error & { code: string }).code = "CONNECTION_CLOSED";
return error;
}
/** A stub standing in for the root postgres.js client's `unsafe` surface. */
function stubSql(behavior: { failures: number; error?: () => Error }) {
let calls = 0;
const state = {
get calls() {
return calls;
},
};
const unsafe = (_query: string, _parameters?: unknown[]) => {
calls += 1;
const attempt = calls;
const shouldFail = attempt <= behavior.failures;
const outcome = () =>
shouldFail
? Promise.reject((behavior.error ?? writeClosedError)())
: Promise.resolve([{ attempt }]);
return {
then: (onFulfilled?: ((value: unknown) => unknown) | null, onRejected?: ((reason: unknown) => unknown) | null) =>
outcome().then(onFulfilled, onRejected),
values: () => outcome().then(() => [[attempt]]),
};
};
const sql = { unsafe } as unknown as Sql;
return { sql, state };
}
describe("isTransientWritePhaseError", () => {
it("matches only the write-phase connection-closed shape", () => {
expect(isTransientWritePhaseError(writeClosedError())).toBe(true);
const midQuery = new Error("read CONNECTION_CLOSED db.example.internal:5432");
(midQuery as Error & { code: string }).code = "CONNECTION_CLOSED";
expect(isTransientWritePhaseError(midQuery)).toBe(false);
const ended = new Error("write CONNECTION_ENDED db.example.internal:5432");
(ended as Error & { code: string }).code = "CONNECTION_ENDED";
expect(isTransientWritePhaseError(ended)).toBe(false);
expect(isTransientWritePhaseError(new Error("write CONNECTION_CLOSED x:1"))).toBe(false); // no code
expect(isTransientWritePhaseError(null)).toBe(false);
});
});
describe("withTransientWriteRetry", () => {
it("replays a query whose socket write failed, and returns the replay's rows", async () => {
const { sql, state } = stubSql({ failures: 1 });
const rows = await withTransientWriteRetry(sql).unsafe("select 1", []);
expect(rows).toEqual([{ attempt: 2 }]);
expect(state.calls).toBe(2);
});
it("replays the .values() form drizzle uses", async () => {
const { sql, state } = stubSql({ failures: 2 });
const values = await withTransientWriteRetry(sql).unsafe("select 1", []).values();
expect(values).toEqual([[3]]);
expect(state.calls).toBe(3);
});
it("gives up after the attempt budget and surfaces the driver error", async () => {
const { sql, state } = stubSql({ failures: Number.POSITIVE_INFINITY });
await expect(withTransientWriteRetry(sql).unsafe("select 1", [])).rejects.toThrow(
"write CONNECTION_CLOSED",
);
expect(state.calls).toBe(3);
});
it("does not replay an error that may have reached the server", async () => {
const midQuery = () => {
const error = new Error("read CONNECTION_CLOSED db.example.internal:5432");
(error as Error & { code: string }).code = "CONNECTION_CLOSED";
return error;
};
const { sql, state } = stubSql({ failures: Number.POSITIVE_INFINITY, error: midQuery });
await expect(withTransientWriteRetry(sql).unsafe("select 1", [])).rejects.toThrow(
"read CONNECTION_CLOSED",
);
expect(state.calls).toBe(1);
});
it("runs one execution no matter how many handlers attach to the pending query", async () => {
const { sql, state } = stubSql({ failures: 0 });
const pending = withTransientWriteRetry(sql).unsafe("select 1", []);
await Promise.all([pending, pending.catch(() => undefined), pending.then((rows) => rows)]);
expect(state.calls).toBe(1);
});
it("never executes a query twice when a caller both awaits it and takes its values", async () => {
// `.values()` selects the row shape of the one execution, like the driver.
// Starting a second execution here would repeat a mutation's effect.
const { sql, state } = stubSql({ failures: 0 });
const pending = withTransientWriteRetry(sql).unsafe("insert into t values (1)", []);
const rows = await pending;
const values = await pending.values();
expect(state.calls).toBe(1);
expect(values).toBe(rows);
});
it("takes the values shape when it is chosen before the query runs", async () => {
const { sql, state } = stubSql({ failures: 0 });
const pending = withTransientWriteRetry(sql).unsafe("select 1", []);
expect(await pending.values()).toEqual([[1]]);
expect(state.calls).toBe(1);
});
});
describe("createDb with the retrying client", () => {
// The same minimal wire-protocol fake as client-teardown-registry.test.ts:
// enough of the startup and query flow for drizzle to run a real query
// through the proxied client, proving the retry face is transparent.
function startFakePostgresServer(): Promise<{ server: net.Server; port: number }> {
const authOk = Buffer.from([0x52, 0, 0, 0, 8, 0, 0, 0, 0]);
const readyForQuery = Buffer.from([0x5a, 0, 0, 0, 5, 0x49]);
const emptyQueryReply = Buffer.concat([
Buffer.from([0x31, 0, 0, 0, 4]),
Buffer.from([0x32, 0, 0, 0, 4]),
Buffer.from([0x54, 0, 0, 0, 6, 0, 0]),
Buffer.from([0x43, 0, 0, 0, 0x0d, 0x53, 0x45, 0x4c, 0x45, 0x43, 0x54, 0x20, 0x30, 0]),
]);
const server = net.createServer((socket) => {
let greeted = false;
socket.on("data", () => {
if (!greeted) {
greeted = true;
socket.write(Buffer.concat([authOk, readyForQuery]));
return;
}
socket.write(Buffer.concat([emptyQueryReply, readyForQuery]));
});
socket.on("error", () => {});
});
return new Promise((resolve) => {
server.listen(0, "127.0.0.1", () => {
resolve({ server, port: (server.address() as net.AddressInfo).port });
});
});
}
let server: net.Server | null = null;
let url: string | null = null;
afterEach(async () => {
if (url) await closeRegisteredClients(url);
if (server) await new Promise((resolve) => server!.close(resolve));
server = null;
url = null;
});
it("still answers ordinary drizzle queries through the proxy", async () => {
const started = await startFakePostgresServer();
server = started.server;
url = `postgres://test:test@127.0.0.1:${started.port}/test`;
const db = createDb(url, { connectTimeoutSeconds: 5, prepare: false });
await expect(db.execute(drizzleSql`select 0`)).resolves.toBeDefined();
});
it("leaves every other client surface reachable through the proxy", async () => {
// Drizzle itself only awaits `unsafe()` or takes its `.values()`, but the
// client is reachable as `db.$client`, and callers use it as a tagged
// template, open transactions on it, and end it. A wrapper that broke any
// of those would fail far from here, so pin them against the real driver.
const started = await startFakePostgresServer();
server = started.server;
url = `postgres://test:test@127.0.0.1:${started.port}/test`;
const db = createDb(url, { connectTimeoutSeconds: 5, prepare: false });
const client = (db as unknown as { $client: Sql }).$client;
await expect(client`select 1`).resolves.toBeDefined();
await expect(client.unsafe("select 1", [])).resolves.toBeDefined();
await expect(
client.begin(async (tx) => {
await tx.unsafe("select 1", []);
return "transaction result";
}),
).resolves.toBe("transaction result");
await expect(client.end({ timeout: 1 })).resolves.toBeUndefined();
});
});
+90
View File
@@ -0,0 +1,90 @@
import type { Sql } from "postgres";
/**
* Replays a query whose bytes never reached the server.
*
* Behind a connection pooler (Neon's PgBouncer), the server side of an idle
* connection can be recycled while the client still holds the socket. The
* next query then fails at the socket write — postgres.js reports it as
* `write CONNECTION_CLOSED <host>:<port>` with `code: "CONNECTION_CLOSED"`.
* Because the write itself failed, the server never saw the query, so
* replaying it on a fresh connection cannot double-execute anything — the
* replay is safe for reads and writes alike. Every other error, including a
* connection lost mid-query (where the server may have acted), propagates
* untouched.
*
* Drizzle routes every non-transactional query through the root client's
* `unsafe`; queries inside `db.transaction()` run on the scoped client
* `sql.begin()` hands out, which this wrapper deliberately does not touch —
* a transaction that loses its connection must abort, not replay.
*/
const TRANSIENT_WRITE_ATTEMPTS = 3;
const TRANSIENT_WRITE_BACKOFF_MS = 50;
export function isTransientWritePhaseError(error: unknown): boolean {
return (
error instanceof Error &&
(error as { code?: unknown }).code === "CONNECTION_CLOSED" &&
error.message.startsWith("write CONNECTION_CLOSED")
);
}
type UnsafeFn = (query: string, parameters?: unknown[]) => {
values: () => Promise<unknown>;
} & PromiseLike<unknown>;
export function withTransientWriteRetry<T extends Sql>(sql: T): T {
const runWithRetry = async (execute: () => Promise<unknown>): Promise<unknown> => {
for (let attempt = 0; ; attempt++) {
try {
return await execute();
} catch (error) {
if (attempt >= TRANSIENT_WRITE_ATTEMPTS - 1 || !isTransientWritePhaseError(error)) {
throw error;
}
await new Promise((resolve) => setTimeout(resolve, TRANSIENT_WRITE_BACKOFF_MS * (attempt + 1)));
}
}
};
return new Proxy(sql, {
get(target, property, receiver) {
if (property !== "unsafe") return Reflect.get(target, property, receiver);
const unsafe = target.unsafe.bind(target) as UnsafeFn;
const retryingUnsafe = (query: string, parameters?: unknown[], ...rest: unknown[]) => {
if (rest.length > 0) {
// An options argument selects driver surfaces this wrapper does not
// model; leave those calls exactly as they were.
return (target.unsafe as (...args: unknown[]) => unknown)(query, parameters, ...rest);
}
// postgres.js queries execute lazily, exactly once, and `.values()`
// selects the row shape of that one execution rather than starting a
// second one. Mirror both halves: the execution starts when a consumer
// first settles the query, `.values()` marks the shape and returns the
// same pending query, and every later handler joins the same run. A
// shape chosen after the run started cannot change it — the same as
// the driver, and the reason a mutation can never execute twice here.
let started: Promise<unknown> | undefined;
let wantsValues = false;
const runOnce = () =>
(started ??= runWithRetry(() =>
wantsValues
? unsafe(query, parameters).values()
: Promise.resolve(unsafe(query, parameters)),
));
const pending = {
then: (onFulfilled?: ((value: unknown) => unknown) | null, onRejected?: ((reason: unknown) => unknown) | null) =>
runOnce().then(onFulfilled, onRejected),
catch: (onRejected?: ((reason: unknown) => unknown) | null) => runOnce().catch(onRejected),
finally: (onFinally?: (() => void) | null) => runOnce().finally(onFinally),
values: () => {
if (!started) wantsValues = true;
return pending;
},
};
return pending;
};
return retryingUnsafe;
},
}) as T;
}
+29
View File
@@ -56,6 +56,35 @@ stdin/stdout bridge admits the pinned Claude and Codex ACPX profiles. It
validates the exact model, session identity, tool catalog, structured input,
and terminal settlement at the process boundary. Pi remains unavailable.
Native Claude skill assignments travel in the runtime-context snapshot through
runnerd to the ACPX sidecar. After acquiring the provider lifetime lease, the
host materializes the assigned bundles under the isolated Claude home's
`skills/` directory before launch. Reopening a provider refreshes that snapshot;
project and ambient host settings remain excluded. This path is separate from
the legacy `claude_local` adapter's remote skill staging.
The isolated Claude settings pin both `model` and `availableModels` to the
user's requested ID. This keeps ACP from replacing an exact ID with a picker
alias during selection and verification. Users can keep selecting models from
the normal Claude catalog or entering custom IDs; unavailable models still fail
at the provider rather than silently falling back.
For ACPX Claude, `approve-reads` is shown as **Allow Paperclip reads**. The host
intersects the run's public tools with the implementation catalog's read effects
and writes exact MCP permission rules into the isolated Claude settings. The
`paperclip` connection is always the runner's authenticated tool bridge; ambient
MCP configuration is excluded. Tool hints and provider permission metadata cannot
grant access. Unassigned tools, writes, external tools, and provider-native
operations do not receive automatic read permission. Protocol completion and
task-delivery controls keep their existing separate allowance.
This runtime has no interactive permission handler. An operation that still
requires approval stops the turn with `approval_required`. The server marks the
task blocked, exposes the permission action to the operator, and disables
automatic retry. The operator must review the operation and the agent's
permission setting before retrying. Company access checks still run when each
Paperclip tool executes.
Runnerd selects only qualified provider profiles. Claude Managed and AWS
AgentCore receive immutable company-profile snapshots with explicit retention,
spend, and invocation limits. No provider process receives a Paperclip API
@@ -104,18 +104,18 @@ interface EmbeddedEvalReport {
startedAt: string;
finishedAt: string;
durationMs: number | null;
initialRevision: number;
finalRevision: number;
initialRevision: number | null;
finalRevision: number | null;
finalStateSummary?: string;
usage: {
agentTurns: number;
providerRequests: number | null;
inputTokens: number;
outputTokens: number;
cachedInputTokens: number;
reasoningTokens: number;
inputTokens?: number;
outputTokens?: number;
cachedInputTokens?: number;
reasoningTokens?: number;
providerReportedCostNanodollars?: number;
estimatedCostNanodollars: number;
estimatedCostNanodollars?: number;
pricingVersion: string;
} | null;
};
@@ -43,17 +43,17 @@ export interface EvalInspectorReport {
startedAt: string;
finishedAt: string;
durationMs: number | null;
initialRevision: number;
finalRevision: number;
initialRevision: number | null;
finalRevision: number | null;
usage: {
agentTurns: number;
providerRequests: number | null;
inputTokens: number;
outputTokens: number;
cachedInputTokens: number;
reasoningTokens: number;
inputTokens?: number;
outputTokens?: number;
cachedInputTokens?: number;
reasoningTokens?: number;
providerReportedCostNanodollars?: number;
estimatedCostNanodollars: number;
estimatedCostNanodollars?: number;
pricingVersion?: string;
} | null;
};
@@ -286,15 +286,17 @@ export function EvalReportInspector({
<dd>
{isPublic
? "Withheld from public replay"
: `r${evalReport.run.initialRevision} → r${evalReport.run.finalRevision}`}
: evalReport.run.initialRevision == null || evalReport.run.finalRevision == null
? "unavailable"
: `r${evalReport.run.initialRevision} → r${evalReport.run.finalRevision}`}
</dd>
</div>
<div>
<dt>Tokens</dt>
<dd>
{evalReport.run.usage === null
? "unknown"
: `${evalReport.run.usage.inputTokens} in · ${evalReport.run.usage.outputTokens} out · ${evalReport.run.usage.cachedInputTokens} cached`}
{evalReport.run.usage == null
? "unavailable"
: `${evalReport.run.usage.inputTokens ?? "unavailable"} in · ${evalReport.run.usage.outputTokens ?? "unavailable"} out · ${evalReport.run.usage.cachedInputTokens ?? "unavailable"} cached`}
</dd>
</div>
<div>
@@ -308,9 +310,9 @@ export function EvalReportInspector({
<div>
<dt>Estimated cost</dt>
<dd>
{evalReport.run.usage === null
? "unknown"
: `$${(evalReport.run.usage.estimatedCostNanodollars / 1_000_000_000).toFixed(6)}${evalReport.run.usage.pricingVersion ? ` · ${evalReport.run.usage.pricingVersion}` : ""}`}
{evalReport.run.usage?.estimatedCostNanodollars == null
? "unavailable"
: `$${(evalReport.run.usage.estimatedCostNanodollars / 1_000_000_000).toFixed(6)}${evalReport.run.usage.pricingVersion ? ` · ${evalReport.run.usage.pricingVersion}` : ""}`}
</dd>
</div>
<div>
@@ -585,8 +585,9 @@ export function TurnGroup({
<section className="pit-turn" data-turn-id={turn.id} aria-label={`Turn ${turn.ordinal}`}>
<h2 className="pit-turn-header">
<span className="pit-turn-label">
Turn {turn.ordinal} · {turn.mode} · {turn.toolCallCount} tool call
{turn.toolCallCount === 1 ? "" : "s"} ·{" "}
Turn {turn.ordinal} · {turn.mode} · {turn.mode === "replay" && turn.toolCallCount === 0
? "no recorded tool calls"
: `${turn.toolCallCount} tool call${turn.toolCallCount === 1 ? "" : "s"}`} ·{" "}
{new Date(turn.at).toISOString().slice(11, 19)}
</span>
{turn.stoppedByUser ? (
@@ -1 +1 @@
[{"annotations":{"exposure":"always","operationId":"get_task_context","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the active task and actor, including the exact approved Markdown revision when this issue has an accepted plan.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"get_task_context","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"get_task_history","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded comments on the active mock task.","inputSchema":{"additionalProperties":false,"properties":{"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":[],"type":"object"},"name":"get_task_history","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_documents","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List revisioned documents on the active mock task.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_documents","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"read_document","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the current revision of one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"}},"required":["key"],"type":"object"},"name":"read_document","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_document_revisions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded revision history for one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"},"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":["key"],"type":"object"},"name":"list_document_revisions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"report_progress","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append a durable progress comment to the active mock task.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Multiline progress update.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"report_progress","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable mock command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Mock entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"answer_status_question","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append the answer to a status-only wake without changing task disposition.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Concise status answer.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"answer_status_question","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable mock command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Mock entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationLine truncated
[{"annotations":{"exposure":"always","operationId":"get_task_context","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the active task and actor, including the exact approved Markdown revision when this issue has an accepted plan.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"get_task_context","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"get_task_history","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded comments on the active task.","inputSchema":{"additionalProperties":false,"properties":{"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":[],"type":"object"},"name":"get_task_history","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_documents","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List revisioned documents on the active task.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_documents","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"read_document","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the current revision of one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"}},"required":["key"],"type":"object"},"name":"read_document","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_document_revisions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded revision history for one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"},"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":["key"],"type":"object"},"name":"list_document_revisions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"report_progress","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append a durable progress comment to the active task.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Multiline progress update.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"report_progress","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"answer_status_question","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append the answer to a status-only wake without changing task disposition.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Concise status answer.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"answer_status_question","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"write_document","requiredClaimLine truncated
@@ -24,7 +24,7 @@
"prpVersion": 1,
"nativeExecutionVersion": 1,
"catalogVersion": 1,
"catalogSha256": "sha256:155849f666fffed8133d497c4323d42639eae7696699f9df049649f836e2edbc",
"catalogSha256": "sha256:d86fe304faa341cb46f92c93b6c80a6b8666770ebd915ffe61c7c9d3399e6524",
"driverContractVersion": 1,
"driverKind": "paperclip-deterministic",
"driverVersion": "1.0.0"
@@ -160,7 +160,7 @@
},
{
"path": "fixtures/evals/native-execution-seeded.json",
"sha256": "43bda8e713605d690a5e755f2d47eaea012d28fef81bd7dc787a5f9cacc507a7",
"sha256": "dcba9c50475b86260067a7e0789577fc7a5c0624fe3e3552bf3e111c8f9da0ac",
"expectation": "accept",
"compatibilityCase": "canonical"
},
@@ -248,6 +248,7 @@ impl AcpxProviderDescriptor {
permission_mode: self.permission_mode,
permission_mode_pinned: self.permission_mode_pinned,
system_instructions: self.instructions.clone(),
runtime_context: self.runtime_context.clone(),
tool_set,
expected_identity,
})
@@ -63,6 +63,7 @@ pub struct AcpxProviderSessionConfig {
pub permission_mode: AcpxPermissionMode,
pub permission_mode_pinned: bool,
pub system_instructions: String,
pub runtime_context: Value,
pub tool_set: AuthorizedToolSet,
pub expected_identity: Option<AcpxProviderSessionIdentity>,
}
@@ -992,7 +993,7 @@ fn bootstrap(
"permissionMode": config.permission_mode,
"permissionModePinned": config.permission_mode_pinned,
"systemInstructions": config.system_instructions,
"runtimeContext": Value::Null,
"runtimeContext": config.runtime_context,
"tools": &sidecar_tools,
"expectedIdentity": config.expected_identity,
}),
@@ -120,6 +120,9 @@ impl AcpxSidecarTransport {
"RUST_BACKTRACE",
"PAPERCLIP_NATIVE_MCP_NAME",
"PAPERCLIP_NATIVE_MCP_URL",
// The qualified sidecar configures the runner-owned gateway. Keep
// its credential with the name/URL; unrelated secrets stay excluded.
"PAPERCLIP_NATIVE_MCP_TOKEN",
"PAPERCLIP_ACPX_PROVIDER_PACKAGE_ROOT",
"PAPERCLIP_ACPX_PROVIDER_PACKAGE_MANIFEST",
];
@@ -84,6 +84,21 @@ fn run() -> Result<(), Box<dyn std::error::Error>> {
continue;
}
match mode {
"mcp-environment" => {
write_json(
&mut stdout,
&json!({
"protocolVersion": GENERATED_ACPX_SIDECAR_PROTOCOL_VERSION,
"id": id, "ok": true,
"result": {
"name": std::env::var("PAPERCLIP_NATIVE_MCP_NAME").ok(),
"url": std::env::var("PAPERCLIP_NATIVE_MCP_URL").ok(),
"hasToken": std::env::var("PAPERCLIP_NATIVE_MCP_TOKEN").is_ok(),
"hasUnrelatedSecret": std::env::var("UNRELATED_EVAL_SECRET").is_ok(),
}
}),
)?;
}
"silent" => continue,
"wrong-id" => {
write_json(&mut stdout, &success(id + 1, command, &request))?;
@@ -1140,6 +1140,15 @@ fn run() -> Result<(), Box<dyn std::error::Error>> {
send(json!({"id": id, "result": {"data": turns, "nextCursor": null}}))?;
}
"thread/read" => {
if descendant_notifications
&& message.pointer("/params/threadId").and_then(Value::as_str)
== Some("descendant-1")
{
send(json!({"id": id, "result": {"thread": {
"id": "descendant-1", "parentThreadId": state.thread_id
}}}))?;
continue;
}
if args
.iter()
.any(|arg| arg == "--require-lightweight-history")
@@ -1182,7 +1191,7 @@ fn run() -> Result<(), Box<dyn std::error::Error>> {
{
send(json!({"method": "thread/started", "params": {"thread": {
"id": "descendant-overflow",
"source": {"subAgent": {"thread_spawn": {"parent_thread_id": state.thread_id}}}
"parentThreadId": state.thread_id
}}}))?;
}
}
@@ -1431,10 +1440,26 @@ fn run() -> Result<(), Box<dyn std::error::Error>> {
"params": {"turn": {"id": provider_turn_id}}
}))?;
if descendant_notifications {
for index in 0..300 {
// Codex can announce a helper through the root's spawn receipt
// before emitting any thread/started notification for that helper.
send(json!({"method": "item/completed", "params": {
"threadId": state.thread_id, "turnId": provider_turn_id,
"item": {"id": "spawn-first-child", "type": "collabAgentToolCall",
"tool": "spawnAgent", "status": "completed",
"senderThreadId": state.thread_id,
"receiverThreadIds": ["descendant-0"]}
}}))?;
send(json!({"method": "turn/started", "params": {
"threadId": "descendant-0", "turnId": "first-child-turn"
}}))?;
// A helper turn may arrive before either spawn completion or thread/started.
send(json!({"method": "turn/started", "params": {
"threadId": "descendant-1", "turnId": "second-child-turn"
}}))?;
for index in 2..300 {
send(json!({"method": "thread/started", "params": {"thread": {
"id": format!("descendant-{index}"),
"source": {"subAgent": {"thread_spawn": {"parent_thread_id": state.thread_id}}}
"parentThreadId": state.thread_id
}}}))?;
}
send(json!({"method": "turn/completed", "params": {
@@ -1444,7 +1469,7 @@ fn run() -> Result<(), Box<dyn std::error::Error>> {
if args.iter().any(|value| value == "--descendant-overflow") {
send(json!({"method": "thread/started", "params": {"thread": {
"id": "descendant-overflow",
"source": {"subAgent": {"thread_spawn": {"parent_thread_id": state.thread_id}}}
"parentThreadId": state.thread_id
}}}))?;
}
if fail_after_second_turn_start && turn_start_count == 2 {
@@ -65,6 +65,39 @@ fn remember_descendant_thread(ids: &mut BTreeSet<String>, id: &str) -> Result<bo
}
Ok(ids.insert(id.to_owned()))
}
// Call only after validating the outer notification against the active root turn.
fn remember_spawned_descendants(
ids: &mut BTreeSet<String>,
root: &str,
method: &str,
params: &Value,
) -> Result<(), &'static str> {
let item = &params["item"];
if method != "item/completed"
|| item["type"] != "collabAgentToolCall"
|| item["tool"] != "spawnAgent"
|| item["status"] != "completed"
{
return Ok(());
}
if params["threadId"] != root || item["senderThreadId"] != root {
return Err("invalid_spawn_lineage");
}
let receivers = item["receiverThreadIds"]
.as_array()
.ok_or("invalid_spawn_lineage")?;
if receivers.iter().any(|id| {
id.as_str()
.is_none_or(|id| id.is_empty() || id.len() > 240 || id == root)
}) {
return Err("invalid_spawn_lineage");
}
for id in receivers {
remember_descendant_thread(ids, id.as_str().expect("validated receiver"))?;
}
Ok(())
}
type QuestionOptionLabels = BTreeMap<String, BTreeMap<String, String>>;
type QuestionSetMapping = (String, Value, QuestionOptionLabels);
@@ -2424,11 +2457,53 @@ impl CodexProvider {
) {
Ok(identity) => identity,
Err(_) => {
return Ok(Some(self.identity_failure(
method,
&params,
"thread_binding_mismatch",
)))
// Helpers can start before Codex publishes their spawn receipt.
// Verify lineage with the provider; never infer authority from
// the arrival of an otherwise foreign execution event.
let candidate = notification_thread_id(&params)
.filter(|id| !id.is_empty() && id.len() <= 240 && *id != self.thread_id)
.map(str::to_owned);
let verified = candidate.as_ref().is_some_and(|candidate| {
self.request(
"thread/read",
json!({"threadId": candidate, "includeTurns": false}),
)
.ok()
.is_some_and(|metadata| {
metadata.pointer("/thread/id").and_then(Value::as_str)
== Some(candidate.as_str())
&& matches!(
classify_notification_thread(
"thread/started",
&self.thread_id,
&self.descendant_thread_ids,
&metadata
),
Ok(NotificationThread::Descendant)
)
})
});
if verified {
let mut known = self.descendant_thread_ids.clone();
known.insert(candidate.expect("verified candidate"));
match classify_notification_thread(method, &self.thread_id, &known, &params)
{
Ok(NotificationThread::Descendant) => NotificationThread::Descendant,
_ => {
return Ok(Some(self.identity_failure(
method,
&params,
"thread_binding_mismatch",
)))
}
}
} else {
return Ok(Some(self.identity_failure(
method,
&params,
"thread_binding_mismatch",
)));
}
}
};
if identity == NotificationThread::Descendant {
@@ -2535,6 +2610,22 @@ impl CodexProvider {
"turn_binding_mismatch",
)));
}
if let Err(code) = remember_spawned_descendants(
&mut self.descendant_thread_ids,
&self.thread_id,
method,
&params,
) {
if code == "provider_descendant_capacity_exhausted" {
return Ok(Some(CodexProviderEvent::ResourceLimit {
diagnostic: json!({"code": code, "recoverable": false,
"classification": "resource_capacity", "limit": MAX_DESCENDANT_THREAD_IDS,
"message": "Codex reached the child-thread inventory limit.",
"method": bounded_method(method), "expectedThreadId": self.thread_id}),
}));
}
return Ok(Some(self.identity_failure(method, &params, code)));
}
if let Some(terminal_event_type) = terminal_event_type {
if self.active_provider_turn_id.is_none() {
return Err(LocalRunnerError::invalid(
@@ -3214,13 +3305,21 @@ fn classify_notification_thread(
if thread.is_none() || thread == Some(root) {
return Ok(NotificationThread::Root);
}
let parent = [
let parents: Vec<&str> = [
"/thread/parentThreadId",
"/thread/source/subAgent/thread_spawn/parent_thread_id",
"/thread/source/subAgent/threadSpawn/parentThreadId",
"/thread/source/subagent/thread_spawn/parent_thread_id",
]
.iter()
.find_map(|path| params.pointer(path).and_then(Value::as_str));
.filter_map(|path| params.pointer(path).and_then(Value::as_str))
.collect();
if parents.windows(2).any(|pair| pair[0] != pair[1]) {
return Err(LocalRunnerError::invalid(
"Codex notification has conflicting parent identity",
));
}
let parent = parents.first().copied();
if thread.is_some()
&& (thread.is_some_and(|id| descendants.contains(id))
|| (method == "thread/started"
@@ -4781,6 +4880,93 @@ mod notification_identity_tests {
NotificationThread::Root
);
}
#[test]
fn spawn_receipts_require_completed_root_authority_and_preserve_capacity() {
let receipt = json!({"threadId":"root", "turnId":"turn", "item":{
"type":"collabAgentToolCall", "tool":"spawnAgent", "status":"completed",
"senderThreadId":"root", "receiverThreadIds":["helper"]}});
let mut ids = BTreeSet::new();
remember_spawned_descendants(&mut ids, "root", "item/completed", &receipt).unwrap();
assert!(ids.contains("helper"));
assert_eq!(
classify_notification_thread(
"turn/started",
"root",
&ids,
&json!({"threadId":"helper", "turn":{"id":"child-turn"}})
)
.unwrap(),
NotificationThread::Descendant
);
for (field, value) in [
("senderThreadId", "foreign"),
("receiverThreadIds", "malformed"),
] {
let mut bad = receipt.clone();
bad["item"][field] = json!(value);
let mut empty = BTreeSet::new();
assert!(
remember_spawned_descendants(&mut empty, "root", "item/completed", &bad).is_err()
);
assert!(empty.is_empty());
}
for (field, value) in [
("tool", "sendInput"),
("status", "failed"),
("status", "inProgress"),
] {
let mut non_spawn = receipt.clone();
non_spawn["item"][field] = json!(value);
let mut empty = BTreeSet::new();
remember_spawned_descendants(&mut empty, "root", "item/completed", &non_spawn).unwrap();
assert!(empty.is_empty());
}
let mut full: BTreeSet<String> = (0..MAX_DESCENDANT_THREAD_IDS)
.map(|n| format!("child-{n}"))
.collect();
assert_eq!(
remember_spawned_descendants(&mut full, "root", "item/completed", &receipt),
Err("provider_descendant_capacity_exhausted")
);
assert_eq!(full.len(), MAX_DESCENDANT_THREAD_IDS);
}
#[test]
fn recognizes_explicit_parent_thread_lineage_without_granting_root_authority() {
let children = BTreeSet::from(["child".to_owned()]);
for parent in ["root", "child"] {
assert_eq!(
classify_notification_thread(
"thread/started",
"root",
&children,
&json!({"thread":{"id":"helper", "parentThreadId":parent}})
)
.unwrap(),
NotificationThread::Descendant
);
}
assert_eq!(
classify_notification_thread(
"thread/started",
"root",
&children,
&json!({"thread":{"id":"stranger", "parentThreadId":"foreign"}})
)
.unwrap(),
NotificationThread::UnrelatedInformation
);
assert!(classify_notification_thread(
"turn/started",
"root",
&children,
&json!({"threadId":"stranger", "parentThreadId":"root", "turn":{"id":"foreign-turn"}})
)
.is_err());
assert!(classify_notification_thread("thread/started", "root", &children,
&json!({"thread":{"id":"helper", "parentThreadId":"root", "source":{"subAgent":{"thread_spawn":{"parent_thread_id":"foreign"}}}}})).is_err());
}
#[test]
fn classifies_provider_lineage_before_root_authority() {
let children = BTreeSet::from(["child".to_owned()]);
@@ -52,6 +52,7 @@ fn config(directory: &std::path::Path) -> AcpxProviderSessionConfig {
permission_mode: AcpxPermissionMode::ApproveReads,
permission_mode_pinned: true,
system_instructions: "Complete the supplied task.".to_owned(),
runtime_context: serde_json::Value::Null,
tool_set: AuthorizedToolSet {
schema: "paperclip.runner.authorized-tools.v1".to_owned(),
schema_version: 1,
@@ -51,6 +51,7 @@ fn config(mode: &str) -> AcpxProviderSessionConfig {
permission_mode: AcpxPermissionMode::ApproveReads,
permission_mode_pinned: true,
system_instructions: "Complete the supplied task.".to_owned(),
runtime_context: serde_json::Value::Null,
tool_set: tool_set(),
expected_identity: None,
}
@@ -46,6 +46,7 @@ fn config(mode: &str) -> AcpxProviderSessionConfig {
permission_mode: AcpxPermissionMode::ApproveReads,
permission_mode_pinned: true,
system_instructions: "Complete the supplied task.".to_owned(),
runtime_context: serde_json::Value::Null,
tool_set: tool_set(),
expected_identity: None,
}
@@ -27,6 +27,7 @@ fn config(mode: &str) -> AcpxProviderSessionConfig {
permission_mode: AcpxPermissionMode::ApproveReads,
permission_mode_pinned: true,
system_instructions: "Complete the supplied task.".to_owned(),
runtime_context: serde_json::Value::Null,
tool_set: AuthorizedToolSet {
schema: "paperclip.runner.authorized-tools.v1".to_owned(),
schema_version: 1,
Loaded 100 of 488 files, more files were not shown because too many files have changed in this diff. Show more