mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-10 20:50:08 +02:00
## Thinking Path > - Paperclip manages agents and durable tasks across provider runs. > - Human questions must preserve task ownership and stop work until a real answer arrives. > - The continuation eval checks the final work, but its lifecycle oracle misses several broken waiting states. > - A correct final answer can hide a stale execution lock or a lost intermediate answer receipt. > - This pull request checks each wait and retains every question and run identity. > - A source audit records which waiting operations belong to native runtimes and which still require legacy API calls. ## Linked Issues or Issue Description **What existing behavior does this improve?** The existing Product E2E continuation oracle for human question and approval journeys. **Current behavior** A pending interaction can pass the waiting check even when its task has the wrong status, a stale lock, or a retry. Only the first pending question and first run receipts are checked at the final checkpoint. **Proposed behavior** Require a healthy wait on the same task and assignee. Preserve all intermediate run receipts and every pending question's answered identity. Allow a paused native provider question only when its pending runtime request identifies the running native run and execution lock. Related: #15548 and #15554 cover earlier bookkeeping slices. #15544 changes production continuation summaries; this PR changes the eval oracle and does not overlap that fix. ## What Changed - Check waiting state, execution ownership, and all answer/run identities in the existing lifecycle oracle. - Add negative calibrations for broken records and positive coverage for both semantic waits and paused provider questions. - Retain task activity at each checkpoint for inspection of successful persisted mutations. - Verify 1,000-cent company and agent budget limits before continuation work. Disable automatic cell rerolls. - Document runtime ownership, existing coverage, and the remaining instruction decision. Keep production instructions unchanged. ## Verification - `pnpm test:e2e:runner:unit`: 1,867 Vitest tests pass, one skips; 128 Node tests pass. - `pnpm test:e2e:runner:typecheck`: passes. - `pnpm -r typecheck`: passes. - `pnpm build`: passes. - Negative calibration: 27 added cases fail against the prior oracle and pass with these checks. - Four existing local continuation cells are selected for a separate bounded live canary. Live results are pending; no behavioral pass is claimed here. - The full local `pnpm test:run` suite was not repeated because embedded PostgreSQL was unavailable in the preceding workspace verification. Required Linux PR CI must pass before readiness. ## Risks The stronger oracle can expose existing product or fixture defects. A paused provider question and a terminal semantic wait have distinct valid states. The four-cell canary does not qualify approval/review, dependency unblock, crash races, remote execution, or general task quality. Task activity records successful persisted writes, not failed API attempts; repeated progress comments are not automatically defects. Historical eval grades remain unchanged. No production scheduling, prompt, tool, schema or migration changes. ## Model Used OpenAI Codex, GPT-6-based, with tool use and code execution. The exact serving model ID and context-window size are not exposed in this session. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [ ] All Paperclip CI gates are green - [ ] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip <noreply@paperclip.ing>
203 lines
7.7 KiB
TypeScript
203 lines
7.7 KiB
TypeScript
import { describe, expect, it } from "vitest";
|
|
import {
|
|
gradeLifecycleBaseline,
|
|
type LifecycleCheckpoint,
|
|
} from "./lifecycle-baseline.js";
|
|
const idle = {
|
|
executionRunId: null,
|
|
scheduledRetry: null,
|
|
activeRecoveryAction: null,
|
|
monitorNextCheckAt: null,
|
|
};
|
|
function recording(): LifecycleCheckpoint[] {
|
|
const base = {
|
|
children: [],
|
|
documents: [],
|
|
attachments: [],
|
|
comments: [],
|
|
lifecycle: { ...idle },
|
|
runs: [{ id: "run-1", status: "succeeded", runtimeMode: "native" }],
|
|
};
|
|
return [
|
|
{
|
|
...structuredClone(base),
|
|
phase: "initial",
|
|
issue: { id: "task", status: "in_review", assigneeAgentId: "agent" },
|
|
interactions: [
|
|
{ id: "question", kind: "ask_user_questions", status: "pending" },
|
|
],
|
|
},
|
|
{
|
|
...structuredClone(base),
|
|
phase: "final",
|
|
issue: { id: "task", status: "done", assigneeAgentId: "agent" },
|
|
interactions: [
|
|
{ id: "question", kind: "ask_user_questions", status: "answered" },
|
|
],
|
|
runs: [
|
|
...base.runs,
|
|
{ id: "run-2", status: "succeeded", runtimeMode: "native" },
|
|
],
|
|
},
|
|
];
|
|
}
|
|
const failures = (rows: LifecycleCheckpoint[]) =>
|
|
gradeLifecycleBaseline(rows)
|
|
.filter((c) => !c.passed)
|
|
.map((c) => c.id);
|
|
describe("LCA Product E2E evidence calibration", () => {
|
|
it("LCA-03 accepts durable question and matching answered identity", () =>
|
|
expect(failures(recording())).toEqual([]));
|
|
it("LCA-03 rejects prose-only waiting even if final task completes", () => {
|
|
const r = recording();
|
|
r[0].interactions = [];
|
|
expect(failures(r)).toContain("lifecycle.initial.durable-wait");
|
|
});
|
|
it("LCA-03 rejects answer on a different question", () => {
|
|
const r = recording();
|
|
r[1].interactions = [{ id: "other", status: "answered" }];
|
|
expect(failures(r)).toContain("lifecycle.answer:question");
|
|
});
|
|
it("LCA-01 fails closed without API evidence", () => {
|
|
const r = recording();
|
|
delete r[1].lifecycle;
|
|
expect(failures(r)).toContain("lifecycle.evidence-present");
|
|
});
|
|
it.each([
|
|
"executionRunId",
|
|
"scheduledRetry",
|
|
"activeRecoveryAction",
|
|
"monitorNextCheckAt",
|
|
] as const)("LCA-01 rejects leftover %s", (field) => {
|
|
const r = recording();
|
|
Object.assign(r[1].lifecycle!, { [field]: "unexpected" });
|
|
expect(failures(r)).toContain("lifecycle.final.no-active-path");
|
|
});
|
|
it("LCA-12 rejects lost original receipt", () => {
|
|
const r = recording();
|
|
r[1].runs.shift();
|
|
expect(failures(r)).toContain("lifecycle.final.preserved-runs");
|
|
});
|
|
it("LCA-05 accepts current revision and rejects stale or foreign plan targets", () => {
|
|
const r = recording();
|
|
r[0].documents = [{ key: "plan", body: "Draft", latestRevisionId: "v2" }];
|
|
const target = {
|
|
type: "issue_document",
|
|
issueId: "task",
|
|
key: "plan",
|
|
revisionId: "v2",
|
|
};
|
|
r[0].interactions.push({
|
|
id: "plan-approval",
|
|
kind: "request_confirmation",
|
|
status: "pending",
|
|
payload: { target },
|
|
});
|
|
expect(failures(r)).toEqual([]);
|
|
target.revisionId = "v1";
|
|
expect(failures(r)).toContain("lifecycle.initial.revision:plan-approval");
|
|
target.revisionId = "v2";
|
|
target.issueId = "other";
|
|
expect(failures(r)).toContain("lifecycle.initial.revision:plan-approval");
|
|
});
|
|
});
|
|
|
|
|
|
describe("settled waiting and every resumed question", () => {
|
|
it.each(["todo", "in_progress", "blocked", "done", "cancelled"])(
|
|
"rejects a settled question left %s even when the final result succeeds", (status) => {
|
|
const r = recording();
|
|
r[0].issue.status = status;
|
|
expect(failures(r)).toContain("lifecycle.initial.wait-state");
|
|
},
|
|
);
|
|
it.each(["executionRunId", "scheduledRetry", "activeRecoveryAction", "monitorNextCheckAt"] as const)(
|
|
"rejects a waiting task with leftover %s", (field) => {
|
|
const r = recording();
|
|
Object.assign(r[0].lifecycle!, { [field]: "unexpected" });
|
|
expect(failures(r)).toContain("lifecycle.initial.wait-state");
|
|
},
|
|
);
|
|
it.each(["queued", "running", "failed", "cancelled", "timed_out"])(
|
|
"rejects an unexplained %s run during a settled wait", (status) => {
|
|
const r = recording();
|
|
r[0].runs[0].status = status;
|
|
expect(failures(r)).toContain("lifecycle.initial.wait-state");
|
|
},
|
|
);
|
|
it.each(["id", "assigneeAgentId"] as const)("rejects changed task %s", field => {
|
|
const r = recording();
|
|
r[1].issue[field] = "other";
|
|
expect(failures(r)).toContain("lifecycle.task-owner-preserved");
|
|
});
|
|
it("fails closed when waiting-state or assignee evidence is absent", () => {
|
|
const r = recording();
|
|
delete r[0].lifecycle;
|
|
delete r[0].issue.assigneeAgentId;
|
|
expect(failures(r)).toContain("lifecycle.initial.wait-state");
|
|
expect(failures(r)).toContain("lifecycle.task-owner-preserved");
|
|
});
|
|
function pausedRecording() {
|
|
const r = recording();
|
|
r[0].issue.status = "in_progress";
|
|
r[0].runs[0].status = "running";
|
|
r[0].lifecycle!.executionRunId = "run-1";
|
|
r[0].interactions = [{ id: "question", kind: "ask_user_questions", status: "pending",
|
|
sourceRunId: "run-1", payload: { runtimeRequestId: "request-1" } }];
|
|
r[1].runs = [r[1].runs[0]];
|
|
return r;
|
|
}
|
|
it("accepts the native provider-question bridge paused in its original run", () => {
|
|
expect(failures(pausedRecording())).toEqual([]);
|
|
});
|
|
it.each(["sourceRunId", "runtimeRequestId", "runtimeMode", "executionRunId"])(
|
|
"rejects an unbound provider pause: %s", field => {
|
|
const r = pausedRecording();
|
|
if (field === "runtimeMode") r[0].runs[0].runtimeMode = "legacy";
|
|
else if (field === "executionRunId") r[0].lifecycle!.executionRunId = "other";
|
|
else {
|
|
const interaction = r[0].interactions[0] as Record<string, any>;
|
|
if (field === "sourceRunId") interaction.sourceRunId = "other";
|
|
else delete interaction.payload.runtimeRequestId;
|
|
}
|
|
expect(failures(r)).toContain("lifecycle.initial.wait-state");
|
|
},
|
|
);
|
|
function twoQuestions() {
|
|
const r = recording();
|
|
const middle = structuredClone(r[1]);
|
|
middle.phase = "answered";
|
|
middle.issue.status = "in_review";
|
|
middle.interactions.push({ id: "second-question", kind: "ask_user_questions", status: "pending" });
|
|
r[1].interactions.push({ id: "second-question", kind: "ask_user_questions", status: "answered" });
|
|
r[1].runs.push({ id: "run-3", status: "succeeded", runtimeMode: "native" });
|
|
return [r[0], middle, r[1]];
|
|
}
|
|
it("accepts two sequential questions with all run and answer receipts retained", () => {
|
|
expect(failures(twoQuestions())).toEqual([]);
|
|
});
|
|
it.each(["missing", "replaced", "cancelled", "duplicate"])(
|
|
"rejects a %s second answer although the first answer and final task are correct", mutation => {
|
|
const r = twoQuestions();
|
|
const final = r[2].interactions as Array<Record<string, any>>;
|
|
if (mutation === "missing") final.pop();
|
|
else if (mutation === "replaced") final[1].id = "replacement";
|
|
else if (mutation === "cancelled") final[1].status = "cancelled";
|
|
else final.push({ ...final[1] });
|
|
expect(failures(r)).toContain("lifecycle.answer:second-question");
|
|
},
|
|
);
|
|
it("rejects losing the intermediate run even when the first receipt survives", () => {
|
|
const r = twoQuestions();
|
|
r[2].runs = r[2].runs.filter(run => run.id !== "run-2");
|
|
expect(failures(r)).toContain("lifecycle.final.preserved-runs");
|
|
});
|
|
it("rejects missing or empty question IDs", () => {
|
|
const r = recording();
|
|
r[0].interactions = [{ id: "", kind: "ask_user_questions", status: "pending" }];
|
|
r[1].interactions = [{ id: "", kind: "ask_user_questions", status: "answered" }];
|
|
expect(failures(r)).toContain("lifecycle.initial.durable-wait");
|
|
expect(failures(r)).toContain("lifecycle.answer:");
|
|
});
|
|
});
|