mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-10 20:50:08 +02:00
## Thinking Path > - Paperclip manages agents and durable tasks across provider runs. > - Human questions must preserve task ownership and stop work until a real answer arrives. > - The continuation eval checks the final work, but its lifecycle oracle misses several broken waiting states. > - A correct final answer can hide a stale execution lock or a lost intermediate answer receipt. > - This pull request checks each wait and retains every question and run identity. > - A source audit records which waiting operations belong to native runtimes and which still require legacy API calls. ## Linked Issues or Issue Description **What existing behavior does this improve?** The existing Product E2E continuation oracle for human question and approval journeys. **Current behavior** A pending interaction can pass the waiting check even when its task has the wrong status, a stale lock, or a retry. Only the first pending question and first run receipts are checked at the final checkpoint. **Proposed behavior** Require a healthy wait on the same task and assignee. Preserve all intermediate run receipts and every pending question's answered identity. Allow a paused native provider question only when its pending runtime request identifies the running native run and execution lock. Related: #15548 and #15554 cover earlier bookkeeping slices. #15544 changes production continuation summaries; this PR changes the eval oracle and does not overlap that fix. ## What Changed - Check waiting state, execution ownership, and all answer/run identities in the existing lifecycle oracle. - Add negative calibrations for broken records and positive coverage for both semantic waits and paused provider questions. - Retain task activity at each checkpoint for inspection of successful persisted mutations. - Verify 1,000-cent company and agent budget limits before continuation work. Disable automatic cell rerolls. - Document runtime ownership, existing coverage, and the remaining instruction decision. Keep production instructions unchanged. ## Verification - `pnpm test:e2e:runner:unit`: 1,867 Vitest tests pass, one skips; 128 Node tests pass. - `pnpm test:e2e:runner:typecheck`: passes. - `pnpm -r typecheck`: passes. - `pnpm build`: passes. - Negative calibration: 27 added cases fail against the prior oracle and pass with these checks. - Four existing local continuation cells are selected for a separate bounded live canary. Live results are pending; no behavioral pass is claimed here. - The full local `pnpm test:run` suite was not repeated because embedded PostgreSQL was unavailable in the preceding workspace verification. Required Linux PR CI must pass before readiness. ## Risks The stronger oracle can expose existing product or fixture defects. A paused provider question and a terminal semantic wait have distinct valid states. The four-cell canary does not qualify approval/review, dependency unblock, crash races, remote execution, or general task quality. Task activity records successful persisted writes, not failed API attempts; repeated progress comments are not automatically defects. Historical eval grades remain unchanged. No production scheduling, prompt, tool, schema or migration changes. ## Model Used OpenAI Codex, GPT-6-based, with tool use and code execution. The exact serving model ID and context-window size are not exposed in this session. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [ ] All Paperclip CI gates are green - [ ] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip <noreply@paperclip.ing>
155 lines
6.1 KiB
TypeScript
155 lines
6.1 KiB
TypeScript
import type { ContinuationCheckpoint } from "./continuation-scoring.js";
|
|
export interface LifecycleSnapshot {
|
|
executionRunId: string | null;
|
|
scheduledRetry: unknown;
|
|
activeRecoveryAction: unknown;
|
|
monitorNextCheckAt: string | null;
|
|
}
|
|
export type LifecycleCheckpoint = ContinuationCheckpoint & {
|
|
lifecycle?: LifecycleSnapshot;
|
|
// Successful persisted mutations; not a count of failed tool/HTTP attempts.
|
|
activity?: unknown[];
|
|
};
|
|
type Interaction = {
|
|
id?: string;
|
|
kind?: string;
|
|
status?: string;
|
|
sourceRunId?: string;
|
|
payload?: {
|
|
runtimeRequestId?: string;
|
|
target?: {
|
|
type?: string;
|
|
issueId?: string;
|
|
key?: string;
|
|
revisionId?: string;
|
|
};
|
|
};
|
|
};
|
|
const nonempty = (value: unknown): value is string =>
|
|
typeof value === "string" && value.trim().length > 0;
|
|
|
|
/** Semantic questions yield the run; provider-native questions instead pause
|
|
* inside it. Only a pending runtime request bound to that running native run
|
|
* permits retaining its execution lock at a waiting checkpoint. */
|
|
function hasHealthyWait(checkpoint: LifecycleCheckpoint, pending: Interaction[]) {
|
|
const state = checkpoint.lifecycle;
|
|
if (!state || state.scheduledRetry !== null || state.activeRecoveryAction !== null ||
|
|
state.monitorNextCheckAt !== null || checkpoint.runs.length === 0) return false;
|
|
const active = checkpoint.runs.filter(run => run.status !== "succeeded");
|
|
if (active.length === 0)
|
|
return checkpoint.issue.status === "in_review" && state.executionRunId === null;
|
|
if (active.length !== 1) return false;
|
|
const run = active[0];
|
|
return run.status === "running" && run.runtimeMode === "native" && nonempty(run.id) &&
|
|
state.executionRunId === run.id && ["in_progress", "in_review"].includes(checkpoint.issue.status) &&
|
|
pending.some(interaction => interaction.kind === "ask_user_questions" &&
|
|
nonempty(interaction.id) && interaction.sourceRunId === run.id &&
|
|
nonempty(interaction.payload?.runtimeRequestId));
|
|
}
|
|
|
|
/** Independent durable-state oracle. It never interprets response prose. */
|
|
export function gradeLifecycleBaseline(checkpoints: LifecycleCheckpoint[]) {
|
|
const checks: Array<{ id: string; passed: boolean; detail: string }> = [];
|
|
const check = (id: string, passed: boolean, detail: string) =>
|
|
checks.push({ id, passed, detail });
|
|
const final = checkpoints.find((c) => c.phase === "final");
|
|
const waiting = checkpoints.filter((c) => c.phase !== "final");
|
|
check(
|
|
"lifecycle.evidence-present",
|
|
!!final?.lifecycle &&
|
|
waiting.length > 0 &&
|
|
waiting.every((c) => !!c.lifecycle),
|
|
"Every checkpoint must retain lifecycle evidence from the public task API.",
|
|
);
|
|
const first = waiting[0];
|
|
check(
|
|
"lifecycle.task-owner-preserved",
|
|
nonempty(first?.issue.id) && nonempty(first?.issue.assigneeAgentId) &&
|
|
checkpoints.every(c => c.issue.id === first.issue.id &&
|
|
c.issue.assigneeAgentId === first.issue.assigneeAgentId),
|
|
"Waiting and resuming retain the same task and assigned agent.",
|
|
);
|
|
for (const c of waiting) {
|
|
const pending = (c.interactions as Interaction[]).filter(
|
|
(i) => i.status === "pending",
|
|
);
|
|
check(
|
|
`lifecycle.${c.phase}.durable-wait`,
|
|
pending.some(
|
|
(i) =>
|
|
nonempty(i.id) &&
|
|
[
|
|
"ask_user_questions",
|
|
"request_confirmation",
|
|
"request_approval",
|
|
].includes(i.kind ?? ""),
|
|
),
|
|
"Waiting must have an identifiable pending interaction, not only an assistant message.",
|
|
);
|
|
check(
|
|
`lifecycle.${c.phase}.wait-state`,
|
|
hasHealthyWait(c, pending),
|
|
"A settled wait is in_review with no execution lock, retry, recovery or monitor. A bound native provider question may keep its paused run and lock.",
|
|
);
|
|
for (const i of pending.filter(
|
|
(i) => i.payload?.target?.type === "issue_document",
|
|
)) {
|
|
const target = i.payload!.target!;
|
|
check(
|
|
`lifecycle.${c.phase}.revision:${i.id}`,
|
|
target.issueId === c.issue.id &&
|
|
c.documents.some(
|
|
(d) =>
|
|
d.key === target.key &&
|
|
typeof d.latestRevisionId === "string" &&
|
|
d.latestRevisionId === target.revisionId,
|
|
),
|
|
"Plan confirmation binds this task and the recorded current revision.",
|
|
);
|
|
}
|
|
}
|
|
check(
|
|
"lifecycle.final.no-active-path",
|
|
!!final?.lifecycle &&
|
|
final.issue.status === "done" &&
|
|
final.lifecycle.executionRunId === null &&
|
|
final.lifecycle.scheduledRetry === null &&
|
|
final.lifecycle.activeRecoveryAction === null &&
|
|
final.lifecycle.monitorNextCheckAt === null &&
|
|
final.runs.length > 0 &&
|
|
final.runs.every((r) => !["queued", "running"].includes(r.status)),
|
|
"Completed work has no live run, execution lock, scheduled retry, recovery or monitor.",
|
|
);
|
|
check(
|
|
"lifecycle.final.no-pending-interaction",
|
|
!!final &&
|
|
!(final.interactions as Interaction[]).some(
|
|
(i) => i.status === "pending",
|
|
),
|
|
"Completion has no unresolved interaction.",
|
|
);
|
|
if (first && final) {
|
|
const questions = new Map(waiting.flatMap(c =>
|
|
(c.interactions as Interaction[]).filter(i => i.kind === "ask_user_questions" && i.status === "pending")
|
|
.map(i => [i.id, i] as const)));
|
|
for (const question of questions.values()) {
|
|
const answers = (final.interactions as Interaction[]).filter(i => i.id === question.id);
|
|
check(
|
|
`lifecycle.answer:${question.id}`,
|
|
nonempty(question.id) && answers.length === 1 &&
|
|
answers[0].kind === "ask_user_questions" && answers[0].status === "answered",
|
|
"Every observed pending question retains exactly one answered identity, including later questions.",
|
|
);
|
|
}
|
|
const finalIds = new Set(final.runs.map(r => r.id));
|
|
check(
|
|
"lifecycle.final.preserved-runs",
|
|
checkpoints.every(c => c.runs.length > 0 && c.runs.every(r => nonempty(r.id)) &&
|
|
new Set(c.runs.map(r => r.id)).size === c.runs.length) &&
|
|
waiting.every(c => c.runs.every(r => finalIds.has(r.id))),
|
|
"All intermediate run receipts survive in the final snapshot without duplicated or empty IDs.",
|
|
);
|
|
}
|
|
return checks;
|
|
}
|