mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-08 00:54:38 +02:00
## Thinking Path > - Paperclip manages AI agents and records their durable work outcomes. > - Product E2E checks those outcomes through the browser and public API. > - A successful long run can emit more than 1,000 durable events. > - The harness read one page and missed the later completion evidence. > - This pull request reads every page before it checks runtime invariants. > - Invalid or incomplete capture still fails. A missing page cannot produce a pass. ## Linked Issues or Issue Description Related: #13882 (Grok qualification). A duplicate search found no existing event-pagination fix. **What happened?** The local structured-question case in [campaign 36071063537](https://github.com/paperclipai/paperclip/actions/runs/36071063537) completed the task and passed its six outcome matchers. It failed native runtime invariants because the capture contained exactly 1,000 events. The last captured event preceded the run's completion by more than a minute. The API caps each response at 1,000 rows. The harness did not request the next page. **Expected behavior** Read the complete durable event stream through the public API before checking semantic-result and terminal-event counts. Reject incomplete or malformed evidence. **Steps to reproduce** 1. Complete a native task that emits more than 1,000 durable events. 2. Place the semantic-result and terminal events after row 1,000. 3. Capture the run with the Product E2E harness. 4. Before this fix, the invariant checker sees only the first page. **Paperclip version or commit** Observed at `4196a4cd76db434854b679035e4146c7f69689ce`. The same single-page capture exists on master. The original failed result remains unchanged; missing historical tail evidence is not reconstructed or graded as a pass. ## What Changed - Add a bounded event collector that advances through the public `afterSeq` cursor. - Use it for task success/failure evidence and shared chat run evidence. - Reject invalid pages, missing or non-increasing sequence numbers, repeated cursors, failed later requests, and an exhausted page limit. - Test completion events beyond the first page, exact page boundaries, and malformed evidence. - Correct the existing Everyday catalog test from 38 to the maintained 47 cells. The suite stays explicit-only. - Document the complete-capture requirement and its bound. ## Verification - Product harness typecheck passes. - The full credential-free harness suite passed 477 tests. After adding the chat integration regression, all 34 chat evidence tests pass. - Seventeen pagination tests cover the valid tail and malformed-evidence cases. - `git diff --check` passes. - [Full repository CI](https://github.com/paperclipai/paperclip/actions/runs/36080683422) passes at `aed6f79c089226be79e75dcf390969561ec4f787`: 52 successful checks and two intentional skips, including Rust, typecheck, build, server tests, and browser shards. Greptile gives this exact head 5/5 with no findings. No local Docker, Rust build, browser suite, or model invocation was used for this change. - [Live Grok requalification](https://github.com/paperclipai/paperclip/actions/runs/36080870743) is running on combined source `1b0551bb7c8de3c54f4bee64dbe2c88328b3645e`, with the runner, scheduler, and evidence fixes. Its full credential-free harness passes all 493 tests and typecheck. Live results are pending; the original campaign remains a failed measurement. ## Risks Long runs need more read-only API requests and larger private evidence files. Capture stops with an explicit error after 100 full pages. This changes neither production APIs nor provider behavior. It does not relax an invariant or change a historical grade. ## Model Used OpenAI GPT-6 through Codex, with repository tools and code execution. The exact serving identifier and context-window size are not exposed in this session. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [x] All Paperclip CI gates are green - [x] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge Co-authored-by: Paperclip <noreply@paperclip.ing>
305 lines
9.2 KiB
TypeScript
305 lines
9.2 KiB
TypeScript
/** Read the complete durable stream; never grade a silently truncated page. */
|
|
export async function collectRunEvents<T extends { seq?: number }>(
|
|
loadPage: (afterSeq: number, limit: number) => Promise<unknown>,
|
|
): Promise<T[]> {
|
|
const pageSize = 1000;
|
|
const events: T[] = [];
|
|
let afterSeq = 0;
|
|
for (let pageNumber = 0; pageNumber < 100; pageNumber += 1) {
|
|
const page = await loadPage(afterSeq, pageSize);
|
|
if (!Array.isArray(page) || page.length > pageSize) {
|
|
throw new Error("Run event evidence returned an invalid page");
|
|
}
|
|
for (const event of page) {
|
|
const seq = event?.seq;
|
|
if (!Number.isSafeInteger(seq) || seq <= afterSeq) {
|
|
throw new Error("Run event evidence has a missing or non-increasing sequence");
|
|
}
|
|
afterSeq = seq;
|
|
events.push(event as T);
|
|
}
|
|
if (page.length < pageSize) return events;
|
|
}
|
|
throw new Error("Run event evidence exceeded 100 pages; refusing incomplete evidence");
|
|
}
|
|
|
|
export interface ObservableRunState {
|
|
status?: string | null;
|
|
errorCode?: string | null;
|
|
}
|
|
|
|
export interface ObservableRunEvent {
|
|
eventType?: string;
|
|
payload?: Record<string, unknown> | null;
|
|
}
|
|
|
|
export interface ObservableProviderSessionRun {
|
|
id: string;
|
|
sessionIdBefore?: string | null;
|
|
sessionIdAfter?: string | null;
|
|
contextSnapshot?: Record<string, unknown> | null;
|
|
}
|
|
|
|
export interface ObservableMatcherResult {
|
|
matcher: {
|
|
kind?: string;
|
|
expected?: unknown;
|
|
count?: unknown;
|
|
};
|
|
passed: boolean;
|
|
}
|
|
|
|
export interface ObservableInteraction {
|
|
kind?: string | null;
|
|
status?: string | null;
|
|
payload?: unknown;
|
|
}
|
|
|
|
export interface OpenRouterHelloTerminalVarianceObservation {
|
|
suiteId: string;
|
|
profileId: string;
|
|
taskId: string;
|
|
expectedMarker: string;
|
|
finalRunMessage: string;
|
|
allAgentMessages: string;
|
|
semanticSummary: unknown;
|
|
issueStatus: string;
|
|
runStatuses: readonly string[];
|
|
matcherResults: readonly ObservableMatcherResult[];
|
|
invariantFailures: readonly string[];
|
|
}
|
|
|
|
export function acceptedPlanSessionResetFailures(
|
|
provider: "codex" | "opencode" | "acpx",
|
|
previousSessionId: string | null | undefined,
|
|
current: ObservableProviderSessionRun,
|
|
): string[] | null {
|
|
const context = record(current.contextSnapshot);
|
|
const acceptedPlanReset =
|
|
context.forceFreshSession === true &&
|
|
context.workspaceRefreshReason === "accepted_plan_confirmation" &&
|
|
context.source === "issue.interaction.accept" &&
|
|
context.interactionStatus === "accepted";
|
|
if (!acceptedPlanReset) return null;
|
|
|
|
const failures: string[] = [];
|
|
if (current.sessionIdBefore) {
|
|
failures.push(
|
|
`expected accepted Plan run ${current.id} to start without a prior provider session`,
|
|
);
|
|
}
|
|
if (
|
|
previousSessionId &&
|
|
current.sessionIdAfter &&
|
|
current.sessionIdAfter === previousSessionId
|
|
) {
|
|
failures.push(
|
|
`expected accepted Plan run ${current.id} to rotate the ${provider} provider session`,
|
|
);
|
|
}
|
|
return failures;
|
|
}
|
|
|
|
function record(value: unknown): Record<string, unknown> {
|
|
return value && typeof value === "object" && !Array.isArray(value)
|
|
? (value as Record<string, unknown>)
|
|
: {};
|
|
}
|
|
|
|
export function isNonExecutingReviewFenceRun(run: ObservableRunState) {
|
|
return (
|
|
run.status === "cancelled" &&
|
|
run.errorCode === "issue_continuation_waiting_on_review"
|
|
);
|
|
}
|
|
|
|
export function hasTerminalMalformedPlanConfirmation(input: {
|
|
runs: readonly ObservableRunState[];
|
|
interactions: readonly ObservableInteraction[];
|
|
minimumRunCount: number;
|
|
}) {
|
|
if (
|
|
input.runs.length < input.minimumRunCount ||
|
|
!input.runs.every((run) => run.status === "succeeded")
|
|
) {
|
|
return false;
|
|
}
|
|
|
|
return input.interactions.some((interaction) => {
|
|
if (
|
|
interaction.kind !== "request_confirmation" ||
|
|
interaction.status !== "pending"
|
|
) {
|
|
return false;
|
|
}
|
|
const target = record(record(interaction.payload).target);
|
|
return !(
|
|
target.type === "issue_document" &&
|
|
target.key === "plan" &&
|
|
typeof target.revisionId === "string" &&
|
|
target.revisionId.trim().length > 0
|
|
);
|
|
});
|
|
}
|
|
|
|
function normalizeMessage(value: string) {
|
|
return value
|
|
.replace(/\r\n/g, "\n")
|
|
.replace(/\\_/g, "_")
|
|
.replace(/[ \t]+/g, " ")
|
|
.trim();
|
|
}
|
|
|
|
function countOccurrences(value: string, expected: string) {
|
|
if (!expected) return 0;
|
|
let count = 0;
|
|
let cursor = 0;
|
|
while (cursor <= value.length - expected.length) {
|
|
const index = value.indexOf(expected, cursor);
|
|
if (index < 0) break;
|
|
count += 1;
|
|
cursor = index + expected.length;
|
|
}
|
|
return count;
|
|
}
|
|
|
|
export function isOpenRouterDeepSeekHelloTerminalVariance(
|
|
observation: OpenRouterHelloTerminalVarianceObservation,
|
|
) {
|
|
const marker = normalizeMessage(observation.expectedMarker);
|
|
const finalRunMessage = normalizeMessage(observation.finalRunMessage);
|
|
const allAgentMessages = normalizeMessage(observation.allAgentMessages);
|
|
const failedMatchers = observation.matcherResults.filter(
|
|
(result) => !result.passed,
|
|
);
|
|
const hasExpectedExactFailure = failedMatchers.some(
|
|
(result) =>
|
|
result.matcher.kind === "message_exact" &&
|
|
result.matcher.expected === observation.expectedMarker,
|
|
);
|
|
const hasExpectedOccurrenceFailure = failedMatchers.some(
|
|
(result) =>
|
|
result.matcher.kind === "message_occurrences" &&
|
|
result.matcher.expected === observation.expectedMarker &&
|
|
result.matcher.count === 1,
|
|
);
|
|
|
|
return (
|
|
observation.suiteId === "openrouter-model-breadth" &&
|
|
observation.profileId === "openrouter-deepseek-deepseek-v4-flash-0731" &&
|
|
observation.taskId === "hello-complete" &&
|
|
observation.issueStatus === "done" &&
|
|
observation.runStatuses.length === 1 &&
|
|
observation.runStatuses[0] === "succeeded" &&
|
|
observation.semanticSummary === observation.expectedMarker &&
|
|
finalRunMessage.length > 0 &&
|
|
countOccurrences(finalRunMessage, marker) === 0 &&
|
|
countOccurrences(allAgentMessages, marker) === 0 &&
|
|
observation.invariantFailures.length === 0 &&
|
|
failedMatchers.length === 2 &&
|
|
hasExpectedExactFailure &&
|
|
hasExpectedOccurrenceFailure
|
|
);
|
|
}
|
|
|
|
export function isControlPlaneGovernedResponseWait(
|
|
events: readonly ObservableRunEvent[],
|
|
) {
|
|
const accepted = events.filter(
|
|
(event) => event.eventType === "run.result.accepted",
|
|
);
|
|
if (accepted.length !== 1) return false;
|
|
const envelope = record(accepted[0]?.payload?.prpEvent);
|
|
const result = record(record(envelope.payload).result);
|
|
const continuation = record(result.continuation);
|
|
const idempotencyKey = continuation.idempotencyKey;
|
|
if (
|
|
typeof idempotencyKey !== "string" ||
|
|
!idempotencyKey.startsWith("interaction-response:")
|
|
) {
|
|
return false;
|
|
}
|
|
const interactionId = idempotencyKey.slice("interaction-response:".length);
|
|
if (!interactionId) return false;
|
|
const interactionRef = `interaction:${interactionId}`;
|
|
const hasEvidence =
|
|
Array.isArray(result.evidence) &&
|
|
result.evidence.some((value) => record(value).ref === interactionRef);
|
|
const hasInteractionArtifact =
|
|
Array.isArray(result.artifacts) &&
|
|
result.artifacts.some((value) => {
|
|
const artifact = record(value);
|
|
return (
|
|
artifact.kind === "issue_thread_interaction" &&
|
|
artifact.ref === interactionRef
|
|
);
|
|
});
|
|
return (
|
|
envelope.schema === "paperclip.prp.event.v1" &&
|
|
envelope.eventType === "run.result.accepted" &&
|
|
envelope.sourceKind === "control_plane" &&
|
|
result.schema === "paperclip.run_result.v1" &&
|
|
result.reportedWorkDisposition === "yielded" &&
|
|
continuation.kind === "response_wake" &&
|
|
hasEvidence &&
|
|
hasInteractionArtifact
|
|
);
|
|
}
|
|
|
|
export function providerSessionContinuityFailures(
|
|
provider: "codex" | "opencode",
|
|
runs: readonly ObservableProviderSessionRun[],
|
|
): string[] {
|
|
const failures: string[] = [];
|
|
for (let index = 0; index < runs.length; index += 1) {
|
|
const current = runs[index]!;
|
|
const currentSessionId = current.sessionIdAfter;
|
|
if (!currentSessionId) {
|
|
failures.push(
|
|
`expected ${provider} run ${current.id} to record provider session identity`,
|
|
);
|
|
continue;
|
|
}
|
|
if (index === 0) continue;
|
|
|
|
const previousSessionId = runs[index - 1]?.sessionIdAfter;
|
|
const acceptedPlanResetFailures = acceptedPlanSessionResetFailures(
|
|
provider,
|
|
previousSessionId,
|
|
current,
|
|
);
|
|
if (acceptedPlanResetFailures) {
|
|
failures.push(...acceptedPlanResetFailures);
|
|
continue;
|
|
}
|
|
|
|
if (!previousSessionId || currentSessionId !== previousSessionId) {
|
|
failures.push(
|
|
`expected ${provider} to preserve its provider session for run ${current.id}`,
|
|
);
|
|
continue;
|
|
}
|
|
if (
|
|
current.sessionIdBefore &&
|
|
current.sessionIdBefore !== previousSessionId
|
|
) {
|
|
failures.push(
|
|
`expected ${provider} run ${current.id} to resume provider session ${previousSessionId}`,
|
|
);
|
|
}
|
|
}
|
|
return failures;
|
|
}
|
|
|
|
export function numberedPlanStepCount(body: string | null | undefined) {
|
|
return (body ?? "").split(/\r?\n/).filter((line) => {
|
|
const normalized = line
|
|
.replaceAll("**", "")
|
|
.replaceAll("__", "")
|
|
.replaceAll("`", "");
|
|
return /^\s*(?:#{1,6}\s*)?(?:[-*+]\s*)?(?:step\s+)?\d+(?:[.)]|\s*[-—:])(?:\s|$)/i.test(
|
|
normalized,
|
|
);
|
|
}).length;
|
|
}
|