mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-06 21:05:21 +02:00
Return completed Agent Chat handoffs through a durable outbox and scope each generated update to its supplied tasks. Add recovery, browser delivery, result access, and calibrated quality coverage. Validated with two consecutive ten-case Claude/Codex campaigns, all CI checks, and a 5/5 review. Co-Authored-By: Paperclip <noreply@paperclip.ing>
82 lines
4.9 KiB
TypeScript
82 lines
4.9 KiB
TypeScript
/** Mechanical delivery/access evidence, not a prose-quality or truthfulness judge.
|
|
* Keep replies verbatim for semantic review; never accept a marker alone as output.
|
|
*/
|
|
type Row = Record<string, any>;
|
|
export type CompletionObservation = {
|
|
sourceId: string;
|
|
worker: Row;
|
|
documents: Row[];
|
|
comments: Row[];
|
|
runs: Row[];
|
|
marker: string;
|
|
/** Synthetic user request and released brief, never inferred from the reply. */
|
|
fixtureRequest?: string;
|
|
/** Only other tasks explicitly delegated by this isolated fixture. */
|
|
relatedTasks?: Array<{ task: Row; documents: Row[] }>;
|
|
renderedLinks?: Array<{ commentId: string; href: string }>;
|
|
};
|
|
export const completionReviewRubric = [
|
|
"Does the final source-thread answer accurately say the requested work is finished?",
|
|
"Does it describe the saved result rather than repeat an earlier handoff promise?",
|
|
"Can the user access that result without asking another question?",
|
|
"Does it avoid inventing verification, publication, or other work not in the evidence?",
|
|
"Does it avoid repeated completion announcements or redundant acknowledgements, while allowing corrections and updates about different tasks?",
|
|
] as const;
|
|
|
|
/** The start time exists only in the brief released after the source is idle. */
|
|
export function completionOutputUsesReleasedBrief(body: string): boolean {
|
|
return /\b(?:10\s*[:.h]\s*30|ten[-\s]+thirty|half[\s-]+past[\s-]+(?:ten|10))\b/i.test(body);
|
|
}
|
|
|
|
export function completionDelivery(observation: CompletionObservation) {
|
|
const { worker, documents, comments, runs, sourceId, marker } = observation;
|
|
const completedAt = Date.parse(worker.completedAt);
|
|
const outputs = documents.filter(d => d.issueId === worker.id &&
|
|
!new Set(["plan", "summary", "proposal"]).has(String(d.key ?? "").toLowerCase()) && typeof d.body === "string" && d.body.includes(marker));
|
|
const responses = comments.filter(c => c.issueId === sourceId && c.authorAgentId &&
|
|
Number.isFinite(completedAt) && Date.parse(c.createdAt) >= completedAt &&
|
|
runs.some(r => r.id === c.createdByRunId && r.agentId === c.authorAgentId &&
|
|
r.contextSnapshot?.issueId === sourceId && r.status === "succeeded"));
|
|
responses.sort((a, b) => a.createdAt.localeCompare(b.createdAt) || a.id.localeCompare(b.id));
|
|
const latestResponse = responses.at(-1);
|
|
const correlated = responses.filter(c => runs.some(r => r.id === c.createdByRunId &&
|
|
Array.isArray(r.contextSnapshot?.chatCompletionUpdates) &&
|
|
r.contextSnapshot.chatCompletionUpdates.some((update: Row) => update.id === worker.id && update.status === "done")));
|
|
const normalize = (text: string) => text.replace(/\s+/g, " ").trim();
|
|
// Delivery remains valid if a subsequent clarification omits the same link.
|
|
// Keep the latest reply separately for semantic review of the whole exchange.
|
|
const deliveries = correlated.map(response => {
|
|
const body = String(response.body ?? "");
|
|
const links = (observation.renderedLinks ?? []).filter(link => link.commentId === response.id).map(link => link.href);
|
|
const resultLinks = links.filter(link => {
|
|
try {
|
|
const url = new URL(link, "http://fixture.invalid");
|
|
const parts = url.pathname.split("/").map(decodeURIComponent);
|
|
const index = parts.indexOf("issues");
|
|
return index >= 0 && [worker.id, worker.identifier].filter(Boolean).includes(parts[index + 1]);
|
|
} catch { return false; }
|
|
});
|
|
const quotedOutput = outputs.some(d => normalize(d.body).length >= 60 && normalize(body).includes(normalize(d.body)));
|
|
return { response, resultLinks, quotedOutput };
|
|
});
|
|
const delivered = deliveries.filter(d => d.resultLinks.length > 0 || d.quotedOutput).at(-1);
|
|
const final = delivered?.response ?? latestResponse;
|
|
const resultLinks = delivered?.resultLinks ?? [];
|
|
const quotedOutput = delivered?.quotedOutput ?? false;
|
|
return {
|
|
checks: [
|
|
{ id: "completion-worker-done", passed: worker.status === "done" && Number.isFinite(completedAt), detail: "The delegated task is durably Done, not merely a successful run" },
|
|
{ id: "completion-output-saved", passed: outputs.length > 0, detail: "The worker saved a non-plan deliverable containing the requested reference" },
|
|
{ id: "completion-source-response", passed: Boolean(final), detail: "The originating thread has a run-attributed agent reply after task completion" },
|
|
{ id: "completion-source-correlated", passed: correlated.length > 0, detail: "A reply run received authoritative completion facts for this task; a late initial handoff reply does not qualify" },
|
|
{ id: "completion-result-access", passed: Boolean(final) && (resultLinks.length > 0 || quotedOutput), detail: "That reply links to the completed task/output or includes the actual saved output" },
|
|
],
|
|
response: final ?? null,
|
|
latestResponse: latestResponse ?? null,
|
|
responses,
|
|
resultLinks,
|
|
quotedOutput,
|
|
semanticReview: { status: "required", rubric: completionReviewRubric },
|
|
};
|
|
}
|