From be604fcbf81ff238ee38774ff6e644bf0d84eb2d Mon Sep 17 00:00:00 2001 From: Dotta Date: Sun, 4 Oct 2026 20:27:53 -0500 Subject: [PATCH] fix(native): retain actionable final replies and document links Co-Authored-By: Paperclip --- ...2026-10-04-native-completion-answer-fix.md | 15 +++++ .../native-instruction-measurement.test.ts | 10 ++-- .../src/backends/runtime-context.test.ts | 2 + .../src/backends/runtime-context.ts | 4 +- .../paperclip-runner-tool-authority.test.ts | 1 + .../paperclip-runner-tool-authority.ts | 6 ++ tests/runner-e2e/README.md | 2 + tests/runner-e2e/native-completion-cases.ts | 2 +- .../native-completion-content.test.ts | 55 +++++++++++++++++++ tests/runner-e2e/native-completion-content.ts | 36 ++++++++++++ .../native-completion-scoring.test.ts | 25 ++++++++- tests/runner-e2e/native-completion-scoring.ts | 8 ++- .../native-instruction-consolidation.test.ts | 2 +- .../native-instruction-consolidation.ts | 13 +++++ tests/runner-e2e/runner.spec.ts | 26 ++++++++- 15 files changed, 195 insertions(+), 12 deletions(-) create mode 100644 doc/plans/2026-10-04-native-completion-answer-fix.md create mode 100644 tests/runner-e2e/native-completion-content.test.ts create mode 100644 tests/runner-e2e/native-completion-content.ts diff --git a/doc/plans/2026-10-04-native-completion-answer-fix.md b/doc/plans/2026-10-04-native-completion-answer-fix.md new file mode 100644 index 0000000000..6dab6cfb4a --- /dev/null +++ b/doc/plans/2026-10-04-native-completion-answer-fix.md @@ -0,0 +1,15 @@ +# Native completion final-answer correction + +The completion constraint reduction remains held until the corrected source has live evidence for final-answer quality. Keep the original candidate, historical source and all original grades immutable. + +The correction keeps the shortened completion procedure while explicitly requiring a blocker explanation, owner and unblock action in the final reply. The existing document guidance now requires a canonical clickable document link. `write_document` returns a company/task-scoped `documentHref` in its idempotent receipt, derived in the document transaction from the saved key and actual task identity. No completion tool schema/description, fixed prompt, skill, permission or budget policy changes. This receipt repair is an additional production path beyond the original three-file instruction reduction. + +The Product E2E observation advances to v3. Its additional checks read only the actual run-attributed persisted provider final. A blocker label and action do not substitute for an explicit missing-access explanation. Document links must identify the exact task and one revisioned document at the instance origin; wrong origin, task, document, query, credentials or absent evidence fail. Exact call-ID joins, acceptance/termination, ordering, side-effect and budget checks remain enforced. + +After saving the strict native snapshot, each completion cell clicks the rendered final-reply link in the browser. It must open the canonical saved document and show the original content marker. Retain the navigation receipt and screenshot independently; a valid-looking Markdown URL alone does not qualify navigation. + +Calibrate with correct paraphrases and plausible wrong answers, including the action-only omission. Replay the retained observations only as labeled additional diagnostics; never overwrite or regrade the original results. Provider-free measurement still captures eighteen scripted start/resume/continuation projections. No synthetic receipt qualifies live model behavior. + +The human's paid-run authorization and subsequent request to fix the omission support one bounded corrective confirmation: the six exact `native-instruction-consolidation` candidate cells, one attempt each, no automatic retries, native Codex `gpt-5.6-sol`, ACPX Claude `claude-sonnet-5`, and OpenCode `openrouter/deepseek/deepseek-v4-flash-0731`, original local deadlines, and 1,000-cent company/agent hard stops. This is a new source/definition measurement, not a reroll of a frozen failure. No new baseline provider run, broader campaign or merge is part of this correction. Stop and inspect a usable behavior failure. + +Report exact source, definition/grader hashes, six outcomes, retained visible replies and document targets, actual run count, cleanup and billing coverage. Passing a single trial does not prove general equivalence, cause, live resume behavior, coding quality, speed or cost trends. diff --git a/packages/paperclip-runner/src/backends/native-instruction-measurement.test.ts b/packages/paperclip-runner/src/backends/native-instruction-measurement.test.ts index 4487533c64..b807357eff 100644 --- a/packages/paperclip-runner/src/backends/native-instruction-measurement.test.ts +++ b/packages/paperclip-runner/src/backends/native-instruction-measurement.test.ts @@ -158,16 +158,18 @@ afterAll(() => { const output = process.env.PAPERCLIP_NATIVE_INSTRUCTION_REPORT; if (!output) return; if (receipts.length !== 18) throw new Error("Incomplete native instruction measurement; refusing a partial receipt"); - const sourcePaths = ["runtime-context.ts", "codex-native-backend.ts", "opencode-native-backend.ts"]; - const source = inspectNativeCompletionSourceMetadata({ repositoryRoot: new URL("../../../../", import.meta.url).pathname, - sourceFiles: sourcePaths.map(file => `packages/paperclip-runner/src/backends/${file}`), + const repositoryRoot = new URL("../../../../", import.meta.url); + const sourcePaths = ["runtime-context.ts", "codex-native-backend.ts", "opencode-native-backend.ts"].map(file => `packages/paperclip-runner/src/backends/${file}`); + sourcePaths.push("server/src/services/native-runtime/paperclip-runner-tool-authority.ts"); + const source = inspectNativeCompletionSourceMetadata({ repositoryRoot: repositoryRoot.pathname, + sourceFiles: sourcePaths, baseSha: "2a8a99e4a5f69aa803b3f10b982f583e75a87042", variant: "measurement" }); writeFileSync(output, `${JSON.stringify({ schema: "paperclip.native-instruction-measurement.v1", sourceSha: execFileSync("git", ["rev-parse", "HEAD"], { encoding: "utf8" }).trim(), sourceDirty: !source.immutable, sourceMetadata: source.sourceMetadata, - sourceHashes: Object.fromEntries(sourcePaths.map(file => [file, sha256(readFileSync(new URL(file, import.meta.url)))])), + sourceHashes: Object.fromEntries(sourcePaths.map(file => [file.split("/").at(-1)!, sha256(readFileSync(new URL(file, repositoryRoot)))])), fixtureSha256: sha256(readFileSync(new URL(import.meta.url))), boundary: "scripted runnerd RPC; complete Paperclip instructions, fixture core tool schemas and turn input", providerCalls: 0, tokenCount: null, vendorStockPrompt: "unavailable", modelBehavior: "not_measured", diff --git a/packages/paperclip-runner/src/backends/runtime-context.test.ts b/packages/paperclip-runner/src/backends/runtime-context.test.ts index f62f87b7e3..2fc3ac4726 100644 --- a/packages/paperclip-runner/src/backends/runtime-context.test.ts +++ b/packages/paperclip-runner/src/backends/runtime-context.test.ts @@ -68,6 +68,7 @@ describe("native runtime context files", () => { "Obtain one accepted result from paperclip_finish or paperclip_block before writing", ); expect(constraints).toContain("before writing the complete user-facing final response."); + expect(constraints).toContain("If blocked, explain why work cannot continue, name the owner and give the unblock action."); for (const description of [PRP_COMPLETION_TOOL_DESCRIPTION, PRP_BLOCK_TOOL_DESCRIPTION]) { expect(description).toContain("If rejected, correct the report and retry."); expect(description).toContain("After acceptance, read the returned outcome"); @@ -89,6 +90,7 @@ describe("native runtime context files", () => { expect(constraints).toContain("register_deliverable"); expect(constraints).toContain("deliverable:"); expect(constraints).toContain("download link"); + expect(constraints).toContain("returned documentHref as a clickable link in your final response"); }); it("marks only authoritative answered-question envelopes as resolved in the outer task", () => { diff --git a/packages/paperclip-runner/src/backends/runtime-context.ts b/packages/paperclip-runner/src/backends/runtime-context.ts index 446aab6dbb..a047b29ff3 100644 --- a/packages/paperclip-runner/src/backends/runtime-context.ts +++ b/packages/paperclip-runner/src/backends/runtime-context.ts @@ -34,7 +34,7 @@ export function nativeTaskConstraints(input: NativeExecutionInput): string[] { // Keep the discovery/ordering rule in each turn. The completion tools own // reporting, rejection feedback, approval handling and final-response details. const finalResponseConstraint = - "Obtain one accepted result from paperclip_finish or paperclip_block before writing the complete user-facing final response. Follow that tool's reporting and final-response instructions."; + "Obtain one accepted result from paperclip_finish or paperclip_block before writing the complete user-facing final response. Follow that tool's reporting and final-response instructions. If blocked, explain why work cannot continue, name the owner and give the unblock action."; const answeredQuestions = Array.isArray(input.interactionResponses) ? input.interactionResponses.flatMap((response, responseIndex) => { if ( @@ -112,7 +112,7 @@ export function nativeTaskConstraints(input: NativeExecutionInput): string[] { : `For this turn, the editable agent instruction file is ${input.runtimeContext.instructions.workingCopy.rootPath}/${input.runtimeContext.instructions.workingCopy.entryPath}. This replaces any private working-copy path from a previous turn. Ordinary edits save after the provider stops and only with a durable revision receipt. Use the agent instruction tools for immediate saves. Shared instruction assets and repository instructions are not collected.`, ] : []), "Use Paperclip semantic tools for coordination and finalization.", - "Save requested plans and Paperclip documents directly with write_document. A saved Paperclip document is already a durable deliverable. Do not create a local file, compute file hashes, or call register_deliverable for it unless the user also requests a downloadable file. Cite the saved document in your completion evidence and final response.", + "Save requested plans and Paperclip documents directly with write_document. A saved Paperclip document is already a durable deliverable. Do not create a local file, compute file hashes, or call register_deliverable for it unless the user also requests a downloadable file. Cite the saved document in your completion evidence and include its returned documentHref as a clickable link in your final response.", "When the requested result is a file, use register_deliverable before paperclip_finish. Compute its exact byte size and SHA-256, register the workspace-relative file, cite deliverable: from the receipt as completion evidence, and include /api/attachments//content as the download link in your answer. A bare workspace filename is not a delivered result. For repository edits, cite an accessible PR or registered work product. Preserve existing work; do not upload unrelated files. If file publication fails, fix it or report the concrete blocker instead of claiming the file is delivered.", ...(answeredQuestionConstraint ? [answeredQuestionConstraint] : []), finalResponseConstraint, diff --git a/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts b/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts index 00d042b45d..a0f982637e 100644 --- a/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts +++ b/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts @@ -830,6 +830,7 @@ describe("PaperclipRunnerToolAuthority", () => { disposition: "applied", created: true, document: { key: "plan", body }, + documentHref: "/RNT/issues/RNT-1#document-plan", }); expect( await db diff --git a/server/src/services/native-runtime/paperclip-runner-tool-authority.ts b/server/src/services/native-runtime/paperclip-runner-tool-authority.ts index c3dab94a6a..e67e02b54f 100644 --- a/server/src/services/native-runtime/paperclip-runner-tool-authority.ts +++ b/server/src/services/native-runtime/paperclip-runner-tool-authority.ts @@ -48,6 +48,7 @@ import { agentWakeupRequests, chatEndpoints, chatConversations, + companies, documentRevisions, heartbeatRuns, issueApprovals, @@ -928,10 +929,15 @@ export class PaperclipRunnerToolAuthority { }, }); publication = activity.publication; + const [target] = await tx.select({ identifier: issues.identifier, issuePrefix: companies.issuePrefix }) + .from(issues).innerJoin(companies, eq(companies.id, issues.companyId)) + .where(and(eq(issues.id, this.binding.issueId), eq(issues.companyId, this.binding.companyId))); + if (!target) throw new Error("paperclip_runner_document_task_not_found"); return { disposition: "applied", created: write.created, document: write.document, + documentHref: `/${encodeURIComponent(target.issuePrefix)}/issues/${encodeURIComponent(target.identifier ?? this.binding.issueId)}#document-${encodeURIComponent(write.document.key)}`, }; }); if (publication) publishActivity(publication); diff --git a/tests/runner-e2e/README.md b/tests/runner-e2e/README.md index 46dcfeff96..f18ddc4648 100644 --- a/tests/runner-e2e/README.md +++ b/tests/runner-e2e/README.md @@ -426,6 +426,8 @@ The independent, explicit-only `native-completion` suite qualifies native finish The separate, explicit-only `native-instruction-consolidation` suite reuses those original tasks and strict graders to compare completion constraints on the production defaults at `2a8a99e4a5f69aa803b3f10b982f583e75a87042`. It declares six local cells: document completion and whole-task blocking on native Codex, ACPX Claude, and OpenCode. Each cell allows one attempt and applies a 1,000-cent company and agent budget hard stop. Its source gate rejects dirty, mixed, unknown, or unrelated source changes before credentials load. A provider-free fixture captures the complete Paperclip instruction/tool/message projection at the scripted runnerd RPC boundary on start, full-task resume and compact user-follow-up continuation for native input v4 and v5. That capture measures bytes; it does not measure vendor-owned prompts, tokens, billing, or model behavior. See the [comparison plan](../../doc/plans/2026-10-03-native-completion-consolidation.md) for exact scope and live qualification limits. Existing `native-completion` results do not qualify this new reduction. +The corrected source variant adds explicit blocker explanations and canonical document citations. Observation v3 independently requires the persisted provider final to explain missing release/deployment access and, for completion, link this task's one revisioned document on the same origin. A correct structured blocker, an unblock action alone, or an unbound/foreign document URL cannot pass. These stricter checks supplement the unchanged task prompts and durable-output oracle. Replay of retained v2 evidence is a separate diagnostic, never a replacement for its original verdict. See the [answer correction](../../doc/plans/2026-10-04-native-completion-answer-fix.md). + `context-integrity` is an explicit-only local suite with two bounded cases across ten listed legacy/native profiles (20 cells). Six cells are pending-prerequisite profiles and are listed for discovery but rejected before provider credentials are diff --git a/tests/runner-e2e/native-completion-cases.ts b/tests/runner-e2e/native-completion-cases.ts index 8cea25b7c9..faf396d249 100644 --- a/tests/runner-e2e/native-completion-cases.ts +++ b/tests/runner-e2e/native-completion-cases.ts @@ -38,7 +38,7 @@ export const nativeCompletionTasks: readonly RunnerTaskFixture[] = [ ]; export function nativeCompletionDefinitionDigest() { const hash = createHash("sha256"); - for (const file of ["native-completion-cases.ts", "native-completion-scoring.ts", "native-completion-defaults.ts", + for (const file of ["native-completion-cases.ts", "native-completion-scoring.ts", "native-completion-content.ts", "native-completion-defaults.ts", "native-blocker-visible.ts", "native-completion-admission.ts", "native-completion-checks.mjs", "native-completion-source-contract.mjs", "automatic-retry.ts", "context-integrity-cases.ts", "context-integrity-flow.ts", "context-integrity-scoring.ts", "types.ts", "catalog.ts", "live-fixtures.ts", "runner.spec.ts", "launch.ts"]) { hash.update(file); hash.update(readFileSync(new URL(file, import.meta.url))); diff --git a/tests/runner-e2e/native-completion-content.test.ts b/tests/runner-e2e/native-completion-content.test.ts new file mode 100644 index 0000000000..34eba9f2b2 --- /dev/null +++ b/tests/runner-e2e/native-completion-content.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it } from "vitest"; +import { explainsMissingReleaseAccess, linksSavedNativeDocument, type NativeDocumentLinkContext } from "./native-completion-content.js"; + +describe("native final-answer content calibration", () => { + it.each([ + "Task blocked: required deployment access is missing.", + "Deployment cannot proceed because required release access has not been granted.", + "This task is blocked: required release access has not been granted.", + "Deployment remains blocked until access is granted.", + "The required access is unavailable.", + "Missing required deployment access prevents this release.", + "We lack release access.", + "Release access hasn't been granted.", + ])("accepts an explicit unresolved access reason: %s", text => expect(explainsMissingReleaseAccess(text)).toBe(true)); + it.each([ + "The whole task is blocked. Owner: Release Owner. Unblock action: Grant deployment access.", + "Release Owner must Grant deployment access.", + "Deployment is blocked. Grant deployment access.", + "Task complete. Release access is available.", + "Billing access is missing. Release Owner must Grant deployment access.", + "The task is blocked, but required deployment access is not missing.", + "Required deployment access is already granted, but the task is blocked.", + "BLOCKED_probe", + "", + ])("rejects a label/action or different cause: %s", text => expect(explainsMissingReleaseAccess(text)).toBe(false)); + + const context = (): NativeDocumentLinkContext => ({ appOrigin: "https://paperclip.example", issuePrefix: "RUN", issueIdentifier: "RUN-1", + documents: [{ key: "output", latestRevisionId: "revision", latestRevisionNumber: 1 }] }); + it.each([ + "[Document](/RUN/issues/RUN-1#document-output)", + "[Document](https://paperclip.example/RUN/issues/RUN-1#document-output)", + '[Document]( "Saved output")', + ])("accepts the canonical saved-document link: %s", text => expect(linksSavedNativeDocument(text, context())).toBe(true)); + it.each([ + "Saved output.", + "/RUN/issues/RUN-1#document-output", + "[Document](/RUN/issues/RUN-2#document-output)", + "[Document](/OTHER/issues/RUN-1#document-output)", + "[Document](/RUN/issues/RUN-1#document-other)", + "[Document](https://other.example/RUN/issues/RUN-1#document-output)", + "[Document](https://user@paperclip.example/RUN/issues/RUN-1#document-output)", + "[Document](/RUN/issues/RUN-1?redirect=other#document-output)", + "[Document](file:///RUN/issues/RUN-1#document-output)", + ])("rejects absent or misdirected output citation: %s", text => expect(linksSavedNativeDocument(text, context())).toBe(false)); + it.each(["origin", "no-documents", "extra-document", "revision-id", "revision-number"])("fails closed for missing %s evidence", scenario => { + const value = context(); + if (scenario === "origin") value.appOrigin = ""; + if (scenario === "no-documents") value.documents = []; + if (scenario === "extra-document") value.documents = [...value.documents, { key: "other", latestRevisionId: "other", latestRevisionNumber: 1 }]; + if (scenario === "revision-id") value.documents = [{ key: "output", latestRevisionNumber: 1 }]; + if (scenario === "revision-number") value.documents = [{ key: "output", latestRevisionId: "revision", latestRevisionNumber: 0 }]; + expect(linksSavedNativeDocument("[Document](/RUN/issues/RUN-1#document-output)", value)).toBe(false); + expect(linksSavedNativeDocument("[Document](/RUN/issues/RUN-1#document-output)", undefined)).toBe(false); + }); +}); diff --git a/tests/runner-e2e/native-completion-content.ts b/tests/runner-e2e/native-completion-content.ts new file mode 100644 index 0000000000..db4aa244c4 --- /dev/null +++ b/tests/runner-e2e/native-completion-content.ts @@ -0,0 +1,36 @@ +/** Bounded release-access fixture: an action alone is not a blocker explanation. */ +export function explainsMissingReleaseAccess(text: string): boolean { + // The requested action must not supply the access fact that the explanation lacks. + const explanation = text.replaceAll("Grant deployment access", ""); + const access = String.raw`(?:required\s+access|(?:required\s+)?(?:release|deployment)\s+access)`; + const absent = String.raw`(?:missing|unavailable|absent|not\s+(?:yet\s+)?(?:(?:been|being)\s+)?(?:granted|provided|available)|hasn['’]t\s+been\s+granted)`; + if (new RegExp(String.raw`\b${access}\s+(?:is|was|has\s+been)\s+(?:not\s+(?:missing|unavailable|absent)|(?:already\s+)?(?:available|granted|provided))\b`, "i").test(explanation)) return false; + return new RegExp(String.raw`\b${access}\b[^.!?\n]{0,100}\b${absent}\b`, "i").test(explanation) + || new RegExp(String.raw`\b(?:missing|unavailable|lack(?:ing)?)\b[^.!?\n]{0,60}\b${access}\b`, "i").test(explanation) + || /\b(?:deployment|release)\b[^.!?\n]{0,60}\b(?:blocked|cannot proceed|can't proceed)\b[^.!?\n]{0,60}\b(?:until|without)\b[^.!?\n]{0,60}\baccess\b/i.test(explanation); +} + +export interface NativeDocumentLinkContext { + appOrigin: string; + issuePrefix: string; + issueIdentifier: string; + documents: readonly { key: string; latestRevisionId?: string | null; latestRevisionNumber?: number | null }[]; +} + +/** A citation must resolve to this task's one saved, revisioned document. */ +export function linksSavedNativeDocument(text: string, context: NativeDocumentLinkContext | undefined): boolean { + if (!context?.appOrigin || !context.issuePrefix || !context.issueIdentifier || context.documents.length !== 1) return false; + const document = context.documents[0]!; + if (!document.key || !document.latestRevisionId || !Number.isSafeInteger(document.latestRevisionNumber) || Number(document.latestRevisionNumber) < 1) return false; + let origin: URL; + try { origin = new URL(context.appOrigin); } catch { return false; } + if (!["http:", "https:"].includes(origin.protocol) || origin.username || origin.password || origin.pathname !== "/" || origin.search || origin.hash) return false; + const route = `/${encodeURIComponent(context.issuePrefix)}/issues/${encodeURIComponent(context.issueIdentifier)}#document-${encodeURIComponent(document.key)}`; + const links = text.matchAll(/\[[^\]]+\]\((?:<([^>]+)>|([^\s)]+))(?:\s+["'][^"']*["'])?\)/g); + return [...links].some(link => { + try { + const url = new URL(link[1] ?? link[2]!, origin); + return url.origin === origin.origin && !url.username && !url.password && !url.search && `${url.pathname}${url.hash}` === route; + } catch { return false; } + }); +} diff --git a/tests/runner-e2e/native-completion-scoring.test.ts b/tests/runner-e2e/native-completion-scoring.test.ts index 8c476cbc32..dc874dfd43 100644 --- a/tests/runner-e2e/native-completion-scoring.test.ts +++ b/tests/runner-e2e/native-completion-scoring.test.ts @@ -4,7 +4,7 @@ import { rehydrateRunnerdItemNotification } from "../../packages/paperclip-runne import { gradeNativeCompletion, type NativeCompletionObservation } from "./native-completion-scoring.js"; type Row = Record; function sample(blocked = true, compatibility = false): NativeCompletionObservation { - const body = blocked ? "Deployment remains blocked until access is granted. Release Owner must Grant deployment access. BLOCKED_probe" : "Saved the requested document."; + const body = blocked ? "Deployment remains blocked until access is granted. Release Owner must Grant deployment access. BLOCKED_probe" : "Saved [the requested document](/RUN/issues/RUN-1#document-output)."; const result = { reportedWorkDisposition: blocked ? "blocked" : "done", ...(blocked ? { blocker: { owner: { name: "Release Owner" }, unblockAction: "Grant deployment access", scope: "task_wide" } } : {}) }; const event = (seq: number, eventType: string, payload: Row, sourceKind = "runner"): Row => ({ seq, eventType, protocolSchemaVersion: 1, sourceEventId: `event${seq}`, sourceInstanceId: sourceKind, sourceSeq: seq, @@ -15,6 +15,8 @@ function sample(blocked = true, compatibility = false): NativeCompletionObservat issue: { id: "issue", companyId: "company", assigneeAgentId: "agent", status: result.reportedWorkDisposition }, runs: [{ id: "run", nativeIssueId: "issue", companyId: "company", agentId: "agent", status: "succeeded", runtimeMode: "native", resultJson: { nativeResult: result } }], comments: [{ createdByRunId: "run", authorAgentId: "agent", body }], + documentLinkContext: { appOrigin: "https://paperclip.example", issuePrefix: "RUN", issueIdentifier: "RUN-1", + documents: blocked ? [] : [{ key: "output", latestRevisionId: "revision", latestRevisionNumber: 1 }] }, initial: { issueIds: [], agentIds: ["agent"] }, state: { issueIds: ["issue"], agentIds: ["agent"], documentCount: blocked ? 0 : 1, interactionCount: 0 }, workspaceChanged: false, events: [compatibility ? event(1, "item.started", { kind: "tool_call", item: { type: "tool_call", id: "tool", name: tool } }) : event(1, "tool.execution.started", { name: tool, executionId: "tool" }), event(2, "run.result.proposed", result), compatibility ? event(3, "item.completed", { kind: "tool_result", item: { type: "tool_result", status: "completed", id: "tool" } }) : event(3, "tool.execution.completed", { name: tool, status: "completed", executionId: "tool" }), @@ -24,6 +26,27 @@ function sample(blocked = true, compatibility = false): NativeCompletionObservat } function payload(input: NativeCompletionObservation, index: number): Row { return ((input.events[index]!.payload as Row).prpEvent as Row).payload as Row; } describe("native completion independent oracle", () => { + it("rejects a correct structured blocker whose visible reply only labels it blocked and repeats the action", () => { + const value = sample(); + const text = "The whole task is blocked. Owner: Release Owner.\n\nUnblock action: Grant deployment access\n\nBLOCKED_probe"; + (payload(value, 3).item as Row).text = text; value.comments[0]!.body = text; + const grade = gradeNativeCompletion(value); + expect(grade.checks.find(check => check.id === "visible-blocker-content")?.passed).toBe(true); + expect(grade.checks.find(check => check.id === "exact-blocker")?.passed).toBe(true); + expect(grade.checks.find(check => check.id === "visible-blocker-reason")?.passed).toBe(false); + expect(grade.passed).toBe(false); + }); + it.each(["missing-link", "wrong-document", "wrong-reply-run", "missing-link-context", "missing-revision"])("rejects a saved document with %s in the final", scenario => { + const value = sample(false); + if (scenario === "missing-link" || scenario === "wrong-document") { + const text = scenario === "missing-link" ? "Saved the requested document." : "Saved [document](/RUN/issues/RUN-1#document-other)."; + (payload(value, 3).item as Row).text = text; value.comments[0]!.body = text; + } + if (scenario === "wrong-reply-run") value.comments[0]!.createdByRunId = "other"; + if (scenario === "missing-link-context") delete value.documentLinkContext; + if (scenario === "missing-revision") value.documentLinkContext!.documents = [{ key: "output", latestRevisionNumber: 1 }]; + expect(gradeNativeCompletion(value).passed).toBe(false); + }); it.each([[true, false], [true, true], [false, false], [false, true]])("accepts disposition %s compatibility %s with final before late control-plane acceptance", (blocked, compatibility) => expect(gradeNativeCompletion(sample(blocked, compatibility)).passed).toBe(true)); it.each(["paperclip_finish", "paperclip_block"])("carries normalized %s identity through rehydration into the exact-call oracle", name => { // The Rust normalization calibration asserts these exact fixture bytes. diff --git a/tests/runner-e2e/native-completion-scoring.ts b/tests/runner-e2e/native-completion-scoring.ts index 86740c4580..20e4f8c6f4 100644 --- a/tests/runner-e2e/native-completion-scoring.ts +++ b/tests/runner-e2e/native-completion-scoring.ts @@ -1,4 +1,5 @@ import { isValidNativePrpEnvelope } from "./native-event-envelope.js"; +import { explainsMissingReleaseAccess, linksSavedNativeDocument, type NativeDocumentLinkContext } from "./native-completion-content.js"; type Row = Record; const row = (value: unknown): Row => value && typeof value === "object" && !Array.isArray(value) ? value as Row : {}; @@ -14,6 +15,7 @@ export interface NativeCompletionObservation { initial: { issueIds: string[]; agentIds: string[] }; workspaceChanged: boolean; marker: string; + documentLinkContext?: NativeDocumentLinkContext; } /** Observable tool/result/final sequence, not a claim that the provider consumed feedback. */ @@ -115,6 +117,10 @@ export function gradeNativeCompletion(input: NativeCompletionObservation) { && /\b(?:blocked|cannot proceed|can't proceed|missing|required access|not (?:yet )?granted|awaiting|waiting|unavailable)\b/i.test(finalText) && !/\b(?:not blocked|no longer blocked|access (?:is |has been |was )?already granted|completed Grant deployment access)\b/i.test(finalText), "The final explains the actual unresolved blocker and its owner/action, rather than supplying only a marker."); + if (blocked) check("visible-blocker-reason", explainsMissingReleaseAccess(finalText), + "The persisted provider final independently explains the missing release/deployment access; a blocked label or unblock action alone is insufficient."); + else check("saved-document-final-link", linksSavedNativeDocument(finalText, input.documentLinkContext), + "The persisted provider final links this task's one saved, revisioned document at the canonical same-origin anchor."); const same = (a: string[], b: string[]) => a.length === b.length && a.every(id => b.includes(id)) && new Set(a).size === a.length; check("bounded-durable-work", same(input.state.issueIds, [...input.initial.issueIds, String(input.issue.id)]) && same(input.state.agentIds, input.initial.agentIds) && input.state.documentCount === (blocked ? 0 : 1) @@ -130,6 +136,6 @@ export function gradeNativeCompletion(input: NativeCompletionObservation) { check("no-deployment-or-file-work", !input.workspaceChanged && !forbidden, "The fixture workspace is unchanged and no process/file/deployment or extra-work tool is observed."); } - return { schema: "paperclip.native-completion-observation.v2", passed: checks.every(value => value.passed), checks, + return { schema: "paperclip.native-completion-observation.v3", passed: checks.every(value => value.passed), checks, limitations: ["Exact provider feedback identity/consumption is not measured by the public sequence."] }; } diff --git a/tests/runner-e2e/native-instruction-consolidation.test.ts b/tests/runner-e2e/native-instruction-consolidation.test.ts index c44de76f03..8a2cd4db97 100644 --- a/tests/runner-e2e/native-instruction-consolidation.test.ts +++ b/tests/runner-e2e/native-instruction-consolidation.test.ts @@ -37,7 +37,7 @@ describe("native instruction comparison admission", () => { expect(nativeInstructionVariant(file => baseline.get(file)!)).toBe("baseline"); const candidate = new Map(files.map(file => [file, readFileSync(new URL(`../../${file}`, import.meta.url))])); const currentVariant = nativeInstructionVariant(file => candidate.get(file)!); - expect(["baseline", "candidate"]).toContain(currentVariant); + expect(["baseline", "candidate", "corrected"]).toContain(currentVariant); candidate.set(files[0]!, currentVariant === "candidate" ? baseline.get(files[0]!)! : Buffer.from("unknown source")); expect(() => nativeInstructionVariant(file => candidate.get(file)!)).toThrow("Mixed or unknown"); baseline.set(files[0]!, Buffer.from("unknown source")); diff --git a/tests/runner-e2e/native-instruction-consolidation.ts b/tests/runner-e2e/native-instruction-consolidation.ts index 6eb719914e..272b052e9d 100644 --- a/tests/runner-e2e/native-instruction-consolidation.ts +++ b/tests/runner-e2e/native-instruction-consolidation.ts @@ -18,11 +18,19 @@ export const NATIVE_INSTRUCTION_VARIANTS = { "packages/paperclip-runner/src/backends/runtime-context.ts": "e7c46e81cc9c93f9a4f805b5091ef08c59ea70c62f208cfda70d6edeed363898", "packages/paperclip-runner/src/backends/codex-native-backend.ts": "f276d436391c9e2f0a7e58ce1302b5c9f1a39ac01c148be257f9d455515a31b9", "packages/paperclip-runner/src/backends/opencode-native-backend.ts": "d54ddde3a3fdffcda3490d813d27b3b4891448e11cd05f61f9bb4b292d9fd6c2", + "server/src/services/native-runtime/paperclip-runner-tool-authority.ts": "2da6963c690a3988bf2617d39e85025600b12a056a61c7b528c3d94ac636f777", }, candidate: { "packages/paperclip-runner/src/backends/runtime-context.ts": "cae9075fac25f168972c2be4ddb58052ef2940554458e2757d88a5d0e39805c2", "packages/paperclip-runner/src/backends/codex-native-backend.ts": "affecc515a623e0dfeb338f553ea53ebb7d18baa3d13e4395d17b21000dbee93", "packages/paperclip-runner/src/backends/opencode-native-backend.ts": "d5dc4cc2c06b37d22be47c9f15f828c06b439d8241a8d34498ca4339273eed96", + "server/src/services/native-runtime/paperclip-runner-tool-authority.ts": "2da6963c690a3988bf2617d39e85025600b12a056a61c7b528c3d94ac636f777", + }, + corrected: { + "packages/paperclip-runner/src/backends/runtime-context.ts": "bbdab79c5b1bd57c4ddbc44edfe745b40eb694aa28a465452964109aedd6104b", + "packages/paperclip-runner/src/backends/codex-native-backend.ts": "affecc515a623e0dfeb338f553ea53ebb7d18baa3d13e4395d17b21000dbee93", + "packages/paperclip-runner/src/backends/opencode-native-backend.ts": "d5dc4cc2c06b37d22be47c9f15f828c06b439d8241a8d34498ca4339273eed96", + "server/src/services/native-runtime/paperclip-runner-tool-authority.ts": "d2360cdaa63902cbf0c6a6109da452ba7e0b2a5aaa76ee5cab9c63b4bf12e584", }, } as const; @@ -36,10 +44,15 @@ const comparisonFiles = new Set([ "tests/runner-e2e/native-completion-git-source.d.mts", "tests/runner-e2e/native-completion-defaults.ts", "tests/runner-e2e/native-completion-defaults.test.ts", + "tests/runner-e2e/native-completion-cases.ts", "tests/runner-e2e/native-completion-scoring.ts", + "tests/runner-e2e/native-completion-scoring.test.ts", "tests/runner-e2e/native-completion-content.ts", + "tests/runner-e2e/native-completion-content.test.ts", + "server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts", "tests/runner-e2e/catalog.ts", "tests/runner-e2e/catalog.test.ts", "tests/runner-e2e/launch.ts", "tests/runner-e2e/runner.spec.ts", "tests/runner-e2e/live-fixtures.ts", "tests/runner-e2e/README.md", "doc/evals.md", "doc/plans/2026-10-03-native-completion-consolidation.md", + "doc/plans/2026-10-04-native-completion-answer-fix.md", ]); const git = (...args: string[]) => execFileSync("git", ["--no-replace-objects", ...args], { cwd: root, encoding: "utf8", timeout: 60_000, diff --git a/tests/runner-e2e/runner.spec.ts b/tests/runner-e2e/runner.spec.ts index 44e848c84a..2c047a047a 100644 --- a/tests/runner-e2e/runner.spec.ts +++ b/tests/runner-e2e/runner.spec.ts @@ -2699,12 +2699,34 @@ for (const execution of executions) { companyId: fixtures.company.id, agentId: fixtures.agent.id, issue: currentIssue as unknown as Record, runs: detailedRuns as unknown as Record[], comments: comments as unknown as Record[], events: events as unknown as Record[], initial: nativeInitial, state: { issueIds: issues.map(value => value.id), agentIds: agents.map(value => value.id), documentCount: documents.length, interactionCount: interactions.length }, - workspaceChanged: workspaceDigest !== nativeInitial.workspaceDigest, marker }; + workspaceChanged: workspaceDigest !== nativeInitial.workspaceDigest, marker, + documentLinkContext: { appOrigin: new URL(page.url()).origin, issuePrefix: fixtures.company.issuePrefix ?? "", + issueIdentifier: currentIssue.identifier ?? "", documents } }; const grade = gradeNativeCompletion(observation); const integrity = detailedRuns.flatMap(candidate => nativeRunEventIntegrityFailures(candidate, events)); await writeSanitizedJson(snapshotsDir, "native-completion.json", { observation, grade, integrity }, secrets); matcherResults.push(...grade.checks.map(check => ({ matcher: { kind: "json_path" as const, path: `nativeCompletion.${check.id}`, expected: true }, passed: check.passed, detail: check.detail }))); - if (!grade.passed || integrity.length) throw new Error(`Native completion qualification failed: ${[...grade.checks.filter(check => !check.passed).map(check => check.id), ...integrity].join("; ")}`); + if (!grade.passed || integrity.length) throw new Error(`Native completion matcher failure: ${[...grade.checks.filter(check => !check.passed).map(check => check.id), ...integrity].join("; ")}`); + if (execution.task.id === "assigned-skill-explicit-invocation") { + const document = documents[0]!; + const href = `/${encodeURIComponent(fixtures.company.issuePrefix!)}/issues/${encodeURIComponent(currentIssue.identifier!)}#document-${encodeURIComponent(document.key)}`; + let opened = false; + try { + const link = page.getByTestId("task-chat-agent-bubble").locator(`a[href=${JSON.stringify(href)}]`).last(); + await expect(link).toBeVisible({ timeout: 30_000 }); + await link.click(); + await expect(page).toHaveURL(new URL(href, observation.documentLinkContext.appOrigin).href); + const target = page.locator(`[id=${JSON.stringify(`document-${document.key}`)}]`); + await expect(target).toBeVisible(); + await expect(target).toContainText(marker); + opened = true; + await captureScreenshot("document-final-link", "Final reply link opens the saved document", "document-final-link.png"); + } finally { + matcherResults.push({ matcher: { kind: "json_path", path: "nativeCompletion.visible-document-navigation", expected: true }, passed: opened, + detail: "The rendered final reply link opens this task's saved document and shows its original content marker." }); + await writeSanitizedJson(snapshotsDir, "native-document-navigation.json", { href, documentKey: document.key, revisionId: document.latestRevisionId, opened }, secrets); + } + } } if (runsCompletionUpdateProbe(execution) && credentials.OPENAI_API_KEY) { const qualification = completionQualityStatus(completionQuality);