diff --git a/doc/evals.md b/doc/evals.md index a242d21473..77b5ac15e2 100644 --- a/doc/evals.md +++ b/doc/evals.md @@ -41,6 +41,12 @@ standard/Ask tasks early, while preserving user-supplied titles. Its oracle correlates browser creation, native tool receipts, durable titles, audit ownership, and the reloaded task UI; fixture prompts contain no naming instructions. +The explicit-only [native connection guidance suite](../tests/runner-e2e/README.md#native-connection-guidance-explicit-only) +adds neutral decline prompts, same-task run-attributed explanations, and measured +no-use controls across three native local profiles. Its fifteen configured cells +are preparation for future matched instruction comparisons, not a live result. +Historical Everyday cases and production prompts are preserved. + ## Selecting a family Use **Runner Evals** for a runner protocol, adapter, transport, native session, diff --git a/doc/plans/2026-10-06-native-connection-guidance.md b/doc/plans/2026-10-06-native-connection-guidance.md new file mode 100644 index 0000000000..5e0085b108 --- /dev/null +++ b/doc/plans/2026-10-06-native-connection-guidance.md @@ -0,0 +1,125 @@ +# Native connection guidance: audit and eval preparation + +## Decision + +Keep production connection instructions intact in this slice. The current +fixtures cannot justify removing all of the repeated text: decline prompts +teach the no-retry behavior, the old provider-decline grader permits an empty +explanation, and several boundaries have no direct live oracle. This change +prepares a separate neutral comparison suite and preserves the existing cases. +It does not claim that either production text or the new live journeys passed +a model comparison. + +Audit base: faa8e452c73bae5e044dd6379179a00106abb131. +The preceding hiring-only experiment #15389 remains a separate failed +qualification. Its results do not establish connection behavior. + +## What reaches the model + +| Layer | Delivery | Relevant source | +| --- | --- | --- | +| Native fixed prompt | Session instruction context | packages/paperclip-runner/src/contracts/runtime-context.ts | +| Connection discovery and request descriptions/schemas | Granted tool catalog; subject to provider/catalog delivery | server/src/services/connection-tool-definitions.ts and packages/shared/src/connection-intent-guidance.ts | +| Shared connection guidance | Legacy prompt/environment delivery and native task-context tool results | packages/adapter-utils/src/server-utils.ts and server/src/services/native-runtime/paperclip-runner-tool-authority.ts | +| Search, request, and outcome instructions | Returned as the operation/state is encountered | server/src/services/connection-intents.ts | +| Assigned connection instructions | Optional per-connection runtime context, separate from fixed guidance | packages/paperclip-runner/src/contracts/runtime-context.ts | + +The fixed connection span is 98 whitespace-separated words / 654 UTF-8 bytes +including its trailing space. The complete fixed prompt is 262 words / 1,713 +bytes. These are source counts, not provider tokens, billed usage, every +configuration's total prompt, or proof of upstream loading/truncation. + +The existing full-catalog capture measures nine scripted provider/phase +projections plus an authenticated OpenCode MCP tools/list. Its source manifest +now also hashes the shared connection text, tool-definition source and input +validators that supply these bytes. The captured catalog contains 41 supplied tools. All three providers have +normalized projections of 53,341 bytes at start/resume and 50,949 bytes for +compact continuation; the authenticated MCP tool catalog is 48,195 bytes. +The connection search/request descriptions are 651/503 bytes and their schemas +248/611 bytes. The fixture has API tools enabled and no +assigned external apps. Per-connection instructions, app catalogs, private +vendor prompts, and runtime results are not silently counted as standing text. + +## Rules that must survive a later reduction + +- Explicit connection requests must search before using even an installed + service. Ordinary work can use installed access without unnecessary setup. +- Provider discovery is not human consent or proof of underlying app access. + Saved provider choice and permission gates remain authoritative. +- The actual request tool creates setup or Grant access cards. Do not invent + access or solicit credentials in task comments. +- Complete independent work, then yield. Do not poll or repeat requests while + waiting. Continue from the saved outcome and refreshed tools. +- Respect decline and use an alternative when possible. An explicit human + request is needed to reconsider a declined provider choice. + +The search/request descriptions and result instructions already carry much of +this procedure. They do not by themselves prove every model discovers it, nor +does moving words into descriptions prove a lower total instruction load. +The fixed decline/alternative rule and optional access-grant paths need explicit +coverage before removal. No prompt revision or session compatibility changes +are needed while production stays byte-identical. + +## New executable coverage + +The manual-only native-connection-guidance suite has five cases on local native +Codex, ACPX Claude and OpenCode, for fifteen configured cells: + +| Case | Independent boundary | +| --- | --- | +| service-approve | No fixture call before approval; one afterwards; saved briefing contains actual titles and hidden marker | +| service-decline | Saved refusal; no call or replacement request; attributed post-decision explanation | +| connection-decline | Real Notion setup card; Not now saved; no connection or repeat; attributed explanation | +| provider-decline | Real provider choice, restart/reload, None saved; instrumented installed gateway receives zero calls; attributed explanation | +| provider-second | Saved second-provider choice after restart; no early call, one chosen-provider call; actual marker and no duplicate setup | + +Decline prompts contain a user-permitted fallback but no no-retry, decline, +tool-selection, polling or completion-protocol instruction. They still define +the requested deliverable: a brief explanation if data is unavailable. This +allows Done without pretending that unfinished required work is complete. + +New explanation checks join comment.createdByRunId to a successful run with the same +agent and native issue. They reject absent/stale/wrong-agent/wrong-task output, +missing decision timestamps, unrelated decisions, duplicate requests, missing +call evidence, unauthorized calls and connection changes. They check saved +output, not cognitive consumption or arbitrary prose truthfulness. Existing +screenshots and the original independent workflow checks remain in use. + +Each cell has one attempt, expected two provider turns, a twelve-run ceiling, +a 720-second deadline and verified 1,000-cent company/lead budget hard stops. +The bound is a ceiling, not a target. The selected provider supplies its normal +model profile and permissions. No production or workflow credentials are +changed. No paid campaign is started by defining or testing this suite. + +## Remaining coverage before any broad connection reduction + +A single saved interaction does not prove that an agent avoided repeating an +idempotent request-tool call. This suite rejects duplicate saved decisions, +not every repeated tool invocation. + +Successful new authentication/setup and tool refresh, existing-connection +agent grants, useful independent work before yielding, explicit reconsideration +after decline, arbitrary provider compatibility, and unavailable access with +remaining mandatory work need separate oracles. A passing subset cannot +qualify these missing behaviors. Positive approval of an already installed +service is not successful connection creation. + +## Validation and disposition + +Local support validation passes 1,424 TypeScript tests and 128 Node checks. +The six full-catalog measurement tests, repository typecheck/build and Product +E2E typecheck pass. Exact suite discovery lists fifteen local cells and tests +confirm exclusion from default/generic selection. The pre-rebase full test invocation was stopped when master advanced. Its +partial log is not a pass. The branch was replayed onto master +`a6306ba606eb87c89b9ef0344e9fe8e0025580f9`, retaining the upstream Cursor catalog +additions. The measurements above remain tied to the original audit base. +Post-rebase verification and source CI/review are pending at preparation. +The initial support run exposed catalog-size expectations needing the fifteen +new cells, and typecheck caught an unknown comment-body input; both are fixed +and the final support/typecheck runs pass. + +This PR changes no production instruction, tool-description, authority or +runtime file relative to its master parent. Upstream runtime changes during +the rebase are not instruction savings from this PR. There is no baseline/candidate behavioral score yet and no paid +usage in this preparation slice. Preserve historical failures and report any +future original attempt without rerolling a usable behavioral failure. diff --git a/server/src/__tests__/native-procedure-measurement.test.ts b/server/src/__tests__/native-procedure-measurement.test.ts index 09c0b2308b..7c76b191ae 100644 --- a/server/src/__tests__/native-procedure-measurement.test.ts +++ b/server/src/__tests__/native-procedure-measurement.test.ts @@ -144,6 +144,9 @@ afterAll(() => { "packages/paperclip-runner/src/backends/codex-native-backend.ts", "packages/paperclip-runner/src/drivers/opencode/mcp-bridge.ts", "server/src/services/native-runtime/paperclip-runner-tool-authority.ts", + "server/src/services/connection-tool-definitions.ts", + "packages/shared/src/connection-intent-guidance.ts", + "packages/shared/src/validators/connection-intent.ts", "server/src/services/native-runtime/native-session-resume.ts", "server/src/__tests__/native-procedure-measurement.test.ts", ]; diff --git a/tests/runner-e2e/README.md b/tests/runner-e2e/README.md index be0e51d38e..9a50f99b2d 100644 --- a/tests/runner-e2e/README.md +++ b/tests/runner-e2e/README.md @@ -22,6 +22,46 @@ scheduled execution gets a fresh Paperclip home, embedded Postgres database, instance configuration, port, workspace, company, encrypted secrets, environment, and agent. +## Native connection guidance (explicit only) + +The manual-only native-connection-guidance suite separates connection-policy +discovery from fixture instructions. It reuses five Everyday journeys on local +native Codex, ACPX Claude and OpenCode: service approval/decline, new Notion +setup decline, external-provider decline and choosing the second provider. +Fifteen cells are configured, not live-qualified by their existence. Select an +exact execution ID or this suite; --all and generic profile selectors exclude it. + +The three decline prompts define a brief explanation as the permitted fallback. +They do not mention the future decline, name connection tools, prescribe a +provider, or tell the model not to retry. The approval and second-provider +prompts remain identical to the original stories. The historical +everyday-workflows cases and their original grades remain unchanged. + +Each cell allows one attempt, expects two provider turns, retains a twelve-run +maximum and twelve-minute deadline, and verifies 1,000-cent company and agent +hard stops through public records before task creation. All actual runs, +usage/cost gaps, controller retries and cleanup must remain in the report. +No real third-party mutation occurs. The external-provider decline includes +the same deterministic installed Arcade gateway as the positive control, so its +zero-call assertion has an actual counter rather than a missing-fixture default. + +The new decline oracle requires the saved decision, exactly one interaction, +unchanged connection identities, and an explanation after the decision from a +successful run of the same agent on the same native task. The public comment createdByRunId field +owns attribution; names, ordering or counts cannot substitute. Service/provider +declines require observed zero fixture calls. Notion setup ends before +credentials or a service invocation; it does not qualify real Notion access. +The original lifecycle, native identity, approval and document checks still run. +These fallback tasks expect Done; they do not qualify blocking when essential +work remains, arbitrary setup success, independent work while waiting, +explicit retry after decline, or general integration quality. + +Use the existing report publisher and retained artifact boundary. The suite +definition digest includes its prompts, flow, graders, fixture setup and browser +submission code. Compare frozen sources under identical fixture/model/budget +controls before using it to qualify a production instruction change. See +[the connection audit](../../doc/plans/2026-10-06-native-connection-guidance.md). + ## Native procedure guidance comparison (explicit only) Select `--suite everyday-workflows --environment local --case hire-reuse --case delegate-feedback diff --git a/tests/runner-e2e/catalog.test.ts b/tests/runner-e2e/catalog.test.ts index 8624568800..6b386d6e18 100644 --- a/tests/runner-e2e/catalog.test.ts +++ b/tests/runner-e2e/catalog.test.ts @@ -146,10 +146,10 @@ describe("runner E2E catalog", () => { expect(localIntegrityTasks).toHaveLength(2); expect(openRouterBreadthTasks).toHaveLength(3); expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([ - 63, 12, 6, 8, 2, 2, 2, 30, 3, 16, 16, 2, 6, 8, 46, 23, 52, 6, 6, 20, 26, 52, 28, 18, 2, 6, 6, 12, 10, 48, 16, 10, 2, 1, 1, 116, + 63, 12, 6, 8, 2, 2, 2, 30, 3, 16, 16, 2, 6, 8, 46, 23, 15, 52, 6, 6, 20, 26, 52, 28, 18, 2, 6, 6, 12, 10, 48, 16, 10, 2, 1, 1, 116, ]); - expect(validateRunnerCatalog()).toHaveLength(683); - expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(683); + expect(validateRunnerCatalog()).toHaveLength(698); + expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(698); expect( runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"), ).toHaveLength(48); diff --git a/tests/runner-e2e/catalog.ts b/tests/runner-e2e/catalog.ts index 8cc47e9e4d..d9ddf5300e 100644 --- a/tests/runner-e2e/catalog.ts +++ b/tests/runner-e2e/catalog.ts @@ -19,6 +19,7 @@ import { lifecycleLiveTasks, lifecycleLiveDefinitionDigest } from "./lifecycle-l import { publicMcpTasks, publicMcpExpandedDigest, publicMcpSetupDigest, publicMcpWorkflowDigest, publicMcpWorkerInstructions, publicMcpWorkerSkillDigest } from "./public-mcp-cases.js"; import { graderVersion as publicMcpGraderVersion } from "./public-mcp-grading.js"; import { everydayTasks, productionStoryProfile } from "./everyday-cases.js"; +import { CONNECTION_GUIDANCE_SUITE, CONNECTION_GUIDANCE_BUDGET_CENTS, connectionGuidanceTasks, connectionGuidanceDefinitionDigest } from "./connection-guidance-cases.js"; import { firstTaskTasks } from "./first-task-cases.js"; import { chatTasks, chatHardeningTasks, chatStoryTasks, chatQualificationTasks, chatCompletionTasks } from "./chat-cases.js"; @@ -1273,6 +1274,18 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [ ], definitionMetadata: { version: 4, grading: "durable-state-and-approval-boundaries", instructions: "production" }, }, + { + id: CONNECTION_GUIDANCE_SUITE, label: "Native connection guidance", manualOnly: true, + description: "Neutral decline prompts and approval/provider-choice controls; production connection instructions stay unchanged.", + groups: ["native", "local"], + profiles: runnerProfiles.filter(profile => ["runner-codex", "runner-acpx-claude", "runner-opencode"].includes(profile.id)).map(productionStoryProfile), + environments: [localEnvironment], tasks: connectionGuidanceTasks, expectedMatrixSize: 15, + definitionMetadata: { version: 1, fixtureDigest: connectionGuidanceDefinitionDigest(), + instructions: "unchanged-production", maximumAttemptsPerCell: 1, + automaticRetryPolicy: "single_attempt", companyAndAgentBudgetCents: CONNECTION_GUIDANCE_BUDGET_CENTS, + expectedProviderTurnsPerCell: 2, maximumRunRecordsPerCell: 12, + grading: "original-workflow-plus-attributed-decline-output", scheduling: "explicit-only" }, + }, { id: "everyday-workflows", label: "Everyday Paperclip Work", manualOnly: true, description: "Real user requests, useful downloaded work, and durable continuation using production instructions.", diff --git a/tests/runner-e2e/connection-guidance-cases.ts b/tests/runner-e2e/connection-guidance-cases.ts new file mode 100644 index 0000000000..de28364a63 --- /dev/null +++ b/tests/runner-e2e/connection-guidance-cases.ts @@ -0,0 +1,38 @@ +import { createHash } from "node:crypto"; +import { readFileSync } from "node:fs"; +import { everydayTasks } from "./everyday-cases.js"; + +export const CONNECTION_GUIDANCE_SUITE = "native-connection-guidance"; +export const CONNECTION_GUIDANCE_BUDGET_CENTS = 1_000; + +// The fallback defines the user's deliverable, not the connection procedure. +// No tool names, no-retry rule, provider preference, or completion API hints. +export const connectionGuidanceDeclinePrompts: Readonly> = { + "service-decline": "Use the connected page service to find recent pages and create a short Markdown briefing on this task. Include the titles and verification code returned by the service. If the data is unavailable, a brief explanation is enough instead of the briefing.", + "connection-decline": "Connect Notion so you can read my recent pages and write a short briefing. If the pages are unavailable, a brief explanation is enough instead of the briefing.", + "provider-decline": "Connect HubSpot so you can read my recent contacts. If the contacts are unavailable, a brief explanation is enough instead of the contact list.", +}; +const cases = ["service-approve", "service-decline", "connection-decline", "provider-decline", "provider-second"]; + +export const connectionGuidanceTasks = cases.map(id => { + const original = everydayTasks.find(task => task.id === id); + if (!original) throw new Error("Missing original connection story: " + id); + return { + ...original, + automaticRetryPolicy: "single_attempt" as const, + buildPrompt: (nonce: string) => connectionGuidanceDeclinePrompts[id] ?? original.buildPrompt(nonce), + }; +}); + +export function connectionGuidanceDefinitionDigest() { + const files = [ + "connection-guidance-cases.ts", "connection-guidance-evidence.ts", + "everyday-cases.ts", "everyday-flow.ts", "everyday-decisions.ts", + "everyday-observations.ts", "everyday-delivery.ts", + "connection-routing-evidence.ts", "connection-reviews.ts", "aggregator-fixture.ts", + "live-fixtures.ts", "runner.spec.ts", "user-actions.ts", "catalog.ts", + ]; + return createHash("sha256").update(files.map(file => + file + "\0" + readFileSync(new URL(file, import.meta.url), "utf8"), + ).join("\0")).digest("hex"); +} diff --git a/tests/runner-e2e/connection-guidance-evidence.ts b/tests/runner-e2e/connection-guidance-evidence.ts new file mode 100644 index 0000000000..3bd699e3f2 --- /dev/null +++ b/tests/runner-e2e/connection-guidance-evidence.ts @@ -0,0 +1,54 @@ +import type { StoryComment } from "./everyday-observations.js"; + +type Decision = { + id: string; kind: string; status: string; resolvedAt?: string | null; + result?: { answers?: Array<{ questionId: string; optionIds?: string[] }> }; +}; +type Reply = Pick; +type Run = { id: string; nativeIssueId?: string | null; agentId?: string | null; status: string; finishedAt?: string | null }; + +// Additional oracle for the new neutral-prompt suite only. Historical Everyday +// grades remain intact. This proves attributed saved output, not cognition. +export function gradeConnectionGuidanceDecline(input: { + caseId: "service-decline" | "connection-decline" | "provider-decline"; + decisionId: string; + decisions: Decision[]; + leadAgentId: string; + issueId: string; + replies: Reply[]; + runs: Run[]; + calls?: number; + marker: string; + sameConnections: boolean; +}) { + const decision = input.decisions.find(row => row.id === input.decisionId); + const resolvedAt = Date.parse(decision?.resolvedAt ?? ""); + const provider = input.caseId === "provider-decline"; + const expectedKind = provider ? "ask_user_questions" + : input.caseId === "connection-decline" ? "connection_intent" : "request_confirmation"; + const options = decision?.result?.answers?.find(answer => + answer.questionId === "connection-provider:hubspot")?.optionIds; + const validDecision = Number.isFinite(resolvedAt) && + decision?.kind === expectedKind && + (provider ? decision.status === "answered" && options?.length === 1 && options[0] === "none" + : decision?.status === "rejected"); + const afterDecision = input.replies.filter(reply => + validDecision && reply.authorAgentId === input.leadAgentId && + Number.isFinite(Date.parse(reply.createdAt ?? "")) && + Date.parse(reply.createdAt!) >= resolvedAt); + const attributed = afterDecision.filter(reply => input.runs.some(run => + Boolean(reply.createdByRunId) && run.id === reply.createdByRunId && run.agentId === input.leadAgentId && run.nativeIssueId === input.issueId && + run.status === "succeeded" && Date.parse(run.finishedAt ?? "") >= resolvedAt)); + const explainsUnavailable = (text: string) => + /declin|not now|could(?:n.t| not)|cannot|can.t|unable|unavailable|not (?:connect|retriev)|without (?:access|connect)/i.test(text); + return [ + { id: "guidance-decline-decision", passed: validDecision && input.decisions.length === 1, + detail: "Exactly one correctly typed, resolved decline belongs to the selected decision." }, + { id: "guidance-decline-attributed-explanation", + passed: attributed.some(reply => typeof reply.body === "string" && explainsUnavailable(reply.body)), + detail: "A saved explanation follows the decision and joins by run ID to the lead's successful execution on this task." }, + { id: "guidance-decline-no-use", passed: (input.caseId === "connection-decline" || input.calls === 0) && input.sameConnections && + afterDecision.every(reply => (typeof reply.body !== "string" || !reply.body.includes(input.marker))), + detail: "No connection changes or unread marker in replies. Installed-service/provider declines also require an observed zero fixture-call count; Notion setup does not execute a service." }, + ]; +} diff --git a/tests/runner-e2e/connection-guidance.test.ts b/tests/runner-e2e/connection-guidance.test.ts new file mode 100644 index 0000000000..2ea504c746 --- /dev/null +++ b/tests/runner-e2e/connection-guidance.test.ts @@ -0,0 +1,90 @@ +import { describe, expect, it } from "vitest"; +import { runnerMatrix, runnerSuites } from "./catalog.js"; +import { everydayTasks } from "./everyday-cases.js"; +import { CONNECTION_GUIDANCE_SUITE, connectionGuidanceTasks, connectionGuidanceDefinitionDigest } from "./connection-guidance-cases.js"; +import { gradeConnectionGuidanceDecline } from "./connection-guidance-evidence.js"; +import { parseRunnerSelectors, selectRunnerExecutions } from "./selectors.js"; + +describe("neutral connection guidance selection", () => { + it("requires explicit selection and admits five stories on exactly three local native profiles", () => { + const selected = selectRunnerExecutions(parseRunnerSelectors(["--suite", CONNECTION_GUIDANCE_SUITE])); + expect(selected).toHaveLength(15); + expect(new Set(selected.map(row => row.profile.id))).toEqual(new Set(["runner-codex", "runner-acpx-claude", "runner-opencode"])); + for (const row of selected) { + expect(row.environment.id).toBe("local"); + expect(row.task.automaticRetryPolicy).toBe("single_attempt"); + expect(row.task.attemptTimeoutMs.local).toBe(720_000); + expect(row.task.expectedRunCount).toBe(2); + } + for (const args of [["--all"], ["--profile", "runner-opencode"]]) { + expect(selectRunnerExecutions(parseRunnerSelectors(args), runnerMatrix).some(row => row.suite.id === CONNECTION_GUIDANCE_SUITE)).toBe(false); + } + expect(runnerSuites.find(suite => suite.id === CONNECTION_GUIDANCE_SUITE)?.definitionMetadata) + .toMatchObject({ fixtureDigest: connectionGuidanceDefinitionDigest(), companyAndAgentBudgetCents: 1_000, maximumAttemptsPerCell: 1 }); + }); + it("keeps workflow instructions out of new decline prompts and preserves the historical prompts", () => { + for (const task of connectionGuidanceTasks) { + const original = everydayTasks.find(row => row.id === task.id)!; + if (task.id.endsWith("decline")) { + expect(task.buildPrompt("nonce")).toMatch(/brief explanation is enough/); + expect(task.buildPrompt("nonce")).not.toMatch(/declin|Not now|None for now|do not|retry|try again|connection_request|connections_search|paperclip_|yield|poll/i); + expect(task.buildPrompt("nonce")).not.toBe(original.buildPrompt("nonce")); + } else expect(task.buildPrompt("nonce")).toBe(original.buildPrompt("nonce")); + } + expect(everydayTasks.find(task => task.id === "service-decline")!.buildPrompt("nonce")).toContain("do not try again"); + expect(everydayTasks.find(task => task.id === "connection-decline")!.buildPrompt("nonce")).toContain("do not try again"); + }); +}); + +const valid = { + caseId: "provider-decline" as const, decisionId: "decision", leadAgentId: "lead", issueId: "task", + decisions: [{ id: "decision", kind: "ask_user_questions", status: "answered", resolvedAt: "2026-10-06T22:00:00Z", + result: { answers: [{ questionId: "connection-provider:hubspot", optionIds: ["none"] }] } }], + replies: [{ authorAgentId: "lead", createdByRunId: "run", createdAt: "2026-10-06T22:00:02Z", body: "Contacts are unavailable after your decision." }], + runs: [{ id: "run", nativeIssueId: "task", agentId: "lead", status: "succeeded", finishedAt: "2026-10-06T22:00:03Z" }], + calls: 0, marker: "PRIVATE_MARKER", sameConnections: true, +}; +const passes = (input: Parameters[0]) => + gradeConnectionGuidanceDecline(input).every(check => check.passed); + +describe("neutral decline evidence", () => { + it("accepts an attributed explanation after each kind of saved decline", () => { + expect(passes(valid)).toBe(true); + for (const [caseId, kind] of [["service-decline", "request_confirmation"], ["connection-decline", "connection_intent"]] as const) { + expect(passes({ ...valid, caseId, decisions: [{ ...valid.decisions[0], kind, status: "rejected" }] })).toBe(true); + } + }); + it("rejects missing, stale, unattributed, wrong-agent, or unsuccessful explanations", () => { + for (const input of [ + { ...valid, replies: [] }, + { ...valid, replies: [{ ...valid.replies[0], body: "Done." }] }, + { ...valid, replies: [{ ...valid.replies[0], body: { text: "Unavailable" } }] }, + { ...valid, replies: [{ ...valid.replies[0], createdAt: "2026-10-06T21:59:00Z" }] }, + { ...valid, replies: [{ ...valid.replies[0], createdByRunId: undefined }] }, + { ...valid, replies: [{ ...valid.replies[0], createdByRunId: null, runId: "run" }] }, + { ...valid, replies: [{ ...valid.replies[0], createdByRunId: "other-run" }] }, + { ...valid, replies: [{ ...valid.replies[0], authorAgentId: "worker" }] }, + { ...valid, replies: [{ ...valid.replies[0], createdAt: "invalid" }] }, + { ...valid, runs: [{ ...valid.runs[0], agentId: "worker" }] }, + { ...valid, runs: [{ ...valid.runs[0], nativeIssueId: "other-task" }] }, + { ...valid, runs: [{ ...valid.runs[0], nativeIssueId: undefined }] }, + { ...valid, runs: [{ ...valid.runs[0], status: "failed" }] }, + { ...valid, runs: [{ ...valid.runs[0], finishedAt: "2026-10-06T21:59:00Z" }] }, + ]) expect(passes(input)).toBe(false); + }); + it("rejects wrong or repeated decisions, early use, missing call evidence and connection changes", () => { + for (const input of [ + { ...valid, decisionId: "other" }, + { ...valid, decisions: [] }, + { ...valid, decisions: [...valid.decisions, ...valid.decisions] }, + { ...valid, decisions: [{ ...valid.decisions[0], resolvedAt: undefined }] }, + { ...valid, decisions: [{ ...valid.decisions[0], kind: "request_confirmation" }] }, + { ...valid, decisions: [{ ...valid.decisions[0], status: "pending" }] }, + { ...valid, decisions: [{ ...valid.decisions[0], result: { answers: [{ questionId: "connection-provider:hubspot", optionIds: ["via:arcade:hubspot"] }] } }] }, + { ...valid, calls: 1 }, + { ...valid, calls: undefined }, + { ...valid, sameConnections: false }, + { ...valid, replies: [{ ...valid.replies[0], body: "Unavailable, but PRIVATE_MARKER" }] }, + ]) expect(passes(input)).toBe(false); + }); +}); diff --git a/tests/runner-e2e/everyday-flow.ts b/tests/runner-e2e/everyday-flow.ts index 22f9734815..050d079a12 100644 --- a/tests/runner-e2e/everyday-flow.ts +++ b/tests/runner-e2e/everyday-flow.ts @@ -1,4 +1,6 @@ import { gradeAgentmailSetup } from "./agentmail-setup-evidence.js"; +import { CONNECTION_GUIDANCE_SUITE, CONNECTION_GUIDANCE_BUDGET_CENTS } from "./connection-guidance-cases.js"; +import { gradeConnectionGuidanceDecline } from "./connection-guidance-evidence.js"; import { expect, type Page } from "@playwright/test"; import { runnerApiToolsEnabled } from "../../server/src/services/native-runtime/runner-api-rollout.js"; import { spawn } from "node:child_process"; @@ -62,6 +64,7 @@ export interface EverydayEvidence { fixtureConfiguration?: { apiToolsEnabled: boolean; aiConnection?: LiveFixtureValues["aiConnection"]; + connectionGuidanceBudgets?: { companyMonthlyCents: unknown; agentMonthlyCents: unknown }; }; documents?: Row[]; checks: StoryCheck[]; @@ -565,6 +568,20 @@ export async function runEverydayFlow(input: Input) { "connection-reviews.ts", "catalog.ts", ]; + if (execution.suite.id === CONNECTION_GUIDANCE_SUITE) { + harnessFiles.push("connection-guidance-cases.ts", "connection-guidance-evidence.ts"); + const [company, agent] = await Promise.all([ + api.get("/api/companies/" + fixtures.company.id), + api.get("/api/agents/" + fixtures.agent.id), + ]); + ev.fixtureConfiguration!.connectionGuidanceBudgets = { + companyMonthlyCents: company.budgetMonthlyCents, agentMonthlyCents: agent.budgetMonthlyCents, + }; + const budgetsMatch = company.budgetMonthlyCents === CONNECTION_GUIDANCE_BUDGET_CENTS && + agent.budgetMonthlyCents === CONNECTION_GUIDANCE_BUDGET_CENTS; + check("guidance-budget-hard-stops", budgetsMatch, "Public company and lead records retain both 1,000-cent hard stops before task creation."); + if (!budgetsMatch) throw new Error("Connection guidance budget admission failed before task creation"); + } ev.harnessDigest = createHash("sha256") .update( ( @@ -649,8 +666,8 @@ export async function runEverydayFlow(input: Input) { marker: `Pages: Roadmap, Meeting notes. Verification code: SERVICE_${nonce}`, authenticated: true, }); - if (providerChoice || nativeProviderCase) { - if (caseId === "provider-second") aggregatorFixture = await setupAggregatorFixture(api, fixtures.company.id, fixtures.agent.id, `CONTACTS_${nonce}`); + if (providerChoice || nativeProviderCase || (execution.suite.id === CONNECTION_GUIDANCE_SUITE && Boolean(review))) { + if (caseId === "provider-second" || (execution.suite.id === CONNECTION_GUIDANCE_SUITE && caseId === "provider-decline")) aggregatorFixture = await setupAggregatorFixture(api, fixtures.company.id, fixtures.agent.id, `CONTACTS_${nonce}`); const state = await api.get<{connections:Row[]}>(`/api/companies/${fixtures.company.id}/tools/connections`); initialConnections = state.connections.map(c=>c.id); } @@ -1116,6 +1133,18 @@ export async function runEverydayFlow(input: Input) { sameConnections:isDeepStrictEqual(state.connections.map(c=>c.id).sort(), initialConnections.sort()), })); } + if (execution.suite.id === CONNECTION_GUIDANCE_SUITE && + (caseId === "service-decline" || caseId === "connection-decline" || caseId === "provider-decline")) { + const issue = ev.issues.find(i => i.id === parent!.id)!; + const state = await api.get<{ connections: Row[] }>("/api/companies/" + fixtures.company.id + "/tools/connections"); + ev.checks.push(...gradeConnectionGuidanceDecline({ + caseId, decisionId: decisionId!, decisions: issue.interactions as any, + leadAgentId: fixtures.agent.id, issueId: parent!.id, replies: issue.comments ?? [], + runs: ev.runs, calls: review?.invocationCount() ?? aggregatorFixture?.invocationCount(), + marker: (caseId === "provider-decline" ? "CONTACTS_" : "SERVICE_") + nonce, + sameConnections: isDeepStrictEqual(state.connections.map(c => c.id).sort(), initialConnections.sort()), + })); + } if (declining) { const issue = ev.issues.find((i) => i.id === parent!.id)!; const requests = issue.interactions as Row[]; diff --git a/tests/runner-e2e/everyday-observations.ts b/tests/runner-e2e/everyday-observations.ts index adbaaeed81..33ae797704 100644 --- a/tests/runner-e2e/everyday-observations.ts +++ b/tests/runner-e2e/everyday-observations.ts @@ -1,3 +1,5 @@ +import type { IssueComment } from "../../packages/shared/src/types/issue.js"; + export interface StoryCheck { id: string; passed: boolean; @@ -24,9 +26,8 @@ export interface StoryIssue { wakeDiagnostics?: StoryWakeDiagnostics; blockedTransitionAt?: string | null; } -export interface StoryComment { +export interface StoryComment extends Partial> { id?: string; - authorAgentId?: string | null; body?: unknown; createdAt?: string; } diff --git a/tests/runner-e2e/live-fixtures.test.ts b/tests/runner-e2e/live-fixtures.test.ts index a2c231dc3a..d3ac65d6d7 100644 --- a/tests/runner-e2e/live-fixtures.test.ts +++ b/tests/runner-e2e/live-fixtures.test.ts @@ -27,6 +27,28 @@ describe("live runner fixtures", () => { }, ); + it.each(["runner-codex", "runner-acpx-claude", "runner-opencode"])( + "sets both connection-guidance budget stops for %s before execution", async profile => { + const execution = runnerMatrix.find(row => row.suite.id === "native-connection-guidance" && row.profile.id === profile)!; + let companyBudget: unknown; + let agentBudget: unknown; + const api = { + async get() { return [{ id: "local", driver: "local" }]; }, + async postSensitive() { return { id: "secret" }; }, + async post(url: string, data: any) { + if (url === "/api/companies") { companyBudget = data.budgetMonthlyCents; return { id: "company", name: "Test" }; } + if (url.endsWith("/agents")) { agentBudget = data.budgetMonthlyCents; return { id: "lead", ...data }; } + throw new Error("Unexpected POST " + url); + }, + } as unknown as RunnerApi; + const fixtures = await setupLiveFixtures({ api, execution, executionNonce: "nonce", workspacePath: "/tmp/test", + credentials: { [execution.profile.credential]: "test-value" } }); + expect(companyBudget).toBe(1_000); + expect(agentBudget).toBe(1_000); + await fixtures.teardown(); + }, + ); + it.each(["runner-codex", "legacy-codex", "runner-acpx-claude", "legacy-opencode"])( "creates a production-default %s hire with company and agent budget stops", async (profile) => { const execution = runnerMatrix.find(row => row.suite.id === "stock-harness" && row.profile.id === profile)!; diff --git a/tests/runner-e2e/live-fixtures.ts b/tests/runner-e2e/live-fixtures.ts index 3f466a63dd..26512d6079 100644 --- a/tests/runner-e2e/live-fixtures.ts +++ b/tests/runner-e2e/live-fixtures.ts @@ -139,7 +139,7 @@ export async function setupLiveFixtures(input: { return api.post("/api/companies", { name: `Runner E2E ${execution.id} ${input.executionNonce}`, description: "Ephemeral paid full-stack runner acceptance fixture", - budgetMonthlyCents: ["native-completion", "native-instruction-consolidation"].includes(execution.suite.id) + budgetMonthlyCents: ["native-completion", "native-instruction-consolidation", "native-connection-guidance"].includes(execution.suite.id) || (execution.suite.id === "everyday-workflows" && ["hire-reuse", "delegate-feedback"].includes(execution.task.id)) ? NATIVE_COMPLETION_BUDGET_CENTS : execution.suite.id === "task-titles" ? TASK_TITLE_BUDGET_CENTS : execution.suite.id === "stock-harness" ? 1_000 : 0, @@ -295,7 +295,7 @@ export async function setupLiveFixtures(input: { secretRefs, executionId: input.executionNonce, }); - if (execution.suite.id === "stock-harness" + if (["stock-harness", "native-connection-guidance"].includes(execution.suite.id) || (execution.suite.id === "everyday-workflows" && ["hire-reuse", "delegate-feedback"].includes(execution.task.id))) { agent.budgetMonthlyCents = 1_000; }