diff --git a/.github/workflows/runner-full-stack-e2e.yml b/.github/workflows/runner-full-stack-e2e.yml index 711d0aff2a..decdd6865f 100644 --- a/.github/workflows/runner-full-stack-e2e.yml +++ b/.github/workflows/runner-full-stack-e2e.yml @@ -891,7 +891,7 @@ jobs: run: node packages/paperclip-runner/scripts/materialize-opencode-binary.mjs - name: Install checksum-verified Grok executable - if: matrix.environmentId == 'local' && matrix.profileId == 'runner-acpx-grok' + if: matrix.environmentId == 'local' && (matrix.profileId == 'runner-acpx-grok' || matrix.profileId == 'runner-acpx-grok-subscription') run: node packages/grok-acp/install.mjs - name: Download immutable campaign outputs @@ -1044,7 +1044,7 @@ jobs: NODE - name: Prepare pinned Python artifact oracle image - if: (matrix.suiteId == 'everyday-workflows' || matrix.suiteId == 'grok-qualification') && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect') + if: (matrix.suiteId == 'everyday-workflows' || matrix.suiteId == 'grok-qualification' || matrix.suiteId == 'grok-subscription-qualification') && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect') run: | set -euo pipefail oracle_image='python@sha256:9d2e5553305c7c7b0097999bb17187c69b921ccd6bc9d40e4bb5ebe652c00285' @@ -1058,6 +1058,7 @@ jobs: ANTHROPIC_API_KEY: ${{ matrix.credentialName == 'ANTHROPIC_API_KEY' && secrets.ANTHROPIC_API_KEY || '' }} OPENROUTER_API_KEY: ${{ matrix.credentialName == 'OPENROUTER_API_KEY' && secrets.OPENROUTER_API_KEY || '' }} XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }} + GROK_AUTH_JSON: ${{ matrix.credentialName == 'GROK_AUTH_JSON' && secrets.GROK_AUTH_JSON || '' }} DAYTONA_API_KEY: ${{ matrix.environmentId == 'daytona' && secrets.DAYTONA_API_KEY || '' }} PAPERCLIP_E2E_DAYTONA_IMAGE: ${{ needs.daytona_image.outputs.image }} PAPERCLIP_RUNNER_REMOTE_PROVIDER_PACK_PATH: ${{ github.workspace }}/packages/paperclip-runner/provider-pack diff --git a/.github/workflows/runner-protocol-live-evals.yml b/.github/workflows/runner-protocol-live-evals.yml index c3bd0e01bf..73763827db 100644 --- a/.github/workflows/runner-protocol-live-evals.yml +++ b/.github/workflows/runner-protocol-live-evals.yml @@ -18,6 +18,10 @@ on: type: string default: "all" required: false + max_parallel: + description: "Optional lower concurrency (at least 2, no higher than the configured campaign limit)" + type: string + required: false max_infrastructure_retries: description: "Automatic retries only for explicitly retryable infrastructure failures (0-3)" type: number @@ -250,6 +254,7 @@ jobs: PAPERCLIP_PROTOCOL_EVALS_SHA: ${{ needs.authorize.outputs.evals_sha }} MAX_PARALLEL: ${{ vars.RUNNER_E2E_MAX_PARALLEL || needs.authorize.outputs.max_parallel_default }} MAX_PARALLEL_LIMIT: ${{ needs.authorize.outputs.max_parallel_limit }} + REQUESTED_MAX_PARALLEL: ${{ inputs.max_parallel }} ROSTERS: ${{ inputs.rosters || 'all' }} run: | set -euo pipefail @@ -257,6 +262,13 @@ jobs: echo "RUNNER_E2E_MAX_PARALLEL must be an integer from 2 through $MAX_PARALLEL_LIMIT for the two-shard direct suite." >&2 exit 1 fi + if [ -n "${REQUESTED_MAX_PARALLEL:-}" ]; then + if ! [[ "$REQUESTED_MAX_PARALLEL" =~ ^[1-9][0-9]{0,2}$ ]] || [ "$REQUESTED_MAX_PARALLEL" -lt 2 ] || [ "$REQUESTED_MAX_PARALLEL" -gt "$MAX_PARALLEL" ]; then + echo "max_parallel must be an integer from 2 through the configured campaign limit." >&2 + exit 1 + fi + MAX_PARALLEL="$REQUESTED_MAX_PARALLEL" + fi node packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs catalog \ --evals-root .paperclip-evals \ --rosters "$ROSTERS" \ @@ -332,6 +344,12 @@ jobs: - name: Materialize the pinned OpenCode executable before packaging run: node packages/paperclip-runner/scripts/materialize-opencode-binary.mjs + - name: Materialize the pinned Grok executable when the target includes it + run: | + if [ -f packages/grok-acp/install.mjs ]; then + node packages/grok-acp/install.mjs + fi + - name: Build runner CLI, daemon, and canonical attempt viewer run: | set -euo pipefail @@ -523,6 +541,7 @@ jobs: OPENAI_API_KEY: ${{ matrix.credentialName == 'OPENAI_API_KEY' && secrets.OPENAI_API_KEY || '' }} ANTHROPIC_API_KEY: ${{ matrix.credentialName == 'ANTHROPIC_API_KEY' && secrets.ANTHROPIC_API_KEY || '' }} OPENROUTER_API_KEY: ${{ matrix.credentialName == 'OPENROUTER_API_KEY' && secrets.OPENROUTER_API_KEY || '' }} + XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }} PAPERCLIP_CLAUDE_MANAGED_PROFILE_ID: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_PROFILE_ID }} PAPERCLIP_CLAUDE_MANAGED_AGENT_ID: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_AGENT_ID }} PAPERCLIP_CLAUDE_MANAGED_AGENT_VERSION: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_AGENT_VERSION }} diff --git a/packages/paperclip-runner/docs/runner-protocol-live-evals.md b/packages/paperclip-runner/docs/runner-protocol-live-evals.md index 98270c3201..5480b53109 100644 --- a/packages/paperclip-runner/docs/runner-protocol-live-evals.md +++ b/packages/paperclip-runner/docs/runner-protocol-live-evals.md @@ -205,6 +205,7 @@ The paid jobs read only the credential selected for each roster: - `OPENAI_API_KEY` for native Codex and ACPX Codex; - `ANTHROPIC_API_KEY` for ACPX Claude and Claude Managed; - `OPENROUTER_API_KEY` for native OpenCode and ACPX Pi; +- `XAI_API_KEY` for an explicitly selected ACPX Grok roster; - short-lived GitHub OIDC workload identity for AWS AgentCore. Claude Managed also requires the four nonsecret @@ -308,3 +309,14 @@ pnpm --filter @paperclipai/paperclip-runner \ --campaign-id gha-1-1 \ --output /tmp/runner-protocol-eval-catalog.json ``` + + +For Grok qualification, select `protocol-live-acpx-grok` and its exact eval +revision. Set `max_parallel: 2` when the API key has a low custom rate limit; +this admits one case per shard. The override can only lower the configured +campaign ceiling and cannot change the key's provider limits. Retain any +rate-limited attempts as failures. The build installs the target's pinned, +checksum-verified Grok binary before packaging the portable runtime. Targets +without the Grok package keep their existing build behavior. This protocol +workflow uses the explicitly selected API key; subscription credentials are +not delivered by it. diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs index 5656395b1e..3c0a50faf8 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs @@ -64,6 +64,7 @@ export function credentialForConfig(config) { if (config.provider === "acpx") { if (config.acpxAgent === "pi") return "OPENROUTER_API_KEY"; if (config.acpxAgent === "claude") return "ANTHROPIC_API_KEY"; + if (config.acpxAgent === "grok") return "XAI_API_KEY"; if (config.acpxAgent === "codex") return "OPENAI_API_KEY"; } throw new Error( diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs index a8717ef6ba..a97049d4c8 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs @@ -85,6 +85,7 @@ async function fixture() { test("maps every qualified driver to one explicit credential boundary", () => { assert.equal(credentialForConfig({ provider: "codex" }), "OPENAI_API_KEY"); + assert.equal(credentialForConfig({ provider: "acpx", acpxAgent: "grok" }), "XAI_API_KEY"); assert.equal( credentialForConfig({ provider: "opencode" }), "OPENROUTER_API_KEY", diff --git a/scripts/__tests__/release-verify-workflow.test.mjs b/scripts/__tests__/release-verify-workflow.test.mjs index d93b4bf5da..9ca66885e1 100644 --- a/scripts/__tests__/release-verify-workflow.test.mjs +++ b/scripts/__tests__/release-verify-workflow.test.mjs @@ -1,4 +1,5 @@ import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; import { existsSync, readdirSync, readFileSync } from "node:fs"; import path from "node:path"; import { fileURLToPath } from "node:url"; @@ -320,7 +321,7 @@ test("Runner eval workflows pin actions and gate paid live execution", () => { ]; const paidWorkflowNameSet = new Set(paidWorkflowNames); const providerSecretReference = - /secrets(?:\.(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|DAYTONA_API_KEY)\b|\[['"](?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|DAYTONA_API_KEY)['"]\])/g; + /secrets(?:\.(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|GROK_AUTH_JSON|DAYTONA_API_KEY)\b|\[['"](?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|GROK_AUTH_JSON|DAYTONA_API_KEY)['"]\])/g; for (const name of readdirSync(path.join(repoRoot, ".github/workflows"))) { if (!/\.ya?ml$/.test(name)) continue; const workflow = readWorkflow(name); @@ -430,3 +431,27 @@ test("Runner eval workflows pin actions and gate paid live execution", () => { } } }); + + +test("direct Grok qualification installs the pinned binary and scopes its API key", () => { + const workflow = readWorkflow("runner-protocol-live-evals.yml"); + assert.ok(workflow.includes("XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }}")); + assert.ok(workflow.includes("if [ -f packages/grok-acp/install.mjs ]; then")); + assert.ok(workflow.indexOf("node packages/grok-acp/install.mjs") < workflow.indexOf("pnpm --filter @paperclipai/paperclip-runner deploy --prod")); + assert.ok(!workflow.includes("secrets.GROK_AUTH_JSON")); +}); + +test("direct protocol concurrency override only lowers the configured ceiling", () => { + const workflow = readWorkflow("runner-protocol-live-evals.yml"); + const start = workflow.indexOf(' if [ -n "${REQUESTED_MAX_PARALLEL:-}" ]; then'); + const end = workflow.indexOf(" node packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs catalog", start); + assert.ok(start > 0 && end > start); + const script = workflow.slice(start, end) + '\nprintf "%s" "$MAX_PARALLEL"\n'; + for (const [requested, expected] of [["", "8"], ["2", "2"], ["8", "8"], ["1", null], ["9", null], ["0", null], ["-1", null], ["2.5", null], ["garbage", null], ["9999999999999999999999", null]]) { + const result = spawnSync("bash", ["-eu", "-c", script], { + env: { ...process.env, MAX_PARALLEL: "8", REQUESTED_MAX_PARALLEL: requested }, encoding: "utf8", + }); + assert.equal(result.status, expected === null ? 1 : 0, requested); + if (expected !== null) assert.equal(result.stdout, expected); + } +}); diff --git a/tests/runner-e2e/billing.ts b/tests/runner-e2e/billing.ts index 71729e77bb..72d130b01d 100644 --- a/tests/runner-e2e/billing.ts +++ b/tests/runner-e2e/billing.ts @@ -40,6 +40,14 @@ export interface CampaignBillingSummary { testsWithCompleteBilling: number; } +/** Render observed subtotals without presenting absent measurements as zero. */ +export function billingCoverageLabel(value: string, covered: number, total: number): string { + if (!Number.isInteger(covered) || !Number.isInteger(total) || covered <= 0 || total <= 0 || covered > total) { + return "Unavailable"; + } + return covered < total ? `${value} (partial: ${covered}/${total} runs)` : value; +} + function record(value: unknown): Record { return value && typeof value === "object" && !Array.isArray(value) ? (value as Record) diff --git a/tests/runner-e2e/dashboard-regenerate.ts b/tests/runner-e2e/dashboard-regenerate.ts index e145c37bf8..4c59220fad 100644 --- a/tests/runner-e2e/dashboard-regenerate.ts +++ b/tests/runner-e2e/dashboard-regenerate.ts @@ -21,6 +21,7 @@ interface PublishedResult extends RunnerE2EResult { interface PublishedCampaign { schema?: string; campaignId?: string; + source?: RunnerE2ECampaign["source"]; generatedAt: string; expected: string[]; results: PublishedResult[]; @@ -94,6 +95,16 @@ export async function regenerateRunnerDashboard(input: { expected, results, }); + // A rendering refresh must not inherit the renderer's checkout or CI event. + const retainedSource = results.find((result) => + result.source?.sha || result.source?.ref || result.source?.workflowRunUrl, + )?.source; + campaign.source = normalized.source ?? { + sha: retainedSource?.sha ?? null, + ref: retainedSource?.ref ?? null, + workflowRunUrl: retainedSource?.workflowRunUrl ?? null, + eventName: null, + }; const history = input.historyFile === null ? undefined diff --git a/tests/runner-e2e/dashboard.ts b/tests/runner-e2e/dashboard.ts index 4ad7912c25..0b1e9d988f 100644 --- a/tests/runner-e2e/dashboard.ts +++ b/tests/runner-e2e/dashboard.ts @@ -5,6 +5,7 @@ import { type ReportExecution, } from "./report-catalog.js"; import type { + RunnerE2EBillingSummary, RunnerE2ECampaign, RunnerE2EHistoryIndex, RunnerE2EResult, @@ -12,6 +13,7 @@ import type { } from "./types.js"; import { aggregateCampaignBilling, + billingCoverageLabel, summarizeExecutionBilling, } from "./billing.js"; @@ -57,8 +59,13 @@ function durationLabel(durationMs: number) { return `${Math.floor(seconds / 60)}m ${seconds % 60}s`; } -function tokenLabel(value: number) { - return new Intl.NumberFormat("en-US").format(value); +function tokenLabel(value: number, llm?: RunnerE2EBillingSummary["llm"]) { + const formatted = new Intl.NumberFormat("en-US").format(value); + return llm ? billingCoverageLabel(formatted, llm.runsWithTokenUsage, llm.runCount) : formatted; +} + +function llmCostLabel(value: number, llm: RunnerE2EBillingSummary["llm"]) { + return billingCoverageLabel(usdLabel(value), llm.runsWithReportedCost, llm.runCount); } function usdLabel(value: number | null) { @@ -264,7 +271,7 @@ function renderCase( data-gallery-runtime="${html(entry?.result.runtimeMode ?? execution.profile.expectedRuntimeMode)}" data-gallery-status="${html(label)}" data-gallery-duration="${html(entry ? durationLabel(entry.result.durationMs) : "Not run")}" - data-gallery-tokens="${html(billing ? `${tokenLabel(billing.llm.inputTokens)} in · ${tokenLabel(billing.llm.outputTokens)} out` : "Unavailable")}" + data-gallery-tokens="${html(billing ? `${tokenLabel(billing.llm.inputTokens, billing.llm)} in · ${tokenLabel(billing.llm.outputTokens, billing.llm)} out` : "Unavailable")}" data-gallery-matchers="${html(matcherResults.length > 0 ? `${passedMatchers}/${matcherResults.length} matchers passed${notReached ? ` · ${notReached} not reached` : ""}` : "No matchers recorded")}" aria-label="Open ${html(item.label)} for ${html(execution.id)} in gallery" > @@ -276,8 +283,8 @@ function renderCase( : ""; const billingStrip = billing ? `
-
Tokens${html(tokenLabel(billing.llm.inputTokens))} in · ${html(tokenLabel(billing.llm.outputTokens))} out${html(tokenLabel(billing.llm.cachedInputTokens))} cached · ${billing.llm.runsWithTokenUsage}/${billing.llm.runCount} runs covered
-
LLM spend${billing.llm.runsWithReportedCost > 0 ? html(usdLabel(billing.reportedCostUsd)) : html(billing.llm.costStatus)}${billing.llm.runsWithReportedCost}/${billing.llm.runCount} runs provider-priced
+
Tokens${html(tokenLabel(billing.llm.inputTokens, billing.llm))} in · ${html(tokenLabel(billing.llm.outputTokens, billing.llm))} out${html(tokenLabel(billing.llm.cachedInputTokens, billing.llm))} cached · ${billing.llm.runsWithTokenUsage}/${billing.llm.runCount} runs covered
+
LLM spend${html(llmCostLabel(billing.reportedCostUsd, billing.llm))}${billing.llm.runsWithReportedCost}/${billing.llm.runCount} runs provider-priced
Execution${billing.runtime.estimatedListCostUsd === undefined ? html(billing.runtime.costStatus === "not_metered" ? "Local · not metered" : "Cost unavailable") : `${html(usdLabel(billing.runtime.estimatedListCostUsd))} est.`}${html(durationLabel(billing.runtime.agentRunDurationMs))} agent${billing.runtime.leaseDurationMs === null ? "" : ` · ${html(durationLabel(billing.runtime.leaseDurationMs))} lease`}
` : ""; @@ -516,8 +523,8 @@ function renderHistory(history: RunnerE2EHistoryIndex | undefined) { ${html(campaign.campaignId)}${html(new Date(campaign.generatedAt).toLocaleString("en-US", { timeZone: "UTC" }))} UTC ${sha ? `${html(sha.slice(0, 10))}` : "Unknown"}${html(campaign.source.ref ?? "unknown ref")} ${status}${campaign.passed}/${campaign.selected} passed${campaign.incomplete ? ` · ${campaign.incomplete} incomplete` : ""} · ${campaign.complete ? "complete" : "partial"} - ${html(tokenLabel(campaign.billing.llm.inputTokens))} / ${html(tokenLabel(campaign.billing.llm.outputTokens))}input / output · ${html(tokenLabel(campaign.billing.llm.cachedInputTokens))} cached - ${html(usdLabel(campaign.billing.reportedLlmCostUsd))}${html(usdLabel(campaign.billing.estimatedRuntimeCostUsd))} runtime estimate + ${html(tokenLabel(campaign.billing.llm.inputTokens, campaign.billing.llm))} / ${html(tokenLabel(campaign.billing.llm.outputTokens, campaign.billing.llm))}input / output · ${html(tokenLabel(campaign.billing.llm.cachedInputTokens, campaign.billing.llm))} cached + ${html(llmCostLabel(campaign.billing.reportedLlmCostUsd, campaign.billing.llm))}${html(usdLabel(campaign.billing.estimatedRuntimeCostUsd))} runtime estimate ${html(durationLabel(campaign.billing.agentRunDurationMs))}${html(durationLabel(campaign.billing.leaseDurationMs))} lease `; }) @@ -607,8 +614,8 @@ function renderSuiteMatrix(input: { const summaryHtml = summary ? `
Pass rate${summary.selected > 0 ? ((summary.passed / summary.selected) * 100).toFixed(1) : "0.0"}%${summary.passed}/${summary.selected} passed${summary.incomplete ? ` · ${summary.incomplete} incomplete` : ""}
-
Tokens${html(tokenLabel(summary.billing.llm.totalTokens))}${html(tokenLabel(summary.billing.llm.inputTokens))} input · ${html(tokenLabel(summary.billing.llm.outputTokens))} output
-
Cost${html(usdLabel(summary.billing.observedAndEstimatedCostUsd))}reported LLM + runtime${summary.billing.judge ? " + judge" : ""} estimate
+
Tokens${html(tokenLabel(summary.billing.llm.totalTokens, summary.billing.llm))}${html(tokenLabel(summary.billing.llm.inputTokens, summary.billing.llm))} input · ${html(tokenLabel(summary.billing.llm.outputTokens, summary.billing.llm))} output
+
Known spend${html(summary.billing.llm.runsWithReportedCost > 0 || summary.billing.estimatedRuntimeCostUsd > 0 || summary.billing.judge ? `${usdLabel(summary.billing.observedAndEstimatedCostUsd)}${summary.billing.testsWithCompleteBilling < summary.billing.testCount ? " (partial)" : ""}` : "Unavailable")}reported LLM + runtime${summary.billing.judge ? " + judge" : ""} estimate
Agent time${html(durationLabel(summary.billing.agentRunDurationMs))}${html(durationLabel(summary.billing.leaseDurationMs))} lease
Execution${summary.executed}/${summary.selected}${summary.retries} retries · cleanup ${summary.cleanupPassed ? "passed" : "failed"}
` @@ -1118,10 +1125,10 @@ export function renderRunnerE2EDashboard(input: RunnerDashboardInput) { ${publicSummaryImageHref ? `
Runner E2E campaign status summary
` : ""}
-
${html(tokenLabel(campaignBilling.llm.inputTokens))}Input tokens
-
${html(tokenLabel(campaignBilling.llm.outputTokens))}Output tokens
-
${html(tokenLabel(campaignBilling.llm.cachedInputTokens))}Cached tokens
-
${html(usdLabel(campaignBilling.reportedLlmCostUsd))}LLM reported subtotal
+
${html(tokenLabel(campaignBilling.llm.inputTokens, campaignBilling.llm))}Input tokens
+
${html(tokenLabel(campaignBilling.llm.outputTokens, campaignBilling.llm))}Output tokens
+
${html(tokenLabel(campaignBilling.llm.cachedInputTokens, campaignBilling.llm))}Cached tokens
+
${html(llmCostLabel(campaignBilling.reportedLlmCostUsd, campaignBilling.llm))}LLM reported subtotal
${html(usdLabel(campaignBilling.estimatedRuntimeCostUsd))}Daytona list estimate
${campaignBilling.judge ? `
${html(usdLabel(campaignBilling.judge.estimatedCostUsd))}Judge estimate · ${campaignBilling.judge.attempts} attempts · ${campaignBilling.judge.attemptsWithUnknownUsage} unknown usage${html(tokenLabel(campaignBilling.judge.inputTokens))} in / ${html(tokenLabel(campaignBilling.judge.outputTokens))} out · $${campaignBilling.judge.reservedCostUsd.toFixed(6)} reserved
` : ""}
${html(durationLabel(campaignBilling.agentRunDurationMs))}Agent execution time
diff --git a/tests/runner-e2e/history.test.ts b/tests/runner-e2e/history.test.ts index 0ade0e0e06..27aa8a7bae 100644 --- a/tests/runner-e2e/history.test.ts +++ b/tests/runner-e2e/history.test.ts @@ -67,6 +67,55 @@ function result(execution: MatrixExecution, status: "passed" | "failed") { } describe("runner E2E campaign history", () => { + it("does not let empty legacy provenance hide a later measured source", async () => { + const root = await mkdtemp(path.join(os.tmpdir(), "runner-legacy-source-")); + temporaryDirectories.push(root); + const source = { + sha: "measured-source-sha", ref: "refs/heads/measured-source", + workflowRunUrl: "https://example.test/actions/runs/1", + }; + const results = [ + { ...result(runnerMatrix[0]!, "passed"), schema: "paperclip.runner-e2e.result/v1" }, + { ...result(runnerMatrix[1]!, "passed"), source }, + ]; + await writeFile(path.join(root, "normalized-results.json"), JSON.stringify({ + campaignId: "legacy-source", generatedAt: results[0]!.finishedAt, + expected: results.map((entry) => entry.executionId), results, + })); + vi.stubEnv("PAPERCLIP_RUNNER_E2E_SOURCE_SHA", "renderer-sha"); + vi.stubEnv("GITHUB_EVENT_NAME", "push"); + await regenerateRunnerDashboard({ bundle: root }); + const regenerated = JSON.parse(await readFile(path.join(root, "normalized-results.json"), "utf8")); + expect(regenerated.source).toEqual({ ...source, eventName: null }); + expect(regenerated.results[0].source).toEqual({ sha: null, ref: null, workflowRunUrl: null }); + expect(regenerated.passed).toBe(2); + }); + + it.each(["workflow_dispatch", null, undefined])("preserves retained source during regeneration (event %s)", async (eventName) => { + const root = await mkdtemp(path.join(os.tmpdir(), "runner-retained-source-")); + temporaryDirectories.push(root); + const execution = runnerMatrix[0]!; + const source = { + sha: "measured-source-sha", ref: "refs/heads/measured-source", + workflowRunUrl: "https://example.test/actions/runs/1", + }; + const retainedResult = { ...result(execution, "passed"), source }; + const campaign = buildRunnerCampaign({ + campaignId: "retained-source", generatedAt: retainedResult.finishedAt, + expected: [execution.id], results: [retainedResult], + }); + const published = { ...campaign, source: eventName === undefined ? undefined : { ...source, eventName } }; + await writeFile(path.join(root, "normalized-results.json"), JSON.stringify(published)); + vi.stubEnv("PAPERCLIP_RUNNER_E2E_SOURCE_SHA", "renderer-sha"); + vi.stubEnv("PAPERCLIP_RUNNER_E2E_SOURCE_REF", "refs/heads/renderer"); + vi.stubEnv("GITHUB_EVENT_NAME", "push"); + await regenerateRunnerDashboard({ bundle: root }); + const regenerated = JSON.parse(await readFile(path.join(root, "normalized-results.json"), "utf8")); + expect(regenerated.source).toEqual({ ...source, eventName: eventName ?? null }); + expect(regenerated.results).toEqual(campaign.results.map((entry) => ({ ...entry, evidenceValid: true, evidenceErrors: [] }))); + expect(regenerated.generatedAt).toBe(campaign.generatedAt); + }); + it("retains incomplete journeys in campaign, suite and history without marking them green", () => { const execution = runnerMatrix.find(e => e.suite.id === "first-task")!; const incomplete: RunnerE2EResult = { diff --git a/tests/runner-e2e/report.test.ts b/tests/runner-e2e/report.test.ts index fe898ef3e4..7bbea6e5a7 100644 --- a/tests/runner-e2e/report.test.ts +++ b/tests/runner-e2e/report.test.ts @@ -18,6 +18,49 @@ afterEach(async () => { }); describe("runner E2E report aggregation", () => { + it.each([ + { name: "missing", usage: null, runIds: ["run-1"], tokens: "Unavailable", cost: "Unavailable", htmlTokens: "Unavailable" }, + { name: "partial", usage: { runs: [{ usage: { inputTokens: 1250, outputTokens: 75, cachedInputTokens: 500, costUsd: 0.0125 } }, { usage: null }] }, runIds: ["run-1", "run-2"], tokens: "1250 input / 75 output / 500 cached (partial: 1/2 runs)", cost: "$0.012500 (partial: 1/2 runs)", htmlTokens: "1,250 (partial: 1/2 runs)" }, + { name: "reported", usage: { inputTokens: 1250, outputTokens: 75, cachedInputTokens: 500, costUsd: 0.0125 }, runIds: ["run-1"], tokens: "1250 input / 75 output / 500 cached", cost: "$0.012500", htmlTokens: "1,250" }, + { name: "reported zero cost", usage: { inputTokens: 1, outputTokens: 0, costUsd: 0 }, runIds: ["run-1"], tokens: "1 input / 0 output / 0 cached", cost: "$0.000000", htmlTokens: "1" }, + ])("renders $name usage with its actual coverage", async ({ usage, runIds, tokens, cost, htmlTokens }) => { + const root = await mkdtemp(path.join(os.tmpdir(), "runner-billing-report-")); + cleanupDirectories.push(root); + const executionId = "core-compatibility.runner-codex.local.message-marker"; + const directory = path.join(root, "attempt-1"); + await mkdir(directory); + const result: RunnerE2EResult = { + schema: "paperclip.runner-e2e.result/v2", executionId, suiteId: "core-compatibility", + attempt: 1, status: "passed", profileId: "runner-codex", environmentId: "local", + caseId: "message-marker", provider: "codex", model: "fixture-model", runtimeMode: "native", + startedAt: "2026-09-23T00:00:00Z", finishedAt: "2026-09-23T00:00:01Z", durationMs: 1000, + cleanup: "passed", runIds, usage, + }; + const raw = JSON.stringify(result); + await writeFile(path.join(directory, "result.json"), raw); + await writeFile(path.join(directory, "final-state.png"), "fake-png"); + await writeFile(path.join(directory, "evidence-manifest.json"), JSON.stringify({ files: ["final-state.png"], leaks: [], missing: [] })); + const output = path.join(root, "merged"); + await execFileAsync(process.execPath, [path.join(repositoryRoot, "cli/node_modules/tsx/dist/cli.mjs"), path.join(repositoryRoot, "tests/runner-e2e/report.ts")], { + cwd: repositoryRoot, + env: { ...process.env, PAPERCLIP_RUNNER_E2E_REPORT_ROOT: root, PAPERCLIP_RUNNER_E2E_REPORT_OUT: output, PAPERCLIP_RUNNER_E2E_EXPECTED_IDS: JSON.stringify([executionId]) }, + }); + const markdown = await readFile(path.join(output, "summary.md"), "utf8"); + expect(markdown.split("\n").find((line) => line.startsWith("Tokens: "))).toBe(`Tokens: ${tokens}`); + expect(markdown.split("\n").find((line) => line.startsWith("Provider-reported LLM cost: "))).toBe(`Provider-reported LLM cost: ${cost}`); + const dashboard = await readFile(path.join(output, "index.html"), "utf8"); + expect(dashboard).toContain(`${htmlTokens}Input tokens`); + if (usage === null) { + expect(markdown).not.toContain("0/0 | $0.000000"); + expect(dashboard).toContain("UnavailableLLM reported subtotal"); + expect(dashboard).toContain("Known spendUnavailable"); + } + expect(await readFile(path.join(directory, "result.json"), "utf8")).toBe(raw); + const normalized = JSON.parse(await readFile(path.join(output, "normalized-results.json"), "utf8")); + expect(normalized.passed).toBe(1); + expect(normalized.results[0].usage).toEqual(usage); + }); + it("keeps interrupted journeys incomplete unless their evidence is invalid", async () => { const root = await mkdtemp(path.join(os.tmpdir(), "runner-incomplete-report-")); cleanupDirectories.push(root); diff --git a/tests/runner-e2e/report.ts b/tests/runner-e2e/report.ts index 5be243a256..7024f8f277 100644 --- a/tests/runner-e2e/report.ts +++ b/tests/runner-e2e/report.ts @@ -8,7 +8,7 @@ import { writeFile, } from "node:fs/promises"; import { runnerMatrix } from "./catalog.js"; -import { summarizeExecutionBilling } from "./billing.js"; +import { billingCoverageLabel, summarizeExecutionBilling } from "./billing.js"; import { renderRunnerE2EDashboard } from "./dashboard.js"; import { buildRunnerCampaign, @@ -368,9 +368,9 @@ async function main() { "", ] : []), - `Tokens: ${billing.llm.inputTokens} input / ${billing.llm.outputTokens} output / ${billing.llm.cachedInputTokens} cached`, + `Tokens: ${billingCoverageLabel(`${billing.llm.inputTokens} input / ${billing.llm.outputTokens} output / ${billing.llm.cachedInputTokens} cached`, billing.llm.runsWithTokenUsage, billing.llm.runCount)}`, "", - `Provider-reported LLM cost: $${billing.reportedLlmCostUsd.toFixed(6)} (${billing.llm.runsWithReportedCost}/${billing.llm.runCount} runs priced)`, + `Provider-reported LLM cost: ${billingCoverageLabel(`$${billing.reportedLlmCostUsd.toFixed(6)}`, billing.llm.runsWithReportedCost, billing.llm.runCount)}`, "", `Estimated Daytona list-price runtime cost: $${billing.estimatedRuntimeCostUsd.toFixed(6)}`, ...(billing.judge ? [`Estimated judge cost: ${billing.judge.estimatedCostUsd === null ? "unknown" : `$${billing.judge.estimatedCostUsd.toFixed(6)}`}; ${billing.judge.attempts} attempts; ${billing.judge.attemptsWithUnknownUsage} with unknown usage; $${billing.judge.reservedCostUsd.toFixed(6)} reserved`] : []), @@ -385,7 +385,7 @@ async function main() { const cell = publicCampaignUrl ? `[${resolved.executionId}](${publicCampaignUrl}#execution-${encodeURIComponent(resolved.executionId)})` : resolved.executionId; - return `| ${cell} | ${resolved.attempt} | ${entry.valid ? "pass" : "fail"} | ${resolved.runtimeMode} | ${Math.round(resolved.durationMs / 1000)}s | ${cellBilling.llm.inputTokens}/${cellBilling.llm.outputTokens} | $${cellBilling.reportedCostUsd.toFixed(6)} (${cellBilling.llm.costStatus}) | ${runtimeCost === undefined ? cellBilling.runtime.costStatus : `$${runtimeCost.toFixed(6)} est.`} | ${detail} |`; + return `| ${cell} | ${resolved.attempt} | ${entry.valid ? "pass" : "fail"} | ${resolved.runtimeMode} | ${Math.round(resolved.durationMs / 1000)}s | ${billingCoverageLabel(`${cellBilling.llm.inputTokens}/${cellBilling.llm.outputTokens}`, cellBilling.llm.runsWithTokenUsage, cellBilling.llm.runCount)} | ${billingCoverageLabel(`$${cellBilling.reportedCostUsd.toFixed(6)}`, cellBilling.llm.runsWithReportedCost, cellBilling.llm.runCount)} | ${runtimeCost === undefined ? cellBilling.runtime.costStatus : `$${runtimeCost.toFixed(6)} est.`} | ${detail} |`; }), "", ]; diff --git a/tests/runner-e2e/workflow-security.test.ts b/tests/runner-e2e/workflow-security.test.ts index dc0ef205f8..0616e9d47a 100644 --- a/tests/runner-e2e/workflow-security.test.ts +++ b/tests/runner-e2e/workflow-security.test.ts @@ -305,7 +305,7 @@ describe("public repository paid workflow security", () => { const grokPreparation = paidJob.indexOf("- name: Install checksum-verified Grok executable"); expect(grokPreparation).toBeGreaterThan(paidInstall); expect(paidExecution).toBeGreaterThan(grokPreparation); - expect(paidJob).toContain("if: matrix.environmentId == 'local' && matrix.profileId == 'runner-acpx-grok'"); + expect(paidJob).toContain("if: matrix.environmentId == 'local' && (matrix.profileId == 'runner-acpx-grok' || matrix.profileId == 'runner-acpx-grok-subscription')"); expect(paidJob).toContain("run: node packages/grok-acp/install.mjs"); const everydayOracleStep = paidJob.slice( @@ -313,7 +313,7 @@ describe("public repository paid workflow security", () => { paidExecution, ); expect(everydayOracleStep).toContain( - "if: (matrix.suiteId == 'everyday-workflows' || matrix.suiteId == 'grok-qualification') && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect')", + "if: (matrix.suiteId == 'everyday-workflows' || matrix.suiteId == 'grok-qualification' || matrix.suiteId == 'grok-subscription-qualification') && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect')", ); expect(everydayOracleStep).toContain( `oracle_image='${everydayOracleImage}'`, @@ -530,6 +530,7 @@ describe("public repository paid workflow security", () => { ANTHROPIC_API_KEY: "matrix.credentialName == 'ANTHROPIC_API_KEY'", OPENROUTER_API_KEY: "matrix.credentialName == 'OPENROUTER_API_KEY'", XAI_API_KEY: "matrix.credentialName == 'XAI_API_KEY'", + GROK_AUTH_JSON: "matrix.credentialName == 'GROK_AUTH_JSON'", DAYTONA_API_KEY: "matrix.environmentId == 'daytona'", })) { expect(fullStack).toContain( @@ -557,7 +558,7 @@ describe("public repository paid workflow security", () => { ); const providerSecretReferences = [ ...contents.matchAll( - /secrets(?:\.(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|DAYTONA_API_KEY)\b|\[['"](?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|DAYTONA_API_KEY)['"]\])/g, + /secrets(?:\.(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|GROK_AUTH_JSON|DAYTONA_API_KEY)\b|\[['"](?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|GROK_AUTH_JSON|DAYTONA_API_KEY)['"]\])/g, ), ]; if (providerSecretReferences.length > 0) {