diff --git a/.github/workflows/runner-protocol-live-evals.yml b/.github/workflows/runner-protocol-live-evals.yml index 73763827db..fb063185a4 100644 --- a/.github/workflows/runner-protocol-live-evals.yml +++ b/.github/workflows/runner-protocol-live-evals.yml @@ -22,6 +22,14 @@ on: description: "Optional lower concurrency (at least 2, no higher than the configured campaign limit)" type: string required: false + grok_authentication: + description: "Explicit Grok credential source; subscription uses protected GROK_AUTH_JSON" + type: choice + options: + - api_key + - subscription + default: api_key + required: false max_infrastructure_retries: description: "Automatic retries only for explicitly retryable infrastructure failures (0-3)" type: number @@ -256,6 +264,7 @@ jobs: MAX_PARALLEL_LIMIT: ${{ needs.authorize.outputs.max_parallel_limit }} REQUESTED_MAX_PARALLEL: ${{ inputs.max_parallel }} ROSTERS: ${{ inputs.rosters || 'all' }} + GROK_AUTHENTICATION: ${{ inputs.grok_authentication || 'api_key' }} run: | set -euo pipefail if ! [[ "$MAX_PARALLEL" =~ ^[1-9][0-9]*$ ]] || [ "$MAX_PARALLEL" -lt 2 ] || [ "$MAX_PARALLEL" -gt "$MAX_PARALLEL_LIMIT" ]; then @@ -272,6 +281,7 @@ jobs: node packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs catalog \ --evals-root .paperclip-evals \ --rosters "$ROSTERS" \ + --grok-authentication "$GROK_AUTHENTICATION" \ --campaign-id "gha-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" \ --max-parallel "$MAX_PARALLEL" \ --output runner-protocol-eval-catalog.json @@ -542,6 +552,7 @@ jobs: ANTHROPIC_API_KEY: ${{ matrix.credentialName == 'ANTHROPIC_API_KEY' && secrets.ANTHROPIC_API_KEY || '' }} OPENROUTER_API_KEY: ${{ matrix.credentialName == 'OPENROUTER_API_KEY' && secrets.OPENROUTER_API_KEY || '' }} XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }} + PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET: ${{ matrix.credentialName == 'PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET' && secrets.GROK_AUTH_JSON || '' }} PAPERCLIP_CLAUDE_MANAGED_PROFILE_ID: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_PROFILE_ID }} PAPERCLIP_CLAUDE_MANAGED_AGENT_ID: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_AGENT_ID }} PAPERCLIP_CLAUDE_MANAGED_AGENT_VERSION: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_AGENT_VERSION }} @@ -582,19 +593,36 @@ jobs: --runner-package runner-protocol-build/extracted/paperclip-runner.tgz \ --runnerd runner-protocol-build/extracted/paperclip-runnerd \ --runs-root cell-output/runs \ + --summary-path cell-output/roster-summary.json \ --max-infrastructure-retries "$MAX_INFRASTRUCTURE_RETRIES" \ --run-id "gha-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}-${CELL_ID}" status=$? set -e CELL_EXIT_CODE="$status" node --input-type=module <<'NODE' - import { writeFileSync } from "node:fs"; + import { readFileSync, writeFileSync } from "node:fs"; + let authenticationMode; + let authenticationEvidenceFailure; + if (["XAI_API_KEY", "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET"].includes(process.env.CREDENTIAL_NAME)) { + const expected = process.env.CREDENTIAL_NAME === "XAI_API_KEY" ? "api_key" : "subscription"; + try { + const observed = JSON.parse(readFileSync("cell-output/roster-summary.json", "utf8")).authenticationMode; + // Never copy arbitrary provider output into cell metadata or errors. + if (["api_key", "subscription"].includes(observed)) authenticationMode = observed; + if (authenticationMode !== expected) authenticationEvidenceFailure = "grok_authentication_evidence_mismatch"; + } catch { + authenticationEvidenceFailure = "grok_authentication_evidence_unreadable"; + } + } writeFileSync("cell-output/cell.json", `${JSON.stringify({ schema: "paperclip.runner-protocol-eval.cell/v1", cellId: process.env.CELL_ID, rosterFile: process.env.ROSTER_FILE, caseId: process.env.CASE_ID, + ...(authenticationMode ? { authenticationMode } : {}), + ...(authenticationEvidenceFailure ? { authenticationEvidenceFailure } : {}), exitCode: Number(process.env.CELL_EXIT_CODE), }, null, 2)}\n`, { mode: 0o600 }); + if (authenticationEvidenceFailure) throw new Error(authenticationEvidenceFailure); NODE exit "$status" diff --git a/packages/paperclip-runner/docs/runner-protocol-live-evals.md b/packages/paperclip-runner/docs/runner-protocol-live-evals.md index 5480b53109..36b36fbded 100644 --- a/packages/paperclip-runner/docs/runner-protocol-live-evals.md +++ b/packages/paperclip-runner/docs/runner-protocol-live-evals.md @@ -119,6 +119,20 @@ from the default branch and provide: marked disabled; - `max_infrastructure_retries`: zero through three, applied only when an attempt explicitly reports a retryable infrastructure failure. +- `grok_authentication`: `api_key` (the compatibility default) or `subscription`. + Subscription requires an explicitly selected Grok roster and the owner-approved + `GROK_AUTH_JSON` secret in the protected `runner-e2e-paid` environment. API mode + uses only `XAI_API_KEY`. A missing selected credential fails; it never changes + authentication mode or falls back to another credential. + +For subscription qualification, pass `-f grok_authentication=subscription` when +dispatching the default-branch workflow. The immutable catalog, retained cell, +campaign roster, and result record carry `authenticationMode`. The cell reads it +from the pinned eval program's actual roster summary, and aggregation rejects +missing or mismatched authentication evidence. Select an eval revision that +supports Grok subscription admission and records this summary field. Keep API +and subscription campaigns separate when interpreting results. Remove temporary +subscription test secrets after the authorized qualification completes. The authorization job resolves the Paperclip branch to a commit and verifies the supplied eval commit before any checkout. A short-lived bot token generated diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs index 3c0a50faf8..2ea5c4f7cb 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs @@ -54,7 +54,14 @@ function inside(root, candidate, label) { return resolve(candidate); } -export function credentialForConfig(config) { +function validateGrokAuthenticationMode(mode) { + if (mode !== "api_key" && mode !== "subscription") { + throw new Error("Grok authentication mode must be api_key or subscription"); + } + return mode; +} + +export function credentialForConfig(config, grokAuthenticationMode = "api_key") { if (config.provider === "opencode") return "OPENROUTER_API_KEY"; if (config.provider === "claude_managed") return "ANTHROPIC_API_KEY"; if (config.provider === "aws_agentcore") return "AWS_AGENTCORE_OIDC"; @@ -64,7 +71,11 @@ export function credentialForConfig(config) { if (config.provider === "acpx") { if (config.acpxAgent === "pi") return "OPENROUTER_API_KEY"; if (config.acpxAgent === "claude") return "ANTHROPIC_API_KEY"; - if (config.acpxAgent === "grok") return "XAI_API_KEY"; + if (config.acpxAgent === "grok") { + return validateGrokAuthenticationMode(grokAuthenticationMode) === "subscription" + ? "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET" + : "XAI_API_KEY"; + } if (config.acpxAgent === "codex") return "OPENAI_API_KEY"; } throw new Error( @@ -117,8 +128,10 @@ export async function buildProtocolEvalCatalog({ campaignId, source = {}, maxParallel = 100, + grokAuthenticationMode = "api_key", }) { safeId(campaignId, "campaign ID"); + validateGrokAuthenticationMode(grokAuthenticationMode); if ( !Number.isSafeInteger(maxParallel) || maxParallel < 2 || @@ -159,7 +172,7 @@ export async function buildProtocolEvalCatalog({ `Config for ${rosterId}`, ); const config = await loadObject(configPath); - const credentialName = credentialForConfig(config); + const credentialName = credentialForConfig(config, grokAuthenticationMode); const cases = roster.cases.map((caseId) => safeId(caseId, "case ID")); if (new Set(cases).size !== cases.length) { throw new Error(`Live roster ${rosterId} repeats a case`); @@ -172,6 +185,9 @@ export async function buildProtocolEvalCatalog({ provider: String(config.provider ?? "codex"), driver: String(config.driver ?? "codex_app_server"), credentialName, + ...(config.provider === "acpx" && config.acpxAgent === "grok" + ? { authenticationMode: grokAuthenticationMode } + : {}), cases, }); } @@ -185,6 +201,9 @@ export async function buildProtocolEvalCatalog({ } } if (rosters.length === 0) throw new Error("No live rosters were selected"); + if (grokAuthenticationMode === "subscription" && !rosters.some((roster) => roster.authenticationMode === "subscription")) { + throw new Error("Subscription selection requires an explicit Grok roster"); + } const cells = rosters.flatMap((roster) => roster.cases.map((caseId) => ({ @@ -196,6 +215,7 @@ export async function buildProtocolEvalCatalog({ provider: roster.provider, driver: roster.driver, credentialName: roster.credentialName, + ...(roster.authenticationMode ? { authenticationMode: roster.authenticationMode } : {}), })), ); const shards = [[], []]; @@ -209,7 +229,7 @@ export async function buildProtocolEvalCatalog({ schema: "paperclip.runner-protocol-eval.catalog/v1", campaignId, source, - selection: { kind: requested === null ? "maintained_full" : "subset", rosters: rosterSelection }, + selection: { kind: requested === null ? "maintained_full" : "subset", rosters: rosterSelection, grokAuthenticationMode }, rosters, cells, matrices: shards.map((include) => ({ include })), @@ -399,6 +419,10 @@ export async function aggregateProtocolEvalCampaign({ ) { throw new Error(`Downloaded cell metadata drifted for ${cellId}`); } + if (expected.authenticationMode !== undefined && + (status.authenticationMode !== expected.authenticationMode || status.authenticationEvidenceFailure !== undefined)) { + throw new Error(`Invalid Grok authentication evidence for ${cellId}; retained cell metadata and attempts require inspection`); + } const attemptRoot = resolve(dirname(statusPath), "runs"); const attemptIds = []; const attemptMetadata = await lstat(attemptRoot).catch(() => null); @@ -473,6 +497,7 @@ export async function aggregateProtocolEvalCampaign({ model: cell.model, provider: cell.provider, driver: cell.driver, + ...(cell.authenticationMode ? { authenticationMode: cell.authenticationMode } : {}), attemptIds, finalAttemptId, disposition: score.disposition, @@ -513,6 +538,7 @@ export async function aggregateProtocolEvalCampaign({ model: roster.model, provider: roster.provider, driver: roster.driver, + ...(roster.authenticationMode ? { authenticationMode: roster.authenticationMode } : {}), selected: roster.cases.length, passed: results.filter( (result) => result.rosterId === roster.rosterId && result.passed, @@ -699,6 +725,7 @@ async function main() { rosterSelection: argument(args, "--rosters", "all"), campaignId: argument(args, "--campaign-id", `local-${Date.now()}`), maxParallel: Number(argument(args, "--max-parallel", "100")), + grokAuthenticationMode: argument(args, "--grok-authentication", "api_key"), source: { paperclipSha: process.env.PAPERCLIP_PROTOCOL_EVAL_SOURCE_SHA ?? null, evalsSha: process.env.PAPERCLIP_PROTOCOL_EVALS_SHA ?? null, diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs index a97049d4c8..6814929deb 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs @@ -104,6 +104,71 @@ test("maps every qualified driver to one explicit credential boundary", () => { ); }); +async function grokFixture() { + const value = await fixture(); + value.config = { + schema: "paperclip-runner/eval-config/v1", id: "live-grok", + provider: "acpx", driver: "acpx_runtime", acpxAgent: "grok", model: "grok-4.7", + }; + value.roster.model = value.config.model; + await writeFile(join(value.program, "configs/live-opencode-model.json"), JSON.stringify(value.config)); + await writeFile(join(value.program, "rosters/live-opencode-model.json"), JSON.stringify(value.roster)); + return value; +} + +test("selects exactly one Grok credential and preserves authentication provenance", async () => { + const { root } = await grokFixture(); + for (const [mode, credential] of [ + ["api_key", "XAI_API_KEY"], + ["subscription", "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET"], + ]) { + const catalog = await buildProtocolEvalCatalog({ evalsRoot: root, campaignId: "gha-42-1", grokAuthenticationMode: mode }); + assert.equal(catalog.selection.grokAuthenticationMode, mode); + assert.equal(catalog.cells[0].credentialName, credential); + assert.equal(catalog.cells[0].authenticationMode, mode); + assert.equal(catalog.rosters[0].authenticationMode, mode); + assert.equal(catalog.matrices[0].include[0].credentialName, credential); + } + for (const mode of ["", "auto", "subscription,api_key", null]) { + await assert.rejects(buildProtocolEvalCatalog({ evalsRoot: root, campaignId: "gha-42-1", grokAuthenticationMode: mode }), /authentication mode/); + } + const other = await fixture(); + await assert.rejects(buildProtocolEvalCatalog({ evalsRoot: other.root, campaignId: "gha-42-1", grokAuthenticationMode: "subscription" }), /requires an explicit Grok roster/); +}); + +test("subscription evidence cannot be substituted with an API or missing authentication record", async () => { + const { root, config, evalCase } = await grokFixture(); + const catalog = await buildProtocolEvalCatalog({ evalsRoot: root, campaignId: "gha-42-1", grokAuthenticationMode: "subscription" }); + const catalogPath = join(root, "catalog.json"); + await writeFile(catalogPath, JSON.stringify(catalog)); + const cell = catalog.cells[0]; + const download = join(root, "downloads/cell"); + const attemptId = "grok-attempt-01"; + const attempt = join(download, "runs", attemptId); + await mkdir(attempt, { recursive: true }); + for (const [file, value] of Object.entries({ + "artifact.json": { attemptId, usage: null }, + "score.json": { attemptId, caseId: evalCase.id, passed: true, disposition: "passed" }, + "case.json": evalCase, "config.json": config, + })) await writeFile(join(attempt, file), JSON.stringify(value)); + const aggregate = () => aggregateProtocolEvalCampaign({ + catalogPath, downloadsRoot: join(root, "downloads"), evalsRoot: root, + runsOut: join(root, "merged"), campaignOut: join(root, "campaign.json"), source: {}, + }); + const status = { cellId: cell.cellId, rosterFile: cell.rosterFile, caseId: cell.caseId, exitCode: 0 }; + await writeFile(join(download, "cell.json"), JSON.stringify({ ...status, authenticationMode: "subscription" })); + const passed = await aggregate(); + assert.equal(passed.totals.passed, 1); + assert.equal(passed.results[0].authenticationMode, "subscription"); + assert.equal(passed.rosters[0].authenticationMode, "subscription"); + await writeFile(join(download, "cell.json"), JSON.stringify({ ...status, authenticationMode: "subscription", authenticationEvidenceFailure: "grok_authentication_evidence_unreadable" })); + await assert.rejects(aggregate(), /Invalid Grok authentication evidence/); + for (const authenticationMode of [undefined, "api_key", "auto"]) { + await writeFile(join(download, "cell.json"), JSON.stringify({ ...status, authenticationMode })); + await assert.rejects(aggregate(), /Invalid Grok authentication evidence/); + } +}); + test("catalogs roster plus case cells and emits bounded balanced shards", async () => { const { root } = await fixture(); const catalog = await buildProtocolEvalCatalog({ diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs index 2d45293ef8..210b7954b5 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs @@ -1,5 +1,7 @@ import assert from "node:assert/strict"; -import { readFile } from "node:fs/promises"; +import { mkdtemp, mkdir, readFile, rm, writeFile } from "node:fs/promises"; +import { spawnSync } from "node:child_process"; +import { tmpdir } from "node:os"; import { resolve } from "node:path"; import test from "node:test"; @@ -13,6 +15,21 @@ const trustedPrWorkflowPath = resolve( ".github/workflows/pr-trusted.yml", ); +test("Grok subscription credentials require explicit catalog selection and observed auth evidence", async () => { + const workflow = await readFile(workflowPath, "utf8"); + const paid = workflow.slice(workflow.indexOf(" steps: &direct_eval_steps"), workflow.indexOf(" eval_shard_1:")); + assert.match(workflow, /grok_authentication:\n[\s\S]*?type: choice\n[\s\S]*?default: api_key/u); + assert.match(workflow, /--grok-authentication "\$GROK_AUTHENTICATION"/u); + assert.ok(paid.includes("PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET: ${{ matrix.credentialName == 'PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET' && secrets.GROK_AUTH_JSON || '' }}")); + assert.ok(paid.includes("XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }}")); + assert.match(paid, /--summary-path cell-output\/roster-summary\.json/u); + assert.match(paid, /JSON\.parse\(readFileSync\("cell-output\/roster-summary\.json", "utf8"\)\)\.authenticationMode/u); + assert.match(paid, /authenticationMode !== expected/u); + for (const job of [workflow.slice(0, workflow.indexOf(" eval_shard_0:")), workflow.slice(workflow.indexOf(" report:"))]) { + assert.doesNotMatch(job, /secrets\.GROK_AUTH_JSON/u); + } +}); + test("direct live eval workflow keeps paid execution behind stable actor authorization", async () => { const workflow = await readFile(workflowPath, "utf8"); assert.match(workflow, /^\s{2}authorize:/mu); @@ -228,3 +245,41 @@ test("trusted catalog, direct eval, and report orchestration stay on workflow re assert.doesNotMatch(section, /ref: \$\{\{ needs\.authorize\.outputs\.target_sha \}\}/u, `${job} must not execute target orchestration code`); } }); + + +test("authentication failures retain cell metadata without leaking malformed summary content", async () => { + const workflow = await readFile(workflowPath, "utf8"); + const script = workflow.match(/CELL_EXIT_CODE="\$status" node --input-type=module <<'NODE'\n([\s\S]*?) NODE/u)?.[1]; + assert.ok(script); + const root = await mkdtemp(resolve(tmpdir(), "grok-cell-evidence-")); + try { + await mkdir(resolve(root, "cell-output")); + const summary = resolve(root, "cell-output/roster-summary.json"); + for (const [content, expectedFailure, expectedMode] of [ + [undefined, "grok_authentication_evidence_unreadable", undefined], + ["{SENSITIVE_SENTINEL", "grok_authentication_evidence_unreadable", undefined], + ["null", "grok_authentication_evidence_unreadable", undefined], + ['{"authenticationMode":"SENSITIVE_SENTINEL"}', "grok_authentication_evidence_mismatch", undefined], + ['{"authenticationMode":"api_key"}', "grok_authentication_evidence_mismatch", "api_key"], + ['{"authenticationMode":"subscription"}', undefined, "subscription"], + ]) { + if (content === undefined) await rm(summary, { force: true }); + else await writeFile(summary, content); + const result = spawnSync(process.execPath, ["--input-type=module", "-e", script], { + cwd: root, encoding: "utf8", + env: { CREDENTIAL_NAME: "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET", CELL_ID: "cell-1", CASE_ID: "context", ROSTER_FILE: "grok.json", CELL_EXIT_CODE: "7" }, + }); + assert.equal(result.status, expectedFailure ? 1 : 0); + const retained = await readFile(resolve(root, "cell-output/cell.json"), "utf8"); + const metadata = JSON.parse(retained); + assert.equal(metadata.cellId, "cell-1"); + assert.equal(metadata.caseId, "context"); + assert.equal(metadata.exitCode, 7); + assert.equal(metadata.authenticationEvidenceFailure, expectedFailure); + assert.equal(metadata.authenticationMode, expectedMode); + assert.doesNotMatch(retained + result.stdout + result.stderr, /SENSITIVE_SENTINEL/u); + } + } finally { + await rm(root, { recursive: true, force: true }); + } +}); diff --git a/scripts/__tests__/release-verify-workflow.test.mjs b/scripts/__tests__/release-verify-workflow.test.mjs index 9ca66885e1..2712c8214c 100644 --- a/scripts/__tests__/release-verify-workflow.test.mjs +++ b/scripts/__tests__/release-verify-workflow.test.mjs @@ -433,12 +433,13 @@ test("Runner eval workflows pin actions and gate paid live execution", () => { }); -test("direct Grok qualification installs the pinned binary and scopes its API key", () => { +test("direct Grok qualification installs the pinned binary and scopes the selected credential", () => { const workflow = readWorkflow("runner-protocol-live-evals.yml"); assert.ok(workflow.includes("XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }}")); assert.ok(workflow.includes("if [ -f packages/grok-acp/install.mjs ]; then")); assert.ok(workflow.indexOf("node packages/grok-acp/install.mjs") < workflow.indexOf("pnpm --filter @paperclipai/paperclip-runner deploy --prod")); - assert.ok(!workflow.includes("secrets.GROK_AUTH_JSON")); + assert.ok(workflow.includes("PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET: ${{ matrix.credentialName == 'PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET' && secrets.GROK_AUTH_JSON || '' }}")); + assert.equal((workflow.match(/secrets\.GROK_AUTH_JSON/gu) ?? []).length, 1); }); test("direct protocol concurrency override only lowers the configured ceiling", () => {