From db8f8fe5b73a2697684a30261b0d306a9c631aba Mon Sep 17 00:00:00 2001 From: Dotta <34892728+cryppadotta@users.noreply.github.com> Date: Wed, 23 Sep 2026 16:43:21 -0700 Subject: [PATCH] fix(evals): select Grok subscription protocol credentials explicitly (#13901) ## Thinking Path > - Paperclip manages agent work through shared runner contracts. > - Direct protocol evals qualify provider behavior against a mock control plane. > - Grok supports API keys and company subscription credentials. > - The hosted protocol workflow selected an API key for every Grok cell. > - Product subscription support did not enable subscription protocol runs. > - This change adds explicit subscription selection and checks the recorded authentication mode. ## Linked Issues or Issue Description Refs #13878, #13882, #12618. The direct Grok protocol roster cannot run with subscription authentication through the trusted default-branch workflow. Add an explicit selector while keeping API-key dispatches compatible. Keep the actor allowlist, protected environment, immutable source revisions, and publication gates. ## What Changed - Add `grok_authentication` with `api_key` and `subscription` choices. Keep `api_key` as the compatibility default. - Deliver the protected `GROK_AUTH_JSON` secret only to a subscription-selected Grok cell. Do not provide an API key to that cell. - Read authentication mode from the pinned eval program's actual roster summary. Retain it in the cell, catalog, campaign roster, and result. - Reject missing or mismatched authentication evidence during aggregation. Preserve cell metadata and an allowlisted failure reason before failing a cell, so malformed evidence cannot hide the retained attempt. - Document credential setup, source separation, and temporary-secret cleanup. ## Verification - `node --test packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs scripts/__tests__/release-verify-workflow.test.mjs`: 34 tests passed. - Validated all 39 Grok cells at eval revision `3213dbec7e8ca1865ea95e6db7e7d34b095eb47a`; every selected cell requests only the subscription credential. Validation made zero provider calls. - Negative coverage rejects invalid selectors and missing or API authentication evidence in an otherwise passing subscription attempt. - `git diff --check` passed. All 53 current-head checks passed; the unchanged callback-drain timing test passed its bounded rerun, and the failed attempt is retained. Greptile reviewed `ecd3dcc0e998f07cf56fcb1f087946f50f388bec` at 5/5 with no remaining findings. - No Docker or broad builds ran on the developer machine. CI performs repository checks on the configured fleet. ## Risks Grok runs require an eval revision that records `authenticationMode` in the roster summary. Missing evidence fails closed. The credential contains account access and refresh tokens; an owner must approve its delivery to the protected environment before a live run. The change adds no PR trigger or authorization bypass. Live subscription protocol qualification remains pending this workflow reaching master. ## Model Used OpenAI GPT-6 through Codex, with tool use and code execution. The exact serving model identifier and context-window size are not exposed in this session. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [x] All Paperclip CI gates are green - [x] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip --- .../workflows/runner-protocol-live-evals.yml | 30 ++++++++- .../docs/runner-protocol-live-evals.md | 14 ++++ .../scripts/runner-protocol-eval-campaign.mjs | 35 ++++++++-- .../runner-protocol-eval-campaign.test.mjs | 65 +++++++++++++++++++ ...r-protocol-eval-workflow-security.test.mjs | 57 +++++++++++++++- .../release-verify-workflow.test.mjs | 5 +- 6 files changed, 198 insertions(+), 8 deletions(-) diff --git a/.github/workflows/runner-protocol-live-evals.yml b/.github/workflows/runner-protocol-live-evals.yml index 73763827db..fb063185a4 100644 --- a/.github/workflows/runner-protocol-live-evals.yml +++ b/.github/workflows/runner-protocol-live-evals.yml @@ -22,6 +22,14 @@ on: description: "Optional lower concurrency (at least 2, no higher than the configured campaign limit)" type: string required: false + grok_authentication: + description: "Explicit Grok credential source; subscription uses protected GROK_AUTH_JSON" + type: choice + options: + - api_key + - subscription + default: api_key + required: false max_infrastructure_retries: description: "Automatic retries only for explicitly retryable infrastructure failures (0-3)" type: number @@ -256,6 +264,7 @@ jobs: MAX_PARALLEL_LIMIT: ${{ needs.authorize.outputs.max_parallel_limit }} REQUESTED_MAX_PARALLEL: ${{ inputs.max_parallel }} ROSTERS: ${{ inputs.rosters || 'all' }} + GROK_AUTHENTICATION: ${{ inputs.grok_authentication || 'api_key' }} run: | set -euo pipefail if ! [[ "$MAX_PARALLEL" =~ ^[1-9][0-9]*$ ]] || [ "$MAX_PARALLEL" -lt 2 ] || [ "$MAX_PARALLEL" -gt "$MAX_PARALLEL_LIMIT" ]; then @@ -272,6 +281,7 @@ jobs: node packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs catalog \ --evals-root .paperclip-evals \ --rosters "$ROSTERS" \ + --grok-authentication "$GROK_AUTHENTICATION" \ --campaign-id "gha-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}" \ --max-parallel "$MAX_PARALLEL" \ --output runner-protocol-eval-catalog.json @@ -542,6 +552,7 @@ jobs: ANTHROPIC_API_KEY: ${{ matrix.credentialName == 'ANTHROPIC_API_KEY' && secrets.ANTHROPIC_API_KEY || '' }} OPENROUTER_API_KEY: ${{ matrix.credentialName == 'OPENROUTER_API_KEY' && secrets.OPENROUTER_API_KEY || '' }} XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }} + PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET: ${{ matrix.credentialName == 'PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET' && secrets.GROK_AUTH_JSON || '' }} PAPERCLIP_CLAUDE_MANAGED_PROFILE_ID: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_PROFILE_ID }} PAPERCLIP_CLAUDE_MANAGED_AGENT_ID: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_AGENT_ID }} PAPERCLIP_CLAUDE_MANAGED_AGENT_VERSION: ${{ vars.PAPERCLIP_CLAUDE_MANAGED_AGENT_VERSION }} @@ -582,19 +593,36 @@ jobs: --runner-package runner-protocol-build/extracted/paperclip-runner.tgz \ --runnerd runner-protocol-build/extracted/paperclip-runnerd \ --runs-root cell-output/runs \ + --summary-path cell-output/roster-summary.json \ --max-infrastructure-retries "$MAX_INFRASTRUCTURE_RETRIES" \ --run-id "gha-${GITHUB_RUN_ID}-${GITHUB_RUN_ATTEMPT}-${CELL_ID}" status=$? set -e CELL_EXIT_CODE="$status" node --input-type=module <<'NODE' - import { writeFileSync } from "node:fs"; + import { readFileSync, writeFileSync } from "node:fs"; + let authenticationMode; + let authenticationEvidenceFailure; + if (["XAI_API_KEY", "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET"].includes(process.env.CREDENTIAL_NAME)) { + const expected = process.env.CREDENTIAL_NAME === "XAI_API_KEY" ? "api_key" : "subscription"; + try { + const observed = JSON.parse(readFileSync("cell-output/roster-summary.json", "utf8")).authenticationMode; + // Never copy arbitrary provider output into cell metadata or errors. + if (["api_key", "subscription"].includes(observed)) authenticationMode = observed; + if (authenticationMode !== expected) authenticationEvidenceFailure = "grok_authentication_evidence_mismatch"; + } catch { + authenticationEvidenceFailure = "grok_authentication_evidence_unreadable"; + } + } writeFileSync("cell-output/cell.json", `${JSON.stringify({ schema: "paperclip.runner-protocol-eval.cell/v1", cellId: process.env.CELL_ID, rosterFile: process.env.ROSTER_FILE, caseId: process.env.CASE_ID, + ...(authenticationMode ? { authenticationMode } : {}), + ...(authenticationEvidenceFailure ? { authenticationEvidenceFailure } : {}), exitCode: Number(process.env.CELL_EXIT_CODE), }, null, 2)}\n`, { mode: 0o600 }); + if (authenticationEvidenceFailure) throw new Error(authenticationEvidenceFailure); NODE exit "$status" diff --git a/packages/paperclip-runner/docs/runner-protocol-live-evals.md b/packages/paperclip-runner/docs/runner-protocol-live-evals.md index 5480b53109..36b36fbded 100644 --- a/packages/paperclip-runner/docs/runner-protocol-live-evals.md +++ b/packages/paperclip-runner/docs/runner-protocol-live-evals.md @@ -119,6 +119,20 @@ from the default branch and provide: marked disabled; - `max_infrastructure_retries`: zero through three, applied only when an attempt explicitly reports a retryable infrastructure failure. +- `grok_authentication`: `api_key` (the compatibility default) or `subscription`. + Subscription requires an explicitly selected Grok roster and the owner-approved + `GROK_AUTH_JSON` secret in the protected `runner-e2e-paid` environment. API mode + uses only `XAI_API_KEY`. A missing selected credential fails; it never changes + authentication mode or falls back to another credential. + +For subscription qualification, pass `-f grok_authentication=subscription` when +dispatching the default-branch workflow. The immutable catalog, retained cell, +campaign roster, and result record carry `authenticationMode`. The cell reads it +from the pinned eval program's actual roster summary, and aggregation rejects +missing or mismatched authentication evidence. Select an eval revision that +supports Grok subscription admission and records this summary field. Keep API +and subscription campaigns separate when interpreting results. Remove temporary +subscription test secrets after the authorized qualification completes. The authorization job resolves the Paperclip branch to a commit and verifies the supplied eval commit before any checkout. A short-lived bot token generated diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs index 3c0a50faf8..2ea5c4f7cb 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.mjs @@ -54,7 +54,14 @@ function inside(root, candidate, label) { return resolve(candidate); } -export function credentialForConfig(config) { +function validateGrokAuthenticationMode(mode) { + if (mode !== "api_key" && mode !== "subscription") { + throw new Error("Grok authentication mode must be api_key or subscription"); + } + return mode; +} + +export function credentialForConfig(config, grokAuthenticationMode = "api_key") { if (config.provider === "opencode") return "OPENROUTER_API_KEY"; if (config.provider === "claude_managed") return "ANTHROPIC_API_KEY"; if (config.provider === "aws_agentcore") return "AWS_AGENTCORE_OIDC"; @@ -64,7 +71,11 @@ export function credentialForConfig(config) { if (config.provider === "acpx") { if (config.acpxAgent === "pi") return "OPENROUTER_API_KEY"; if (config.acpxAgent === "claude") return "ANTHROPIC_API_KEY"; - if (config.acpxAgent === "grok") return "XAI_API_KEY"; + if (config.acpxAgent === "grok") { + return validateGrokAuthenticationMode(grokAuthenticationMode) === "subscription" + ? "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET" + : "XAI_API_KEY"; + } if (config.acpxAgent === "codex") return "OPENAI_API_KEY"; } throw new Error( @@ -117,8 +128,10 @@ export async function buildProtocolEvalCatalog({ campaignId, source = {}, maxParallel = 100, + grokAuthenticationMode = "api_key", }) { safeId(campaignId, "campaign ID"); + validateGrokAuthenticationMode(grokAuthenticationMode); if ( !Number.isSafeInteger(maxParallel) || maxParallel < 2 || @@ -159,7 +172,7 @@ export async function buildProtocolEvalCatalog({ `Config for ${rosterId}`, ); const config = await loadObject(configPath); - const credentialName = credentialForConfig(config); + const credentialName = credentialForConfig(config, grokAuthenticationMode); const cases = roster.cases.map((caseId) => safeId(caseId, "case ID")); if (new Set(cases).size !== cases.length) { throw new Error(`Live roster ${rosterId} repeats a case`); @@ -172,6 +185,9 @@ export async function buildProtocolEvalCatalog({ provider: String(config.provider ?? "codex"), driver: String(config.driver ?? "codex_app_server"), credentialName, + ...(config.provider === "acpx" && config.acpxAgent === "grok" + ? { authenticationMode: grokAuthenticationMode } + : {}), cases, }); } @@ -185,6 +201,9 @@ export async function buildProtocolEvalCatalog({ } } if (rosters.length === 0) throw new Error("No live rosters were selected"); + if (grokAuthenticationMode === "subscription" && !rosters.some((roster) => roster.authenticationMode === "subscription")) { + throw new Error("Subscription selection requires an explicit Grok roster"); + } const cells = rosters.flatMap((roster) => roster.cases.map((caseId) => ({ @@ -196,6 +215,7 @@ export async function buildProtocolEvalCatalog({ provider: roster.provider, driver: roster.driver, credentialName: roster.credentialName, + ...(roster.authenticationMode ? { authenticationMode: roster.authenticationMode } : {}), })), ); const shards = [[], []]; @@ -209,7 +229,7 @@ export async function buildProtocolEvalCatalog({ schema: "paperclip.runner-protocol-eval.catalog/v1", campaignId, source, - selection: { kind: requested === null ? "maintained_full" : "subset", rosters: rosterSelection }, + selection: { kind: requested === null ? "maintained_full" : "subset", rosters: rosterSelection, grokAuthenticationMode }, rosters, cells, matrices: shards.map((include) => ({ include })), @@ -399,6 +419,10 @@ export async function aggregateProtocolEvalCampaign({ ) { throw new Error(`Downloaded cell metadata drifted for ${cellId}`); } + if (expected.authenticationMode !== undefined && + (status.authenticationMode !== expected.authenticationMode || status.authenticationEvidenceFailure !== undefined)) { + throw new Error(`Invalid Grok authentication evidence for ${cellId}; retained cell metadata and attempts require inspection`); + } const attemptRoot = resolve(dirname(statusPath), "runs"); const attemptIds = []; const attemptMetadata = await lstat(attemptRoot).catch(() => null); @@ -473,6 +497,7 @@ export async function aggregateProtocolEvalCampaign({ model: cell.model, provider: cell.provider, driver: cell.driver, + ...(cell.authenticationMode ? { authenticationMode: cell.authenticationMode } : {}), attemptIds, finalAttemptId, disposition: score.disposition, @@ -513,6 +538,7 @@ export async function aggregateProtocolEvalCampaign({ model: roster.model, provider: roster.provider, driver: roster.driver, + ...(roster.authenticationMode ? { authenticationMode: roster.authenticationMode } : {}), selected: roster.cases.length, passed: results.filter( (result) => result.rosterId === roster.rosterId && result.passed, @@ -699,6 +725,7 @@ async function main() { rosterSelection: argument(args, "--rosters", "all"), campaignId: argument(args, "--campaign-id", `local-${Date.now()}`), maxParallel: Number(argument(args, "--max-parallel", "100")), + grokAuthenticationMode: argument(args, "--grok-authentication", "api_key"), source: { paperclipSha: process.env.PAPERCLIP_PROTOCOL_EVAL_SOURCE_SHA ?? null, evalsSha: process.env.PAPERCLIP_PROTOCOL_EVALS_SHA ?? null, diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs index a97049d4c8..6814929deb 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-campaign.test.mjs @@ -104,6 +104,71 @@ test("maps every qualified driver to one explicit credential boundary", () => { ); }); +async function grokFixture() { + const value = await fixture(); + value.config = { + schema: "paperclip-runner/eval-config/v1", id: "live-grok", + provider: "acpx", driver: "acpx_runtime", acpxAgent: "grok", model: "grok-4.7", + }; + value.roster.model = value.config.model; + await writeFile(join(value.program, "configs/live-opencode-model.json"), JSON.stringify(value.config)); + await writeFile(join(value.program, "rosters/live-opencode-model.json"), JSON.stringify(value.roster)); + return value; +} + +test("selects exactly one Grok credential and preserves authentication provenance", async () => { + const { root } = await grokFixture(); + for (const [mode, credential] of [ + ["api_key", "XAI_API_KEY"], + ["subscription", "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET"], + ]) { + const catalog = await buildProtocolEvalCatalog({ evalsRoot: root, campaignId: "gha-42-1", grokAuthenticationMode: mode }); + assert.equal(catalog.selection.grokAuthenticationMode, mode); + assert.equal(catalog.cells[0].credentialName, credential); + assert.equal(catalog.cells[0].authenticationMode, mode); + assert.equal(catalog.rosters[0].authenticationMode, mode); + assert.equal(catalog.matrices[0].include[0].credentialName, credential); + } + for (const mode of ["", "auto", "subscription,api_key", null]) { + await assert.rejects(buildProtocolEvalCatalog({ evalsRoot: root, campaignId: "gha-42-1", grokAuthenticationMode: mode }), /authentication mode/); + } + const other = await fixture(); + await assert.rejects(buildProtocolEvalCatalog({ evalsRoot: other.root, campaignId: "gha-42-1", grokAuthenticationMode: "subscription" }), /requires an explicit Grok roster/); +}); + +test("subscription evidence cannot be substituted with an API or missing authentication record", async () => { + const { root, config, evalCase } = await grokFixture(); + const catalog = await buildProtocolEvalCatalog({ evalsRoot: root, campaignId: "gha-42-1", grokAuthenticationMode: "subscription" }); + const catalogPath = join(root, "catalog.json"); + await writeFile(catalogPath, JSON.stringify(catalog)); + const cell = catalog.cells[0]; + const download = join(root, "downloads/cell"); + const attemptId = "grok-attempt-01"; + const attempt = join(download, "runs", attemptId); + await mkdir(attempt, { recursive: true }); + for (const [file, value] of Object.entries({ + "artifact.json": { attemptId, usage: null }, + "score.json": { attemptId, caseId: evalCase.id, passed: true, disposition: "passed" }, + "case.json": evalCase, "config.json": config, + })) await writeFile(join(attempt, file), JSON.stringify(value)); + const aggregate = () => aggregateProtocolEvalCampaign({ + catalogPath, downloadsRoot: join(root, "downloads"), evalsRoot: root, + runsOut: join(root, "merged"), campaignOut: join(root, "campaign.json"), source: {}, + }); + const status = { cellId: cell.cellId, rosterFile: cell.rosterFile, caseId: cell.caseId, exitCode: 0 }; + await writeFile(join(download, "cell.json"), JSON.stringify({ ...status, authenticationMode: "subscription" })); + const passed = await aggregate(); + assert.equal(passed.totals.passed, 1); + assert.equal(passed.results[0].authenticationMode, "subscription"); + assert.equal(passed.rosters[0].authenticationMode, "subscription"); + await writeFile(join(download, "cell.json"), JSON.stringify({ ...status, authenticationMode: "subscription", authenticationEvidenceFailure: "grok_authentication_evidence_unreadable" })); + await assert.rejects(aggregate(), /Invalid Grok authentication evidence/); + for (const authenticationMode of [undefined, "api_key", "auto"]) { + await writeFile(join(download, "cell.json"), JSON.stringify({ ...status, authenticationMode })); + await assert.rejects(aggregate(), /Invalid Grok authentication evidence/); + } +}); + test("catalogs roster plus case cells and emits bounded balanced shards", async () => { const { root } = await fixture(); const catalog = await buildProtocolEvalCatalog({ diff --git a/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs b/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs index 2d45293ef8..210b7954b5 100644 --- a/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs +++ b/packages/paperclip-runner/scripts/runner-protocol-eval-workflow-security.test.mjs @@ -1,5 +1,7 @@ import assert from "node:assert/strict"; -import { readFile } from "node:fs/promises"; +import { mkdtemp, mkdir, readFile, rm, writeFile } from "node:fs/promises"; +import { spawnSync } from "node:child_process"; +import { tmpdir } from "node:os"; import { resolve } from "node:path"; import test from "node:test"; @@ -13,6 +15,21 @@ const trustedPrWorkflowPath = resolve( ".github/workflows/pr-trusted.yml", ); +test("Grok subscription credentials require explicit catalog selection and observed auth evidence", async () => { + const workflow = await readFile(workflowPath, "utf8"); + const paid = workflow.slice(workflow.indexOf(" steps: &direct_eval_steps"), workflow.indexOf(" eval_shard_1:")); + assert.match(workflow, /grok_authentication:\n[\s\S]*?type: choice\n[\s\S]*?default: api_key/u); + assert.match(workflow, /--grok-authentication "\$GROK_AUTHENTICATION"/u); + assert.ok(paid.includes("PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET: ${{ matrix.credentialName == 'PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET' && secrets.GROK_AUTH_JSON || '' }}")); + assert.ok(paid.includes("XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }}")); + assert.match(paid, /--summary-path cell-output\/roster-summary\.json/u); + assert.match(paid, /JSON\.parse\(readFileSync\("cell-output\/roster-summary\.json", "utf8"\)\)\.authenticationMode/u); + assert.match(paid, /authenticationMode !== expected/u); + for (const job of [workflow.slice(0, workflow.indexOf(" eval_shard_0:")), workflow.slice(workflow.indexOf(" report:"))]) { + assert.doesNotMatch(job, /secrets\.GROK_AUTH_JSON/u); + } +}); + test("direct live eval workflow keeps paid execution behind stable actor authorization", async () => { const workflow = await readFile(workflowPath, "utf8"); assert.match(workflow, /^\s{2}authorize:/mu); @@ -228,3 +245,41 @@ test("trusted catalog, direct eval, and report orchestration stay on workflow re assert.doesNotMatch(section, /ref: \$\{\{ needs\.authorize\.outputs\.target_sha \}\}/u, `${job} must not execute target orchestration code`); } }); + + +test("authentication failures retain cell metadata without leaking malformed summary content", async () => { + const workflow = await readFile(workflowPath, "utf8"); + const script = workflow.match(/CELL_EXIT_CODE="\$status" node --input-type=module <<'NODE'\n([\s\S]*?) NODE/u)?.[1]; + assert.ok(script); + const root = await mkdtemp(resolve(tmpdir(), "grok-cell-evidence-")); + try { + await mkdir(resolve(root, "cell-output")); + const summary = resolve(root, "cell-output/roster-summary.json"); + for (const [content, expectedFailure, expectedMode] of [ + [undefined, "grok_authentication_evidence_unreadable", undefined], + ["{SENSITIVE_SENTINEL", "grok_authentication_evidence_unreadable", undefined], + ["null", "grok_authentication_evidence_unreadable", undefined], + ['{"authenticationMode":"SENSITIVE_SENTINEL"}', "grok_authentication_evidence_mismatch", undefined], + ['{"authenticationMode":"api_key"}', "grok_authentication_evidence_mismatch", "api_key"], + ['{"authenticationMode":"subscription"}', undefined, "subscription"], + ]) { + if (content === undefined) await rm(summary, { force: true }); + else await writeFile(summary, content); + const result = spawnSync(process.execPath, ["--input-type=module", "-e", script], { + cwd: root, encoding: "utf8", + env: { CREDENTIAL_NAME: "PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET", CELL_ID: "cell-1", CASE_ID: "context", ROSTER_FILE: "grok.json", CELL_EXIT_CODE: "7" }, + }); + assert.equal(result.status, expectedFailure ? 1 : 0); + const retained = await readFile(resolve(root, "cell-output/cell.json"), "utf8"); + const metadata = JSON.parse(retained); + assert.equal(metadata.cellId, "cell-1"); + assert.equal(metadata.caseId, "context"); + assert.equal(metadata.exitCode, 7); + assert.equal(metadata.authenticationEvidenceFailure, expectedFailure); + assert.equal(metadata.authenticationMode, expectedMode); + assert.doesNotMatch(retained + result.stdout + result.stderr, /SENSITIVE_SENTINEL/u); + } + } finally { + await rm(root, { recursive: true, force: true }); + } +}); diff --git a/scripts/__tests__/release-verify-workflow.test.mjs b/scripts/__tests__/release-verify-workflow.test.mjs index 9ca66885e1..2712c8214c 100644 --- a/scripts/__tests__/release-verify-workflow.test.mjs +++ b/scripts/__tests__/release-verify-workflow.test.mjs @@ -433,12 +433,13 @@ test("Runner eval workflows pin actions and gate paid live execution", () => { }); -test("direct Grok qualification installs the pinned binary and scopes its API key", () => { +test("direct Grok qualification installs the pinned binary and scopes the selected credential", () => { const workflow = readWorkflow("runner-protocol-live-evals.yml"); assert.ok(workflow.includes("XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }}")); assert.ok(workflow.includes("if [ -f packages/grok-acp/install.mjs ]; then")); assert.ok(workflow.indexOf("node packages/grok-acp/install.mjs") < workflow.indexOf("pnpm --filter @paperclipai/paperclip-runner deploy --prod")); - assert.ok(!workflow.includes("secrets.GROK_AUTH_JSON")); + assert.ok(workflow.includes("PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET: ${{ matrix.credentialName == 'PAPERCLIP_ACPX_GROK_AUTH_JSON_SECRET' && secrets.GROK_AUTH_JSON || '' }}")); + assert.equal((workflow.match(/secrets\.GROK_AUTH_JSON/gu) ?? []).length, 1); }); test("direct protocol concurrency override only lowers the configured ceiling", () => {