From 950ccb8eef8ee09bfad49cebbe108fdfec887ee3 Mon Sep 17 00:00:00 2001 From: Dotta <34892728+cryppadotta@users.noreply.github.com> Date: Tue, 22 Sep 2026 18:17:25 -0700 Subject: [PATCH] ci: enable Grok qualification in the trusted paid workflow (#13845) ## Thinking Path > - Paperclip manages AI agents and their work. > - Product E2E tests verify tasks through the browser, server, and runner. > - These paid tests use a trusted workflow from master and an isolated target commit. > - The Grok target branch selects XAI_API_KEY, but the trusted workflow does not deliver that credential. > - Its artifact test also needs the pinned Python verifier before execution. > - This pull request adds both bindings inside the existing paid boundary. > - The tests can then run on the configured EC2 fleet without laptop Docker. ## Linked Issues or Issue Description Related evaluation infrastructure: https://github.com/paperclipai/paperclip/pull/11297. No duplicate Grok paid-workflow change was found. **What existing behavior does this improve?** Branch-targeted Grok Product E2E qualification in the existing paid workflow. **Current behavior** Grok cells cannot receive their selected API credential. The Grok build-revise case also misses the artifact-verifier setup step. **Proposed behavior** Deliver XAI_API_KEY only when the selected matrix credential is XAI_API_KEY. Prepare the existing pinned verifier for the Grok qualification suite. **Reason and benefit** Run the controller, browser, runner and artifact checks on the EC2 fleet. Preserve default-branch workflow authorization and protected environment secret access. ## What Changed - Bind the selected XAI credential only in the paid test step. - Install the checksum-verified Grok binary for local cells before provider access. - Include Grok qualification in the existing pinned artifact-verifier preparation. - Add an optional max_parallel input that can only lower the configured campaign concurrency. Use 1 for the Grok test key. - Extend security assertions and document setup. ## Verification - Ran the Product E2E workflow-security tests: 11 passed. - Checked the diff for whitespace errors. - Reviewed credential selection, setup ordering, numeric actor gates, target commit pinning, and trusted report checkout. - Live Grok execution follows after this workflow is available on master. This PR does not claim completed Grok qualification. ## Risks The paid test step can use the selected XAI credential and incur provider charges. The credential remains in runner-e2e-paid and is absent from setup, build, and reporting jobs. The default-branch gate and existing environment restrictions remain in place. No database migration or product behavior changes. ## Model Used OpenAI Codex, GPT-6 family, with reasoning, code editing, and tool execution. The exact deployment model ID and context-window size are not exposed in this session. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [ ] All Paperclip CI gates are green - [ ] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip --- .github/workflows/runner-full-stack-e2e.yml | 19 ++++++++++++++++++- tests/runner-e2e/README.md | 14 ++++++++++++++ tests/runner-e2e/SECURITY.md | 4 ++-- tests/runner-e2e/workflow-security.test.ts | 15 +++++++++++++-- 4 files changed, 47 insertions(+), 5 deletions(-) diff --git a/.github/workflows/runner-full-stack-e2e.yml b/.github/workflows/runner-full-stack-e2e.yml index 9016efa30c..5556752cbc 100644 --- a/.github/workflows/runner-full-stack-e2e.yml +++ b/.github/workflows/runner-full-stack-e2e.yml @@ -9,6 +9,10 @@ on: description: "Branch in paperclipai/paperclip to test; the trusted workflow still runs from master" type: string required: false + max_parallel: + description: "Optional lower concurrency for this campaign (cannot exceed the configured limit)" + type: string + required: false all: description: "Run the complete paid matrix when no narrower selector is supplied" type: boolean @@ -286,6 +290,7 @@ jobs: SELECT_ID: ${{ inputs.id }} MAX_PARALLEL: ${{ vars.RUNNER_E2E_MAX_PARALLEL || needs.authorize.outputs.max_parallel_default }} MAX_PARALLEL_LIMIT: ${{ needs.authorize.outputs.max_parallel_limit }} + REQUESTED_MAX_PARALLEL: ${{ inputs.max_parallel }} run: | set -euo pipefail args=(--matrix-json) @@ -346,6 +351,13 @@ jobs: echo "RUNNER_E2E_MAX_PARALLEL must be an integer from 1 through $MAX_PARALLEL_LIMIT for the selected runner." >&2 exit 1 fi + if [ -n "${REQUESTED_MAX_PARALLEL:-}" ]; then + if ! [[ "$REQUESTED_MAX_PARALLEL" =~ ^[1-9][0-9]{0,2}$ ]] || [ "$REQUESTED_MAX_PARALLEL" -gt "$MAX_PARALLEL" ]; then + echo "max_parallel must be an integer from 1 through the configured campaign limit." >&2 + exit 1 + fi + MAX_PARALLEL="$REQUESTED_MAX_PARALLEL" + fi echo "max_parallel=$MAX_PARALLEL" >> "$GITHUB_OUTPUT" daytona_image: @@ -874,6 +886,10 @@ jobs: if: matrix.environmentId == 'local' && (matrix.profileId == 'legacy-opencode' || matrix.profileId == 'runner-opencode' || matrix.suiteId == 'openrouter-model-breadth') run: node packages/paperclip-runner/scripts/materialize-opencode-binary.mjs + - name: Install checksum-verified Grok executable + if: matrix.environmentId == 'local' && matrix.profileId == 'runner-acpx-grok' + run: node packages/grok-acp/install.mjs + - name: Download immutable campaign outputs if: startsWith(matrix.profileId, 'runner-') || matrix.suiteId == 'openrouter-model-breadth' uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8 @@ -1024,7 +1040,7 @@ jobs: NODE - name: Prepare pinned Python artifact oracle image - if: matrix.suiteId == 'everyday-workflows' && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect') + if: (matrix.suiteId == 'everyday-workflows' || matrix.suiteId == 'grok-qualification') && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect') run: | set -euo pipefail oracle_image='python@sha256:9d2e5553305c7c7b0097999bb17187c69b921ccd6bc9d40e4bb5ebe652c00285' @@ -1037,6 +1053,7 @@ jobs: OPENAI_API_KEY: ${{ matrix.credentialName == 'OPENAI_API_KEY' && secrets.OPENAI_API_KEY || '' }} ANTHROPIC_API_KEY: ${{ matrix.credentialName == 'ANTHROPIC_API_KEY' && secrets.ANTHROPIC_API_KEY || '' }} OPENROUTER_API_KEY: ${{ matrix.credentialName == 'OPENROUTER_API_KEY' && secrets.OPENROUTER_API_KEY || '' }} + XAI_API_KEY: ${{ matrix.credentialName == 'XAI_API_KEY' && secrets.XAI_API_KEY || '' }} DAYTONA_API_KEY: ${{ matrix.environmentId == 'daytona' && secrets.DAYTONA_API_KEY || '' }} PAPERCLIP_E2E_DAYTONA_IMAGE: ${{ needs.daytona_image.outputs.image }} PAPERCLIP_RUNNER_REMOTE_PROVIDER_PACK_PATH: ${{ github.workspace }}/packages/paperclip-runner/provider-pack diff --git a/tests/runner-e2e/README.md b/tests/runner-e2e/README.md index d10684439e..00c1ffc3d2 100644 --- a/tests/runner-e2e/README.md +++ b/tests/runner-e2e/README.md @@ -1001,3 +1001,17 @@ requires the saved content and usable composer to remain visible. Run it with th standard `tests/e2e/playwright.config.ts`; no provider or Daytona credentials are needed. Browser-support tests separately exercise blank-root/pending-module failure evidence, so a future blank page is distinguishable from a loaded task. + +### Grok branch qualification on EC2 + +The trusted default-branch workflow can run the explicit `grok-qualification` +suite from a selected target branch. Store `XAI_API_KEY` only in the protected +`runner-e2e-paid` environment. The paid step delivers it only to a profile whose +credential name is `XAI_API_KEY`. The Grok `build-revise` cells prepare the same +pinned Python artifact verifier used by Everyday Workflows, before credentials +are exposed. Local Grok cells also run the checksum-verifying binary installer +before receiving credentials. With `RUNNER_E2E_AWS_ENABLED=true`, the controller, browser and +artifact verifier run on the existing EC2 fleet; no developer laptop Docker +service is required. Set the optional `max_parallel` dispatch input to `1` for +keys with low request limits. It can only lower the configured campaign limit. +Keep subscription qualification separate from API-key results. diff --git a/tests/runner-e2e/SECURITY.md b/tests/runner-e2e/SECURITY.md index 57ed632d78..e781d54fc6 100644 --- a/tests/runner-e2e/SECURITY.md +++ b/tests/runner-e2e/SECURITY.md @@ -1,6 +1,6 @@ # Runner E2E security for a public repository -This suite can spend provider money, expose four API credentials to isolated +This suite can spend provider money, expose selected API credentials to isolated test processes, publish a container, retain private visual evidence, and write public structured evidence. Treat changes to the workflow, harness, fixture prompts, evidence packager, and publisher as security-sensitive production @@ -64,7 +64,7 @@ rejects mutable tag or branch references. ## Secrets and protected environments Create `runner-e2e-paid`, restrict deployments to the default branch, and put -only `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `OPENROUTER_API_KEY`, and +only `OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, `OPENROUTER_API_KEY`, `XAI_API_KEY`, and `DAYTONA_API_KEY` in it. Do not duplicate these credentials as repository- or organization-level Actions secrets: environment scoping is the boundary that prevents branch or pull-request jobs from requesting them. Require approval diff --git a/tests/runner-e2e/workflow-security.test.ts b/tests/runner-e2e/workflow-security.test.ts index 1a9c021dd0..5e3fd4084f 100644 --- a/tests/runner-e2e/workflow-security.test.ts +++ b/tests/runner-e2e/workflow-security.test.ts @@ -294,12 +294,18 @@ describe("public repository paid workflow security", () => { expect(paidExecution).toBeGreaterThan(awsFfmpegInstall); expect(paidExecution).toBeGreaterThan(daytonaPluginPreparation); expect(paidExecution).toBeGreaterThan(everydayOraclePreparation); + const grokPreparation = paidJob.indexOf("- name: Install checksum-verified Grok executable"); + expect(grokPreparation).toBeGreaterThan(paidInstall); + expect(paidExecution).toBeGreaterThan(grokPreparation); + expect(paidJob).toContain("if: matrix.environmentId == 'local' && matrix.profileId == 'runner-acpx-grok'"); + expect(paidJob).toContain("run: node packages/grok-acp/install.mjs"); + const everydayOracleStep = paidJob.slice( everydayOraclePreparation, paidExecution, ); expect(everydayOracleStep).toContain( - "if: matrix.suiteId == 'everyday-workflows' && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect')", + "if: (matrix.suiteId == 'everyday-workflows' || matrix.suiteId == 'grok-qualification') && (matrix.caseId == 'build-revise' || matrix.caseId == 'delegate-feedback' || matrix.caseId == 'agent-review-handoff' || matrix.caseId == 'hire-reuse' || matrix.caseId == 'recover-controller' || matrix.caseId == 'stop-redirect')", ); expect(everydayOracleStep).toContain( `oracle_image='${everydayOracleImage}'`, @@ -367,6 +373,10 @@ describe("public repository paid workflow security", () => { ); expect(authorizeJob).toContain('echo "max_parallel_limit=100"'); expect(fullStack).toContain('[ "$MAX_PARALLEL_LIMIT" -gt 100 ]'); + expect(fullStack).toContain("REQUESTED_MAX_PARALLEL: ${{ inputs.max_parallel }}"); + expect(fullStack).toContain('[ "$REQUESTED_MAX_PARALLEL" -gt "$MAX_PARALLEL" ]'); + expect(fullStack).toContain('[[ "$REQUESTED_MAX_PARALLEL" =~ ^[1-9][0-9]{0,2}$ ]]'); + expect(fullStack).toContain( '[ "$MAX_PARALLEL" -gt "$MAX_PARALLEL_LIMIT" ]', ); @@ -511,6 +521,7 @@ describe("public repository paid workflow security", () => { OPENAI_API_KEY: "matrix.credentialName == 'OPENAI_API_KEY'", ANTHROPIC_API_KEY: "matrix.credentialName == 'ANTHROPIC_API_KEY'", OPENROUTER_API_KEY: "matrix.credentialName == 'OPENROUTER_API_KEY'", + XAI_API_KEY: "matrix.credentialName == 'XAI_API_KEY'", DAYTONA_API_KEY: "matrix.environmentId == 'daytona'", })) { expect(fullStack).toContain( @@ -538,7 +549,7 @@ describe("public repository paid workflow security", () => { ); const providerSecretReferences = [ ...contents.matchAll( - /secrets(?:\.(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|DAYTONA_API_KEY)\b|\[['"](?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|DAYTONA_API_KEY)['"]\])/g, + /secrets(?:\.(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|DAYTONA_API_KEY)\b|\[['"](?:OPENAI_API_KEY|ANTHROPIC_API_KEY|OPENROUTER_API_KEY|XAI_API_KEY|DAYTONA_API_KEY)['"]\])/g, ), ]; if (providerSecretReferences.length > 0) {