Files
PaperClipAI/.github/workflows/runner-live-evals.yml
dependabot[bot] 5c15be3715 build(deps): bump actions/cache from 5.1.0 to 6.1.0 (#12962)
Bumps [actions/cache](https://github.com/actions/cache) from 5.1.0 to
6.1.0.
<details>
<summary>Release notes</summary>
<p><em>Sourced from <a
href="https://github.com/actions/cache/releases">actions/cache's
releases</a>.</em></p>
<blockquote>
<h2>v6.1.0</h2>
<h2>What's Changed</h2>
<ul>
<li>Bump <code>@​actions/cache</code> to v6.1.0 - handle read-only cache
access by <a
href="https://github.com/jasongin"><code>@​jasongin</code></a> in <a
href="https://redirect.github.com/actions/cache/pull/1768">actions/cache#1768</a></li>
</ul>
<p><strong>Full Changelog</strong>: <a
href="https://github.com/actions/cache/compare/v6...v6.1.0">https://github.com/actions/cache/compare/v6...v6.1.0</a></p>
<h2>v6.0.0</h2>
<h2>What's Changed</h2>
<ul>
<li>Update packages, migrate to ESM by <a
href="https://github.com/Samirat"><code>@​Samirat</code></a> in <a
href="https://redirect.github.com/actions/cache/pull/1760">actions/cache#1760</a></li>
</ul>
<p><strong>Full Changelog</strong>: <a
href="https://github.com/actions/cache/compare/v5...v6.0.0">https://github.com/actions/cache/compare/v5...v6.0.0</a></p>
</blockquote>
</details>
<details>
<summary>Commits</summary>
<ul>
<li><a
href="https://github.com/actions/cache/commit/55cc8345863c7cc4c66a329aec7e433d2d1c52a9"><code>55cc834</code></a>
Merge pull request <a
href="https://redirect.github.com/actions/cache/issues/1768">#1768</a>
from jasongin/readonly-cache</li>
<li><a
href="https://github.com/actions/cache/commit/d8cd72f230726cdf4457ebb61ec1b593a8d12337"><code>d8cd72f</code></a>
Bump <code>@​actions/cache</code> to v6.1.0 - handle cache write error
due to RO token</li>
<li><a
href="https://github.com/actions/cache/commit/2c8a9bd7457de244a408f35966fab2fb45fda9c8"><code>2c8a9bd</code></a>
Merge pull request <a
href="https://redirect.github.com/actions/cache/issues/1760">#1760</a>
from actions/samirat/esm_migration_and_package_update</li>
<li><a
href="https://github.com/actions/cache/commit/e9b91fdc3fea7d79165fceb79042ef45c2d51023"><code>e9b91fd</code></a>
Prettier fixes</li>
<li><a
href="https://github.com/actions/cache/commit/e4884b8ff7f92ef6b52c79eda480bbc86e685adb"><code>e4884b8</code></a>
Rebuild dist</li>
<li><a
href="https://github.com/actions/cache/commit/10baf0191a3c426ea0fa4a3253a5c04233b6e18f"><code>10baf01</code></a>
Fixed licenses</li>
<li><a
href="https://github.com/actions/cache/commit/e39b386c9004d72a15d864ade8c0b3a702d47a37"><code>e39b386</code></a>
Fix test mock return order</li>
<li><a
href="https://github.com/actions/cache/commit/b6928203372a8571ff984c0c883ef3a1adfb0c06"><code>b692820</code></a>
PR feedback</li>
<li><a
href="https://github.com/actions/cache/commit/60749128a44d25d3c520a489e576380cf00ff3f1"><code>6074912</code></a>
Rebuild dist bundles as ESM to match type:module</li>
<li><a
href="https://github.com/actions/cache/commit/5a912e8b4af820fa082a0e75cfd2c782f8fbfe0e"><code>5a912e8</code></a>
Fix lint and jest issues</li>
<li>Additional commits viewable in <a
href="https://github.com/actions/cache/compare/v5.1.0...v6.1.0">compare
view</a></li>
</ul>
</details>
<br />

Signed-off-by: dependabot[bot] <support@github.com>
Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com>
2026-10-06 15:36:53 -07:00

193 lines
7.7 KiB
YAML

name: Runner Live Evals
on:
schedule:
- cron: "17 6 * * 0"
workflow_dispatch:
inputs:
candidate:
description: "Comma-separated live candidate IDs (for example codex-luna)"
type: string
required: false
case:
description: "Comma-separated workflow case IDs"
type: string
required: false
limit:
description: "Maximum executions after candidate/case filtering"
type: string
required: false
concurrency:
group: runner-live-evals-${{ github.ref }}
cancel-in-progress: true
jobs:
authorize:
name: Authorize paid campaign
if: github.event_name != 'schedule' || vars.RUNNER_LIVE_EVALS_NIGHTLY_ENABLED == 'true'
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
outputs:
eval_runner: ${{ steps.runner.outputs.runner }}
steps:
- name: Require default branch and allowlisted numeric actor IDs
env:
GH_TOKEN: ${{ github.token }}
REF: ${{ github.ref }}
DEFAULT_BRANCH: ${{ github.event.repository.default_branch }}
ACTOR: ${{ github.actor }}
ACTOR_ID: ${{ github.actor_id }}
TRIGGERING_ACTOR: ${{ github.triggering_actor }}
ALLOWED_ACTOR_IDS: ${{ vars.RUNNER_E2E_ALLOWED_ACTOR_IDS }}
run: |
set -euo pipefail
if [ "$REF" != "refs/heads/$DEFAULT_BRANCH" ]; then
echo "Paid runner live evals may run only from the default branch." >&2
exit 1
fi
if ! jq -e 'type == "array" and length > 0 and all(.[]; type == "number" and . > 0 and floor == .)' <<< "${ALLOWED_ACTOR_IDS:-}" >/dev/null; then
echo "RUNNER_E2E_ALLOWED_ACTOR_IDS must be a non-empty JSON array of numeric GitHub user IDs." >&2
exit 1
fi
triggering_actor_id="$(gh api "users/$TRIGGERING_ACTOR" --jq .id)"
if [ "$triggering_actor_id" != "$ACTOR_ID" ] && [ "$TRIGGERING_ACTOR" = "$ACTOR" ]; then
echo "GitHub actor identity contexts disagree; refusing the paid run." >&2
exit 1
fi
candidates=("$triggering_actor_id" "$ACTOR_ID")
for candidate in "${candidates[@]}"; do
if ! jq -e --argjson candidate "$candidate" 'index($candidate) != null' <<< "$ALLOWED_ACTOR_IDS" >/dev/null; then
echo "The initiating GitHub account is not authorized to run paid runner live evals." >&2
exit 1
fi
done
- name: Select paid eval runner
id: runner
env:
AWS_PAID_RUNNER_ENABLED: ${{ vars.RUNNER_E2E_AWS_ENABLED }}
run: |
set -euo pipefail
github_runner='ubuntu-latest'
aws_runner='runs-on/fleet=paperclip-public-pr-x64/env=public-ci'
if [ "$AWS_PAID_RUNNER_ENABLED" = true ]; then
echo "runner=$aws_runner" >> "$GITHUB_OUTPUT"
echo '::notice title=Paid eval routing::Using an ephemeral RunsOn Fleet runner'
else
echo "runner=$github_runner" >> "$GITHUB_OUTPUT"
echo '::notice title=Paid eval routing::RUNNER_E2E_AWS_ENABLED is not true; using the proven GitHub-hosted runner'
fi
live_matrix:
name: Balanced provider/model matrix
needs: authorize
if: github.event_name != 'schedule' || vars.RUNNER_LIVE_EVALS_NIGHTLY_ENABLED == 'true'
# The authorize job selects one of two literal, reviewed runner labels;
# dispatch inputs and repository variables cannot inject an arbitrary label.
runs-on: ${{ needs.authorize.outputs.eval_runner }}
timeout-minutes: 180
permissions:
contents: read
environment:
name: runner-e2e-paid
steps:
- name: Reauthorize paid execution before provider access
env:
GH_TOKEN: ${{ github.token }}
REF: ${{ github.ref }}
DEFAULT_BRANCH: ${{ github.event.repository.default_branch }}
ACTOR_ID: ${{ github.actor_id }}
TRIGGERING_ACTOR: ${{ github.triggering_actor }}
ALLOWED_ACTOR_IDS: ${{ vars.RUNNER_E2E_ALLOWED_ACTOR_IDS }}
run: |
set -euo pipefail
test "$REF" = "refs/heads/$DEFAULT_BRANCH"
jq -e 'type == "array" and length > 0 and all(.[]; type == "number" and . > 0 and floor == .)' <<< "${ALLOWED_ACTOR_IDS:-}" >/dev/null
triggering_actor_id="$(gh api "users/$TRIGGERING_ACTOR" --jq .id)"
jq -e --argjson candidate "$triggering_actor_id" 'index($candidate) != null' <<< "$ALLOWED_ACTOR_IDS" >/dev/null
jq -e --argjson candidate "$ACTOR_ID" 'index($candidate) != null' <<< "$ALLOWED_ACTOR_IDS" >/dev/null
- name: Checkout repository
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
- name: Generate private eval-repository token
id: evals_token
env:
COMMITPERCLIP_KEY: ${{ secrets.COMMITPERCLIP_KEY }}
GH_REPO: ${{ github.repository }}
run: |
set -euo pipefail
token="$(node .github/scripts/get-bot-token.mjs)"
echo "::add-mask::$token"
echo "value=$token" >> "$GITHUB_OUTPUT"
- name: Checkout canonical Evalbook reporter
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
with:
repository: paperclipai/paperclip-evals
ref: 0c8dfea0ef71a73e909b59a5c0484554cbee199b
path: .paperclip-evals
token: ${{ steps.evals_token.outputs.value }}
persist-credentials: false
- name: Setup pnpm
uses: pnpm/action-setup@0977fd99725f1db4007ccb2928dbb4e90d06cc86 # v6
with:
version: 9.15.4
- name: Setup Node.js
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7
with:
node-version: 24
- name: Install dependencies
run: pnpm install --frozen-lockfile
- name: Restore compatible weekly baseline
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
with:
path: packages/paperclip-runner/.paperclip-local/evals/workflows/history
key: runner-live-eval-history-${{ github.ref_name }}-${{ github.run_id }}
restore-keys: |
runner-live-eval-history-${{ github.ref_name }}-
- name: Build provider-neutral eval kernel
run: pnpm --filter @paperclipai/paperclip-eval-kernel build
- name: Run trend-only live matrix
env:
OPENROUTER_API_KEY: ${{ secrets.OPENROUTER_API_KEY }}
ANTHROPIC_API_KEY: ${{ secrets.ANTHROPIC_API_KEY }}
OPENAI_API_KEY: ${{ secrets.OPENAI_API_KEY }}
PAPERCLIP_EVAL_BASELINE_READY: "true"
PAPERCLIP_EVAL_RUNNER_BUILD: ${{ github.sha }}
PAPERCLIP_EVAL_MAX_CAMPAIGN_COST_USD: "12"
PAPERCLIP_EVAL_SCHEDULE_SEED: runner-live-seven-week-v1
PAPERCLIP_EVAL_CANDIDATE: ${{ inputs.candidate }}
PAPERCLIP_EVAL_CASE: ${{ inputs.case }}
PAPERCLIP_EVAL_LIMIT: ${{ inputs.limit }}
PAPERCLIP_EVALBOOK_PROGRAM: ${{ github.workspace }}/.paperclip-evals/evals/paperclip-runner/tools/eval_program.py
run: pnpm --filter @paperclipai/paperclip-runner report:runner-live-evals
- name: Publish job summary
if: always()
run: |
summary=packages/paperclip-runner/.paperclip-local/evals/workflows/github-live-summary.md
if [ -f "$summary" ]; then
cat "$summary" >> "$GITHUB_STEP_SUMMARY"
fi
- name: Upload safe live eval bundle
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7
with:
name: runner-live-evals-${{ github.run_id }}
path: packages/paperclip-runner/.paperclip-local/evals/workflows/
retention-days: 30
if-no-files-found: error