mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-06 10:48:12 +02:00
feat(runner): add offline evaluation tooling (#12653)
## Thinking Path > - Paperclip is the open source app people use to manage AI agents for work. > - Paperclip Runner needs repeatable evaluation contracts. > - Evaluation code must stay separate from provider launch and production orchestration. > - Offline fixtures need stable compatibility, scoring, traceability, and report rules. > - Published Runner consumers need only the supported evaluation contract surface. > - This pull request adds offline evaluation tooling and a workspace-private matrix kernel. > - The benefit is deterministic evaluation without credentials or paid provider calls. ## Linked Issues or Issue Description Refs #11297 This pull request extracts the offline evaluation unit from the earlier aggregate Runner work. ## What Changed - Add a workspace-private, provider-neutral evaluation matrix kernel. - Add the public `@paperclipai/paperclip-runner/evals` compatibility and native execution contracts. - Add fail-closed runnerd artifact and protocol compatibility checks. - Add deterministic workflow catalogs, scoring, traceability, and report generation. - Add sanitized Codex, OpenCode, and ACPX fixtures. - Add package-boundary and clean-consumer checks. - Add the eval package manifest to the Docker dependency stage. - Add the generated protocol fixture digest without changing the lockfile. ## Verification GitHub Actions must run: - Runner TypeScript and Rust type checks. - Runner unit and protocol tests. - Evaluation kernel tests. - Workflow traceability checks. - Clean-consumer and package-boundary checks. - Repository test, type-check, build, policy, and security gates. No local test command was run. The repository owner requested GitHub-only verification. ## Risks This is a large greenfield review surface with 51 files. The code does not launch a live provider or load credentials. Package and protocol drift fail closed. The workspace lockfile remains under the existing CI-owned process. ## Model Used OpenAI Codex with the GPT-5 agent model. The work used high reasoning, repository inspection, tool use, and parallel code review. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [ ] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [ ] All Paperclip CI gates are green - [ ] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge
This commit is contained in:
1 parent
1ed29abaa6
commit
5458940a6e
51 files changed
+6544
-27
No files matched your search
@@ -0,0 +1,14 @@
|
||||
# Paperclip Eval Kernel
|
||||
|
||||
`@paperclipai/paperclip-eval-kernel` is the workspace-private, provider-neutral
|
||||
matrix orchestrator owned by Paperclip Evals. It contains no Paperclip scenario
|
||||
corpus, provider configuration, product fixture, scorer, or report template.
|
||||
|
||||
Consumers pass scenario and candidate values plus execution and scoring
|
||||
callbacks. Candidate `preflight` hooks should call the runner package's
|
||||
`assertPaperclipRunnerCompatibility` before any provider work starts. This keeps
|
||||
catalog, protocol, runner-client, control-plane-adapter, testkit, corpus, and
|
||||
provider-operation incompatibilities explicit.
|
||||
|
||||
Paperclip App may consume this package only as a development dependency for CI
|
||||
or parity tests. `@paperclipai/paperclip-runner` has no runtime dependency on it.
|
||||
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"name": "@paperclipai/paperclip-eval-kernel",
|
||||
"version": "0.1.0",
|
||||
"private": true,
|
||||
"description": "Provider-neutral evaluation matrix kernel without Paperclip scenario content",
|
||||
"type": "module",
|
||||
"files": [
|
||||
"dist",
|
||||
"README.md"
|
||||
],
|
||||
"exports": {
|
||||
".": {
|
||||
"types": "./dist/index.d.ts",
|
||||
"import": "./dist/index.js"
|
||||
}
|
||||
},
|
||||
"scripts": {
|
||||
"build": "tsc -p tsconfig.json",
|
||||
"typecheck": "tsc -p tsconfig.json --noEmit",
|
||||
"test": "pnpm run build && node --test test/*.test.mjs"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=24.11.0"
|
||||
},
|
||||
"license": "MIT",
|
||||
"devDependencies": {
|
||||
"@types/node": "^24.0.0",
|
||||
"typescript": "^7.0.2"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
export const PAPERCLIP_EVAL_KERNEL_COMPATIBILITY = Object.freeze({
|
||||
schema: "paperclip.eval-kernel.compatibility.v1" as const,
|
||||
packageName: "@paperclipai/paperclip-eval-kernel" as const,
|
||||
packageVersion: "0.1.0" as const,
|
||||
apiVersion: 1 as const,
|
||||
});
|
||||
|
||||
export interface PaperclipEvalScenario<TInput = unknown> {
|
||||
readonly id: string;
|
||||
readonly input: TInput;
|
||||
}
|
||||
|
||||
export interface PaperclipEvalCandidate<TCandidate = unknown> {
|
||||
readonly id: string;
|
||||
readonly config: TCandidate;
|
||||
/** Fail-closed runner/catalog/provider compatibility check. */
|
||||
readonly preflight?: () => void | Promise<void>;
|
||||
}
|
||||
|
||||
export interface PaperclipEvalResult<TOutput = unknown, TScore = unknown> {
|
||||
readonly scenarioId: string;
|
||||
readonly candidateId: string;
|
||||
readonly output: TOutput;
|
||||
readonly score: TScore;
|
||||
}
|
||||
|
||||
export class PaperclipEvalKernelConfigurationError extends Error {
|
||||
readonly code = "paperclip_eval_kernel_configuration_invalid" as const;
|
||||
|
||||
constructor(message: string) {
|
||||
super(message);
|
||||
this.name = "PaperclipEvalKernelConfigurationError";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Generic deterministic matrix orchestration. Scenario definitions, provider
|
||||
* configuration, scorers, reports, and persistence remain caller-owned.
|
||||
*/
|
||||
export async function runPaperclipEvalMatrix<
|
||||
TInput,
|
||||
TCandidate,
|
||||
TOutput,
|
||||
TScore,
|
||||
>(input: {
|
||||
readonly scenarios: readonly PaperclipEvalScenario<TInput>[];
|
||||
readonly candidates: readonly PaperclipEvalCandidate<TCandidate>[];
|
||||
readonly execute: (context: {
|
||||
readonly scenario: PaperclipEvalScenario<TInput>;
|
||||
readonly candidate: PaperclipEvalCandidate<TCandidate>;
|
||||
}) => Promise<TOutput>;
|
||||
readonly score: (context: {
|
||||
readonly scenario: PaperclipEvalScenario<TInput>;
|
||||
readonly candidate: PaperclipEvalCandidate<TCandidate>;
|
||||
readonly output: TOutput;
|
||||
}) => Promise<TScore> | TScore;
|
||||
}): Promise<readonly PaperclipEvalResult<TOutput, TScore>[]> {
|
||||
assertUniqueNonEmptyIds("scenario", input.scenarios);
|
||||
assertUniqueNonEmptyIds("candidate", input.candidates);
|
||||
|
||||
for (const candidate of input.candidates) {
|
||||
await candidate.preflight?.();
|
||||
}
|
||||
|
||||
const results: PaperclipEvalResult<TOutput, TScore>[] = [];
|
||||
for (const scenario of input.scenarios) {
|
||||
for (const candidate of input.candidates) {
|
||||
const output = await input.execute({ scenario, candidate });
|
||||
const score = await input.score({ scenario, candidate, output });
|
||||
results.push(Object.freeze({
|
||||
scenarioId: scenario.id,
|
||||
candidateId: candidate.id,
|
||||
output,
|
||||
score,
|
||||
}));
|
||||
}
|
||||
}
|
||||
return Object.freeze(results);
|
||||
}
|
||||
|
||||
function assertUniqueNonEmptyIds(
|
||||
kind: "scenario" | "candidate",
|
||||
values: readonly { readonly id: string }[],
|
||||
): void {
|
||||
if (values.length === 0) {
|
||||
throw new PaperclipEvalKernelConfigurationError(`${kind} list must not be empty`);
|
||||
}
|
||||
const ids = new Set<string>();
|
||||
for (const value of values) {
|
||||
if (value.id.trim().length === 0) {
|
||||
throw new PaperclipEvalKernelConfigurationError(`${kind} id must not be empty`);
|
||||
}
|
||||
if (ids.has(value.id)) {
|
||||
throw new PaperclipEvalKernelConfigurationError(`duplicate ${kind} id: ${value.id}`);
|
||||
}
|
||||
ids.add(value.id);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
import assert from "node:assert/strict";
|
||||
import test from "node:test";
|
||||
|
||||
import {
|
||||
PAPERCLIP_EVAL_KERNEL_COMPATIBILITY,
|
||||
PaperclipEvalKernelConfigurationError,
|
||||
runPaperclipEvalMatrix,
|
||||
} from "../dist/index.js";
|
||||
|
||||
test("runs a caller-owned scenario/candidate matrix", async () => {
|
||||
const results = await runPaperclipEvalMatrix({
|
||||
scenarios: [{ id: "scenario-a", input: { value: 2 } }],
|
||||
candidates: [{ id: "candidate-a", config: { multiplier: 3 } }],
|
||||
execute: async ({ scenario, candidate }) => scenario.input.value * candidate.config.multiplier,
|
||||
score: ({ output }) => ({ passed: output === 6 }),
|
||||
});
|
||||
assert.equal(PAPERCLIP_EVAL_KERNEL_COMPATIBILITY.apiVersion, 1);
|
||||
assert.deepEqual(results, [{
|
||||
scenarioId: "scenario-a",
|
||||
candidateId: "candidate-a",
|
||||
output: 6,
|
||||
score: { passed: true },
|
||||
}]);
|
||||
});
|
||||
|
||||
test("fails before execution when compatibility preflight fails", async () => {
|
||||
let executed = false;
|
||||
await assert.rejects(
|
||||
runPaperclipEvalMatrix({
|
||||
scenarios: [{ id: "scenario-a", input: null }],
|
||||
candidates: [{
|
||||
id: "candidate-a",
|
||||
config: null,
|
||||
preflight: () => { throw new Error("paperclip_runner_incompatible"); },
|
||||
}],
|
||||
execute: async () => { executed = true; },
|
||||
score: () => null,
|
||||
}),
|
||||
/paperclip_runner_incompatible/,
|
||||
);
|
||||
assert.equal(executed, false);
|
||||
});
|
||||
|
||||
test("rejects duplicate scenario ids", async () => {
|
||||
await assert.rejects(
|
||||
runPaperclipEvalMatrix({
|
||||
scenarios: [{ id: "duplicate", input: 1 }, { id: "duplicate", input: 2 }],
|
||||
candidates: [{ id: "candidate-a", config: null }],
|
||||
execute: async () => null,
|
||||
score: () => null,
|
||||
}),
|
||||
PaperclipEvalKernelConfigurationError,
|
||||
);
|
||||
});
|
||||
@@ -0,0 +1,8 @@
|
||||
{
|
||||
"extends": "../../tsconfig.base.json",
|
||||
"compilerOptions": {
|
||||
"outDir": "dist",
|
||||
"rootDir": "src"
|
||||
},
|
||||
"include": ["src/**/*.ts"]
|
||||
}
|
||||
Reference in new issue
Block a user