diff --git a/doc/plans/2026-10-02-native-completion-live-comparison.json b/doc/plans/2026-10-02-native-completion-live-comparison.json new file mode 100644 index 0000000000..f43cc0e92b --- /dev/null +++ b/doc/plans/2026-10-02-native-completion-live-comparison.json @@ -0,0 +1,114 @@ +{ + "schema": "paperclip.native-completion.safe-comparison/v1", + "asOf": "2026-10-02T19:41:00Z", + "partial": true, + "expectedCellsPerVariant": 6, + "expectedProviderTurnsPerVariant": 6, + "candidateSha": "ae34843731ac338bd806a4106cb447134bff402c", + "candidatePilotSha": "55ce97b675e1ed68cc171fe44b729147209a7e24", + "baselineSha": "8792aed8ac8e9d004406afcb4b9a4ffddccd72c8", + "candidate": [ + { + "executionId": "stock-harness.runner-codex.local.assigned-skill-explicit-invocation", + "sourceSha": "55ce97b675e1ed68cc171fe44b729147209a7e24", + "status": "passed", + "model": "gpt-5.6-sol", + "durationMs": 46812, + "providerDurationMs": 25483, + "cleanup": "passed", + "billing": { + "llm": { + "runCount": 1, + "runsWithTokenUsage": 1, + "runsWithReportedCost": 1, + "inputTokens": 156457, + "outputTokens": 1288, + "cachedInputTokens": 133102, + "totalTokens": 290847, + "reportedCostUsd": 0, + "costStatus": "reported" + }, + "runtime": { + "provider": "local", + "agentRunDurationMs": 25483, + "leaseDurationMs": null, + "leaseCount": 0, + "costStatus": "not_metered", + "costSource": "local_not_metered" + }, + "reportedCostUsd": 0, + "estimatedRuntimeCostUsd": 0, + "observedAndEstimatedCostUsd": 0, + "complete": true + }, + "checks": 7, + "allBehaviorChecksPass": true, + "publicTaskDocuments": 1, + "prerequisites": { + "passed": true, + "sourceSha": "55ce97b675e1ed68cc171fe44b729147209a7e24", + "sourceFingerprint": "ca46783b9d6529260667716a45d47c7fec43b36d6c813549212f6dbd4d2a25d0", + "providerCallsBeforeAdmission": 0, + "tests": 585 + }, + "resultSha256": "3a59532dd61292e551e1de678b5f84df49c01b760ae03ea4a784a039ba0e55d1" + } + ], + "baseline": [], + "pendingCandidate": [ + "stock-harness.runner-codex.local.native-blocked-report", + "stock-harness.runner-acpx-claude.local.assigned-skill-explicit-invocation", + "stock-harness.runner-acpx-claude.local.native-blocked-report", + "stock-harness.runner-opencode.local.assigned-skill-explicit-invocation", + "stock-harness.runner-opencode.local.native-blocked-report" + ], + "pendingBaseline": [ + "stock-harness.runner-codex.local.assigned-skill-explicit-invocation", + "stock-harness.runner-codex.local.native-blocked-report", + "stock-harness.runner-acpx-claude.local.assigned-skill-explicit-invocation", + "stock-harness.runner-acpx-claude.local.native-blocked-report", + "stock-harness.runner-opencode.local.assigned-skill-explicit-invocation", + "stock-harness.runner-opencode.local.native-blocked-report" + ], + "fixtureMatchProof": { + "schema": "paperclip.native-completion.fixture-match-proof/v1", + "cells": [ + { + "executionId": "stock-harness.runner-codex.local.assigned-skill-explicit-invocation", + "fixtureConfigSha256": "8fc61211aefb339a310469695d716c3faea15c2a547c543d70c599857a7e071d", + "model": "gpt-5.6-sol" + }, + { + "executionId": "stock-harness.runner-codex.local.native-blocked-report", + "fixtureConfigSha256": "46c90b7162ad096dcae386cc850e1bd64bcc1c571c0118ddd5988fec45706047", + "model": "gpt-5.6-sol" + }, + { + "executionId": "stock-harness.runner-opencode.local.assigned-skill-explicit-invocation", + "fixtureConfigSha256": "891eb6dd154aa5926a3de2eae79780661b436618b6eac7733998750c6c341ba3", + "model": "openrouter/deepseek/deepseek-v4-flash-0731" + }, + { + "executionId": "stock-harness.runner-opencode.local.native-blocked-report", + "fixtureConfigSha256": "3e093398bf6e775b4367f40d1837f6c96dde675226951beb441a8a57108d1b82", + "model": "openrouter/deepseek/deepseek-v4-flash-0731" + }, + { + "executionId": "stock-harness.runner-acpx-claude.local.assigned-skill-explicit-invocation", + "fixtureConfigSha256": "c190a1e731ef450ea6ae0c24b5a8ae0f21db725b47350645cdbb9273415c5acd", + "model": "claude-sonnet-5" + }, + { + "executionId": "stock-harness.runner-acpx-claude.local.native-blocked-report", + "fixtureConfigSha256": "eb9c1d8579092f13a00b7a3ca24c96449d2897659c0514f7d6f1a4ce6d779d9b", + "model": "claude-sonnet-5" + } + ] + }, + "limitations": [ + "One graded candidate cell, no baseline or blocker results yet.", + "Native reported zero cost has unknown actual provider billing; local runtime is unmetered.", + "Pilot source precedes blocker-only grader correction; its completion fixture and production guidance are unchanged.", + "Single trials do not establish general coding quality, reliability, speed or cost improvement." + ] +} diff --git a/doc/plans/2026-10-02-native-completion-live-comparison.md b/doc/plans/2026-10-02-native-completion-live-comparison.md index 1ddd892e6d..12960cd9b8 100644 --- a/doc/plans/2026-10-02-native-completion-live-comparison.md +++ b/doc/plans/2026-10-02-native-completion-live-comparison.md @@ -1,7 +1,8 @@ # Native completion guidance qualification — 2026-10-02 -**Partial report, updated 19:34 UTC.** No completed native qualification result -is available yet. The selected scope is three native profiles, each with one +**Partial report, updated 19:41 UTC.** One of six candidate results is available: +native Codex completion passes. No baseline or blocker result is available yet. +The selected scope is three native profiles, each with one completion and one concrete blocker case: six cells per variant. Missing cells are pending, not passing. [Draft PR #14961](https://github.com/paperclipai/paperclip/pull/14961) remains stacked on [draft PR #14948](https://github.com/paperclipai/paperclip/pull/14948). @@ -16,7 +17,7 @@ results do not measure this native completion-tool change. | Native profile | Completion baseline | Completion candidate | Blocker baseline | Blocker candidate | | --- | --- | --- | --- | --- | -| Codex | Pending | Running at `55ce97b675e1ed68cc171fe44b729147209a7e24` | Pending | Pending | +| Codex | Pending | Pass at `55ce97b675e1ed68cc171fe44b729147209a7e24` | Pending | Pending | | ACPX Claude | Pending | Pending | Pending | Pending | | OpenCode | Pending | Pending | Pending | Pending | @@ -25,7 +26,21 @@ results do not measure this native completion-tool change. - The protected [candidate completion pilot](https://github.com/paperclipai/paperclip/actions/runs/37053897522) runs from the trusted default-branch workflow against immutable candidate `55ce97b675e1ed68cc171fe44b729147209a7e24`. It selects only native Codex's skill - completion case. Full exact-head prerequisites precede provider admission. + completion case. All 585 exact-head prerequisites pass before provider admission. + All seven independent task/skill checks pass; one Paperclip task document is + saved, public hire/budget receipts pass, and cleanup passes. Provider time is + 25.483 seconds; cell time is 46.812 seconds. Its reported zero cost does not + establish zero actual billing. Result SHA-256: + `3a59532dd61292e551e1de678b5f84df49c01b760ae03ea4a784a039ba0e55d1`. + The completion fixture and production guidance are unchanged by the subsequent + blocker-grader correction. +- The [remaining five candidate cells](https://github.com/paperclipai/paperclip/actions/runs/37055582470) + run at `ae34843731ac338bd806a4106cb447134bff402c` on the frozen + `codex/native-completion-qualified-cells` branch. This separate target avoids + superseding the pilot while its report publication finishes. +- The [corrected six-cell historical comparison](https://github.com/paperclipai/paperclip/actions/runs/37055273989) + runs at `8792aed8ac8e9d004406afcb4b9a4ffddccd72c8` on + `codex/native-completion-previous-guidance`, concurrently with candidate setup. - The first [historical setup](https://github.com/paperclipai/paperclip/actions/runs/37054642871) selected six cells at `9060f7ee4` on `codex/native-completion-previous-guidance`. It was cancelled before provider cells started after review identified a gap @@ -35,9 +50,8 @@ results do not measure this native completion-tool change. - The comparison restores only five native production sources containing tool descriptions and their session fingerprint. The tiny hire manual, reduced shared prompts, and merged Codex base fix #14920 remain constant. Twelve - behavioral/fixture sources were byte-identical before the shared grader fix; - all six profile/model/auth/effort fixture configurations match. Recheck those - hashes after the fix before dispatch. + behavioral/fixture sources are byte-identical between the corrected baseline + and `ae34843731`; all six profile/model/auth/effort fixture configurations match. - Historical structural assertions expect the prior descriptions in actual authenticated/wire/provider catalogs. The historical v13 branch tests its prior compatibility cases; the candidate separately requires v13-to-v14 @@ -62,6 +76,11 @@ grader defect. The initial failed calibration remains retained. Candidate production instructions and independent completion graders were not changed. Fresh CI/review for the corrected fixture and paid outcomes remain pending. -Cost is unknown until retained results arrive; no speed, spending, coding-quality, +At corrected source `ae34843731`, all 589 credential-free prerequisite checks pass +locally (588 TypeScript and one Rust), with zero provider calls. Its receipt is +retained under `tests/runner-e2e/results/stock-harness-preflight-2026-10-02T19-37-13.267Z/`. +The safe [partial evidence projection](2026-10-02-native-completion-live-comparison.json) +records the graded result and explicitly pending cells. Actual cost remains +unknown; no speed, spending, coding-quality, or broad reliability improvement is claimed. Native finish/block documentation must remain separate from legacy Paperclip skill/API completion guidance.