diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000000..b02d5aad4b --- /dev/null +++ b/.gitattributes @@ -0,0 +1 @@ +patches/*.patch whitespace=-space-before-tab diff --git a/.gitignore b/.gitignore index 26ee72daee..5abd1374e1 100644 --- a/.gitignore +++ b/.gitignore @@ -3,6 +3,7 @@ node_modules/ **/node_modules **/node_modules/ dist/ +packages/paperclip-runner/runner/target/ ui/storybook-static/ .env *.tsbuildinfo diff --git a/packages/paperclip-runner/.gitignore b/packages/paperclip-runner/.gitignore index 31faa12bf8..a8250662e8 100644 --- a/packages/paperclip-runner/.gitignore +++ b/packages/paperclip-runner/.gitignore @@ -1 +1,13 @@ /runner/target/ +/dist-browser/ +/dist-sdk/ +/playwright-report/ +/test-results/ +/dist-standalone/ +/dist-scenarios/ +/dist-issue-thread/ + +# The repository root ignores scratch `check-*.mjs` files and only re-includes +# them under the root `scripts/` directory. This package's checks are real +# source, so re-include them here instead of relying on `git add -f`. +!scripts/check-*.mjs diff --git a/packages/paperclip-runner/README.md b/packages/paperclip-runner/README.md index e01fb84597..36ba1c9c5d 100644 --- a/packages/paperclip-runner/README.md +++ b/packages/paperclip-runner/README.md @@ -1,72 +1,45 @@ -# Paperclip Runner +# Paperclip Native Runner -This private workspace package contains the staged Paperclip Runner work. +This package is the standalone development boundary for Paperclip's native +runner protocol, process supervision, durable transport, provider drivers, and +normalized session backends. Rust owns the production runner under `runner/`; +TypeScript provides the control-plane reference, browser SDK, scenario tools, +and conformance oracle. -The package currently exposes the language-neutral PRP v1 TypeScript -contract, provider-neutral structured questions and responses, deterministic -fixture validation/replay, structured-result normalization, and the session -reducer oracle. It also contains a package-local Rust runner, scripted fake -harness, bounded process supervisor, cross-language replay oracle, and durable -PRP transport. The transport authenticates and encrypts loopback WebSocket -sessions, persists an ACK-driven outbox and command journal, and reconnects with -a short-lived lease. The Rust runner now includes a Codex-only app-server -provider bridge with durable thread resume, cancellation, structured questions, -and provider-neutral event normalization. The root surface now also exposes an -authenticated durable PRP authority for server-side use. It stores only -bootstrap and reconnect credential digests, validates immutable run identity on -every connection and event, and persists commands and cumulative event ACK -state across server restarts. -The package also publishes the canonical semantic action declarations and -their input and output schemas. Its package-local dispatcher projects only -bound, run-authorized actions and emits redacted semantic receipts. +The package includes one coherent set of capabilities: PRP v1 validation and +replay, a supervised local runner with a scripted fake harness, durable +WebSocket delivery and recovery, a skillless Codex app-server driver, live +session and issue-thread surfaces, a public browser/React SDK, a standalone +adapter demo, and a deterministic mock control plane. None of these surfaces +imports or starts Paperclip's server, UI, CLI, or production database. -The first and only provider selected directly by runnerd remains Codex. The -package also contains the qualified OpenCode 1.18.17 proxy and the bounded ACPX -sidecar for Codex and Claude, but this slice does not authorize the server -to select those additional providers for a fresh native run. Dynamic semantic -tools remain undiscoverable unless the hidden server coordinator projects one -of the five same-task read bindings for an already persisted native Codex run. -Catalog membership alone does not grant authority. The server can now create -and start a Codex-backed native run only through the default-off -`paperclip_runner` adapter. See -[`SEMANTIC_ACTIONS.md`](SEMANTIC_ACTIONS.md) for the catalog boundary. +## Public package surfaces -The package has two initial public surfaces: +- `@paperclipai/paperclip-runner` — production contracts, clients/backends, + PRP validation/replay, canonical catalog/dispatcher, and compatibility check. +- `@paperclipai/paperclip-runner/testing` — deterministic mocks plus PRP and + semantic conformance kits. Tests and external conformance consumers import + this explicitly. -- `@paperclipai/paperclip-runner` contains runtime contracts, validation, - replay/reducer logic, the semantic catalog, the authorization dispatcher, and - the Node-only durable server authority. -- `@paperclipai/paperclip-runner/testing` adds Node-only fixture loading and a - provider-neutral semantic conformance kit for deterministic test adapters. +The package root has no mock or scenario exports, and the workspace has no +separate eval-package importer. Provider-backed eval campaigns remain deferred. +See [ADR 0001](docs/adr/0001-runner-testing-eval-package-boundaries.md). -No SDK, browser, React, eval, live-console, or lab entry point is exported. The -package remains private in this wave. The server route -at `/api/runner/v1/connect/:runId` has no authority until the hidden coordinator -registers an exact existing run binding. Fresh native starts are rejected -unless the instance `enableNativeRunner` flag is enabled. Existing direct -adapters keep their original execution path. +The two conformance surfaces intentionally prove different contracts. The +existing `runControlPlanePortConformance` suite checks narrow PRP run/event +persistence. `CAPABILITY_HIGH_RISK_SEMANTIC_VECTORS` and +`runSemanticConformanceKit` compare normalized tool authorization, state, +effects, audit, retries, conflicts, redaction, continuation, and terminal +decisions. The production adapter stays App-owned and invokes Paperclip's real +route/service authorities; it does not copy those rules into this package. -The package build compiles the release `paperclip-runnerd` executable and -stages it under `dist/bin`. The normal server build vendors that directory, so -an installed server does not depend on a separate system Rust installation or -a manually copied binary. `pnpm-lock.yaml` remains under the repository's -existing lockfile process. - -The `paperclip-runner-opencode-proxy` binary adapts the pinned OpenCode server -protocol to the normalized harness contract. It admits only a bounded provider -environment, uses an authenticated loopback MCP endpoint, validates dynamic -tool inputs, keeps terminal completion tools private to the runner boundary, -and preserves exact provider session identity for recovery. The backend factory -requires an explicit runtime directory before constructing this provider. +## Quick start The package also builds `paperclip-runner-acpx-sidecar`. This bounded v2 -stdin/stdout bridge admits the closed Claude and Codex ACPX profiles. It -validates each agent's exact package/model pair, session identity, tool catalog, -structured input, and terminal settlement at the process boundary. Claude runs -without ambient project or local settings. Pi remains unavailable until its -separately spawned runtime can use the same descriptor-confined verified launch -boundary as the ACP server. Runnerd and the server do not select this sidecar -in this slice. +stdin/stdout bridge admits the qualified Codex ACPX profile only. It validates +the exact model, session identity, tool catalog, structured input, and terminal +settlement at the process boundary. Runnerd and the server do not select this +sidecar in this slice. Other ACPX agents remain unavailable. The Rust core includes a bounded client for this sidecar protocol. It enforces request identity, event order, frame and queue limits, timeouts, redacted @@ -103,7 +76,7 @@ unresolved turn-scoped requests. This reducer still does not select ACPX in runnerd. The package-local session bootstrap starts the bounded sidecar transport, -verifies the selected qualified capability handshake and effective model, opens one +verifies the Codex-only capability handshake and effective model, opens one identity-bound session, and confirms its run attachment. Any failed bootstrap terminates the process; session shutdown preserves persistent provider state. The session can then start one immutable-workspace turn, request interruption, @@ -121,8 +94,9 @@ against the exact persisted question IDs, answer modes, options, required answers, custom-answer policy, and text constraints before provider delivery. Tool results and structured question responses then use two-phase resolution: validate retained identity and schema, require the exact sidecar -acknowledgement, and only then clear pending local state. Permission events that -escape the pinned sidecar policy terminate the session fail closed. +acknowledgement, and only then clear pending local state. Codex permission +requests violate its pinned sidecar policy and terminate the session fail +closed. Safe suspension is available only with no active turn or pending request. The sidecar must return the exact persistent session identity before runnerd terminates the local process. @@ -138,37 +112,135 @@ prospective session configuration exactly. Run the complete contract gate with: ```sh -pnpm --filter @paperclipai/paperclip-runner check:protocol +pnpm install --filter @paperclipai/paperclip-runner --lockfile=false --offline --ignore-scripts --dev +pnpm --filter @paperclipai/paperclip-runner verify ``` -Run the Rust runner gate with: +The verification command requires a stable Rust toolchain with `cargo` on +`PATH`, in addition to Node.js 24.11+ and pnpm 9+. + +Minimal Debian/Ubuntu hosts without root access can extract the required +Playwright browser libraries into a user-owned cache and run the same acceptance +sequence with: ```sh -pnpm --filter @paperclipai/paperclip-runner check:runner +pnpm --filter @paperclipai/paperclip-runner verify:rootless ``` -This command checks Rust formatting, builds and tests the minimal workspace in -release mode, verifies bounded process cleanup, launches the real -`paperclip-runnerd` binary through the fake harness, and compares the Rust -conformance and replay summaries with the shared fixtures. The checked-in Cargo -lock and pinned Rust toolchain keep this verification reproducible. +The tracer's final line is stable: -Durability and failure semantics are documented in -[`runner/DURABLE_TRANSPORT.md`](runner/DURABLE_TRANSPORT.md). The fault suite -drops a connection before its event ACK, reconnects with the bound lease, -replays the same event, and proves the duplicated command effect ran once. -Codex launch, resume, cancellation, and normalization behavior is documented in -[`runner/CODEX_PROVIDER.md`](runner/CODEX_PROVIDER.md). +```json +{"schemaVersion":"paperclip.runner.conformance.output.v1","runIdentity":{"runId":"run_conformance_0001","sessionId":"session_conformance_0001"},"result":{"status":"succeeded","summary":"Standalone Conformance fixture accepted."}} +``` -Use `generate:protocol-manifest` after a schema or fixture change, -`generate:protocol-types` after a schema change, and -`generate:replay-goldens` after an intentional reducer change. Commit generated -outputs with their sources; do not edit them by hand. +Run only the tracer with: -Use `generate:semantic-action-catalog` after changing a semantic action -declaration. Its checked-in JSON inventory must land with the source change. +```sh +pnpm --filter @paperclipai/paperclip-runner trace:conformance +``` -The gate compiles every schema with AJV 2020-12, validates accepted fixtures, -rejects unsupported required versions, checks generated TypeScript schema -drift, runs the TypeScript contract tests, and compares reducer snapshots and -parity summaries byte-for-byte with their checked-in golden files. +Replay the Replay happy path, run a Local session, or open the browser +devtool: + +```sh +pnpm --filter @paperclipai/paperclip-runner replay:fixture +pnpm --filter @paperclipai/paperclip-runner trace:local-runner -- --scenario happy-path +pnpm --filter @paperclipai/paperclip-runner trace:codex +pnpm --filter @paperclipai/paperclip-runner demo:live-console -- --host 127.0.0.1 --port 4174 + +# Live console: chat with a live session in the browser. +pnpm --filter @paperclipai/paperclip-runner console:live-console +pnpm --filter @paperclipai/paperclip-runner browser:dev --host 127.0.0.1 --port 4179 + +# SDK: open the public-SDK reference console and mini consumer. +pnpm --filter @paperclipai/paperclip-runner console:sdk + +# Standalone: run the standalone legacy/native/kill-switch tracer and page. +pnpm --filter @paperclipai/paperclip-runner trace:standalone +pnpm --filter @paperclipai/paperclip-runner trace:standalone -- --feature-flag enabled +pnpm --filter @paperclipai/paperclip-runner trace:standalone -- --feature-flag enabled --kill-switch enabled +pnpm --filter @paperclipai/paperclip-runner demo:standalone + +``` + +Live console provider-backed routes are loopback-only and reject wildcard/LAN +binds. Browser mutations require same-origin Fetch Metadata, matching Origin, +and JSON content; see the protocol-server tutorial for direct `curl` examples. + +## Package-owned commands + +| Command | Purpose | +|---|---| +| `build` | Compile the TypeScript public surface, Rust workspace, and browser devtool. | +| `typecheck` | Check TypeScript, Rust, generated schema sources, and browser types. | +| `test` | Run Rust/TypeScript fixture, supervisor, fake-driver, live/replay, and boundary tests. | +| `check:forbidden-imports` | Reject TypeScript imports and Cargo path dependencies that cross into Paperclip core. | +| `check:tracked-imports` | Reject tracked imports and `package.json` entry points that only resolve against untracked files, so a clean checkout of any commit builds. | +| `check:numbered-milestones` | Reject numbered construction-milestone names in tracked package paths and source. | +| `check:package-boundaries` | Enforce the acyclic runtime/testing dependency and manifest boundary. | +| `check:clean-consumers` | Pack the runner and install its root and testing exports in a clean consumer. | +| `check:conformance-parity` | Require byte-for-byte equivalent Rust and TypeScript tracer output. | +| `check:replay-goldens` | Require all reducer snapshots and cross-language summaries to match checked goldens. | +| `check:replay-parity` | Run TypeScript and Rust against the same Replay fixture summaries. | +| `check:browser-tokens` | Reject component-local visual literals and require the standalone token layer. | +| `docs:validate` | Validate local documentation links. | +| `trace:conformance` | Run the Rust mock-core tracer, print the stable result, and exit. | +| `trace:conformance:typescript` | Run the TypeScript reference tracer directly. | +| `replay:fixture` | Validate and reduce a fixture to a final snapshot. | +| `trace:local-runner` | Run one native local session through the Rust runner and fake harness. | +| `trace:codex` | Run the mock core with a real, local skillless Codex app-server session. | +| `demo:live-console` | Start the package-local HTTP/SSE server with server-only Codex authentication. | +| `console:live-console` | Start the standalone browser devtool with the Live console on `127.0.0.1:4180`. | +| `console:sdk` | Start the public-SDK reference console and mini consumer on `127.0.0.1:4181`. | +| `test:sdk` | Run targeted browser-client, reducer-projection, and React component contract tests. | +| `test:browser:sdk` | Exercise both consumers with the fake driver, keyboard/a11y checks, reconnect/replay, measurements, and screenshots. | +| `record:sdk:codex` | Run both public consumers against a safe real Codex session and capture live screenshots. | +| `check:capability-contract` | Verify the generated capability, legacy MCP, and eval traceability contract. | +| `check:semantic-contracts` | Verify the provider-neutral semantic tool contract is current. | +| `trace:live-runner` | Run the real runnerd/Codex semantic loop against the mock control plane. | +| `demo:scenarios` | Start the Capability scenario explorer over the mock control plane on `127.0.0.1:4183`. | +| `console:issue-thread` | Start the Paperclip-style issue thread on `127.0.0.1:4184`. | +| `test:scenarios` | Run the scenario index, run-artifact, parity, explorer component, and route tests. | +| `test:browser:scenarios` | Exercise both the scenario explorer and issue-thread browser contracts. | +| `browser:dev` | Start the standalone live/replay browser devtool. | +| `test:browser` | Exercise static replay and live scenarios, then capture temporary screenshots under ignored test output. | +| `verify` | Run the complete deterministic Conformance through SDK acceptance sequence. | +| `verify:rootless` | Extract Debian/Ubuntu browser libraries without root, then run `verify`. | + +## Navigate + +- [Architecture and dependency boundary](docs/architecture.md) +- [ADR 0001: runner and testing package boundaries](docs/adr/0001-runner-testing-eval-package-boundaries.md) +- [Tutorial index](docs/index.md) +- [Conformance hand-run tutorial](docs/tutorials/conformance-standalone-tracer.md) +- [Replay hand-run tutorial](docs/tutorials/replay.md) +- [Local runner hand-run tutorial](docs/tutorials/local-runner.md) +- [Local protocol reference](docs/local-runner.md) +- [Durable transport reference](docs/durable-recovery.md) +- [Codex skillless Codex tutorial](docs/tutorials/codex.md) +- [Codex skillless Codex driver reference](docs/codex-driver.md) +- [Live console protocol/server tutorial](docs/tutorials/live-console-protocol-server.md) +- [Live console protocol/server reference](docs/live-console-protocol-server.md) +- [Live console tutorial](docs/tutorials/live-console.md) +- [Live console reference](docs/live-console.md) +- [SDK console tutorial](docs/tutorials/sdk-console.md) +- [SDK browser SDK reference](docs/sdk.md) +- [SDK component decision record](docs/design/sdk-component-decisions.md) +- [Capability semantic catalog and authorization](docs/capability-semantic-catalog.md) +- [Capability live runnerd/Codex loop](docs/capability-live-runnerd-codex.md) +- [Capability issue-thread UI](docs/capability-issue-thread-ui.md) +- [PRP compatibility/versioning policy](docs/protocol-compatibility.md) +- [Adding a harness and permission-mode requirements](docs/adding-a-harness.md) +- [PRP v1 expressiveness audit](spec/prp-v1-expressiveness-audit.md) +- [Cumulative end-to-end tutorial](docs/tutorials/end-to-end.md) + +Codex adds the package-local real-model reference driver, Live console adds the +package-local browser console, and SDK extracts a reusable public SDK plus +two standalone consumers. Runtime production Paperclip integration remains +deferred; the App-owned production conformance adapter is test-only. + +The SDK reference console opens in direct chat mode. Enter a normal prompt, +then open the protocol inspector to review events and reducer state. Expand a +Terminal row and its nested **Debug details** disclosure to inspect every +canonical event retained for that command. The header marker `🖇️ v0.1.2` +identifies the current console iteration. diff --git a/packages/paperclip-runner/devtools/browser/index.html b/packages/paperclip-runner/devtools/browser/index.html new file mode 100644 index 0000000000..301dbcf965 --- /dev/null +++ b/packages/paperclip-runner/devtools/browser/index.html @@ -0,0 +1,13 @@ + + + + + + + Paperclip Runner · PRP Static Replay + + +
+ + + diff --git a/packages/paperclip-runner/devtools/browser/live-console.spec.ts b/packages/paperclip-runner/devtools/browser/live-console.spec.ts new file mode 100644 index 0000000000..23ff80c386 --- /dev/null +++ b/packages/paperclip-runner/devtools/browser/live-console.spec.ts @@ -0,0 +1,398 @@ +import { expect, test, type Page } from "@playwright/test"; + +const DESKTOP = { width: 1440, height: 900 }; +const MOBILE = { width: 390, height: 844 }; + +function shot(page: Page, name: string) { + return page.screenshot({ path: test.info().outputPath(`${name}.png`), fullPage: true }); +} + +async function openConsole(page: Page) { + await page.goto("/"); + await page.getByRole("button", { name: "Live console" }).click(); + await expect(page.getByRole("heading", { name: "Live Codex protocol console" })).toBeVisible(); +} + +async function runManifest(page: Page, id: string) { + await expect(page.getByTestId("manifest-list")).toBeVisible(); + await page.getByTestId(`manifest-${id}`).click(); + await page.getByTestId("run-manifest").click(); + await expect(page.getByTestId("connection-status")).toHaveText("connected", { timeout: 15_000 }); +} + +test.describe("Live console", () => { + test.beforeEach(async ({ page }) => { + await page.setViewportSize(DESKTOP); + }); + + test("shows the empty state and runs a completed turn", async ({ page }) => { + await openConsole(page); + await expect(page.getByTestId("empty-transcript")).toBeVisible(); + await expect(page.getByTestId("composer-send")).toBeDisabled(); + await shot(page, "desktop-01-idle-empty"); + + await runManifest(page, "completion"); + await expect(page.getByTestId("reasoning-item")).toBeVisible(); + await expect(page.getByTestId("tool-item")).toBeVisible(); + await shot(page, "desktop-02-streaming-turn"); + + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 20_000 }); + await expect(page.getByTestId("live-transcript")).toContainText("finished the task"); + await expect(page.getByTestId("live-transcript")).toContainText( + "Summarize what the Live console demo server proves.", + ); + await expect(page.getByTestId("tool-item")).toHaveAttribute("data-status", "completed"); + await shot(page, "desktop-03-turn-completed"); + + // The browser must never receive a provider credential. + const body = await page.locator("body").innerText(); + expect(body).not.toContain("Bearer "); + expect(body).not.toContain("auth.json"); + }); + + test("acknowledges steering and keeps rejected text recoverable", async ({ page }) => { + await openConsole(page); + await runManifest(page, "steering"); + await expect(page.getByTestId("composer-send")).toHaveText("Steer", { timeout: 15_000 }); + + await page.getByTestId("composer-input").fill("Keep it to one sentence."); + await page.getByTestId("composer-send").click(); + const chip = page.getByTestId("steering-chip").first(); + await expect(chip).toContainText("acknowledged", { timeout: 15_000 }); + await shot(page, "desktop-04-steering-acknowledged"); + + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 25_000 }); + // The composer state machine now offers Send again; force a stale steer. + await page.evaluate(() => window.history.replaceState({}, "", "/")); + }); + + test("rejects a stale steer and offers to resend it", async ({ page }) => { + await openConsole(page); + await runManifest(page, "steering"); + await expect(page.getByTestId("composer-send")).toHaveText("Steer", { timeout: 15_000 }); + await page.getByTestId("composer-input").fill("Typed while the turn was running."); + + // Let the turn finish before the steer is sent: the classic stale race. + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 25_000 }); + await expect(page.getByTestId("composer-send")).toHaveText("Send"); + }); + + test("cancels a turn interrupted before it starts", async ({ page }) => { + await openConsole(page); + await runManifest(page, "interrupt-before-start"); + await expect(page.getByTestId("composer-stop")).toBeEnabled(); + await page.getByTestId("composer-stop").click(); + await expect(page.getByTestId("turn-cancelled")).toBeVisible({ timeout: 15_000 }); + await expect(page.getByTestId("turn-cancelled")).toContainText("Cancelled before start"); + await shot(page, "desktop-05-interrupt-before-start"); + }); + + test("keeps partial text when a streaming turn is stopped", async ({ page }) => { + await openConsole(page); + await runManifest(page, "interrupt-during-generation"); + await expect(page.getByTestId("live-transcript")).toContainText("The tracer keeps", { + timeout: 15_000, + }); + await page.getByTestId("composer-stop").click(); + await expect(page.getByTestId("turn-interrupted")).toBeVisible({ timeout: 15_000 }); + await expect(page.getByTestId("interrupted-divider").first()).toBeVisible(); + await expect(page.getByTestId("live-transcript")).not.toContainText("And this final one."); + await shot(page, "desktop-06-interrupt-during-generation"); + }); + + test("keeps an interrupted tool item without a fabricated result", async ({ page }) => { + await openConsole(page); + await runManifest(page, "interrupt-during-tool"); + await expect(page.getByTestId("tool-item")).toHaveAttribute("data-status", "running", { + timeout: 15_000, + }); + await page.getByTestId("composer-stop").click(); + await expect(page.getByTestId("tool-item")).toHaveAttribute("data-status", "interrupted", { + timeout: 15_000, + }); + await expect(page.getByTestId("turn-interrupted")).toBeVisible(); + await shot(page, "desktop-07-interrupt-during-tool"); + }); + + test("resolves approvals exactly once and collapses the resolved card", async ({ page }) => { + await openConsole(page); + await runManifest(page, "approvals"); + const card = page.getByTestId("request-card").first(); + await expect(card).toHaveAttribute("data-status", "pending", { timeout: 15_000 }); + await expect(page.getByTestId("pending-request-banner")).toBeVisible(); + await expect(page.getByTestId("request-action-accept_for_session")).toBeVisible(); + await shot(page, "desktop-08-request-pending"); + + await page.getByTestId("request-action-accept").first().click(); + await expect(page.getByTestId("request-card").first()).toHaveAttribute( + "data-status", + "resolved", + { timeout: 15_000 }, + ); + await expect(page.getByTestId("request-card").first()).toContainText("resolved — Approve"); + await shot(page, "desktop-09-request-resolved"); + + // Second request only offers accept and decline. + const second = page.getByTestId("request-card").nth(1); + await expect(second).toHaveAttribute("data-status", "pending", { timeout: 15_000 }); + await expect(second.getByTestId("request-action-accept_for_session")).toHaveCount(0); + }); + + test("submits user-input and elicitation answers and lets an unanswered request expire", async ({ page }) => { + test.setTimeout(60_000); + await openConsole(page); + await runManifest(page, "user-input"); + await expect(page.getByTestId("request-answer")).toBeVisible({ timeout: 15_000 }); + await page.getByTestId("request-answer").fill("staging"); + await page.getByTestId("request-action-submit").click(); + await expect(page.getByTestId("request-card").first()).toHaveAttribute( + "data-status", + "resolved", + { timeout: 15_000 }, + ); + + // An elicitation submits `content`, not `answers`. The runner validates the + // body against the request kind, so a card that sent the user-input shape + // would be rejected and stay pending here. + const elicitation = page.getByTestId("request-card").nth(1); + await expect(elicitation).toHaveAttribute("data-status", "pending", { timeout: 15_000 }); + await elicitation.getByTestId("request-answer").fill("stable"); + await elicitation.getByTestId("request-action-submit").click(); + await expect(elicitation).toHaveAttribute("data-status", "resolved", { timeout: 15_000 }); + await expect(elicitation).toContainText("resolved — Submit"); + + await expect(page.getByTestId("request-card").nth(2)).toHaveAttribute( + "data-status", + "expired", + { timeout: 25_000 }, + ); + await expect(page.getByTestId("request-card").nth(2)).toContainText("expired before response"); + await shot(page, "desktop-10-request-expired"); + }); + + test("moves a goal through its lifecycle", async ({ page }) => { + await openConsole(page); + await runManifest(page, "goals"); + await expect(page.getByTestId("goal-banner")).toHaveText("No goal set", { timeout: 15_000 }); + + await page.locator("#goal-menu").click(); + await page.getByRole("menuitem", { name: "Set goal…" }).click(); + await page.getByTestId("goal-objective").fill("Keep the demo goal visible"); + await page.getByTestId("goal-set-confirm").click(); + await expect(page.getByTestId("goal-banner")).toContainText("Keep the demo goal visible", { + timeout: 15_000, + }); + await expect(page.getByTestId("goal-banner")).toContainText("active"); + + await page.locator("#goal-menu").click(); + await page.getByRole("menuitem", { name: "Pause" }).click(); + await expect(page.getByTestId("goal-banner")).toContainText("paused", { timeout: 15_000 }); + await shot(page, "desktop-11-goal-supported"); + }); + + test("disables the goal menu with the exact upstream diagnostic", async ({ page }) => { + await openConsole(page); + await runManifest(page, "goals-unsupported"); + const trigger = page.locator("#goal-menu"); + await expect(trigger).toHaveAttribute("aria-disabled", "true", { timeout: 15_000 }); + const describedBy = await trigger.getAttribute("aria-describedby"); + expect(describedBy).not.toBeNull(); + await expect(page.locator(`#${describedBy}`)).toHaveText( + "Goal operations unsupported: app-server 0.132.0 does not advertise the goals capability", + ); + + await page.getByTestId("toggle-inspector").click(); + await page.getByRole("tab", { name: "Capabilities" }).click(); + await expect(page.getByTestId("inspector-capabilities")).toContainText( + "does not advertise the goals capability", + ); + await shot(page, "desktop-12-goal-unsupported"); + }); + + test("shows child threads and refuses to emulate child steering", async ({ page }) => { + await openConsole(page); + await runManifest(page, "subagents"); + await expect(page.getByTestId("thread-thread-child-b")).toBeVisible({ timeout: 20_000 }); + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 20_000 }); + await expect(page.getByTestId("thread-thread-child-a")).toContainText("Completed"); + await expect(page.getByTestId("thread-thread-child-b")).toContainText("Failed"); + + await page.getByTestId("thread-thread-child-a").click(); + await expect(page.getByTestId("breadcrumb")).toContainText("thread-child-a"); + await expect(page.getByTestId("composer-reason")).toHaveText( + "Direct steering of child threads is not supported by this app-server.", + ); + await expect(page.getByTestId("composer-input")).toBeDisabled(); + await shot(page, "desktop-13-lineage-child"); + }); + + test("shows the failed item and diagnostic in the record", async ({ page }) => { + await openConsole(page); + await runManifest(page, "failure"); + await expect(page.getByTestId("turn-failed")).toBeVisible({ timeout: 20_000 }); + await expect(page.getByTestId("tool-failure")).toContainText("cannot find module"); + await shot(page, "desktop-14-failed-turn"); + }); + + test("resumes the same session after a transport drop", async ({ page }) => { + await openConsole(page); + await runManifest(page, "completion"); + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 20_000 }); + const before = await page.getByTestId("live-transcript").innerText(); + + await page.getByTestId("drop-connection").click(); + await expect(page.getByTestId("reconnect-banner")).toBeVisible(); + await expect(page.getByTestId("composer-input")).toBeDisabled(); + await shot(page, "desktop-15-reconnect-banner"); + + await page.getByTestId("retry-now").click(); + await expect(page.getByTestId("connection-status")).toHaveText("connected", { + timeout: 20_000, + }); + // Live and resumed transcripts must be identical, not merely similar. + expect(await page.getByTestId("live-transcript").innerText()).toContain(before.trim()); + }); + + test("restores the same session across a page refresh", async ({ page }) => { + await openConsole(page); + await runManifest(page, "completion"); + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 20_000 }); + const before = await page.getByTestId("live-transcript").innerText(); + + await page.reload(); + await page.getByRole("button", { name: "Live console" }).click(); + await expect(page.getByTestId("live-transcript")).toContainText("finished the task", { + timeout: 20_000, + }); + expect(await page.getByTestId("live-transcript").innerText()).toContain(before.trim()); + await shot(page, "desktop-16-refresh-replay"); + }); + + test("replays a recorded session with a stepper", async ({ page }) => { + await openConsole(page); + await runManifest(page, "completion"); + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 20_000 }); + + await page.getByTestId("toggle-replay").click(); + await expect(page.getByTestId("replay-badge")).toBeVisible(); + await expect(page.getByTestId("replay-stepper")).toBeVisible(); + await expect(page.getByTestId("empty-transcript")).toBeVisible(); + + await page.getByRole("button", { name: "Step forward" }).click(); + await page.getByRole("button", { name: "Step forward" }).click(); + await expect(page.getByTestId("empty-transcript")).toHaveCount(0); + await shot(page, "desktop-17-replay-mode"); + + await page.getByTestId("toggle-replay").click(); + await expect(page.getByTestId("replay-badge")).toHaveCount(0); + await expect(page.getByTestId("turn-completed")).toBeVisible(); + }); + + test("inspects canonical events and reports reducer parity", async ({ page }) => { + await openConsole(page); + await runManifest(page, "completion"); + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 20_000 }); + + await page.getByTestId("toggle-inspector").click(); + await expect(page.getByTestId("inspector-events")).toBeVisible(); + await page.getByTestId("inspector-filter").fill("turn.completed"); + await expect(page.getByTestId("inspector-events").getByRole("listitem")).toHaveCount(1); + await shot(page, "desktop-18-inspector-events"); + + await page.getByTestId("inspector-filter").fill(""); + await page.getByRole("tab", { name: "Session" }).click(); + await expect(page.getByTestId("inspector-session")).toContainText("Credentials in browser"); + await expect(page.getByTestId("replay-parity")).toHaveText("match"); + await shot(page, "desktop-19-inspector-session"); + }); + + test("meets the keyboard contract", async ({ page }) => { + await openConsole(page); + await runManifest(page, "steering"); + await expect(page.getByTestId("composer-send")).toHaveText("Steer", { timeout: 15_000 }); + + // K2: Escape moves focus to Stop and never interrupts directly. + await page.getByTestId("composer-input").focus(); + await page.keyboard.press("Escape"); + await expect(page.getByTestId("composer-stop")).toBeFocused(); + await expect(page.getByTestId("turn-interrupted")).toHaveCount(0); + + // K2: Enter steers the running turn. + await page.getByTestId("composer-input").fill("Steered with the keyboard."); + await page.getByTestId("composer-input").press("Enter"); + await expect(page.getByTestId("steering-chip").first()).toContainText("acknowledged", { + timeout: 15_000, + }); + + // K7: inspector tabs follow the arrow-key tabs pattern. + await page.getByTestId("toggle-inspector").click(); + await page.getByRole("tab", { name: "Events" }).focus(); + await page.keyboard.press("ArrowRight"); + await expect(page.getByRole("tab", { name: "Requests" })).toBeFocused(); + await expect(page.getByRole("tab", { name: "Requests" })).toHaveAttribute( + "aria-selected", + "true", + ); + + // K5: lineage tree responds to arrow keys. + await page.getByTestId("lineage-tree").getByRole("treeitem").first().focus(); + await page.keyboard.press("ArrowDown"); + + // K6/A1: the transcript is a labelled log landmark. + await expect(page.getByTestId("live-transcript")).toHaveAttribute("role", "log"); + await expect(page.getByTestId("live-transcript")).toHaveAttribute("aria-live", "polite"); + }); + + test("resets demo state behind a naming confirmation", async ({ page }) => { + await openConsole(page); + await runManifest(page, "completion"); + await expect(page.getByTestId("turn-completed")).toBeVisible({ timeout: 20_000 }); + + await page.getByRole("button", { name: "Reset demo state", exact: true }).first().click(); + await expect(page.getByRole("dialog")).toContainText( + "Recorded evidence files are not touched.", + ); + await page.getByTestId("reset-confirm").click(); + await expect(page.getByTestId("empty-transcript")).toBeVisible(); + }); +}); + +test.describe("Live console on a small screen", () => { + test("keeps the transcript and composer usable at 390 by 844", async ({ page }) => { + await page.setViewportSize(MOBILE); + await openConsole(page); + await expect(page.getByRole("tab", { name: "Chat" })).toBeVisible(); + await expect(page.getByTestId("manifest-list")).toHaveCount(0); + await shot(page, "mobile-01-idle-empty"); + + await page.getByRole("tab", { name: "Session" }).click(); + await expect(page.getByTestId("manifest-list")).toBeVisible(); + await page.getByTestId("manifest-approvals").click(); + await page.getByTestId("run-manifest").click(); + await shot(page, "mobile-02-session-segment"); + + await page.getByRole("tab", { name: "Chat" }).click(); + await expect(page.getByTestId("request-card").first()).toHaveAttribute( + "data-status", + "pending", + { timeout: 20_000 }, + ); + await shot(page, "mobile-03-request-pending"); + await page.getByTestId("request-action-accept").first().click(); + await expect(page.getByTestId("request-card").first()).toHaveAttribute( + "data-status", + "resolved", + { timeout: 15_000 }, + ); + + await page.getByRole("tab", { name: "Inspector" }).click(); + await expect(page.getByTestId("inspector-events")).toBeVisible(); + await shot(page, "mobile-04-inspector"); + + // WCAG 1.4.10 reflow: no horizontal scrolling of the page itself. + const overflow = await page.evaluate( + () => document.documentElement.scrollWidth - document.documentElement.clientWidth, + ); + expect(overflow).toBeLessThanOrEqual(1); + }); +}); diff --git a/packages/paperclip-runner/devtools/browser/playwright.config.ts b/packages/paperclip-runner/devtools/browser/playwright.config.ts new file mode 100644 index 0000000000..bcd43122dc --- /dev/null +++ b/packages/paperclip-runner/devtools/browser/playwright.config.ts @@ -0,0 +1,29 @@ +import { defineConfig } from "@playwright/test"; + +export default defineConfig({ + testDir: ".", + testMatch: ["static-replay.spec.ts", "live-console.spec.ts"], + fullyParallel: false, + workers: 1, + reporter: "line", + use: { + baseURL: "http://127.0.0.1:4179", + viewport: { width: 1440, height: 1000 }, + }, + webServer: { + command: + "pnpm exec vite preview --config vite.config.ts --host 127.0.0.1 --port 4179", + env: { + // Keep streamed chunks slow enough that interrupt races are reachable + // from a test, and fast enough that the suite stays quick. + PAPERCLIP_LIVE_CONSOLE_CHUNK_DELAY_MS: "60", + // One server serves the whole suite, and every test leaves its session + // behind when its page closes, so the interactive default of 16 is + // exhausted partway through. + PAPERCLIP_LIVE_CONSOLE_MAX_SESSIONS: "64", + }, + url: "http://127.0.0.1:4179", + reuseExistingServer: false, + timeout: 30_000, + }, +}); diff --git a/packages/paperclip-runner/devtools/browser/src/App.tsx b/packages/paperclip-runner/devtools/browser/src/App.tsx new file mode 100644 index 0000000000..020e41e651 --- /dev/null +++ b/packages/paperclip-runner/devtools/browser/src/App.tsx @@ -0,0 +1,642 @@ +import { useMemo, useRef, useState } from "react"; + +import type { PrpEvent, PrpRequest } from "../../../src/protocol/replay-contract"; +import type { SessionSnapshot } from "../../../src/reducer/session-reducer"; +import type { + LocalRunnerRunMetadata, + LocalRunnerRunTrace, + LocalRunnerScenario, +} from "../../../src/contracts/local-runner"; +import { + applyLocalRunnerLiveEvent, + createLocalRunnerLiveSnapshot, + localRunnerSnapshotsMatch, + replayLocalRunnerEvents, +} from "../../../src/tracer/local-runner-live"; +import { replayReplayFixtureText } from "../../../src/tracer/replay"; +import { Badge } from "./components/ui/badge"; +import { Button } from "./components/ui/button"; +import { + Card, + CardContent, + CardDescription, + CardHeader, + CardTitle, +} from "./components/ui/card"; +import { Textarea } from "./components/ui/textarea"; +import { LiveConsole } from "./live/LiveConsole"; + +const fixtureFiles = import.meta.glob( + "../../../protocol/fixtures/replay/*.json", + { eager: true, query: "?raw", import: "default" }, +) as Record; + +const fixtureOrder = [ + "happy-path", + "failed-run", + "interrupted-run", + "duplicate-event", + "source-gap", + "unknown-optional-fields", + "unsupported-required-version", +]; + +const liveScenarios: Array<{ key: LocalRunnerScenario; label: string }> = [ + { key: "happy-path", label: "Happy path" }, + { key: "permission-input", label: "Permission and input" }, + { key: "interrupted", label: "Interruption" }, + { key: "error", label: "Scripted error" }, + { key: "duplicate-terminal", label: "Duplicate terminal guard" }, +]; + +type BrowserMode = "console" | "live" | "replay"; +type LiveStatus = "idle" | "starting" | "running" | "terminal" | "error"; + +const modeCopy: Record< + BrowserMode, + { eyebrow: string; title: string; description: string } +> = { + console: { + eyebrow: "Paperclip Runner Protocol · Live console", + title: "Live Codex protocol console", + description: + "Chat with a live session, steer it, stop it, answer its requests, and inspect the canonical protocol behind every surface.", + }, + live: { + eyebrow: "Paperclip Runner Protocol · Local runner", + title: "Live runner diagnostics", + description: + "Run the local harness, stream validated events, and compare the live and replayed state.", + }, + replay: { + eyebrow: "Paperclip Runner Protocol · Replay", + title: "Static protocol replay", + description: + "Validate a protocol fixture, replay its events, and inspect the resulting session state.", + }, +}; + +interface BrowserFixture { + key: string; + label: string; + source: string; +} + +type LocalRunnerStreamRecord = + | { kind: "event"; event: unknown } + | { kind: "diagnostic"; message: string } + | { kind: "trace"; trace: LocalRunnerRunTrace } + | { kind: "error"; message: string }; + +const browserFixtures = Object.entries(fixtureFiles) + .map(([path, source]): BrowserFixture => { + const key = path.split("/").at(-1)?.replace(/\.json$/, "") ?? path; + const parsed = JSON.parse(source) as { name?: string }; + return { key, label: parsed.name ?? key, source }; + }) + .sort((left, right) => fixtureOrder.indexOf(left.key) - fixtureOrder.indexOf(right.key)); + +function terminalTone(snapshot: SessionSnapshot) { + if (snapshot.integrity === "gap_detected") { + return "warning" as const; + } + if (snapshot.terminal?.runTerminalState === "succeeded") { + return "success" as const; + } + if (snapshot.terminal?.runTerminalState === "failed") { + return "danger" as const; + } + return "neutral" as const; +} + +function humanizeProtocolLabel(value: string): string { + const words = value.replaceAll(/[_-]/g, " "); + return words.length === 0 ? words : `${words[0]?.toUpperCase()}${words.slice(1)}`; +} + +function liveStatusTone(status: LiveStatus, snapshot: SessionSnapshot | null) { + if (status === "error") { + return "danger" as const; + } + if (status !== "terminal") { + return "neutral" as const; + } + if (snapshot?.terminal?.runTerminalState === "succeeded") { + return "success" as const; + } + if (snapshot?.terminal?.runTerminalState === "failed") { + return "danger" as const; + } + return "neutral" as const; +} + +function SnapshotSummary({ + snapshot, + label = "Session snapshot", +}: { + snapshot: SessionSnapshot; + label?: string; +}) { + return ( + <> +
+
+

{label}

+

{snapshot.fixtureName}

+
+ + {humanizeProtocolLabel( + snapshot.integrity === "gap_detected" + ? "Gap detected" + : (snapshot.terminal?.runTerminalState ?? snapshot.runPhase), + )} + +
+ +
+
+
Run
+
{snapshot.identity.runId}
+
+
+
Session
+
{snapshot.identity.normalizedSessionId}
+
+
+
Turn
+
{snapshot.terminal?.turnTerminalState ?? snapshot.turnState}
+
+
+
Events
+
{snapshot.timeline.length}
+
+
+ + {snapshot.proposedResult ? ( +
+ + Reported {humanizeProtocolLabel(snapshot.proposedResult.reportedWorkDisposition)} + +

{snapshot.proposedResult.summary}

+
+ ) : null} + + {snapshot.gaps.length > 0 ? ( +
+ Missing source sequence {snapshot.gaps.flatMap((gap) => gap.missing).join(", ")}. + Replay remains inspectable but is not complete. +
+ ) : null} + +
+

Ordered timeline

+ {snapshot.duplicateEventIds.length} duplicate events ignored +
+
    + {snapshot.timeline.map((event) => ( +
  1. + +
    +
    + {event.eventType} + +
    +

    {event.summary}

    +
    +
  2. + ))} +
+ +
+ Inspect snapshot JSON +
{JSON.stringify(snapshot, null, 2)}
+
+ + ); +} + +function pendingRuntimeRequest(events: readonly PrpEvent[]): PrpRequest | null { + const resolved = new Set( + events + .filter((event) => event.eventType === "runtime_request.resolved") + .map((event) => (event.payload as Record).requestId) + .filter((value): value is string => typeof value === "string"), + ); + for (const event of events.toReversed()) { + if (event.eventType !== "runtime_request.created") { + continue; + } + const request = (event.payload as Record).request; + if ( + typeof request === "object" && + request !== null && + "requestId" in request && + typeof request.requestId === "string" && + !resolved.has(request.requestId) + ) { + return request as PrpRequest; + } + } + return null; +} + +function LiveRunner() { + const [scenario, setScenario] = useState("happy-path"); + const [status, setStatus] = useState("idle"); + const [runId, setRunId] = useState(null); + const [snapshot, setSnapshot] = useState(null); + const [events, setEvents] = useState([]); + const [error, setError] = useState(null); + const [diagnostic, setDiagnostic] = useState(null); + const [input, setInput] = useState("local-runner-live-trace"); + const [trace, setTrace] = useState(null); + const [parity, setParity] = useState(null); + const snapshotRef = useRef(null); + const eventsRef = useRef([]); + + const request = useMemo(() => pendingRuntimeRequest(events), [events]); + + async function startRun() { + setStatus("starting"); + setError(null); + setDiagnostic(null); + setTrace(null); + setParity(null); + setRunId(null); + setEvents([]); + eventsRef.current = []; + setSnapshot(null); + snapshotRef.current = null; + + try { + const startResponse = await fetch("/api/localRunner/runs", { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify({ scenario }), + }); + const started = (await startResponse.json()) as { + id?: string; + metadata?: LocalRunnerRunMetadata; + error?: string; + }; + if (!startResponse.ok || started.id === undefined || started.metadata === undefined) { + throw new Error(started.error ?? "The local runner did not start."); + } + setRunId(started.id); + const initial = createLocalRunnerLiveSnapshot(started.metadata); + setSnapshot(initial); + snapshotRef.current = initial; + setStatus("running"); + + const streamResponse = await fetch(`/api/localRunner/runs/${started.id}/events`); + if (!streamResponse.ok || streamResponse.body === null) { + throw new Error("The live event stream did not open."); + } + const reader = streamResponse.body.getReader(); + const decoder = new TextDecoder(); + let buffered = ""; + while (true) { + const part = await reader.read(); + buffered += decoder.decode(part.value, { stream: !part.done }); + const lines = buffered.split("\n"); + buffered = lines.pop() ?? ""; + for (const line of lines) { + if (line.trim().length > 0) { + handleRecord(JSON.parse(line) as LocalRunnerStreamRecord, started.metadata); + } + } + if (part.done) { + if (buffered.trim().length > 0) { + handleRecord(JSON.parse(buffered) as LocalRunnerStreamRecord, started.metadata); + } + break; + } + } + } catch (cause) { + setStatus("error"); + setError(cause instanceof Error ? cause.message : String(cause)); + } + } + + function handleRecord(record: LocalRunnerStreamRecord, metadata: LocalRunnerRunMetadata) { + if (record.kind === "event") { + const current = snapshotRef.current; + if (current === null) { + return; + } + const applied = applyLocalRunnerLiveEvent(current, record.event); + if (!applied.ok) { + setStatus("error"); + setError(applied.issues.join("; ")); + return; + } + snapshotRef.current = applied.snapshot; + setSnapshot(applied.snapshot); + eventsRef.current = [...eventsRef.current, applied.event]; + setEvents(eventsRef.current); + return; + } + if (record.kind === "diagnostic") { + setDiagnostic(record.message); + return; + } + if (record.kind === "error") { + setStatus("error"); + setError(record.message); + return; + } + setTrace(record.trace); + const replay = replayLocalRunnerEvents(metadata, record.trace.events); + setParity( + snapshotRef.current === null + ? false + : localRunnerSnapshotsMatch(snapshotRef.current, replay), + ); + setStatus("terminal"); + } + + async function postAction(action: "interrupt" | "resolve", body?: unknown) { + if (runId === null) { + return; + } + const response = await fetch(`/api/localRunner/runs/${runId}/${action}`, { + method: "POST", + headers: { "content-type": "application/json" }, + body: JSON.stringify(body ?? {}), + }); + if (!response.ok) { + const failure = (await response.json()) as { error?: string }; + setError(failure.error ?? `The ${action} command failed.`); + setStatus("error"); + } + } + + return ( +
+ + + Local runner controls + + Start one Rust runner and one scripted fake harness. No network or model is used. + + + + + +
+ + +
+ +
+ Runner status + + {humanizeProtocolLabel(status)} + +
+ + {request?.requestKind === "runtime" ? ( +
+
+

{request.type} request

+

{request.prompt}

+
+ {request.type === "permission" ? ( + + ) : ( + <> + + setInput(event.target.value)} + /> + + + )} +
+ ) : null} + + {trace?.harnessProcessExit ? ( +
+
+
Harness process exit
+
{trace.harnessProcessExit.exitCode ?? "signal"}
+
+
+
Semantic result
+
{trace.result?.reportedWorkDisposition ?? "missing"}
+
+
+ ) : null} + + {parity !== null ? ( +
+ Live and replay reducer output + + {parity ? "Match" : "Mismatch"} + +
+ ) : null} + {diagnostic ?

{diagnostic}

: null} + {error ?

{error}

: null} +
+
+ +
+ {snapshot ? ( + + ) : ( +
+

Validated live snapshot

+

Start a local run

+

Each incoming event will pass the Replay validator before the reducer applies it.

+
+ )} +
+
+ ); +} + +function StaticReplay() { + const defaultFixture = browserFixtures[0]; + if (defaultFixture === undefined) { + throw new Error("No Replay fixtures were bundled"); + } + const [selectedFixture, setSelectedFixture] = useState(defaultFixture.key); + const [source, setSource] = useState(defaultFixture.source); + const [replay, setReplay] = useState(() => replayReplayFixtureText(defaultFixture.source)); + const selected = useMemo( + () => browserFixtures.find((fixture) => fixture.key === selectedFixture), + [selectedFixture], + ); + + function chooseFixture(key: string) { + const fixture = browserFixtures.find((candidate) => candidate.key === key); + if (fixture === undefined) { + return; + } + setSelectedFixture(key); + setSource(fixture.source); + setReplay(replayReplayFixtureText(fixture.source)); + } + + return ( +
+ + + Fixture input + + Choose a conformance case or edit the JSON, then validate and replay it. + + + + + + +