= {
+ running: "Running",
+ completed: "Completed",
+ failed: "Failed",
+ interrupted: "Interrupted",
+};
+
+/**
+ * Source-adapted AI Elements `Tool`. The header/input/output anatomy is kept;
+ * the status enum comes from canonical item events instead of an AI SDK tool
+ * part, and payloads stay in `` so nothing is prettified away.
+ */
+export function ToolItem({
+ name,
+ status,
+ input,
+ output,
+ failure,
+}: {
+ name: string;
+ status: ToolStatus;
+ input: string | null;
+ output: string | null;
+ failure: string | null;
+}) {
+ return (
+
+
+
+ Tool {name}
+
+ {STATUS_LABEL[status]}
+
+ {input === null ? null : (
+ <>
+ Input
+ {input}
+ >
+ )}
+ {output === null ? null : (
+ <>
+ Output
+ {output}
+ >
+ )}
+ {failure === null ? null : (
+ <>
+ Diagnostic
+
+ {failure}
+
+ >
+ )}
+ {status === "interrupted" ? (
+
+ Interrupted — the last known status is shown and no result was fabricated.
+
+ ) : null}
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/components/ui/tooltip.tsx b/packages/paperclip-runner/devtools/browser/src/components/ui/tooltip.tsx
new file mode 100644
index 0000000000..6f367d8ada
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/components/ui/tooltip.tsx
@@ -0,0 +1,25 @@
+import * as React from "react";
+
+/**
+ * Source-adapted shadcn Tooltip. The content is always mirrored into a
+ * described-by element, so a tooltip is never the only channel for a
+ * diagnostic (interaction map A5).
+ */
+export function Tooltip({
+ content,
+ id,
+ children,
+}: {
+ content: string;
+ id: string;
+ children: React.ReactElement<{ "aria-describedby"?: string }>;
+}) {
+ return (
+
+ {React.cloneElement(children, { "aria-describedby": id })}
+
+ {content}
+
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/live/Inspector.tsx b/packages/paperclip-runner/devtools/browser/src/live/Inspector.tsx
new file mode 100644
index 0000000000..57d0cb1104
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/Inspector.tsx
@@ -0,0 +1,229 @@
+import * as React from "react";
+
+import type { PrpEvent } from "../../../../src/protocol/replay-contract";
+import type { SessionSnapshot } from "../../../../src/reducer/session-reducer";
+import { Badge } from "../components/ui/badge";
+import { Tabs } from "../components/ui/tabs";
+import type { LiveConnectionStatus, LiveSessionState } from "./protocol";
+import { capabilityRows } from "./transcript-model";
+
+const TABS = [
+ { id: "events", label: "Events" },
+ { id: "requests", label: "Requests" },
+ { id: "capabilities", label: "Capabilities" },
+ { id: "session", label: "Session" },
+];
+
+function CopyButton({ label, value }: { label: string; value: string }) {
+ const [copied, setCopied] = React.useState(false);
+ return (
+ {
+ void navigator.clipboard?.writeText(value).then(
+ () => setCopied(true),
+ () => setCopied(false),
+ );
+ }}
+ >
+ {copied ? "Copied" : "Copy"}
+
+ );
+}
+
+/**
+ * The debugging surface. Redaction is the server's job: the inspector renders
+ * redaction markers verbatim and never tries to reconstruct a redacted value.
+ */
+export function Inspector({
+ events,
+ snapshot,
+ state,
+ connection,
+ reconnectAttempt,
+ replayParity,
+ highlightItemId,
+ onHighlightItem,
+}: {
+ events: readonly PrpEvent[];
+ snapshot: SessionSnapshot | null;
+ state: LiveSessionState | null;
+ connection: LiveConnectionStatus;
+ reconnectAttempt: number;
+ replayParity: boolean | null;
+ highlightItemId: string | null;
+ onHighlightItem: (itemId: string | null) => void;
+}) {
+ const [tab, setTab] = React.useState("events");
+ const [filter, setFilter] = React.useState("");
+ const [facet, setFacet] = React.useState(null);
+
+ const facets = React.useMemo(
+ () => [...new Set(events.map((event) => event.eventType))].sort(),
+ [events],
+ );
+
+ const visible = React.useMemo(
+ () =>
+ events.filter((event) => {
+ if (facet !== null && event.eventType !== facet) return false;
+ if (filter.length === 0) return true;
+ return JSON.stringify(event).toLowerCase().includes(filter.toLowerCase());
+ }),
+ [events, facet, filter],
+ );
+
+ return (
+
+ {tab === "events" ? (
+
+
+ Filter events
+
+
setFilter(event.target.value)}
+ />
+
+ setFacet(null)}
+ >
+ All
+
+ {facets.map((candidate) => (
+ setFacet(candidate)}
+ >
+ {candidate}
+
+ ))}
+
+
+ {visible.map((event) => (
+
+
+ onHighlightItem(
+ (toggle.currentTarget as HTMLDetailsElement).open
+ ? (event.itemId ?? null)
+ : null,
+ )
+ }
+ >
+
+ {event.sourceSeq}
+ {event.eventType}
+ {event.itemId ?? event.turnId ?? "—"}
+
+ {JSON.stringify(event, null, 2)}
+
+
+
+ ))}
+
+
+ ) : null}
+
+ {tab === "requests" ? (
+
+ {(snapshot?.requests ?? []).map((request) => (
+
+
+ {request.requestId}
+
+
+ {request.type} — {request.status}
+
+
+ ))}
+ {(snapshot?.requests.length ?? 0) === 0 ? No runtime requests yet.
: null}
+
+ ) : null}
+
+ {tab === "capabilities" ? (
+
+ {capabilityRows(state?.capabilities ?? { resume: false, typedEvents: false, steering: false, interruption: false, structuredResult: false }).map(
+ (row) => (
+
+ {row.name}
+
+ {row.supported ? "supported" : "unsupported"}
+
+ {row.diagnostic === null ? null : (
+ {row.diagnostic}
+ )}
+
+ ),
+ )}
+
+ ) : null}
+
+ {tab === "session" ? (
+
+
+
Run
+ {state?.runId ?? "—"}
+
+
+
Normalized session
+ {state?.normalizedSessionId ?? "—"}
+
+
+
Driver session
+ {state?.driverSession.driverSessionId ?? "—"}
+
+
+
Source cursors
+ {JSON.stringify(snapshot?.sourceCursors ?? {})}
+
+
+
Integrity
+ {snapshot?.integrity ?? "—"}
+
+
+
Gaps
+ {snapshot?.gaps.length ?? 0}
+
+
+
Connection
+
+ {connection}
+ {connection === "reconnecting" ? ` (attempt ${reconnectAttempt})` : ""}
+
+
+
+
Credentials in browser
+ {state?.credentialsExposed === true ? "yes" : "no"}
+
+
+
Live and replay reducer
+
+ {replayParity === null ? "not compared yet" : replayParity ? "match" : "mismatch"}
+
+
+
+ ) : null}
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/live/LiveConsole.tsx b/packages/paperclip-runner/devtools/browser/src/live/LiveConsole.tsx
new file mode 100644
index 0000000000..7eb0b5f1e7
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/LiveConsole.tsx
@@ -0,0 +1,487 @@
+import * as React from "react";
+
+import { Badge } from "../components/ui/badge";
+import { Banner, BannerActions, BannerText } from "../components/ui/banner";
+import { Button } from "../components/ui/button";
+import { Composer } from "../components/ui/composer";
+import { Conversation } from "../components/ui/conversation";
+import { Dialog } from "../components/ui/dialog";
+import { Message, MessageText } from "../components/ui/message";
+import { ReasoningItem } from "../components/ui/reasoning-item";
+import { Tabs } from "../components/ui/tabs";
+import { ToolItem, type ToolStatus } from "../components/ui/tool-item";
+import { Inspector } from "./Inspector";
+import { RequestCard } from "./RequestCard";
+import { ConnectionBlock, GoalBlock, LineageTree, ManifestList } from "./SessionPanel";
+import { useLiveConsole } from "./use-live-console";
+import type { TranscriptEntry } from "./transcript-model";
+
+const MOBILE_SEGMENTS = [
+ { id: "chat", label: "Chat" },
+ { id: "session", label: "Session" },
+ { id: "inspector", label: "Inspector" },
+];
+
+const TURN_TONE = {
+ completed: "success",
+ failed: "danger",
+ interrupted: "warning",
+ cancelled: "warning",
+} as const;
+
+const TURN_LABEL = {
+ completed: "Turn completed",
+ failed: "Turn failed",
+ interrupted: "Turn interrupted",
+ cancelled: "Cancelled before start",
+} as const;
+
+/**
+ * Renders exactly one layout. Hiding a second copy with CSS would duplicate
+ * the transcript log landmark and every control in the DOM.
+ */
+function useCompactLayout(): boolean {
+ const [compact, setCompact] = React.useState(
+ () => window.matchMedia?.("(max-width: 64rem)").matches === true,
+ );
+ React.useEffect(() => {
+ const query = window.matchMedia?.("(max-width: 64rem)");
+ if (query === undefined) return undefined;
+ const onChange = () => setCompact(query.matches);
+ onChange();
+ query.addEventListener("change", onChange);
+ return () => query.removeEventListener("change", onChange);
+ }, []);
+ return compact;
+}
+
+function toolStatus(entry: Extract): ToolStatus {
+ if (entry.interrupted) return "interrupted";
+ if (entry.item.status === "failed") return "failed";
+ if (entry.item.status === "running") return "running";
+ return "completed";
+}
+
+export function LiveConsole() {
+ const console_ = useLiveConsole();
+ const [draft, setDraft] = React.useState("");
+ const [inspectorOpen, setInspectorOpen] = React.useState(false);
+ const [segment, setSegment] = React.useState("chat");
+ const [resetOpen, setResetOpen] = React.useState(false);
+ const [highlightItemId, setHighlightItemId] = React.useState(null);
+ const [outcomes, setOutcomes] = React.useState>({});
+ const oldestPendingRef = React.useRef(null);
+ const compact = useCompactLayout();
+
+ React.useEffect(() => {
+ const stored = window.localStorage.getItem("paperclip-runner.live-console.inspector");
+ if (stored === "open") setInspectorOpen(true);
+ }, []);
+
+ React.useEffect(() => {
+ window.localStorage.setItem(
+ "paperclip-runner.live-console.inspector",
+ inspectorOpen ? "open" : "closed",
+ );
+ }, [inspectorOpen]);
+
+ const manifestId = console_.state?.manifest ?? console_.selectedManifestId;
+ const activeManifest = console_.manifests.find((entry) => entry.id === manifestId);
+ const pendingRequests = console_.transcript.filter(
+ (entry) => entry.kind === "request" && entry.status === "pending",
+ );
+ const viewingChild = console_.selectedThreadId !== null;
+ const childSteeringSupported = false;
+
+ React.useEffect(() => {
+ const snapshot = console_.snapshot;
+ if (snapshot === null || console_.state === null) return;
+ if (!["completed", "failed", "interrupted", "cancelled"].includes(snapshot.turnState)) return;
+ setOutcomes((current) => ({
+ ...current,
+ [console_.state!.manifest ?? "unknown"]:
+ snapshot.turnState === "completed" ? "passed observations" : snapshot.turnState,
+ }));
+ }, [console_.snapshot, console_.state]);
+
+ function submit() {
+ const text = draft.trim();
+ if (text.length === 0) return;
+ setDraft("");
+ if (console_.composer === "active-turn") void console_.steer(text);
+ else void console_.send(text);
+ }
+
+ function runManifest(id: string) {
+ const manifest = console_.manifests.find((entry) => entry.id === id);
+ void console_.start({ manifestId: id, message: manifest?.prompt ?? "Run the demo." });
+ }
+
+ const composerReason = viewingChild && !childSteeringSupported
+ ? "Direct steering of child threads is not supported by this app-server."
+ : null;
+
+ const transcriptBody = (
+
+ {console_.transcript.length === 0 ? (
+
+
No session yet
+
Pick a demo chat or start a blank session
+
+ The browser never holds a provider credential. Only the demo server starts a driver
+ and owns the working directory.
+
+
+ ) : null}
+
+ {console_.transcript.map((entry) => {
+ if (entry.kind === "user") {
+ return (
+
+
+
+
+
+ );
+ }
+ if (entry.kind === "turn") {
+ return (
+
+ {TURN_LABEL[entry.state]}
+ {entry.detail}
+
+ );
+ }
+ if (entry.kind === "diagnostic") {
+ return (
+
+
+
+
+
+ );
+ }
+ if (entry.kind === "request") {
+ const isOldestPending =
+ entry.status === "pending" && pendingRequests[0]?.key === entry.key;
+ return (
+
+ void console_.resolve(entry.requestId, entry.turnId ?? "", resolution)
+ }
+ />
+ );
+ }
+ if (entry.role === "reasoning") {
+ return (
+
+
+
+ );
+ }
+ if (entry.role === "tool") {
+ return (
+
+
+
+ );
+ }
+ return (
+
+
+
+ {entry.failure === null ? null : (
+ {entry.failure}
+ )}
+
+
+ );
+ })}
+ {console_.steeringChips.map((chip) => (
+
+
+ {chip.status === "rejected"
+ ? `rejected — ${chip.detail ?? "turn already ended"}`
+ : chip.status}
+
+ {chip.text}
+ {chip.status === "rejected" || chip.status === "failed" ? (
+ {
+ console_.dismissSteeringChip(chip.id);
+ void console_.send(chip.text);
+ }}
+ >
+ Send as new message
+
+ ) : null}
+
+ ))}
+
+
+ );
+
+ const composerBlock = (
+
+ {console_.connection === "disconnected" || console_.connection === "reconnecting" ? (
+
+
+ {console_.connection === "reconnecting"
+ ? `Connection lost — reconnecting (attempt ${console_.reconnectAttempt})…`
+ : "Connection lost — the session is still on the server."}
+
+
+
+ Retry now
+
+
+
+ ) : null}
+ {pendingRequests.length > 0 ? (
+
+ {pendingRequests.length} request waiting — Review
+
+ oldestPendingRef.current?.focus()}
+ data-testid="review-request"
+ >
+ Review
+
+
+
+ ) : null}
+ void console_.interrupt()}
+ disabledReason={composerReason}
+ />
+
+ );
+
+ const sessionRail = (
+ <>
+
+ oldestPendingRef.current?.focus()}
+ />
+ void console_.goal(operation, body)} />
+
+
+ >
+ );
+
+ const inspectorPanel = (
+
+ );
+
+ return (
+
+
+
+ {console_.replay.active ? (
+
+ REPLAY
+
+ ) : null}
+
{activeManifest?.name ?? "Live console"}
+ {viewingChild ? (
+
+ root ▸ {console_.selectedThreadId}
+
+ ) : null}
+
+
+ setInspectorOpen((current) => !current)}
+ >
+ Protocol inspector
+
+ (console_.replay.active ? console_.replay.exit() : console_.replay.enter())}
+ >
+ {console_.replay.active ? "Exit replay" : "Replay"}
+
+
+
+
+
+ {console_.announcement}
+
+ {console_.error === null ? null : (
+
+ {console_.error}
+
+ )}
+
+ {console_.replay.active ? (
+
+
+ {console_.replay.playing ? "Pause" : "Play"}
+
+ console_.replay.step(-1)}>
+ Step back
+
+ console_.replay.step(1)}>
+ Step forward
+
+ Position
+ console_.replay.setPosition(Number(event.target.value))}
+ />
+
+ {console_.replay.position} / {console_.replay.length}
+
+
+ ) : null}
+
+ {compact ? (
+
+
+ {segment === "chat" ? (
+ <>
+ {transcriptBody}
+ {composerBlock}
+ >
+ ) : null}
+ {segment === "session" ? sessionRail : null}
+ {segment === "inspector" ? inspectorPanel : null}
+
+
+ ) : (
+
+
+
+ {transcriptBody}
+ {composerBlock}
+
+ {inspectorOpen ? (
+
+ ) : null}
+
+ )}
+
+
setResetOpen(false)}
+ title="Reset demo state"
+ description="Discards the current live session transcript. Recorded evidence files are not touched."
+ footer={
+ <>
+ setResetOpen(false)}>
+ Keep the session
+
+ {
+ setResetOpen(false);
+ setDraft("");
+ void console_.reset();
+ }}
+ >
+ Reset demo state
+
+ >
+ }
+ />
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/live/RequestCard.tsx b/packages/paperclip-runner/devtools/browser/src/live/RequestCard.tsx
new file mode 100644
index 0000000000..abe7969ba7
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/RequestCard.tsx
@@ -0,0 +1,168 @@
+import * as React from "react";
+
+import { Badge } from "../components/ui/badge";
+import { Button } from "../components/ui/button";
+import { runtimeRequestSubmission, type TranscriptRequestEntry } from "./transcript-model";
+
+const KIND_LABELS: Record = {
+ command_approval: "Command",
+ file_approval: "File change",
+ permission_approval: "Permission",
+ user_input: "Input",
+ elicitation: "Input",
+};
+
+const ACTION_LABELS: Record = {
+ accept: "Approve",
+ accept_for_session: "Approve for session",
+ decline: "Reject",
+ cancel: "Cancel",
+ submit: "Submit",
+};
+
+const STATUS_TONE = {
+ pending: "accent",
+ resolved: "success",
+ expired: "warning",
+ cancelled: "warning",
+} as const;
+
+function detailBlock(details: Record): string | null {
+ const command = details.command;
+ if (typeof command === "string") return command;
+ const diff = details.diff;
+ if (typeof diff === "string") {
+ const path = typeof details.path === "string" ? `${details.path}\n` : "";
+ return `${path}${diff}`;
+ }
+ return null;
+}
+
+/**
+ * One card renders all five request kinds. Actions come from the upstream
+ * request contract, so the card can never invent an affordance the runner
+ * cannot honour, and the first click locks the row until the canonical
+ * resolved event arrives.
+ */
+export function RequestCard({
+ entry,
+ onResolve,
+ focusRef,
+}: {
+ entry: TranscriptRequestEntry;
+ onResolve: (resolution: Record) => void;
+ focusRef?: React.Ref;
+}) {
+ const [answer, setAnswer] = React.useState("");
+ const [resolving, setResolving] = React.useState(false);
+ const answerId = React.useId();
+ const kindLabel = KIND_LABELS[entry.requestKind] ?? "Request";
+ const pending = entry.status === "pending";
+ const isInput = entry.requestKind === "user_input" || entry.requestKind === "elicitation";
+ const payload = detailBlock(entry.details);
+
+ React.useEffect(() => {
+ if (!pending) setResolving(false);
+ }, [pending]);
+
+ function submit(resolution: Record) {
+ setResolving(true);
+ onResolve(resolution);
+ }
+
+ if (!pending) {
+ return (
+
+
+
+
+ {kindLabel} · {entry.status === "resolved"
+ ? `resolved — ${ACTION_LABELS[entry.resolvedAction ?? ""] ?? entry.resolvedAction ?? "unknown"}`
+ : entry.status === "expired"
+ ? "expired before response"
+ : "cancelled"}
+
+ {entry.status}
+
+ {entry.prompt}
+
+ {entry.requestId}
+ {entry.resolvedAt === null ? null : {entry.resolvedAt} }
+
+ {payload === null ? null : {payload} }
+
+
+ );
+ }
+
+ return (
+
+
+ {kindLabel}
+ pending
+
+ {entry.prompt}
+
+ {entry.requestId}
+
+ {payload === null ? null : {payload} }
+ {isInput ? (
+
+ Answer
+ setAnswer(event.target.value)}
+ />
+
+ ) : null}
+
+ {entry.actions.map((action) => (
+
+ submit(
+ action === "submit"
+ ? runtimeRequestSubmission(entry.requestKind, answer)
+ : { action },
+ )
+ }
+ >
+ {ACTION_LABELS[action] ?? action}
+
+ ))}
+
+ {resolving ? (
+
+ Resolving — waiting for the canonical resolved event.
+
+ ) : null}
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/live/SessionPanel.tsx b/packages/paperclip-runner/devtools/browser/src/live/SessionPanel.tsx
new file mode 100644
index 0000000000..f5939551b9
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/SessionPanel.tsx
@@ -0,0 +1,346 @@
+import * as React from "react";
+
+import { Badge } from "../components/ui/badge";
+import { Button } from "../components/ui/button";
+import { Dialog } from "../components/ui/dialog";
+import { Menu } from "../components/ui/menu";
+import type {
+ GoalOperation,
+ LiveConnectionStatus,
+ LiveLineageEntry,
+ LiveManifestSummary,
+ LiveSessionState,
+} from "./protocol";
+import { GOAL_OPERATION_LABELS } from "./protocol";
+import { goalAvailability } from "./transcript-model";
+
+const CONNECTION_TONE = {
+ idle: "neutral",
+ connecting: "accent",
+ connected: "success",
+ reconnecting: "warning",
+ disconnected: "danger",
+ terminal: "neutral",
+} as const;
+
+const THREAD_STATE_LABEL: Record = {
+ starting: "Starting",
+ running: "Running",
+ completed: "Completed",
+ failed: "Failed",
+};
+
+export function ManifestList({
+ manifests,
+ selectedId,
+ onSelect,
+ onRun,
+ lastOutcome,
+ disabled,
+}: {
+ manifests: readonly LiveManifestSummary[];
+ selectedId: string;
+ onSelect: (id: string) => void;
+ onRun: (id: string) => void;
+ lastOutcome: Record;
+ disabled: boolean;
+}) {
+ return (
+
+
+ Demo chats
+
+
+ {manifests.map((manifest) => (
+
+ onSelect(manifest.id)}
+ >
+ {manifest.name}
+ {manifest.purpose}
+
+ {manifest.scenarios.map((scenario) => (
+ {scenario}
+ ))}
+
+ {lastOutcome[manifest.id] ?? "not run"}
+
+
+
+ {manifest.id === selectedId ? (
+ <>
+
+ Expected observations
+
+ {manifest.expectedObservations.map((observation) => (
+ {observation}
+ ))}
+
+
+ onRun(manifest.id)}
+ >
+ Run this demo chat
+
+ >
+ ) : null}
+
+ ))}
+
+
+ );
+}
+
+export function ConnectionBlock({
+ connection,
+ reconnectAttempt,
+ state,
+ pendingRequestCount,
+ onDrop,
+ onRetry,
+ onScrollToOldestRequest,
+}: {
+ connection: LiveConnectionStatus;
+ reconnectAttempt: number;
+ state: LiveSessionState | null;
+ pendingRequestCount: number;
+ onDrop: () => void;
+ onRetry: () => void;
+ onScrollToOldestRequest: () => void;
+}) {
+ return (
+
+
+ Session
+
+
+ Connection
+
+ {connection === "reconnecting" ? `reconnecting (attempt ${reconnectAttempt})` : connection}
+
+
+
+ Pending requests
+
+ {pendingRequestCount}
+
+
+
+ Credentials in browser
+
+ {state?.credentialsExposed === true ? "exposed" : "none"}
+
+
+
+
+
Run
+ {state?.runId ?? "—"}
+
+
+
Session
+ {state?.normalizedSessionId ?? "—"}
+
+
+
+
+ Simulate connection loss
+
+
+ Resume this session
+
+
+
+ );
+}
+
+export function GoalBlock({
+ state,
+ onGoal,
+}: {
+ state: LiveSessionState | null;
+ onGoal: (operation: GoalOperation, body?: Record) => void;
+}) {
+ const [dialogOpen, setDialogOpen] = React.useState(false);
+ const [objective, setObjective] = React.useState("");
+ const objectiveId = React.useId();
+ const capabilities = state?.capabilities ?? {
+ resume: false,
+ typedEvents: false,
+ steering: false,
+ interruption: false,
+ structuredResult: false,
+ };
+ const availability = goalAvailability(capabilities, state?.goal?.status ?? null);
+ const unsupported = capabilities.goals !== true;
+ const unsupportedReason = availability[0]?.reason ?? null;
+
+ return (
+
+
+
+ Goal
+
+ ({
+ id: entry.operation,
+ label: GOAL_OPERATION_LABELS[entry.operation],
+ enabled: entry.enabled,
+ reason: entry.reason,
+ }))}
+ onSelect={(id) => {
+ if (id === "set") setDialogOpen(true);
+ else onGoal(id as GoalOperation);
+ }}
+ />
+
+ {state === null ? null : state.goal === null ? (
+
+ {unsupported ? "Goal operations are unsupported here." : "No goal set"}
+
+ ) : (
+
+
{state.goal.objective}
+
+
+ {state.goal.status}
+
+
+ {new Date(state.goal.updatedAt).toISOString().slice(11, 19)}
+
+
+
+ )}
+ setDialogOpen(false)}
+ title="Set goal"
+ description="The goal is durable on the thread and every change is also a transcript event."
+ footer={
+ <>
+ setDialogOpen(false)}>
+ Cancel
+
+ {
+ onGoal("set", { objective: objective.trim() });
+ setObjective("");
+ setDialogOpen(false);
+ }}
+ >
+ Set goal
+
+ >
+ }
+ >
+ Objective
+
+
+ );
+}
+
+export function LineageTree({
+ lineage,
+ selectedThreadId,
+ onSelect,
+}: {
+ lineage: readonly LiveLineageEntry[];
+ selectedThreadId: string | null;
+ onSelect: (threadId: string | null) => void;
+}) {
+ const rootId = lineage.find((entry) => entry.depth === 0)?.threadId ?? null;
+ const current = selectedThreadId ?? rootId;
+ const treeRef = React.useRef(null);
+
+ function onKeyDown(event: React.KeyboardEvent) {
+ const rows = [
+ ...(treeRef.current?.querySelectorAll('[role="treeitem"]') ?? []),
+ ];
+ const index = rows.indexOf(document.activeElement as HTMLButtonElement);
+ if (event.key === "ArrowDown") {
+ event.preventDefault();
+ rows[Math.min(index + 1, rows.length - 1)]?.focus();
+ } else if (event.key === "ArrowUp") {
+ event.preventDefault();
+ rows[Math.max(index - 1, 0)]?.focus();
+ }
+ }
+
+ return (
+
+
+ Threads
+
+
+ {lineage.length === 0 ? No threads yet. : null}
+ {lineage.map((entry) => (
+
+ onSelect(entry.depth === 0 ? null : entry.threadId)}
+ >
+
+ {entry.nickname ?? entry.threadId}
+
+ {THREAD_STATE_LABEL[entry.status] ?? entry.status}
+
+
+
+ ))}
+
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/live/client.ts b/packages/paperclip-runner/devtools/browser/src/live/client.ts
new file mode 100644
index 0000000000..b183aaaa57
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/client.ts
@@ -0,0 +1,165 @@
+import type { PrpEvent } from "../../../../src/protocol/replay-contract";
+import type {
+ GoalOperation,
+ LiveManifestSummary,
+ LiveSessionState,
+} from "./protocol";
+
+const BASE = "/api/liveConsole";
+
+export class LiveConsoleError extends Error {
+ readonly status: number;
+ readonly code: string;
+
+ constructor(status: number, code: string, message: string) {
+ super(message);
+ this.name = "LiveConsoleError";
+ this.status = status;
+ this.code = code;
+ }
+}
+
+async function send(path: string, init?: RequestInit): Promise {
+ const response = await fetch(`${BASE}${path}`, {
+ ...init,
+ headers: init?.body === undefined ? undefined : { "content-type": "application/json" },
+ });
+ const body = (await response.json()) as Record;
+ if (!response.ok) {
+ throw new LiveConsoleError(
+ response.status,
+ typeof body.error === "string" ? body.error : "request_failed",
+ typeof body.message === "string" ? body.message : `Request to ${path} failed.`,
+ );
+ }
+ return body as T;
+}
+
+export async function fetchManifests(): Promise {
+ const body = await send<{ manifests: LiveManifestSummary[] }>("/manifests");
+ return body.manifests;
+}
+
+export async function createSession(input: {
+ manifest: string;
+ objective: string;
+ message?: string;
+ startTurn?: boolean;
+}): Promise {
+ return send("/sessions", {
+ method: "POST",
+ body: JSON.stringify(input),
+ });
+}
+
+export async function readSession(sessionId: string): Promise {
+ return send(`/sessions/${sessionId}`);
+}
+
+export async function readEvents(
+ sessionId: string,
+ after = 0,
+): Promise<{ events: PrpEvent[]; cursor: number; replay: boolean }> {
+ return send(`/sessions/${sessionId}/events?after=${after}`);
+}
+
+export async function startTurn(
+ sessionId: string,
+ text: string,
+): Promise {
+ return send(`/sessions/${sessionId}/turns`, {
+ method: "POST",
+ body: JSON.stringify({ text }),
+ });
+}
+
+export async function steerTurn(
+ sessionId: string,
+ turnId: string,
+ text: string,
+): Promise {
+ return send(`/sessions/${sessionId}/steer`, {
+ method: "POST",
+ body: JSON.stringify({ turnId, text }),
+ });
+}
+
+export async function interruptTurn(
+ sessionId: string,
+ turnId: string | null,
+): Promise {
+ return send(`/sessions/${sessionId}/interrupt`, {
+ method: "POST",
+ body: JSON.stringify({
+ reason: "browser_operator",
+ ...(turnId === null ? {} : { turnId }),
+ }),
+ });
+}
+
+export async function resolveRequest(
+ sessionId: string,
+ requestId: string,
+ turnId: string,
+ resolution: Record,
+): Promise {
+ return send(
+ `/sessions/${sessionId}/requests/${encodeURIComponent(requestId)}/resolve`,
+ { method: "POST", body: JSON.stringify({ turnId, resolution }) },
+ );
+}
+
+export async function goalOperation(
+ sessionId: string,
+ operation: GoalOperation,
+ body: Record = {},
+): Promise {
+ return send(`/sessions/${sessionId}/goal/${operation}`, {
+ method: "POST",
+ body: JSON.stringify(body),
+ });
+}
+
+export async function reconnectSession(sessionId: string): Promise {
+ return send(`/sessions/${sessionId}/reconnect`, {
+ method: "POST",
+ body: JSON.stringify({}),
+ });
+}
+
+export async function closeSession(sessionId: string): Promise {
+ await send(`/sessions/${sessionId}/close`, {
+ method: "POST",
+ body: JSON.stringify({}),
+ });
+}
+
+export interface EventStreamHandle {
+ close(): void;
+}
+
+/**
+ * Subscribes to the canonical event stream from a durable cursor. The stream
+ * never carries provider credentials; the server redacts every frame.
+ */
+export function openEventStream(
+ sessionId: string,
+ after: number,
+ handlers: {
+ onEvent: (event: PrpEvent) => void;
+ onOpen?: () => void;
+ onError?: () => void;
+ },
+): EventStreamHandle {
+ const source = new EventSource(`${BASE}/sessions/${sessionId}/stream?after=${after}`);
+ source.onopen = () => handlers.onOpen?.();
+ source.onmessage = (message) => {
+ try {
+ handlers.onEvent(JSON.parse(message.data) as PrpEvent);
+ } catch {
+ handlers.onError?.();
+ }
+ };
+ source.onerror = () => handlers.onError?.();
+ return { close: () => source.close() };
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/live/protocol.ts b/packages/paperclip-runner/devtools/browser/src/live/protocol.ts
new file mode 100644
index 0000000000..d793d59db4
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/protocol.ts
@@ -0,0 +1,115 @@
+import type { PrpEvent } from "../../../../src/protocol/replay-contract";
+import type { SessionSnapshot } from "../../../../src/reducer/session-reducer";
+
+/**
+ * The browser only ever sees the demo server's public state and the canonical
+ * PRP event stream. There is no second event model and no client-side session
+ * cache: every surface reads these shapes or the reducer snapshot they carry.
+ */
+export interface LiveCapabilities {
+ resume: boolean;
+ typedEvents: boolean;
+ steering: boolean;
+ interruption: boolean;
+ structuredResult: boolean;
+ read?: boolean;
+ reconciliation?: boolean;
+ usage?: boolean;
+ runtimeRequestResolution?: boolean;
+ runtimeRequestHandoff?: boolean;
+ goals?: boolean;
+ threadLineage?: boolean;
+ unsupported?: string[];
+}
+
+export interface LivePendingRequest {
+ requestId: string;
+ requestKind: string;
+ method: string;
+ turnId: string;
+ itemId: string;
+ status: string;
+ prompt: string;
+ details: Record;
+}
+
+export interface LiveGoal {
+ threadId: string;
+ objective: string;
+ status: string;
+ tokenBudget: number | null;
+ tokensUsed: number;
+ timeUsedSeconds: number;
+ createdAt: number;
+ updatedAt: number;
+}
+
+export interface LiveLineageEntry {
+ threadId: string;
+ providerSessionId: string | null;
+ parentThreadId: string | null;
+ depth: number;
+ nickname: string | null;
+ role: string | null;
+ status: string;
+}
+
+export interface LiveSessionState {
+ sessionId: string;
+ runId: string;
+ normalizedSessionId: string;
+ manifest: string | null;
+ providerAuthentication: string;
+ credentialsExposed: boolean;
+ capabilities: LiveCapabilities;
+ driverSession: {
+ driverSessionId: string;
+ providerSessionId?: string | null;
+ displayId?: string | null;
+ };
+ activeTurnId: string | null;
+ pendingRequests: LivePendingRequest[];
+ goal: LiveGoal | null;
+ lineage: LiveLineageEntry[];
+ cursor: number;
+ snapshot: SessionSnapshot;
+}
+
+export interface LiveManifestSummary {
+ id: string;
+ name: string;
+ purpose: string;
+ scenarios: string[];
+ objective: string;
+ prompt: string;
+ expectedObservations: string[];
+}
+
+export type LiveConnectionStatus =
+ | "idle"
+ | "connecting"
+ | "connected"
+ | "reconnecting"
+ | "disconnected"
+ | "terminal";
+
+export type GoalOperation = "get" | "set" | "pause" | "resume" | "clear";
+
+export type LiveEventList = readonly PrpEvent[];
+
+export const GOAL_OPERATIONS: readonly GoalOperation[] = [
+ "set",
+ "get",
+ "pause",
+ "resume",
+ "clear",
+];
+
+/** Human labels for the five fixed goal verbs (interaction map §5). */
+export const GOAL_OPERATION_LABELS: Record = {
+ set: "Set goal…",
+ get: "View",
+ pause: "Pause",
+ resume: "Resume",
+ clear: "Clear",
+};
diff --git a/packages/paperclip-runner/devtools/browser/src/live/transcript-model.test.ts b/packages/paperclip-runner/devtools/browser/src/live/transcript-model.test.ts
new file mode 100644
index 0000000000..119382d37b
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/transcript-model.test.ts
@@ -0,0 +1,293 @@
+import { describe, expect, it } from "vitest";
+
+import { LiveConsoleScriptedDriver } from "../../../../src/mock-core/live-console-scripted-driver";
+import {
+ parseHarnessRuntimeRequestResolution,
+ type HarnessSession,
+} from "../../../../src/contracts/harness-driver";
+import type { PrpEvent } from "../../../../src/protocol/replay-contract";
+import {
+ applyPrpEvent,
+ createSessionSnapshotFromMetadata,
+ type SessionSnapshot,
+} from "../../../../src/reducer/session-reducer";
+import type { LiveCapabilities } from "./protocol";
+import {
+ buildTranscript,
+ capabilityRows,
+ composerState,
+ goalAvailability,
+ runtimeRequestSubmission,
+ transcriptRole,
+ type TranscriptEntry,
+} from "./transcript-model";
+
+const FULL: LiveCapabilities = {
+ resume: true,
+ typedEvents: true,
+ steering: true,
+ interruption: true,
+ structuredResult: true,
+ runtimeRequestResolution: true,
+ goals: true,
+ threadLineage: true,
+};
+
+function seed(runId: string, sessionId: string): SessionSnapshot {
+ return createSessionSnapshotFromMetadata({
+ fixtureName: "live-console-demo",
+ identity: {
+ schema: "paperclip.prp.identity.v1",
+ companyId: "package-local-demo",
+ issueId: "live-console-demo",
+ runId,
+ environmentLeaseId: "package-local",
+ runnerInstanceId: "live-console-demo-server",
+ normalizedSessionId: sessionId,
+ driverSessionId: `thread-${sessionId}`,
+ },
+ capabilities: {
+ schema: "paperclip.prp.capabilities.v1",
+ sessionReusePolicy: "reuse_per_issue",
+ driver: { kind: "live-console-demo", version: "1" },
+ steer: true,
+ interrupt: true,
+ resume: true,
+ runtimeRequests: true,
+ structuredResult: true,
+ typedEvents: true,
+ },
+ });
+}
+
+async function drive(
+ manifestId: string,
+ interact?: (session: HarnessSession, turnId: string) => Promise,
+): Promise<{ events: PrpEvent[]; snapshot: SessionSnapshot }> {
+ const driver = new LiveConsoleScriptedDriver({ manifestId, chunkDelayMs: 0 });
+ const session = await driver.openSession({
+ runId: `run-${manifestId}`,
+ normalizedSessionId: `session-${manifestId}`,
+ workingDirectory: "/demo",
+ });
+ const events: PrpEvent[] = [];
+ const pump = (async () => {
+ for await (const event of session.events()) events.push(event);
+ })();
+ const { turnId } = await session.startTurn({ message: { role: "user", text: "Go." } });
+ if (interact !== undefined) await interact(session, turnId);
+ await new Promise((resolve) => setTimeout(resolve, 60));
+ await session.close({ reason: "test" });
+ await pump;
+ return {
+ events,
+ snapshot: events.reduce(applyPrpEvent, seed(`run-${manifestId}`, `session-${manifestId}`)),
+ };
+}
+
+function kinds(entries: readonly TranscriptEntry[]): string[] {
+ return entries.map((entry) => (entry.kind === "item" ? `item:${entry.role}` : entry.kind));
+}
+
+describe("transcriptRole", () => {
+ it("maps canonical item kinds onto transcript roles", () => {
+ expect(transcriptRole("reasoning")).toBe("reasoning");
+ expect(transcriptRole("agentMessage")).toBe("assistant");
+ expect(transcriptRole("commandExecution")).toBe("tool");
+ expect(transcriptRole("steering_acknowledgement")).toBe("steering");
+ expect(transcriptRole("thread_lineage")).toBe("lineage");
+ expect(transcriptRole("something_new")).toBe("system");
+ });
+});
+
+describe("buildTranscript", () => {
+ it("orders entries by the reducer timeline and shows the submitted message", async () => {
+ const { events, snapshot } = await drive("completion");
+ const entries = buildTranscript(snapshot, events);
+ expect(kinds(entries)).toEqual([
+ "user",
+ "item:reasoning",
+ "item:tool",
+ "item:assistant",
+ "turn",
+ ]);
+ const user = entries[0];
+ expect(user.kind === "user" && user.text).toBe("Go.");
+ const tool = entries[2];
+ expect(tool.kind === "item" && tool.input).toBe("docs/live-console-protocol-server.md");
+ expect(tool.kind === "item" && tool.output).toContain("server-sent events");
+ });
+
+ it("marks an interrupted item and never claims it completed cleanly", async () => {
+ const { events, snapshot } = await drive("interrupt-during-tool", async (session, turnId) => {
+ await new Promise((resolve) => setTimeout(resolve, 20));
+ await session.interrupt?.({ turnId });
+ });
+ const entries = buildTranscript(snapshot, events);
+ const tool = entries.find((entry) => entry.kind === "item" && entry.role === "tool");
+ expect(tool?.kind === "item" && tool.interrupted).toBe(true);
+ expect(tool?.kind === "item" && tool.output).toBeNull();
+ const turn = entries.find((entry) => entry.kind === "turn");
+ expect(turn?.kind === "turn" && turn.state).toBe("interrupted");
+ });
+
+ it("keeps a resolved request in the record with the chosen action", async () => {
+ const { events, snapshot } = await drive("approvals", async (session, turnId) => {
+ await new Promise((resolve) => setTimeout(resolve, 20));
+ await session.resolveRuntimeRequest?.({
+ requestId: "request-command",
+ turnId,
+ resolution: { action: "accept_for_session" },
+ });
+ });
+ const request = buildTranscript(snapshot, events).find(
+ (entry) => entry.kind === "request" && entry.requestId === "request-command",
+ );
+ expect(request?.kind === "request" && request.status).toBe("resolved");
+ expect(request?.kind === "request" && request.resolvedAction).toBe("accept_for_session");
+ expect(request?.kind === "request" && request.actions).toEqual([
+ "accept",
+ "accept_for_session",
+ "decline",
+ "cancel",
+ ]);
+ });
+
+ it("renders a failed item with its exact diagnostic", async () => {
+ const { events, snapshot } = await drive("failure");
+ const failed = buildTranscript(snapshot, events).find(
+ (entry) => entry.kind === "item" && entry.failure !== null,
+ );
+ expect(failed?.kind === "item" && failed.failure).toContain("cannot find module");
+ });
+
+ it("keeps a replayed prefix a real prefix of the finished transcript", async () => {
+ // Replay mode reduces events[0..n]. Stepping forward must only ever add
+ // entries in the same positions the live session produced.
+ const { events, snapshot } = await drive("completion");
+ const full = buildTranscript(snapshot, events);
+
+ for (let cut = 1; cut <= events.length; cut += 1) {
+ const prefix = events.slice(0, cut);
+ const stepped = buildTranscript(
+ prefix.reduce(applyPrpEvent, seed("run-completion", "session-completion")),
+ prefix,
+ );
+ expect(stepped.length).toBeLessThanOrEqual(full.length);
+ stepped.forEach((entry, index) => {
+ expect(entry.key).toBe(full[index]?.key);
+ expect(entry.position).toBe(full[index]?.position);
+ });
+ }
+ expect(
+ buildTranscript(
+ events.reduce(applyPrpEvent, seed("run-completion", "session-completion")),
+ events,
+ ),
+ ).toEqual(full);
+ });
+});
+
+describe("composerState", () => {
+ const base = {
+ connection: "connected" as const,
+ submitting: false,
+ interrupting: false,
+ activeTurnId: null,
+ sessionState: "running" as const,
+ };
+
+ it("shows exactly one state per protocol situation", () => {
+ expect(composerState(base)).toBe("idle");
+ expect(composerState({ ...base, submitting: true })).toBe("submitting");
+ expect(composerState({ ...base, activeTurnId: "turn-1" })).toBe("active-turn");
+ expect(composerState({ ...base, interrupting: true, activeTurnId: "turn-1" })).toBe(
+ "interrupting",
+ );
+ expect(composerState({ ...base, connection: "disconnected" })).toBe("disconnected");
+ expect(composerState({ ...base, sessionState: "closed" })).toBe("terminal");
+ });
+
+ it("puts a terminal session ahead of every other state", () => {
+ expect(
+ composerState({
+ ...base,
+ sessionState: "failed",
+ activeTurnId: "turn-1",
+ connection: "disconnected",
+ }),
+ ).toBe("terminal");
+ });
+});
+
+describe("goalAvailability", () => {
+ it("disables every verb with the exact diagnostic when goals are unsupported", () => {
+ const rows = goalAvailability(
+ {
+ ...FULL,
+ goals: false,
+ unsupported: ["Goal operations unsupported: app-server 0.132.0 does not advertise goals"],
+ },
+ null,
+ );
+ expect(rows).toHaveLength(5);
+ expect(rows.every((row) => !row.enabled)).toBe(true);
+ expect(rows[0]?.reason).toContain("does not advertise goals");
+ });
+
+ it("disables only the verbs that are invalid for the current goal state", () => {
+ const active = goalAvailability(FULL, "active");
+ expect(active.find((row) => row.operation === "pause")?.enabled).toBe(true);
+ expect(active.find((row) => row.operation === "resume")?.enabled).toBe(false);
+
+ const paused = goalAvailability(FULL, "paused");
+ expect(paused.find((row) => row.operation === "resume")?.enabled).toBe(true);
+ expect(paused.find((row) => row.operation === "pause")?.enabled).toBe(false);
+
+ const none = goalAvailability(FULL, null);
+ expect(none.find((row) => row.operation === "set")?.enabled).toBe(true);
+ expect(none.find((row) => row.operation === "clear")?.enabled).toBe(false);
+ });
+});
+
+describe("runtimeRequestSubmission", () => {
+ it("shapes each submit body the way its request kind is validated", async () => {
+ expect(runtimeRequestSubmission("user_input", "staging")).toEqual({
+ action: "submit",
+ answers: { answer: { answers: ["staging"] } },
+ });
+ expect(runtimeRequestSubmission("elicitation", "stable")).toEqual({
+ action: "submit",
+ content: { answer: "stable" },
+ });
+
+ // The driver is the authority on shape, so both bodies are checked against
+ // the real validator rather than against this test's expectations alone.
+ for (const [requestKind, answer] of [
+ ["user_input", "staging"],
+ ["elicitation", "stable"],
+ ] as const) {
+ expect(() =>
+ parseHarnessRuntimeRequestResolution(
+ requestKind,
+ runtimeRequestSubmission(requestKind, answer),
+ )).not.toThrow();
+ // ...and the other kind's body is rejected, which is what the browser
+ // used to send for elicitation.
+ const wrongKind = requestKind === "user_input" ? "elicitation" : "user_input";
+ expect(() =>
+ parseHarnessRuntimeRequestResolution(
+ requestKind,
+ runtimeRequestSubmission(wrongKind, answer),
+ )).toThrow(/rejected its resolution/);
+ }
+ });
+});
+
+describe("capabilityRows", () => {
+ it("gives every unsupported capability a diagnostic string", () => {
+ const rows = capabilityRows({ ...FULL, goals: false, usage: false });
+ expect(rows.filter((row) => !row.supported).every((row) => row.diagnostic !== null)).toBe(true);
+ expect(rows.find((row) => row.name === "steer")?.supported).toBe(true);
+ });
+});
diff --git a/packages/paperclip-runner/devtools/browser/src/live/transcript-model.ts b/packages/paperclip-runner/devtools/browser/src/live/transcript-model.ts
new file mode 100644
index 0000000000..9f0d853bcf
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/transcript-model.ts
@@ -0,0 +1,402 @@
+import type { PrpEvent } from "../../../../src/protocol/replay-contract";
+import type {
+ SessionItemSnapshot,
+ SessionSnapshot,
+} from "../../../../src/reducer/session-reducer";
+import type { GoalOperation, LiveCapabilities } from "./protocol";
+
+/**
+ * A projection of reducer state plus the canonical events that produced it.
+ * It never stores anything the protocol did not send, so a replayed session
+ * and a live session with the same events project identically.
+ */
+export type TranscriptRole =
+ | "user"
+ | "assistant"
+ | "reasoning"
+ | "tool"
+ | "steering"
+ | "interrupt"
+ | "goal"
+ | "lineage"
+ | "system";
+
+export interface TranscriptItemEntry {
+ kind: "item";
+ key: string;
+ position: number;
+ role: TranscriptRole;
+ item: SessionItemSnapshot;
+ interrupted: boolean;
+ streaming: boolean;
+ toolName: string | null;
+ input: string | null;
+ output: string | null;
+ failure: string | null;
+}
+
+export interface TranscriptUserEntry {
+ kind: "user";
+ key: string;
+ position: number;
+ turnId: string | null;
+ text: string;
+}
+
+export interface TranscriptRequestEntry {
+ kind: "request";
+ key: string;
+ position: number;
+ requestId: string;
+ requestKind: string;
+ type: string;
+ prompt: string;
+ status: "pending" | "resolved" | "expired" | "cancelled";
+ actions: string[];
+ details: Record;
+ resolvedAction: string | null;
+ resolvedAt: string | null;
+ turnId: string | null;
+}
+
+export interface TranscriptTurnEntry {
+ kind: "turn";
+ key: string;
+ position: number;
+ turnId: string | null;
+ state: "completed" | "failed" | "interrupted" | "cancelled";
+ detail: string;
+}
+
+export interface TranscriptDiagnosticEntry {
+ kind: "diagnostic";
+ key: string;
+ position: number;
+ message: string;
+ code: string | null;
+}
+
+export type TranscriptEntry =
+ | TranscriptItemEntry
+ | TranscriptUserEntry
+ | TranscriptRequestEntry
+ | TranscriptTurnEntry
+ | TranscriptDiagnosticEntry;
+
+const TOOL_KINDS = new Set([
+ "commandExecution",
+ "command_execution",
+ "fileChange",
+ "file_change",
+ "mcpToolCall",
+ "tool",
+ "webSearch",
+ "patchApply",
+]);
+
+const ASSISTANT_KINDS = new Set(["agentMessage", "assistant", "assistantMessage", "message"]);
+
+export function transcriptRole(kind: string): TranscriptRole {
+ if (kind === "reasoning") return "reasoning";
+ if (ASSISTANT_KINDS.has(kind)) return "assistant";
+ if (TOOL_KINDS.has(kind)) return "tool";
+ if (kind === "steering_acknowledgement") return "steering";
+ if (kind === "interrupt_acknowledgement") return "interrupt";
+ if (kind === "goal" || kind === "goal_cleared") return "goal";
+ if (kind === "thread_lineage") return "lineage";
+ return "system";
+}
+
+function payloadRecord(value: unknown): Record {
+ return typeof value === "object" && value !== null && !Array.isArray(value)
+ ? (value as Record)
+ : {};
+}
+
+function payloadText(value: unknown): string | null {
+ return typeof value === "string" && value.length > 0 ? value : null;
+}
+
+interface ItemFacts {
+ interrupted: boolean;
+ toolName: string | null;
+ input: string | null;
+ output: string | null;
+ failure: string | null;
+ terminal: boolean;
+}
+
+function collectItemFacts(events: readonly PrpEvent[]): Map {
+ const facts = new Map();
+ for (const event of events) {
+ if (event.itemId === undefined || !event.eventType.startsWith("item.")) continue;
+ const payload = payloadRecord(event.payload);
+ const current = facts.get(event.itemId) ?? {
+ interrupted: false,
+ toolName: null,
+ input: null,
+ output: null,
+ failure: null,
+ terminal: false,
+ };
+ if (payload.interrupted === true || payload.status === "interrupted") {
+ current.interrupted = true;
+ }
+ current.toolName = payloadText(payload.toolName) ?? current.toolName;
+ current.input = payloadText(payload.input) ?? current.input;
+ current.output = payloadText(payload.output) ?? current.output;
+ if (event.eventType === "item.failed") {
+ current.failure = payloadText(payload.message) ?? "The item failed.";
+ }
+ if (event.eventType === "item.completed" || event.eventType === "item.failed") {
+ current.terminal = true;
+ }
+ facts.set(event.itemId, current);
+ }
+ return facts;
+}
+
+/**
+ * Builds the submit body a request kind actually accepts. Elicitation answers
+ * travel as `content` and user-input answers as `answers`; the runner validates
+ * both against the request kind, so sending the wrong one fails closed rather
+ * than resolving the request with an empty response.
+ */
+export function runtimeRequestSubmission(
+ requestKind: string,
+ answer: string,
+): Record {
+ if (requestKind === "elicitation") {
+ return { action: "submit", content: { answer } };
+ }
+ return { action: "submit", answers: { answer: { answers: [answer] } } };
+}
+
+function requestActions(details: unknown): string[] {
+ const record = payloadRecord(details);
+ return Array.isArray(record.actions)
+ ? record.actions.filter((action): action is string => typeof action === "string")
+ : [];
+}
+
+/**
+ * Builds the ordered transcript. Order comes from the reducer timeline, which
+ * is authoritative; nothing is re-sorted client-side.
+ */
+export function buildTranscript(
+ snapshot: SessionSnapshot,
+ events: readonly PrpEvent[],
+): TranscriptEntry[] {
+ const facts = collectItemFacts(events);
+ const items = new Map(snapshot.items.map((item) => [item.itemId, item]));
+ const seenItems = new Set();
+ const entries: TranscriptEntry[] = [];
+ const eventsById = new Map(events.map((event) => [event.sourceEventId, event]));
+
+ for (const timelineEntry of snapshot.timeline) {
+ const event = eventsById.get(timelineEntry.sourceEventId);
+ const payload = payloadRecord(event?.payload);
+ const position = timelineEntry.position;
+
+ if (timelineEntry.eventType === "turn.submitted") {
+ entries.push({
+ kind: "user",
+ key: `user:${timelineEntry.sourceEventId}`,
+ position,
+ turnId: typeof payload.turnId === "string" ? payload.turnId : null,
+ text: payloadText(payload.text) ?? "(no message text in the canonical event)",
+ });
+ continue;
+ }
+
+ if (
+ ["turn.completed", "turn.failed", "turn.interrupted", "turn.cancelled"].includes(
+ timelineEntry.eventType,
+ )
+ ) {
+ const state = timelineEntry.eventType.split(".").at(-1) as TranscriptTurnEntry["state"];
+ entries.push({
+ kind: "turn",
+ key: `turn:${timelineEntry.sourceEventId}`,
+ position,
+ turnId: typeof payload.turnId === "string" ? payload.turnId : null,
+ state,
+ detail:
+ payloadText(payload.message) ??
+ payloadText(payload.reason) ??
+ timelineEntry.summary,
+ });
+ continue;
+ }
+
+ if (timelineEntry.eventType === "runtime_request.created") {
+ const request = payloadRecord(payload.request ?? payload);
+ const requestId = typeof request.requestId === "string" ? request.requestId : "";
+ const projected = snapshot.requests.find(
+ (candidate) => candidate.requestId === requestId,
+ );
+ const resolution = events.find(
+ (candidate) =>
+ candidate.eventType.startsWith("runtime_request.") &&
+ candidate.eventType !== "runtime_request.created" &&
+ payloadRecord(candidate.payload).requestId === requestId,
+ );
+ entries.push({
+ kind: "request",
+ key: `request:${requestId}`,
+ position,
+ requestId,
+ requestKind: typeof request.requestKind === "string" ? request.requestKind : "runtime",
+ type: typeof request.type === "string" ? request.type : "runtime",
+ prompt: payloadText(request.prompt) ?? "Runtime request",
+ status: (projected?.status ?? "pending") as TranscriptRequestEntry["status"],
+ actions: requestActions(request),
+ details: payloadRecord(request.details),
+ resolvedAction: payloadText(payloadRecord(resolution?.payload).action),
+ resolvedAt: resolution?.emittedAt ?? null,
+ turnId: typeof request.turnId === "string" ? request.turnId : null,
+ });
+ continue;
+ }
+
+ if (
+ timelineEntry.eventType === "harness.diagnostic" ||
+ timelineEntry.eventType === "runner.diagnostic"
+ ) {
+ entries.push({
+ kind: "diagnostic",
+ key: `diagnostic:${timelineEntry.sourceEventId}`,
+ position,
+ message: payloadText(payload.message) ?? timelineEntry.summary,
+ code: payloadText(payload.code),
+ });
+ continue;
+ }
+
+ const itemId = timelineEntry.itemId;
+ if (itemId === undefined || seenItems.has(itemId)) continue;
+ const item = items.get(itemId);
+ if (item === undefined) continue;
+ seenItems.add(itemId);
+ const itemFacts = facts.get(itemId);
+ entries.push({
+ kind: "item",
+ key: `item:${itemId}`,
+ position,
+ role: transcriptRole(item.kind),
+ item,
+ interrupted: itemFacts?.interrupted ?? false,
+ streaming: item.status === "running" && itemFacts?.terminal !== true,
+ toolName: itemFacts?.toolName ?? null,
+ input: itemFacts?.input ?? null,
+ output: itemFacts?.output ?? null,
+ failure: itemFacts?.failure ?? null,
+ });
+ }
+
+ return entries;
+}
+
+export type ComposerState =
+ | "idle"
+ | "submitting"
+ | "active-turn"
+ | "interrupting"
+ | "disconnected"
+ | "terminal";
+
+export interface ComposerStateInput {
+ connection: "idle" | "connecting" | "connected" | "reconnecting" | "disconnected" | "terminal";
+ submitting: boolean;
+ interrupting: boolean;
+ activeTurnId: string | null;
+ sessionState: SessionSnapshot["sessionState"];
+}
+
+/** One state machine with exactly one visible primary action. */
+export function composerState(input: ComposerStateInput): ComposerState {
+ if (input.sessionState === "closed" || input.sessionState === "failed") return "terminal";
+ if (input.connection === "disconnected" || input.connection === "reconnecting") {
+ return "disconnected";
+ }
+ if (input.interrupting) return "interrupting";
+ if (input.activeTurnId !== null) return "active-turn";
+ if (input.submitting) return "submitting";
+ return "idle";
+}
+
+export interface GoalOperationAvailability {
+ operation: GoalOperation;
+ enabled: boolean;
+ reason: string | null;
+}
+
+const GOAL_UNSUPPORTED_FALLBACK =
+ "Goal operations unsupported: this app-server does not advertise the goals capability";
+
+/**
+ * Capability gating for the goal menu. An unsupported verb is disabled with
+ * the exact upstream diagnostic — never hidden and never emulated.
+ */
+export function goalAvailability(
+ capabilities: LiveCapabilities,
+ goalStatus: string | null,
+): GoalOperationAvailability[] {
+ const unsupported = capabilities.unsupported ?? [];
+ if (capabilities.goals !== true) {
+ const reason = unsupported[0] ?? GOAL_UNSUPPORTED_FALLBACK;
+ return (["set", "get", "pause", "resume", "clear"] as GoalOperation[]).map((operation) => ({
+ operation,
+ enabled: false,
+ reason,
+ }));
+ }
+ return (["set", "get", "pause", "resume", "clear"] as GoalOperation[]).map((operation) => {
+ if (operation === "set") return { operation, enabled: true, reason: null };
+ if (goalStatus === null) {
+ return {
+ operation,
+ enabled: false,
+ reason: "No goal is set on this thread yet.",
+ };
+ }
+ if (operation === "pause" && goalStatus !== "active") {
+ return { operation, enabled: false, reason: `The goal is already ${goalStatus}.` };
+ }
+ if (operation === "resume" && goalStatus === "active") {
+ return { operation, enabled: false, reason: "The goal is already active." };
+ }
+ return { operation, enabled: true, reason: null };
+ });
+}
+
+export interface CapabilityRow {
+ name: string;
+ supported: boolean;
+ diagnostic: string | null;
+}
+
+export function capabilityRows(capabilities: LiveCapabilities): CapabilityRow[] {
+ const unsupported = capabilities.unsupported ?? [];
+ const rows: Array<[string, boolean | undefined]> = [
+ ["steer", capabilities.steering],
+ ["interrupt", capabilities.interruption],
+ ["resume", capabilities.resume],
+ ["runtime requests", capabilities.runtimeRequestResolution],
+ ["runtime request handoff", capabilities.runtimeRequestHandoff],
+ ["goals", capabilities.goals],
+ ["child threads", capabilities.threadLineage],
+ ["structured result", capabilities.structuredResult],
+ ["typed events", capabilities.typedEvents],
+ ["usage", capabilities.usage],
+ ];
+ return rows.map(([name, supported]) => ({
+ name,
+ supported: supported === true,
+ diagnostic:
+ supported === true
+ ? null
+ : (unsupported.find((entry) => entry.toLowerCase().includes(name.split(" ")[0]!)) ??
+ `${name} is not advertised by this app-server`),
+ }));
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/live/use-live-console.test.ts b/packages/paperclip-runner/devtools/browser/src/live/use-live-console.test.ts
new file mode 100644
index 0000000000..43d7b46fac
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/use-live-console.test.ts
@@ -0,0 +1,25 @@
+import { describe, expect, it } from "vitest";
+
+import type { PrpEvent } from "../../../../src/protocol/replay-contract";
+import {
+ acceptLiveEvent,
+ createLiveEventCursor,
+} from "./use-live-console";
+
+function event(sourceEventId: string): PrpEvent {
+ return { sourceEventId } as PrpEvent;
+}
+
+describe("Live console reconnect cursor", () => {
+ it("does not advance past an unseen event when replay delivers a duplicate", () => {
+ const first = event("source:event-1");
+ const second = event("source:event-2");
+ const cursor = createLiveEventCursor([first], 1);
+
+ expect(acceptLiveEvent(cursor, first)).toBe(false);
+ expect(cursor.cursor).toBe(1);
+
+ expect(acceptLiveEvent(cursor, second)).toBe(true);
+ expect(cursor.cursor).toBe(2);
+ });
+});
diff --git a/packages/paperclip-runner/devtools/browser/src/live/use-live-console.ts b/packages/paperclip-runner/devtools/browser/src/live/use-live-console.ts
new file mode 100644
index 0000000000..a2c6a903a9
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/live/use-live-console.ts
@@ -0,0 +1,579 @@
+import { useCallback, useEffect, useMemo, useRef, useState } from "react";
+
+import type { PrpEvent } from "../../../../src/protocol/replay-contract";
+import {
+ applyPrpEvent,
+ createSessionSnapshotFromMetadata,
+ type SessionSnapshot,
+} from "../../../../src/reducer/session-reducer";
+import * as client from "./client";
+import type {
+ GoalOperation,
+ LiveConnectionStatus,
+ LiveManifestSummary,
+ LiveSessionState,
+} from "./protocol";
+import {
+ buildTranscript,
+ composerState,
+ type ComposerState,
+ type TranscriptEntry,
+} from "./transcript-model";
+
+const SESSION_STORAGE_KEY = "paperclip-runner.live-console.session";
+const RECONNECT_DELAY_MS = 900;
+const MAX_RECONNECT_ATTEMPTS = 5;
+
+export interface LiveEventCursor {
+ cursor: number;
+ seenSourceEventIds: Set;
+}
+
+export function createLiveEventCursor(
+ events: readonly PrpEvent[] = [],
+ cursor = 0,
+): LiveEventCursor {
+ return {
+ cursor,
+ seenSourceEventIds: new Set(events.map((event) => event.sourceEventId)),
+ };
+}
+
+export function acceptLiveEvent(cursor: LiveEventCursor, event: PrpEvent): boolean {
+ if (cursor.seenSourceEventIds.has(event.sourceEventId)) return false;
+ cursor.seenSourceEventIds.add(event.sourceEventId);
+ cursor.cursor += 1;
+ return true;
+}
+
+export type SteeringChipStatus = "pending" | "acknowledged" | "rejected" | "failed";
+
+export interface SteeringChip {
+ id: string;
+ text: string;
+ expectedTurnId: string;
+ status: SteeringChipStatus;
+ detail: string | null;
+}
+
+function reduceEvents(
+ state: LiveSessionState | null,
+ events: readonly PrpEvent[],
+): SessionSnapshot | null {
+ if (state === null) return null;
+ const seed = createSessionSnapshotFromMetadata({
+ fixtureName: state.snapshot.fixtureName,
+ identity: state.snapshot.identity,
+ capabilities: state.snapshot.capabilities,
+ });
+ return events.reduce(applyPrpEvent, seed);
+}
+
+function errorMessage(cause: unknown): string {
+ if (cause instanceof client.LiveConsoleError) return cause.message;
+ return cause instanceof Error ? cause.message : String(cause);
+}
+
+export interface LiveConsole {
+ manifests: LiveManifestSummary[];
+ selectedManifestId: string;
+ selectManifest: (id: string) => void;
+ state: LiveSessionState | null;
+ events: PrpEvent[];
+ snapshot: SessionSnapshot | null;
+ transcript: TranscriptEntry[];
+ connection: LiveConnectionStatus;
+ reconnectAttempt: number;
+ replayParity: boolean | null;
+ composer: ComposerState;
+ steeringChips: SteeringChip[];
+ announcement: string;
+ error: string | null;
+ busy: boolean;
+ selectedThreadId: string | null;
+ selectThread: (threadId: string | null) => void;
+ replay: {
+ active: boolean;
+ position: number;
+ length: number;
+ playing: boolean;
+ enter: () => void;
+ exit: () => void;
+ setPosition: (position: number) => void;
+ togglePlay: () => void;
+ step: (delta: number) => void;
+ };
+ start: (input: { manifestId: string; message: string }) => Promise;
+ send: (text: string) => Promise;
+ steer: (text: string) => Promise;
+ interrupt: () => Promise;
+ resolve: (
+ requestId: string,
+ turnId: string,
+ resolution: Record,
+ ) => Promise;
+ goal: (operation: GoalOperation, body?: Record) => Promise;
+ dropConnection: () => void;
+ retryNow: () => void;
+ reset: () => Promise;
+ dismissSteeringChip: (id: string) => void;
+}
+
+export function useLiveConsole(): LiveConsole {
+ const [manifests, setManifests] = useState([]);
+ const [selectedManifestId, setSelectedManifestId] = useState("completion");
+ const [state, setState] = useState(null);
+ const [events, setEvents] = useState([]);
+ const [connection, setConnection] = useState("idle");
+ const [reconnectAttempt, setReconnectAttempt] = useState(0);
+ const [steeringChips, setSteeringChips] = useState([]);
+ const [announcement, setAnnouncement] = useState("");
+ const [error, setError] = useState(null);
+ const [busy, setBusy] = useState(false);
+ const [interrupting, setInterrupting] = useState(false);
+ const [submitting, setSubmitting] = useState(false);
+ const [selectedThreadId, setSelectedThreadId] = useState(null);
+ const [replayActive, setReplayActive] = useState(false);
+ const [replayPosition, setReplayPosition] = useState(0);
+ const [replayPlaying, setReplayPlaying] = useState(false);
+
+ const streamRef = useRef(null);
+ const eventCursorRef = useRef(createLiveEventCursor());
+ const sessionIdRef = useRef(null);
+ const retryTimerRef = useRef | null>(null);
+
+ const closeStream = useCallback(() => {
+ streamRef.current?.close();
+ streamRef.current = null;
+ if (retryTimerRef.current !== null) {
+ clearTimeout(retryTimerRef.current);
+ retryTimerRef.current = null;
+ }
+ }, []);
+
+ const subscribe = useCallback(
+ (sessionId: string, attempt = 0) => {
+ closeStream();
+ setConnection(attempt === 0 ? "connecting" : "reconnecting");
+ setReconnectAttempt(attempt);
+ streamRef.current = client.openEventStream(sessionId, eventCursorRef.current.cursor, {
+ onOpen: () => {
+ setConnection("connected");
+ setReconnectAttempt(0);
+ setAnnouncement("Connected");
+ },
+ onEvent: (event) => {
+ if (!acceptLiveEvent(eventCursorRef.current, event)) return;
+ setEvents((current) => [...current, event]);
+ },
+ onError: () => {
+ closeStream();
+ if (attempt >= MAX_RECONNECT_ATTEMPTS) {
+ setConnection("disconnected");
+ setAnnouncement("Disconnected");
+ return;
+ }
+ setConnection("reconnecting");
+ setReconnectAttempt(attempt + 1);
+ setAnnouncement(`Reconnecting, attempt ${attempt + 1}`);
+ retryTimerRef.current = setTimeout(
+ () => subscribe(sessionId, attempt + 1),
+ RECONNECT_DELAY_MS,
+ );
+ },
+ });
+ },
+ [closeStream],
+ );
+
+ const adopt = useCallback(
+ (next: LiveSessionState) => {
+ setState(next);
+ setError(null);
+ },
+ [],
+ );
+
+ const attach = useCallback(
+ async (sessionId: string, resume: boolean) => {
+ const [next, history] = await Promise.all([
+ client.readSession(sessionId),
+ client.readEvents(sessionId, 0),
+ ]);
+ sessionIdRef.current = sessionId;
+ window.sessionStorage.setItem(SESSION_STORAGE_KEY, sessionId);
+ adopt(next);
+ setEvents(history.events);
+ eventCursorRef.current = createLiveEventCursor(history.events, history.cursor);
+ if (resume) setAnnouncement("Replayed the durable transcript");
+ subscribe(sessionId, 0);
+ },
+ [adopt, subscribe],
+ );
+
+ useEffect(() => {
+ let cancelled = false;
+ void client
+ .fetchManifests()
+ .then((list) => {
+ if (cancelled) return;
+ setManifests(list);
+ if (list[0] !== undefined) setSelectedManifestId((current) => current || list[0]!.id);
+ })
+ .catch((cause) => {
+ if (!cancelled) setError(errorMessage(cause));
+ });
+ const stored = window.sessionStorage.getItem(SESSION_STORAGE_KEY);
+ if (stored !== null) {
+ void attach(stored, true).catch(() => {
+ window.sessionStorage.removeItem(SESSION_STORAGE_KEY);
+ });
+ }
+ return () => {
+ cancelled = true;
+ };
+ // Runs once: session restoration is a mount concern.
+ // eslint-disable-next-line react-hooks/exhaustive-deps
+ }, []);
+
+ useEffect(() => () => closeStream(), [closeStream]);
+
+ // Lineage, goal, pending requests, and the server cursor live in the public
+ // state, so a stream that only carries events would leave them stale.
+ useEffect(() => {
+ const sessionId = sessionIdRef.current;
+ if (sessionId === null || events.length === 0) return undefined;
+ const timer = setTimeout(() => {
+ void client
+ .readSession(sessionId)
+ .then((next) => setState(next))
+ .catch(() => undefined);
+ }, 150);
+ return () => clearTimeout(timer);
+ }, [events.length]);
+
+ const liveSnapshot = useMemo(() => reduceEvents(state, events), [state, events]);
+ const visibleEvents = useMemo(
+ () => (replayActive ? events.slice(0, replayPosition) : events),
+ [events, replayActive, replayPosition],
+ );
+ const snapshot = useMemo(
+ () => (replayActive ? reduceEvents(state, visibleEvents) : liveSnapshot),
+ [replayActive, state, visibleEvents, liveSnapshot],
+ );
+
+ const replayParity = useMemo(() => {
+ if (state === null || liveSnapshot === null) return null;
+ if (state.cursor !== events.length) return null;
+ return (
+ JSON.stringify(liveSnapshot.timeline) === JSON.stringify(state.snapshot.timeline) &&
+ liveSnapshot.integrity === state.snapshot.integrity
+ );
+ }, [state, liveSnapshot, events.length]);
+
+ const transcript = useMemo(
+ () => (snapshot === null ? [] : buildTranscript(snapshot, visibleEvents)),
+ [snapshot, visibleEvents],
+ );
+
+ useEffect(() => {
+ if (!replayPlaying) return undefined;
+ if (replayPosition >= events.length) {
+ setReplayPlaying(false);
+ return undefined;
+ }
+ const timer = setTimeout(() => setReplayPosition((current) => current + 1), 140);
+ return () => clearTimeout(timer);
+ }, [replayPlaying, replayPosition, events.length]);
+
+ // Steering chips resolve against canonical acknowledgements only.
+ useEffect(() => {
+ setSteeringChips((chips) => {
+ if (chips.every((chip) => chip.status !== "pending")) return chips;
+ const acknowledged = events.filter(
+ (event) =>
+ event.eventType === "item.completed" &&
+ (event.payload as { kind?: string }).kind === "steering_acknowledgement",
+ );
+ let index = 0;
+ return chips.map((chip) => {
+ if (chip.status !== "pending") return chip;
+ const match = acknowledged[index];
+ index += 1;
+ return match === undefined
+ ? chip
+ : { ...chip, status: "acknowledged" as const, detail: null };
+ });
+ });
+ }, [events]);
+
+ // The reducer is authoritative for the active turn. Falling back to the
+ // last public-state read would keep a finished turn alive in the composer.
+ const activeTurnId = snapshot === null ? (state?.activeTurnId ?? null) : snapshot.activeTurnId;
+
+ // A submitted turn that has not been accepted yet is a protocol fact, not a
+ // local flag: it is what makes interrupt-before-start reachable.
+ const awaitingTurnStart = useMemo(() => {
+ let pending = false;
+ for (const entry of snapshot?.timeline ?? []) {
+ if (entry.eventType === "turn.submitted") pending = true;
+ else if (entry.eventType.startsWith("turn.")) pending = false;
+ }
+ return pending;
+ }, [snapshot]);
+
+ const composer = composerState({
+ connection,
+ submitting: submitting || awaitingTurnStart,
+ interrupting,
+ activeTurnId,
+ sessionState: snapshot?.sessionState ?? "not_started",
+ });
+
+ const guard = useCallback(async (task: () => Promise) => {
+ setBusy(true);
+ try {
+ await task();
+ } catch (cause) {
+ setError(errorMessage(cause));
+ } finally {
+ setBusy(false);
+ }
+ }, []);
+
+ const start = useCallback(
+ async (input: { manifestId: string; message: string }) => {
+ await guard(async () => {
+ closeStream();
+ setEvents([]);
+ eventCursorRef.current = createLiveEventCursor();
+ setSteeringChips([]);
+ setReplayActive(false);
+ setReplayPosition(0);
+ setSelectedThreadId(null);
+ setSubmitting(true);
+ const manifest = manifests.find((entry) => entry.id === input.manifestId);
+ try {
+ const created = await client.createSession({
+ manifest: input.manifestId,
+ objective: manifest?.objective ?? "Run the Live console demo.",
+ message: input.message,
+ });
+ sessionIdRef.current = created.sessionId;
+ window.sessionStorage.setItem(SESSION_STORAGE_KEY, created.sessionId);
+ adopt(created);
+ setAnnouncement("Turn running");
+ subscribe(created.sessionId, 0);
+ } finally {
+ setSubmitting(false);
+ }
+ });
+ },
+ [adopt, closeStream, guard, manifests, subscribe],
+ );
+
+ const refreshState = useCallback(async () => {
+ const sessionId = sessionIdRef.current;
+ if (sessionId === null) return;
+ adopt(await client.readSession(sessionId));
+ }, [adopt]);
+
+ const send = useCallback(
+ async (text: string) => {
+ const sessionId = sessionIdRef.current;
+ if (sessionId === null) {
+ await start({ manifestId: selectedManifestId, message: text });
+ return;
+ }
+ await guard(async () => {
+ setSubmitting(true);
+ try {
+ adopt(await client.startTurn(sessionId, text));
+ setAnnouncement("Turn running");
+ } finally {
+ setSubmitting(false);
+ }
+ });
+ },
+ [adopt, guard, selectedManifestId, start],
+ );
+
+ const steer = useCallback(
+ async (text: string) => {
+ const sessionId = sessionIdRef.current;
+ const expectedTurnId = activeTurnId;
+ if (sessionId === null || expectedTurnId === null) return;
+ const chipId = `steer-${Date.now()}-${steeringChips.length}`;
+ setSteeringChips((chips) => [
+ ...chips,
+ { id: chipId, text, expectedTurnId, status: "pending", detail: null },
+ ]);
+ try {
+ adopt(await client.steerTurn(sessionId, expectedTurnId, text));
+ setAnnouncement("Steering sent");
+ } catch (cause) {
+ const stale =
+ cause instanceof client.LiveConsoleError &&
+ (cause.code === "stale_turn" || cause.code === "already_terminal");
+ setSteeringChips((chips) =>
+ chips.map((chip) =>
+ chip.id === chipId
+ ? {
+ ...chip,
+ status: stale ? "rejected" : "failed",
+ detail: stale ? "turn already ended" : errorMessage(cause),
+ }
+ : chip,
+ ),
+ );
+ setAnnouncement(stale ? "Steering rejected: turn already ended" : "Steering failed");
+ await refreshState().catch(() => undefined);
+ }
+ },
+ [activeTurnId, adopt, refreshState, steeringChips.length],
+ );
+
+ const interrupt = useCallback(async () => {
+ const sessionId = sessionIdRef.current;
+ if (sessionId === null) return;
+ setInterrupting(true);
+ setAnnouncement("Stopping");
+ try {
+ adopt(await client.interruptTurn(sessionId, activeTurnId));
+ setAnnouncement("Turn stopped");
+ } catch (cause) {
+ setError(errorMessage(cause));
+ } finally {
+ setInterrupting(false);
+ }
+ }, [activeTurnId, adopt]);
+
+ const resolve = useCallback(
+ async (requestId: string, turnId: string, resolution: Record) => {
+ const sessionId = sessionIdRef.current;
+ if (sessionId === null) return;
+ await guard(async () => {
+ adopt(await client.resolveRequest(sessionId, requestId, turnId, resolution));
+ setAnnouncement("Request resolved");
+ });
+ },
+ [adopt, guard],
+ );
+
+ const goal = useCallback(
+ async (operation: GoalOperation, body: Record = {}) => {
+ const sessionId = sessionIdRef.current;
+ if (sessionId === null) return;
+ await guard(async () => {
+ adopt(await client.goalOperation(sessionId, operation, body));
+ setAnnouncement(`Goal ${operation}`);
+ });
+ },
+ [adopt, guard],
+ );
+
+ const dropConnection = useCallback(() => {
+ closeStream();
+ setConnection("disconnected");
+ setAnnouncement("Connection lost");
+ }, [closeStream]);
+
+ const retryNow = useCallback(() => {
+ const sessionId = sessionIdRef.current;
+ if (sessionId === null) return;
+ void guard(async () => {
+ const resumed = await client.reconnectSession(sessionId);
+ adopt(resumed);
+ const history = await client.readEvents(sessionId, 0);
+ setEvents(history.events);
+ eventCursorRef.current = createLiveEventCursor(history.events, history.cursor);
+ subscribe(sessionId, 0);
+ setAnnouncement("Resumed the same session");
+ });
+ }, [adopt, guard, subscribe]);
+
+ const reset = useCallback(async () => {
+ const sessionId = sessionIdRef.current;
+ closeStream();
+ if (sessionId !== null) {
+ try {
+ await client.closeSession(sessionId);
+ } catch (cause) {
+ setConnection("disconnected");
+ setError(errorMessage(cause));
+ setAnnouncement("Demo reset failed");
+ return;
+ }
+ }
+ window.sessionStorage.removeItem(SESSION_STORAGE_KEY);
+ sessionIdRef.current = null;
+ eventCursorRef.current = createLiveEventCursor();
+ setState(null);
+ setEvents([]);
+ setSteeringChips([]);
+ setConnection("idle");
+ setReplayActive(false);
+ setReplayPosition(0);
+ setSelectedThreadId(null);
+ setError(null);
+ setAnnouncement("Demo state reset");
+ }, [closeStream]);
+
+ const replay = useMemo(
+ () => ({
+ active: replayActive,
+ position: replayPosition,
+ length: events.length,
+ playing: replayPlaying,
+ enter: () => {
+ setReplayActive(true);
+ setReplayPosition(0);
+ setReplayPlaying(false);
+ },
+ exit: () => {
+ setReplayActive(false);
+ setReplayPlaying(false);
+ },
+ setPosition: (position: number) =>
+ setReplayPosition(Math.min(Math.max(position, 0), events.length)),
+ togglePlay: () => setReplayPlaying((current) => !current),
+ step: (delta: number) =>
+ setReplayPosition((current) =>
+ Math.min(Math.max(current + delta, 0), events.length),
+ ),
+ }),
+ [events.length, replayActive, replayPlaying, replayPosition],
+ );
+
+ return {
+ manifests,
+ selectedManifestId,
+ selectManifest: setSelectedManifestId,
+ state,
+ events,
+ snapshot,
+ transcript,
+ connection,
+ reconnectAttempt,
+ replayParity,
+ composer,
+ steeringChips,
+ announcement,
+ error,
+ busy,
+ selectedThreadId,
+ selectThread: setSelectedThreadId,
+ replay,
+ start,
+ send,
+ steer,
+ interrupt,
+ resolve,
+ goal,
+ dropConnection,
+ retryNow,
+ reset,
+ dismissSteeringChip: (id: string) =>
+ setSteeringChips((chips) => chips.filter((chip) => chip.id !== id)),
+ };
+}
diff --git a/packages/paperclip-runner/devtools/browser/src/main.tsx b/packages/paperclip-runner/devtools/browser/src/main.tsx
new file mode 100644
index 0000000000..8bcbe49a01
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/main.tsx
@@ -0,0 +1,16 @@
+import { StrictMode } from "react";
+import { createRoot } from "react-dom/client";
+
+import { App } from "./App";
+import "./styles.css";
+
+const root = document.getElementById("root");
+if (root === null) {
+ throw new Error("Browser replay root element is missing");
+}
+
+createRoot(root).render(
+
+
+ ,
+);
diff --git a/packages/paperclip-runner/devtools/browser/src/styles.css b/packages/paperclip-runner/devtools/browser/src/styles.css
new file mode 100644
index 0000000000..a06ecfa662
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/src/styles.css
@@ -0,0 +1,1430 @@
+:root {
+ color-scheme: light;
+ --background: #f7f7f5;
+ --foreground: #191917;
+ --card: #ffffff;
+ --card-foreground: #191917;
+ --muted: #efefeb;
+ --muted-foreground: #66665f;
+ --border: #deded7;
+ --primary: #242420;
+ --primary-foreground: #ffffff;
+ --success: #17633a;
+ --success-surface: #e9f5ed;
+ --warning: #8a5a10;
+ --warning-surface: #fff4dc;
+ --danger: #a02f2f;
+ --danger-surface: #fceaea;
+ /* Pending and streaming are not warnings; Live console needs their own pair. */
+ --accent: #2f5fa8;
+ --accent-surface: #e9f0fa;
+ --surface-raised: #fcfcfa;
+ --ring: #66665f;
+ --font-sans: Inter, ui-sans-serif, system-ui, sans-serif;
+ --font-mono: "SFMono-Regular", Consolas, "Liberation Mono", monospace;
+ --font-size-xs: 0.75rem;
+ --font-size-sm: 0.875rem;
+ --font-size-md: 1rem;
+ --font-size-lg: 1.25rem;
+ --font-size-xl: 1.75rem;
+ --line-tight: 1.2;
+ --line-normal: 1.5;
+ --weight-medium: 500;
+ --weight-semibold: 600;
+ --space-1: 0.25rem;
+ --space-2: 0.5rem;
+ --space-3: 0.75rem;
+ --space-4: 1rem;
+ --space-5: 1.25rem;
+ --space-6: 1.5rem;
+ --space-8: 2rem;
+ --space-10: 2.5rem;
+ --radius-sm: 0.375rem;
+ --radius-md: 0.625rem;
+ --shadow-sm: 0 0.0625rem 0.125rem rgb(0 0 0 / 0.05);
+ --editor-height: 32rem;
+ --page-width: 90rem;
+ --control-height: 2.5rem;
+ --timeline-column: 2rem;
+ --motion-fast: 120ms;
+ --motion-medium: 200ms;
+ --z-banner: 10;
+ --z-dialog: 20;
+ --rail-width: 17rem;
+ --center-width: 48rem;
+ --inspector-width: 24rem;
+ --transcript-height: 34rem;
+ --touch-target: 2.75rem;
+ --facet-height: 6rem;
+}
+
+* {
+ box-sizing: border-box;
+}
+
+body {
+ margin: 0;
+ min-width: 20rem;
+ background: var(--background);
+ color: var(--foreground);
+ font-family: var(--font-sans);
+ font-size: var(--font-size-sm);
+ line-height: var(--line-normal);
+}
+
+button,
+input,
+select,
+textarea {
+ font: inherit;
+}
+
+h1,
+h2,
+h3,
+p {
+ margin: 0;
+}
+
+.app-shell {
+ width: min(100%, var(--page-width));
+ margin-inline: auto;
+ padding: var(--space-8);
+}
+
+.page-header,
+.snapshot-heading,
+.timeline-heading,
+.editor-actions,
+.timeline-title {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--space-4);
+}
+
+.page-header {
+ align-items: flex-start;
+ margin-bottom: var(--space-8);
+}
+
+.mode-switch,
+.scenario-actions,
+.live-status,
+.parity-line {
+ display: flex;
+ align-items: center;
+ gap: var(--space-3);
+}
+
+.mode-switch {
+ flex-wrap: wrap;
+ margin-bottom: var(--space-6);
+}
+
+.mode-switch .ui-button[aria-pressed="false"],
+.scenario-actions .ui-button + .ui-button {
+ background: var(--muted);
+ color: var(--muted-foreground);
+}
+
+.page-header > div {
+ display: grid;
+ gap: var(--space-2);
+}
+
+.page-header h1 {
+ font-size: var(--font-size-xl);
+ line-height: var(--line-tight);
+ letter-spacing: -0.02em;
+}
+
+.page-header p:not(.eyebrow),
+.ui-card-description,
+.timeline-heading span,
+.editor-actions span,
+dt {
+ color: var(--muted-foreground);
+}
+
+.eyebrow {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+ letter-spacing: 0.08em;
+ text-transform: uppercase;
+}
+
+.workspace-grid {
+ display: grid;
+ grid-template-columns: minmax(22rem, 0.8fr) minmax(28rem, 1.2fr);
+ gap: var(--space-8);
+ align-items: start;
+}
+
+.ui-card {
+ display: grid;
+ gap: var(--space-5);
+ padding: var(--space-6);
+ background: var(--card);
+ color: var(--card-foreground);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+ box-shadow: var(--shadow-sm);
+}
+
+.ui-card-header,
+.ui-card-content,
+.editor-content {
+ display: grid;
+ gap: var(--space-3);
+}
+
+.ui-card-title,
+.snapshot-heading h2 {
+ font-size: var(--font-size-lg);
+ line-height: var(--line-tight);
+ font-weight: var(--weight-semibold);
+}
+
+.ui-card-description {
+ font-size: var(--font-size-sm);
+}
+
+.ui-button {
+ min-height: var(--control-height);
+ padding: var(--space-2) var(--space-4);
+ border: 0;
+ border-radius: var(--radius-sm);
+ background: var(--primary);
+ color: var(--primary-foreground);
+ font-weight: var(--weight-medium);
+ cursor: pointer;
+ transition: opacity var(--motion-fast) ease;
+}
+
+.ui-button:hover {
+ opacity: 0.88;
+}
+
+.ui-button:disabled {
+ cursor: not-allowed;
+ opacity: 0.5;
+}
+
+.ui-button:focus-visible,
+select:focus-visible,
+.request-card input:focus-visible,
+.ui-textarea:focus-visible {
+ outline: 0.125rem solid var(--ring);
+ outline-offset: 0.125rem;
+}
+
+.ui-badge {
+ display: inline-flex;
+ align-items: center;
+ width: fit-content;
+ min-height: 1.625rem;
+ padding-inline: var(--space-2);
+ border-radius: 999rem;
+ background: var(--muted);
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+ text-transform: capitalize;
+ white-space: nowrap;
+}
+
+.ui-badge[data-tone="success"] {
+ background: var(--success-surface);
+ color: var(--success);
+}
+
+.ui-badge[data-tone="warning"] {
+ background: var(--warning-surface);
+ color: var(--warning);
+}
+
+.ui-badge[data-tone="danger"] {
+ background: var(--danger-surface);
+ color: var(--danger);
+}
+
+label,
+dt {
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+}
+
+select,
+.request-card input,
+.ui-textarea {
+ width: 100%;
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-sm);
+ background: var(--card);
+ color: var(--card-foreground);
+}
+
+select {
+ min-height: var(--control-height);
+ padding: var(--space-2) var(--space-3);
+}
+
+.request-card input {
+ min-height: var(--control-height);
+ padding: var(--space-2) var(--space-3);
+}
+
+.ui-textarea {
+ min-height: var(--editor-height);
+ resize: vertical;
+ padding: var(--space-3);
+ font-family: var(--font-mono);
+ font-size: var(--font-size-xs);
+ line-height: var(--line-normal);
+ tab-size: 2;
+}
+
+.replay-panel {
+ display: grid;
+ gap: var(--space-6);
+}
+
+.live-status,
+.parity-line {
+ justify-content: space-between;
+ padding-block: var(--space-2);
+}
+
+.live-status > span,
+.parity-line > span {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+}
+
+.request-card {
+ display: grid;
+ gap: var(--space-3);
+ padding-block: var(--space-4);
+ border-block: 0.0625rem solid var(--border);
+}
+
+.request-card > div {
+ display: grid;
+ gap: var(--space-1);
+}
+
+.process-facts {
+ display: grid;
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ gap: var(--space-4);
+ margin: 0;
+ padding-block: var(--space-3);
+}
+
+.empty-live-state {
+ display: grid;
+ gap: var(--space-3);
+ padding-block: var(--space-6);
+}
+
+.empty-live-state h2 {
+ font-size: var(--font-size-lg);
+}
+
+.empty-live-state p:last-child {
+ color: var(--muted-foreground);
+}
+
+.snapshot-heading {
+ align-items: flex-start;
+}
+
+.snapshot-heading > div {
+ display: grid;
+ gap: var(--space-2);
+}
+
+.snapshot-grid {
+ display: grid;
+ grid-template-columns: repeat(4, minmax(0, 1fr));
+ gap: var(--space-4);
+ margin: 0;
+ padding-block: var(--space-5);
+ border-block: 0.0625rem solid var(--border);
+}
+
+.snapshot-grid div {
+ min-width: 0;
+}
+
+dd {
+ margin: var(--space-1) 0 0;
+ overflow: hidden;
+ font-family: var(--font-mono);
+ font-size: var(--font-size-xs);
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.result-summary {
+ display: grid;
+ gap: var(--space-1);
+}
+
+.result-summary span {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+ text-transform: uppercase;
+}
+
+.result-summary p {
+ font-size: var(--font-size-md);
+}
+
+.diagnostic,
+.validation-errors {
+ padding: var(--space-4);
+ border-radius: var(--radius-sm);
+ background: var(--warning-surface);
+ color: var(--warning);
+}
+
+.validation-errors {
+ display: grid;
+ gap: var(--space-4);
+ background: var(--danger-surface);
+ color: var(--danger);
+}
+
+.validation-errors ul {
+ margin: 0;
+ padding-left: var(--space-5);
+}
+
+.timeline-heading {
+ margin-top: var(--space-2);
+}
+
+.timeline-heading h3 {
+ font-size: var(--font-size-md);
+}
+
+.timeline-heading span {
+ font-size: var(--font-size-xs);
+}
+
+.timeline {
+ display: grid;
+ gap: 0;
+ margin: 0;
+ padding: 0;
+ list-style: none;
+}
+
+.timeline li {
+ display: grid;
+ grid-template-columns: var(--timeline-column) minmax(0, 1fr);
+ gap: var(--space-3);
+ padding-block: var(--space-3);
+}
+
+.timeline li + li {
+ border-top: 0.0625rem solid var(--border);
+}
+
+.sequence {
+ display: inline-grid;
+ place-items: center;
+ align-self: start;
+ width: var(--timeline-column);
+ height: var(--timeline-column);
+ border-radius: 999rem;
+ background: var(--muted);
+ color: var(--muted-foreground);
+ font-family: var(--font-mono);
+ font-size: var(--font-size-xs);
+}
+
+.timeline-title code,
+.snapshot-json pre,
+.validation-errors code {
+ font-family: var(--font-mono);
+}
+
+.timeline-title code {
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+}
+
+.timeline-title time,
+.timeline li p {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+}
+
+.timeline li p {
+ margin-top: var(--space-1);
+}
+
+.snapshot-json {
+ border-top: 0.0625rem solid var(--border);
+ padding-top: var(--space-4);
+}
+
+.snapshot-json summary {
+ cursor: pointer;
+ font-weight: var(--weight-medium);
+}
+
+.snapshot-json pre {
+ max-height: var(--editor-height);
+ overflow: auto;
+ padding: var(--space-4);
+ border-radius: var(--radius-sm);
+ background: var(--muted);
+ font-size: var(--font-size-xs);
+}
+
+@media (max-width: 60rem) {
+ .workspace-grid {
+ grid-template-columns: 1fr;
+ }
+
+ .ui-textarea {
+ min-height: 20rem;
+ }
+}
+
+@media (max-width: 36rem) {
+ .app-shell {
+ padding: var(--space-5);
+ }
+
+ .page-header,
+ .snapshot-heading,
+ .timeline-heading {
+ align-items: flex-start;
+ flex-direction: column;
+ }
+
+ .snapshot-grid {
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ }
+}
+
+@media (prefers-reduced-motion: reduce) {
+ :root {
+ --motion-fast: 0ms;
+ }
+}
+
+/* ---------------------------------------------------------------------------
+ Live console
+ Every value below comes from the token layer above. Adapted shadcn/ui and
+ AI Elements components carry no visual values of their own.
+ --------------------------------------------------------------------------- */
+
+.ui-badge[data-tone="accent"] {
+ background: var(--accent-surface);
+ color: var(--accent);
+}
+
+.ui-button--danger {
+ background: var(--danger);
+ color: var(--primary-foreground);
+}
+
+.ui-button--secondary {
+ background: var(--muted);
+ color: var(--foreground);
+}
+
+.ui-button--quiet {
+ background: transparent;
+ color: var(--foreground);
+ border: 0.0625rem solid var(--border);
+}
+
+.ui-visually-hidden {
+ position: absolute;
+ width: 0.0625rem;
+ height: 0.0625rem;
+ margin: -0.0625rem;
+ padding: 0;
+ overflow: hidden;
+ clip-path: inset(50%);
+ white-space: nowrap;
+}
+
+.live-console {
+ display: grid;
+ gap: var(--space-4);
+}
+
+.live-console-header,
+.live-console-title,
+.live-console-header-actions {
+ display: flex;
+ align-items: center;
+ gap: var(--space-3);
+}
+
+.live-console-header {
+ justify-content: space-between;
+ flex-wrap: wrap;
+}
+
+.live-console-title h2 {
+ font-size: var(--font-size-lg);
+ font-weight: var(--weight-semibold);
+}
+
+.ui-breadcrumb {
+ color: var(--muted-foreground);
+ font-family: var(--font-mono);
+ font-size: var(--font-size-xs);
+}
+
+.live-console-desktop {
+ display: grid;
+ grid-template-columns: var(--rail-width) minmax(0, 1fr);
+ gap: var(--space-5);
+ align-items: start;
+}
+
+.live-console[data-inspector="true"] .live-console-desktop {
+ grid-template-columns: var(--rail-width) minmax(0, 1fr) var(--inspector-width);
+}
+
+.live-console-rail,
+.live-console-inspector {
+ display: grid;
+ /* An auto track would size to the widest identifier and overflow the rail. */
+ grid-template-columns: minmax(0, 1fr);
+ gap: var(--space-4);
+ min-width: 0;
+}
+
+.live-console-center {
+ display: grid;
+ gap: var(--space-3);
+ min-width: 0;
+ max-width: var(--center-width);
+}
+
+.ui-panel {
+ display: grid;
+ grid-template-columns: minmax(0, 1fr);
+ min-width: 0;
+ gap: var(--space-3);
+ padding: var(--space-4);
+ background: var(--card);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+}
+
+.ui-panel-header {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--space-2);
+}
+
+.ui-panel-title {
+ font-size: var(--font-size-sm);
+ font-weight: var(--weight-semibold);
+}
+
+.ui-panel-actions {
+ display: grid;
+ gap: var(--space-2);
+}
+
+.ui-status-row {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--space-2);
+}
+
+.ui-status-row > span:first-child {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+}
+
+.ui-identity-grid {
+ display: grid;
+ grid-template-columns: minmax(0, 1fr);
+ gap: var(--space-2);
+ margin: 0;
+ min-width: 0;
+}
+
+.ui-identity-grid > div,
+.ui-inspector-grid > div,
+.ui-status-row,
+.ui-tree-row,
+.ui-capability-list li {
+ min-width: 0;
+}
+
+.ui-identity-grid dd,
+.ui-inspector-grid dd {
+ overflow-wrap: anywhere;
+ white-space: normal;
+}
+
+/* Manifest list ---------------------------------------------------------- */
+
+.ui-manifest-list,
+.ui-tree,
+.ui-transcript,
+.ui-inspector-list,
+.ui-observation-list {
+ display: grid;
+ gap: var(--space-2);
+ margin: 0;
+ padding: 0;
+ list-style: none;
+}
+
+.ui-manifest-row {
+ display: grid;
+ grid-template-columns: minmax(0, 1fr);
+ gap: var(--space-1);
+ width: 100%;
+ min-width: 0;
+ min-height: var(--touch-target);
+ padding: var(--space-2);
+ border: 0.0625rem solid transparent;
+ border-radius: var(--radius-sm);
+ background: transparent;
+ color: inherit;
+ text-align: left;
+ cursor: pointer;
+}
+
+.ui-manifest-row[aria-pressed="true"] {
+ border-color: var(--accent);
+ background: var(--accent-surface);
+}
+
+.ui-manifest-name {
+ font-weight: var(--weight-semibold);
+}
+
+.ui-manifest-purpose {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ overflow-wrap: anywhere;
+}
+
+.ui-manifest-badges {
+ display: flex;
+ flex-wrap: wrap;
+ gap: var(--space-1);
+}
+
+.ui-observation-list {
+ gap: var(--space-1);
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+}
+
+/* Lineage tree ----------------------------------------------------------- */
+
+.ui-tree-row {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ gap: var(--space-2);
+ width: 100%;
+ min-height: var(--touch-target);
+ padding: var(--space-2);
+ border: 0.0625rem solid transparent;
+ border-radius: var(--radius-sm);
+ background: transparent;
+ color: inherit;
+ cursor: pointer;
+}
+
+.ui-tree li[data-depth="1"] .ui-tree-row {
+ padding-left: var(--space-5);
+}
+
+.ui-tree-row[aria-current="true"] {
+ border-color: var(--border);
+ background: var(--muted);
+}
+
+.ui-tree-state,
+.ui-tree-empty {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+}
+
+.ui-state-dot {
+ display: inline-block;
+ width: var(--space-2);
+ height: var(--space-2);
+ border-radius: 999rem;
+ background: var(--muted-foreground);
+}
+
+.ui-state-dot[data-state="running"] {
+ background: var(--accent);
+}
+
+.ui-state-dot[data-state="completed"] {
+ background: var(--success);
+}
+
+.ui-state-dot[data-state="failed"] {
+ background: var(--danger);
+}
+
+/* Conversation ----------------------------------------------------------- */
+
+.ui-conversation {
+ position: relative;
+}
+
+.ui-conversation-viewport {
+ max-height: var(--transcript-height);
+ overflow-y: auto;
+ padding: var(--space-4);
+ background: var(--card);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+}
+
+.ui-jump-to-latest {
+ position: absolute;
+ bottom: var(--space-3);
+ left: 50%;
+ translate: -50% 0;
+ z-index: var(--z-banner);
+ min-height: var(--touch-target);
+ padding-inline: var(--space-4);
+ border: 0.0625rem solid var(--border);
+ border-radius: 999rem;
+ background: var(--card);
+ color: var(--foreground);
+ cursor: pointer;
+}
+
+.ui-empty-transcript {
+ display: grid;
+ gap: var(--space-2);
+ padding-block: var(--space-6);
+}
+
+.ui-empty-transcript h3 {
+ font-size: var(--font-size-md);
+}
+
+.ui-empty-transcript p:last-child {
+ color: var(--muted-foreground);
+}
+
+.ui-transcript {
+ gap: var(--space-3);
+}
+
+.ui-message {
+ display: grid;
+ gap: var(--space-1);
+ padding: var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+ background: var(--surface-raised);
+}
+
+.ui-message[data-role="user"] {
+ margin-left: var(--space-10);
+ background: var(--muted);
+}
+
+.ui-message[data-state="failed"] {
+ border-color: var(--danger);
+ background: var(--danger-surface);
+ color: var(--danger);
+}
+
+.ui-message[data-state="interrupted"] {
+ border-color: var(--warning);
+}
+
+.ui-message-role {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+ letter-spacing: 0.08em;
+ text-transform: uppercase;
+}
+
+.ui-message-text {
+ white-space: pre-wrap;
+}
+
+.ui-message-divider {
+ padding-top: var(--space-2);
+ border-top: 0.0625rem solid var(--warning);
+ color: var(--warning);
+ font-size: var(--font-size-xs);
+}
+
+.ui-stream-cursor {
+ display: inline-block;
+ width: var(--space-2);
+ height: var(--font-size-md);
+ margin-left: var(--space-1);
+ background: var(--accent);
+ vertical-align: text-bottom;
+ animation: ui-pulse 1s steps(2, end) infinite;
+}
+
+@keyframes ui-pulse {
+ 50% {
+ opacity: 0.2;
+ }
+}
+
+.ui-turn-marker {
+ display: flex;
+ align-items: center;
+ gap: var(--space-2);
+ padding-block: var(--space-2);
+ border-top: 0.0625rem solid var(--border);
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+}
+
+.ui-steering-chip {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ gap: var(--space-2);
+ padding: var(--space-2);
+ border: 0.0625rem dashed var(--border);
+ border-radius: var(--radius-sm);
+ font-size: var(--font-size-xs);
+}
+
+/* Disclosures ------------------------------------------------------------ */
+
+.ui-disclosure {
+ padding: var(--space-2) var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+ background: var(--surface-raised);
+}
+
+.ui-disclosure-summary {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--space-2);
+ min-height: var(--touch-target);
+ cursor: pointer;
+ font-weight: var(--weight-medium);
+}
+
+.ui-disclosure-preview,
+.ui-disclosure-label {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ font-weight: var(--weight-semibold);
+}
+
+.ui-disclosure-preview {
+ overflow: hidden;
+ min-width: 0;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.ui-payload {
+ max-height: var(--transcript-height);
+ margin: var(--space-2) 0 0;
+ overflow: auto;
+ padding: var(--space-3);
+ border-radius: var(--radius-sm);
+ background: var(--muted);
+ font-family: var(--font-mono);
+ font-size: var(--font-size-xs);
+ white-space: pre-wrap;
+ word-break: break-word;
+}
+
+.ui-payload--danger {
+ background: var(--danger-surface);
+ color: var(--danger);
+}
+
+/* Request cards ---------------------------------------------------------- */
+
+.ui-request {
+ display: grid;
+ gap: var(--space-2);
+ padding: var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-left: var(--space-1) solid var(--accent);
+ border-radius: var(--radius-md);
+ background: var(--card);
+}
+
+.ui-request--settled {
+ border-left-color: var(--border);
+ background: var(--surface-raised);
+}
+
+.ui-request[data-status="resolved"] {
+ border-left-color: var(--success);
+}
+
+.ui-request-header,
+.ui-request-summary,
+.ui-request-actions,
+.ui-request-meta {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ gap: var(--space-2);
+}
+
+.ui-request-summary {
+ justify-content: space-between;
+ min-height: var(--touch-target);
+ cursor: pointer;
+}
+
+.ui-request-kind {
+ font-weight: var(--weight-semibold);
+}
+
+.ui-request-meta,
+.ui-request-resolving {
+ color: var(--muted-foreground);
+ font-family: var(--font-mono);
+ font-size: var(--font-size-xs);
+}
+
+.ui-request-input {
+ display: grid;
+ gap: var(--space-1);
+}
+
+.ui-request-input input {
+ min-height: var(--control-height);
+ padding: var(--space-2) var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-sm);
+ background: var(--card);
+ color: var(--card-foreground);
+}
+
+.ui-request-actions .ui-button {
+ min-height: var(--touch-target);
+}
+
+/* Composer --------------------------------------------------------------- */
+
+.ui-composer-region {
+ display: grid;
+ gap: var(--space-2);
+}
+
+.ui-composer {
+ display: grid;
+ gap: var(--space-2);
+ padding: var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+ background: var(--card);
+}
+
+.ui-composer[data-state="active-turn"] {
+ border-color: var(--accent);
+}
+
+.ui-composer-input {
+ width: 100%;
+ padding: var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-sm);
+ background: var(--card);
+ color: var(--card-foreground);
+ resize: none;
+}
+
+.ui-composer-actions {
+ display: grid;
+ gap: var(--space-2);
+}
+
+.ui-composer-hint,
+.ui-composer-reason,
+.ui-menu-reason,
+.ui-capability-diagnostic {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+}
+
+.ui-composer-reason,
+.ui-capability-diagnostic {
+ color: var(--warning);
+}
+
+.ui-composer-buttons {
+ display: flex;
+ justify-content: flex-end;
+ gap: var(--space-3);
+}
+
+.ui-composer-buttons .ui-button {
+ min-width: var(--touch-target);
+ min-height: var(--touch-target);
+}
+
+/* Banners, menus, dialogs, tooltips -------------------------------------- */
+
+.ui-banner {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--space-3);
+ z-index: var(--z-banner);
+ padding: var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+ background: var(--muted);
+ transition: opacity var(--motion-medium) ease;
+}
+
+.ui-banner[data-tone="accent"] {
+ border-color: var(--accent);
+ background: var(--accent-surface);
+ color: var(--accent);
+}
+
+.ui-banner[data-tone="warning"] {
+ border-color: var(--warning);
+ background: var(--warning-surface);
+ color: var(--warning);
+}
+
+.ui-banner[data-tone="danger"] {
+ border-color: var(--danger);
+ background: var(--danger-surface);
+ color: var(--danger);
+}
+
+.ui-banner-actions {
+ display: flex;
+ gap: var(--space-2);
+}
+
+.ui-menu {
+ position: relative;
+}
+
+.ui-menu-list {
+ position: absolute;
+ right: 0;
+ z-index: var(--z-banner);
+ display: grid;
+ gap: var(--space-1);
+ min-width: var(--rail-width);
+ padding: var(--space-2);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+ background: var(--card);
+ box-shadow: var(--shadow-sm);
+}
+
+.ui-menu-item {
+ min-height: var(--touch-target);
+ padding: var(--space-2);
+ border: 0;
+ border-radius: var(--radius-sm);
+ background: transparent;
+ color: inherit;
+ text-align: left;
+ cursor: pointer;
+}
+
+.ui-menu-item[aria-disabled="true"] {
+ color: var(--muted-foreground);
+ cursor: not-allowed;
+}
+
+.ui-menu-item:hover:not([aria-disabled="true"]) {
+ background: var(--muted);
+}
+
+.ui-dialog {
+ z-index: var(--z-dialog);
+ max-width: var(--center-width);
+ padding: 0;
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-md);
+ background: var(--card);
+ color: var(--card-foreground);
+}
+
+.ui-dialog-body {
+ display: grid;
+ gap: var(--space-3);
+ padding: var(--space-5);
+}
+
+.ui-dialog-title {
+ font-size: var(--font-size-lg);
+ font-weight: var(--weight-semibold);
+}
+
+.ui-dialog-description {
+ color: var(--muted-foreground);
+}
+
+.ui-dialog-input {
+ width: 100%;
+ padding: var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-sm);
+ background: var(--card);
+ color: var(--card-foreground);
+}
+
+.ui-dialog-footer {
+ display: flex;
+ justify-content: flex-end;
+ gap: var(--space-3);
+}
+
+.ui-tooltip {
+ position: relative;
+}
+
+.ui-tooltip-content {
+ position: absolute;
+ left: 0;
+ top: 100%;
+ z-index: var(--z-banner);
+ display: none;
+ min-width: var(--rail-width);
+ padding: var(--space-2);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-sm);
+ background: var(--card);
+ font-size: var(--font-size-xs);
+}
+
+.ui-tooltip:hover .ui-tooltip-content,
+.ui-tooltip:focus-within .ui-tooltip-content {
+ display: block;
+}
+
+/* Tabs and inspector ----------------------------------------------------- */
+
+.ui-tabs {
+ display: grid;
+ gap: var(--space-3);
+}
+
+.ui-tablist {
+ display: flex;
+ flex-wrap: wrap;
+ gap: var(--space-1);
+ padding: var(--space-1);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-sm);
+ background: var(--muted);
+}
+
+.ui-tab {
+ min-height: var(--touch-target);
+ padding-inline: var(--space-3);
+ border: 0;
+ border-radius: var(--radius-sm);
+ background: transparent;
+ color: var(--muted-foreground);
+ cursor: pointer;
+}
+
+.ui-tab[aria-selected="true"] {
+ background: var(--card);
+ color: var(--foreground);
+ font-weight: var(--weight-semibold);
+}
+
+.ui-inspector-events {
+ display: grid;
+ gap: var(--space-2);
+}
+
+.ui-inspector-filter {
+ width: 100%;
+ min-height: var(--control-height);
+ padding: var(--space-2) var(--space-3);
+ border: 0.0625rem solid var(--border);
+ border-radius: var(--radius-sm);
+ background: var(--card);
+ color: var(--card-foreground);
+}
+
+.ui-facets {
+ display: flex;
+ flex-wrap: wrap;
+ gap: var(--space-1);
+ max-height: var(--facet-height);
+ overflow-y: auto;
+ padding-bottom: var(--space-1);
+}
+
+.ui-facet,
+.ui-chip,
+.ui-copy {
+ min-height: var(--space-6);
+ padding-inline: var(--space-2);
+ border: 0.0625rem solid var(--border);
+ border-radius: 999rem;
+ background: var(--card);
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+ cursor: pointer;
+}
+
+.ui-facet[aria-pressed="true"] {
+ border-color: var(--accent);
+ background: var(--accent-surface);
+ color: var(--accent);
+}
+
+.ui-inspector-list {
+ max-height: var(--transcript-height);
+ overflow-y: auto;
+}
+
+.ui-inspector-list li[data-highlighted="true"] {
+ background: var(--accent-surface);
+}
+
+.ui-inspector-row {
+ display: grid;
+ grid-template-columns: var(--timeline-column) minmax(0, 1fr) minmax(0, 1fr);
+ gap: var(--space-2);
+ align-items: center;
+ min-height: var(--touch-target);
+ cursor: pointer;
+ font-family: var(--font-mono);
+ font-size: var(--font-size-xs);
+}
+
+.ui-inspector-id {
+ overflow: hidden;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.ui-seq {
+ color: var(--muted-foreground);
+}
+
+.ui-inspector-grid {
+ display: grid;
+ grid-template-columns: minmax(0, 1fr);
+ gap: var(--space-3);
+ margin: 0;
+ min-width: 0;
+}
+
+.ui-capability-list {
+ display: grid;
+ gap: var(--space-2);
+ margin: 0;
+ padding: 0;
+ list-style: none;
+}
+
+.ui-capability-list li {
+ display: grid;
+ grid-template-columns: minmax(0, 1fr) auto;
+ gap: var(--space-2);
+ align-items: center;
+ padding-block: var(--space-2);
+ border-bottom: 0.0625rem solid var(--border);
+}
+
+.ui-capability-list p {
+ grid-column: 1 / -1;
+}
+
+.ui-replay-stepper {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ gap: var(--space-3);
+ padding: var(--space-3);
+ border: 0.0625rem solid var(--warning);
+ border-radius: var(--radius-md);
+ background: var(--warning-surface);
+}
+
+.ui-goal-banner {
+ display: grid;
+ gap: var(--space-2);
+}
+
+.ui-goal-empty {
+ color: var(--muted-foreground);
+ font-size: var(--font-size-xs);
+}
+
+.ui-transcript li[data-highlighted="true"] {
+ outline: 0.125rem solid var(--accent);
+ outline-offset: 0.125rem;
+}
+
+.ui-tab:focus-visible,
+.ui-menu-item:focus-visible,
+.ui-tree-row:focus-visible,
+.ui-manifest-row:focus-visible,
+.ui-facet:focus-visible,
+.ui-chip:focus-visible,
+.ui-copy:focus-visible,
+.ui-jump-to-latest:focus-visible,
+.ui-request:focus-visible,
+.ui-conversation-viewport:focus-visible,
+.ui-composer-input:focus-visible,
+.ui-inspector-filter:focus-visible,
+.ui-dialog-input:focus-visible,
+.ui-request-input input:focus-visible,
+.ui-tabpanel:focus-visible,
+summary:focus-visible {
+ outline: 0.125rem solid var(--ring);
+ outline-offset: 0.125rem;
+}
+
+@media (max-width: 64rem) {
+ .ui-message[data-role="user"] {
+ margin-left: var(--space-5);
+ }
+}
+
+@media (prefers-reduced-motion: reduce) {
+ :root {
+ --motion-medium: 0ms;
+ }
+
+ .ui-stream-cursor {
+ animation: none;
+ }
+}
diff --git a/packages/paperclip-runner/devtools/browser/static-replay.spec.ts b/packages/paperclip-runner/devtools/browser/static-replay.spec.ts
new file mode 100644
index 0000000000..98dfaf7288
--- /dev/null
+++ b/packages/paperclip-runner/devtools/browser/static-replay.spec.ts
@@ -0,0 +1,112 @@
+import { expect, test } from "@playwright/test";
+
+test("validates and renders the shared Replay fixture", async ({ page }, testInfo) => {
+ await page.goto("/");
+ await expect(page.getByRole("heading", { name: "Live runner diagnostics" })).toBeVisible();
+ await page.getByRole("button", { name: "Static replay" }).click();
+ await expect(page.getByRole("heading", { name: "Static protocol replay" })).toBeVisible();
+ await expect(page.getByTestId("terminal-badge")).toHaveText("Succeeded");
+ await expect(page.getByTestId("timeline").getByRole("listitem")).toHaveCount(9);
+ await expect(page.getByTestId("timeline")).not.toContainText("workspace_preparing");
+ await expect(page.getByTestId("result-summary")).toContainText(
+ "The scripted run completed successfully.",
+ );
+ await page.screenshot({
+ path: testInfo.outputPath("replay-static-replay.png"),
+ fullPage: true,
+ });
+});
+
+test("shows duplicate and unsupported-version replay states", async ({ page }) => {
+ await page.goto("/");
+ await page.getByRole("button", { name: "Static replay" }).click();
+ await page.getByLabel("Fixture", { exact: true }).selectOption("duplicate-event");
+ await expect(page.getByText("1 duplicate events ignored")).toBeVisible();
+ await page
+ .getByLabel("Fixture", { exact: true })
+ .selectOption("unsupported-required-version");
+ await expect(page.getByRole("heading", { name: "Fixture cannot be replayed" })).toBeVisible();
+ await expect(page.getByText(/protocolVersion 2 is unsupported/)).toBeVisible();
+});
+
+test("streams a live run and proves replay parity", async ({ page }, testInfo) => {
+ await page.goto("/");
+ await page.getByRole("button", { name: "Start local run" }).click();
+ await expect(page.getByTestId("live-status")).toHaveText("Terminal");
+ await expect(page.getByTestId("terminal-badge")).toHaveText("Succeeded");
+ await expect(page.getByTestId("process-facts")).toContainText("Harness process exit");
+ await expect(page.getByTestId("process-facts")).toContainText("done");
+ await expect(page.getByTestId("parity-result")).toContainText("Match");
+ await page.screenshot({
+ path: testInfo.outputPath("local-runner-live-complete.png"),
+ fullPage: true,
+ });
+});
+
+test("resolves live permission and input requests", async ({ page }, testInfo) => {
+ await page.goto("/");
+ await page.getByLabel("Scenario").selectOption("permission-input");
+ await page.getByRole("button", { name: "Start local run" }).click();
+ await expect(page.getByTestId("runtime-request")).toContainText("permission request");
+ await expect(page.getByTestId("terminal-badge")).toHaveText("Executing");
+ await expect(
+ page.getByTestId("timeline").getByRole("listitem").filter({
+ hasText: "runtime_request.created",
+ }).first(),
+ ).toContainText("permission: Allow the fake driver to write its local fixture?");
+ await page.screenshot({
+ path: testInfo.outputPath("local-runner-live-permission.png"),
+ fullPage: true,
+ });
+ await page.getByRole("button", { name: "Allow" }).click();
+ await expect(page.getByTestId("runtime-request")).toContainText("input request");
+ await expect(
+ page.getByTestId("timeline").getByRole("listitem").filter({
+ hasText: "runtime_request.resolved",
+ }).first(),
+ ).toContainText("Resolved permission: Allow the fake driver to write its local fixture?");
+ await page.waitForTimeout(10_500);
+ await page.getByLabel("Response").fill("local-runner-browser-trace");
+ await page.getByRole("button", { name: "Send input" }).click();
+ await expect(page.getByTestId("live-status")).toHaveText("Terminal");
+ await expect(page.getByTestId("parity-result")).toContainText("Match");
+});
+
+test("interrupts a live turn without duplicating terminal state", async ({ page }, testInfo) => {
+ await page.goto("/");
+ await page.getByLabel("Scenario").selectOption("interrupted");
+ await page.getByRole("button", { name: "Start local run" }).click();
+ const interrupt = page.getByRole("button", { name: "Interrupt turn" });
+ await expect(interrupt).toBeEnabled();
+ await interrupt.click();
+ await expect(page.getByTestId("live-status")).toHaveText("Terminal");
+ await expect(page.getByTestId("terminal-badge")).toHaveText("Cancelled");
+ await expect(page.getByTestId("live-status")).toHaveAttribute("data-tone", "neutral");
+ await expect(page.getByTestId("terminal-badge")).toHaveAttribute("data-tone", "neutral");
+ await expect(page.getByTestId("timeline").getByText("run.terminal")).toHaveCount(1);
+ await page.screenshot({
+ path: testInfo.outputPath("local-runner-live-interrupted.png"),
+ fullPage: true,
+ });
+});
+
+test("shows failed outcomes as danger and names the duplicate-terminal guard", async ({ page }) => {
+ await page.goto("/");
+ await page.getByLabel("Scenario").selectOption("error");
+ await page.getByRole("button", { name: "Start local run" }).click();
+ await expect(page.getByTestId("live-status")).toHaveText("Terminal");
+ await expect(page.getByTestId("live-status")).toHaveAttribute("data-tone", "danger");
+ await expect(page.getByTestId("terminal-badge")).toHaveText("Failed");
+ await expect(page.getByTestId("terminal-badge")).toHaveAttribute("data-tone", "danger");
+
+ await page.getByLabel("Scenario").selectOption("duplicate-terminal");
+ await page.getByRole("button", { name: "Start local run" }).click();
+ await expect(page.getByTestId("live-status")).toHaveText("Terminal");
+ await expect(
+ page.getByTestId("timeline").getByRole("listitem").filter({
+ hasText: "harness.diagnostic",
+ }),
+ ).toContainText(
+ "Duplicate terminal event ignored; the first terminal event remains authoritative.",
+ );
+});
diff --git a/packages/paperclip-runner/devtools/issue-thread/index.html b/packages/paperclip-runner/devtools/issue-thread/index.html
new file mode 100644
index 0000000000..ff5100b74d
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/index.html
@@ -0,0 +1,13 @@
+
+
+
+
+
+
+ 🧯 Mock Paperclip · Issue thread
+
+
+
+
+
+
diff --git a/packages/paperclip-runner/devtools/issue-thread/issue-thread.spec.ts b/packages/paperclip-runner/devtools/issue-thread/issue-thread.spec.ts
new file mode 100644
index 0000000000..fa9c0f2551
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/issue-thread.spec.ts
@@ -0,0 +1,1234 @@
+import { createRequire } from "node:module";
+
+import { expect, test, type Page } from "@playwright/test";
+
+import { CAPABILITY_UI_SHOT_SLUGS } from "../../src/issue-thread/fixtures";
+import type {
+ CapabilityIssueThreadSnapshot,
+ CapabilityThreadItem,
+} from "../../src/issue-thread/types";
+
+const require = createRequire(import.meta.url);
+const AXE_PATH = require.resolve("axe-core/axe.min.js");
+
+const DESKTOP = { width: 1440, height: 900 };
+const MOBILE = { width: 390, height: 844 };
+
+function route(slug: string, params: Record = {}): string {
+ const query = new URLSearchParams({ shot: slug, capture: "1", ...params });
+ return `/#/issue/hb-baseline?${query.toString()}`;
+}
+
+/** Every route settles on a data attribute, never on a timeout (§10.1). */
+async function open(page: Page, slug: string, params: Record = {}) {
+ await page.goto(route(slug, params));
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible();
+}
+
+interface AxeViolation {
+ id: string;
+ impact: string | null;
+ nodes: unknown[];
+}
+
+async function seriousAxeViolations(page: Page): Promise {
+ await page.addScriptTag({ path: AXE_PATH });
+ const violations = await page.evaluate(async () => {
+ const axe = (window as unknown as { axe: { run: (context: unknown, options: unknown) => Promise<{ violations: AxeViolation[] }> } }).axe;
+ const results = await axe.run(document, {
+ resultTypes: ["violations"],
+ runOnly: { type: "tag", values: ["wcag2a", "wcag2aa", "wcag21a", "wcag21aa"] },
+ });
+ return results.violations.map((violation) => ({
+ id: violation.id,
+ impact: violation.impact ?? null,
+ nodes: violation.nodes.length,
+ }));
+ });
+ return (violations as unknown as AxeViolation[]).filter(
+ (violation) => violation.impact === "serious" || violation.impact === "critical",
+ );
+}
+
+test.describe("Capability issue thread", () => {
+ test("loads the bundled sans and mono faces before the thread settles", async ({ page }) => {
+ await open(page, "thread-baseline");
+
+ const probe = await page.evaluate(async () => {
+ const normalizeFamily = (family: string) => family.trim().replace(/^['"]|['"]$/g, "");
+ const styles = getComputedStyle(document.documentElement);
+ const sansFamily = normalizeFamily(styles.getPropertyValue("--pit-font-sans").split(",")[0]);
+ const monoFamily = normalizeFamily(styles.getPropertyValue("--pit-font-mono").split(",")[0]);
+ const symbolFamily = normalizeFamily(styles.getPropertyValue("--pit-font-symbols"));
+ await document.fonts.load(`400 16px "${symbolFamily}"`, "◐⏳\uFE0E");
+ const statuses = (family: string) =>
+ [...document.fonts]
+ .filter((face) => normalizeFamily(face.family) === family)
+ .map((face) => face.status);
+
+ return {
+ sansFamily,
+ monoFamily,
+ symbolFamily,
+ sansStatuses: statuses(sansFamily),
+ monoStatuses: statuses(monoFamily),
+ symbolStatuses: statuses(symbolFamily),
+ sansWeightsLoaded: [400, 500, 600, 700].every((weight) =>
+ document.fonts.check(`${weight} 16px "${sansFamily}"`, "Paperclip"),
+ ),
+ monoWeightsLoaded: [400, 700].every((weight) =>
+ document.fonts.check(`${weight} 16px "${monoFamily}"`, "TASK-17003"),
+ ),
+ symbolsLoaded: document.fonts.check(`400 16px "${symbolFamily}"`, "◐⏳\uFE0E"),
+ };
+ });
+
+ expect(probe).toEqual({
+ sansFamily: "Paperclip Issue Thread Inter",
+ monoFamily: "Paperclip Issue Thread DejaVu Sans Mono",
+ symbolFamily: "Paperclip Issue Thread Symbols",
+ sansStatuses: ["loaded"],
+ monoStatuses: ["loaded", "loaded"],
+ symbolStatuses: ["loaded", "loaded"],
+ sansWeightsLoaded: true,
+ monoWeightsLoaded: true,
+ symbolsLoaded: true,
+ });
+ });
+
+ test("baseline thread renders identity, turns, and the §3 item types", async ({ page }) => {
+ await open(page, "thread-baseline");
+
+ await expect(page.locator('[data-session-mode="fake"]').first()).toBeVisible();
+ const chips = page.getByTestId("identity-chips");
+ await expect(chips.getByTestId("agent-chip")).toHaveText("Fake agent");
+ await expect(chips.getByTestId("runner-chip")).toContainText("In-process runner");
+ await expect(chips.getByTestId("control-plane-chip")).toHaveText("Mock Paperclip");
+
+ await expect(page.locator("h1")).toHaveCount(1);
+ await expect(page.locator('[data-turn-id]')).toHaveCount(3);
+ for (const kind of ["user_message", "agent_message", "durable_comment", "tool_activity"]) {
+ await expect(page.locator(`[data-thread-item="${kind}"]`).first()).toBeVisible();
+ }
+ await expect(page.locator('[data-composer-state="ready"]')).toHaveCount(1);
+ });
+
+ test("a durable progress comment is visually distinct from model prose", async ({ page }) => {
+ await open(page, "thread-baseline");
+ await expect(
+ page.locator('[data-thread-item="durable_comment"]').getByText("Recorded to mock thread"),
+ ).toBeVisible();
+ await expect(
+ page.locator('[data-thread-item="agent_message"]').first().getByText("Recorded to mock thread"),
+ ).toHaveCount(0);
+ });
+
+ test("tool strips are collapsed disclosures that deep-link into Evidence", async ({ page }) => {
+ await open(page, "thread-baseline");
+ const strip = page.locator('[data-tool-strip="get_task_context"]').first();
+ await expect(strip).not.toHaveAttribute("open", "");
+ await strip.locator("summary").click();
+ await expect(strip).toHaveAttribute("open", "");
+
+ await page.getByRole("button", { name: "View in Evidence" }).first().click();
+ const panel = page.getByTestId("evidence-panel");
+ await expect(panel).toBeVisible();
+ await expect(panel.locator('[data-record-id="call-1"][data-highlighted="true"]')).toBeVisible();
+ });
+
+ test("a pending question card resolves inline and releases the composer", async ({ page }) => {
+ await open(page, "interaction-question-pending");
+
+ await expect(page.locator('[data-composer-state="waiting"]')).toBeVisible();
+ const card = page.locator('[data-interaction-id="ix-questions-01"]');
+ await expect(card).toHaveAttribute("data-interaction-state", "pending");
+ await expect(card.getByTestId("interaction-state-chip")).toContainText("Waiting for you");
+
+ // The submit is rejected until every required question is answered.
+ await card.getByRole("button", { name: "Submit answers" }).click();
+ await expect(card.getByRole("alert")).toContainText("required");
+
+ await card.getByRole("radio", { name: "Yes — runner owns it" }).check();
+ await card.getByRole("combobox").selectOption("hb-baseline");
+ await card.getByRole("button", { name: "Submit answers" }).click();
+
+ await expect(card).toHaveAttribute("data-interaction-state", "answered");
+ await expect(card.getByTestId("interaction-state-chip")).toContainText("Answered");
+ await expect(page.locator('[data-composer-state="ready"]').first()).toBeVisible();
+ });
+
+ test("a revision-bound confirmation names its target and requires a reject reason", async ({
+ page,
+ }) => {
+ await open(page, "interaction-confirmation-pending");
+ const card = page.locator('[data-interaction-id="ix-confirmation-plan-01"]');
+ await expect(card.getByTestId("interaction-target")).toHaveText("plan · r4");
+
+ await card.getByRole("button", { name: "Request changes" }).click();
+ await expect(card.getByRole("alert")).toContainText("reason is required");
+ await expect(card).toHaveAttribute("data-interaction-state", "pending");
+
+ await card.getByRole("textbox").fill("Split the acceptance section first.");
+ await card.getByRole("button", { name: "Request changes" }).click();
+ await expect(card).toHaveAttribute("data-interaction-state", "rejected");
+ await expect(card).toContainText("Split the acceptance section first.");
+ });
+
+ test("resolved, rejected, stale, and superseded cards stay distinct in history", async ({
+ page,
+ }) => {
+ await open(page, "interaction-resolved-mixed");
+ for (const state of ["answered", "accepted", "rejected", "stale_target", "superseded_by_comment"]) {
+ await expect(
+ page.locator(`[data-interaction-state="${state}"]`),
+ `missing ${state}`,
+ ).toHaveCount(1);
+ }
+ // Expired treatments keep their controls removed but stay reachable.
+ await expect(
+ page.locator('[data-interaction-state="stale_target"] button', { hasText: "Approve" }),
+ ).toHaveCount(0);
+ await expect(
+ page.locator('[data-interaction-state="stale_target"]').getByRole("button", {
+ name: "View request evidence",
+ }),
+ ).toBeVisible();
+ });
+
+ test("the deliverable strip and card report one size in one unit", async ({ page }) => {
+ await open(page, "deliverable-registered");
+ const card = page.locator('[data-thread-item="deliverable"]');
+ await expect(card).toContainText("18.0 kB");
+ await expect(
+ page.locator('[data-tool-strip="register_deliverable"]'),
+ ).toContainText("18.0 kB");
+ });
+
+ test("a denial quotes the authorization record verbatim and badges Evidence", async ({
+ page,
+ }) => {
+ await open(page, "denial-optional-tool");
+ await expect(page.getByTestId("denial-reason")).toHaveText(
+ "denied: missing grant su:read_or_write",
+ );
+ await expect(page.getByTestId("evidence-toggle")).toContainText("(1)");
+
+ await page.locator('[data-thread-item="denial"]').getByRole("button", { name: "View in Evidence" }).click();
+ const record = page.getByTestId("evidence-panel").locator('[data-record-id="authz-deny-1"]');
+ await expect(record).toHaveAttribute("data-allowed", "false");
+ await expect(record).toContainText("denied: missing grant su:read_or_write");
+ });
+
+ test("Evidence exposes eight sections, a turn selector, and the withheld control-plane list", async ({
+ page,
+ }) => {
+ await open(page, "debug-panel-open", { panel: "authorization" });
+ const panel = page.getByTestId("evidence-panel");
+ await expect(panel).toBeVisible();
+ await expect(panel.locator("[data-evidence-section]")).toHaveCount(8);
+ await expect(panel.locator("#evidence-turn-selector")).toBeVisible();
+
+ // The Tools section is open by default; assert its groups directly.
+ await expect(
+ panel.locator('[data-evidence-section="tools"] button').first(),
+ ).toHaveAttribute("aria-expanded", "true");
+ await expect(panel.getByText("Control plane (not exposed to the agent)")).toBeVisible();
+ await expect(
+ panel.locator('[data-disposition="control_plane_owned"]', { hasText: "create_task" }),
+ ).toBeVisible();
+ await expect(panel.getByText("Agent tool — always")).toBeVisible();
+ await expect(panel.getByText("Agent tool — granted")).toBeVisible();
+
+ });
+
+ test("tool exposure groups are deduplicated, foldable, and open catalog details", async ({ page }) => {
+ await open(page, "debug-panel-open", { panel: "tools" });
+ const panel = page.getByTestId("evidence-panel");
+ const always = panel.getByRole("button", { name: /Agent tool — always/ });
+ await expect(always).toHaveCount(1);
+ await always.click();
+ await expect(panel.getByRole("button", { name: /get_task_context/ })).toHaveCount(0);
+ await always.click();
+ await panel.getByRole("button", { name: /get_task_context/ }).click();
+ const dialog = page.getByRole("dialog", { name: "Get active task context" });
+ await expect(dialog).toContainText("Read the active mock task, actor, wake, ancestors, budget, and interaction results.");
+ await expect(dialog).toContainText("Input schema");
+ await dialog.getByRole("button", { name: "Close tool details" }).click();
+ await expect(dialog).toHaveCount(0);
+ });
+
+ test("Evidence records link back to their thread anchor", async ({ page }) => {
+ await open(page, "debug-panel-open", { panel: "calls" });
+ const panel = page.getByTestId("evidence-panel");
+ await panel.locator('[data-record-id="call-1"]').getByRole("button", { name: "Show in thread" }).click();
+ await expect(page.locator('[data-tool-strip="get_task_context"]')).toBeInViewport();
+ });
+
+ test("the splitter resizes the panel from the keyboard", async ({ page }) => {
+ await open(page, "debug-panel-open", { panel: "tools" });
+ const splitter = page.getByRole("separator", { name: "Resize the evidence panel" });
+ const before = await splitter.getAttribute("aria-valuenow");
+ await splitter.focus();
+ await page.keyboard.press("ArrowLeft");
+ await expect(splitter).not.toHaveAttribute("aria-valuenow", before ?? "");
+ });
+
+ test("the splitter drag makes the DevTools panel wider", async ({ page }) => {
+ await open(page, "debug-panel-open", { panel: "tools" });
+ const splitter = page.getByRole("separator", { name: "Resize the evidence panel" });
+ const panel = page.getByTestId("evidence-panel");
+ const handle = await splitter.boundingBox();
+ const before = await panel.boundingBox();
+ expect(handle).not.toBeNull();
+ expect(before).not.toBeNull();
+ await page.mouse.move(handle!.x + handle!.width / 2, handle!.y + handle!.height / 2);
+ await page.mouse.down();
+ await page.mouse.move(handle!.x - 140, handle!.y + handle!.height / 2, { steps: 5 });
+ await page.mouse.up();
+ const after = await panel.boundingBox();
+ expect(after!.width).toBeGreaterThan(before!.width + 100);
+ });
+
+ test("reset is confirmed, escapable, and clears the thread", async ({ page }) => {
+ await open(page, "disposition-terminal");
+ await page.getByTestId("reset-button").click();
+ const dialog = page.getByTestId("reset-dialog");
+ await expect(dialog).toBeVisible();
+ await expect(dialog).toContainText("The transcript will be lost.");
+ await expect(page.getByRole("button", { name: "Cancel" })).toBeFocused();
+
+ await page.keyboard.press("Escape");
+ await expect(dialog).toHaveCount(0);
+
+ await page.getByTestId("reset-button").click();
+ await page
+ .getByTestId("reset-dialog")
+ .getByRole("button", { name: "Reset scenario", exact: true })
+ .click();
+ await expect(page.locator('[data-thread-item="disposition"]')).toHaveCount(0);
+ await expect(page.locator('[data-composer-state="ready"]').first()).toBeVisible();
+ });
+
+ test("stop keeps the partial turn and marks it", async ({ page }) => {
+ await open(page, "turn-streaming");
+ await expect(page.locator('[data-composer-state="streaming"]')).toBeVisible();
+ await expect(page.getByTestId("composer-stop")).toBeVisible();
+ await expect(page.locator('[data-composer-state="streaming"] textarea')).toBeEnabled();
+
+ await page.getByTestId("composer-stop").click();
+ await expect(page.getByTestId("stopped-marker")).toBeVisible();
+ await expect(page.locator('[data-thread-item="agent_message"]').last()).toContainText(
+ "Two dependent tasks are still open",
+ );
+ });
+
+ test("reconnect pins an amber banner and disables the composer", async ({ page }) => {
+ await open(page, "reconnect-banner");
+ await expect(page.getByTestId("reconnect-banner")).toContainText("attempt 2");
+ await expect(page.locator('[data-composer-state="reconnecting"]')).toBeVisible();
+ await expect(page.locator('[data-composer-state="reconnecting"] textarea')).toBeDisabled();
+
+ await page.getByRole("button", { name: "Retry now" }).click();
+ await expect(page.getByText("Reconnected")).toBeVisible();
+ });
+
+ test("replay is read-only and labelled as fake-derived", async ({ page }) => {
+ await open(page, "replay-mode");
+ await expect(page.getByTestId("agent-chip")).toContainText("Replay · fake source");
+ await expect(page.getByTestId("composer-reason")).toHaveText("Replay is read-only");
+ await expect(page.getByTestId("replay-strip")).toContainText("Replay 12/18");
+ await expect(page.locator('[data-composer-state="disabled"] textarea')).toBeDisabled();
+ });
+
+ test("the replay strip steps, advances, and plays through to the end", async ({ page }) => {
+ await open(page, "replay-mode");
+ const strip = page.getByTestId("replay-strip");
+ const progress = page.getByRole("progressbar", { name: "Replay progress" });
+
+ await page.getByTestId("replay-step-back").click();
+ await expect(strip).toContainText("Replay 11/18");
+ await expect(progress).toHaveAttribute("aria-valuenow", "11");
+
+ await page.getByTestId("replay-next-turn").click();
+ await expect(strip).toContainText("Replay 12/18");
+
+ // Play-all runs the recording out and parks itself at the last ordinal
+ // instead of leaving the operator to click Next turn eighteen times (§6).
+ const playAll = page.getByTestId("replay-play-all");
+ await playAll.click();
+ await expect(playAll).toHaveAttribute("aria-pressed", "true");
+ await expect(strip).toContainText("Replay 18/18", { timeout: 15_000 });
+ await expect(playAll).toHaveAttribute("aria-pressed", "false");
+ await expect(page.getByTestId("replay-next-turn")).toBeDisabled();
+ });
+
+ test("a terminal disposition keeps the composer available", async ({ page }) => {
+ await open(page, "disposition-terminal");
+ await expect(page.locator('[data-thread-item="disposition"]')).toContainText("Done");
+ await expect(page.locator("#composer-input")).toBeEnabled();
+ await expect(page.getByText("This task is done. You can still continue the conversation.")).toBeVisible();
+ });
+
+ test("composer drafts survive a refresh", async ({ page }) => {
+ await open(page, "thread-baseline");
+ await page.locator("#composer-input").fill("Draft that must survive F5.");
+ await page.reload();
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible();
+ await expect(page.locator("#composer-input")).toHaveValue("Draft that must survive F5.");
+ });
+
+ test("mobile keeps the thread usable with no horizontal page scroll", async ({ page }) => {
+ await page.setViewportSize(MOBILE);
+ await open(page, "turn-streaming");
+
+ const overflow = await page.evaluate(() => {
+ const element = document.scrollingElement as HTMLElement;
+ return element.scrollWidth - element.clientWidth;
+ });
+ expect(overflow).toBeLessThanOrEqual(0);
+
+ // Stop stays outside the overflow menu while a turn is active (§2.2).
+ await expect(page.getByTestId("stop-button")).toBeVisible();
+ await expect(page.getByTestId("stop-button")).toBeEnabled();
+ await expect(page.getByTestId("overflow-menu")).toHaveCount(0);
+
+ await page.getByTestId("segment-evidence").click();
+ await expect(page.getByTestId("evidence-panel")).toBeVisible();
+ await expect(page.getByTestId("evidence-panel")).toHaveAttribute("data-layout", "segment");
+ });
+
+ test("mobile surfaces the denial count on the Evidence segment", async ({ page }) => {
+ await page.setViewportSize(MOBILE);
+ await open(page, "denial-optional-tool");
+ await expect(page.getByTestId("segment-denial-badge")).toHaveText("1");
+ });
+
+ test("every actionable control meets the 44px touch target on mobile", async ({ page }) => {
+ await page.setViewportSize(MOBILE);
+ await open(page, "interaction-question-pending");
+ const undersized = await page.evaluate(() => {
+ const selectors = "button:not([hidden]), select, textarea, [role='radio']";
+ return [...document.querySelectorAll(selectors)]
+ .filter((node) => {
+ const element = node as HTMLElement;
+ if (element.offsetParent === null) return false;
+ const rect = element.getBoundingClientRect();
+ return rect.height < 44;
+ })
+ .map((node) => (node as HTMLElement).className || node.tagName);
+ });
+ expect(undersized).toEqual([]);
+ });
+
+ test("the mobile overflow menu closes with Escape", async ({ page }) => {
+ await page.setViewportSize(MOBILE);
+ await open(page, "thread-baseline");
+ await page.getByTestId("overflow-menu-button").click();
+ await expect(page.getByTestId("overflow-menu")).toBeVisible();
+ await page.keyboard.press("Escape");
+ await expect(page.getByTestId("overflow-menu")).toHaveCount(0);
+ });
+
+ test("opening Evidence moves focus to its heading", async ({ page }) => {
+ await open(page, "thread-baseline");
+ await page.getByTestId("evidence-toggle").click();
+ await expect(page.getByRole("heading", { name: "Developer tools" })).toBeFocused();
+ });
+
+ test("Escape closes the desktop overlay sheet and returns focus to its toggle", async ({
+ page,
+ }) => {
+ // Between the mobile segment and the docked side panel, Evidence renders as
+ // an overlay sheet; §9.1 makes Escape dismiss it.
+ await page.setViewportSize({ width: 1000, height: 800 });
+ await open(page, "thread-baseline");
+ await page.getByTestId("evidence-toggle").click();
+
+ const panel = page.getByTestId("evidence-panel");
+ await expect(panel).toHaveAttribute("data-layout", "overlay");
+ await expect(page.getByRole("heading", { name: "Developer tools" })).toBeFocused();
+
+ await page.keyboard.press("Escape");
+ await expect(panel).toHaveCount(0);
+ await expect(page.getByTestId("evidence-toggle")).toBeFocused();
+ });
+
+ test("resolving an interaction card moves focus to its state chip", async ({ page }) => {
+ await open(page, "interaction-question-pending");
+ const card = page.locator('[data-interaction-id="ix-questions-01"]');
+
+ await card.getByRole("radio", { name: "Yes — runner owns it" }).check();
+ await card.getByRole("combobox").selectOption("hb-baseline");
+ await card.getByRole("button", { name: "Submit answers" }).click();
+
+ await expect(card).toHaveAttribute("data-interaction-state", "answered");
+ await expect(card.getByTestId("interaction-state-chip")).toBeFocused();
+ });
+
+ test("the waiting composer anchor focuses the pending card", async ({ page }) => {
+ await open(page, "interaction-question-pending");
+ await page.getByTestId("pending-anchor").click();
+ await expect(page.locator('[data-interaction-id="ix-questions-01"]')).toBeInViewport();
+ });
+
+ for (const slug of CAPABILITY_UI_SHOT_SLUGS) {
+ test(`axe reports no serious or critical violation on ${slug} (desktop)`, async ({ page }) => {
+ await page.setViewportSize(DESKTOP);
+ await open(page, slug, slug === "debug-panel-open" ? { panel: "authorization" } : {});
+ expect(await seriousAxeViolations(page)).toEqual([]);
+ });
+
+ test(`axe reports no serious or critical violation on ${slug} (mobile)`, async ({ page }) => {
+ await page.setViewportSize(MOBILE);
+ await open(page, slug, slug === "debug-panel-open" ? { panel: "authorization", seg: "evidence" } : {});
+ expect(await seriousAxeViolations(page)).toEqual([]);
+ });
+ }
+});
+
+/* ------------------------------------------------------- Capability clean room */
+
+/**
+ * The clean room is live-only by construction, so these tests stub the package
+ * session route rather than starting a real Codex process: what is under test
+ * here is the surface — entry, blank state, evidence-on-demand, identity
+ * rotation, failure honesty, and the narrow layout — not the tool loop, which
+ * `smoke:capability:cleanroom` and the server suite cover against the real thing.
+ */
+
+const CLEAN_ROOM_API = "**/api/capability/ui/cleanroom/session*";
+const DEVTOOLS_API = "**/api/capability/ui/devtools?*";
+
+function devtoolsPayload() {
+ const state = {
+ company: { id: "company-1", name: "Mock Paperclip" },
+ actors: [], tasks: [], comments: [], interactions: [], approvals: [], artifacts: [],
+ workProducts: [], blockers: [], workspaceServices: [], budgets: [], runs: [], wakes: [],
+ audit: [], decisions: [], idempotency: [], faults: [],
+ documents: [{
+ id: "document-1",
+ key: "plan",
+ title: "Runner plan",
+ revisions: [{ id: "revision-1", revision: 1, body: "# Real document body\n\nInspectable from DevTools.", changeSummary: "Initial draft", createdAt: "2026-08-10T12:00:00.000Z" }],
+ }],
+ };
+ return {
+ schema: "paperclip.capability.devtools.v1",
+ currentRevision: 4,
+ revisions: [{ revision: 4, at: "2026-08-10T12:00:00.000Z", turnId: "turn-1", operationId: "write_document", state }],
+ protocol: [], runtime: {}, authority: {},
+ };
+}
+
+function cleanRoomView(identifier: string, withTurn: boolean) {
+ const guard = {
+ id: `network-guard-session-${identifier}`,
+ turnId: "turn-0",
+ category: "session",
+ outcome: "no_real_paperclip_request",
+ reason: "Real Paperclip API requests: 0. Child PAPERCLIP_* environment keys: none.",
+ stateRevision: 3,
+ threadAnchorId: null,
+ };
+ return {
+ schema: "paperclip.capability.issue-thread-view.v1",
+ sessionId: `session-${identifier}`,
+ mode: "live",
+ identity: {
+ agentLabel: "Real Codex",
+ runnerLabel: "Real runnerd",
+ runnerAttached: true,
+ controlPlaneLabel: "Mock Paperclip",
+ controlPlaneTooltip: "All issue records are mock. No real Paperclip API is reachable.",
+ replaySource: null,
+ },
+ issue: {
+ identifier,
+ title: "Clean-room chat",
+ status: "in_progress",
+ priority: "medium",
+ assignee: "Mock Agent",
+ runState: `run-${identifier} · idle`,
+ scenarioId: "capability-clean-room",
+ fixtureProfile: "clean-room",
+ },
+ turns: withTurn
+ ? [
+ {
+ id: "turn-1",
+ ordinal: 1,
+ mode: "live",
+ toolCallCount: 1,
+ at: "2026-08-10T12:00:00.000Z",
+ stoppedByUser: false,
+ items: [
+ {
+ kind: "user_message",
+ id: "transcript-1",
+ at: "2026-08-10T12:00:00.000Z",
+ author: "You (board user)",
+ body: "Read this issue and record a status.",
+ },
+ {
+ kind: "tool_activity",
+ id: "item-call-call-1",
+ at: "2026-08-10T12:00:01.000Z",
+ status: "ok",
+ operationId: "report_progress",
+ summary: "state revision 3",
+ input: { body: "first status" },
+ result: { ok: true, stateRevision: 3 },
+ evidenceRef: { section: "calls", recordId: "call-1" },
+ },
+ {
+ kind: "agent_message",
+ id: "transcript-2",
+ at: "2026-08-10T12:00:02.000Z",
+ author: "Real Codex",
+ body: "Recorded a first status on the mock issue.",
+ streaming: false,
+ },
+ ],
+ },
+ ]
+ : [],
+ composer: { state: "ready", helper: null, reason: null, pendingInteractionId: null },
+ evidence: {
+ tools: [],
+ calls: [],
+ authorization: [],
+ control_plane: [guard],
+ runner: [],
+ state: [],
+ traceability: [],
+ parity: [],
+ },
+ connection: { state: "connected", attempt: 0 },
+ replay: null,
+ renderedAt: "2026-08-10T12:00:02.000Z",
+ };
+}
+
+function cleanRoomPayload(identifier: string, withTurn = false) {
+ return {
+ sessionId: `session-${identifier}`,
+ surface: "cleanroom",
+ identity: {
+ token: identifier.toLowerCase().replace("mck-", "tok"),
+ sequence: Number(identifier.slice(4)),
+ companyId: `company-cleanroom-${identifier}`,
+ actorId: `actor-cleanroom-${identifier}`,
+ taskId: `task-cleanroom-${identifier}`,
+ identifier,
+ },
+ limits: { maxTurns: 24, maxMessageBytes: 8192 },
+ view: cleanRoomView(identifier, withTurn),
+ };
+}
+
+/**
+ * A turn now answers with the NDJSON stream, so a stubbed turn has to speak it.
+ * These stubs deliver one settled frame: they exercise what a turn *produced*,
+ * while incremental delivery is proved against the real server below.
+ */
+function turnStreamBody(payload: unknown, interim: unknown[] = []): string {
+ const frames = interim.map((view, index) => ({
+ schema: "paperclip.capability.turn-stream.v1",
+ type: "frame",
+ seq: index + 1,
+ reason: "delta",
+ turnId: "turn-1",
+ view,
+ }));
+ return [
+ ...frames,
+ {
+ schema: "paperclip.capability.turn-stream.v1",
+ type: "settled",
+ seq: frames.length + 1,
+ payload,
+ },
+ ]
+ .map((frame) => `${JSON.stringify(frame)}\n`)
+ .join("");
+}
+
+async function stubCleanRoom(page: Page, identifiers: string[]) {
+ let opened = 0;
+ await page.route(CLEAN_ROOM_API, async (route) => {
+ const requestedSessionId = new URL(route.request().url()).searchParams.get("sessionId");
+ const identifier = requestedSessionId?.replace(/^session-/, "")
+ ?? identifiers[Math.min(opened, identifiers.length - 1)]
+ ?? "MCK-1000";
+ if (requestedSessionId === null) opened += 1;
+ await route.fulfill({
+ status: route.request().method() === "POST" ? 201 : 200,
+ contentType: "application/json",
+ body: JSON.stringify(cleanRoomPayload(identifier)),
+ });
+ });
+}
+
+async function openCleanRoom(page: Page, identifiers: string[] = ["MCK-1000"]) {
+ await stubCleanRoom(page, identifiers);
+ await page.addInitScript(() => window.localStorage.clear());
+ await page.goto("/#/chat");
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible();
+}
+
+test.describe("Capability clean-room chat", () => {
+ test("the landing surface offers a clean-room entry that needs no scenario", async ({ page }) => {
+ await open(page, "thread-baseline");
+ const entry = page.getByTestId("surface-chat-link");
+ await expect(entry).toBeVisible();
+ await expect(entry).toHaveRole("button");
+ await expect(entry).toContainText("New chat");
+ });
+
+ test("a new chat opens a blank live thread with no scenario controls", async ({ page }) => {
+ await openCleanRoom(page);
+
+ await expect(page).toHaveTitle("🫧 Mock Paperclip · Issue thread");
+ await expect(page.locator('[data-surface="chat"]')).toBeVisible();
+ await expect(page.getByTestId("clean-room-empty")).toBeVisible();
+ await expect(page.locator("[data-turn-id]")).toHaveCount(0);
+ await expect(page.getByTestId("scenario-picker")).toHaveCount(0);
+ await expect(page.getByTestId("replay-button")).toHaveCount(0);
+ await expect(page.getByTestId("replay-strip")).toHaveCount(0);
+
+ const chips = page.getByTestId("identity-chips");
+ await expect(chips.getByTestId("agent-chip")).toHaveText("Real Codex");
+ await expect(chips.getByTestId("runner-chip")).toContainText("Real runnerd");
+ await expect(chips.getByTestId("control-plane-chip")).toHaveText("Mock Paperclip");
+ await expect(page.locator('[data-composer-state="ready"]')).toHaveCount(1);
+ });
+
+ test("evidence stays collapsed until it is asked for", async ({ page }) => {
+ await openCleanRoom(page);
+
+ await expect(page.locator(".pit-panel")).toHaveCount(0);
+ await expect(page.getByTestId("evidence-toggle")).toHaveAttribute("aria-expanded", "false");
+
+ await page.getByTestId("evidence-toggle").click();
+ await expect(page.locator(".pit-panel")).toBeVisible();
+ await page.getByRole("button", { name: /Control plane/ }).click();
+ await expect(page.getByText("Real Paperclip API requests: 0")).toBeVisible();
+ });
+
+ test("documents written to company state are readable in DevTools", async ({ page }) => {
+ await page.route(DEVTOOLS_API, async (route) => {
+ await route.fulfill({ status: 200, contentType: "application/json", body: JSON.stringify(devtoolsPayload()) });
+ });
+ await openCleanRoom(page);
+ await page.getByTestId("evidence-toggle").click();
+ const panel = page.getByTestId("evidence-panel");
+ const tabs = panel.getByRole("tab");
+ await expect(tabs.first()).toHaveText(/Evidence/);
+ await expect(tabs.nth(1)).toHaveText(/Timeline/);
+ await expect(panel.locator(".pit-devtools-tabs .pit-icon")).toHaveCount(8);
+ const centerOffsets = await tabs.evaluateAll((elements) => elements.map((element) => {
+ const icon = element.querySelector(".pit-tab-glyph")?.getBoundingClientRect();
+ const label = element.querySelector(".pit-tab-glyph + span")?.getBoundingClientRect();
+ return icon === undefined || label === undefined
+ ? Number.POSITIVE_INFINITY
+ : Math.abs((icon.top + icon.height / 2) - (label.top + label.height / 2));
+ }));
+ expect(Math.max(...centerOffsets)).toBeLessThanOrEqual(1);
+ await expect(tabs.first()).toHaveAttribute("aria-selected", "true");
+ await expect(panel.locator("#evidence-turn-selector")).toBeVisible();
+ await panel.getByRole("tab", { name: /Timeline/ }).click();
+ await expect(panel.locator("#evidence-turn-selector")).toHaveCount(0);
+ await panel.getByRole("tab", { name: /Evidence/ }).click();
+ await expect(panel.locator("#evidence-turn-selector")).toBeVisible();
+ await page.getByRole("tab", { name: /Documents/ }).click();
+ await expect(page.getByRole("heading", { name: "Runner plan" })).toBeVisible();
+ await expect(page.getByText("Inspectable from DevTools.")).toBeVisible();
+ });
+
+ test("Runner events expand completely while stream deltas stay grouped", async ({ page }) => {
+ await page.addInitScript(() => window.localStorage.clear());
+ await page.route(CLEAN_ROOM_API, async (route) => {
+ const payload = cleanRoomPayload("MCK-1000");
+ const view = payload.view as CapabilityIssueThreadSnapshot;
+ view.evidence.runner = [
+ {
+ id: "session-1",
+ turnId: "turn-0",
+ kind: "session",
+ ordinal: 1,
+ detail: "session · started",
+ details: [
+ { label: "Action", value: "started" },
+ { label: "Runner", value: "paperclip-runnerd" },
+ ],
+ },
+ ...[2, 3].map((ordinal) => ({
+ id: `delta-${ordinal}`,
+ turnId: "turn-1",
+ kind: "provider_event",
+ ordinal,
+ detail: "provider event · assistant_delta",
+ details: [{ label: "Event", value: "assistant_delta" }],
+ })),
+ ];
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(payload),
+ });
+ });
+ await page.goto("/#/chat");
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible();
+
+ await page.getByTestId("evidence-toggle").click();
+ await page.getByRole("button", { name: /Runner & events/ }).click();
+
+ const session = page.locator('[data-record-id="session-1"]');
+ await session.locator("summary").click();
+ await expect(session).toContainText("paperclip-runnerd");
+
+ const group = page.locator('[data-runner-delta-group="assistant_delta"]');
+ await expect(group).toContainText("2 streamed updates");
+ await group.locator(":scope > summary").click();
+ await expect(group.locator(".pit-runner-event")).toHaveCount(2);
+ await group.locator(".pit-runner-event").first().locator("summary").click();
+ await expect(group.locator(".pit-runner-event").first()).toContainText("assistant_delta");
+ });
+
+ test("a sent message renders the live turn it produced", async ({ page }) => {
+ await openCleanRoom(page);
+ await page.route("**/api/capability/ui/message", async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/x-ndjson",
+ body: turnStreamBody({
+ sessionId: "session-MCK-1000",
+ surface: "cleanroom",
+ identity: cleanRoomPayload("MCK-1000").identity,
+ limits: { maxTurns: 24, maxMessageBytes: 8192 },
+ view: cleanRoomView("MCK-1000", true),
+ }),
+ });
+ });
+
+ await page.locator("#composer-input").fill("Read this issue and record a status.");
+ await page.getByTestId("composer-send").click();
+
+ await expect(page.getByTestId("clean-room-empty")).toHaveCount(0);
+ await expect(page.locator('[data-thread-item="user_message"]')).toBeVisible();
+ await expect(page.locator('[data-thread-item="agent_message"]')).toBeVisible();
+ await expect(page.locator('[data-tool-strip="report_progress"]')).toBeVisible();
+ });
+
+ test("New chat visibly rotates the mock identity", async ({ page }) => {
+ await openCleanRoom(page, ["MCK-1000", "MCK-2000"]);
+ await expect(page.locator(".pit-identifier")).toHaveText("MCK-1000");
+
+ await page.getByTestId("surface-chat-link").click();
+ await expect(page.locator(".pit-identifier")).toHaveText("MCK-2000");
+ await expect(page.getByTestId("clean-room-empty")).toBeVisible();
+ const history = page.getByRole("region", { name: "Session history" });
+ await expect(history.getByRole("button")).toHaveCount(2);
+
+ await history.getByRole("button", { name: /MCK-1000/ }).click();
+ await expect(page.locator(".pit-identifier")).toHaveText("MCK-1000");
+ await expect(page.locator("#composer-input")).toBeEnabled();
+
+ await history.getByRole("button", { name: /MCK-2000/ }).click();
+ await expect(page.locator(".pit-identifier")).toHaveText("MCK-2000");
+ await expect(page.locator("#composer-input")).toBeEnabled();
+ });
+
+ test("a failed live start reports the failure instead of a fixture", async ({ page }) => {
+ await page.addInitScript(() => window.localStorage.clear());
+ await page.route(CLEAN_ROOM_API, async (route) => {
+ await route.fulfill({
+ status: 500,
+ contentType: "application/json",
+ body: JSON.stringify({
+ error: "capability_issue_thread_unavailable",
+ message: "paperclip-runnerd exited before the Codex app-server was ready",
+ }),
+ });
+ });
+ await page.goto("/#/chat");
+
+ const error = page.getByTestId("surface-error");
+ await expect(error).toBeVisible();
+ await expect(error).toContainText("could not start the selected provider");
+ await expect(page.getByTestId("surface-error-retry")).toBeVisible();
+ await expect(page.locator("[data-turn-id]")).toHaveCount(0);
+ });
+
+ test("the narrow layout keeps thread and composer usable without horizontal scroll", async ({
+ page,
+ }) => {
+ await page.setViewportSize(MOBILE);
+ await openCleanRoom(page);
+
+ await expect(page.getByTestId("clean-room-empty")).toBeVisible();
+ await expect(page.locator("#composer-input")).toBeVisible();
+ const overflow = await page.evaluate(
+ () => document.documentElement.scrollWidth - document.documentElement.clientWidth,
+ );
+ expect(overflow).toBeLessThanOrEqual(0);
+
+ await page.getByTestId("segment-evidence").click();
+ await expect(page.locator(".pit-panel")).toBeVisible();
+ expect(
+ await page.evaluate(
+ () => document.documentElement.scrollWidth - document.documentElement.clientWidth,
+ ),
+ ).toBeLessThanOrEqual(0);
+ });
+
+ test("axe reports no serious or critical violation on the clean room", async ({ page }) => {
+ await page.setViewportSize(DESKTOP);
+ await openCleanRoom(page);
+ expect(await seriousAxeViolations(page)).toEqual([]);
+
+ await page.setViewportSize(MOBILE);
+ await expect(page.getByTestId("clean-room-empty")).toBeVisible();
+ expect(await seriousAxeViolations(page)).toEqual([]);
+ });
+});
+
+/* ------------------------------------ pending-request live region (TASK-16978) */
+
+/**
+ * Clean-room QA (TASK-16974) heard the visually hidden status region still
+ * announce "A request is waiting for your answer." after the request had been
+ * answered: the live path re-renders with the same request still pending while
+ * the submit is in flight, which used to re-arm the pending copy and leave it
+ * standing once the request settled. These tests pin the live region to the
+ * request's actual state across pending → answered → a later turn.
+ */
+
+const PENDING_ANNOUNCEMENT = "A request is waiting for your answer.";
+
+function questionsItem(state: "pending" | "answered" | "withdrawn"): CapabilityThreadItem {
+ return {
+ kind: "interaction",
+ id: "item-ix-1",
+ at: "2026-08-10T12:00:03.000Z",
+ interactionId: "ix-questions-01",
+ interactionKind: "questions",
+ title: "One scoping question",
+ prompt: "Answer this before I continue.",
+ payload: {
+ kind: "questions",
+ submitLabel: "Submit answers",
+ questions: [
+ {
+ id: "q1",
+ prompt: "Should the runner own process cleanup?",
+ control: "radio",
+ options: ["Yes — runner owns it", "No — harness owns it"],
+ required: true,
+ },
+ ],
+ },
+ state,
+ target: null,
+ stateLabel:
+ state === "pending" ? "Waiting for you" : state === "answered" ? "Answered" : "Withdrawn",
+ resolvedSummary:
+ state === "answered" ? ["Should the runner own process cleanup? Yes — runner owns it"] : [],
+ reason: null,
+ supersededBy: null,
+ evidenceRef: { section: "authorization", recordId: "authz-ix-1" },
+ };
+}
+
+/** A clean-room view carrying one request in the given state. */
+function interactionView(
+ identifier: string,
+ state: "pending" | "answered",
+ options: { withdrawn?: boolean; extraTurn?: boolean } = {},
+): CapabilityIssueThreadSnapshot {
+ const view = JSON.parse(
+ JSON.stringify(cleanRoomView(identifier, true)),
+ ) as CapabilityIssueThreadSnapshot;
+ view.turns[0].items.push(questionsItem(options.withdrawn === true ? "withdrawn" : state));
+ if (options.extraTurn === true) {
+ view.turns.push({
+ id: "turn-2",
+ ordinal: 2,
+ mode: "live",
+ toolCallCount: 0,
+ at: "2026-08-10T12:01:00.000Z",
+ stoppedByUser: false,
+ items: [
+ {
+ kind: "user_message",
+ id: "transcript-3",
+ at: "2026-08-10T12:01:00.000Z",
+ author: "You (board user)",
+ body: "Carry on with the plan.",
+ },
+ {
+ kind: "agent_message",
+ id: "transcript-4",
+ at: "2026-08-10T12:01:01.000Z",
+ author: "Real Codex",
+ body: "Folded the answer into the plan.",
+ streaming: false,
+ },
+ ],
+ });
+ }
+ view.composer =
+ state === "pending"
+ ? {
+ state: "waiting",
+ helper: "Answer the pending request above to continue.",
+ reason: null,
+ pendingInteractionId: "ix-questions-01",
+ }
+ : { state: "ready", helper: null, reason: null, pendingInteractionId: null };
+ return view;
+}
+
+async function openWithPendingRequest(page: Page) {
+ await page.addInitScript(() => window.localStorage.clear());
+ await page.route(CLEAN_ROOM_API, async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify({
+ ...cleanRoomPayload("MCK-1000"),
+ view: interactionView("MCK-1000", "pending"),
+ }),
+ });
+ });
+ await page.goto("/#/chat");
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible();
+}
+
+test.describe("pending-request live region", () => {
+ test("the pending announcement is dropped once the request is answered", async ({ page }) => {
+ await openWithPendingRequest(page);
+
+ const live = page.getByTestId("thread-live-region");
+ await expect(live).toHaveText(PENDING_ANNOUNCEMENT);
+ await expect(page.locator('[data-composer-state="waiting"]')).toBeVisible();
+
+ let interactionRequests = 0;
+ await page.route("**/api/capability/ui/interaction", async (route) => {
+ interactionRequests += 1;
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify({
+ ...cleanRoomPayload("MCK-1000"),
+ view: interactionView("MCK-1000", "answered"),
+ }),
+ });
+ });
+
+ const card = page.locator('[data-interaction-id="ix-questions-01"]');
+ await card.getByRole("radio", { name: "Yes — runner owns it" }).check();
+ await card.getByRole("button", { name: "Submit answers" }).click();
+
+ await expect(card).toHaveAttribute("data-interaction-state", "answered");
+ expect(interactionRequests).toBe(1);
+ await expect(page.locator('[data-composer-state="ready"]')).toHaveCount(1);
+ await expect(live).not.toContainText(PENDING_ANNOUNCEMENT);
+ await expect(live).toHaveText("Your answer was recorded.");
+
+ // A later turn must not resurrect the pending guidance either.
+ await page.route("**/api/capability/ui/message", async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/x-ndjson",
+ body: turnStreamBody({
+ ...cleanRoomPayload("MCK-1000"),
+ view: interactionView("MCK-1000", "answered", { extraTurn: true }),
+ }),
+ });
+ });
+ await page.locator("#composer-input").fill("Carry on with the plan.");
+ await page.getByTestId("composer-send").click();
+
+ await expect(page.locator("[data-turn-id]")).toHaveCount(2);
+ await expect(live).not.toContainText(PENDING_ANNOUNCEMENT);
+ });
+
+ test("a request withdrawn server-side replaces the pending announcement", async ({ page }) => {
+ // Nobody answers here: the request goes away while the session is
+ // reconnecting, so the correction has to come from the settled snapshot.
+ await page.addInitScript(() => window.localStorage.clear());
+ await page.route(CLEAN_ROOM_API, async (route) => {
+ const view = interactionView("MCK-1000", "pending");
+ view.composer.state = "reconnecting";
+ view.connection = { state: "reconnecting", attempt: 2 };
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify({ ...cleanRoomPayload("MCK-1000"), view }),
+ });
+ });
+ await page.goto("/#/chat");
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible();
+
+ const live = page.getByTestId("thread-live-region");
+ await expect(live).toHaveText(PENDING_ANNOUNCEMENT);
+
+ await page.route("**/api/capability/ui/reconnect", async (route) => {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify({
+ ...cleanRoomPayload("MCK-1000"),
+ view: interactionView("MCK-1000", "answered", { withdrawn: true }),
+ }),
+ });
+ });
+ await page.getByRole("button", { name: "Retry now" }).click();
+
+ await expect(page.locator('[data-interaction-state="withdrawn"]')).toBeVisible();
+ await expect(live).not.toContainText(PENDING_ANNOUNCEMENT);
+ await expect(live).toHaveText("The pending request is resolved. Withdrawn.");
+ });
+});
+
+/* -------------------------------------------- Capability streamed live turn */
+
+/**
+ * Streaming is proved against the real package server, not a stub: port 4185
+ * runs the same built bundle and the same session middleware with a scripted
+ * Codex provider (`vite.issue-thread-stream.config.ts`), so the NDJSON turn
+ * stream, the live session, the projection, and the browser client are all the
+ * shipped code. A `route.fulfill` stub cannot show this — it can only deliver a
+ * body in one piece, which is exactly the behaviour under repair.
+ */
+
+const STREAM_ORIGIN = "http://127.0.0.1:4185";
+const STREAM_REPLY =
+ "Reading the clean-room issue. It is blank, with one mock agent and one mock task. " +
+ "Recording a first status against the mock control plane. Done — every record stayed in the mock port.";
+
+async function openStreamingCleanRoom(page: Page): Promise {
+ await page.addInitScript(() => window.localStorage.clear());
+ await page.goto(`${STREAM_ORIGIN}/#/chat`);
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible({ timeout: 60_000 });
+}
+
+test.describe("Capability streamed live turn", () => {
+ test("shows sanitized thinking progress before assistant text arrives", async ({ page }) => {
+ await openStreamingCleanRoom(page);
+
+ await page.locator("#composer-input").fill("Show progress while you inspect this issue.");
+ await page.getByTestId("composer-send").click();
+ const liveActivity = page.getByTestId("live-activity");
+ await expect(liveActivity).toBeVisible();
+
+ const progress = page.locator('[data-thread-item="progress_activity"][data-activity="thinking"]');
+ await expect(progress).toBeVisible({ timeout: 30_000 });
+ await expect(progress).toHaveAttribute("data-status", "running");
+ await expect(progress).toContainText("Thinking");
+ await expect(liveActivity).toContainText(/Reasoning|Thinking|Codex/);
+ await expect(page.locator('[data-thread-item="agent_message"]')).toHaveCount(0);
+ await expect(page.locator("body")).not.toContainText(
+ "PRIVATE reasoning text must never reach the browser.",
+ );
+
+ await expect(page.locator('[data-thread-item="agent_message"]')).toContainText(STREAM_REPLY, {
+ timeout: 60_000,
+ });
+ await expect(progress).toHaveAttribute("data-status", "complete");
+ });
+
+ test("one assistant card grows while the turn POST is still open", async ({ page }) => {
+ await openStreamingCleanRoom(page);
+
+ const states: Array<{ id: string; body: string }> = [];
+ let pendingWhileGrowing = 0;
+ let settledWhileGrowing = 0;
+ let turnFinished = false;
+ // Sample the rendered card, not the network: what has to be proved is that
+ // a reader sees the reply grow, and only the DOM can say that.
+ const sampler = (async () => {
+ while (!turnFinished) {
+ const sample = await page.evaluate(() => {
+ const card = document.querySelector('[data-thread-item="agent_message"]');
+ const app = document.querySelector(".pit-app");
+ return {
+ id: card?.id ?? null,
+ body: card?.querySelector(".pit-card-body")?.textContent ?? null,
+ threadState: app?.getAttribute("data-thread-state") ?? null,
+ composer:
+ document.querySelector("[data-composer-state]")?.getAttribute("data-composer-state") ??
+ null,
+ };
+ });
+ if (sample.id !== null && sample.body !== null && sample.body.length > 0) {
+ if (states.at(-1)?.body !== sample.body) states.push({ id: sample.id, body: sample.body });
+ if (sample.body !== STREAM_REPLY) {
+ pendingWhileGrowing += 1;
+ if (sample.threadState === "settled") settledWhileGrowing += 1;
+ }
+ }
+ await page.waitForTimeout(60);
+ }
+ })();
+
+ await page.locator("#composer-input").fill("Read this issue and record a status.");
+ await page.getByTestId("composer-send").click();
+
+ // The whole reply only appears at the end; the card gets there in pieces.
+ await expect(page.locator('[data-thread-item="agent_message"]')).toContainText(STREAM_REPLY, {
+ timeout: 60_000,
+ });
+ turnFinished = true;
+ await sampler;
+
+ const bodies = states.map((state) => state.body);
+ expect(bodies.length).toBeGreaterThanOrEqual(3);
+ // One card: every observed state belonged to the same transcript entry.
+ expect(new Set(states.map((state) => state.id)).size).toBe(1);
+ for (let index = 1; index < bodies.length; index += 1) {
+ expect(bodies[index]!.startsWith(bodies[index - 1]!)).toBe(true);
+ }
+ expect(bodies.at(-1)).toBe(STREAM_REPLY);
+ // Partial states were observed, and the surface never claimed to be
+ // settled while one of them was on screen.
+ expect(pendingWhileGrowing).toBeGreaterThanOrEqual(2);
+ expect(settledWhileGrowing).toBe(0);
+
+ // Settled arrives only after the terminal frame.
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible();
+ await expect(page.locator('[data-composer-state="ready"]')).toHaveCount(1);
+ await expect(page.locator('[data-thread-item="agent_message"]')).toHaveCount(1);
+ await expect(page.locator('[data-thread-item="agent_message"]')).toHaveAttribute(
+ "data-streaming",
+ "false",
+ );
+ });
+
+ test("Stop interrupts the streamed turn and keeps the partial reply", async ({ page }) => {
+ await openStreamingCleanRoom(page);
+
+ await page.locator("#composer-input").fill("Start a long answer so I can stop it.");
+ await page.getByTestId("composer-send").click();
+
+ const card = page.locator('[data-thread-item="agent_message"]');
+ await expect(card).toBeVisible({ timeout: 30_000 });
+ await expect(card).toHaveAttribute("data-streaming", "true");
+ const partial = ((await card.locator(".pit-card-body").textContent()) ?? "").trim();
+ expect(partial.length).toBeGreaterThan(0);
+ expect(STREAM_REPLY.startsWith(partial)).toBe(true);
+
+ await page.getByTestId("composer-stop").click();
+
+ // The stopped turn settles on what it had said, and nothing lands later.
+ await expect(page.locator('[data-thread-state="settled"]')).toBeVisible({ timeout: 30_000 });
+ await expect(page.getByTestId("stopped-marker")).toBeVisible();
+ const stopped = ((await card.locator(".pit-card-body").textContent()) ?? "").trim();
+ expect(STREAM_REPLY.startsWith(stopped)).toBe(true);
+ expect(stopped).not.toBe(STREAM_REPLY);
+ await page.waitForTimeout(1_500);
+ expect(((await card.locator(".pit-card-body").textContent()) ?? "").trim()).toBe(stopped);
+ await expect(page.locator('[data-thread-item="agent_message"]')).toHaveCount(1);
+ });
+});
diff --git a/packages/paperclip-runner/devtools/issue-thread/playwright.config.ts b/packages/paperclip-runner/devtools/issue-thread/playwright.config.ts
new file mode 100644
index 0000000000..ddd35cecbc
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/playwright.config.ts
@@ -0,0 +1,46 @@
+import { defineConfig } from "@playwright/test";
+
+/**
+ * Focused browser suite for the Capability issue thread.
+ *
+ * The suite runs the built app through `vite preview` on loopback. Fake-mode
+ * routes render from package fixtures, so no provider, runnerd, or Codex
+ * process is started here.
+ */
+export default defineConfig({
+ testDir: ".",
+ testMatch: ["issue-thread.spec.ts"],
+ fullyParallel: false,
+ workers: 1,
+ reporter: "line",
+ use: {
+ baseURL: "http://127.0.0.1:4184",
+ viewport: { width: 1440, height: 900 },
+ launchOptions: {
+ // Hosts without the Playwright chromium system libraries can point at a
+ // preinstalled Chromium instead of running `pnpm verify:rootless`.
+ ...(process.env.PAPERCLIP_RUNNER_CHROMIUM_PATH === undefined
+ ? {}
+ : { executablePath: process.env.PAPERCLIP_RUNNER_CHROMIUM_PATH }),
+ },
+ },
+ webServer: [
+ {
+ command:
+ "pnpm exec vite preview --config vite.issue-thread.config.ts --host 127.0.0.1 --port 4184",
+ url: "http://127.0.0.1:4184",
+ reuseExistingServer: false,
+ timeout: 60_000,
+ },
+ {
+ // Same app and same package server, with a scripted Codex provider, so
+ // the streaming checks can drive a real NDJSON turn without a provider
+ // process. See `vite.issue-thread-stream.config.ts`.
+ command:
+ "pnpm exec vite preview --config vite.issue-thread-stream.config.ts --host 127.0.0.1 --port 4185",
+ url: "http://127.0.0.1:4185",
+ reuseExistingServer: false,
+ timeout: 60_000,
+ },
+ ],
+});
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/App.tsx b/packages/paperclip-runner/devtools/issue-thread/src/App.tsx
new file mode 100644
index 0000000000..5ebeca9d49
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/App.tsx
@@ -0,0 +1,1589 @@
+import { useCallback, useEffect, useLayoutEffect, useMemo, useRef, useState } from "react";
+
+import {
+ CAPABILITY_DEFAULT_FIXTURE_PROFILE,
+ capabilityIssueThreadFixture,
+} from "../../../src/issue-thread/fixtures";
+import type {
+ CapabilityEvidenceSectionId,
+ CapabilityIssueThreadSnapshot,
+} from "../../../src/issue-thread/types";
+import type { CapabilityDevtoolsSnapshot } from "../../../src/devtools";
+import { capabilityDenialCount } from "../../../src/issue-thread/types";
+import { Composer } from "./Composer";
+import { EvidencePanel } from "./EvidencePanel";
+import { Icon } from "./Icons";
+import { IssueHeader } from "./IssueHeader";
+import { applyFakeInteractionResponse } from "./fake-store";
+import type { CapabilityInteractionResponse } from "./InteractionCard";
+import {
+ capabilityLiveClient,
+ recallSession,
+ rememberSession,
+ type CapabilityCleanRoomIdentity,
+ type CapabilityHarnessConfiguration,
+} from "./live-client";
+import { parseCapabilityRoute, capabilityRouteHref, type CapabilityRoute } from "./route";
+import { SurfaceNav, type CapabilityChatHistoryItem } from "./SurfaceNav";
+import { EvalAssertions, TurnGroup, type EvalAssertion } from "./ThreadItems";
+
+const PANEL_OPEN_KEY = "paperclip-runner.capability.panel.open";
+/**
+ * The clean room keeps its own panel preference. Evidence is collapsed by
+ * default there on purpose (revision 5: "the default view reads as plain
+ * chat"), and inheriting an explorer session's opened drawer would quietly
+ * break that on the very first visit.
+ */
+const CHAT_PANEL_OPEN_KEY = "paperclip-runner.capability.chat.panel.open";
+const PANEL_WIDTH_KEY = "paperclip-runner.capability.panel.width";
+const PANEL_MIN = 320;
+const PANEL_MAX = 960;
+/** Play-all cadence for the replay strip (§6); slow enough to read a turn. */
+const REPLAY_STEP_MS = 800;
+/**
+ * The pending-request guidance is a constant because the settle path has to
+ * recognise its own stale announcement: a screen reader must not still be told
+ * a request is waiting after it was answered (TASK-16978).
+ */
+const PENDING_ANNOUNCEMENT = "A request is waiting for your answer.";
+const ANSWERED_ANNOUNCEMENT = "Your answer was recorded.";
+const SETTLED_ANNOUNCEMENT = "The pending request is resolved.";
+const SCENARIOS = ["hb-baseline", "dp-documents", "ix-interactions", "ar-artifacts", "wc-workspace-changes", "fr-file-reference", "pe-structured-plan", "te-command-execution", "mp-mcp-progress", "rs-web-research", "da-subagent-delegation", "mr-model-routing", "cc-context-compaction", "ag-generated-artifact", "rv-review-mode", "hk-hook-lifecycle", "mc-memory-citations", "sr-safety-review", "ti-terminal-input", "wt-intentional-wait", "pn-provider-notices", "ax-acpx-lifecycle", "ax-acpx-permissions", "ax-acpx-tools", "ax-acpx-events", "ax-acpx-failures"];
+const CHAT_HISTORY_KEY = "paperclip-runner.capability.chat.history.v1";
+const CHAT_HARNESS_KEY = "paperclip-runner.capability.chat.harness.v1";
+const MODEL_PRESETS = {
+ codex: ["gpt-5.4-mini", "gpt-5.4"],
+ opencode: ["openrouter/deepseek/deepseek-v4-flash-0731"],
+ acpx: ["openrouter/deepseek/deepseek-v4-flash-0731", "claude-sonnet-5", "gpt-5.6-sol"],
+} as const;
+const ACPX_AGENT_MODELS = {
+ claude: "claude-sonnet-5",
+ codex: "gpt-5.6-sol",
+} as const;
+
+function acpxLabel(configuration: CapabilityHarnessConfiguration | null | undefined): string {
+ if (configuration?.provider !== "acpx") return configuration?.provider ?? "starting";
+ const agent = configuration.acpxAgent ?? "codex";
+ return `Real ${agent === "claude" ? "Claude" : "Codex"} via ACPX`;
+}
+
+interface EmbeddedEvalCheck {
+ id: string;
+ title: string;
+ description: string;
+ passed: boolean;
+ detail: string;
+ definition: Record;
+ anchor: { kind: "item" | "turn" | "run"; id: string };
+}
+
+interface EmbeddedEvalReport {
+ attemptId: string;
+ caseId: string;
+ disposition: string;
+ passed: boolean;
+ checks: EmbeddedEvalCheck[];
+ run: {
+ model: string;
+ provider: string;
+ driver: string;
+ providerVersion: string | null;
+ runnerProvider: string;
+ configuration: string;
+ sessionId: string;
+ providerSessionId: string | null;
+ agentVersion: string | null;
+ retainedSession: boolean | null;
+ retainedSessionStatus: string | null;
+ fixtureDigest: string;
+ runnerPackageDigest: string;
+ runnerdDigest: string;
+ runnerBuild: string;
+ startedAt: string;
+ finishedAt: string;
+ durationMs: number;
+ initialRevision: number;
+ finalRevision: number;
+ usage: {
+ agentTurns: number;
+ providerRequests: number | null;
+ inputTokens: number;
+ outputTokens: number;
+ cachedInputTokens: number;
+ reasoningTokens: number;
+ providerReportedCostNanodollars?: number;
+ estimatedCostNanodollars: number;
+ pricingVersion: string;
+ } | null;
+ };
+ view: CapabilityIssueThreadSnapshot;
+ devtools: CapabilityDevtoolsSnapshot;
+ navigation: { suiteHref: string; previous: { label: string; href: string } | null; next: { label: string; href: string } | null };
+}
+
+declare global {
+ interface Window { __PAPERCLIP_EVAL_REPORT__?: EmbeddedEvalReport }
+}
+
+interface StoredChatSession {
+ sessionId: string;
+ snapshot: CapabilityIssueThreadSnapshot;
+ identity: CapabilityCleanRoomIdentity | null;
+ updatedAt: string;
+ configuration?: CapabilityHarnessConfiguration;
+ runtime?: Awaited>["runtime"];
+}
+
+function defaultHarness(provider: CapabilityHarnessConfiguration["provider"] = "codex"): CapabilityHarnessConfiguration {
+ return {
+ provider,
+ model: MODEL_PRESETS[provider][0],
+ ...(provider === "acpx" ? { acpxAgent: "codex" as const } : {}),
+ lifecyclePolicy: { mode: "warm", idleTimeoutMs: 300_000 },
+ };
+}
+
+function readHarnessConfiguration(): CapabilityHarnessConfiguration {
+ try {
+ const value = JSON.parse(window.localStorage.getItem(CHAT_HARNESS_KEY) ?? "null") as Partial | null;
+ if (value && (value.provider === "codex" || value.provider === "opencode" || value.provider === "acpx") && typeof value.model === "string") {
+ const lifecyclePolicy = value.lifecyclePolicy?.mode === "per_turn"
+ ? { mode: "per_turn" as const, idleTimeoutMs: null }
+ : value.lifecyclePolicy?.mode === "warm" && Number.isSafeInteger(value.lifecyclePolicy.idleTimeoutMs) && Number(value.lifecyclePolicy.idleTimeoutMs) > 0
+ ? { mode: "warm" as const, idleTimeoutMs: Number(value.lifecyclePolicy.idleTimeoutMs) }
+ : { mode: "warm" as const, idleTimeoutMs: 300_000 };
+ return {
+ provider: value.provider,
+ model: value.model,
+ ...(value.provider === "acpx" ? {
+ acpxAgent: value.acpxAgent === "claude" ? value.acpxAgent : "codex",
+ } : {}),
+ lifecyclePolicy,
+ };
+ }
+ } catch {
+ // Invalid preferences fall back to the qualified defaults.
+ }
+ return defaultHarness();
+}
+
+function persistHarnessConfiguration(configuration: CapabilityHarnessConfiguration): void {
+ try {
+ window.localStorage.setItem(CHAT_HARNESS_KEY, JSON.stringify(configuration));
+ } catch {
+ // Provider/model preference is convenient but not authority-bearing.
+ }
+}
+
+function readChatHistory(): StoredChatSession[] {
+ try {
+ const parsed = JSON.parse(window.localStorage.getItem(CHAT_HISTORY_KEY) ?? "[]") as unknown;
+ if (!Array.isArray(parsed)) return [];
+ return parsed.filter((item): item is StoredChatSession =>
+ typeof item === "object" && item !== null &&
+ typeof (item as StoredChatSession).sessionId === "string" &&
+ typeof (item as StoredChatSession).updatedAt === "string" &&
+ typeof (item as StoredChatSession).snapshot === "object",
+ ).slice(0, 12);
+ } catch {
+ return [];
+ }
+}
+
+function persistChatHistory(history: StoredChatSession[]): void {
+ try {
+ window.localStorage.setItem(CHAT_HISTORY_KEY, JSON.stringify(history.slice(0, 12)));
+ } catch {
+ // History is a convenience; a blocked or full storage quota must not stop chat.
+ }
+}
+
+function readStoredNumber(key: string, fallback: number): number {
+ try {
+ const raw = window.localStorage.getItem(key);
+ const parsed = raw === null ? Number.NaN : Number.parseInt(raw, 10);
+ return Number.isFinite(parsed) ? Math.min(PANEL_MAX, Math.max(PANEL_MIN, parsed)) : fallback;
+ } catch {
+ return fallback;
+ }
+}
+
+function readStoredFlag(key: string, fallback: boolean): boolean {
+ try {
+ const raw = window.localStorage.getItem(key);
+ return raw === null ? fallback : raw === "true";
+ } catch {
+ return fallback;
+ }
+}
+
+/**
+ * Scroll an Evidence target into the panel viewport. `start` alignment has to
+ * account for the sticky panel head, which would otherwise sit on top of the
+ * section header the deep link was supposed to reveal.
+ */
+function scrollEvidenceIntoView(target: Element, block: "start" | "center"): void {
+ target.scrollIntoView({ block });
+ if (block !== "start") return;
+ const panel = target.closest(".pit-panel");
+ const head = panel?.querySelector(".pit-panel-chrome");
+ if (panel == null || head == null) return;
+ panel.scrollTop = Math.max(0, panel.scrollTop - head.offsetHeight);
+}
+
+function describe(cause: unknown): string {
+ return String(cause instanceof Error ? cause.message : cause);
+}
+
+/** Resolution copy for a request that settled without this client answering it. */
+function settledAnnouncement(snapshot: CapabilityIssueThreadSnapshot, interactionId: string): string {
+ for (const turn of snapshot.turns) {
+ for (const item of turn.items) {
+ if (item.kind === "interaction" && item.interactionId === interactionId) {
+ return `${SETTLED_ANNOUNCEMENT} ${item.stateLabel}.`;
+ }
+ }
+ }
+ return SETTLED_ANNOUNCEMENT;
+}
+
+function currentTurnActivity(snapshot: CapabilityIssueThreadSnapshot): string {
+ const harness = snapshot.identity.agentLabel.replace(/^Real /, "");
+ const turn = snapshot.turns.at(-1);
+ if (turn === undefined) return `Dispatching turn to ${harness}`;
+ const activity = [...turn.items].reverse().find((item) =>
+ item.kind === "tool_activity" || item.kind === "progress_activity" || item.kind === "provider_activity" ||
+ (item.kind === "agent_message" && item.streaming),
+ );
+ if (activity?.kind === "tool_activity") {
+ return activity.status === "running"
+ ? `Paperclip tool · ${activity.operationId}`
+ : `Paperclip tool completed · ${activity.operationId}`;
+ }
+ if (activity?.kind === "progress_activity") return activity.summary;
+ if (activity?.kind === "agent_message") return `Receiving ${harness} response`;
+ return `Waiting for ${harness} activity`;
+}
+
+function useRoute(): CapabilityRoute {
+ const [route, setRoute] = useState(() => parseCapabilityRoute(window.location));
+ useEffect(() => {
+ const onChange = () => setRoute(parseCapabilityRoute(window.location));
+ window.addEventListener("hashchange", onChange);
+ window.addEventListener("popstate", onChange);
+ return () => {
+ window.removeEventListener("hashchange", onChange);
+ window.removeEventListener("popstate", onChange);
+ };
+ }, []);
+ return route;
+}
+
+function useLayout(): "side" | "overlay" | "segment" {
+ const [layout, setLayout] = useState<"side" | "overlay" | "segment">(() =>
+ window.innerWidth <= 767 ? "segment" : window.innerWidth <= 1100 ? "overlay" : "side",
+ );
+ useEffect(() => {
+ const onResize = () =>
+ setLayout(window.innerWidth <= 767 ? "segment" : window.innerWidth <= 1100 ? "overlay" : "side");
+ window.addEventListener("resize", onResize);
+ return () => window.removeEventListener("resize", onResize);
+ }, []);
+ return layout;
+}
+
+export function App() {
+ const embeddedEval = window.__PAPERCLIP_EVAL_REPORT__ ?? null;
+ const route = useRoute();
+ const chat = route.surface === "chat";
+ const layout = useLayout();
+
+ useEffect(() => {
+ if (embeddedEval !== null) {
+ document.title = `${embeddedEval.passed ? "✓" : "✕"} ${embeddedEval.caseId} · paperclip-runner eval`;
+ return;
+ }
+ document.title = chat
+ ? "🫧 Mock Paperclip · Issue thread"
+ : "🧯 Mock Paperclip · Issue thread";
+ }, [chat, embeddedEval]);
+
+ const [snapshot, setSnapshot] = useState(null);
+ const [devtools, setDevtools] = useState(null);
+ const [identity, setIdentity] = useState(null);
+ const initialHarnessRef = useRef(readHarnessConfiguration());
+ const [harness, setHarness] = useState(initialHarnessRef.current);
+ const [activeHarness, setActiveHarness] = useState(null);
+ const [providerRuntime, setProviderRuntime] = useState>["runtime"]>(undefined);
+ const [chatHistory, setChatHistory] = useState(readChatHistory);
+ const [historicSessionId, setHistoricSessionId] = useState(null);
+ const liveRoomRef = useRef(null);
+ /**
+ * Which surface produced `snapshot`. A hash change commits the new route in
+ * the same frame that still renders the old snapshot, so without this a
+ * capture or a test can catch one frame of "settled" that belongs to the
+ * surface it just navigated away from.
+ */
+ const [snapshotSurface, setSnapshotSurface] = useState(route.surface);
+ const [error, setError] = useState(null);
+ const [actionError, setActionError] = useState(null);
+ const [loadNonce, setLoadNonce] = useState(0);
+ const [settled, setSettled] = useState(false);
+ const [panelOpen, setPanelOpen] = useState(() => embeddedEval !== null ||
+ readStoredFlag(route.surface === "chat" ? CHAT_PANEL_OPEN_KEY : PANEL_OPEN_KEY, false));
+ const [panelWidth, setPanelWidth] = useState(() => readStoredNumber(PANEL_WIDTH_KEY, 384));
+ const [segment, setSegment] = useState<"thread" | "evidence">(route.segment);
+ const [openSections, setOpenSections] = useState(["tools"]);
+ const [selectedTurnId, setSelectedTurnId] = useState("all");
+ const [highlightedRecordId, setHighlightedRecordId] = useState(route.record);
+ const [focusInteractionId, setFocusInteractionId] = useState(null);
+ const [confirmReset, setConfirmReset] = useState(false);
+ const [announcement, setAnnouncement] = useState("");
+ const [showJump, setShowJump] = useState(false);
+ const [reconnected, setReconnected] = useState(false);
+ const [playing, setPlaying] = useState(false);
+ /** True while a turn stream is open, so the surface never reads as settled. */
+ const [streamingTurn, setStreamingTurn] = useState(false);
+ const scrollRef = useRef(null);
+ const cancelResetRef = useRef(null);
+ /** Last announced pending request, so the live region tracks transitions. */
+ const announcedPendingRef = useRef(null);
+ /**
+ * Turn generation. Reset, `New chat`, and leaving the surface all bump it, so
+ * a frame from a turn that has since been abandoned is dropped instead of
+ * appended to a thread it no longer belongs to.
+ */
+ const turnGenerationRef = useRef(0);
+ const turnAbortRef = useRef(null);
+
+ useEffect(() => {
+ if (embeddedEval !== null) setPanelOpen(true);
+ }, [embeddedEval]);
+
+ /** Abandons any open turn stream: the server sees the disconnect and stops. */
+ const abandonTurn = useCallback(() => {
+ turnGenerationRef.current += 1;
+ const controller = turnAbortRef.current;
+ turnAbortRef.current = null;
+ setStreamingTurn(false);
+ controller?.abort();
+ }, []);
+
+ /* --------------------------------------------------------------- loading */
+
+ useEffect(() => {
+ let cancelled = false;
+ setSettled(false);
+ setError(null);
+ if (embeddedEval !== null) {
+ setIdentity(null);
+ setSnapshot({
+ ...embeddedEval.view,
+ composer: { state: "disabled", helper: null, reason: "Immutable eval recording", pendingInteractionId: null },
+ });
+ setDevtools(embeddedEval.devtools);
+ setSnapshotSurface("issue");
+ } else if (chat) {
+ // The clean room has no fixture fallback: if real Codex and real runnerd
+ // cannot start, the surface says so rather than rendering a canned thread.
+ void (async () => {
+ try {
+ const response = await capabilityLiveClient.loadCleanRoom(
+ recallSession("cleanroom"),
+ initialHarnessRef.current,
+ );
+ if (cancelled) return;
+ rememberSession(response.sessionId, "cleanroom");
+ setIdentity(response.identity ?? null);
+ const configuration = response.configuration ?? initialHarnessRef.current;
+ setActiveHarness(configuration);
+ // The server-returned immutable session configuration is authoritative
+ // for both the active status and the options initially shown for the
+ // next chat after a restore.
+ setHarness(configuration);
+ setProviderRuntime(response.runtime);
+ setHistoricSessionId(null);
+ setSnapshot(response.view);
+ setSnapshotSurface("chat");
+ } catch (cause) {
+ if (!cancelled) setError(String(cause instanceof Error ? cause.message : cause));
+ }
+ })();
+ } else if (route.mode === "live") {
+ void (async () => {
+ try {
+ const response = await capabilityLiveClient.load(recallSession());
+ if (cancelled) return;
+ rememberSession(response.sessionId);
+ setSnapshot(response.view);
+ setSnapshotSurface("issue");
+ } catch (cause) {
+ if (!cancelled) setError(String(cause instanceof Error ? cause.message : cause));
+ }
+ })();
+ } else {
+ setIdentity(null);
+ setSnapshot(capabilityIssueThreadFixture(route.shot ?? "thread-baseline", route.fixtureProfile));
+ setSnapshotSurface("issue");
+ }
+ return () => {
+ cancelled = true;
+ // Leaving the surface (or reloading it) must not leave a turn streaming
+ // into a thread that is no longer on screen.
+ abandonTurn();
+ };
+ }, [abandonTurn, chat, embeddedEval, route.mode, route.shot, route.fixtureProfile, loadNonce]);
+
+ /* ------------------------------------------------------- deep-link params */
+
+ useEffect(() => {
+ if (route.panel === null) return;
+ setPanelOpen(true);
+ setSegment("evidence");
+ setOpenSections((current) =>
+ current.includes(route.panel as CapabilityEvidenceSectionId)
+ ? current
+ : [...current, route.panel as CapabilityEvidenceSectionId],
+ );
+ setHighlightedRecordId(route.record);
+ }, [route.panel, route.record]);
+
+ useLayoutEffect(() => {
+ // A deep link has to land on what it addressed. With `rec` that is the
+ // record; without it, the opened section's header — otherwise the link
+ // half-arrives with the section still below the fold. Wait for the bundled
+ // faces before measuring: a host fallback can otherwise move the retained
+ // scroll position by a pixel even after the final webfonts replace it.
+ if (route.panel === null || snapshot === null) return;
+ const panel = route.panel;
+ const record = route.record;
+ let cancelled = false;
+ void (async () => {
+ if (typeof document.fonts?.ready?.then === "function") {
+ await document.fonts.ready;
+ }
+ if (cancelled) return;
+ const target =
+ record === null
+ ? document.querySelector(`[data-evidence-section="${CSS.escape(panel)}"]`)
+ : document.querySelector(`[data-record-id="${CSS.escape(record)}"]`);
+ if (target !== null) {
+ scrollEvidenceIntoView(target, record === null ? "start" : "center");
+ }
+ })();
+ return () => {
+ cancelled = true;
+ };
+ }, [route.panel, route.record, snapshot, panelOpen, segment, openSections]);
+
+ useEffect(() => {
+ setSegment(route.segment);
+ }, [route.segment]);
+
+ /* -------------------------------------------------- settle + auto-follow */
+
+ useEffect(() => {
+ if (snapshot === null) return;
+ let cancelled = false;
+ const scroller = scrollRef.current;
+ void (async () => {
+ if (typeof document.fonts?.ready?.then === "function") {
+ await document.fonts.ready;
+ }
+ if (cancelled) return;
+ if (scroller !== null) scroller.scrollTop = scroller.scrollHeight;
+ await new Promise((resolve) => requestAnimationFrame(() => resolve()));
+ if (!cancelled) setSettled(true);
+ })();
+ return () => {
+ cancelled = true;
+ };
+ }, [snapshot]);
+
+ useEffect(() => {
+ if (!panelOpen || snapshot === null || route.mode !== "live") return;
+ if (historicSessionId !== null) return;
+ let cancelled = false;
+ void capabilityLiveClient.devtools(snapshot.sessionId)
+ .then((next) => { if (!cancelled) setDevtools(next); })
+ .catch((cause) => { if (!cancelled) setActionError(describe(cause)); });
+ return () => { cancelled = true; };
+ }, [historicSessionId, panelOpen, route.mode, snapshot?.renderedAt, snapshot?.sessionId]);
+
+ useEffect(() => {
+ if (!chat || snapshot === null || historicSessionId !== null) return;
+ const record: StoredChatSession = {
+ sessionId: snapshot.sessionId,
+ snapshot,
+ identity,
+ updatedAt: snapshot.renderedAt,
+ ...(activeHarness === null ? {} : { configuration: activeHarness }),
+ ...(providerRuntime === undefined ? {} : { runtime: providerRuntime }),
+ };
+ liveRoomRef.current = record;
+ setChatHistory((current) => {
+ const next = [record, ...current.filter((item) => item.sessionId !== record.sessionId)].slice(0, 12);
+ persistChatHistory(next);
+ return next;
+ });
+ }, [activeHarness, chat, historicSessionId, identity, providerRuntime, snapshot?.renderedAt, snapshot?.sessionId]);
+
+ useEffect(() => {
+ if (snapshot === null) return;
+ const pending = snapshot.composer.pendingInteractionId;
+ const announced = announcedPendingRef.current;
+ announcedPendingRef.current = pending;
+
+ if (pending !== null) {
+ // Announce a request once, when it arrives. Re-announcing on every
+ // snapshot would talk over the outcome the user just triggered: a live
+ // submit re-renders with the same request still pending.
+ if (pending !== announced) setAnnouncement(PENDING_ANNOUNCEMENT);
+ return;
+ }
+ if (snapshot.connection.state === "reconnecting") {
+ setAnnouncement(`Connection lost — retrying (attempt ${snapshot.connection.attempt}).`);
+ return;
+ }
+ if (announced === null) return;
+ // The request settled. Anything the resolution itself announced stands;
+ // only the now-false "still waiting" guidance is replaced, so a screen
+ // reader never carries pending-state copy into later turns (TASK-16978).
+ setAnnouncement((current) =>
+ current === PENDING_ANNOUNCEMENT ? settledAnnouncement(snapshot, announced) : current,
+ );
+ }, [snapshot]);
+
+ /* -------------------------------------------------------------- handlers */
+
+ const persistPanel = useCallback(
+ (open: boolean, width: number) => {
+ try {
+ window.localStorage.setItem(chat ? CHAT_PANEL_OPEN_KEY : PANEL_OPEN_KEY, String(open));
+ window.localStorage.setItem(PANEL_WIDTH_KEY, String(width));
+ } catch {
+ // Panel preference is a convenience, never a correctness requirement.
+ }
+ },
+ [chat],
+ );
+
+ const openEvidence = useCallback(
+ (section: CapabilityEvidenceSectionId, recordId: string) => {
+ setPanelOpen(true);
+ setSegment("evidence");
+ setSelectedTurnId("all");
+ setOpenSections((current) => (current.includes(section) ? current : [...current, section]));
+ setHighlightedRecordId(recordId);
+ persistPanel(true, panelWidth);
+ window.setTimeout(() => {
+ document
+ .querySelector(`[data-record-id="${CSS.escape(recordId)}"]`)
+ ?.scrollIntoView({ block: "center" });
+ }, 0);
+ },
+ [panelWidth, persistPanel],
+ );
+
+ const closeEvidence = useCallback(() => {
+ setPanelOpen(false);
+ persistPanel(false, panelWidth);
+ if (layout === "segment") setSegment("thread");
+ // Focus came into the panel when it opened (§9.2), so hand it back to the
+ // visible control that owns the panel rather than dropping it on `body`.
+ window.setTimeout(() => {
+ const candidates = ['[data-testid="evidence-toggle"]', '[data-testid="segment-thread"]'];
+ for (const selector of candidates) {
+ const control = document.querySelector(selector);
+ if (control !== null && control.offsetParent !== null) {
+ control.focus();
+ return;
+ }
+ }
+ }, 0);
+ }, [layout, panelWidth, persistPanel]);
+
+ const jumpToThread = useCallback((anchorId: string) => {
+ setSegment("thread");
+ const anchor = document.getElementById(anchorId) ?? document.querySelector(`#${CSS.escape(anchorId)}`);
+ anchor?.scrollIntoView({ block: "center" });
+ }, []);
+
+ const respond = useCallback(
+ (response: CapabilityInteractionResponse) => {
+ if (snapshot === null) return;
+ if (route.mode !== "live") {
+ setSnapshot(applyFakeInteractionResponse(snapshot, response));
+ setAnnouncement(ANSWERED_ANNOUNCEMENT);
+ return;
+ }
+
+ // Network effects must never live inside a React state updater. Strict
+ // Mode intentionally invokes updater functions more than once in
+ // development, which previously submitted every interaction twice: the
+ // first request resolved it and the duplicate produced "not pending".
+ const sessionId = snapshot.sessionId;
+ setActionError(null);
+ setSnapshot((current) => current === null ? current : {
+ ...current,
+ turns: current.turns.map((turn) => ({
+ ...turn,
+ items: turn.items.map((item) =>
+ item.kind === "interaction" && item.interactionId === response.interactionId
+ ? { ...item, state: "submitting" as const, stateLabel: "Submitting…" }
+ : item,
+ ),
+ })),
+ });
+ void capabilityLiveClient
+ .respond(sessionId, response.interactionId, response.outcome, response.result)
+ .then((next) => {
+ setSnapshot(next.view);
+ setAnnouncement(ANSWERED_ANNOUNCEMENT);
+ })
+ .catch((cause) => {
+ setActionError(describe(cause));
+ setSnapshot((current) => current === null ? current : {
+ ...current,
+ turns: current.turns.map((turn) => ({
+ ...turn,
+ items: turn.items.map((item) =>
+ item.kind === "interaction" &&
+ item.interactionId === response.interactionId &&
+ item.state === "submitting"
+ ? { ...item, state: "pending" as const, stateLabel: "Waiting for you" }
+ : item,
+ ),
+ })),
+ });
+ });
+ },
+ [route.mode, snapshot],
+ );
+
+ /**
+ * Sends one message and renders the turn as it arrives.
+ *
+ * Each frame is a whole server projection, so the surface still never patches
+ * state locally — it just gets more than one projection per turn. The settled
+ * payload is applied last and stays the authority.
+ */
+ const send = useCallback(
+ (message: string) => {
+ if (snapshot === null) return;
+ if (route.mode !== "live") return;
+ setActionError(null);
+ abandonTurn();
+ const generation = turnGenerationRef.current;
+ const controller = new AbortController();
+ turnAbortRef.current = controller;
+ setStreamingTurn(true);
+ setSnapshot((current) =>
+ current === null ? current : { ...current, composer: { ...current.composer, state: "sending" } },
+ );
+ void capabilityLiveClient
+ .send(snapshot.sessionId, message, {
+ signal: controller.signal,
+ onFrame: (view) => {
+ if (turnGenerationRef.current === generation) setSnapshot(view);
+ },
+ })
+ .then((next) => {
+ if (turnGenerationRef.current === generation) {
+ setSnapshot(next.view);
+ setProviderRuntime(next.runtime);
+ }
+ })
+ .catch((cause) => {
+ if (turnGenerationRef.current !== generation || controller.signal.aborted) return;
+ setActionError(describe(cause));
+ // A refused turn must not strand the composer in `sending`.
+ setSnapshot((stale) =>
+ stale === null ? stale : { ...stale, composer: { ...stale.composer, state: "ready" } },
+ );
+ })
+ .finally(() => {
+ if (turnGenerationRef.current !== generation) return;
+ turnAbortRef.current = null;
+ setStreamingTurn(false);
+ });
+ },
+ [abandonTurn, route.mode, snapshot],
+ );
+
+ const stop = useCallback(() => {
+ setSnapshot((current) => {
+ if (current === null) return current;
+ if (route.mode === "live") {
+ // Stop reaches the provider interrupt through its own request; the open
+ // turn stream keeps rendering and its terminal frame — not this
+ // response — decides what the stopped turn finally looks like.
+ const streaming = turnAbortRef.current !== null;
+ void capabilityLiveClient
+ .stop(current.sessionId)
+ .then((next) => {
+ if (!streaming && turnAbortRef.current === null) setSnapshot(next.view);
+ })
+ .catch((cause) => setActionError(describe(cause)));
+ return current;
+ }
+ const turns = current.turns.map((turn, index) =>
+ index === current.turns.length - 1 ? { ...turn, stoppedByUser: true } : turn,
+ );
+ return {
+ ...current,
+ turns,
+ composer: { state: "ready", helper: null, reason: null, pendingInteractionId: null },
+ };
+ });
+ setAnnouncement("Turn stopped. Partial output is preserved.");
+ }, [route.mode]);
+
+ /**
+ * Adopts a rotated clean room. Reset and `New chat` both land here: the
+ * server always answers with a new session id and new mock identities, so the
+ * client's job is only to forget the old ones.
+ */
+ const adoptCleanRoom = useCallback((next: Awaited>) => {
+ rememberSession(next.sessionId, "cleanroom");
+ setIdentity(next.identity ?? null);
+ const configuration = next.configuration ?? harness;
+ setActiveHarness(configuration);
+ setHarness(configuration);
+ setProviderRuntime(next.runtime);
+ setHistoricSessionId(null);
+ setSnapshot(next.view);
+ setDevtools(null);
+ setError(null);
+ setActionError(null);
+ setAnnouncement(
+ `New clean-room chat started on ${next.view.issue.identifier}. The previous session was closed.`,
+ );
+ }, [harness]);
+
+ const newChat = useCallback(() => {
+ const model = harness.model?.trim() ?? "";
+ if (!model) {
+ setActionError("Choose or enter a model before starting a new chat.");
+ return;
+ }
+ if (harness.provider === "opencode" && !model.includes("/")) {
+ setActionError("OpenCode models must use provider/model form.");
+ return;
+ }
+ if (harness.provider === "acpx") {
+ const agent = harness.acpxAgent ?? "codex";
+ if (model !== ACPX_AGENT_MODELS[agent]) {
+ setActionError(`The qualified ACPX ${agent} profile requires exact model ${ACPX_AGENT_MODELS[agent]}.`);
+ return;
+ }
+ }
+ const configuration = { ...harness, model };
+ persistHarnessConfiguration(configuration);
+ setConfirmReset(false);
+ // The room this turn belongs to is about to be retired, so the stream is
+ // dropped before the request that retires it.
+ abandonTurn();
+ setAnnouncement("Starting a new clean-room chat…");
+ void capabilityLiveClient
+ .newCleanRoom(liveRoomRef.current?.sessionId ?? snapshot?.sessionId ?? null, configuration)
+ .then(adoptCleanRoom)
+ .catch((cause) => setActionError(describe(cause)));
+ }, [abandonTurn, adoptCleanRoom, harness, snapshot]);
+
+ const selectChatHistory = useCallback((sessionId: string) => {
+ const live = liveRoomRef.current;
+ if (live?.sessionId === sessionId) {
+ setSnapshot(live.snapshot);
+ setIdentity(live.identity);
+ setActiveHarness(live.configuration ?? null);
+ if (live.configuration !== undefined) setHarness(live.configuration);
+ setProviderRuntime(live.runtime);
+ setDevtools(null);
+ setHistoricSessionId(null);
+ setActionError(null);
+ return;
+ }
+ const archived = chatHistory.find((item) => item.sessionId === sessionId);
+ if (archived === undefined) return;
+ abandonTurn();
+ const configuration = archived.configuration ?? defaultHarness();
+ void capabilityLiveClient.loadCleanRoom(sessionId, configuration)
+ .then((restored) => {
+ rememberSession(restored.sessionId, "cleanroom");
+ setSnapshot(restored.view);
+ setIdentity(restored.identity ?? archived.identity);
+ setActiveHarness(restored.configuration ?? configuration);
+ setHarness(restored.configuration ?? configuration);
+ setProviderRuntime(restored.runtime);
+ setDevtools(null);
+ setHistoricSessionId(null);
+ setActionError(null);
+ setAnnouncement("Archived chat restored. The runner will resume when you send a message.");
+ })
+ .catch((cause) => setActionError(describe(cause)));
+ }, [abandonTurn, chatHistory]);
+
+ const reset = useCallback(() => {
+ setConfirmReset(false);
+ abandonTurn();
+ if (chat && snapshot !== null) {
+ void capabilityLiveClient
+ .reset(snapshot.sessionId)
+ .then(adoptCleanRoom)
+ .catch((cause) => setActionError(describe(cause)));
+ return;
+ }
+ if (route.mode === "live" && snapshot !== null) {
+ void capabilityLiveClient
+ .reset(snapshot.sessionId)
+ .then((next) => {
+ rememberSession(next.sessionId);
+ setSnapshot(next.view);
+ })
+ .catch((cause) => setActionError(describe(cause)));
+ return;
+ }
+ setSnapshot(capabilityIssueThreadFixture("thread-baseline", route.fixtureProfile));
+ setAnnouncement("Scenario reset. The mock state is back to its clean seed.");
+ }, [abandonTurn, adoptCleanRoom, chat, route.fixtureProfile, route.mode, snapshot]);
+
+ const retry = useCallback(() => {
+ if (route.mode === "live" && snapshot !== null) {
+ void capabilityLiveClient
+ .reconnect(snapshot.sessionId)
+ .then((next) => setSnapshot(next.view))
+ .catch((cause) => setActionError(describe(cause)));
+ return;
+ }
+ setSnapshot((current) =>
+ current === null
+ ? current
+ : {
+ ...current,
+ connection: { state: "connected", attempt: 0 },
+ composer: { state: "ready", helper: null, reason: null, pendingInteractionId: null },
+ },
+ );
+ setReconnected(true);
+ window.setTimeout(() => setReconnected(false), 3_000);
+ }, [route.mode, snapshot]);
+
+ useEffect(() => {
+ if (!confirmReset) return;
+ cancelResetRef.current?.focus();
+ function onKeyDown(event: KeyboardEvent) {
+ if (event.key === "Escape") setConfirmReset(false);
+ }
+ document.addEventListener("keydown", onKeyDown);
+ return () => document.removeEventListener("keydown", onKeyDown);
+ }, [confirmReset]);
+
+ /* ----------------------------------------------------------- replay (§6) */
+
+ // `?at=` is the source of truth for where the recording is parked,
+ // so step / next-turn / play-all and the deep link all move the same value.
+ const replay = useMemo(() => {
+ if (snapshot === null || snapshot.replay === null) return null;
+ const { total } = snapshot.replay;
+ const ordinal =
+ route.at === null ? snapshot.replay.ordinal : Math.min(total, Math.max(0, route.at));
+ return { ordinal, total };
+ }, [snapshot, route.at]);
+
+ const seekReplay = useCallback(
+ (ordinal: number) => {
+ setPlaying(false);
+ window.location.hash = capabilityRouteHref(route, { at: ordinal });
+ },
+ [route],
+ );
+
+ useEffect(() => {
+ if (!playing || replay === null) return;
+ if (replay.ordinal >= replay.total) {
+ setPlaying(false);
+ return;
+ }
+ const timer = window.setTimeout(() => {
+ window.location.hash = capabilityRouteHref(route, { at: replay.ordinal + 1 });
+ }, REPLAY_STEP_MS);
+ return () => window.clearTimeout(timer);
+ }, [playing, replay, route]);
+
+ useEffect(() => {
+ if (replay === null) setPlaying(false);
+ }, [replay]);
+
+ /* ---------------------------------------------------------------- render */
+
+ const denialCount = useMemo(
+ () => (snapshot === null ? 0 : capabilityDenialCount(snapshot.evidence, null)),
+ [snapshot],
+ );
+ const historyItems: CapabilityChatHistoryItem[] = chatHistory.map((item) => ({
+ sessionId: item.sessionId,
+ identifier: item.snapshot.issue.identifier,
+ title: item.snapshot.issue.title,
+ updatedAt: item.updatedAt,
+ current: liveRoomRef.current?.sessionId === item.sessionId,
+ }));
+ const surfaceNav = (
+ { window.location.hash = "#/chat"; }}
+ onSelectHistory={selectChatHistory}
+ />
+ );
+
+ if (error !== null) {
+ return (
+
+ {surfaceNav}
+
+
+ {chat
+ ? `The clean-room chat could not start the selected provider: ${error}`
+ : error}
+
+ {chat ? (
+ <>
+
+ The clean room only runs against the selected real provider through real runnerd, so it does not fall
+ back to a fixture or a recording.
+
+
+
+ Provider
+ {
+ const provider = event.target.value as CapabilityHarnessConfiguration["provider"];
+ setHarness((current) => ({
+ ...defaultHarness(provider),
+ lifecyclePolicy: current.lifecyclePolicy,
+ }));
+ }}
+ >
+ Codex
+ OpenCode
+ ACPX
+
+
+ {harness.provider === "acpx" ? (
+
+ ACP agent
+ {
+ const acpxAgent = event.target.value as NonNullable;
+ setHarness((current) => ({
+ ...current,
+ acpxAgent,
+ model: ACPX_AGENT_MODELS[acpxAgent],
+ }));
+ }}
+ >
+ Claude
+ Codex (control)
+
+
+ ) : null}
+
+ Execution
+ setHarness((current) => ({
+ ...current,
+ lifecyclePolicy: event.target.value === "per_turn"
+ ? { mode: "per_turn", idleTimeoutMs: null }
+ : { mode: "warm", idleTimeoutMs: 300_000 },
+ }))}
+ >
+ Warm session
+ Turn by turn
+
+
+ {harness.lifecyclePolicy.mode === "warm" ? (
+
+ Idle timeout (seconds)
+ setHarness((current) => ({
+ ...current,
+ lifecyclePolicy: {
+ mode: "warm",
+ idleTimeoutMs: Math.max(1, Number(event.target.value)) * 1_000,
+ },
+ }))}
+ />
+
+ ) : null}
+
+ Model
+ setHarness((current) => ({ ...current, model: event.target.value }))}
+ />
+
+ {MODEL_PRESETS[harness.provider].map((model) => )}
+
+
+
+ Start new chat
+
+
+ >
+ ) : null}
+ setLoadNonce((current) => current + 1)}
+ >
+ Try again
+
+
+
+ );
+ }
+
+ if (snapshot === null) {
+ return (
+
+ {surfaceNav}
+
+
+ {chat
+ ? "Starting real runnerd and the selected provider for a fresh mock tenant…"
+ : "Loading…"}
+
+
+
+ );
+ }
+
+ const showThread = layout !== "segment" || segment === "thread";
+ const showPanel = layout === "segment" ? segment === "evidence" : panelOpen;
+
+ return (
+
+ {embeddedEval === null ? surfaceNav : null}
+
+
{
+ const next = !panelOpen;
+ setPanelOpen(next);
+ persistPanel(next, panelWidth);
+ if (layout === "segment") setSegment(next ? "evidence" : "thread");
+ }}
+ onSelectScenario={(scenario) => {
+ window.location.hash = capabilityRouteHref(route, { fixtureProfile: scenario });
+ }}
+ onReplay={() => {
+ window.location.hash = capabilityRouteHref(route, {
+ shot: "replay-mode",
+ mode: "replay",
+ at: 12,
+ });
+ }}
+ onReset={() => setConfirmReset(true)}
+ onStop={stop}
+ onSelectSegment={setSegment}
+ />
+
+ {chat && embeddedEval === null ? (
+
+
+ Provider
+ {
+ const provider = event.target.value as CapabilityHarnessConfiguration["provider"];
+ setHarness((current) => ({
+ ...defaultHarness(provider),
+ lifecyclePolicy: current.lifecyclePolicy,
+ }));
+ }}
+ >
+ Codex
+ OpenCode
+ ACPX
+
+
+ {harness.provider === "acpx" ? (
+
+ ACP agent
+ {
+ const acpxAgent = event.target.value as NonNullable;
+ setHarness((current) => ({
+ ...current,
+ acpxAgent,
+ model: ACPX_AGENT_MODELS[acpxAgent],
+ }));
+ }}
+ >
+ Claude
+ Codex (control)
+
+
+ ) : null}
+
+ Execution
+ setHarness((current) => ({
+ ...current,
+ lifecyclePolicy: event.target.value === "per_turn"
+ ? { mode: "per_turn", idleTimeoutMs: null }
+ : { mode: "warm", idleTimeoutMs: 300_000 },
+ }))}
+ >
+ Warm session
+ Turn by turn
+
+
+ {harness.lifecyclePolicy.mode === "warm" ? (
+
+ Idle timeout (seconds)
+ {
+ const seconds = Math.max(1, Number.parseInt(event.target.value || "1", 10));
+ setHarness((current) => ({
+ ...current,
+ lifecyclePolicy: { mode: "warm", idleTimeoutMs: seconds * 1_000 },
+ }));
+ }}
+ />
+
+ ) : null}
+
+ Model
+ setHarness((current) => ({ ...current, model: event.target.value }))}
+ />
+
+ {MODEL_PRESETS[harness.provider].map((model) => )}
+
+
+
+ Start new chat
+
+
+ Active: {acpxLabel(activeHarness)}
+ {activeHarness?.model ? ` · ${activeHarness.model}` : ""}
+ {activeHarness ? ` · ${activeHarness.lifecyclePolicy.mode === "warm" ? `warm ${Math.round(activeHarness.lifecyclePolicy.idleTimeoutMs / 1_000)}s` : "turn by turn"}` : ""}
+ {providerRuntime?.runnerPid ? ` · runner PID ${providerRuntime.runnerPid}` : ""}
+ {activeHarness?.provider === "acpx" && providerRuntime?.sidecarPid ? ` · sidecar PID ${providerRuntime.sidecarPid}` : ""}
+ {activeHarness?.provider === "acpx" && providerRuntime?.agentPid ? ` · agent PID ${providerRuntime.agentPid}` : ""}
+ {activeHarness?.provider === "acpx" && providerRuntime?.driverSessionId ? ` · ACPX record ${providerRuntime.driverSessionId}` : ""}
+ {providerRuntime?.providerSessionId ? ` · session ${providerRuntime.providerSessionId}` : ""}
+ {activeHarness?.provider === "acpx" && providerRuntime?.providerVersion ? ` · ACPX ${providerRuntime.providerVersion}` : ""}
+ {activeHarness?.provider === "acpx" && providerRuntime?.agentServerVersion ? ` · agent ${providerRuntime.agentServerVersion}` : ""}
+ {activeHarness?.provider === "acpx" && providerRuntime?.acpProtocolVersion ? ` · ACP ${providerRuntime.acpProtocolVersion}` : ""}
+ {providerRuntime?.status ? ` · ${providerRuntime.status}` : ""}
+
+
+ ) : null}
+
+ {embeddedEval !== null ? (
+
+ ← All results
+ {embeddedEval.navigation.previous ? ← {embeddedEval.navigation.previous.label} : }
+ {embeddedEval.passed ? "PASS" : "FAIL"} · {embeddedEval.attemptId}
+ {embeddedEval.navigation.next ? {embeddedEval.navigation.next.label} → : }
+
+ ) : null}
+
+ {actionError !== null ? (
+
+ ⚠
+ {actionError}
+ setActionError(null)}
+ >
+ Dismiss
+
+
+ ) : null}
+
+ {snapshot.connection.state === "reconnecting" ? (
+
+ ⏳
+ Connection lost — retrying (attempt {snapshot.connection.attempt})
+
+ ) : reconnected ? (
+
+ ✓
+ Reconnected
+
+ ) : null}
+
+ {streamingTurn ? (
+
+
+
+ {currentTurnActivity(snapshot)}
+ live
+
+ ) : null}
+
+ {replay !== null ? (
+
+
+ Replay {replay.ordinal}/{replay.total}
+
+
+
seekReplay(replay.ordinal - 1)}
+ >
+ Step back
+
+
= replay.total}
+ data-testid="replay-next-turn"
+ onClick={() => seekReplay(replay.ordinal + 1)}
+ >
+ Next turn
+
+
= replay.total}
+ data-testid="replay-play-all"
+ onClick={() => setPlaying((current) => !current)}
+ >
+ {playing ? "❚❚" : "▶"}
+ {playing ? "Pause" : "Play all"}
+
+
+ ) : null}
+
+
+
+
+
{
+ const element = event.currentTarget;
+ const distance =
+ element.scrollHeight - element.scrollTop - element.clientHeight;
+ setShowJump(distance > 300);
+ }}
+ >
+
+ {embeddedEval !== null ? (
+
Eval execution
+ ) : null}
+ {snapshot.turns.length === 0 ? (
+
+ Start a clean-room chat
+
+ This is a blank thread on a brand-new mock tenant:{" "}
+ {snapshot.issue.identifier} in{" "}
+ Mock Paperclip (clean room) . Nothing has been said, called, or
+ recorded yet.
+
+
+ Your first message starts a real {snapshot.identity.agentLabel.replace(/^Real /, "")} turn through real runnerd. The agent may
+ call the semantic tools this session exposes, and every record it creates lands
+ in the mock control plane only — never a real Paperclip API. Detailed tool,
+ policy, event, and state evidence stays in the Evidence drawer until you open
+ it.
+
+
+ ) : (
+ snapshot.turns.map((turn) => (
+
check.anchor.kind === "item")
+ .map((check) => [check.anchor.id, [{ ...check } satisfies EvalAssertion]]) ?? [])}
+ terminalAssertions={embeddedEval?.checks
+ .filter((check) => check.anchor.kind === "turn" && check.anchor.id === turn.id)
+ .map((check) => ({ ...check })) ?? []}
+ />
+ ))
+ )}
+ {embeddedEval !== null ? (
+
+ Post-run state
+ Final mock control-plane revision {embeddedEval.run.finalRevision}
+ check.anchor.kind === "run")} />
+
+ ) : null}
+
+
+ {showJump ? (
+
{
+ const element = scrollRef.current;
+ if (element !== null) element.scrollTop = element.scrollHeight;
+ setShowJump(false);
+ }}
+ >
+ Jump to latest
+
+ ) : null}
+
+
+ (chat ? newChat() : setConfirmReset(true))}
+ resetLabel={chat ? "New chat" : "Reset scenario"}
+ onFocusPending={(interactionId) => {
+ setFocusInteractionId(interactionId);
+ document
+ .getElementById(`interaction-${interactionId}`)
+ ?.scrollIntoView({ block: "center" });
+ }}
+ />
+
+
+ {showPanel && layout === "side" ? (
+
{
+ const splitter = event.currentTarget;
+ splitter.setPointerCapture(event.pointerId);
+ const resize = (pointer: PointerEvent) => {
+ const next = Math.min(PANEL_MAX, Math.max(PANEL_MIN, window.innerWidth - pointer.clientX));
+ setPanelWidth(next);
+ };
+ const finish = () => {
+ splitter.removeEventListener("pointermove", resize);
+ splitter.removeEventListener("pointerup", finish);
+ splitter.removeEventListener("pointercancel", finish);
+ setPanelWidth((current) => {
+ persistPanel(panelOpen, current);
+ return current;
+ });
+ };
+ splitter.addEventListener("pointermove", resize);
+ splitter.addEventListener("pointerup", finish);
+ splitter.addEventListener("pointercancel", finish);
+ }}
+ onKeyDown={(event) => {
+ const step = event.shiftKey ? 64 : 16;
+ if (event.key === "ArrowLeft") {
+ event.preventDefault();
+ setPanelWidth((current) => {
+ const next = Math.min(PANEL_MAX, current + step);
+ persistPanel(panelOpen, next);
+ return next;
+ });
+ } else if (event.key === "ArrowRight") {
+ event.preventDefault();
+ setPanelWidth((current) => {
+ const next = Math.max(PANEL_MIN, current - step);
+ persistPanel(panelOpen, next);
+ return next;
+ });
+ }
+ }}
+ />
+ ) : null}
+
+ {showPanel ? (
+ {
+ abandonTurn();
+ setActionError(null);
+ void capabilityLiveClient.fork(snapshot.sessionId, revision)
+ .then((next) => {
+ rememberSession(next.sessionId, chat ? "cleanroom" : "issue");
+ setSnapshot(next.view);
+ setIdentity(next.identity ?? null);
+ setDevtools(null);
+ })
+ .catch((cause) => setActionError(describe(cause)));
+ }}
+ layout={layout}
+ width={panelWidth}
+ selectedTurnId={selectedTurnId}
+ openSections={openSections}
+ highlightedRecordId={highlightedRecordId}
+ onSelectTurn={setSelectedTurnId}
+ onToggleSection={(section) =>
+ setOpenSections((current) =>
+ current.includes(section)
+ ? current.filter((entry) => entry !== section)
+ : [...current, section],
+ )
+ }
+ onClose={closeEvidence}
+ onJumpToThread={jumpToThread}
+ onInvokeTool={route.mode === "live" && historicSessionId === null ? async (operationId, input) => {
+ setActionError(null);
+ const next = await capabilityLiveClient.invokeTool(snapshot.sessionId, operationId, input);
+ setSnapshot(next.view);
+ setSelectedTurnId(next.toolTurnId);
+ return next.toolResult;
+ } : undefined}
+ />
+ ) : null}
+
+
+ {confirmReset ? (
+
+
+
+ {chat ? "Reset this chat?" : "Reset scenario?"}
+
+
+ {chat
+ ? "This stops any active turn, closes this session's authority, and opens a new mock tenant with new identities. The transcript and its mock records will be lost."
+ : "This clears the mock state and starts a clean session. The transcript will be lost."}
+
+
+ setConfirmReset(false)}
+ >
+ Cancel
+
+
+ {chat ? "Reset chat" : "Reset scenario"}
+
+
+
+
+ ) : null}
+
+
+ {announcement}
+
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/Composer.tsx b/packages/paperclip-runner/devtools/issue-thread/src/Composer.tsx
new file mode 100644
index 0000000000..e1c5242b38
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/Composer.tsx
@@ -0,0 +1,161 @@
+import { useEffect, useRef, useState, type FormEvent } from "react";
+
+import type { CapabilityComposerModel } from "../../../src/issue-thread/types";
+
+/**
+ * Composer (contract §4). `data-composer-state` is the single test and
+ * screenshot hook. Draft text survives refresh through localStorage keyed by
+ * session id; Stop never discards the transcript.
+ */
+
+export interface ComposerProps {
+ model: CapabilityComposerModel;
+ sessionId: string;
+ onSend: (message: string) => void;
+ onStop: () => void;
+ onRetry: () => void;
+ onReset: () => void;
+ onFocusPending: (interactionId: string) => void;
+ /** The clean room resets into a new tenant, so it names the action itself. */
+ resetLabel?: string;
+}
+
+function draftKey(sessionId: string): string {
+ return `paperclip-runner.capability.draft.${sessionId}`;
+}
+
+function readDraft(sessionId: string): string {
+ try {
+ return window.localStorage.getItem(draftKey(sessionId)) ?? "";
+ } catch {
+ return "";
+ }
+}
+
+export function Composer(props: ComposerProps) {
+ const {
+ model,
+ sessionId,
+ onSend,
+ onStop,
+ onRetry,
+ onReset,
+ onFocusPending,
+ resetLabel = "Reset scenario",
+ } = props;
+ const [value, setValue] = useState(() => readDraft(sessionId));
+ const inputRef = useRef(null);
+
+ useEffect(() => {
+ setValue(readDraft(sessionId));
+ }, [sessionId]);
+
+ useEffect(() => {
+ try {
+ window.localStorage.setItem(draftKey(sessionId), value);
+ } catch {
+ // A blocked storage quota must not break the session.
+ }
+ }, [sessionId, value]);
+
+ const editable = model.state === "ready" || model.state === "streaming";
+ const canSend = editable && value.trim().length > 0;
+
+ function submit(event: FormEvent) {
+ event.preventDefault();
+ if (!canSend) return;
+ onSend(value.trim());
+ setValue("");
+ }
+
+ return (
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/DevtoolsInspector.tsx b/packages/paperclip-runner/devtools/issue-thread/src/DevtoolsInspector.tsx
new file mode 100644
index 0000000000..acfd180775
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/DevtoolsInspector.tsx
@@ -0,0 +1,646 @@
+import { useEffect, useMemo, useState } from "react";
+
+import type { CapabilityDevtoolsSnapshot } from "../../../src/devtools";
+import { Icon, type IconName } from "./Icons";
+import { EvalAssertions, type EvalAssertion } from "./ThreadItems";
+
+export type CapabilityDevtoolsTab =
+ | "eval"
+ | "evidence"
+ | "timeline"
+ | "state"
+ | "diff"
+ | "documents"
+ | "protocol"
+ | "trace"
+ | "runtime"
+ | "authority";
+type Json = null | boolean | number | string | Json[] | { [key: string]: Json };
+
+export interface EvalInspectorReport {
+ attemptId: string;
+ caseId: string;
+ disposition: string;
+ passed: boolean;
+ checks: EvalAssertion[];
+ run: {
+ model: string;
+ provider: string;
+ driver: string;
+ providerVersion: string | null;
+ runnerProvider: string;
+ configuration: string;
+ sessionId: string;
+ providerSessionId: string | null;
+ agentVersion: string | null;
+ retainedSession: boolean | null;
+ retainedSessionStatus: string | null;
+ fixtureDigest: string;
+ runnerPackageDigest: string;
+ runnerdDigest: string;
+ runnerBuild: string;
+ startedAt: string;
+ finishedAt: string;
+ durationMs: number;
+ initialRevision: number;
+ finalRevision: number;
+ usage: {
+ agentTurns: number;
+ providerRequests: number | null;
+ inputTokens: number;
+ outputTokens: number;
+ cachedInputTokens: number;
+ reasoningTokens: number;
+ providerReportedCostNanodollars?: number;
+ estimatedCostNanodollars: number;
+ pricingVersion: string;
+ } | null;
+ };
+}
+
+function JsonTree({ value, name = "root" }: { value: Json; name?: string }) {
+ if (value === null || typeof value !== "object") {
+ return {JSON.stringify(value)} ;
+ }
+ const entries = Array.isArray(value)
+ ? value.map((entry, index) => [String(index), entry] as const)
+ : Object.entries(value);
+ return (
+
+
+ {name}{" "}
+
+ {Array.isArray(value) ? `[${entries.length}]` : `{${entries.length}}`}
+
+
+
+ {entries.map(([key, entry]) => (
+
+ {key}
+ {entry !== null && typeof entry === "object" ? (
+
+ ) : (
+
+ )}
+
+ ))}
+
+
+ );
+}
+
+interface DiffRow {
+ path: string;
+ before: Json | undefined;
+ after: Json | undefined;
+}
+
+function diff(
+ before: Json | undefined,
+ after: Json | undefined,
+ path = "state",
+): DiffRow[] {
+ if (JSON.stringify(before) === JSON.stringify(after)) return [];
+ if (
+ before !== null &&
+ after !== null &&
+ typeof before === "object" &&
+ typeof after === "object" &&
+ !Array.isArray(before) &&
+ !Array.isArray(after)
+ ) {
+ const keys = new Set([...Object.keys(before), ...Object.keys(after)]);
+ return [...keys].flatMap((key) =>
+ diff(before[key], after[key], `${path}.${key}`),
+ );
+ }
+ return [{ path, before, after }];
+}
+
+function download(snapshot: CapabilityDevtoolsSnapshot) {
+ const blob = new Blob([JSON.stringify(snapshot, null, 2)], {
+ type: "application/json",
+ });
+ const url = URL.createObjectURL(blob);
+ const anchor = document.createElement("a");
+ anchor.href = url;
+ anchor.download = `paperclip-devtools-r${snapshot.currentRevision}.json`;
+ anchor.click();
+ URL.revokeObjectURL(url);
+}
+
+const TABS: ReadonlyArray<{
+ id: CapabilityDevtoolsTab;
+ icon: IconName;
+ label: string;
+}> = [
+ { id: "evidence", icon: "evidence", label: "Evidence" },
+ { id: "timeline", icon: "timeline", label: "Timeline" },
+ { id: "state", icon: "state", label: "State" },
+ { id: "diff", icon: "diff", label: "Diff" },
+ { id: "documents", icon: "documents", label: "Documents" },
+ { id: "protocol", icon: "protocol", label: "Protocol" },
+ { id: "trace", icon: "protocol", label: "Provider trace" },
+ { id: "runtime", icon: "terminal", label: "Runtime" },
+ { id: "authority", icon: "authority", label: "Authority" },
+];
+
+function documentsOf(state: Json): Array<{
+ id: string;
+ key: string;
+ title: string;
+ revisions: Array<{
+ id: string;
+ revision: number;
+ body: string;
+ changeSummary: string | null;
+ createdAt: string;
+ }>;
+}> {
+ if (state === null || typeof state !== "object" || Array.isArray(state))
+ return [];
+ const documents = state.documents;
+ if (!Array.isArray(documents)) return [];
+ return documents.flatMap((value) => {
+ if (value === null || typeof value !== "object" || Array.isArray(value))
+ return [];
+ const revisions = Array.isArray(value.revisions)
+ ? value.revisions.flatMap((candidate) => {
+ if (
+ candidate === null ||
+ typeof candidate !== "object" ||
+ Array.isArray(candidate)
+ )
+ return [];
+ return [
+ {
+ id: String(candidate.id ?? ""),
+ revision: Number(candidate.revision ?? 0),
+ body: String(candidate.body ?? ""),
+ changeSummary:
+ candidate.changeSummary === null
+ ? null
+ : String(candidate.changeSummary ?? ""),
+ createdAt: String(candidate.createdAt ?? ""),
+ },
+ ];
+ })
+ : [];
+ return [
+ {
+ id: String(value.id ?? ""),
+ key: String(value.key ?? ""),
+ title: String(value.title ?? "Untitled"),
+ revisions,
+ },
+ ];
+ });
+}
+
+export function DevtoolsInspector({
+ snapshot,
+ onFork,
+ tab,
+ onTabChange,
+ evalReport,
+}: {
+ snapshot: CapabilityDevtoolsSnapshot;
+ onFork: (revision: number) => void;
+ tab: CapabilityDevtoolsTab;
+ onTabChange: (tab: CapabilityDevtoolsTab) => void;
+ evalReport?: EvalInspectorReport | null;
+}) {
+ const [query, setQuery] = useState("");
+ const latest = snapshot.revisions.at(-1)!;
+ const [revision, setRevision] = useState(latest.revision);
+ const [following, setFollowing] = useState(true);
+ const [compareRevision, setCompareRevision] = useState(
+ snapshot.revisions[0]!.revision,
+ );
+ const selected =
+ snapshot.revisions.find((entry) => entry.revision === revision) ?? latest;
+ const compared =
+ snapshot.revisions.find((entry) => entry.revision === compareRevision) ??
+ snapshot.revisions[0]!;
+ const rows = useMemo(
+ () => diff(compared.state as Json, selected.state as Json),
+ [compared.state, selected.state],
+ );
+ const normalized = query.trim().toLowerCase();
+ const revisions = snapshot.revisions.filter(
+ (entry) =>
+ normalized.length === 0 ||
+ JSON.stringify(entry).toLowerCase().includes(normalized),
+ );
+ const protocol = snapshot.protocol.filter(
+ (entry) =>
+ normalized.length === 0 ||
+ JSON.stringify(entry).toLowerCase().includes(normalized),
+ );
+ const documents = useMemo(
+ () => documentsOf(selected.state as Json),
+ [selected.state],
+ );
+ const [documentId, setDocumentId] = useState("");
+ const document =
+ documents.find((entry) => entry.id === documentId) ?? documents[0] ?? null;
+ const [documentRevision, setDocumentRevision] = useState(0);
+ const visibleDocumentRevision =
+ document?.revisions.find((entry) => entry.revision === documentRevision) ??
+ document?.revisions.at(-1) ??
+ null;
+ useEffect(() => {
+ if (following) setRevision(latest.revision);
+ }, [following, latest.revision]);
+
+ return (
+
+
+
+ State monitor
+
+ setQuery(event.target.value)}
+ placeholder="operation, entity, turn…"
+ />
+ setFollowing((value) => !value)}
+ title={following ? "Pause live updates" : "Follow latest revision"}
+ aria-label={
+ following ? "Pause live updates" : "Follow latest revision"
+ }
+ >
+
+
+ download(snapshot)}
+ title="Export redacted snapshot"
+ aria-label="Export redacted snapshot"
+ >
+
+
+ onFork(revision)}
+ >
+ Fork r{revision}
+
+
+
+ {[
+ ...(evalReport
+ ? [
+ {
+ id: "eval" as const,
+ icon: "evidence" as const,
+ label: "Eval",
+ },
+ ]
+ : []),
+ ...TABS,
+ ].map(({ id, icon, label }) => (
+ onTabChange(id)}
+ >
+
+
+
+ {label}
+
+ ))}
+
+ {tab === "eval" && evalReport ? (
+
+
+
← Eval suite
+
+ {evalReport.passed
+ ? "PASS"
+ : evalReport.disposition.replaceAll("_", " ").toUpperCase()}
+
+
{evalReport.attemptId}
+
+
+
+
Model
+
+ {evalReport.run.model.startsWith(`${evalReport.run.provider}/`)
+ ? evalReport.run.model
+ : `${evalReport.run.provider}/${evalReport.run.model}`}
+
+
+
+
Configuration
+ {evalReport.run.configuration}
+
+
+
Session
+ {evalReport.run.sessionId}
+
+
+
Provider session
+ {evalReport.run.providerSessionId ?? "unavailable"}
+
+
+
Driver
+
+ {evalReport.run.driver}
+ {evalReport.run.providerVersion
+ ? ` · ${evalReport.run.providerVersion}`
+ : ""}
+
+
+ {evalReport.run.agentVersion ? (
+
+
Agent version
+ {evalReport.run.agentVersion}
+
+ ) : null}
+
+
Retained session
+
+ {evalReport.run.retainedSession === true
+ ? (evalReport.run.retainedSessionStatus ?? "retained")
+ : "not applicable"}
+
+
+
+
Duration
+ {evalReport.run.durationMs} ms
+
+
+
Fixture
+ {evalReport.run.fixtureDigest}
+
+
+
State
+
+ r{evalReport.run.initialRevision} → r
+ {evalReport.run.finalRevision}
+
+
+
+
Tokens
+
+ {evalReport.run.usage === null
+ ? "unknown"
+ : `${evalReport.run.usage.inputTokens} in · ${evalReport.run.usage.outputTokens} out · ${evalReport.run.usage.cachedInputTokens} cached`}
+
+
+
+
Agent turns
+ {evalReport.run.usage?.agentTurns ?? "unknown"}
+
+
+
Provider requests
+ {evalReport.run.usage?.providerRequests ?? "unavailable"}
+
+
+
Estimated cost
+
+ {evalReport.run.usage === null
+ ? "unknown"
+ : `$${(evalReport.run.usage.estimatedCostNanodollars / 1_000_000_000).toFixed(6)} · ${evalReport.run.usage.pricingVersion}`}
+
+
+
+
Provider list cost
+
+ {typeof evalReport.run.usage
+ ?.providerReportedCostNanodollars !== "number"
+ ? "unknown"
+ : `$${(evalReport.run.usage.providerReportedCostNanodollars / 1_000_000_000).toFixed(6)}`}
+
+
+
+
Runner
+ {evalReport.run.runnerPackageDigest}
+
+
+
Runner build
+ {evalReport.run.runnerBuild}
+
+
+
runnerd
+ {evalReport.run.runnerdDigest}
+
+
+
Assertions
+
+
+ ) : null}
+ {tab === "timeline" ? (
+
+ {revisions.map((entry) => (
+ {
+ setFollowing(false);
+ setRevision(entry.revision);
+ }}
+ >
+ r{entry.revision}
+ {entry.operationId}
+ {entry.turnId ?? "session"}
+
+ ))}
+
+ ) : null}
+ {tab === "state" ? (
+
+
+ Revision
+
+ setRevision(Number(event.target.value))}
+ >
+ {snapshot.revisions.map((entry) => (
+
+ r{entry.revision} · {entry.operationId}
+
+ ))}
+
+
+
+ ) : null}
+ {tab === "diff" ? (
+
+
+
+ setCompareRevision(Number(event.target.value))
+ }
+ >
+ {snapshot.revisions.map((entry) => (
+
+ from r{entry.revision}
+
+ ))}
+
+ setRevision(Number(event.target.value))}
+ >
+ {snapshot.revisions.map((entry) => (
+
+ to r{entry.revision}
+
+ ))}
+
+
+ {rows.map((row) => (
+
+ {row.path}
+ {JSON.stringify(row.before)}
+ {JSON.stringify(row.after)}
+
+ ))}
+ {rows.length === 0 ? (
+
No changes between these revisions.
+ ) : null}
+
+ ) : null}
+ {tab === "documents" ? (
+ documents.length === 0 ? (
+
+
▤
+
No documents exist at r{selected.revision}.
+
+ ) : (
+
+
+ {documents.map((entry) => (
+ {
+ setDocumentId(entry.id);
+ setDocumentRevision(0);
+ }}
+ >
+ {entry.key}
+ {entry.title}
+
+ {entry.revisions.length} revision
+ {entry.revisions.length === 1 ? "" : "s"}
+
+
+ ))}
+
+ {document !== null && visibleDocumentRevision !== null ? (
+
+
+ {visibleDocumentRevision.changeSummary ? (
+
+ {visibleDocumentRevision.changeSummary}
+
+ ) : null}
+
+ {visibleDocumentRevision.body}
+
+
+ ) : null}
+
+ )
+ ) : null}
+ {tab === "protocol" ? (
+
+ {protocol.map((entry) => (
+
+
+ {entry.boundary} · {entry.event}
+
+
+
+ ))}
+
+ ) : null}
+ {tab === "trace" ? (
+ snapshot.providerTrace ? (
+
+
+ {snapshot.providerTrace.status} · expires{" "}
+ {snapshot.providerTrace.expiresAt}
+
+
Raw frame → Interpretation → PRP events → Presentation
+
+
+ ) : (
+
+
◇
+
+ Raw provider capture was off for this run. Canonical protocol
+ records remain available in Protocol.
+
+
+ )
+ ) : null}
+ {tab === "runtime" ? (
+
+
+
+ ) : null}
+ {tab === "authority" ? (
+
+
+
+ ) : null}
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/EvidencePanel.test.ts b/packages/paperclip-runner/devtools/issue-thread/src/EvidencePanel.test.ts
new file mode 100644
index 0000000000..67b0959f25
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/EvidencePanel.test.ts
@@ -0,0 +1,56 @@
+import { describe, expect, it } from "vitest";
+
+import type { CapabilityEvidenceRunnerRecord } from "../../../src/issue-thread/types.js";
+import { groupCapabilityRunnerEvents } from "./EvidencePanel.js";
+
+function record(
+ ordinal: number,
+ event: string,
+ turnId = "turn-1",
+): CapabilityEvidenceRunnerRecord {
+ return {
+ id: `event-${ordinal}`,
+ turnId,
+ kind: "provider_event",
+ ordinal,
+ detail: `provider event · ${event}`,
+ details: [{ label: "Event", value: event }],
+ };
+}
+
+describe("Runner event grouping", () => {
+ it("aggregates each delta category per turn at its first position", () => {
+ const session: CapabilityEvidenceRunnerRecord = {
+ id: "event-1",
+ turnId: "turn-0",
+ kind: "session",
+ ordinal: 1,
+ detail: "session · started",
+ details: [{ label: "Action", value: "started" }],
+ };
+ const groups = groupCapabilityRunnerEvents([
+ session,
+ record(2, "assistant_delta"),
+ record(3, "reasoning_delta"),
+ record(4, "assistant_delta"),
+ record(5, "assistant_delta", "turn-2"),
+ ]);
+
+ expect(groups.map((group) => [group.event, group.records.map((entry) => entry.ordinal)])).toEqual([
+ [null, [1]],
+ ["assistant_delta", [2, 4]],
+ ["reasoning_delta", [3]],
+ ["assistant_delta", [5]],
+ ]);
+ });
+
+ it("leaves non-delta events individually expandable", () => {
+ const groups = groupCapabilityRunnerEvents([
+ record(1, "turn_started"),
+ record(2, "tool_result"),
+ ]);
+
+ expect(groups).toHaveLength(2);
+ expect(groups.every((group) => group.event === null && group.records.length === 1)).toBe(true);
+ });
+});
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/EvidencePanel.tsx b/packages/paperclip-runner/devtools/issue-thread/src/EvidencePanel.tsx
new file mode 100644
index 0000000000..61aa2d7373
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/EvidencePanel.tsx
@@ -0,0 +1,622 @@
+import { useEffect, useMemo, useRef, useState, type ReactNode } from "react";
+
+import type {
+ CapabilityEvidenceModel,
+ CapabilityEvidenceRunnerRecord,
+ CapabilityEvidenceSectionId,
+ CapabilityEvidenceToolRow,
+ CapabilityIssueThreadSnapshot,
+ CapabilityJsonValue,
+ CapabilityToolDisposition,
+} from "../../../src/issue-thread/types";
+import type { CapabilityDevtoolsSnapshot } from "../../../src/devtools";
+import { DevtoolsInspector, type CapabilityDevtoolsTab, type EvalInspectorReport } from "./DevtoolsInspector";
+import { Icon } from "./Icons";
+import { capabilitySemanticToolDescriptor } from "../../../src/semantic-tools/catalog";
+import {
+ CAPABILITY_EVIDENCE_SECTIONS,
+ capabilityDispositionLabel,
+} from "../../../src/issue-thread/types";
+
+/**
+ * Evidence panel (contract §7): eight accordion sections in fixed order with a
+ * turn selector. Every record is rendered from the snapshot; the panel derives
+ * no verdicts of its own.
+ */
+
+const DISPOSITION_ORDER: CapabilityToolDisposition[] = [
+ "always_agent_tool",
+ "optional_agent_tool",
+ "control_plane_owned",
+];
+
+export interface CapabilityRunnerEventGroup {
+ id: string;
+ event: string | null;
+ records: CapabilityEvidenceRunnerRecord[];
+}
+
+/**
+ * Stream deltas are useful evidence but terrible default reading material.
+ * Aggregate each sanitized delta category within a turn at its first position;
+ * expanding the group still exposes every individual event in ordinal order.
+ */
+export function groupCapabilityRunnerEvents(
+ records: readonly CapabilityEvidenceRunnerRecord[],
+): CapabilityRunnerEventGroup[] {
+ const groups: CapabilityRunnerEventGroup[] = [];
+ const deltaGroups = new Map();
+ for (const record of records) {
+ const event = record.details.find((entry) => entry.label === "Event")?.value ?? null;
+ if (event === null || !event.endsWith("_delta")) {
+ groups.push({ id: record.id, event: null, records: [record] });
+ continue;
+ }
+ const key = `${record.turnId}:${event}`;
+ const existing = deltaGroups.get(key);
+ if (existing !== undefined) {
+ existing.records.push(record);
+ continue;
+ }
+ const group = { id: `delta:${key}`, event, records: [record] };
+ deltaGroups.set(key, group);
+ groups.push(group);
+ }
+ return groups;
+}
+
+function RunnerEvent({ record }: { record: CapabilityEvidenceRunnerRecord }) {
+ return (
+
+
+ #{record.ordinal} {record.kind}
+ {record.detail}
+ ›
+
+
+ {record.details.length === 0 ? (
+ <>
+ Detail
+ No additional sanitized fields were retained.
+ >
+ ) : (
+ record.details.map((entry, index) => (
+
+
{entry.label}
+ {entry.value}
+
+ ))
+ )}
+
+
+ );
+}
+
+function RunnerEventGroup({ group }: { group: CapabilityRunnerEventGroup }) {
+ if (group.event === null) return ;
+ return (
+
+
+ ›
+ {group.event}
+
+ {group.records.length} streamed update{group.records.length === 1 ? "" : "s"}
+
+
+
+ {group.records.map((record) => )}
+
+
+ );
+}
+
+export interface EvidencePanelProps {
+ snapshot: CapabilityIssueThreadSnapshot;
+ devtools?: CapabilityDevtoolsSnapshot | null;
+ evalReport?: EvalInspectorReport | null;
+ onForkRevision?: (revision: number) => void;
+ layout: "side" | "overlay" | "segment";
+ width: number;
+ selectedTurnId: string | "all";
+ openSections: CapabilityEvidenceSectionId[];
+ highlightedRecordId: string | null;
+ onSelectTurn: (turnId: string | "all") => void;
+ onToggleSection: (section: CapabilityEvidenceSectionId) => void;
+ onClose: () => void;
+ onJumpToThread: (anchorId: string) => void;
+ onInvokeTool?: (operationId: string, input: CapabilityJsonValue) => Promise;
+}
+
+function sectionCount(
+ evidence: CapabilityEvidenceModel,
+ section: CapabilityEvidenceSectionId,
+ turnId: string | "all",
+): string {
+ if (section === "tools") {
+ const records = filterByTurn(evidence.tools, turnId);
+ return String(new Set(records.flatMap((record) => record.rows.map((row) => row.operationId))).size);
+ }
+ if (section === "parity") {
+ const rows = filterByTurn(evidence.parity, turnId);
+ const passing = rows.filter((row) => row.verdict === "pass").length;
+ return `${passing}/${rows.length}`;
+ }
+ const rows = filterByTurn(
+ evidence[section] as ReadonlyArray<{ turnId: string }>,
+ turnId,
+ );
+ return String(rows.length);
+}
+
+function starterInput(schema: { properties?: Readonly>; required?: readonly string[] } | undefined): Record {
+ const result: Record = {};
+ for (const name of schema?.required ?? []) {
+ const property = schema?.properties?.[name];
+ if (property?.default !== undefined) result[name] = property.default;
+ else if (property?.enum?.[0] !== undefined) result[name] = property.enum[0];
+ else if (name === "idempotencyKey") result[name] = `devtools-${Date.now()}`;
+ else if (property?.type === "integer" || property?.type === "number") result[name] = 0;
+ else if (property?.type === "boolean") result[name] = false;
+ else if (property?.type === "array") result[name] = [];
+ else if (property?.type === "object") result[name] = {};
+ else result[name] = "";
+ }
+ return result;
+}
+
+function ToolsBrowser({ records, onInvoke }: { records: CapabilityEvidenceModel["tools"]; onInvoke?: (operationId: string, input: CapabilityJsonValue) => Promise }) {
+ const [collapsed, setCollapsed] = useState([]);
+ const [selected, setSelected] = useState(null);
+ const [inputText, setInputText] = useState("{}");
+ const [invocationResult, setInvocationResult] = useState(null);
+ const [invocationError, setInvocationError] = useState(null);
+ const [invoking, setInvoking] = useState(false);
+ const rows = useMemo(() => {
+ const unique = new Map();
+ for (const record of records) {
+ for (const row of record.rows) unique.set(row.operationId, row);
+ }
+ return [...unique.values()];
+ }, [records]);
+ const descriptor = selected === null ? undefined : capabilitySemanticToolDescriptor(selected.operationId);
+
+ useEffect(() => {
+ if (selected === null) return;
+ setInputText(JSON.stringify(starterInput(descriptor?.inputSchema), null, 2));
+ setInvocationResult(null);
+ setInvocationError(null);
+ }, [descriptor, selected]);
+
+ useEffect(() => {
+ if (selected === null) return;
+ const close = (event: KeyboardEvent) => { if (event.key === "Escape") setSelected(null); };
+ document.addEventListener("keydown", close);
+ return () => document.removeEventListener("keydown", close);
+ }, [selected]);
+
+ return (
+ <>
+
+ {DISPOSITION_ORDER.map((disposition) => {
+ const group = rows.filter((row) => row.disposition === disposition);
+ if (group.length === 0) return null;
+ const folded = collapsed.includes(disposition);
+ const title = disposition === "control_plane_owned"
+ ? "Control plane (not exposed to the agent)"
+ : capabilityDispositionLabel(disposition);
+ return (
+
+
+ setCollapsed((current) => current.includes(disposition) ? current.filter((entry) => entry !== disposition) : [...current, disposition])}>
+ {folded ? "›" : "⌄"}
+ {title}
+ {group.length}
+
+
+ {!folded ? (
+
+ {group.map((row) => (
+ setSelected(row)}>
+ {row.operationId}
+ {row.description}
+ {row.grant !== null && disposition === "optional_agent_tool" ? {row.grant} : null}
+ ›
+
+ ))}
+
+ ) : null}
+
+ );
+ })}
+
+
+ {selected !== null ? (
+ { if (event.target === event.currentTarget) setSelected(null); }}>
+
+
+
+
+
{selected.operationId}
+
{descriptor?.description ?? selected.description}
+
+
Placement {capabilityDispositionLabel(selected.disposition)}
+
Required claims {descriptor?.requiredClaims.join(", ") || selected.grant || "None"}
+
Task modes {descriptor?.allowedModes.join(", ") || "—"}
+
Actor roles {descriptor?.allowedRoles?.join(", ") || "Any allowed role"}
+
+
+
Input schema
+
{JSON.stringify(descriptor?.inputSchema ?? {}, null, 2)}
+
+
+ {onInvoke !== undefined ? (
+
+
Invoke in conversation
+
Runs through the real semantic dispatcher and records a turn, evidence, and company-state changes.
+
Parameters (JSON)
+
setInputText(event.target.value)} spellCheck={false} />
+ {
+ let parsed: CapabilityJsonValue;
+ try {
+ parsed = JSON.parse(inputText) as CapabilityJsonValue;
+ } catch (error) {
+ setInvocationError(error instanceof Error ? error.message : String(error));
+ return;
+ }
+ setInvoking(true);
+ setInvocationError(null);
+ setInvocationResult(null);
+ void onInvoke(selected.operationId, parsed)
+ .then(setInvocationResult)
+ .catch((error) => setInvocationError(error instanceof Error ? error.message : String(error)))
+ .finally(() => setInvoking(false));
+ }}>{invoking ? "Invoking…" : "Invoke as tool call"}
+ {invocationError !== null ? {invocationError}
: null}
+ {invocationResult !== null ? {JSON.stringify(invocationResult, null, 2)} : null}
+
+ ) : null}
+
+
+
+ ) : null}
+ >
+ );
+}
+
+function filterByTurn(
+ rows: ReadonlyArray,
+ turnId: string | "all",
+): T[] {
+ return turnId === "all" ? [...rows] : rows.filter((row) => row.turnId === turnId);
+}
+
+function Section({
+ id,
+ title,
+ count,
+ open,
+ onToggle,
+ children,
+}: {
+ id: CapabilityEvidenceSectionId;
+ title: string;
+ count: string;
+ open: boolean;
+ onToggle: () => void;
+ children: ReactNode;
+}) {
+ return (
+
+
+
+ {open ? "⌄" : "›"}
+ {title}
+ {count}
+
+
+
+ {children}
+
+
+ );
+}
+
+export function EvidencePanel(props: EvidencePanelProps) {
+ const [devtoolsTab, setDevtoolsTab] = useState(props.evalReport ? "eval" : "evidence");
+ const {
+ snapshot,
+ devtools,
+ evalReport,
+ onForkRevision = () => undefined,
+ layout,
+ width,
+ selectedTurnId,
+ openSections,
+ highlightedRecordId,
+ onSelectTurn,
+ onToggleSection,
+ onClose,
+ onJumpToThread,
+ onInvokeTool,
+ } = props;
+ const headingRef = useRef(null);
+ const evidence = snapshot.evidence;
+
+ useEffect(() => {
+ // Contract §9.2: opening Evidence moves focus to its heading.
+ headingRef.current?.focus();
+ }, []);
+
+ useEffect(() => {
+ // Contract §9.1: Escape closes the desktop overlay sheet. The side panel is
+ // a persistent region rather than a sheet, and the mobile segment is a tab
+ // view, so neither of those is dismissible this way.
+ if (layout !== "overlay") return;
+ function onKeyDown(event: KeyboardEvent) {
+ if (event.key !== "Escape") return;
+ // A modal dialog owns Escape while it is open (§6 reset confirm).
+ if (document.querySelector('[role="dialog"][aria-modal="true"]') !== null) return;
+ onClose();
+ }
+ document.addEventListener("keydown", onKeyDown);
+ return () => document.removeEventListener("keydown", onKeyDown);
+ }, [layout, onClose]);
+
+ const isOpen = (section: CapabilityEvidenceSectionId) => openSections.includes(section);
+
+ return (
+
+
+
Developer tools
+ Paperclip DevTools
+
+
+
+
+
+ {devtools !== undefined ? (
+ <>
+ {devtools === null ? (
+ Loading company state…
+ ) : (
+
+ )}
+ >
+ ) : null}
+
+ {devtoolsTab === "evidence" ?
+
+ Evidence scope
+ onSelectTurn(event.target.value as string | "all")}>
+ {snapshot.turns.map((turn) => Turn {turn.ordinal} )}
+ All turns
+
+
+
+
+
+
onToggleSection("calls")}
+ >
+ {filterByTurn(evidence.calls, selectedTurnId).map((record) => (
+
+
+ {record.operationId} · v{record.version} · {record.outcome}
+
+ {record.providerRequest}
+ {record.dispatchedCommand}
+ {JSON.stringify(record.result)}
+ {record.redactions.length > 0 ? (
+
+ ••• redacted by {record.redactions.join(", ")}
+
+ ) : null}
+ onJumpToThread(record.threadAnchorId)}
+ >
+ Show in thread
+
+
+ ))}
+
+
+
onToggleSection("authorization")}
+ >
+ {filterByTurn(evidence.authorization, selectedTurnId).map((record) => (
+
+
+ {record.allowed ? "✓" : "✕"} {" "}
+ {record.operationId} · {record.phase} · {record.allowed ? "allowed" : "denied"}
+
+ {record.reason}
+
+ claims: {record.claimsConsidered.join(", ") || "none"}
+
+ {record.redactions.length > 0 ? (
+
+ ••• redacted by {record.redactions.join(", ")}
+
+ ) : null}
+ {record.stateChangeRef !== null ? (
+ state: {record.stateChangeRef}
+ ) : null}
+ {record.threadAnchorId.length > 0 ? (
+ onJumpToThread(record.threadAnchorId)}
+ >
+ Show in thread
+
+ ) : null}
+
+ ))}
+
+
+
onToggleSection("control_plane")}
+ >
+ {filterByTurn(evidence.control_plane, selectedTurnId).map((record) => (
+
+
+ {record.category} · {record.outcome} · rev {record.stateRevision}
+
+ {record.reason}
+
+ ))}
+
+
+
onToggleSection("runner")}
+ >
+ {groupCapabilityRunnerEvents(filterByTurn(evidence.runner, selectedTurnId)).map((group) => (
+
+ ))}
+
+
+
onToggleSection("state")}
+ >
+ {filterByTurn(evidence.state, selectedTurnId).map((record) => (
+
+
+ revision {record.fromRevision} → {record.toRevision}
+
+ {record.rows.map((row) => (
+
+
+ {row.entityClass} · {row.entityRef}
+
+
+ {row.before} → {row.after}
+
+
+ ))}
+
+ ))}
+
+
+
onToggleSection("traceability")}
+ >
+ {filterByTurn(evidence.traceability, selectedTurnId).map((record) => (
+
+
+ {record.caseId} · {record.group}
+
+ {record.sourceAnchor}
+ {record.browserEvidenceRecipe}
+
+ expects: {record.expectedSemanticOperations.join(", ")}
+
+
+ forbids: {record.forbiddenOperations.join(", ") || "none"}
+
+
+ grants: {record.requiredCapabilityGrants.join(", ") || "none"}
+
+
+ ))}
+
+
+
onToggleSection("parity")}
+ >
+ {filterByTurn(evidence.parity, selectedTurnId).map((record) => (
+
+
+
+ {record.verdict === "pass" ? "✓" : record.verdict === "fail" ? "✕" : "—"}
+
+ {record.verdict === "intentional_gap" ? "intentional gap" : record.verdict}
+
+ {record.assertion}
+ {record.note !== null ? (
+ {record.note}
+ ) : null}
+
+ ))}
+
+
: null}
+
+ );
+}
+
+export const CAPABILITY_EVIDENCE_SECTION_IDS = CAPABILITY_EVIDENCE_SECTIONS.map(
+ (section) => section.id,
+);
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/Icons.tsx b/packages/paperclip-runner/devtools/issue-thread/src/Icons.tsx
new file mode 100644
index 0000000000..2ce971dc73
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/Icons.tsx
@@ -0,0 +1,32 @@
+export type IconName =
+ | "activity" | "authority" | "branch" | "close" | "diff" | "documents"
+ | "download" | "evidence" | "pause" | "play" | "protocol" | "reset"
+ | "spark" | "state" | "stop" | "terminal" | "timeline";
+
+const PATHS: Record = {
+ activity: "M3 12h3l2-6 4 12 3-9 2 5h4",
+ authority: "M12 3l7 4v5c0 4.6-2.8 7.6-7 9-4.2-1.4-7-4.4-7-9V7zM9 12l2 2 4-4",
+ branch: "M6 3v12a3 3 0 0 0 3 3h6M15 18l-3-3m3 3-3 3M6 7h6a3 3 0 0 0 3-3V3",
+ close: "M5 5l14 14M19 5 5 19",
+ diff: "M8 5v14M4 9h8M17 7v10M14 12h6",
+ documents: "M6 3h9l3 3v15H6zM15 3v4h4M9 11h6M9 15h6",
+ download: "M12 3v12m-5-5 5 5 5-5M5 21h14",
+ evidence: "M4 5h16v14H4zM8 9h8M8 13h5",
+ pause: "M8 5v14M16 5v14",
+ play: "M8 5v14l11-7z",
+ protocol: "M4 8h14m-3-3 3 3-3 3M20 16H6m3-3-3 3 3 3",
+ reset: "M5 8a8 8 0 1 1-1 7M5 8V3m0 5h5",
+ spark: "M12 2l1.7 5.3L19 9l-5.3 1.7L12 16l-1.7-5.3L5 9l5.3-1.7zM19 16l.8 2.2L22 19l-2.2.8L19 22l-.8-2.2L16 19l2.2-.8z",
+ state: "M9 4H6a2 2 0 0 0-2 2v3M15 4h3a2 2 0 0 1 2 2v3M9 20H6a2 2 0 0 1-2-2v-3M15 20h3a2 2 0 0 0 2-2v-3M9 9h6v6H9z",
+ stop: "M7 7h10v10H7z",
+ terminal: "M5 7l4 5-4 5m7 0h7",
+ timeline: "M5 5v14M5 8h5M5 13h9M5 18h13",
+};
+
+export function Icon({ name }: { name: IconName }) {
+ return (
+
+
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/InteractionCard.tsx b/packages/paperclip-runner/devtools/issue-thread/src/InteractionCard.tsx
new file mode 100644
index 0000000000..bd464a1f6c
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/InteractionCard.tsx
@@ -0,0 +1,528 @@
+import { useEffect, useId, useRef, useState } from "react";
+
+import type {
+ CapabilityThreadInteractionCard,
+ CapabilityThreadInteractionState,
+} from "../../../src/issue-thread/types";
+import { Chip } from "./primitives";
+
+/**
+ * Native interaction cards (contract §5). All five `request_human_input`
+ * kinds render typed controls; the card never asks the user to type an answer
+ * into the composer.
+ *
+ * The submit handler posts to the package server, which stores the response in
+ * the mock control plane before the runner is resumed. The card only leaves
+ * `submitting` when the server acknowledges — there is no optimistic
+ * resolution here.
+ */
+
+export interface CapabilityInteractionResponse {
+ interactionId: string;
+ outcome: "answered" | "accepted" | "rejected";
+ result: Record;
+}
+
+const STATE_GLYPH: Record = {
+ pending: "⏳",
+ submitting: "⏳",
+ accepted: "✓",
+ answered: "✓",
+ rejected: "✕",
+ stale_target: "⌛",
+ superseded_by_comment: "⌛",
+ expired: "⌛",
+ withdrawn: "⌛",
+ issue_closed: "⌛",
+};
+
+const RESOLVED_STATES = new Set([
+ "accepted",
+ "answered",
+ "rejected",
+ "stale_target",
+ "superseded_by_comment",
+ "expired",
+ "withdrawn",
+ "issue_closed",
+]);
+
+function stateTone(state: CapabilityThreadInteractionState) {
+ if (state === "accepted" || state === "answered") return "success" as const;
+ if (state === "rejected") return "danger" as const;
+ if (state === "pending" || state === "submitting") return "accent" as const;
+ return undefined;
+}
+
+export function InteractionCard({
+ card,
+ onRespond,
+ onOpenEvidence,
+ autoFocus,
+}: {
+ card: CapabilityThreadInteractionCard;
+ onRespond?: (response: CapabilityInteractionResponse) => void;
+ onOpenEvidence?: (section: string, recordId: string) => void;
+ autoFocus?: boolean;
+}) {
+ const titleId = useId();
+ const [answers, setAnswers] = useState>({});
+ const [selected, setSelected] = useState(() =>
+ card.payload.kind === "checkbox"
+ ? card.payload.options.filter((option) => option.defaultSelected).map((option) => option.id)
+ : card.payload.kind === "suggest_tasks"
+ ? card.payload.tasks.map((task) => task.id)
+ : [],
+ );
+ const [verdicts, setVerdicts] = useState>({});
+ const [reasons, setReasons] = useState>({});
+ const [rejectReason, setRejectReason] = useState("");
+ const [error, setError] = useState(null);
+ const firstControl = useRef(null);
+ const stateChip = useRef(null);
+ const previousState = useRef(card.state);
+ const resolved = RESOLVED_STATES.has(card.state);
+ const busy = card.state === "submitting";
+
+ useEffect(() => {
+ // Contract §9.2: resolving a card moves focus to its state chip. The
+ // controls the user just used unmount on resolve, so without this the
+ // keyboard caret would fall back to `body` exactly when the card changes.
+ const from = previousState.current;
+ previousState.current = card.state;
+ if (from === card.state) return;
+ if (!RESOLVED_STATES.has(from) && RESOLVED_STATES.has(card.state)) {
+ stateChip.current?.focus();
+ }
+ }, [card.state]);
+
+ function respond(response: CapabilityInteractionResponse) {
+ setError(null);
+ onRespond?.(response);
+ }
+
+ function toggle(list: string[], id: string): string[] {
+ return list.includes(id) ? list.filter((entry) => entry !== id) : [...list, id];
+ }
+
+ return (
+
+
+
+ {card.title}
+
+
+ {STATE_GLYPH[card.state]}
+ {card.stateLabel}
+
+
+
+ {card.prompt}
+
+ {card.target !== null ? (
+
+ Target:{" "}
+
+ {card.target.label}
+
+
+ ) : null}
+
+ {card.payload.kind === "confirmation" ? (
+ {card.payload.targetSummary}
+ ) : null}
+
+ {resolved ? (
+
+ {card.resolvedSummary.length > 0 ? (
+
+ {card.resolvedSummary.map((entry) => (
+ {entry}
+ ))}
+
+ ) : null}
+ {card.reason !== null ?
{card.reason} : null}
+ {card.supersededBy !== null ? (
+
+ {card.supersededBy.label}
+
+ ) : null}
+
+ ) : (
+
+ {card.payload.kind === "questions"
+ ? card.payload.questions.map((question, index) => (
+
+
+ {question.prompt}
+
+ {question.control === "text" ? (
+
+ setAnswers((current) => ({ ...current, [question.id]: event.target.value }))
+ }
+ />
+ ) : question.control === "select" ? (
+
+ setAnswers((current) => ({ ...current, [question.id]: event.target.value }))
+ }
+ >
+ Choose…
+ {(question.options ?? []).map((option) => (
+
+ {option}
+
+ ))}
+
+ ) : (
+
+ {(question.options ?? []).map((option, optionIndex) => (
+
+ {
+ firstControl.current = node;
+ if (autoFocus === true && node !== null) node.focus();
+ }
+ : undefined
+ }
+ checked={answers[question.id] === option}
+ onChange={() =>
+ setAnswers((current) => ({ ...current, [question.id]: option }))
+ }
+ />
+ {option}
+
+ ))}
+
+ )}
+
+ ))
+ : null}
+
+ {card.payload.kind === "checkbox"
+ ? card.payload.options.map((option, index) => (
+
+ {
+ if (autoFocus === true && node !== null) node.focus();
+ }
+ : undefined
+ }
+ onChange={() => setSelected((current) => toggle(current, option.id))}
+ />
+ {option.label}
+
+ ))
+ : null}
+
+ {card.payload.kind === "suggest_tasks"
+ ? card.payload.tasks.map((task) => (
+
+ setSelected((current) => toggle(current, task.id))}
+ />
+
+ {task.title}
+
+ {task.description}
+
+
+ ))
+ : null}
+
+ {card.payload.kind === "item_verdicts"
+ ? card.payload.items.map((item) => (
+
+
+ {item.title}
+
+
+ {(["approve", "reject", "defer"] as const).map((verdict) => (
+
+ setVerdicts((current) => ({ ...current, [item.id]: verdict }))
+ }
+ >
+ {verdict}
+
+ ))}
+
+ {item.requireReason === true ? (
+
+ setReasons((current) => ({ ...current, [item.id]: event.target.value }))
+ }
+ />
+ ) : null}
+
+ ))
+ : null}
+
+ {card.payload.kind === "confirmation" && card.payload.requireRejectReason ? (
+
+
+ Reason (required to request changes)
+
+ setRejectReason(event.target.value)}
+ />
+
+ ) : null}
+
+ {error !== null ? (
+
+ {error}
+
+ ) : null}
+
+
+ {card.payload.kind === "confirmation" ? (
+ <>
+ {
+ if (autoFocus === true && node !== null && firstControl.current === null) {
+ firstControl.current = node;
+ node.focus();
+ }
+ }}
+ onClick={() =>
+ respond({
+ interactionId: card.interactionId,
+ outcome: "accepted",
+ result: { accepted: true },
+ })
+ }
+ >
+ {card.payload.acceptLabel}
+
+ {
+ if (
+ card.payload.kind === "confirmation" &&
+ card.payload.requireRejectReason &&
+ rejectReason.trim().length === 0
+ ) {
+ setError("A reason is required to request changes.");
+ return;
+ }
+ respond({
+ interactionId: card.interactionId,
+ outcome: "rejected",
+ result: { accepted: false, reason: rejectReason },
+ });
+ }}
+ >
+ {card.payload.rejectLabel}
+
+ >
+ ) : null}
+
+ {card.payload.kind === "questions" ? (
+ {
+ const missing =
+ card.payload.kind === "questions"
+ ? card.payload.questions.filter(
+ (question) =>
+ question.required === true &&
+ (answers[question.id] ?? "").trim().length === 0,
+ )
+ : [];
+ if (missing.length > 0) {
+ setError(`Answer every required question (${missing.length} remaining).`);
+ return;
+ }
+ respond({
+ interactionId: card.interactionId,
+ outcome: "answered",
+ result: { answers },
+ });
+ }}
+ >
+ {card.payload.submitLabel}
+
+ ) : null}
+
+ {card.payload.kind === "checkbox" ? (
+ <>
+ {
+ if (card.payload.kind !== "checkbox") return;
+ if (selected.length < card.payload.minSelected) {
+ setError(`Select at least ${card.payload.minSelected}.`);
+ return;
+ }
+ if (selected.length > card.payload.maxSelected) {
+ setError(`Select at most ${card.payload.maxSelected}.`);
+ return;
+ }
+ respond({
+ interactionId: card.interactionId,
+ outcome: "accepted",
+ result: { selected },
+ });
+ }}
+ >
+ {card.payload.acceptLabel}
+
+
+ respond({
+ interactionId: card.interactionId,
+ outcome: "rejected",
+ result: { selected: [] },
+ })
+ }
+ >
+ {card.payload.rejectLabel}
+
+ >
+ ) : null}
+
+ {card.payload.kind === "suggest_tasks" ? (
+
+ respond({
+ interactionId: card.interactionId,
+ outcome: "accepted",
+ result: { acceptedTaskIds: selected },
+ })
+ }
+ >
+ {card.payload.acceptLabel}
+
+ ) : null}
+
+ {card.payload.kind === "item_verdicts" ? (
+ {
+ if (card.payload.kind !== "item_verdicts") return;
+ const submitted = card.payload.items.filter(
+ (item) => item.lockedVerdict == null && verdicts[item.id] !== undefined,
+ );
+ const missingReason = submitted.filter(
+ (item) =>
+ item.requireReason === true && (reasons[item.id] ?? "").trim().length === 0,
+ );
+ if (missingReason.length > 0) {
+ setError(`A reason is required for ${missingReason.length} item(s).`);
+ return;
+ }
+ if (submitted.length === 0) {
+ setError("Choose a verdict for at least one item.");
+ return;
+ }
+ respond({
+ interactionId: card.interactionId,
+ outcome: "answered",
+ result: {
+ verdicts: submitted.map((item) => ({
+ id: item.id,
+ verdict: verdicts[item.id],
+ reason: reasons[item.id] ?? null,
+ })),
+ },
+ });
+ }}
+ >
+ {card.payload.submitLabel}
+
+ ) : null}
+
+
+ )}
+
+ onOpenEvidence?.(card.evidenceRef.section, card.evidenceRef.recordId)}
+ >
+ View request evidence
+
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/IssueHeader.tsx b/packages/paperclip-runner/devtools/issue-thread/src/IssueHeader.tsx
new file mode 100644
index 0000000000..6013871557
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/IssueHeader.tsx
@@ -0,0 +1,213 @@
+import { useEffect, useRef, useState } from "react";
+
+import type { CapabilityIssueThreadSnapshot } from "../../../src/issue-thread/types";
+import { Icon } from "./Icons";
+import { Chip, PriorityIcon, StatusBadge } from "./primitives";
+
+/**
+ * Sticky issue header (contract §1 and §2). The three identity chips are never
+ * hidden by scroll, mode is data rather than styling, and on mobile Stop stays
+ * outside the overflow menu whenever a turn is active.
+ */
+
+export interface IssueHeaderProps {
+ snapshot: CapabilityIssueThreadSnapshot;
+ scenarios: string[];
+ /** `chat` is the clean room: no preset scenario, no recording to replay. */
+ surface?: "issue" | "chat";
+ /** Immutable eval recordings use the same inspector without live controls. */
+ readOnly?: boolean;
+ /** Clean-room tenant token, rendered so identity rotation is visible. */
+ cleanRoomToken?: string | null;
+ evidenceOpen: boolean;
+ denialCount: number;
+ segment: "thread" | "evidence";
+ onToggleEvidence: () => void;
+ onSelectScenario: (scenario: string) => void;
+ onReplay: () => void;
+ onReset: () => void;
+ onStop: () => void;
+ onSelectSegment: (segment: "thread" | "evidence") => void;
+}
+
+export function IssueHeader(props: IssueHeaderProps) {
+ const {
+ snapshot,
+ scenarios,
+ surface = "issue",
+ readOnly = false,
+ cleanRoomToken = null,
+ evidenceOpen,
+ denialCount,
+ segment,
+ onToggleEvidence,
+ onSelectScenario,
+ onReplay,
+ onReset,
+ onStop,
+ onSelectSegment,
+ } = props;
+ const chat = surface === "chat";
+ const [menuOpen, setMenuOpen] = useState(false);
+ const menuRef = useRef(null);
+ const turnActive = snapshot.composer.state === "streaming" || snapshot.composer.state === "sending";
+
+ useEffect(() => {
+ if (!menuOpen) return;
+ function onKeyDown(event: KeyboardEvent) {
+ if (event.key === "Escape") setMenuOpen(false);
+ }
+ document.addEventListener("keydown", onKeyDown);
+ return () => document.removeEventListener("keydown", onKeyDown);
+ }, [menuOpen]);
+
+ return (
+
+
+
+ {snapshot.issue.identifier}
+
{snapshot.issue.title}
+
+
+
+ {!chat && !readOnly ? (
+ <>
+ Scenario
+ onSelectScenario(event.target.value)} data-testid="scenario-picker">
+ {scenarios.map((scenario) => {scenario} )}
+
+ Replay
+ >
+ ) : null}
+ {!readOnly ? : null}
+ {!readOnly ? : null}
+ DevTools{denialCount > 0 ? ` (${denialCount})` : ""}
+
+
+
+
+
+
+ {snapshot.identity.agentLabel}
+ {snapshot.identity.replaySource !== null
+ ? ` · ${snapshot.identity.replaySource} source`
+ : ""}
+
+
+
+ {snapshot.identity.runnerLabel}
+
+ {snapshot.identity.runnerAttached ? "attached" : "detached"}
+
+
+
+ {snapshot.identity.controlPlaneLabel}
+
+ {chat && cleanRoomToken !== null ? (
+
+ Clean room {cleanRoomToken}
+
+ ) : null}
+
+
+ {snapshot.issue.assignee !== null ?
{snapshot.issue.assignee} : null}
+
{snapshot.issue.runState}
+
+
+
setMenuOpen((current) => !current)}
+ data-testid="overflow-menu-button"
+ >
+ ⋯
+ More actions
+
+ {menuOpen ? (
+
+ {chat ? null : (
+ <>
+
+ Scenario
+ {
+ setMenuOpen(false);
+ onSelectScenario(event.target.value);
+ }}
+ >
+ {scenarios.map((scenario) => (
+
+ {scenario}
+
+ ))}
+
+
+ {
+ setMenuOpen(false);
+ onReplay();
+ }}
+ >
+ Replay
+
+ >
+ )}
+ {
+ setMenuOpen(false);
+ onReset();
+ }}
+ >
+ {chat ? "Reset chat" : "Reset scenario"}
+
+
+ ) : null}
+
+
+
+
+
+ onSelectSegment("thread")}
+ data-testid="segment-thread"
+ >
+ Thread
+
+ onSelectSegment("evidence")}
+ data-testid="segment-evidence"
+ >
+ Evidence
+ {denialCount > 0 ? (
+
+ {denialCount}
+
+ ) : null}
+
+
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/SurfaceNav.tsx b/packages/paperclip-runner/devtools/issue-thread/src/SurfaceNav.tsx
new file mode 100644
index 0000000000..9c0cf47c67
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/SurfaceNav.tsx
@@ -0,0 +1,57 @@
+import type { CapabilitySurface } from "./route";
+import { Icon } from "./Icons";
+
+/**
+ * Top-level surface switch (Capability plan revision 5).
+ *
+ * Capability now has two primary paths, not one path with a mode toggle: the
+ * preset scenario explorer and a clean-room chat that starts from nothing. They
+ * are different jobs, so they get links rather than tabs — the clean-room entry
+ * has to be reachable from the landing page without first choosing a scenario.
+ *
+ * These are real anchors on purpose. The entry must survive a hard refresh, be
+ * copyable, and open in a new tab, which a button handler would not give.
+ */
+export interface CapabilityChatHistoryItem {
+ sessionId: string;
+ identifier: string;
+ title: string;
+ updatedAt: string;
+ current: boolean;
+}
+
+export function SurfaceNav({ surface, history = [], activeSessionId = null, onNewChat, onSelectHistory }: {
+ surface: CapabilitySurface;
+ history?: CapabilityChatHistoryItem[];
+ activeSessionId?: string | null;
+ onNewChat?: () => void;
+ onSelectHistory?: (sessionId: string) => void;
+}) {
+ return (
+
+ Runner lab
+
+
+ Scenario explorer
+
+
+ New chat
+
+
+ {surface === "chat" ? (
+
+ Session history
+
+ {history.map((item) => (
+
onSelectHistory?.(item.sessionId)}>
+ {item.title}
+ {item.identifier}{item.current ? " · current" : ""}
+
+ ))}
+ {history.length === 0 ?
No chats yet.
: null}
+
+
+ ) : null}
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/ThreadItems.tsx b/packages/paperclip-runner/devtools/issue-thread/src/ThreadItems.tsx
new file mode 100644
index 0000000000..a5efd9b300
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/ThreadItems.tsx
@@ -0,0 +1,616 @@
+import { useState } from "react";
+import ReactMarkdown from "react-markdown";
+import remarkGfm from "remark-gfm";
+
+import type {
+ CapabilityEvidenceSectionId,
+ CapabilityThreadItem,
+ CapabilityThreadTurn,
+} from "../../../src/issue-thread/types";
+import { InteractionCard, type CapabilityInteractionResponse } from "./InteractionCard";
+import { Chip, StatusBadge, Timestamp, formatBytes } from "./primitives";
+
+export interface EvalAssertion {
+ id: string;
+ title: string;
+ description: string;
+ passed: boolean;
+ detail: string;
+ definition: Record;
+}
+
+export function EvalAssertions({ assertions }: { assertions: EvalAssertion[] }) {
+ const [selected, setSelected] = useState(null);
+ if (assertions.length === 0) return null;
+ return (
+ <>
+
+ {assertions.map((assertion) => (
+ setSelected(assertion)}>
+ {assertion.passed ? "✓ PASS" : "✕ FAIL"} · {assertion.title}
+ {assertion.detail}
+
+ ))}
+
+ {selected !== null ? (
+ { if (event.target === event.currentTarget) setSelected(null); }}>
+
+
+
{selected.description}
+
Observed result
{selected.detail}
+
Original case definition
{JSON.stringify(selected.definition, null, 2)}
+
+
+ ) : null}
+ >
+ );
+}
+
+/** Progressive disclosure budget from contract §3 (T4). */
+const VISIBLE_STRIPS = 3;
+
+const STATUS_GLYPH = { ok: "✓", denied: "✕", running: "⏳" } as const;
+
+function MarkdownBody({ children }: { children: string }) {
+ return (
+
+ {children}
+
+ );
+}
+
+export interface ThreadCallbacks {
+ onOpenEvidence: (section: CapabilityEvidenceSectionId, recordId: string) => void;
+ onRespond: (response: CapabilityInteractionResponse) => void;
+ focusInteractionId: string | null;
+}
+
+function ToolStrip({
+ item,
+ callbacks,
+}: {
+ item: Extract;
+ callbacks: ThreadCallbacks;
+}) {
+ return (
+
+
+
+ {STATUS_GLYPH[item.status]}
+
+ {item.status}
+ {item.operationId}
+ {item.summary}
+ ›
+
+
+
Sanitized tool input and result
+
+
{JSON.stringify(item.input, null, 2)}
+
{JSON.stringify(item.result, null, 2)}
+
+
+ callbacks.onOpenEvidence(item.evidenceRef.section, item.evidenceRef.recordId)
+ }
+ >
+ View in Evidence
+
+
+
+ );
+}
+
+function ProgressItem({
+ item,
+}: {
+ item: Extract;
+}) {
+ return (
+
+
+
+ {item.status === "running" ? "⏳" : "✓"}
+
+ {item.label}
+ {item.summary}
+ ›
+
+
+ {Object.entries(item.details).map(([name, value]) =>
+ name === "withheld" ? null : (
+
+
{name}
+
{typeof value === "string" ? value : JSON.stringify(value, null, 2)}
+
+ ),
+ )}
+ {Array.isArray(item.details.withheld) ? (
+
+ Not available here: {item.details.withheld.join(", ")}.
+
+ ) : null}
+
+
+ );
+}
+
+function valueRecord(value: unknown): Record {
+ return typeof value === "object" && value !== null && !Array.isArray(value) ? value as Record : {};
+}
+
+function ProviderActivity({ item, callbacks }: {
+ item: Extract;
+ callbacks: ThreadCallbacks;
+}) {
+ const payload = valueRecord(item.payload);
+ const steps = Array.isArray(payload.steps) ? payload.steps.map(valueRecord) : [];
+ const children = Array.isArray(payload.children) ? payload.children.map(valueRecord) : [];
+ const sources = Array.isArray(payload.sources) ? payload.sources.map(valueRecord) : [];
+ const output = typeof payload.output === "string" ? payload.output : "";
+ const glyph = item.status === "running" ? "⏳" : item.status === "failed" ? "✕" : item.status === "interrupted" ? "■" : "✓";
+ return (
+
+ {glyph} {item.title} {item.summary} ›
+
+ {steps.length > 0 ?
{steps.map((step, index) => {step.status === "completed" ? "✓" : step.status === "blocked" ? "!" : step.status === "in_progress" ? "◐" : "○"} {String(step.body ?? "")} )} : null}
+ {children.length > 0 ?
{children.map((child, index) => {String(child.role ?? "Child agent")} {String(child.status ?? "unknown")} {String(child.summary ?? "")} )} : null}
+ {sources.length > 0 ?
{sources.map((source, index) => { const url = typeof source.url === "string" && /^https?:\/\//.test(source.url) ? source.url : null; return {url === null ? {String(source.title ?? "Unavailable source")} : {String(source.title ?? url)} }Provider-reported ; })} : null}
+ {output.length > 0 ?
{output} : null}
+ {item.family === "model_identity" ?
Requested {String(payload.requestedModel ?? "—")} Effective {String(payload.effectiveModel ?? "—")} : null}
+ {item.family === "artifact" && typeof payload.reference === "string" ?
Open {payload.reference} : null}
+
callbacks.onOpenEvidence(item.evidenceRef.section, item.evidenceRef.recordId)}>View in Evidence
+
+
+ );
+}
+
+function FileDiff({ diff }: { diff: string | null }) {
+ if (diff === null) return Binary or oversized file; textual diff unavailable.
;
+ return (
+
+ {diff.split("\n").map((line, index) => (
+ {line || " "}{"\n"}
+ ))}
+
+ );
+}
+
+function WorkspaceChanges({
+ item,
+}: {
+ item: Extract;
+}) {
+ const [expanded, setExpanded] = useState(false);
+ const [selectedPath, setSelectedPath] = useState(null);
+ const files = item.changeSet.files;
+ const visible = expanded ? files : files.slice(0, 3);
+ const selected = files.find((file) => file.path === selectedPath) ?? null;
+ const stats = (file: (typeof files)[number]) => (
+
+ {file.additions === null ? null : +{file.additions} }
+ {file.deletions === null ? null : -{file.deletions} }
+
+ );
+
+ return (
+ <>
+
+
+ ▣
+
+ {item.changeSet.complete ? "Edited" : "Editing"} {files.length} file{files.length === 1 ? "" : "s"}
+ {item.changeSet.source === "runner_verified" ? "Verified from workspace" : "Reported by harness"}
+
+
+ {item.changeSet.totals.additions === null ? null : +{item.changeSet.totals.additions} }
+ {item.changeSet.totals.deletions === null ? null : -{item.changeSet.totals.deletions} }
+
+ setSelectedPath(files[0]?.path ?? null)}>Review
+
+
+ {visible.map((file) => (
+ setSelectedPath(file.path)}>
+
+ {file.previousPath === null ? file.path : `${file.previousPath} → ${file.path}`}
+ {file.operation.replace("_", " ")}
+
+ {stats(file)}
+
+ ))}
+
+ {files.length > 3 ? (
+ setExpanded((value) => !value)}>
+ {expanded ? "Show fewer files" : `Show ${files.length - 3} more files`} ⌄
+
+ ) : null}
+
+ {selected !== null ? (
+ { if (event.target === event.currentTarget) setSelectedPath(null); }}>
+
+
+ {files.length > 1 ? (
+
+ {files.map((file) => setSelectedPath(file.path)}>{file.path} )}
+
+ ) : null}
+
+
+
+ ) : null}
+ >
+ );
+}
+
+function WorkspaceFileReference({
+ item,
+}: {
+ item: Extract;
+}) {
+ const [open, setOpen] = useState(false);
+ const extension = item.reference.path.includes(".")
+ ? item.reference.path.split(".").at(-1)?.toUpperCase()
+ : "FILE";
+ const presentation = item.reference.presentation === "document"
+ ? "Document"
+ : item.reference.presentation === "code"
+ ? "Code"
+ : item.reference.presentation === "image"
+ ? "Image"
+ : "File";
+ return (
+ <>
+
+ ▤
+
+ {item.reference.displayName}
+ {presentation} · {extension}{item.reference.line === null ? "" : ` · line ${item.reference.line}`}
+
+ setOpen(true)}>Open
+
+ {open ? (
+ { if (event.target === event.currentTarget) setOpen(false); }}>
+
+
+
+ {item.reference.preview === null
+ ?
Preview unavailable. The reference remains recorded with its normalized workspace path.
+ : item.reference.presentation === "document"
+ ?
{item.reference.preview}
+ :
{item.reference.preview} }
+ {item.reference.previewTruncated ?
Preview truncated by the runner.
: null}
+
+
+
+ ) : null}
+ >
+ );
+}
+
+function ThreadItemView({
+ item,
+ callbacks,
+}: {
+ item: CapabilityThreadItem;
+ callbacks: ThreadCallbacks;
+}) {
+ switch (item.kind) {
+ case "user_message":
+ return (
+
+
+ {item.author}
+
+
+ {item.body}
+
+ );
+
+ case "agent_message":
+ return (
+
+
+ {item.author}
+
+ {item.streaming ? (
+
+ ⏳
+ Streaming
+
+ ) : null}
+
+ {item.body}
+
+ );
+
+ case "durable_comment":
+ return (
+
+
+ {item.author}
+
+
+ ◆
+ Recorded to mock thread
+
+
+ {item.body}
+
+ callbacks.onOpenEvidence(item.evidenceRef.section, item.evidenceRef.recordId)
+ }
+ >
+ View in Evidence
+
+
+ );
+
+ case "tool_activity":
+ return ;
+
+ case "progress_activity":
+ return ;
+
+ case "workspace_changes":
+ return ;
+
+ case "workspace_file_reference":
+ return ;
+
+ case "provider_activity":
+ return ;
+
+ case "denial":
+ return (
+
+
+
+ ✕
+
+ {item.operationId}
+
+ {item.reason}
+
+
+
+
+ callbacks.onOpenEvidence(item.evidenceRef.section, item.evidenceRef.recordId)
+ }
+ >
+ View in Evidence
+
+
+
+ );
+
+ case "interaction":
+ return (
+
+ callbacks.onOpenEvidence(section as CapabilityEvidenceSectionId, recordId)
+ }
+ />
+ );
+
+ case "document":
+ return (
+
+
+ {item.title}
+ {item.documentKey}
+
+
+
+
+ {item.revisionFrom === null ? `r${item.revisionTo}` : `r${item.revisionFrom} → r${item.revisionTo}`}
+
+ {" · "}
+ {item.author}
+ {item.staleBehind !== null ? (
+ <>
+ {" "}
+
+ ⌛
+ Stale — {item.staleBehind} newer revision(s)
+
+ >
+ ) : null}
+
+
+ callbacks.onOpenEvidence(item.evidenceRef.section, item.evidenceRef.recordId)
+ }
+ >
+ View diff
+
+
+ );
+
+ case "deliverable":
+ return (
+
+
+ {item.filename}
+
+
+
+
+ {item.deliverableKind} · {formatBytes(item.byteSize)} · registered by {item.registeredBy}
+
+
+
+ callbacks.onOpenEvidence(item.evidenceRef.section, item.evidenceRef.recordId)
+ }
+ >
+ Download {item.filename}
+
+
+ );
+
+ case "dependency":
+ return (
+
+
+ Delegation
+
+
+
+ {item.createdTasks.map((task) => (
+
+ {task.identifier} {task.title}
+
+ ))}
+ {item.blockerEdges.map((edge) => (
+ {edge}
+ ))}
+
+
+ );
+
+ case "disposition":
+ return (
+
+
+
+ {item.operationId}
+
+
+ {item.body}
+ {item.blockerOwner !== null ? (
+ Blocker owner: {item.blockerOwner}
+ ) : null}
+
+ );
+
+ case "system_notice":
+ return (
+
+ {item.glyph}
+ {item.text}
+
+ callbacks.onOpenEvidence(item.evidenceRef.section, item.evidenceRef.recordId)
+ }
+ >
+ Details
+
+
+ );
+ }
+}
+
+export function TurnGroup({
+ turn,
+ callbacks,
+ assertions = {},
+ terminalAssertions = [],
+}: {
+ turn: CapabilityThreadTurn;
+ callbacks: ThreadCallbacks;
+ assertions?: Record;
+ terminalAssertions?: EvalAssertion[];
+}) {
+ const [showAllStrips, setShowAllStrips] = useState(false);
+ const stripIndexes = turn.items
+ .map((item, index) => (item.kind === "tool_activity" ? index : -1))
+ .filter((index) => index >= 0);
+ const hiddenStripIndexes = new Set(
+ showAllStrips ? [] : stripIndexes.slice(VISIBLE_STRIPS),
+ );
+
+ return (
+
+
+
+ Turn {turn.ordinal} · {turn.mode} · {turn.toolCallCount} tool call
+ {turn.toolCallCount === 1 ? "" : "s"} ·{" "}
+ {new Date(turn.at).toISOString().slice(11, 19)}
+
+ {turn.stoppedByUser ? (
+
+ Stopped by user
+
+ ) : null}
+
+ {turn.items.map((item, index) => hiddenStripIndexes.has(index) ? null : (
+
+
+
+
+ ))}
+ {hiddenStripIndexes.size > 0 ? (
+ setShowAllStrips(true)}
+ >
+ {hiddenStripIndexes.size} more…
+
+ ) : null}
+
+
+ );
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/fake-store.ts b/packages/paperclip-runner/devtools/issue-thread/src/fake-store.ts
new file mode 100644
index 0000000000..b4dd0817c3
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/fake-store.ts
@@ -0,0 +1,82 @@
+import type {
+ CapabilityIssueThreadSnapshot,
+ CapabilityThreadInteractionState,
+ CapabilityThreadItem,
+} from "../../../src/issue-thread/types";
+import type { CapabilityInteractionResponse } from "./InteractionCard";
+
+/**
+ * `fake`-mode interaction store.
+ *
+ * The screenshot matrix runs without a session server, so this module plays
+ * the mock control plane's role for the deterministic fixtures only: it stores
+ * the response and then hands back a new snapshot, mirroring the live authority
+ * path (store first, resume second) rather than resolving the card optimistically
+ * in the card component. Live mode never reaches this code — it posts to the
+ * package server instead.
+ */
+
+function resolvedState(
+ outcome: CapabilityInteractionResponse["outcome"],
+): { state: CapabilityThreadInteractionState; label: string } {
+ if (outcome === "accepted") return { state: "accepted", label: "Accepted" };
+ if (outcome === "rejected") return { state: "rejected", label: "Changes requested" };
+ return { state: "answered", label: "Answered" };
+}
+
+function summarise(response: CapabilityInteractionResponse): string[] {
+ const result = response.result;
+ if (Array.isArray(result.selected)) return result.selected.map((entry) => String(entry));
+ if (Array.isArray(result.acceptedTaskIds)) {
+ return result.acceptedTaskIds.map((entry) => `Accepted ${String(entry)}`);
+ }
+ if (Array.isArray(result.verdicts)) {
+ return result.verdicts.map((entry) => {
+ const verdict = entry as { id?: unknown; verdict?: unknown };
+ return `${String(verdict.id)} — ${String(verdict.verdict)}`;
+ });
+ }
+ if (typeof result.answers === "object" && result.answers !== null) {
+ return Object.entries(result.answers as Record).map(
+ ([key, value]) => `${key} — ${String(value)}`,
+ );
+ }
+ if (result.accepted === true) return ["Accepted"];
+ return [];
+}
+
+function applyToItem(
+ item: CapabilityThreadItem,
+ response: CapabilityInteractionResponse,
+): CapabilityThreadItem {
+ if (item.kind !== "interaction" || item.interactionId !== response.interactionId) return item;
+ const view = resolvedState(response.outcome);
+ return {
+ ...item,
+ state: view.state,
+ stateLabel: view.label,
+ resolvedSummary: summarise(response),
+ reason: typeof response.result.reason === "string" ? response.result.reason : null,
+ };
+}
+
+/** Store the typed response, then return the snapshot the card renders from. */
+export function applyFakeInteractionResponse(
+ snapshot: CapabilityIssueThreadSnapshot,
+ response: CapabilityInteractionResponse,
+): CapabilityIssueThreadSnapshot {
+ const turns = snapshot.turns.map((turn) => ({
+ ...turn,
+ items: turn.items.map((item) => applyToItem(item, response)),
+ }));
+ const stillPending = turns.some(
+ (turn) => turn.items.some((item) => item.kind === "interaction" && item.state === "pending"),
+ );
+ return {
+ ...snapshot,
+ turns,
+ composer: stillPending
+ ? snapshot.composer
+ : { state: "ready", helper: null, reason: null, pendingInteractionId: null },
+ };
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/fonts/NOTICE.md b/packages/paperclip-runner/devtools/issue-thread/src/fonts/NOTICE.md
new file mode 100644
index 0000000000..1fa3ee3b75
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/fonts/NOTICE.md
@@ -0,0 +1,78 @@
+# Capability issue-thread fonts
+
+The Capability issue-thread bundle includes Latin-subset WOFF2 files so captures
+do not depend on fonts installed on the host.
+
+## Inter
+
+- Upstream: https://github.com/rsms/inter
+- Version: 4.1
+- Source: `ui/public/fonts/InterVariable.woff2`
+- License: SIL Open Font License 1.1
+- License text: https://github.com/rsms/inter/blob/v4.1/LICENSE.txt
+
+The bundled file is subset from the repository's unmodified upstream Inter 4.1
+variable WOFF2. CSS exposes weights 400 through 700.
+
+## DejaVu Sans Mono
+
+- Upstream: https://dejavu-fonts.github.io/
+- Version: 2.37
+- Source faces: DejaVu Sans Mono Book and Bold; DejaVu Sans Book for status glyphs
+- License: Bitstream Vera Fonts license; DejaVu changes are public domain
+- License text: https://dejavu-fonts.github.io/License.html
+
+The bundled files are Latin subsets of the Book and Bold faces. The source font
+copyright and permission notice follow.
+
+> Copyright (c) 2003 by Bitstream, Inc. All Rights Reserved. Bitstream Vera is a
+> trademark of Bitstream, Inc.
+>
+> Permission is hereby granted, free of charge, to any person obtaining a copy
+> of the fonts accompanying this license ("Fonts") and associated documentation
+> files (the "Font Software"), to reproduce and distribute the Font Software,
+> including without limitation the rights to use, copy, merge, publish,
+> distribute, and/or sell copies of the Font Software, and to permit persons to
+> whom the Font Software is furnished to do so, subject to the following
+> conditions:
+>
+> The above copyright and trademark notices and this permission notice shall be
+> included in all copies of one or more of the Font Software typefaces.
+>
+> The Font Software may be modified, altered, or added to, and in particular the
+> designs of glyphs or characters in the Fonts may be modified and additional
+> glyphs or characters may be added to the Fonts, only if the fonts are renamed
+> to names not containing either the words "Bitstream" or the word "Vera".
+>
+> This License becomes null and void to the extent applicable to Fonts or Font
+> Software that has been modified and is distributed under the "Bitstream Vera"
+> names.
+>
+> The Font Software may be sold as part of a larger software package but no copy
+> of one or more of the Font Software typefaces may be sold by itself.
+>
+> THE FONT SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
+> OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTIES OF MERCHANTABILITY,
+> FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT OF COPYRIGHT, PATENT,
+> TRADEMARK, OR OTHER RIGHT. IN NO EVENT SHALL BITSTREAM OR THE GNOME FOUNDATION
+> BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, INCLUDING ANY GENERAL,
+> SPECIAL, INDIRECT, INCIDENTAL, OR CONSEQUENTIAL DAMAGES, WHETHER IN AN ACTION
+> OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF THE USE OR INABILITY TO
+> USE THE FONT SOFTWARE OR FROM OTHER DEALINGS IN THE FONT SOFTWARE.
+>
+> Except as contained in this notice, the names of Gnome, the Gnome Foundation,
+> and Bitstream Inc., shall not be used in advertising or otherwise to promote
+> the sale, use or other dealings in the Font Software without prior written
+> authorization from the Gnome Foundation or Bitstream Inc., respectively.
+
+## Noto Sans Symbols 2
+
+- Upstream: https://github.com/notofonts/symbols
+- Version: 2.003
+- Source face: Noto Sans Symbols 2 Regular
+- License: SIL Open Font License 1.1
+- License text: https://github.com/notofonts/symbols/blob/main/OFL.txt
+
+The 1.2 kB bundled subset contains only the two hourglass glyphs that are not
+present in Inter or DejaVu Sans. It is exposed under the same package-specific
+symbol fallback family using a disjoint `unicode-range`.
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-mono-latin-400.woff2 b/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-mono-latin-400.woff2
new file mode 100644
index 0000000000..7700aef0cc
Binary files /dev/null and b/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-mono-latin-400.woff2 differ
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-mono-latin-700.woff2 b/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-mono-latin-700.woff2
new file mode 100644
index 0000000000..83d6fc37b2
Binary files /dev/null and b/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-mono-latin-700.woff2 differ
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-symbols.woff2 b/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-symbols.woff2
new file mode 100644
index 0000000000..a786706688
Binary files /dev/null and b/packages/paperclip-runner/devtools/issue-thread/src/fonts/dejavu-sans-symbols.woff2 differ
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/fonts/inter-latin.woff2 b/packages/paperclip-runner/devtools/issue-thread/src/fonts/inter-latin.woff2
new file mode 100644
index 0000000000..8f09bddd4d
Binary files /dev/null and b/packages/paperclip-runner/devtools/issue-thread/src/fonts/inter-latin.woff2 differ
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/fonts/noto-sans-symbols2-hourglass.woff2 b/packages/paperclip-runner/devtools/issue-thread/src/fonts/noto-sans-symbols2-hourglass.woff2
new file mode 100644
index 0000000000..7c24a451df
Binary files /dev/null and b/packages/paperclip-runner/devtools/issue-thread/src/fonts/noto-sans-symbols2-hourglass.woff2 differ
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/issue-thread.css b/packages/paperclip-runner/devtools/issue-thread/src/issue-thread.css
new file mode 100644
index 0000000000..4b372ef3cc
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/issue-thread.css
@@ -0,0 +1,2546 @@
+/*
+ * Capability issue-thread surface.
+ *
+ * Package-local implementation of the Paperclip visual language (contract §8):
+ * dark default, OKLCH neutral grays, semantic tokens only. Components carry no
+ * raw color literals — `pnpm check:browser-tokens` enforces that.
+ */
+
+@font-face {
+ font-family: "Paperclip Issue Thread Inter";
+ src: url("./fonts/inter-latin.woff2") format("woff2");
+ font-style: normal;
+ font-weight: 400 700;
+ font-display: block;
+}
+
+@font-face {
+ font-family: "Paperclip Issue Thread DejaVu Sans Mono";
+ src: url("./fonts/dejavu-sans-mono-latin-400.woff2") format("woff2");
+ font-style: normal;
+ font-weight: 400;
+ font-display: block;
+}
+
+@font-face {
+ font-family: "Paperclip Issue Thread DejaVu Sans Mono";
+ src: url("./fonts/dejavu-sans-mono-latin-700.woff2") format("woff2");
+ font-style: normal;
+ font-weight: 700;
+ font-display: block;
+}
+
+/* Inter intentionally omits several status glyphs. Keep those deterministic
+ with two tiny, disjoint symbol subsets instead of consulting host fonts. */
+@font-face {
+ font-family: "Paperclip Issue Thread Symbols";
+ src: url("./fonts/dejavu-sans-symbols.woff2") format("woff2");
+ font-style: normal;
+ font-weight: 400;
+ font-display: block;
+ unicode-range: U+00A7, U+00B7, U+2013-2014, U+2022, U+2026, U+203A, U+2192,
+ U+22EF, U+2304, U+25AC, U+25B2, U+25B6, U+25BC, U+25C6, U+25CB, U+25D0-25D1,
+ U+2699, U+26A0, U+2713, U+2715, U+275A;
+}
+
+@font-face {
+ font-family: "Paperclip Issue Thread Symbols";
+ src: url("./fonts/noto-sans-symbols2-hourglass.woff2") format("woff2");
+ font-style: normal;
+ font-weight: 400;
+ font-display: block;
+ unicode-range: U+231B, U+23F3, U+FE0E;
+}
+
+:root {
+ color-scheme: dark;
+
+ /* Neutral ramp */
+ --pit-background: oklch(0.17 0.005 285);
+ --pit-surface: oklch(0.21 0.006 285);
+ --pit-surface-raised: oklch(0.245 0.007 285);
+ --pit-surface-sunken: oklch(0.145 0.005 285);
+ --pit-border: oklch(0.31 0.008 285);
+ --pit-border-strong: oklch(0.42 0.01 285);
+ --pit-foreground: oklch(0.95 0.004 285);
+ --pit-muted-foreground: oklch(0.72 0.008 285);
+ --pit-faint-foreground: oklch(0.64 0.008 285);
+
+ /* Semantic hues — product status table */
+ --pit-todo: oklch(0.72 0.13 245);
+ --pit-in-progress: oklch(0.7 0.15 275);
+ --pit-in-review: oklch(0.72 0.16 300);
+ --pit-done: oklch(0.74 0.15 155);
+ --pit-blocked: oklch(0.68 0.18 25);
+ --pit-neutral: oklch(0.66 0.01 285);
+
+ --pit-accent: oklch(0.7 0.15 275);
+ --pit-accent-surface: oklch(0.28 0.05 275);
+ --pit-success: oklch(0.74 0.15 155);
+ --pit-success-surface: oklch(0.27 0.05 155);
+ --pit-warning: oklch(0.8 0.13 80);
+ --pit-warning-surface: oklch(0.29 0.05 80);
+ --pit-danger: oklch(0.7 0.17 25);
+ --pit-danger-surface: oklch(0.28 0.06 25);
+ --pit-live: oklch(0.78 0.12 200);
+ --pit-live-surface: oklch(0.27 0.05 200);
+ --pit-ring: oklch(0.78 0.11 275);
+
+ /* Type ramp */
+ --pit-font-sans: "Paperclip Issue Thread Inter", "Paperclip Issue Thread Symbols", "Inter", ui-sans-serif, system-ui, sans-serif;
+ --pit-font-mono: "Paperclip Issue Thread DejaVu Sans Mono", "Paperclip Issue Thread Symbols", "DejaVu Sans Mono", ui-monospace, monospace;
+ --pit-font-symbols: "Paperclip Issue Thread Symbols";
+ --pit-text-xs: 0.75rem;
+ --pit-text-sm: 0.875rem;
+ --pit-text-base: 1rem;
+ --pit-text-xl: 1.25rem;
+ --pit-leading-tight: 1.25;
+ --pit-leading-normal: 1.55;
+
+ /* Space and shape */
+ --pit-space-1: 0.25rem;
+ --pit-space-2: 0.5rem;
+ --pit-space-3: 0.75rem;
+ --pit-space-4: 1rem;
+ --pit-space-5: 1.25rem;
+ --pit-space-6: 1.5rem;
+ --pit-radius-sm: 0.375rem;
+ --pit-radius-md: 0.5rem;
+ --pit-radius-lg: 0.75rem;
+ --pit-shadow-sm: 0 1px 2px oklch(0 0 0 / 0.35);
+ --pit-scrim: oklch(0 0 0 / 0.6);
+ --pit-thread-measure: 47.5rem;
+ --pit-panel-min: 20rem;
+ --pit-panel-max: 40rem;
+ --pit-touch-target: 2.75rem;
+ --pit-motion: 160ms;
+}
+
+*,
+*::before,
+*::after {
+ box-sizing: border-box;
+}
+
+[hidden] {
+ /* A layout class must never win over the hidden attribute. */
+ display: none !important;
+}
+
+html,
+body {
+ margin: 0;
+ padding: 0;
+ background: var(--pit-background);
+ color: var(--pit-foreground);
+}
+
+/*
+ * Body carries the type scale, never `html`. Shrinking the root font size
+ * would rescale every `rem`, including the 44px touch-target token.
+ */
+body {
+ font-family: var(--pit-font-sans);
+ font-variant-emoji: text;
+ font-size: var(--pit-text-sm);
+ line-height: var(--pit-leading-normal);
+}
+
+button,
+input,
+select,
+textarea {
+ font: inherit;
+ color: inherit;
+}
+
+:focus-visible {
+ outline: 3px solid var(--pit-ring);
+ outline-offset: 2px;
+}
+
+/*
+ * The app owns the viewport so the thread column scrolls internally and the
+ * composer stays pinned to the bottom of the thread column (§2.1).
+ */
+.pit-app {
+ position: relative;
+ display: flex;
+ flex-direction: column;
+ box-sizing: border-box;
+ padding-left: 15rem;
+ height: 100vh;
+ height: 100dvh;
+ overflow: hidden;
+ background: var(--pit-background);
+}
+
+.pit-app[data-eval-view="true"] { padding-left: 0; }
+
+.pit-eval-nav { display: grid; grid-template-columns:auto auto minmax(0,1fr) auto; align-items:center; gap:12px; min-height:var(--pit-touch-target); padding:0 var(--pit-space-4); border-bottom:1px solid var(--pit-border-strong); background:var(--pit-surface-sunken); font-size:var(--pit-text-xs); }
+.pit-eval-nav strong { text-align:center; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; }
+.pit-eval-nav a { color:var(--pit-live); white-space:nowrap; }
+@media(max-width:800px){.pit-eval-nav{grid-template-columns:auto 1fr auto}.pit-eval-nav>a:not(:first-child),.pit-eval-nav>span{display:none}.pit-eval-nav strong{text-align:left}}
+
+.pit-visually-hidden {
+ position: absolute;
+ width: 1px;
+ height: 1px;
+ margin: -1px;
+ padding: 0;
+ overflow: hidden;
+ clip-path: inset(50%);
+ white-space: nowrap;
+ border: 0;
+}
+
+/* ------------------------------------------------------------- sidebar */
+
+.pit-surface-nav {
+ position: absolute;
+ inset: 0 auto 0 0;
+ z-index: 4;
+ display: flex;
+ flex-direction: column;
+ width: 15rem;
+ box-sizing: border-box;
+ padding: var(--pit-space-3);
+ background: var(--pit-surface-sunken);
+ border-right: 1px solid var(--pit-border);
+ overflow: hidden;
+}
+
+.pit-sidebar-brand {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-2);
+ color: var(--pit-foreground);
+ font-weight: 700;
+}
+
+.pit-sidebar-brand .pit-icon {
+ color: var(--pit-live);
+}
+
+.pit-sidebar-primary {
+ display: grid;
+ gap: var(--pit-space-1);
+ padding-bottom: var(--pit-space-4);
+ border-bottom: 1px solid var(--pit-border);
+}
+
+.pit-surface-link {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ width: 100%;
+ box-sizing: border-box;
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-2);
+ border: 1px solid transparent;
+ border-radius: var(--pit-radius-md);
+ background: transparent;
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+ text-decoration: none;
+ text-align: left;
+ cursor: pointer;
+}
+
+.pit-surface-link:hover {
+ color: var(--pit-foreground);
+ border-color: var(--pit-border-strong);
+}
+
+.pit-surface-link[aria-current="page"] {
+ background: var(--pit-accent-surface);
+ border-color: var(--pit-accent);
+ color: var(--pit-foreground);
+}
+
+.pit-session-history {
+ display: flex;
+ flex: 1;
+ min-height: 0;
+ flex-direction: column;
+ padding-top: var(--pit-space-4);
+}
+
+.pit-session-history h2 {
+ margin: 0 0 var(--pit-space-2);
+ padding: 0 var(--pit-space-2);
+ color: var(--pit-faint-foreground);
+ font-size: var(--pit-text-xs);
+ font-weight: 600;
+ letter-spacing: 0.06em;
+ text-transform: uppercase;
+}
+
+.pit-session-history-list {
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-1);
+ overflow-y: auto;
+}
+
+.pit-session-history-item {
+ display: flex;
+ flex-direction: column;
+ gap: 0.15rem;
+ width: 100%;
+ padding: var(--pit-space-2);
+ border: 1px solid transparent;
+ border-radius: var(--pit-radius-md);
+ background: transparent;
+ color: var(--pit-faint-foreground);
+ font-size: var(--pit-text-xs);
+ text-align: left;
+ cursor: pointer;
+}
+
+.pit-session-history-item:hover,
+.pit-session-history-item[data-selected="true"] {
+ border-color: var(--pit-border);
+ background: var(--pit-surface);
+}
+
+.pit-session-history-item[data-selected="true"] {
+ border-color: var(--pit-accent);
+}
+
+.pit-session-history-title {
+ overflow: hidden;
+ color: var(--pit-foreground);
+ font-weight: 600;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+/* A load failure and the pre-session wait share the centred single-column. */
+.pit-app-error {
+ display: flex;
+ flex: 1;
+ flex-direction: column;
+ align-items: flex-start;
+ gap: var(--pit-space-3);
+ max-width: var(--pit-thread-measure);
+ padding: var(--pit-space-6) var(--pit-space-4);
+}
+
+.pit-muted {
+ margin: 0;
+ color: var(--pit-muted-foreground);
+}
+
+/*
+ * The clean-room blank state. It carries the whole explanation of what the
+ * surface is, because the thread it replaces is deliberately empty.
+ */
+.pit-empty-thread {
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-3);
+ padding: var(--pit-space-5);
+ background: var(--pit-surface);
+ border: 1px dashed var(--pit-border-strong);
+ border-radius: var(--pit-radius-lg);
+}
+
+.pit-empty-thread h2 {
+ margin: 0;
+ font-size: var(--pit-text-base);
+ line-height: var(--pit-leading-tight);
+}
+
+.pit-empty-thread p {
+ margin: 0;
+}
+
+/* ---------------------------------------------------------------- header */
+
+.pit-header {
+ position: sticky;
+ top: 0;
+ z-index: 5;
+ flex: none;
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-3);
+ padding: var(--pit-space-3) var(--pit-space-4);
+ background: var(--pit-surface);
+ border-bottom: 1px solid var(--pit-border);
+ box-shadow: var(--pit-shadow-sm);
+}
+
+.pit-header-primary,
+.pit-header-context {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--pit-space-3);
+ min-width: 0;
+}
+
+.pit-header-context {
+ justify-content: flex-start;
+}
+
+.pit-header-title-row {
+ display: flex;
+ align-items: center;
+ flex-wrap: wrap;
+ gap: var(--pit-space-2);
+ min-width: 0;
+}
+
+.pit-identifier {
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ color: var(--pit-muted-foreground);
+}
+
+.pit-title {
+ margin: 0;
+ font-size: var(--pit-text-xl);
+ font-weight: 700;
+ line-height: var(--pit-leading-tight);
+ min-width: 0;
+}
+
+.pit-icon {
+ width: var(--pit-space-4);
+ height: var(--pit-space-4);
+ flex: none;
+ fill: none;
+ stroke: currentColor;
+ stroke-width: 1.75;
+ stroke-linecap: round;
+ stroke-linejoin: round;
+}
+
+.pit-assignee {
+ font-size: var(--pit-text-xs);
+ color: var(--pit-muted-foreground);
+}
+
+.pit-chip-row {
+ display: flex;
+ flex-wrap: wrap;
+ align-items: center;
+ gap: var(--pit-space-2);
+ min-width: 0;
+}
+
+.pit-chip {
+ display: inline-flex;
+ align-items: center;
+ gap: var(--pit-space-1);
+ padding: 0.125rem var(--pit-space-2);
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface-raised);
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+ white-space: nowrap;
+}
+
+.pit-chip[data-tone="live"] {
+ border-color: var(--pit-live);
+ background: var(--pit-live-surface);
+ color: var(--pit-foreground);
+}
+
+.pit-chip[data-tone="mock"] {
+ border-color: var(--pit-warning);
+ background: var(--pit-warning-surface);
+ color: var(--pit-foreground);
+}
+
+.pit-chip[data-tone="accent"] {
+ border-color: var(--pit-accent);
+ background: var(--pit-accent-surface);
+ color: var(--pit-foreground);
+}
+
+.pit-chip[data-tone="success"] {
+ border-color: var(--pit-success);
+ background: var(--pit-success-surface);
+ color: var(--pit-foreground);
+}
+
+.pit-chip[data-tone="danger"] {
+ border-color: var(--pit-danger);
+ background: var(--pit-danger-surface);
+ color: var(--pit-foreground);
+}
+
+.pit-process-dot {
+ width: 0.5rem;
+ height: 0.5rem;
+ border-radius: 50%;
+ background: var(--pit-neutral);
+ flex: none;
+}
+
+.pit-process-dot[data-attached="true"] {
+ background: var(--pit-live);
+ animation: pit-pulse 1.6s ease-in-out infinite;
+}
+
+@keyframes pit-pulse {
+ 0%,
+ 100% {
+ opacity: 1;
+ }
+ 50% {
+ opacity: 0.35;
+ }
+}
+
+.pit-status-badge {
+ display: inline-flex;
+ align-items: center;
+ gap: var(--pit-space-1);
+ padding: 0.125rem var(--pit-space-2);
+ border-radius: var(--pit-radius-sm);
+ border: 1px solid currentColor;
+ font-size: var(--pit-text-xs);
+ font-weight: 500;
+ white-space: nowrap;
+}
+
+.pit-status-badge[data-status="todo"] {
+ color: var(--pit-todo);
+}
+.pit-status-badge[data-status="in_progress"] {
+ color: var(--pit-in-progress);
+}
+.pit-status-badge[data-status="in_review"] {
+ color: var(--pit-in-review);
+}
+.pit-status-badge[data-status="done"] {
+ color: var(--pit-done);
+}
+.pit-status-badge[data-status="blocked"] {
+ color: var(--pit-blocked);
+}
+.pit-status-badge[data-status="backlog"],
+.pit-status-badge[data-status="cancelled"] {
+ color: var(--pit-neutral);
+}
+
+.pit-priority {
+ font-size: var(--pit-text-xs);
+ color: var(--pit-muted-foreground);
+ white-space: nowrap;
+}
+
+.pit-header-controls {
+ display: flex;
+ align-items: center;
+ flex-wrap: wrap;
+ gap: var(--pit-space-2);
+}
+
+.pit-run-state {
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ color: var(--pit-faint-foreground);
+ max-width: 20rem;
+ overflow: hidden;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.pit-button {
+ display: inline-flex;
+ align-items: center;
+ justify-content: center;
+ gap: var(--pit-space-1);
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-3);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface-raised);
+ color: var(--pit-foreground);
+ font-size: var(--pit-text-xs);
+ cursor: pointer;
+}
+
+.pit-icon-button {
+ display: inline-flex;
+ align-items: center;
+ justify-content: center;
+ width: var(--pit-touch-target);
+ height: var(--pit-touch-target);
+ padding: 0;
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface-raised);
+ color: var(--pit-foreground);
+ cursor: pointer;
+}
+
+.pit-icon-button:hover:not(:disabled) {
+ border-color: var(--pit-accent);
+ background: var(--pit-accent-surface);
+}
+
+.pit-icon-button:disabled {
+ opacity: 0.45;
+ cursor: not-allowed;
+}
+
+.pit-icon-button[data-variant="destructive"] {
+ color: var(--pit-danger);
+}
+
+.pit-button:hover:not(:disabled) {
+ background: var(--pit-surface);
+}
+
+.pit-button:disabled {
+ opacity: 0.5;
+ cursor: not-allowed;
+}
+
+.pit-button[data-variant="primary"] {
+ background: var(--pit-accent);
+ border-color: var(--pit-accent);
+ color: var(--pit-surface-sunken);
+ font-weight: 600;
+}
+
+.pit-button[data-variant="destructive"] {
+ border-color: var(--pit-danger);
+ color: var(--pit-danger);
+}
+
+.pit-select {
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-2);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface-raised);
+ font-size: var(--pit-text-xs);
+}
+
+/* ---------------------------------------------------------------- layout */
+
+.pit-body {
+ display: flex;
+ flex: 1;
+ min-height: 0;
+ min-width: 0;
+ overflow: hidden;
+}
+
+.pit-main {
+ display: flex;
+ flex-direction: column;
+ flex: 1;
+ min-width: 0;
+}
+
+.pit-thread-scroll {
+ flex: 1;
+ overflow-y: auto;
+ overflow-x: hidden;
+ padding: var(--pit-space-5) var(--pit-space-4);
+}
+
+.pit-thread {
+ width: 100%;
+ max-width: var(--pit-thread-measure);
+ margin: 0 auto;
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-5);
+ min-width: 0;
+}
+
+.pit-turn {
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-3);
+ min-width: 0;
+}
+
+.pit-turn-header {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ font-size: var(--pit-text-xs);
+ color: var(--pit-faint-foreground);
+ font-family: var(--pit-font-mono);
+}
+
+.pit-turn-header::after {
+ content: "";
+ flex: 1;
+ height: 1px;
+ background: var(--pit-border);
+}
+
+.pit-stopped-marker {
+ color: var(--pit-warning);
+ font-family: var(--pit-font-sans);
+}
+
+/* ------------------------------------------------------------ thread items */
+
+.pit-card {
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-lg);
+ background: var(--pit-surface);
+ padding: var(--pit-space-3) var(--pit-space-4);
+ box-shadow: var(--pit-shadow-sm);
+ min-width: 0;
+}
+
+.pit-workspace-card {
+ overflow: hidden;
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-lg);
+ background: var(--pit-surface);
+ box-shadow: var(--pit-shadow-sm);
+}
+
+.pit-workspace-head {
+ display: grid;
+ grid-template-columns: auto minmax(0, 1fr) auto auto;
+ align-items: center;
+ gap: var(--pit-space-3);
+ padding: var(--pit-space-3) var(--pit-space-4);
+ border-bottom: 1px solid var(--pit-border);
+}
+
+.pit-workspace-head strong,
+.pit-workspace-head small,
+.pit-workspace-path small { display: block; }
+.pit-workspace-head small,
+.pit-workspace-path small,
+.pit-workspace-binary { color: var(--pit-muted-foreground); }
+.pit-workspace-icon { font-family: var(--pit-font-symbols); font-size: var(--pit-text-xl); }
+.pit-workspace-total,
+.pit-workspace-stats { display: flex; gap: var(--pit-space-2); font-family: var(--pit-font-mono); }
+[data-stat="add"] { color: var(--pit-success); }
+[data-stat="delete"] { color: var(--pit-danger); }
+
+.pit-workspace-review,
+.pit-workspace-more,
+.pit-workspace-files button,
+.pit-workspace-file-nav button {
+ border: 0;
+ background: transparent;
+ cursor: pointer;
+}
+
+.pit-workspace-review {
+ min-height: 2rem;
+ padding: 0 var(--pit-space-3);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-md);
+ background: var(--pit-surface-raised);
+}
+
+.pit-workspace-files button {
+ display: flex;
+ width: 100%;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--pit-space-4);
+ min-height: 2.75rem;
+ padding: var(--pit-space-2) var(--pit-space-4);
+ text-align: left;
+}
+.pit-workspace-files button:hover,
+.pit-workspace-file-nav button:hover,
+.pit-workspace-file-nav button[data-selected="true"] { background: var(--pit-surface-raised); }
+.pit-workspace-path { min-width: 0; overflow-wrap: anywhere; font-family: var(--pit-font-mono); }
+.pit-workspace-path small { font-family: var(--pit-font-sans); }
+.pit-workspace-more { width: 100%; padding: var(--pit-space-2) var(--pit-space-4); border-top: 1px solid var(--pit-border); text-align: left; }
+
+.pit-dialog.pit-workspace-dialog {
+ width: min(70rem, calc(100vw - 2rem));
+ max-width: 70rem;
+ max-height: calc(100vh - 2rem);
+ padding: 0;
+ gap: 0;
+ overflow: hidden;
+}
+.pit-workspace-dialog > .pit-tool-dialog-head { padding: var(--pit-space-4); }
+.pit-workspace-file-nav { display: flex; max-height: 9rem; overflow: auto; border-bottom: 1px solid var(--pit-border); }
+.pit-workspace-file-nav button { padding: var(--pit-space-2) var(--pit-space-3); white-space: nowrap; font-family: var(--pit-font-mono); }
+.pit-workspace-diff { max-height: min(65vh, 44rem); margin: 0; overflow: auto; padding: var(--pit-space-4); background: var(--pit-surface-sunken); font-family: var(--pit-font-mono); font-size: var(--pit-text-xs); line-height: 1.55; }
+.pit-workspace-diff span { display: block; min-width: max-content; }
+.pit-workspace-diff [data-diff-line="addition"] { color: var(--pit-success); background: var(--pit-success-surface); }
+.pit-workspace-diff [data-diff-line="deletion"] { color: var(--pit-danger); background: var(--pit-danger-surface); }
+.pit-workspace-diff [data-diff-line="hunk"] { color: var(--pit-accent); }
+.pit-workspace-binary { padding: var(--pit-space-4); }
+
+.pit-file-reference-card {
+ display: grid;
+ grid-template-columns: auto minmax(0, 1fr) auto;
+ align-items: center;
+ gap: var(--pit-space-3);
+ padding: var(--pit-space-3);
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-lg);
+ background: var(--pit-surface);
+ box-shadow: var(--pit-shadow-sm);
+}
+.pit-file-reference-icon { display: grid; place-items: center; width: 2.75rem; height: 2.75rem; border-radius: var(--pit-radius-md); background: var(--pit-surface-raised); font-family: var(--pit-font-symbols); font-size: var(--pit-text-xl); }
+.pit-file-reference-copy { min-width: 0; }
+.pit-file-reference-copy strong,
+.pit-file-reference-copy small { display: block; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; }
+.pit-file-reference-copy small { color: var(--pit-muted-foreground); }
+.pit-dialog.pit-file-reference-dialog { width: min(58rem, calc(100vw - 2rem)); max-width: 58rem; max-height: calc(100vh - 2rem); padding: 0; gap: 0; overflow: hidden; }
+.pit-file-reference-dialog > .pit-tool-dialog-head { padding: var(--pit-space-4); border-bottom: 1px solid var(--pit-border); }
+.pit-file-reference-preview { overflow: auto; padding: var(--pit-space-4); }
+.pit-file-reference-preview > .pit-code { max-height: 65vh; overflow: auto; white-space: pre; }
+
+/*
+ * Chat messages follow the familiar assistant flow: the board's message is a
+ * compact bubble, while the assistant answer sits directly on the page. The
+ * heavier card remains available for durable control-plane records below.
+ */
+.pit-message {
+ min-width: 0;
+}
+
+.pit-message[data-role="user"] {
+ align-self: flex-end;
+ width: fit-content;
+ max-width: 85%;
+ padding: var(--pit-space-3) var(--pit-space-4);
+ border-radius: var(--pit-radius-lg);
+ background: var(--pit-surface-raised);
+}
+
+.pit-message[data-role="assistant"] {
+ padding: var(--pit-space-2) 0;
+}
+
+.pit-message .pit-card-head {
+ color: var(--pit-muted-foreground);
+}
+
+.pit-app[data-surface="chat"] .pit-turn-label,
+.pit-app[data-surface="chat"] .pit-turn-header::after {
+ display: none;
+}
+
+.pit-card-head {
+ display: flex;
+ align-items: baseline;
+ flex-wrap: wrap;
+ gap: var(--pit-space-2);
+ margin-bottom: var(--pit-space-2);
+}
+
+.pit-card-author {
+ font-size: var(--pit-text-sm);
+ font-weight: 500;
+}
+
+.pit-card-meta {
+ font-size: var(--pit-text-xs);
+ color: var(--pit-muted-foreground);
+ font-family: var(--pit-font-mono);
+}
+
+.pit-card-body {
+ white-space: pre-wrap;
+ overflow-wrap: anywhere;
+}
+
+.pit-markdown {
+ white-space: normal;
+}
+
+.pit-markdown > :first-child { margin-top: 0; }
+.pit-markdown > :last-child { margin-bottom: 0; }
+.pit-markdown pre,
+.pit-markdown code {
+ font-family: var(--pit-font-mono);
+}
+.pit-markdown pre {
+ overflow-x: auto;
+ padding: var(--pit-space-3);
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-background);
+}
+.pit-markdown :not(pre) > code {
+ padding: 0.08rem 0.28rem;
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface-sunken);
+}
+.pit-markdown blockquote {
+ margin-left: 0;
+ padding-left: var(--pit-space-3);
+ border-left: 2px solid var(--pit-accent);
+ color: var(--pit-muted-foreground);
+}
+.pit-markdown table { border-collapse: collapse; }
+.pit-markdown th,
+.pit-markdown td { padding: var(--pit-space-2); border: 1px solid var(--pit-border); }
+
+.pit-durable-tag {
+ display: inline-flex;
+ align-items: center;
+ gap: var(--pit-space-1);
+ padding: 0.0625rem var(--pit-space-2);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-accent-surface);
+ border: 1px solid var(--pit-accent);
+ color: var(--pit-foreground);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-strip {
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-md);
+ background: var(--pit-surface-sunken);
+ min-width: 0;
+}
+
+.pit-strip[data-status="denied"] {
+ border-color: var(--pit-danger);
+}
+
+.pit-strip-button {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ width: 100%;
+ min-height: var(--pit-touch-target);
+ padding: var(--pit-space-2) var(--pit-space-3);
+ background: none;
+ border: 0;
+ text-align: left;
+ cursor: pointer;
+ color: inherit;
+ min-width: 0;
+}
+
+.pit-strip-glyph {
+ font-family: var(--pit-font-mono);
+ flex: none;
+}
+
+.pit-strip[data-status="ok"] .pit-strip-glyph {
+ color: var(--pit-success);
+}
+.pit-strip[data-status="denied"] .pit-strip-glyph {
+ color: var(--pit-danger);
+}
+.pit-strip[data-status="running"] .pit-strip-glyph {
+ color: var(--pit-warning);
+}
+
+.pit-strip-operation {
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ flex: none;
+}
+
+.pit-strip-summary {
+ flex: 1;
+ min-width: 0;
+ overflow: hidden;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-strip-caret {
+ flex: none;
+ color: var(--pit-faint-foreground);
+}
+
+.pit-strip-detail {
+ padding: 0 var(--pit-space-3) var(--pit-space-3);
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-2);
+}
+
+/* Lightweight assistant-style reasoning/tool disclosures. */
+.pit-activity-item {
+ min-width: 0;
+ color: var(--pit-muted-foreground);
+}
+
+.pit-activity-summary {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ min-height: var(--pit-touch-target);
+ padding: var(--pit-space-1) 0;
+ cursor: pointer;
+ list-style: none;
+}
+
+.pit-activity-summary::-webkit-details-marker {
+ display: none;
+}
+
+.pit-activity-glyph {
+ flex: none;
+ width: var(--pit-space-5);
+ color: var(--pit-muted-foreground);
+ font-family: var(--pit-font-mono);
+ text-align: center;
+}
+
+.pit-activity-item[data-status="ok"] .pit-activity-glyph,
+.pit-activity-item[data-status="complete"] .pit-activity-glyph {
+ color: var(--pit-success);
+}
+
+.pit-activity-item[data-status="denied"] .pit-activity-glyph {
+ color: var(--pit-danger);
+}
+
+.pit-activity-item[data-status="running"] .pit-activity-glyph {
+ color: var(--pit-live);
+ animation: pit-pulse 1.6s ease-in-out infinite;
+}
+
+.pit-activity-operation {
+ flex: none;
+ color: var(--pit-foreground);
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ font-weight: 600;
+}
+
+.pit-progress-activity .pit-activity-operation {
+ font-family: var(--pit-font-sans);
+}
+
+.pit-activity-description {
+ flex: 1;
+ min-width: 0;
+ overflow: hidden;
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.pit-activity-caret {
+ flex: none;
+ color: var(--pit-faint-foreground);
+ transition: transform var(--pit-motion) ease-out;
+}
+
+.pit-activity-item[open] > .pit-activity-summary .pit-activity-caret {
+ transform: rotate(90deg);
+}
+
+.pit-activity-detail {
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-2);
+ margin-left: calc(var(--pit-space-5) + var(--pit-space-2));
+ padding: 0 0 var(--pit-space-3) var(--pit-space-3);
+ border-left: 1px solid var(--pit-border);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-provider-activity[data-family="plan"] { border-left: 2px solid var(--pit-accent); }
+.pit-provider-activity[data-status="failed"] { border-left: 2px solid var(--pit-danger); }
+.pit-provider-plan, .pit-provider-tree, .pit-provider-sources { margin: 0 0 12px; padding-left: 22px; display: grid; gap: 8px; }
+.pit-provider-plan li { display: grid; grid-template-columns: 18px 1fr; gap: 6px; }
+.pit-provider-plan li[data-plan-status="blocked"] { color: var(--pit-danger); }
+.pit-provider-tree li { display: grid; grid-template-columns: minmax(100px, 1fr) auto; gap: 4px 12px; }
+.pit-provider-tree small { grid-column: 1 / -1; color: var(--pit-muted-foreground); }
+.pit-provider-sources li { display: flex; align-items: baseline; gap: 8px; }
+.pit-provider-sources small { color: var(--pit-muted-foreground); }
+.pit-provider-output { max-height: 260px; overflow: auto; margin: 0 0 12px; padding: 12px; white-space: pre-wrap; background: var(--pit-muted); border-radius: 6px; }
+.pit-provider-fields { display: grid; grid-template-columns: min-content 1fr; gap: 6px 12px; margin: 0 0 12px; }
+.pit-provider-fields dt { color: var(--pit-muted-foreground); }
+.pit-provider-fields dd { margin: 0; overflow-wrap: anywhere; }
+
+.pit-activity-detail p {
+ margin: 0;
+}
+
+.pit-activity-field {
+ display: grid;
+ gap: var(--pit-space-1);
+}
+
+.pit-activity-field > span {
+ color: var(--pit-faint-foreground);
+ font-family: var(--pit-font-mono);
+ text-transform: uppercase;
+ letter-spacing: 0.06em;
+}
+
+.pit-withheld-note {
+ color: var(--pit-faint-foreground);
+}
+
+.pit-activity-payloads {
+ display: grid;
+ gap: var(--pit-space-2);
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+}
+
+.pit-code {
+ margin: 0;
+ padding: var(--pit-space-2);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-background);
+ border: 1px solid var(--pit-border);
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ overflow-x: auto;
+ white-space: pre-wrap;
+ overflow-wrap: anywhere;
+}
+
+.pit-more-button {
+ align-self: flex-start;
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-2);
+ background: none;
+ border: 1px dashed var(--pit-border-strong);
+ border-radius: var(--pit-radius-sm);
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+ cursor: pointer;
+}
+
+.pit-notice {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ font-size: var(--pit-text-xs);
+ color: var(--pit-faint-foreground);
+ min-width: 0;
+}
+
+.pit-notice-text {
+ min-width: 0;
+ overflow-wrap: anywhere;
+}
+
+.pit-link-button {
+ background: none;
+ border: 0;
+ padding: 0;
+ color: var(--pit-accent);
+ text-decoration: underline;
+ cursor: pointer;
+ font-size: var(--pit-text-xs);
+}
+
+.pit-terminal-card {
+ border-left: 3px solid var(--pit-done);
+}
+
+.pit-denial {
+ border-left: 3px solid var(--pit-danger);
+}
+
+/* ------------------------------------------------------ interaction cards */
+
+.pit-interaction {
+ border: 1px solid var(--pit-border);
+ border-left: 3px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-lg);
+ background: var(--pit-surface);
+ padding: var(--pit-space-4);
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-3);
+ min-width: 0;
+}
+
+.pit-interaction[data-interaction-state="pending"] {
+ border-left-color: var(--pit-accent);
+}
+
+.pit-interaction[data-interaction-state="accepted"],
+.pit-interaction[data-interaction-state="answered"] {
+ border-left-color: var(--pit-success);
+}
+
+.pit-interaction[data-interaction-state="rejected"] {
+ border-left-color: var(--pit-danger);
+}
+
+/*
+ * Expired-family cards read as history without a blanket opacity. The contract
+ * asks for a 60% dimmed body (§5); a literal opacity drops the card's text to
+ * ~3.2:1 and fails the blocking axe gate (§9.7), so the dim is expressed as a
+ * recessed surface plus muted text that still clears 4.5:1. Recorded as
+ * revision 2 of the contract.
+ */
+.pit-interaction[data-interaction-state="stale_target"],
+.pit-interaction[data-interaction-state="superseded_by_comment"],
+.pit-interaction[data-interaction-state="expired"],
+.pit-interaction[data-interaction-state="withdrawn"],
+.pit-interaction[data-interaction-state="issue_closed"] {
+ border-left-color: var(--pit-neutral);
+ background: var(--pit-surface-sunken);
+ color: var(--pit-muted-foreground);
+}
+
+.pit-interaction[data-interaction-state="stale_target"] .pit-card-meta,
+.pit-interaction[data-interaction-state="superseded_by_comment"] .pit-card-meta,
+.pit-interaction[data-interaction-state="expired"] .pit-card-meta,
+.pit-interaction[data-interaction-state="withdrawn"] .pit-card-meta,
+.pit-interaction[data-interaction-state="issue_closed"] .pit-card-meta,
+.pit-interaction[data-interaction-state="stale_target"] .pit-quote,
+.pit-interaction[data-interaction-state="superseded_by_comment"] .pit-quote,
+.pit-interaction[data-interaction-state="expired"] .pit-quote,
+.pit-interaction[data-interaction-state="withdrawn"] .pit-quote,
+.pit-interaction[data-interaction-state="issue_closed"] .pit-quote {
+ color: var(--pit-muted-foreground);
+}
+
+.pit-interaction-title {
+ margin: 0;
+ font-size: var(--pit-text-sm);
+ font-weight: 600;
+}
+
+.pit-interaction-controls {
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-3);
+}
+
+.pit-field {
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-1);
+ min-width: 0;
+}
+
+.pit-field-label {
+ font-size: var(--pit-text-sm);
+ font-weight: 500;
+}
+
+.pit-option-row {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ min-height: var(--pit-touch-target);
+}
+
+.pit-input,
+.pit-textarea {
+ width: 100%;
+ min-height: var(--pit-touch-target);
+ padding: var(--pit-space-2);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface-sunken);
+ color: var(--pit-foreground);
+}
+
+.pit-textarea {
+ min-height: 4rem;
+ resize: vertical;
+}
+
+.pit-button-row {
+ display: flex;
+ flex-wrap: wrap;
+ gap: var(--pit-space-2);
+}
+
+.pit-summary-list {
+ margin: 0;
+ padding-left: var(--pit-space-5);
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-quote {
+ margin: 0;
+ padding-left: var(--pit-space-3);
+ border-left: 2px solid var(--pit-border-strong);
+ color: var(--pit-muted-foreground);
+}
+
+.pit-segmented-verdicts {
+ display: flex;
+ flex-wrap: wrap;
+ gap: var(--pit-space-2);
+ align-items: center;
+}
+
+/* ------------------------------------------------------------- composer */
+
+.pit-composer {
+ border-top: 1px solid var(--pit-border);
+ background: var(--pit-surface);
+ padding: var(--pit-space-3) var(--pit-space-4)
+ calc(var(--pit-space-3) + env(safe-area-inset-bottom, 0px));
+}
+
+.pit-composer-inner {
+ width: 100%;
+ max-width: var(--pit-thread-measure);
+ margin: 0 auto;
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-2);
+ min-width: 0;
+}
+
+.pit-composer-row {
+ display: flex;
+ gap: var(--pit-space-2);
+ align-items: flex-end;
+ min-width: 0;
+}
+
+.pit-composer-input {
+ flex: 1;
+ min-width: 0;
+ min-height: var(--pit-touch-target);
+ padding: var(--pit-space-2);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-md);
+ background: var(--pit-surface-sunken);
+ color: var(--pit-foreground);
+ resize: none;
+}
+
+.pit-composer-helper {
+ font-size: var(--pit-text-xs);
+ color: var(--pit-muted-foreground);
+}
+
+.pit-composer-reason {
+ font-size: var(--pit-text-xs);
+ color: var(--pit-warning);
+}
+
+.pit-banner {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ padding: var(--pit-space-2) var(--pit-space-4);
+ background: var(--pit-warning-surface);
+ border-bottom: 1px solid var(--pit-warning);
+ color: var(--pit-foreground);
+ font-size: var(--pit-text-xs);
+ animation: pit-slide var(--pit-motion) ease-out;
+}
+
+.pit-banner[data-tone="success"] {
+ background: var(--pit-success-surface);
+ border-bottom-color: var(--pit-success);
+}
+
+.pit-banner[data-tone="danger"] {
+ background: var(--pit-danger-surface);
+ border-bottom-color: var(--pit-danger);
+}
+
+.pit-eval-summary {
+ border-block: 1px solid color-mix(in oklch, var(--pit-success) 45%, var(--pit-border));
+ background: color-mix(in oklch, var(--pit-success) 10%, var(--pit-surface-sunken));
+ padding: 10px 20px;
+}
+
+.pit-eval-summary[data-passed="false"] {
+ border-color: color-mix(in oklch, var(--pit-danger) 45%, var(--pit-border));
+ background: color-mix(in oklch, var(--pit-danger) 10%, var(--pit-surface-sunken));
+}
+
+.pit-eval-summary-head,
+.pit-eval-checks {
+ display: flex;
+ align-items: center;
+ flex-wrap: wrap;
+ gap: 8px 14px;
+}
+
+.pit-eval-summary-head code { color: var(--pit-muted); }
+.pit-eval-checks { margin-top: 8px; }
+.pit-eval-checks span { padding: 3px 8px; border-radius: 999px; background: var(--pit-surface-raised); }
+.pit-eval-checks span[data-passed="true"] { color: var(--pit-success); }
+.pit-eval-checks span[data-passed="false"] { color: var(--pit-danger); }
+
+.pit-eval-run-facts {
+ display: grid;
+ grid-template-columns: repeat(5, minmax(0, 1fr));
+ gap: 8px;
+ margin: 10px 0 0;
+}
+.pit-eval-run-facts div { min-width: 0; }
+.pit-eval-run-facts dt { color: var(--pit-muted-foreground); font-size: var(--pit-text-xs); }
+.pit-eval-run-facts dd { margin: 2px 0 0; overflow: hidden; text-overflow: ellipsis; white-space: nowrap; font-family: var(--pit-font-mono); font-size: var(--pit-text-xs); }
+.pit-eval-boundary { display: grid; gap: 3px; margin: 14px 0; padding: 10px 12px; border-block: 1px solid var(--pit-border-strong); background: var(--pit-surface-sunken); }
+.pit-eval-boundary[data-phase="execution"] { display: flex; margin: 4px 0; padding: 7px 0; border: 0; border-top: 2px solid var(--pit-live); background: transparent; color: var(--pit-live); }
+.pit-eval-boundary[data-phase="post-run"] { border-color: var(--pit-accent); }
+.pit-eval-boundary span { color: var(--pit-muted-foreground); font-family: var(--pit-font-mono); font-size: var(--pit-text-xs); overflow-wrap: anywhere; }
+.pit-eval-item { display: contents; }
+.pit-inline-assertions { display: grid; gap: 6px; margin: 6px 0 12px; }
+.pit-inline-assertions > button { display: flex; align-items: baseline; gap: 10px; width: 100%; padding: 7px 10px; border: 0; border-left: 3px solid var(--pit-danger); background: var(--pit-danger-surface); color: inherit; text-align: left; cursor: pointer; font-size: var(--pit-text-xs); }
+.pit-inline-assertions > button[data-passed="true"] { border-left-color: var(--pit-success); background: var(--pit-success-surface); }
+.pit-inline-assertions > button strong { color: var(--pit-danger); white-space: nowrap; }
+.pit-inline-assertions > button[data-passed="true"] strong { color: var(--pit-success); }
+.pit-inline-assertions > button span { color: var(--pit-muted-foreground); }
+.pit-inline-assertions > button:hover { filter: brightness(1.12); }
+.pit-eval-inspector .pit-eval-run-facts { grid-template-columns: minmax(0, 1fr); }
+.pit-eval-inspector .pit-eval-run-facts dd { white-space: normal; overflow-wrap: anywhere; }
+.pit-eval-check-dialog { width: min(42rem, 100%); }
+.pit-eval-check-dialog .pit-code { max-height: 22rem; overflow: auto; }
+@media(max-width:900px){.pit-eval-run-facts{grid-template-columns:repeat(2,minmax(0,1fr))}}
+
+@keyframes pit-slide {
+ from {
+ transform: translateY(-0.5rem);
+ opacity: 0;
+ }
+ to {
+ transform: translateY(0);
+ opacity: 1;
+ }
+}
+
+.pit-jump-pill {
+ position: absolute;
+ left: 50%;
+ bottom: var(--pit-space-4);
+ transform: translateX(-50%);
+ z-index: 4;
+}
+
+.pit-main-inner {
+ position: relative;
+ display: flex;
+ flex-direction: column;
+ flex: 1;
+ min-height: 0;
+ min-width: 0;
+}
+
+/* -------------------------------------------------------- evidence panel */
+
+.pit-splitter {
+ position: relative;
+ width: var(--pit-space-2);
+ flex: none;
+ cursor: col-resize;
+ touch-action: none;
+ background: var(--pit-surface-sunken);
+ border: 0;
+ padding: 0;
+}
+
+.pit-splitter::after {
+ content: "";
+ position: absolute;
+ inset: 0 auto 0 50%;
+ width: 1px;
+ background: var(--pit-border-strong);
+}
+
+.pit-splitter:hover::after,
+.pit-splitter:focus-visible::after {
+ width: var(--pit-space-1);
+ transform: translateX(-50%);
+ background: var(--pit-accent);
+}
+
+.pit-panel {
+ flex: none;
+ display: flex;
+ flex-direction: column;
+ min-width: 0;
+ border-left: 1px solid var(--pit-border);
+ background: var(--pit-surface-sunken);
+ overflow-y: auto;
+}
+
+.pit-panel-chrome {
+ position: sticky;
+ top: 0;
+ background: var(--pit-surface-sunken);
+ border-bottom: 1px solid var(--pit-border);
+ padding: var(--pit-space-2) var(--pit-space-3);
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--pit-space-2);
+ z-index: 3;
+}
+
+.pit-panel-kicker,
+.pit-devtools-brand {
+ display: inline-flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ color: var(--pit-muted-foreground);
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ font-weight: 600;
+ text-transform: uppercase;
+ letter-spacing: 0.04em;
+}
+
+.pit-panel-title {
+ margin: 0;
+ font-size: var(--pit-text-sm);
+ font-weight: 600;
+}
+
+.pit-section {
+ border-bottom: 1px solid var(--pit-border);
+}
+
+.pit-section-button {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ width: 100%;
+ min-height: var(--pit-touch-target);
+ padding: var(--pit-space-2) var(--pit-space-3);
+ background: none;
+ border: 0;
+ color: inherit;
+ text-align: left;
+ cursor: pointer;
+ font-size: var(--pit-text-sm);
+ font-weight: 500;
+}
+
+.pit-section-count {
+ margin-left: auto;
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ color: var(--pit-muted-foreground);
+}
+
+.pit-section-body {
+ padding: 0 var(--pit-space-3) var(--pit-space-3);
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-2);
+ min-width: 0;
+}
+
+.pit-record {
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface);
+ padding: var(--pit-space-2);
+ font-size: var(--pit-text-xs);
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-1);
+ min-width: 0;
+}
+
+.pit-record[data-highlighted="true"] {
+ border-color: var(--pit-accent);
+}
+
+.pit-record[data-allowed="false"] {
+ border-left: 3px solid var(--pit-danger);
+}
+
+.pit-record-title {
+ font-family: var(--pit-font-mono);
+ overflow-wrap: anywhere;
+}
+
+.pit-record-detail {
+ color: var(--pit-muted-foreground);
+ overflow-wrap: anywhere;
+}
+
+.pit-runner-event,
+.pit-runner-group {
+ min-width: 0;
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-runner-event-summary,
+.pit-runner-group-summary {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ min-height: var(--pit-touch-target);
+ padding: var(--pit-space-2);
+ cursor: pointer;
+ list-style: none;
+ min-width: 0;
+}
+
+.pit-runner-event-summary::-webkit-details-marker,
+.pit-runner-group-summary::-webkit-details-marker {
+ display: none;
+}
+
+.pit-runner-event-summary .pit-record-detail,
+.pit-runner-group-summary .pit-record-detail {
+ flex: 1;
+ min-width: 0;
+}
+
+.pit-runner-event[open] > .pit-runner-event-summary .pit-activity-caret,
+.pit-runner-group[open] > .pit-runner-group-summary .pit-activity-caret {
+ transform: rotate(90deg);
+}
+
+.pit-runner-event-details {
+ display: grid;
+ gap: var(--pit-space-2);
+ margin: 0;
+ padding: 0 var(--pit-space-2) var(--pit-space-2);
+}
+
+.pit-runner-event-details > div {
+ display: grid;
+ grid-template-columns: minmax(0, 7rem) minmax(0, 1fr);
+ gap: var(--pit-space-2);
+}
+
+.pit-runner-event-details dt,
+.pit-runner-event-details dd {
+ margin: 0;
+ overflow-wrap: anywhere;
+}
+
+.pit-runner-event-details dt {
+ color: var(--pit-faint-foreground);
+}
+
+.pit-runner-event-details dd {
+ color: var(--pit-muted-foreground);
+ font-family: var(--pit-font-mono);
+}
+
+.pit-runner-group-events {
+ display: grid;
+ gap: var(--pit-space-2);
+ padding: 0 var(--pit-space-2) var(--pit-space-2);
+}
+
+.pit-runner-group-events .pit-runner-event {
+ background: var(--pit-surface-sunken);
+}
+
+.pit-tool-groups {
+ display: grid;
+ gap: var(--pit-space-5);
+ padding: var(--pit-space-3);
+}
+
+.pit-tool-group h4 {
+ margin: 0;
+}
+
+.pit-tool-group-toggle {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ width: 100%;
+ min-height: var(--pit-touch-target);
+ padding: 0;
+ border: 0;
+ border-bottom: 1px solid var(--pit-border);
+ background: transparent;
+ color: var(--pit-foreground);
+ font-size: var(--pit-text-sm);
+ font-weight: 700;
+ text-align: left;
+ cursor: pointer;
+}
+
+.pit-tool-group[data-group="control_plane_owned"] .pit-tool-group-toggle {
+ color: var(--pit-faint-foreground);
+}
+
+.pit-tool-group-rows {
+ display: grid;
+ padding-top: var(--pit-space-2);
+}
+
+.pit-tool-row {
+ display: grid;
+ grid-template-columns: minmax(9rem, 0.55fr) minmax(0, 1fr) auto auto;
+ gap: var(--pit-space-2);
+ align-items: center;
+ width: 100%;
+ min-height: var(--pit-touch-target);
+ padding: var(--pit-space-2);
+ border: 0;
+ border-bottom: 1px solid var(--pit-border);
+ background: transparent;
+ color: inherit;
+ font-size: var(--pit-text-xs);
+ text-align: left;
+ min-width: 0;
+ cursor: pointer;
+}
+
+.pit-tool-row:hover,
+.pit-tool-row:focus-visible {
+ background: var(--pit-accent-surface);
+}
+
+.pit-tool-row[data-disposition="control_plane_owned"] {
+ color: var(--pit-faint-foreground);
+}
+
+.pit-grant {
+ font-family: var(--pit-font-mono);
+ color: var(--pit-muted-foreground);
+ overflow-wrap: anywhere;
+}
+
+.pit-tool-row-description {
+ color: var(--pit-muted-foreground);
+ overflow: hidden;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.pit-tool-dialog {
+ width: min(74rem, calc(100vw - (2 * var(--pit-space-6))));
+ max-height: calc(100vh - var(--pit-space-6));
+ overflow: auto;
+}
+
+.pit-tool-dialog[data-invokable="false"] {
+ width: min(44rem, calc(100vw - (2 * var(--pit-space-6))));
+}
+
+.pit-tool-dialog-head {
+ display: flex;
+ align-items: flex-start;
+ justify-content: space-between;
+ gap: var(--pit-space-3);
+}
+
+.pit-tool-dialog-head h2,
+.pit-tool-schema-title {
+ margin: 0;
+}
+
+.pit-tool-dialog-grid {
+ display: grid;
+ grid-template-columns: minmax(20rem, 0.9fr) minmax(24rem, 1.1fr);
+ align-items: start;
+ gap: var(--pit-space-5);
+ min-width: 0;
+}
+
+.pit-tool-dialog-grid[data-invokable="false"] {
+ grid-template-columns: minmax(0, 1fr);
+}
+
+.pit-tool-reference {
+ display: flex;
+ min-width: 0;
+ flex-direction: column;
+ gap: var(--pit-space-3);
+}
+
+.pit-tool-reference > p {
+ margin: 0;
+}
+
+.pit-tool-operation {
+ align-self: flex-start;
+ padding: var(--pit-space-1) var(--pit-space-2);
+ border: 1px solid var(--pit-live);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-live-surface);
+ color: var(--pit-foreground);
+}
+
+.pit-tool-facts {
+ display: grid;
+ grid-template-columns: repeat(2, minmax(0, 1fr));
+ gap: var(--pit-space-3);
+ margin: 0;
+}
+
+.pit-tool-facts div {
+ padding: var(--pit-space-2);
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-sm);
+}
+
+.pit-tool-facts dt {
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-tool-facts dd {
+ margin: var(--pit-space-1) 0 0;
+ overflow-wrap: anywhere;
+}
+
+.pit-tool-tester {
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-2);
+ min-width: 0;
+ padding-left: var(--pit-space-5);
+ border-left: 1px solid var(--pit-border-strong);
+}
+
+.pit-tool-tester > p {
+ margin: 0;
+}
+
+.pit-tool-tester > .pit-code {
+ max-height: 24rem;
+ overflow: auto;
+}
+
+.pit-tool-test-input {
+ box-sizing: border-box;
+ width: 100%;
+ min-height: 10rem;
+ resize: vertical;
+ padding: var(--pit-space-3);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-background);
+ color: var(--pit-foreground);
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+}
+
+@media (max-width: 900px) {
+ .pit-tool-dialog {
+ width: min(44rem, calc(100vw - (2 * var(--pit-space-3))));
+ }
+
+ .pit-tool-dialog-grid {
+ grid-template-columns: minmax(0, 1fr);
+ }
+
+ .pit-tool-tester {
+ padding-top: var(--pit-space-4);
+ padding-left: 0;
+ border-top: 1px solid var(--pit-border-strong);
+ border-left: 0;
+ }
+}
+
+.pit-verdict {
+ display: inline-flex;
+ align-items: center;
+ gap: var(--pit-space-1);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-verdict[data-verdict="pass"] {
+ color: var(--pit-success);
+}
+.pit-verdict[data-verdict="fail"] {
+ color: var(--pit-danger);
+}
+.pit-verdict[data-verdict="intentional_gap"] {
+ color: var(--pit-neutral);
+}
+
+.pit-diff-row {
+ display: grid;
+ grid-template-columns: minmax(0, 1fr);
+ gap: var(--pit-space-1);
+}
+
+/* ------------------------------------------------------------- dialog */
+
+.pit-dialog-backdrop {
+ position: fixed;
+ inset: 0;
+ z-index: 20;
+ display: flex;
+ align-items: center;
+ justify-content: center;
+ padding: var(--pit-space-4);
+ background: var(--pit-scrim);
+}
+
+.pit-dialog {
+ width: min(28rem, 100%);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-lg);
+ background: var(--pit-surface);
+ padding: var(--pit-space-5);
+ display: flex;
+ flex-direction: column;
+ gap: var(--pit-space-3);
+}
+
+.pit-dialog-title {
+ margin: 0;
+ font-size: var(--pit-text-base);
+ font-weight: 600;
+}
+
+.pit-menu {
+ position: relative;
+}
+
+.pit-menu-list {
+ position: absolute;
+ right: 0;
+ top: calc(100% + var(--pit-space-1));
+ z-index: 10;
+ min-width: 12rem;
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-md);
+ background: var(--pit-surface-raised);
+ box-shadow: var(--pit-shadow-sm);
+ padding: var(--pit-space-1);
+ display: flex;
+ flex-direction: column;
+}
+
+.pit-menu-item {
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-3);
+ background: none;
+ border: 0;
+ border-radius: var(--pit-radius-sm);
+ color: inherit;
+ text-align: left;
+ cursor: pointer;
+ font-size: var(--pit-text-sm);
+}
+
+.pit-menu-item:hover {
+ background: var(--pit-surface);
+}
+
+/* -------------------------------------------------------- replay strip */
+
+.pit-replay-strip {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ padding: var(--pit-space-2) var(--pit-space-4);
+ border-bottom: 1px solid var(--pit-border);
+ background: var(--pit-surface-raised);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-progress {
+ flex: 1;
+ height: 0.25rem;
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-border);
+ overflow: hidden;
+ min-width: 0;
+}
+
+.pit-progress-fill {
+ height: 100%;
+ background: var(--pit-accent);
+}
+
+/* --------------------------------------------------------- mobile rules */
+
+.pit-segmented {
+ display: none;
+}
+
+/*
+ * §2.2: on mobile the header condenses to identifier + status + `⋯`, and
+ * Scenario/Replay/Reset move inside the overflow menu. Stop stays outside it.
+ */
+.pit-mobile-only {
+ display: none;
+}
+
+@media (max-width: 1100px) {
+ .pit-panel[data-layout="overlay"] {
+ position: fixed;
+ top: 0;
+ right: 0;
+ bottom: 0;
+ z-index: 15;
+ width: min(28rem, 100vw);
+ box-shadow: var(--pit-shadow-sm);
+ }
+
+ .pit-splitter {
+ display: none;
+ }
+}
+
+@media (max-width: 767px) {
+ .pit-app {
+ padding-left: 0;
+ }
+
+ .pit-surface-nav {
+ position: static;
+ width: auto;
+ flex: none;
+ padding: var(--pit-space-1) var(--pit-space-2);
+ border-right: 0;
+ border-bottom: 1px solid var(--pit-border);
+ }
+
+ .pit-sidebar-brand,
+ .pit-session-history {
+ display: none;
+ }
+
+ .pit-sidebar-primary {
+ display: flex;
+ gap: var(--pit-space-1);
+ padding: 0;
+ border: 0;
+ }
+
+ .pit-surface-link {
+ width: auto;
+ }
+
+ .pit-desktop-only {
+ display: none;
+ }
+
+ .pit-mobile-only {
+ display: block;
+ }
+
+ .pit-segmented {
+ display: flex;
+ gap: var(--pit-space-1);
+ padding: var(--pit-space-1);
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-md);
+ background: var(--pit-surface-sunken);
+ }
+
+ .pit-segment {
+ flex: 1;
+ min-height: var(--pit-touch-target);
+ background: none;
+ border: 0;
+ border-radius: var(--pit-radius-sm);
+ color: var(--pit-muted-foreground);
+ cursor: pointer;
+ font-size: var(--pit-text-xs);
+ display: inline-flex;
+ align-items: center;
+ justify-content: center;
+ gap: var(--pit-space-1);
+ }
+
+ .pit-segment[aria-selected="true"] {
+ background: var(--pit-surface-raised);
+ color: var(--pit-foreground);
+ font-weight: 600;
+ }
+
+ .pit-segment-badge {
+ min-width: 1.25rem;
+ padding: 0 var(--pit-space-1);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-danger-surface);
+ border: 1px solid var(--pit-danger);
+ color: var(--pit-foreground);
+ }
+
+ .pit-panel[data-layout="segment"] {
+ position: static;
+ width: 100%;
+ border-left: 0;
+ box-shadow: none;
+ }
+
+ .pit-link-button {
+ display: inline-flex;
+ align-items: center;
+ min-height: var(--pit-touch-target);
+ }
+
+ .pit-title {
+ font-size: var(--pit-text-base);
+ }
+
+ .pit-thread-scroll {
+ padding: var(--pit-space-4) var(--pit-space-3);
+ }
+
+ .pit-message[data-role="user"] {
+ max-width: 92%;
+ }
+
+ .pit-activity-payloads {
+ grid-template-columns: minmax(0, 1fr);
+ }
+
+ .pit-activity-description {
+ white-space: normal;
+ }
+
+ /*
+ * Three transport controls plus a progress bar do not fit on one 390px row —
+ * squeezing them leaves the bar as an unreadable stub. Wrap so the bar keeps
+ * the full width on its own line, below the label and buttons.
+ */
+ .pit-replay-strip {
+ flex-wrap: wrap;
+ padding: var(--pit-space-2) var(--pit-space-3);
+ }
+
+ .pit-replay-strip .pit-progress {
+ order: 1;
+ flex-basis: 100%;
+ }
+}
+
+.pit-devtools {
+ padding: 0;
+ border-bottom: 1px solid var(--pit-border);
+ background: var(--pit-surface-sunken);
+}
+
+.pit-devtools-loading,
+.pit-evidence-heading {
+ margin: var(--pit-space-3);
+}
+
+.pit-devtools-toolbar,
+.pit-devtools-tabs {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ flex-wrap: wrap;
+}
+
+.pit-devtools-toolbar {
+ padding: var(--pit-space-3);
+}
+
+.pit-devtools-tabs {
+ gap: 0;
+ flex-wrap: nowrap;
+ overflow-x: auto;
+ margin: 0;
+ padding: 0 var(--pit-space-2);
+ border-top: 1px solid var(--pit-border);
+ border-bottom: 1px solid var(--pit-border-strong);
+ background: var(--pit-surface);
+}
+
+.pit-input {
+ min-height: var(--pit-touch-target);
+ min-width: 0;
+ flex: 1;
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface);
+ padding: 0 var(--pit-space-3);
+}
+
+.pit-tab {
+ display: inline-flex;
+ align-items: center;
+ justify-content: center;
+ flex: 0 0 auto;
+ min-height: var(--pit-touch-target);
+ line-height: 1;
+ gap: var(--pit-space-2);
+ border: 0;
+ border-bottom: var(--pit-space-1) solid transparent;
+ border-radius: 0;
+ background: transparent;
+ padding: 0 var(--pit-space-3);
+ color: var(--pit-muted-foreground);
+ white-space: nowrap;
+ cursor: pointer;
+}
+
+.pit-tab[aria-selected="true"] {
+ border-bottom-color: var(--pit-accent);
+ background: var(--pit-accent-surface);
+ color: var(--pit-foreground);
+ font-weight: 600;
+}
+
+.pit-tab-glyph {
+ display: inline-flex;
+ align-items: center;
+ justify-content: center;
+ width: var(--pit-space-4);
+ height: var(--pit-space-4);
+ color: var(--pit-live);
+}
+
+.pit-devtools-list {
+ display: grid;
+ gap: var(--pit-space-2);
+ padding: var(--pit-space-3);
+}
+
+.pit-devtools-pane {
+ padding: var(--pit-space-3);
+}
+
+.pit-devtools-event {
+ display: grid;
+ grid-template-columns: auto minmax(0, 1fr) auto;
+ gap: var(--pit-space-2);
+ align-items: center;
+ width: 100%;
+ min-height: var(--pit-touch-target);
+ border: 1px solid var(--pit-border);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface);
+ padding: var(--pit-space-2);
+ text-align: left;
+ cursor: pointer;
+}
+
+.pit-devtools-event[data-selected="true"] {
+ border-color: var(--pit-accent);
+ background: var(--pit-accent-surface);
+}
+
+.pit-json-node {
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ overflow-wrap: anywhere;
+}
+
+.pit-json-node summary {
+ cursor: pointer;
+ min-height: var(--pit-space-6);
+}
+
+.pit-json-children {
+ margin-left: var(--pit-space-3);
+ padding-left: var(--pit-space-2);
+ border-left: 1px solid var(--pit-border);
+}
+
+.pit-json-row {
+ display: grid;
+ grid-template-columns: minmax(5rem, auto) minmax(0, 1fr);
+ gap: var(--pit-space-2);
+}
+
+.pit-json-key {
+ color: var(--pit-live);
+}
+
+.pit-json-value {
+ white-space: pre-wrap;
+}
+
+.pit-diff-row {
+ display: grid;
+ gap: var(--pit-space-1);
+ padding: var(--pit-space-2) 0;
+ border-bottom: 1px solid var(--pit-border);
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ overflow-wrap: anywhere;
+}
+
+.pit-evidence-filter {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--pit-space-3);
+ padding: var(--pit-space-3);
+ border-top: 1px solid var(--pit-border);
+ border-bottom: 1px solid var(--pit-border);
+ background: var(--pit-surface);
+}
+
+.pit-evidence-filter .pit-select {
+ flex: 1;
+}
+
+.pit-document-browser {
+ display: grid;
+ grid-template-columns: minmax(10rem, 0.35fr) minmax(0, 1fr);
+ min-height: 20rem;
+}
+
+.pit-document-sidebar {
+ border-right: 1px solid var(--pit-border);
+ background: var(--pit-surface);
+}
+
+.pit-document-sidebar button {
+ display: grid;
+ width: 100%;
+ gap: var(--pit-space-1);
+ padding: var(--pit-space-3);
+ border: 0;
+ border-bottom: 1px solid var(--pit-border);
+ background: transparent;
+ color: inherit;
+ text-align: left;
+ cursor: pointer;
+}
+
+.pit-document-sidebar button[data-selected="true"] {
+ border-left: var(--pit-space-1) solid var(--pit-accent);
+ background: var(--pit-accent-surface);
+}
+
+.pit-document-sidebar span,
+.pit-document-sidebar small {
+ color: var(--pit-muted-foreground);
+}
+
+.pit-document-viewer {
+ min-width: 0;
+ background: var(--pit-surface-sunken);
+}
+
+.pit-document-viewer header {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--pit-space-3);
+ padding: var(--pit-space-3);
+ border-bottom: 1px solid var(--pit-border);
+}
+
+.pit-document-viewer h4,
+.pit-document-summary {
+ margin: 0;
+}
+
+.pit-document-summary {
+ padding: var(--pit-space-2) var(--pit-space-3);
+ color: var(--pit-muted-foreground);
+ border-bottom: 1px solid var(--pit-border);
+}
+
+.pit-document-body {
+ margin: 0;
+ padding: var(--pit-space-4);
+ white-space: pre-wrap;
+ overflow-wrap: anywhere;
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ line-height: var(--pit-leading-relaxed);
+}
+
+.pit-devtools-empty {
+ display: grid;
+ place-items: center;
+ padding: var(--pit-space-6);
+ color: var(--pit-muted-foreground);
+ text-align: center;
+}
+
+.pit-devtools-empty > span {
+ font-size: var(--pit-text-xl);
+}
+
+.pit-live-activity {
+ display: flex;
+ align-items: center;
+ gap: var(--pit-space-2);
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-4);
+ border-bottom: 1px solid var(--pit-live);
+ background: var(--pit-live-surface);
+ color: var(--pit-foreground);
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-live-activity-pulse {
+ width: var(--pit-space-2);
+ height: var(--pit-space-2);
+ border-radius: 50%;
+ background: var(--pit-live);
+ animation: pit-pulse 0.8s ease-in-out infinite;
+}
+
+.pit-live-activity-tail {
+ margin-left: auto;
+ color: var(--pit-live);
+ text-transform: uppercase;
+}
+
+.pit-harness-picker {
+ display: flex;
+ align-items: end;
+ gap: var(--pit-space-3);
+ padding: var(--pit-space-2) var(--pit-space-4);
+ border-bottom: 1px solid var(--pit-border);
+ background: var(--pit-surface-sunken);
+}
+
+.pit-harness-picker label {
+ display: grid;
+ gap: var(--pit-space-1);
+ color: var(--pit-muted-foreground);
+ font-size: var(--pit-text-xs);
+}
+
+.pit-harness-picker select,
+.pit-harness-picker input {
+ box-sizing: border-box;
+ min-height: var(--pit-touch-target);
+ padding: 0 var(--pit-space-2);
+ border: 1px solid var(--pit-border-strong);
+ border-radius: var(--pit-radius-sm);
+ background: var(--pit-surface-raised);
+}
+
+.pit-harness-model {
+ flex: 1;
+ max-width: 38rem;
+}
+
+.pit-harness-active {
+ align-self: center;
+ min-width: 0;
+ overflow: hidden;
+ color: var(--pit-faint-foreground);
+ font-family: var(--pit-font-mono);
+ font-size: var(--pit-text-xs);
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.pit-harness-notice {
+ flex-basis: 100%;
+ margin: 0;
+ color: var(--pit-faint-foreground);
+ font-size: var(--pit-text-xs);
+ line-height: 1.4;
+}
+
+.pit-managed-session-actions {
+ display: flex;
+ align-items: end;
+ gap: 8px;
+ flex-wrap: wrap;
+ padding-left: 8px;
+ border-left: 1px solid var(--pit-border);
+}
+
+.pit-managed-session-actions label {
+ display: grid;
+ gap: 4px;
+}
+
+.pit-managed-session-actions label > span {
+ color: var(--pit-muted-foreground);
+ font-size: 11px;
+}
+
+.pit-managed-session-actions input {
+ width: 118px;
+}
+
+@media (max-width: 767px) {
+ .pit-harness-picker {
+ align-items: stretch;
+ flex-wrap: wrap;
+ }
+
+ .pit-harness-model {
+ min-width: 12rem;
+ }
+
+ .pit-harness-active {
+ flex-basis: 100%;
+ }
+
+ .pit-header-primary {
+ align-items: flex-start;
+ }
+
+ .pit-header-title-row {
+ flex: 1;
+ }
+
+ .pit-header-context .pit-priority,
+ .pit-header-context .pit-assignee,
+ .pit-header-context .pit-run-state,
+ .pit-chip-row .pit-chip:not([data-testid="agent-chip"]) {
+ display: none;
+ }
+
+ .pit-document-browser {
+ grid-template-columns: minmax(0, 1fr);
+ }
+
+ .pit-document-sidebar {
+ border-right: 0;
+ border-bottom: 1px solid var(--pit-border);
+ }
+
+ .pit-tool-row {
+ grid-template-columns: minmax(0, 1fr) auto;
+ }
+
+ .pit-tool-row-description {
+ grid-column: 1 / -1;
+ white-space: normal;
+ }
+
+ .pit-tool-facts {
+ grid-template-columns: minmax(0, 1fr);
+ }
+}
+
+.pit-diff-row del {
+ color: var(--pit-danger);
+}
+
+.pit-diff-row ins {
+ color: var(--pit-success);
+ text-decoration: none;
+}
+
+@media (prefers-reduced-motion: reduce) {
+ *,
+ *::before,
+ *::after {
+ animation-duration: 0.001ms !important;
+ animation-iteration-count: 1 !important;
+ transition-duration: 0.001ms !important;
+ scroll-behavior: auto !important;
+ }
+
+ .pit-process-dot[data-attached="true"] {
+ animation: none;
+ }
+}
+
+/* Capture mode: the screenshot recorder freezes every animated affordance. */
+:root[data-capture="true"] *,
+:root[data-capture="true"] *::before,
+:root[data-capture="true"] *::after {
+ animation: none !important;
+ transition: none !important;
+ caret-color: transparent !important;
+ scroll-behavior: auto !important;
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/live-client.ts b/packages/paperclip-runner/devtools/issue-thread/src/live-client.ts
new file mode 100644
index 0000000000..0b9f086b2f
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/live-client.ts
@@ -0,0 +1,260 @@
+import type { CapabilityIssueThreadSnapshot } from "../../../src/issue-thread/types";
+import type { CapabilityDevtoolsSnapshot } from "../../../src/devtools";
+import type { CapabilityJsonValue } from "../../../src/mock-core/capability-control-plane-types";
+import {
+ CAPABILITY_TURN_STREAM_ACCEPT,
+ CapabilityTurnStreamError,
+ readCapabilityTurnStream,
+} from "../../../src/live/turn-stream";
+
+/**
+ * Browser client for the package session server.
+ *
+ * Every response is a server-projected `CapabilityIssueThreadSnapshot`. The client
+ * posts intents and renders what comes back; it never patches the snapshot
+ * locally, which is what keeps policy and state authority on the server
+ * (contract §11).
+ *
+ * A turn is the one route that answers with many of those projections instead
+ * of one: `send` consumes NDJSON frames as the provider produces them, hands
+ * each interim view to `onFrame`, and resolves with the settled payload. The
+ * authority rule is unchanged — every frame is still a server projection, and
+ * the settled one is final.
+ */
+
+const BASE = "/api/capability/ui";
+
+/**
+ * Every route is session-scoped and mediated by the per-browser capability
+ * cookie the server sets, so the cookie has to ride along. Stated rather than
+ * left to the default so a future change to this client cannot silently drop
+ * the capability and fall back to session-id-only access (track 7U).
+ */
+const CREDENTIALS = "same-origin" as const;
+
+export interface CapabilityCleanRoomIdentity {
+ token: string;
+ sequence: number;
+ companyId: string;
+ actorId: string;
+ taskId: string;
+ identifier: string;
+}
+
+export interface CapabilityHarnessConfiguration {
+ provider: "codex" | "opencode" | "acpx";
+ model: string | null;
+ acpxAgent?: "claude" | "codex";
+ lifecyclePolicy:
+ | { mode: "per_turn"; idleTimeoutMs: null }
+ | { mode: "warm"; idleTimeoutMs: number };
+}
+
+export interface CapabilityLiveResponse {
+ sessionId: string;
+ view: CapabilityIssueThreadSnapshot;
+ surface?: "issue" | "cleanroom";
+ /** Present on the clean-room surface so the UI can show which tenant is live. */
+ identity?: CapabilityCleanRoomIdentity;
+ limits?: { maxTurns: number; maxMessageBytes: number };
+ turns?: number;
+ configuration?: CapabilityHarnessConfiguration;
+ runtime?: {
+ providerSessionId: string | null;
+ driverSessionId?: string | null;
+ runnerPid: number | null;
+ providerPid: number | null;
+ sidecarPid?: number | null;
+ agentPid?: number | null;
+ providerVersion?: string | null;
+ agentServerVersion?: string | null;
+ agentRuntimeVersion?: string | null;
+ acpProtocolVersion?: number | null;
+ executionKind: "local_process" | "remote_service";
+ status: string;
+ };
+}
+
+export interface CapabilityToolTestResponse extends CapabilityLiveResponse {
+ toolResult: CapabilityJsonValue;
+ toolTurnId: string;
+}
+
+/**
+ * The server names its own failures. Surfacing the code lets the clean room say
+ * "this chat hit its turn limit" instead of "500", and — more importantly —
+ * lets a failure to start real Codex read as a failure rather than quietly
+ * degrading into a fixture.
+ */
+export class CapabilityLiveError extends Error {
+ constructor(
+ readonly code: string,
+ message: string,
+ ) {
+ super(message);
+ this.name = "CapabilityLiveError";
+ }
+}
+
+async function readError(response: Response, path: string): Promise {
+ try {
+ const body = (await response.json()) as { error?: string; message?: string };
+ return new CapabilityLiveError(
+ body.error ?? `http_${response.status}`,
+ body.message ?? `${path} failed with ${response.status}`,
+ );
+ } catch {
+ return new CapabilityLiveError(`http_${response.status}`, `${path} failed with ${response.status}`);
+ }
+}
+
+async function post(path: string, body: unknown): Promise {
+ const response = await fetch(`${BASE}${path}`, {
+ method: "POST",
+ headers: { "content-type": "application/json" },
+ body: JSON.stringify(body),
+ credentials: CREDENTIALS,
+ });
+ if (!response.ok) throw await readError(response, path);
+ return (await response.json()) as T;
+}
+
+/** Interim projection delivered while the turn is still running. */
+export type CapabilityTurnFrameHandler = (view: CapabilityIssueThreadSnapshot) => void;
+
+export interface CapabilitySendOptions {
+ onFrame?: CapabilityTurnFrameHandler;
+ /** Abort the turn stream — used when the surface is torn down or rotated. */
+ signal?: AbortSignal;
+}
+
+async function postTurnStream(
+ path: string,
+ body: unknown,
+ options: CapabilitySendOptions,
+): Promise {
+ const response = await fetch(`${BASE}${path}`, {
+ method: "POST",
+ headers: { "content-type": "application/json", accept: CAPABILITY_TURN_STREAM_ACCEPT },
+ body: JSON.stringify(body),
+ credentials: CREDENTIALS,
+ ...(options.signal === undefined ? {} : { signal: options.signal }),
+ });
+ // Admission failures still answer with a JSON body and a real status code,
+ // because they are decided before the first frame is written.
+ if (!response.ok) throw await readError(response, path);
+ try {
+ return await readCapabilityTurnStream(
+ response,
+ (frame) => options.onFrame?.(frame.view),
+ );
+ } catch (cause) {
+ if (cause instanceof CapabilityTurnStreamError) {
+ throw new CapabilityLiveError(cause.code, cause.message);
+ }
+ throw cause;
+ }
+}
+
+export const capabilityLiveClient = {
+ async devtools(sessionId: string): Promise {
+ const response = await fetch(`${BASE}/devtools?sessionId=${encodeURIComponent(sessionId)}`, {
+ credentials: CREDENTIALS,
+ });
+ if (!response.ok) throw await readError(response, "/devtools");
+ return (await response.json()) as CapabilityDevtoolsSnapshot;
+ },
+ fork(sessionId: string, revision: number): Promise {
+ return post("/devtools/fork", { sessionId, revision });
+ },
+ invokeTool(
+ sessionId: string,
+ operationId: string,
+ input: CapabilityJsonValue,
+ ): Promise {
+ return post("/tool", { sessionId, operationId, input }) as Promise;
+ },
+ async load(sessionId: string | null): Promise {
+ const query = sessionId === null ? "" : `?sessionId=${encodeURIComponent(sessionId)}`;
+ const response = await fetch(`${BASE}/session${query}`, { credentials: CREDENTIALS });
+ if (!response.ok) throw await readError(response, "/session");
+ return (await response.json()) as CapabilityLiveResponse;
+ },
+ create(scenario: string): Promise {
+ return post("/session", { scenario });
+ },
+ /** Reconnect to a clean room, or open one when there is nothing to resume. */
+ async loadCleanRoom(sessionId: string | null, configuration: CapabilityHarnessConfiguration): Promise {
+ const query = new URLSearchParams({ provider: configuration.provider });
+ if (sessionId !== null) query.set("sessionId", sessionId);
+ if (configuration.model !== null) query.set("model", configuration.model);
+ if (configuration.acpxAgent) query.set("acpxAgent", configuration.acpxAgent);
+ query.set("lifecycleMode", configuration.lifecyclePolicy.mode);
+ if (configuration.lifecyclePolicy.mode === "warm") {
+ query.set("idleTimeoutMs", String(configuration.lifecyclePolicy.idleTimeoutMs));
+ }
+ const response = await fetch(`${BASE}/cleanroom/session?${query.toString()}`, { credentials: CREDENTIALS });
+ if (!response.ok) throw await readError(response, "/cleanroom/session");
+ return (await response.json()) as CapabilityLiveResponse;
+ },
+ /** `New chat`: retire the current room and mint a new mock tenant. */
+ newCleanRoom(sessionId: string | null, configuration: CapabilityHarnessConfiguration): Promise {
+ return post("/cleanroom/session", { ...(sessionId === null ? {} : { sessionId }), ...configuration });
+ },
+ /**
+ * Runs one turn. `onFrame` fires for every interim projection the server
+ * writes while the POST is open; the resolved value is the settled payload.
+ */
+ send(
+ sessionId: string,
+ message: string,
+ options: CapabilitySendOptions = {},
+ ): Promise {
+ return postTurnStream("/message", { sessionId, message }, options);
+ },
+ stop(sessionId: string): Promise {
+ return post("/interrupt", { sessionId });
+ },
+ reset(sessionId: string): Promise {
+ return post("/reset", { sessionId });
+ },
+ reconnect(sessionId: string): Promise {
+ return post("/reconnect", { sessionId });
+ },
+ respond(
+ sessionId: string,
+ interactionId: string,
+ outcome: string,
+ result: unknown,
+ ): Promise {
+ return post("/interaction", { sessionId, interactionId, outcome, result });
+ },
+};
+
+const SESSION_STORAGE_KEY = "paperclip-runner.capability.session";
+/**
+ * The clean room keeps its own key. Sharing one would let a scenario session id
+ * be handed to the clean-room route (and the reverse) after a refresh, which is
+ * exactly the cross-surface bleed the isolation criterion forbids.
+ */
+const CLEAN_ROOM_STORAGE_KEY = "paperclip-runner.capability.cleanroom.session";
+
+function storageKey(surface: "issue" | "cleanroom"): string {
+ return surface === "cleanroom" ? CLEAN_ROOM_STORAGE_KEY : SESSION_STORAGE_KEY;
+}
+
+export function rememberSession(sessionId: string, surface: "issue" | "cleanroom" = "issue"): void {
+ try {
+ window.localStorage.setItem(storageKey(surface), sessionId);
+ } catch {
+ // Refresh restore falls back to a fresh session when storage is blocked.
+ }
+}
+
+export function recallSession(surface: "issue" | "cleanroom" = "issue"): string | null {
+ try {
+ return window.localStorage.getItem(storageKey(surface));
+ } catch {
+ return null;
+ }
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/main.tsx b/packages/paperclip-runner/devtools/issue-thread/src/main.tsx
new file mode 100644
index 0000000000..0a62fa2ef8
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/main.tsx
@@ -0,0 +1,24 @@
+import { StrictMode } from "react";
+import { createRoot } from "react-dom/client";
+
+import { App } from "./App";
+import "./issue-thread.css";
+
+// `?capture=1` freezes animation, caret, and smooth scrolling so the
+// screenshot matrix is byte-stable across runs (contract §10.1).
+const params = new URLSearchParams(window.location.search);
+const hashQuery = window.location.hash.split("?")[1] ?? "";
+if (params.get("capture") === "1" || new URLSearchParams(hashQuery).get("capture") === "1") {
+ document.documentElement.dataset.capture = "true";
+}
+
+const root = document.getElementById("root");
+if (root === null) {
+ throw new Error("Capability issue-thread root element is missing");
+}
+
+createRoot(root).render(
+
+
+ ,
+);
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/primitives.tsx b/packages/paperclip-runner/devtools/issue-thread/src/primitives.tsx
new file mode 100644
index 0000000000..331ac8ecb2
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/primitives.tsx
@@ -0,0 +1,91 @@
+import type { ReactNode, Ref } from "react";
+
+import type { CapabilityTaskStatus } from "../../../src/mock-core/capability-control-plane-types";
+
+/**
+ * Package-local equivalents of the product StatusBadge/chip anatomy. Every
+ * state pairs color with a glyph and text so the surface stays readable
+ * without color (contract §9.5).
+ */
+
+const STATUS_COPY: Record = {
+ backlog: { label: "Backlog", glyph: "○" },
+ todo: { label: "Todo", glyph: "○" },
+ in_progress: { label: "In progress", glyph: "◐" },
+ in_review: { label: "In review", glyph: "◑" },
+ done: { label: "Done", glyph: "✓" },
+ blocked: { label: "Blocked", glyph: "✕" },
+ cancelled: { label: "Cancelled", glyph: "–" },
+};
+
+const PRIORITY_GLYPH: Record = {
+ critical: "▲▲",
+ high: "▲",
+ medium: "▬",
+ low: "▼",
+};
+
+export function StatusBadge({ status }: { status: CapabilityTaskStatus }) {
+ const copy = STATUS_COPY[status];
+ return (
+
+ {copy.glyph}
+ {copy.label}
+
+ );
+}
+
+export function PriorityIcon({ priority }: { priority: string }) {
+ return (
+
+ {PRIORITY_GLYPH[priority] ?? "▬"} {" "}
+ {priority.charAt(0).toUpperCase() + priority.slice(1)} priority
+
+ );
+}
+
+export function Chip({
+ tone,
+ children,
+ title,
+ testId,
+ /** Set to -1 so focus management can land on a chip without adding a tab stop. */
+ tabIndex,
+ chipRef,
+}: {
+ tone?: "live" | "mock" | "accent" | "success" | "danger";
+ children: ReactNode;
+ title?: string;
+ testId?: string;
+ tabIndex?: number;
+ chipRef?: Ref;
+}) {
+ return (
+
+ {children}
+
+ );
+}
+
+export function Timestamp({ value }: { value: string }) {
+ // Rendered from fixture data with a fixed locale so captures stay stable.
+ const stamp = new Date(value);
+ const hours = String(stamp.getUTCHours()).padStart(2, "0");
+ const minutes = String(stamp.getUTCMinutes()).padStart(2, "0");
+ const seconds = String(stamp.getUTCSeconds()).padStart(2, "0");
+ return (
+
+ {hours}:{minutes}:{seconds}
+
+ );
+}
+
+/** Re-exported so the thread strip and the deliverable card share one unit. */
+export { capabilityFormatBytes as formatBytes } from "../../../src/issue-thread/types";
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/route.test.ts b/packages/paperclip-runner/devtools/issue-thread/src/route.test.ts
new file mode 100644
index 0000000000..a17710690a
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/route.test.ts
@@ -0,0 +1,75 @@
+import { describe, expect, it } from "vitest";
+
+import { parseCapabilityRoute, capabilityRouteHref } from "./route";
+
+function at(hash: string, search = "") {
+ return parseCapabilityRoute({ hash, search });
+}
+
+describe("Capability issue-thread routes", () => {
+ it("keeps the existing deterministic scenario deep links intact", () => {
+ const route = at("#/issue/dp-documents?shot=document-revision&panel=state&rec=doc-1&at=3&seg=evidence");
+
+ expect(route.surface).toBe("issue");
+ expect(route.fixtureProfile).toBe("dp-documents");
+ expect(route.shot).toBe("document-revision");
+ expect(route.panel).toBe("state");
+ expect(route.record).toBe("doc-1");
+ expect(route.at).toBe(3);
+ expect(route.segment).toBe("evidence");
+ expect(route.mode).toBe("fake");
+ expect(capabilityRouteHref(route, {})).toBe(
+ "#/issue/dp-documents?shot=document-revision&panel=state&rec=doc-1&at=3&seg=evidence",
+ );
+ });
+
+ it("still infers replay mode from the replay shot and honours an explicit mode", () => {
+ expect(at("#/issue/hb-baseline?shot=replay-mode").mode).toBe("replay");
+ expect(at("#/issue/hb-baseline?mode=live").mode).toBe("live");
+ });
+});
+
+describe("Capability clean-room route", () => {
+ it("is reachable without naming a scenario", () => {
+ const route = at("#/chat");
+
+ expect(route.surface).toBe("chat");
+ expect(route.shot).toBe(null);
+ expect(route.at).toBe(null);
+ expect(capabilityRouteHref(route, {})).toBe("#/chat");
+ });
+
+ it("is always live: no URL can downgrade it to a fixture or a recording", () => {
+ for (const hash of [
+ "#/chat",
+ "#/chat?mode=fake",
+ "#/chat?mode=replay",
+ "#/chat?shot=replay-mode",
+ "#/chat?shot=thread-baseline&at=4",
+ ]) {
+ const route = at(hash);
+ expect(route.surface, hash).toBe("chat");
+ expect(route.mode, hash).toBe("live");
+ expect(route.shot, hash).toBe(null);
+ expect(route.at, hash).toBe(null);
+ }
+ expect(at("", "?mode=fake#/chat").surface).toBe("issue");
+ });
+
+ it("carries only the evidence deep-link parameters back into its own href", () => {
+ const route = at("#/chat?panel=calls&rec=call-1&seg=evidence");
+
+ expect(route.panel).toBe("calls");
+ expect(route.record).toBe("call-1");
+ expect(route.segment).toBe("evidence");
+ expect(capabilityRouteHref(route, {})).toBe("#/chat?panel=calls&rec=call-1&seg=evidence");
+ expect(capabilityRouteHref(route, { panel: null, record: null, segment: "thread" })).toBe("#/chat");
+ });
+
+ it("switches surfaces without dragging scenario state across", () => {
+ const chat = at("#/chat");
+ expect(capabilityRouteHref(chat, { surface: "issue", fixtureProfile: "ar-artifacts" })).toBe(
+ "#/issue/ar-artifacts?mode=live",
+ );
+ });
+});
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/route.ts b/packages/paperclip-runner/devtools/issue-thread/src/route.ts
new file mode 100644
index 0000000000..8e6cd9126f
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/src/route.ts
@@ -0,0 +1,98 @@
+/**
+ * Deterministic route scheme from the Capability UX contract §10.1, plus the
+ * Capability clean-room surface.
+ *
+ * `#/issue/?shot=&panel=&rec=&at=&seg=thread|evidence`
+ * `#/chat?panel=&rec=&seg=thread|evidence`
+ *
+ * Parameters are also accepted on the document query string so a capture tool
+ * can address a state without depending on hash-fragment escaping.
+ *
+ * The clean-room surface has no fixture, shot, or replay ordinal by design: it
+ * starts blank and is always a live Codex session. `mode` is pinned to `live`
+ * there rather than parsed, so no URL can talk the clean room into answering
+ * with a fixture or a recording.
+ */
+
+import type { CapabilityEvidenceSectionId } from "../../../src/issue-thread/types";
+import { CAPABILITY_EVIDENCE_SECTIONS } from "../../../src/issue-thread/types";
+import { CAPABILITY_DEFAULT_FIXTURE_PROFILE } from "../../../src/issue-thread/fixtures";
+
+export type CapabilitySurface = "issue" | "chat";
+
+export interface CapabilityRoute {
+ surface: CapabilitySurface;
+ fixtureProfile: string;
+ shot: string | null;
+ panel: CapabilityEvidenceSectionId | null;
+ record: string | null;
+ at: number | null;
+ segment: "thread" | "evidence";
+ /** `live` opts into the package session server; otherwise fixtures render. */
+ mode: "fake" | "live" | "replay";
+}
+
+const SECTION_IDS = new Set(CAPABILITY_EVIDENCE_SECTIONS.map((section) => section.id));
+
+function mergeParams(search: string, hash: string): URLSearchParams {
+ const params = new URLSearchParams(search);
+ const queryIndex = hash.indexOf("?");
+ if (queryIndex >= 0) {
+ for (const [key, value] of new URLSearchParams(hash.slice(queryIndex + 1))) {
+ params.set(key, value);
+ }
+ }
+ return params;
+}
+
+export function parseCapabilityRoute(url: {
+ search: string;
+ hash: string;
+}): CapabilityRoute {
+ const params = mergeParams(url.search, url.hash);
+ const path = url.hash.replace(/^#/, "").split("?")[0] ?? "";
+ const chat = /^\/chat\/?$/.test(path);
+ const match = /^\/issue\/([^/?]+)/.exec(path);
+ const shot = chat ? null : params.get("shot");
+ const requestedMode = params.get("mode");
+ const panel = params.get("panel");
+ const at = params.get("at");
+ const segment = params.get("seg");
+ const mode = chat
+ ? ("live" as const)
+ : requestedMode === "live" || requestedMode === "replay" || requestedMode === "fake"
+ ? requestedMode
+ : shot === "replay-mode"
+ ? "replay"
+ : "fake";
+ return {
+ surface: chat ? "chat" : "issue",
+ fixtureProfile: match?.[1] ?? CAPABILITY_DEFAULT_FIXTURE_PROFILE,
+ shot,
+ panel: panel !== null && SECTION_IDS.has(panel) ? (panel as CapabilityEvidenceSectionId) : null,
+ record: params.get("rec"),
+ at: chat || at === null || !/^\d+$/.test(at) ? null : Number.parseInt(at, 10),
+ segment: segment === "evidence" ? "evidence" : "thread",
+ mode,
+ };
+}
+
+export function capabilityRouteHref(route: CapabilityRoute, overrides: Partial): string {
+ const next = { ...route, ...overrides };
+ const params = new URLSearchParams();
+ if (next.surface === "chat") {
+ if (next.panel !== null) params.set("panel", next.panel);
+ if (next.record !== null) params.set("rec", next.record);
+ if (next.segment !== "thread") params.set("seg", next.segment);
+ const chatQuery = params.toString();
+ return `#/chat${chatQuery.length > 0 ? `?${chatQuery}` : ""}`;
+ }
+ if (next.shot !== null) params.set("shot", next.shot);
+ if (next.mode !== "fake") params.set("mode", next.mode);
+ if (next.panel !== null) params.set("panel", next.panel);
+ if (next.record !== null) params.set("rec", next.record);
+ if (next.at !== null) params.set("at", String(next.at));
+ if (next.segment !== "thread") params.set("seg", next.segment);
+ const query = params.toString();
+ return `#/issue/${next.fixtureProfile}${query.length > 0 ? `?${query}` : ""}`;
+}
diff --git a/packages/paperclip-runner/devtools/issue-thread/stream-fixture-transport.mjs b/packages/paperclip-runner/devtools/issue-thread/stream-fixture-transport.mjs
new file mode 100644
index 0000000000..3d1e70f2bc
--- /dev/null
+++ b/packages/paperclip-runner/devtools/issue-thread/stream-fixture-transport.mjs
@@ -0,0 +1,166 @@
+/**
+ * Scripted Codex transport for the browser streaming test (track 7Q).
+ *
+ * Everything below the provider is the real thing: the same package server
+ * middleware, the same `CapabilityLiveSession`, the same NDJSON turn stream, and
+ * the same built browser bundle. Only the provider is scripted, and only so the
+ * test can decide when a delta arrives — a real Codex process cannot be asked
+ * to emit exactly four deltas 220 ms apart, and a browser assertion needs that
+ * to distinguish "streaming" from "arrived all at once".
+ *
+ * This module is loaded by `vite.issue-thread-stream.config.ts` only. It is not
+ * reachable from the shipped server, the deployed surface, or any route: the
+ * production plugin takes no transport override from a request or an
+ * environment variable.
+ */
+
+const DELTA_INTERVAL_MS = 220;
+export const CAPABILITY_STREAM_FIXTURE_PRIVATE_REASONING =
+ "PRIVATE reasoning text must never reach the browser.";
+
+/** Four deltas: enough for a browser to observe growth twice over, and short. */
+export const CAPABILITY_STREAM_FIXTURE_DELTAS = [
+ "Reading the clean-room issue. ",
+ "It is blank, with one mock agent and one mock task. ",
+ "Recording a first status against the mock control plane. ",
+ "Done — every record stayed in the mock port.",
+];
+
+export const CAPABILITY_STREAM_FIXTURE_REPLY = CAPABILITY_STREAM_FIXTURE_DELTAS.join("");
+
+function sleep(ms) {
+ return new Promise((resolve) => setTimeout(resolve, ms));
+}
+
+class Notifications {
+ #values = [];
+ #waiters = [];
+ #closed = false;
+
+ push(value) {
+ const waiter = this.#waiters.shift();
+ if (waiter) waiter({ value, done: false });
+ else this.#values.push(value);
+ }
+
+ close() {
+ this.#closed = true;
+ for (const waiter of this.#waiters.splice(0)) waiter({ value: undefined, done: true });
+ }
+
+ [Symbol.asyncIterator]() {
+ return {
+ next: async () => {
+ const value = this.#values.shift();
+ if (value) return { value, done: false };
+ if (this.#closed) return { value: undefined, done: true };
+ return new Promise((resolve) => this.#waiters.push(resolve));
+ },
+ };
+ }
+}
+
+class StreamFixtureTransport {
+ #queue = new Notifications();
+ #closed = false;
+ #turns = 0;
+ #interrupted = new Set();
+
+ async request(method, params) {
+ if (method === "initialize") return { user: { sessionId: "stream-fixture-session" } };
+ if (method === "thread/start" || method === "thread/read" || method === "thread/resume") {
+ return { thread: { id: "stream-fixture-thread", sessionId: "stream-fixture-session" } };
+ }
+ if (method === "turn/start") {
+ this.#turns += 1;
+ const turnId = `stream-turn-${this.#turns}`;
+ void this.#runTurn(turnId);
+ return { turn: { id: turnId, status: "inProgress" } };
+ }
+ if (method === "turn/interrupt") {
+ // A stopped turn keeps whatever it had already streamed, so the surface
+ // can be checked for a coherent partial reply.
+ this.#interrupted.add(String(params.turnId));
+ this.#queue.push({
+ method: "turn/completed",
+ params: {
+ threadId: "stream-fixture-thread",
+ turn: { id: String(params.turnId), status: "interrupted" },
+ },
+ });
+ return {};
+ }
+ throw new Error(`unsupported scripted Codex method ${method}`);
+ }
+
+ notify() {}
+
+ notifications() {
+ return this.#queue;
+ }
+
+ setServerRequestHandler() {}
+
+ async close() {
+ this.#closed = true;
+ this.#queue.close();
+ }
+
+ processInfo() {
+ return { pid: 7100, processGroupId: 7100, exited: this.#closed, exitCode: null, signal: null };
+ }
+
+ async #runTurn(turnId) {
+ this.#queue.push({
+ method: "turn/started",
+ params: { threadId: "stream-fixture-thread", turn: { id: turnId, status: "inProgress" } },
+ });
+ await sleep(Math.floor(DELTA_INTERVAL_MS / 2));
+ if (this.#closed || this.#interrupted.has(turnId)) return;
+ this.#queue.push({
+ method: "item/reasoning/summaryTextDelta",
+ params: {
+ threadId: "stream-fixture-thread",
+ turnId,
+ delta: CAPABILITY_STREAM_FIXTURE_PRIVATE_REASONING,
+ },
+ });
+ for (const delta of CAPABILITY_STREAM_FIXTURE_DELTAS) {
+ await sleep(DELTA_INTERVAL_MS);
+ if (this.#closed || this.#interrupted.has(turnId)) return;
+ this.#queue.push({
+ method: "item/agentMessage/delta",
+ params: { threadId: "stream-fixture-thread", turnId, delta },
+ });
+ }
+ await sleep(DELTA_INTERVAL_MS);
+ if (this.#closed || this.#interrupted.has(turnId)) return;
+ this.#queue.push({
+ method: "item/completed",
+ params: {
+ threadId: "stream-fixture-thread",
+ turnId,
+ item: { id: `message-${turnId}`, type: "agentMessage", text: CAPABILITY_STREAM_FIXTURE_REPLY },
+ },
+ });
+ this.#queue.push({
+ method: "turn/completed",
+ params: { threadId: "stream-fixture-thread", turn: { id: turnId, status: "completed" } },
+ });
+ }
+}
+
+export function capabilityStreamFixtureTransportFactory(options = {}) {
+ const evidence = {
+ runnerPid: 7100,
+ runnerProcessGroupId: 7100,
+ codexPid: 7200,
+ runnerExited: false,
+ runnerExitCode: null,
+ runnerSignal: null,
+ childEnvironmentKeys: ["CODEX_HOME", "HOME", "PATH"],
+ diagnostics: [],
+ };
+ options.onEvidence?.(evidence);
+ return { transport: new StreamFixtureTransport(), evidence: () => ({ ...evidence }) };
+}
diff --git a/packages/paperclip-runner/docs/adding-a-harness.md b/packages/paperclip-runner/docs/adding-a-harness.md
new file mode 100644
index 0000000000..81fdf20c17
--- /dev/null
+++ b/packages/paperclip-runner/docs/adding-a-harness.md
@@ -0,0 +1,87 @@
+# Adding a Harness
+
+Every Paperclip Runner harness must declare its permission behavior as part of
+the provider contract. A driver is not complete until its permission modes,
+maximum non-interactive default, durable request translation, recovery
+identity, isolation behavior, and conformance coverage are defined.
+
+## Current permission catalog
+
+| Provider | Agent configuration key | Supported values | Default |
+|---|---|---|---|
+| Codex | `codexPermissionMode` | `never`, `on-request`, `untrusted` | `never` |
+| OpenCode | `opencodePermissionMode` | `allow`, `ask`, `deny` | `allow` |
+| ACPX (Claude, Codex) | `acpxPermissionMode` | `approve-all`, `approve-reads`, `deny-all` | `approve-all` |
+
+The browser-safe source of truth for labels, defaults, and configuration
+validation is `PAPERCLIP_RUNNER_PERMISSION_CAPABILITIES` in
+`@paperclipai/adapter-utils`. Native process boundaries validate the pinned
+provider value again; they must not silently accept an unknown mode.
+
+“Full auto” means the harness does not pause for a duplicate approval inside
+the environment assigned to the run. It does not grant host-wide filesystem
+access, unrestricted network access, Paperclip credentials, or write access in
+read-only planning mode. Workspace containment, protected-path and symlink
+checks, network policy, credential bindings, and authenticated PRP
+authorization remain authoritative and run before any harness auto-approval.
+
+## Driver requirements
+
+When adding or upgrading a harness:
+
+1. Declare the exact native modes supported by the provider. Choose the
+ provider's highest non-interactive mode as the default for a fresh
+ Paperclip Runner execution.
+2. Add its labels, configuration key, options, and default to the shared
+ provider-discriminated capability catalog. The agent form must show only
+ the selected provider's setting and retain stored choices when the provider
+ selection changes.
+3. Pin the effective value in native execution input and provider recovery
+ state. Include it in provider-session compatibility so changing a mode
+ replaces an incompatible warm session rather than mutating an active turn.
+4. Translate every lower-mode prompt into the canonical runtime-request
+ lifecycle: emit one durable request, recover it after reconnect, accept
+ allow-once, allow-for-session, deny, and cancellation where the provider
+ supports them, and send exactly one provider-native resolution.
+5. Authorize Paperclip semantic and question tools through authenticated PRP.
+ These tools may bypass duplicate harness approval, but never their
+ control-plane authorization.
+6. Emit the provider and effective permission mode in session-start
+ diagnostics without including credentials or unredacted provider input.
+7. Keep standalone local adapters unchanged unless their own contract is
+ deliberately revised. Paperclip Runner defaults apply only to
+ `paperclip_runner` executions.
+
+## Compatibility rules
+
+Native execution input v4 pins `approvalPolicy` for Codex and
+`permissionMode` for OpenCode and ACPX. Persisted v1-v3 executions remain
+replayable. Their missing Codex and OpenCode settings use the historical
+effective behavior, and legacy ACPX `permissionPolicy: "interactive"` means
+the historical `approve-reads` behavior. Legacy fields are readable only;
+new agent configuration must not persist them.
+
+Do not rewrite an active execution when configuration changes. A fresh
+execution carries the new policy and replaces an incompatible idle or
+checkpointed session through normal session-compatibility handling.
+
+## Conformance checklist
+
+A new driver must cover:
+
+- agent create/edit defaults, selected-provider visibility, provider switching,
+ invalid-value rejection, and no Runner sandbox-bypass control;
+- native serialization, legacy replay, recovery identity, and replacement
+ after a permission change;
+- start and resume propagation for every supported mode;
+- no runtime request in the maximum mode;
+- once, session, and deny resolution in prompting modes, including every
+ provider event version the driver accepts;
+- immediate rejection in deny mode;
+- workspace and protected-path denial before auto-approval;
+- authenticated semantic-tool behavior; and
+- an end-to-end default-mode write plus validation command that completes
+ without an Input needed permission card.
+
+Run TypeScript, Rust, UI, protocol-generation, durable transport, and
+documentation validation before declaring the harness qualified.
diff --git a/packages/paperclip-runner/docs/adr/0001-runner-testing-eval-package-boundaries.md b/packages/paperclip-runner/docs/adr/0001-runner-testing-eval-package-boundaries.md
new file mode 100644
index 0000000000..c9b65b432f
--- /dev/null
+++ b/packages/paperclip-runner/docs/adr/0001-runner-testing-eval-package-boundaries.md
@@ -0,0 +1,119 @@
+# ADR 0001: Runner and Testing Package Boundaries
+
+- Status: Accepted
+- Date: 2026-08-11
+- Decision owners: Paperclip App
+
+## Context
+
+The runner package previously exported production contracts, deterministic
+mocks, conformance fixtures, scenario/eval helpers, and demo servers from one
+root. A workspace checkout hid two packaging defects: consumers could use
+undeclared deep surfaces, and the public semantic dispatcher loaded `ajv` even
+though `ajv` was only a development dependency.
+
+External conformance consumers must consume App contracts and test helpers
+without making the App runtime depend on scenario corpora, provider
+experiments, or a separate eval package.
+
+## Decision
+
+Ownership is:
+
+| Owner | Stable responsibility |
+|---|---|
+| Paperclip App | PRP schemas and fixtures, canonical semantic catalog and dispatcher, runnerd/client interfaces, `ControlPlanePort`, the production binding, deterministic mock, and mock/real parity fixtures |
+| External eval repositories | Scenario corpus, provider configuration, experiment reports, and provider-backed orchestration |
+
+The production binding remains App code at
+`server/src/services/native-runtime/paperclip-control-plane-port.ts`. It
+implements the App-owned `ControlPlanePort`; the runner package never imports
+the server, database, UI, or CLI.
+
+Public runner exports are:
+
+| Export | Stability and purpose |
+|---|---|
+| `@paperclipai/paperclip-runner` | Runtime contracts, runner clients/backends, PRP validation/replay, canonical catalog/dispatcher, and compatibility preflight |
+| `@paperclipai/paperclip-runner/testing` | Deterministic mocks, PRP port conformance, and provider-neutral semantic conformance kit |
+| `./browser`, `./react`, `./standalone`, `./styles.css` | Existing explicitly named UI/standalone consumers |
+
+Mock adapters and conformance constants are no longer package-root exports.
+Tests and external conformance consumers must use `./testing`. Scenario
+content, provider-backed matrices, reports, and provider configuration are not
+public App package exports.
+
+The testkit remains an App `./testing` subpath rather than a separate package.
+It shares the runner protocol/catalog release cadence, and no independent
+consumer or version cadence currently justifies another package. Split it only
+after an independent release requirement exists; a directory preference is not
+sufficient.
+
+The credential-free matrix engine used by runner conformance remains
+package-local test implementation. A separately versioned eval kernel and any
+provider-backed campaign are deferred. The runner's dependency,
+optional-dependency, peer-dependency, and workspace importer sets remain free
+of eval packages.
+
+The dependency graph is acyclic:
+
+```text
+Paperclip App production binding
+ |
+ v
+@paperclipai/paperclip-runner (runtime contracts)
+ ^
+ |
+External conformance consumers --> @paperclipai/paperclip-runner/testing
+```
+
+No arrow points from App runtime to an external eval repository.
+
+## Compatibility and versioning
+
+Package semver describes distribution compatibility. Independently versioned
+contracts are published in `PAPERCLIP_RUNNER_COMPATIBILITY`:
+
+| Component | Current contract | Compatibility rule |
+|---|---:|---|
+| Canonical catalog | 1 | Operation removals, renames, placement changes, or incompatible schemas require a new contract version |
+| PRP | 1 | Highest overlapping required protocol version; no overlap fails closed |
+| Runner client | 1 | Breaking runnerd/client interface changes require a new version |
+| runnerd artifact | 2 | Binary metadata/package disagreement or digest mismatch fails before launch |
+| Harness driver | 1 | Breaking descriptor/config/session/conformance behavior requires a new version |
+| Native execution | 1 | Breaking App attempt-bundle semantics require a new schema version and converter |
+| Control-plane adapter | 1 | Breaking `ControlPlanePort` or production-binding expectations require a new version |
+| Testkit | 1 | Breaking mock seed, vector, observation, or conformance behavior requires a new version |
+| Eval corpus | 1 | Runner declares a supported inclusive corpus-version range; out-of-range bundles fail before execution |
+
+`assertPaperclipRunnerCompatibility` performs a fail-closed preflight. It emits
+`paperclip_runner_incompatible` with stable issue codes for component mismatch,
+unsupported corpus versions, unknown catalog operations, missing provider
+capability declarations, and provider-operation gaps. Provider-specific runtime
+errors must not stand in for this preflight.
+
+## Clean-consumer proof
+
+Run:
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner check:package-boundaries
+pnpm --filter @paperclipai/paperclip-runner check:clean-consumers
+```
+
+The second command builds and packs the runner, installs its tarball into a
+clean consumer, imports only the root and `./testing` exports, executes
+deterministic PRP, harness-driver, and semantic conformance, and verifies the
+separately staged runnerd artifact digest. The consumer uses no workspace
+protocol, source-relative import, or deep package path. This is the packaging
+gate; workspace tests alone are not proof.
+
+## Consequences
+
+- Existing tests importing mock/conformance values from the package root must
+ migrate to `@paperclipai/paperclip-runner/testing`.
+- `ajv` is a runtime dependency because the public dispatcher imports it.
+- The semantic conformance kit defines normalized comparison; real App service
+ adapters and risk-weighted vectors may evolve behind its testkit version.
+- Scenario corpus changes do not force an App runtime release unless their
+ declared compatibility requirement changes.
diff --git a/packages/paperclip-runner/docs/architecture.md b/packages/paperclip-runner/docs/architecture.md
new file mode 100644
index 0000000000..fd49aa7147
--- /dev/null
+++ b/packages/paperclip-runner/docs/architecture.md
@@ -0,0 +1,298 @@
+# Architecture and Standalone Boundary
+
+## Dependency direction
+
+```text
+ language-neutral protocol fixture
+ |
+ +---------------+---------------+
+ v v
+ Rust runner-core + mock path TypeScript contracts + mock path
+ | |
+ +---------------+---------------+
+ v
+ byte-identical Conformance result
+
+Paperclip App production binding --> implements ControlPlanePort
+External conformance consumers --> use packed runtime + ./testing exports
+```
+
+The dependency arrow always points from an implementation toward a contract.
+The standalone package does not reach backward into a Paperclip implementation.
+The production implementation lives in
+`server/src/services/native-runtime/paperclip-control-plane-port.ts`; it depends
+on the public port, never the reverse. The accepted package/export ownership is
+recorded in [ADR 0001](adr/0001-runner-testing-eval-package-boundaries.md).
+
+## Public package surfaces
+
+- The package root is runtime-only: PRP, runner/client contracts, normalized
+ backends, catalog/dispatcher, and compatibility preflight.
+- `./testing` contains deterministic mocks and conformance kits.
+- Package-local deterministic matrices remain internal test implementation.
+ A separately versioned eval package and provider-backed campaigns are
+ deferred and are not workspace dependencies or public exports.
+
+## Core contracts
+
+- `ControlPlanePort` is the narrow surface through which a runner opens a run,
+ appends ordered events, and submits a terminal structured result.
+- `HarnessDriver` owns a local harness session and its provider-specific
+ identity, event, turn, snapshot, and close behavior.
+- `NativeSessionBackend` normalizes local runner and hosted-provider sessions for
+ a future control-plane consumer. Environment placement is not implied by the
+ backend type.
+
+The TypeScript contracts name responsibility and dependency direction while
+the executable replay path provides a deterministic oracle:
+
+```text
+protocol/schemas/*.json
+ | generate/check | shared fixtures
+ v v
+TypeScript schema constants/types -> validator -> deterministic reducer
+ | |
+ v v
+ CLI browser devtool
+ |
+ golden parity summaries
+ |
+ v
+ Rust runner-core oracle
+```
+
+The Rust `runner-core` crate establishes the production language/package
+boundary and checks the same fixture summaries. Local runner adds the package-local
+`paperclip-runnerd` and `fake-harness` binaries without changing that dependency
+direction.
+
+The clean-consumer gate packs the declared root and `./testing` exports and
+stages the release runnerd executable as a separately checksummed artifact.
+
+## Language ownership
+
+- Rust is the production direction for deterministic runner behavior,
+ supervision, durable delivery, and the eventual `paperclip-runnerd` binary.
+- TypeScript owns the control-plane/browser side and remains a useful reference
+ client/test oracle.
+- JSON Schema and shared fixtures are the language-neutral authority. Conformance
+ keeps its narrow tracer fixture; Replay adds the executable PRP v1 schema and
+ conformance corpus without silently changing the accepted Conformance path.
+- `check:conformance-parity` prevents either implementation from introducing a
+ language-specific observable result.
+- `check:replay-parity` prevents TypeScript replay and the Rust production
+ direction from disagreeing on identity, terminal state, duplicates, or gaps.
+
+## Allowed dependencies
+
+- Rust crates declared by the package-local Cargo workspace.
+- Node.js standard-library modules.
+- Third-party packages declared by this workspace.
+- Files within `packages/paperclip-runner/`.
+- A future explicit generated-schema package only after architecture review and
+ an allowlist change in the boundary checker.
+
+## Forbidden dependencies
+
+The following imports and package dependencies are rejected:
+
+- `server/`, `ui/`, and `cli/` implementation paths;
+- `@paperclipai/db` and production database schema or client modules;
+- `@paperclipai/shared`, adapter utilities, and other Paperclip workspace
+ internals unless a boundary review explicitly allows a public contract;
+- relative or absolute imports that escape `packages/paperclip-runner/`.
+
+This rule applies to type-only imports, exports, dynamic imports, CommonJS
+`require` calls, Rust include/path attributes, and Cargo path dependencies. The
+negative fixtures under `test-fixtures/` intentionally reference `server/` and
+must fail the checker.
+
+## Enforcement
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner check:forbidden-imports
+pnpm --filter @paperclipai/paperclip-runner test
+pnpm --filter @paperclipai/paperclip-runner check:replay-parity
+```
+
+The first command scans the package source, scripts, and manifest. The test
+command additionally asserts that the negative fixture is rejected. The normal
+scan excludes that fixture so a deliberate proof does not make the package fail.
+
+## Conformance process boundary
+
+The mock core is an in-memory adapter, not a Paperclip server. Starting it only
+changes local object state. The tracer performs this sequence:
+
+1. load and validate `protocol/fixtures/conformance-minimal-run.json`;
+2. start the mock adapter;
+3. open the fixture run through `ControlPlanePort`;
+4. append contiguous typed events;
+5. submit the matching terminal result;
+6. print a stable JSON identity/result and stop the adapter.
+
+No socket, database, browser, Paperclip process, or model process is started.
+The default command executes this sequence in Rust. The TypeScript reference
+executes the same sequence, and the parity check compares their complete stdout.
+
+## Static replay boundary
+
+`replayReplayFixtureText` is the single entry point used by the CLI and browser.
+It parses JSON, validates JSON Schema plus cross-record bindings, and only then
+calls the reducer. The reducer is pure: it clones input state, applies an event
+at most once by source event ID, records source gaps/out-of-order deliveries,
+and never performs I/O.
+
+The browser is a Vite application under `devtools/browser/`. Its Button, Badge,
+Card, and Textarea are source-compatible adaptations of shadcn primitives; all
+visual values live in its local `styles.css` token layer. It imports the same
+replay module as the CLI and does not create a browser-only protocol model.
+
+## Local process boundary
+
+```text
+TypeScript mock core
+ | PRP commands over stdin JSONL
+ v
+paperclip-runnerd (Rust supervisor)
+ | fake-harness commands over stdin JSONL
+ v
+fake-harness (Rust scripted driver)
+ | typed messages over stdout JSONL
+ v
+paperclip-runnerd -> canonical PRP events -> mock core
+ |
+ v
+ browser NDJSON stream
+```
+
+The mock core starts one runner process. The runner creates a new process group
+for one fake harness and its workers. The runner clears the inherited
+environment and restores only the path needed to launch local executables. It
+captures stderr and scripted log messages in a bounded tail.
+
+The controller and harness links use newline-delimited JSON over stdio. This is
+the smallest local transport that keeps process ownership clear. The browser
+does not connect to the runner. A package-local Vite middleware exposes an HTTP
+start/action API and an NDJSON event stream from the TypeScript mock core.
+
+The runner publishes the structured semantic result before it publishes the
+harness process exit fact. It then emits one `run.terminal` event. A non-zero
+harness exit can coexist with a valid yielded result. Duplicate commands and
+duplicate terminal messages cannot repeat side effects or close the run twice.
+
+Every browser event passes `validatePrpEvent` and `applyPrpEvent`. When the run
+ends, the browser reduces the complete event list again and compares the replay
+snapshot with the live snapshot.
+
+## Durable transport boundary
+
+```text
+TypeScript mock core
+ | one-time ticket -> short-lived connection lease
+ | PRP v1 hello/welcome, commands, events, cumulative ACKs
+ v
+paperclip-runnerd (Rust WebSocket client)
+ | atomic private JSON state
+ +-- durable outbox and processed-command cache
+ +-- stable runner/session/turn/item identities
+ +-- Local runner fake-harness process for restart proof
+```
+
+The runner initiates the loopback WebSocket. The bootstrap ticket is present
+only in the runner process environment. The returned connection-lease token is
+kept only in runner memory. The mock core stores SHA-256 digests of capabilities
+and the runner state stores neither capability. The runner writes each event and
+command result before network delivery, then removes outbox events only after a
+valid cumulative ACK.
+
+The TypeScript peer is a package-local control-plane implementation, not
+production Paperclip. Its focused tests cover lost ACKs, socket loss, malformed
+input, process restarts, lease expiry, storage pressure, drain, and revoke.
+
+## Capability-model boundary
+
+```text
+Paperclip skill + 7 references Paperclip Evals corpus (106 cases)
+ | |
+ +-------------------+------------------+
+ v
+ generated capability contract (258 rows, 41 MCP aliases)
+ |
+ +-----------------------+-----------------------+
+ v v v
+ semantic tool catalog authorization engine eval conformance suite
+ \ | /
+ \ v /
+ +-----> in-process mock ControlPlanePort <--+
+ |
+ v
+ read-only browser scenario explorer
+```
+
+Capability is a package-local model of a native Paperclip run. It classifies every
+capability as control-plane-owned, always-agent-tool, or optional-agent-tool,
+exposes the always/optional set as a transport-neutral semantic tool catalog,
+gates optional tools behind grants, and proves 106 eval-derived cases against an
+in-process mock `ControlPlanePort`. The mock adapter is the only coupling point,
+so a real adapter can replace it later without touching the catalog,
+authorization rules, or conformance suite. Capability contacts no Paperclip
+service, database, ACPX session, or provider credential; the
+[forbidden-imports checker](#forbidden-dependencies) keeps it that way. Real
+integration is future upload integration (ACPX) and requires separate approval; see
+[the future binding boundary](capability-future-binding-boundary.md).
+
+## Capability live process topology
+
+The live surface adds a real provider turn loop over the same mock core. The
+package server owns every credential and every child process; the browser holds
+none.
+
+```text
+ browser (issue thread + evidence panel; no credential)
+ | HTTP/SSE over the trusted-proxy boundary
+ v
+ package server ── projectCapabilityIssueThread ──> CapabilityIssueThreadSnapshot
+ | (one contract, two producers)
+ | owns session, tool loop, provider auth
+ v
+ paperclip-runnerd (real binary, owns the Codex process group)
+ | newline-delimited JSON-RPC over stdio
+ v
+ codex app-server (real session)
+ | tool request
+ v
+ CapabilitySemanticDispatcher ── typed command ──> in-process mock ControlPlanePort
+ ^ |
+ +──────── typed result / typed denial ─────+
+```
+
+Three actors stay separate at all times: **Real Codex** (the app-server
+session), **Real runnerd** (the package-local binary), and **Mock Paperclip**
+(the in-process `ControlPlanePort`). The same
+`CapabilityIssueThreadSnapshot` is produced by deterministic `fake` fixtures for the
+screenshot matrix and by the server-side projection for a live session; the
+projection reads only durable records and decides nothing, so UI-side state math
+is a defect by construction. The scripted (`fake`) mode drives the conformance
+suite and replay offline; the Codex (live) mode requires a locally authenticated
+Codex. See [execution modes and identity](capability-execution-modes.md) for the
+mode and eligibility rules, and [the live runnerd/Codex loop](capability-live-runnerd-codex.md)
+for the session API. The package server blocks every request to a real
+Paperclip API, and the evidence suite proves no such request occurred. The same
+topology serves the [clean-room chat](capability-clean-room-chat.md): the only
+difference is a mock tenant seeded with a company, an agent, and one blank issue
+instead of a recorded eval case.
+
+## Integration rule
+
+Paperclip core may implement these contracts behind a separately reviewed
+adapter, but this package must remain independently buildable, testable, and
+runnable against the mock adapter.
+
+The proposed Standalone seam is recorded in
+[Standalone Thin Paperclip Adapter Boundary](design/standalone-thin-paperclip-adapter.md).
+It keeps one dependency direction, branches only after Paperclip workspace and
+environment realization, composes a package-owned `NativeSessionBackend` with
+a server-bound `ControlPlanePort`, and returns to the existing Paperclip
+finalization path. The core seam contains no runner behavior. The proposal is
+design-only until the CTO gate accepts it.
diff --git a/packages/paperclip-runner/docs/capability-authorization-and-exposure.md b/packages/paperclip-runner/docs/capability-authorization-and-exposure.md
new file mode 100644
index 0000000000..f802886d08
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-authorization-and-exposure.md
@@ -0,0 +1,98 @@
+# Capability Authorization and Exposure
+
+Two decisions gate every capability: **exposure** (is the tool even offered to
+this actor?) and **invocation** (may this specific call proceed?). The
+authorization engine answers both, produces typed denials that carry no
+protected state, and redacts secrets before any observable boundary.
+
+Sources: `src/tools/capability-tool-authorization.ts`,
+`src/tools/capability-semantic-tool-runtime.ts`, `src/tools/capability-tool-bindings.ts`.
+
+## Three-way classification
+
+- **Control-plane-owned** — a frozen operation list (`checkout_task`,
+ `release_task`, `select_work`, `route_wake`, `enforce_budget`,
+ `append_audit_record`, `persist_run`, `replay_run`, `schedule_blocker_wake`,
+ `reconcile_run`). `authorizeInvocation` short-circuits these to outcome
+ `absent` with reason `control_plane_owned_operation`. No agent tool exists for
+ them and none can be invoked.
+- **Always-agent tool** — exposed to any actor whose task is active.
+- **Optional-agent tool** — exposed only when a grant unlocks it.
+
+## Exposure vs. invocation
+
+- `computeVisibleTools()` walks the whole catalog and records, per tool, an
+ exposure of `exposed` or `absent`, returning only the allowed tools.
+- `authorizeInvocation()` re-evaluates at call time, producing `allowed` or
+ `denied`.
+
+Invocation is evaluated in order and fails at the first rule that rejects:
+policy-denied operation IDs; task-mode membership
+(`operation_not_available_in_task_mode`); role check
+(`actor_role_not_authorized`); policy-denied claims
+(`required_claim_denied_by_policy`); missing grants
+(`required_claim_missing`); `read_secret_value` requires
+`policy.allowSecretValueAccess` (`secret_value_access_disabled`);
+`generic_api_request` requires the escape hatch to be enabled and the target to
+be allowlisted (`generic_escape_hatch_disabled` / `..._target_denied`); and the
+interaction-kind policy for `request_human_input`. Allowed calls carry the
+reason `task_scoped_always_tool` or `required_claims_granted`.
+
+Self-approval is prohibited even with the role and an explicit decision claim.
+
+## How grants unlock optional tools
+
+The actor's grants are the union of `capabilityGrants` (the actor's own) and
+`scenarioGrants` (seeded per scenario). An optional tool unlocks only when
+**every** entry in its `requiredClaims` is present in that union. Each
+authorization record stores the `consideredClaims`, `grantedClaims`, and
+`missingClaims`, so a denial explains exactly which claim was absent.
+
+## Typed denials carry no protected state
+
+A denial is a `CapabilityPolicyDenial`: `ok: false` with an `error.code` of
+`policy_denied`, `operation_absent`, `input_invalid`, or `operation_unsupported`,
+the `operationId`, a generic `reason`, and the authorization record. It carries
+no fixture state and no protected payload. If the underlying mock operation
+throws, the runtime emits `operation_unsupported` with reason
+`mock_operation_rejected` and swallows the underlying error, so protected state
+cannot leak through an exception.
+
+## Secret redaction (the TASK-16909 rule)
+
+Redaction is two layers:
+
+1. **Model-only delivery is separated from every observable boundary.**
+ `invoke()` returns a redacted `observableResult`. The only path to an
+ unredacted secret is `invokeForModel()`, which returns a frozen capsule whose
+ `toJSON()` and default serialization always yield the redacted form; only an
+ explicit `readModelResult()` opens the raw value. The capsule uses a distinct
+ schema (`paperclip.capability.model-tool-result.v1`) so it cannot be assigned to
+ an observable sink by mistake.
+2. **Path redaction rules.** `read_secret_value` redacts `$.value` to
+ `[SECRET_VALUE]` across output, error, and authorization-record channels;
+ `generic_api_request` redacts `$.headers.authorization` and
+ `$.headers.cookie` to `[REDACTED]` across all channels. The scenario runner
+ applies the same redaction again defensively at the explorer artifact
+ boundary and injects only a placeholder secret value.
+
+This is the security-gate outcome (TASK-16902) and its remediation (TASK-16909):
+a real secret value reaches the model surface only, never a trace, artifact, or
+browser view.
+
+## Running the tests
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner exec vitest run \
+ src/tools/capability-semantic-tools.test.ts
+```
+
+The "exposure and authorization" and "security policy" describes cover typed
+denials without protected state, model-only secret delivery, the self-approval
+prohibition, and the test-only, explicitly granted, allowlisted escape hatch.
+
+## Related
+
+- [Capability disposition](capability-disposition.md)
+- [Semantic tool catalog](capability-semantic-tools.md)
+- [Scenario explorer](capability-scenario-explorer.md)
diff --git a/packages/paperclip-runner/docs/capability-clean-room-chat.md b/packages/paperclip-runner/docs/capability-clean-room-chat.md
new file mode 100644
index 0000000000..bcd79002db
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-clean-room-chat.md
@@ -0,0 +1,170 @@
+# Capability clean-room live chat
+
+Capability adds a second primary path beside the preset scenario explorer. A board
+user opens a blank chat, sends a free-form message, and watches real Codex work
+a mock Paperclip issue through real runnerd. There is no scenario to pick, no
+recorded transcript, and no scripted tool tour.
+
+Both paths ship in the same app and share one view contract. What separates them
+is what the session is seeded with, and whether a fixture is allowed to stand in
+for a provider.
+
+| | Preset scenario explorer | Clean-room chat |
+| --- | --- | --- |
+| Route | `#/issue/` | `#/chat` |
+| Seed | A recorded eval case: transcript, scripted calls, parity verdicts | A company, an agent, and one blank issue |
+| Modes | `fake` (default), `replay`, `mode=live` | Live only — no fixture, no recording |
+| Tool calls | Fixed by the scenario | Chosen by the conversation |
+| Evidence drawer | Collapsed by default | Collapsed by default |
+
+## What "clean room" means
+
+Every open mints a new mock tenant — a new company, actor, task, and `MCK-`
+identifier — and seeds nothing else. `assertCapabilityCleanRoomSeedIsBlank` fails the
+open if any comment, document, interaction, approval, artifact, blocker, wake,
+or fault is present, so "the thread starts blank" is enforced at the seam that
+creates the session rather than assumed.
+
+The only mutation an open performs is the run checking the mock issue out, which
+is why a fresh room reads `in_progress` rather than `todo`.
+
+## Live only, and loudly so
+
+The clean room never falls back:
+
+- `parseCapabilityRoute` pins `mode` to `live` on `#/chat` and drops `shot` and
+ `at`. No URL can talk it into rendering a fixture or a recording.
+- The package server has no scripted path for this route. It projects the live
+ session and refuses any view that is not `mode=live` with a `Real Codex`
+ agent label.
+- When runnerd or the Codex app-server cannot start, the surface says exactly
+ that and offers `Try again`. It does not render a canned thread.
+
+## Exposure profile
+
+The profile is deterministic and inspectable even though the calls are not.
+`CAPABILITY_CLEAN_ROOM_CLAIMS` grants the read, comment, document, interaction,
+deliverable, delegation, dependency, wake, and approval-request operations.
+
+Three grants stay withheld on purpose:
+
+| Withheld grant | Effect |
+| --- | --- |
+| `governance:approvals:decide` | The agent may request an approval; the board decides it. |
+| `workspace:control` | No service lifecycle changes from a chat. |
+| `test:generic_api_request` | The escape hatch stays off. |
+
+Withholding them keeps a denial reachable in a conversation nobody scripted: if
+the model reaches for one, it gets a typed denial with a named code and the mock
+state does not move. The Evidence drawer's Tools section lists all three under
+`Control plane (not exposed to the agent)`.
+
+The run itself holds the wider adapter claim set that the mock command boundary
+requires (`capabilityFixtureRunCapabilities`). That union widens the port, never the
+catalog the model sees: effective claims are the intersection of run claims,
+scenario claims, and explicitly delegated claims.
+
+## Session lifecycle
+
+```text
+GET /api/capability/ui/cleanroom/session[?sessionId=…] open or reconnect
+POST /api/capability/ui/cleanroom/session {sessionId?} New chat (retires the caller's room)
+POST /api/capability/ui/message {sessionId, message}
+POST /api/capability/ui/interrupt {sessionId}
+POST /api/capability/ui/reconnect {sessionId}
+POST /api/capability/ui/reset {sessionId}
+POST /api/capability/ui/interaction {sessionId, interactionId, outcome, result}
+```
+
+`Reset` and `New chat` both stop active work, clear the previous session's
+authority, drop its workspace directory, and open a new tenant. A retired
+session id answers `404` on every route afterwards — the rotation is verifiable,
+not just visible. Refresh reconnects to the same durable session; a stale id
+from `localStorage` opens a fresh room rather than dead-ending.
+
+The clean room stores its session id under its own `localStorage` key, so a
+scenario session can never be handed to the chat route or the reverse.
+
+## Bounds
+
+| Bound | Value |
+| --- | --- |
+| Turns per chat | 24, then a named `turn_limit` refusal that points at `New chat` |
+| Message size | 8 KiB |
+| Concurrent clean rooms | 4; the oldest yields to a new board user |
+| Turn timeout | 120 s (`CapabilityLiveSessionService` default) |
+
+## Real-API block
+
+`CapabilityLiveSession` routes every Paperclip operation through the in-process mock
+`ControlPlanePort`; no code path reaches a Paperclip URL. The projection turns
+that into a record rather than a claim: the Control plane section of the
+Evidence drawer carries a `network-guard-` row reading
+`Real Paperclip API requests: 0. Child PAPERCLIP_* environment keys: none.`
+
+The child environment is allowlisted by `createSanitizedCodexEnvironment`, so no
+`PAPERCLIP_*` value reaches runnerd or Codex, and the browser receives no
+provider, runner, or control-plane credential.
+
+## DevTools state inspector
+
+Opening **DevTools** opens a Redux-DevTools-style inspector over the live
+mock company. Its timeline starts at the pristine fixture, adds a revision for
+each successful semantic mutation, and lets the operator inspect or diff the
+complete browser-safe company scaffolding: company, actors, tasks, comments,
+documents, interactions, approvals, artifacts, work products, blockers,
+workspace services, budgets, runs, wakes, audit records, decisions,
+idempotency records, and configured faults. Separate tabs expose the protocol
+boundary records, runner/runtime facts, and effective authority. A dedicated
+**Documents** tab renders every mock document and lets the operator switch
+between its retained revisions without digging through raw JSON.
+
+The Evidence log consolidates repeated tool-exposure snapshots into one
+deduplicated catalog. Its Always, Granted, and Control-plane groups fold
+independently; selecting a tool opens its catalog title, description, placement,
+required claims, allowed task modes and roles, and full input schema.
+
+**Pause** pins a revision while work continues, **Export** downloads the
+redacted DevTools snapshot, and **Fork rN** retires the current mock session and
+starts a new executable branch from that retained state. Provider payloads,
+session checkpoints, artifact content references, working directories, and
+secret-shaped strings remain server-side or are withheld by the explicit
+browser projection.
+
+While a turn runs, a live status rail appears immediately after send and tracks
+the newest safe Codex activity. Reasoning, planning, shell-command, file-change,
+MCP/dynamic-tool, assistant-stream, and Paperclip semantic-tool lifecycle events
+each trigger an interim frame. Discrete tools remain separate rows; only noisy
+text deltas are grouped. Structured shell items show a bounded,
+credential-redacted command preview; raw command output, other provider
+payloads, and chain of thought stay withheld.
+
+## Remote preview gateway
+
+The tailnet preview keeps the package server on loopback and exposes it through
+`dist/capability/tailnet-gateway.js`. Every API call requires the gateway's
+high-entropy HttpOnly capability cookie. Mutations additionally require the
+exact configured `Origin` and `application/json`.
+
+Fetch Metadata is checked as defense in depth: any supplied `Sec-Fetch-Site`,
+`Sec-Fetch-Mode`, or `Sec-Fetch-Dest` value must describe the expected
+same-origin `fetch()` request. Browsers that omit one or all of those optional
+headers are still accepted after the capability, Origin, and content-type
+checks pass. This keeps Safari and embedded/private browser clients working
+without weakening the explicit cross-site denial.
+
+## Verification
+
+| Surface | Command |
+| --- | --- |
+| Seed, exposure profile, and identity rotation | `pnpm --filter @paperclipai/paperclip-runner test:scenarios` |
+| Clean-room HTTP routes end to end (stub provider) | included in `test:scenarios` |
+| Browser entry, blank state, evidence-on-demand, narrow layout, axe | `pnpm --filter @paperclipai/paperclip-runner test:browser:issue-thread` |
+| Real Codex through real runnerd | `pnpm --filter @paperclipai/paperclip-runner smoke:capability:cleanroom` |
+| Live screenshots | `pnpm --filter @paperclipai/paperclip-runner recorded-evidence campaign (deferred)` |
+
+See the [clean-room chat tutorial](tutorials/capability-clean-room-chat.md) for the
+clean-start walkthrough, [execution modes and identity](capability-execution-modes.md)
+for the fake/live eligibility rules, and the
+[issue-thread UI reference](capability-issue-thread-ui.md) for the thread,
+composer, and Evidence panel this surface reuses.
diff --git a/packages/paperclip-runner/docs/capability-contract.md b/packages/paperclip-runner/docs/capability-contract.md
new file mode 100644
index 0000000000..0b73d59fbb
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-contract.md
@@ -0,0 +1,244 @@
+
+
+# Capability Capability Contract
+
+This generated contract is a self-contained derivative of the Paperclip skill, its seven references, the Paperclip Evals corpus, and the legacy MCP tool surface. It does not import or contact the Paperclip control plane.
+
+The skill/reference inventory and eval cases are the only normative behavior sources. Paperclip does not use the legacy MCP calls as a production capability surface; all MCP names below are traceability aliases folded into normative eval rows. Their disposition, grants, assertions, and evidence contract are inherited from the target row rather than classified independently.
+
+## Baseline Counts
+
+- Skill/reference headings: 152
+- Eval cases: 106 across 16 groups
+- Total normative rows: 258
+- Legacy MCP aliases folded into normative rows: 42
+
+| Eval group | Cases |
+| --- | ---: |
+| hb | 5 |
+| co | 6 |
+| st | 8 |
+| cm | 6 |
+| se | 4 |
+| su | 4 |
+| bl | 5 |
+| dp | 3 |
+| ix | 9 |
+| ap | 6 |
+| ar | 4 |
+| er | 9 |
+| rf | 22 |
+| mh | 4 |
+| rs | 3 |
+| wk | 8 |
+
+## Regeneration
+
+- `pnpm --dir packages/paperclip-runner generate:capability-inventory` imports the canonical baselines and rewrites every generated file.
+- `pnpm --dir packages/paperclip-runner check:capability-inventory` validates counts, uniqueness, normative dispositions, one-to-one MCP folds, required fields, and generated-file drift without requiring the external eval repository.
+
+## Skill / Reference Rows
+
+| Capability | Primary disposition | Source anchor |
+| --- | --- | --- |
+| skill:skills/paperclip/SKILL.md:paperclip-skill:10 | optional_agent_tool | skills/paperclip/SKILL.md:10 |
+| skill:skills/paperclip/SKILL.md:terminology:14 | optional_agent_tool | skills/paperclip/SKILL.md:14 |
+| skill:skills/paperclip/SKILL.md:authentication:18 | control_plane_owned | skills/paperclip/SKILL.md:18 |
+| skill:skills/paperclip/SKILL.md:the-heartbeat-procedure:30 | optional_agent_tool | skills/paperclip/SKILL.md:30 |
+| skill:skills/paperclip/SKILL.md:generated-artifacts-and-work-products:101 | always_agent_tool | skills/paperclip/SKILL.md:101 |
+| skill:skills/paperclip/SKILL.md:status-quick-guide:148 | control_plane_owned | skills/paperclip/SKILL.md:148 |
+| skill:skills/paperclip/SKILL.md:monitors-and-watchers-say-only-what-you-actually-scheduled:158 | optional_agent_tool | skills/paperclip/SKILL.md:158 |
+| skill:skills/paperclip/SKILL.md:delegating-review-tasks:171 | always_agent_tool | skills/paperclip/SKILL.md:171 |
+| skill:skills/paperclip/SKILL.md:managing-a-user-s-inbox:182 | control_plane_owned | skills/paperclip/SKILL.md:182 |
+| skill:skills/paperclip/SKILL.md:issue-dependencies-blockers:190 | control_plane_owned | skills/paperclip/SKILL.md:190 |
+| skill:skills/paperclip/SKILL.md:requesting-board-approval:215 | optional_agent_tool | skills/paperclip/SKILL.md:215 |
+| skill:skills/paperclip/SKILL.md:issue-thread-interactions:236 | optional_agent_tool | skills/paperclip/SKILL.md:236 |
+| skill:skills/paperclip/SKILL.md:standalone-decisions:265 | optional_agent_tool | skills/paperclip/SKILL.md:265 |
+| skill:skills/paperclip/SKILL.md:mcp-tool-approval-gates:369 | optional_agent_tool | skills/paperclip/SKILL.md:369 |
+| skill:skills/paperclip/SKILL.md:niche-workflow-pointers:411 | optional_agent_tool | skills/paperclip/SKILL.md:411 |
+| skill:skills/paperclip/SKILL.md:cases:421 | optional_agent_tool | skills/paperclip/SKILL.md:421 |
+| skill:skills/paperclip/SKILL.md:company-skills-workflow:426 | optional_agent_tool | skills/paperclip/SKILL.md:426 |
+| skill:skills/paperclip/SKILL.md:routines:437 | optional_agent_tool | skills/paperclip/SKILL.md:437 |
+| skill:skills/paperclip/SKILL.md:issue-workspace-runtime-controls:448 | optional_agent_tool | skills/paperclip/SKILL.md:448 |
+| skill:skills/paperclip/SKILL.md:proposing-credentials-safely:455 | optional_agent_tool | skills/paperclip/SKILL.md:455 |
+| skill:skills/paperclip/SKILL.md:reading-granted-secrets:462 | optional_agent_tool | skills/paperclip/SKILL.md:462 |
+| skill:skills/paperclip/SKILL.md:critical-rules:488 | optional_agent_tool | skills/paperclip/SKILL.md:488 |
+| skill:skills/paperclip/SKILL.md:comment-style-required:512 | always_agent_tool | skills/paperclip/SKILL.md:512 |
+| skill:skills/paperclip/SKILL.md:update:544 | optional_agent_tool | skills/paperclip/SKILL.md:544 |
+| skill:skills/paperclip/SKILL.md:planning-required-when-planning-requested:554 | optional_agent_tool | skills/paperclip/SKILL.md:554 |
+| skill:skills/paperclip/SKILL.md:key-endpoints-hot-routes:587 | optional_agent_tool | skills/paperclip/SKILL.md:587 |
+| skill:skills/paperclip/SKILL.md:searching-issues:616 | optional_agent_tool | skills/paperclip/SKILL.md:616 |
+| skill:skills/paperclip/SKILL.md:full-reference:626 | optional_agent_tool | skills/paperclip/SKILL.md:626 |
+| skill:skills/paperclip/references/artifacts.md:generated-artifacts-and-work-products:1 | always_agent_tool | skills/paperclip/references/artifacts.md:1 |
+| skill:skills/paperclip/references/artifacts.md:workspace-only-file-references:15 | optional_agent_tool | skills/paperclip/references/artifacts.md:15 |
+| skill:skills/paperclip/references/cases.md:cases:1 | optional_agent_tool | skills/paperclip/references/cases.md:1 |
+| skill:skills/paperclip/references/cases.md:core-model:12 | optional_agent_tool | skills/paperclip/references/cases.md:12 |
+| skill:skills/paperclip/references/cases.md:upsert-semantics:29 | optional_agent_tool | skills/paperclip/references/cases.md:29 |
+| skill:skills/paperclip/references/cases.md:read-and-search:67 | optional_agent_tool | skills/paperclip/references/cases.md:67 |
+| skill:skills/paperclip/references/cases.md:documents:90 | always_agent_tool | skills/paperclip/references/cases.md:90 |
+| skill:skills/paperclip/references/cases.md:fields:118 | optional_agent_tool | skills/paperclip/references/cases.md:118 |
+| skill:skills/paperclip/references/cases.md:issue-links:151 | optional_agent_tool | skills/paperclip/references/cases.md:151 |
+| skill:skills/paperclip/references/cases.md:child-cases:176 | optional_agent_tool | skills/paperclip/references/cases.md:176 |
+| skill:skills/paperclip/references/cases.md:attachments:195 | optional_agent_tool | skills/paperclip/references/cases.md:195 |
+| skill:skills/paperclip/references/cases.md:lifecycle:208 | optional_agent_tool | skills/paperclip/references/cases.md:208 |
+| skill:skills/paperclip/references/cases.md:worked-blog-post-example:222 | optional_agent_tool | skills/paperclip/references/cases.md:222 |
+| skill:skills/paperclip/references/company-skills.md:company-skills-workflow:1 | optional_agent_tool | skills/paperclip/references/company-skills.md:1 |
+| skill:skills/paperclip/references/company-skills.md:what-exists:5 | optional_agent_tool | skills/paperclip/references/company-skills.md:5 |
+| skill:skills/paperclip/references/company-skills.md:permission-model:22 | optional_agent_tool | skills/paperclip/references/company-skills.md:22 |
+| skill:skills/paperclip/references/company-skills.md:core-endpoints:29 | optional_agent_tool | skills/paperclip/references/company-skills.md:29 |
+| skill:skills/paperclip/references/company-skills.md:install-a-skill-into-the-company:65 | optional_agent_tool | skills/paperclip/references/company-skills.md:65 |
+| skill:skills/paperclip/references/company-skills.md:app-shipped-catalog:75 | optional_agent_tool | skills/paperclip/references/company-skills.md:75 |
+| skill:skills/paperclip/references/company-skills.md:external-source-import:101 | optional_agent_tool | skills/paperclip/references/company-skills.md:101 |
+| skill:skills/paperclip/references/company-skills.md:source-types-in-order-of-preference:105 | optional_agent_tool | skills/paperclip/references/company-skills.md:105 |
+| skill:skills/paperclip/references/company-skills.md:example-skills-sh-import-preferred:116 | optional_agent_tool | skills/paperclip/references/company-skills.md:116 |
+| skill:skills/paperclip/references/company-skills.md:example-github-import:138 | optional_agent_tool | skills/paperclip/references/company-skills.md:138 |
+| skill:skills/paperclip/references/company-skills.md:inspect-what-was-installed:164 | optional_agent_tool | skills/paperclip/references/company-skills.md:164 |
+| skill:skills/paperclip/references/company-skills.md:assign-skills-to-an-existing-agent:181 | optional_agent_tool | skills/paperclip/references/company-skills.md:181 |
+| skill:skills/paperclip/references/company-skills.md:include-skills-during-hire-or-create:216 | optional_agent_tool | skills/paperclip/references/company-skills.md:216 |
+| skill:skills/paperclip/references/company-skills.md:notes:256 | optional_agent_tool | skills/paperclip/references/company-skills.md:256 |
+| skill:skills/paperclip/references/issue-workspaces.md:issue-workspace-runtime-controls:1 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:1 |
+| skill:skills/paperclip/references/issue-workspaces.md:discover-the-workspace:5 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:5 |
+| skill:skills/paperclip/references/issue-workspaces.md:control-services:23 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:23 |
+| skill:skills/paperclip/references/issue-workspaces.md:start-all-configured-services-waits-for-configured-readiness-checks:28 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:28 |
+| skill:skills/paperclip/references/issue-workspaces.md:restart-all-configured-services:36 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:36 |
+| skill:skills/paperclip/references/issue-workspaces.md:stop-all-running-services:44 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:44 |
+| skill:skills/paperclip/references/issue-workspaces.md:read-the-url:63 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:63 |
+| skill:skills/paperclip/references/issue-workspaces.md:mcp-tools:72 | optional_agent_tool | skills/paperclip/references/issue-workspaces.md:72 |
+| skill:skills/paperclip/references/routines.md:paperclip-routines:1 | optional_agent_tool | skills/paperclip/references/routines.md:1 |
+| skill:skills/paperclip/references/routines.md:lifecycle:16 | optional_agent_tool | skills/paperclip/references/routines.md:16 |
+| skill:skills/paperclip/references/routines.md:creating-a-routine:27 | optional_agent_tool | skills/paperclip/references/routines.md:27 |
+| skill:skills/paperclip/references/routines.md:concurrency-policies:64 | optional_agent_tool | skills/paperclip/references/routines.md:64 |
+| skill:skills/paperclip/references/routines.md:catch-up-policies:76 | optional_agent_tool | skills/paperclip/references/routines.md:76 |
+| skill:skills/paperclip/references/routines.md:activity-gated-scheduled-runs:87 | optional_agent_tool | skills/paperclip/references/routines.md:87 |
+| skill:skills/paperclip/references/routines.md:example-skip-quiet-nights:107 | optional_agent_tool | skills/paperclip/references/routines.md:107 |
+| skill:skills/paperclip/references/routines.md:adding-triggers:126 | optional_agent_tool | skills/paperclip/references/routines.md:126 |
+| skill:skills/paperclip/references/routines.md:schedule-cron:136 | optional_agent_tool | skills/paperclip/references/routines.md:136 |
+| skill:skills/paperclip/references/routines.md:webhook:150 | optional_agent_tool | skills/paperclip/references/routines.md:150 |
+| skill:skills/paperclip/references/routines.md:api-manual-only:167 | optional_agent_tool | skills/paperclip/references/routines.md:167 |
+| skill:skills/paperclip/references/routines.md:updating-and-deleting-triggers:179 | optional_agent_tool | skills/paperclip/references/routines.md:179 |
+| skill:skills/paperclip/references/routines.md:manual-run:196 | optional_agent_tool | skills/paperclip/references/routines.md:196 |
+| skill:skills/paperclip/references/routines.md:updating-a-routine:212 | optional_agent_tool | skills/paperclip/references/routines.md:212 |
+| skill:skills/paperclip/references/routines.md:reading-routines-and-runs:223 | optional_agent_tool | skills/paperclip/references/routines.md:223 |
+| skill:skills/paperclip/references/workflows.md:paperclip-workflow-playbooks:1 | optional_agent_tool | skills/paperclip/references/workflows.md:1 |
+| skill:skills/paperclip/references/workflows.md:project-setup-ceo-manager:7 | optional_agent_tool | skills/paperclip/references/workflows.md:7 |
+| skill:skills/paperclip/references/workflows.md:openclaw-invite-ceo:22 | optional_agent_tool | skills/paperclip/references/workflows.md:22 |
+| skill:skills/paperclip/references/workflows.md:setting-agent-instructions-path:50 | optional_agent_tool | skills/paperclip/references/workflows.md:50 |
+| skill:skills/paperclip/references/workflows.md:company-import-export:79 | optional_agent_tool | skills/paperclip/references/workflows.md:79 |
+| skill:skills/paperclip/references/workflows.md:self-test-playbook-app-level:106 | optional_agent_tool | skills/paperclip/references/workflows.md:106 |
+| skill:skills/paperclip/references/api-reference.md:paperclip-api-reference:1 | optional_agent_tool | skills/paperclip/references/api-reference.md:1 |
+| skill:skills/paperclip/references/api-reference.md:response-schemas:7 | optional_agent_tool | skills/paperclip/references/api-reference.md:7 |
+| skill:skills/paperclip/references/api-reference.md:agent-record-get-api-agents-me-or-get-api-agents-agentid:9 | optional_agent_tool | skills/paperclip/references/api-reference.md:9 |
+| skill:skills/paperclip/references/api-reference.md:company-portability:42 | optional_agent_tool | skills/paperclip/references/api-reference.md:42 |
+| skill:skills/paperclip/references/api-reference.md:issue-with-ancestors-get-api-issues-issueid:108 | optional_agent_tool | skills/paperclip/references/api-reference.md:108 |
+| skill:skills/paperclip/references/api-reference.md:issue-update-response-patch-api-issues-issueid:194 | optional_agent_tool | skills/paperclip/references/api-reference.md:194 |
+| skill:skills/paperclip/references/api-reference.md:blocker-diagnostics-get-api-issues-issueid-diagnostics-blockers:236 | control_plane_owned | skills/paperclip/references/api-reference.md:236 |
+| skill:skills/paperclip/references/api-reference.md:wake-diagnostics-get-api-issues-issueid-diagnostics-wakes:275 | control_plane_owned | skills/paperclip/references/api-reference.md:275 |
+| skill:skills/paperclip/references/api-reference.md:subtree-diagnostics-get-api-issues-issueid-diagnostics-subtree:319 | optional_agent_tool | skills/paperclip/references/api-reference.md:319 |
+| skill:skills/paperclip/references/api-reference.md:execution-policy-fields-on-an-issue:367 | optional_agent_tool | skills/paperclip/references/api-reference.md:367 |
+| skill:skills/paperclip/references/api-reference.md:cross-agent-review-gates:419 | always_agent_tool | skills/paperclip/references/api-reference.md:419 |
+| skill:skills/paperclip/references/api-reference.md:worked-example-ic-heartbeat:452 | optional_agent_tool | skills/paperclip/references/api-reference.md:452 |
+| skill:skills/paperclip/references/api-reference.md:1-identity-skip-if-already-in-context:457 | control_plane_owned | skills/paperclip/references/api-reference.md:457 |
+| skill:skills/paperclip/references/api-reference.md:2-check-inbox:461 | control_plane_owned | skills/paperclip/references/api-reference.md:461 |
+| skill:skills/paperclip/references/api-reference.md:3-already-have-issue-101-inprogress-highest-priority-continue-it:468 | optional_agent_tool | skills/paperclip/references/api-reference.md:468 |
+| skill:skills/paperclip/references/api-reference.md:4-do-the-actual-work-write-code-run-tests:475 | optional_agent_tool | skills/paperclip/references/api-reference.md:475 |
+| skill:skills/paperclip/references/api-reference.md:5-work-is-done-update-status-and-comment-in-one-call:477 | always_agent_tool | skills/paperclip/references/api-reference.md:477 |
+| skill:skills/paperclip/references/api-reference.md:6-still-have-time-checkout-the-next-task:481 | control_plane_owned | skills/paperclip/references/api-reference.md:481 |
+| skill:skills/paperclip/references/api-reference.md:7-made-partial-progress-not-done-yet-comment-and-exit:488 | always_agent_tool | skills/paperclip/references/api-reference.md:488 |
+| skill:skills/paperclip/references/api-reference.md:worked-example-report-a-board-user-s-mine-inbox:493 | control_plane_owned | skills/paperclip/references/api-reference.md:493 |
+| skill:skills/paperclip/references/api-reference.md:board-user-created-the-requesting-issue:498 | optional_agent_tool | skills/paperclip/references/api-reference.md:498 |
+| skill:skills/paperclip/references/api-reference.md:fetch-the-board-user-s-mine-inbox-issues:502 | control_plane_owned | skills/paperclip/references/api-reference.md:502 |
+| skill:skills/paperclip/references/api-reference.md:summarize-it-back-to-the-board-in-a-comment-or-document:516 | always_agent_tool | skills/paperclip/references/api-reference.md:516 |
+| skill:skills/paperclip/references/api-reference.md:worked-example-archive-a-resolved-inbox-item:521 | control_plane_owned | skills/paperclip/references/api-reference.md:521 |
+| skill:skills/paperclip/references/api-reference.md:the-responsible-user-s-id-is-resolved-from-the-authenticated-agent-run:526 | optional_agent_tool | skills/paperclip/references/api-reference.md:526 |
+| skill:skills/paperclip/references/api-reference.md:reverse-the-archive-if-it-was-premature-or-no-longer-desired:535 | optional_agent_tool | skills/paperclip/references/api-reference.md:535 |
+| skill:skills/paperclip/references/api-reference.md:worked-example-reviewer-approver-heartbeat:545 | always_agent_tool | skills/paperclip/references/api-reference.md:545 |
+| skill:skills/paperclip/references/api-reference.md:worked-example-manager-heartbeat:584 | optional_agent_tool | skills/paperclip/references/api-reference.md:584 |
+| skill:skills/paperclip/references/api-reference.md:1-identity-skip-if-already-in-context:587 | control_plane_owned | skills/paperclip/references/api-reference.md:587 |
+| skill:skills/paperclip/references/api-reference.md:2-check-team-status:591 | optional_agent_tool | skills/paperclip/references/api-reference.md:591 |
+| skill:skills/paperclip/references/api-reference.md:3-agent-42-is-blocked-read-comments:598 | control_plane_owned | skills/paperclip/references/api-reference.md:598 |
+| skill:skills/paperclip/references/api-reference.md:4-unblock-reassign-and-comment:602 | control_plane_owned | skills/paperclip/references/api-reference.md:602 |
+| skill:skills/paperclip/references/api-reference.md:5-check-own-assignments:606 | optional_agent_tool | skills/paperclip/references/api-reference.md:606 |
+| skill:skills/paperclip/references/api-reference.md:6-create-subtasks-and-delegate:613 | optional_agent_tool | skills/paperclip/references/api-reference.md:613 |
+| skill:skills/paperclip/references/api-reference.md:load-tests-depend-on-caching-layer-being-done-first-paperclip-will-auto-wake-agent-55-when-the-blocker-resolves:619 | control_plane_owned | skills/paperclip/references/api-reference.md:619 |
+| skill:skills/paperclip/references/api-reference.md:7-dashboard-for-health-check:624 | optional_agent_tool | skills/paperclip/references/api-reference.md:624 |
+| skill:skills/paperclip/references/api-reference.md:comments-and-mentions:630 | always_agent_tool | skills/paperclip/references/api-reference.md:630 |
+| skill:skills/paperclip/references/api-reference.md:update:637 | optional_agent_tool | skills/paperclip/references/api-reference.md:637 |
+| skill:skills/paperclip/references/api-reference.md:cross-team-work-and-delegation:675 | optional_agent_tool | skills/paperclip/references/api-reference.md:675 |
+| skill:skills/paperclip/references/api-reference.md:receiving-cross-team-work:679 | optional_agent_tool | skills/paperclip/references/api-reference.md:679 |
+| skill:skills/paperclip/references/api-reference.md:escalation:689 | optional_agent_tool | skills/paperclip/references/api-reference.md:689 |
+| skill:skills/paperclip/references/api-reference.md:company-context:699 | optional_agent_tool | skills/paperclip/references/api-reference.md:699 |
+| skill:skills/paperclip/references/api-reference.md:company-branding-ceo-board:711 | optional_agent_tool | skills/paperclip/references/api-reference.md:711 |
+| skill:skills/paperclip/references/api-reference.md:openclaw-invite-prompt-ceo:731 | optional_agent_tool | skills/paperclip/references/api-reference.md:731 |
+| skill:skills/paperclip/references/api-reference.md:setting-agent-instructions-path:750 | optional_agent_tool | skills/paperclip/references/api-reference.md:750 |
+| skill:skills/paperclip/references/api-reference.md:project-setup-create-workspace:783 | optional_agent_tool | skills/paperclip/references/api-reference.md:783 |
+| skill:skills/paperclip/references/api-reference.md:option-a-one-call-create-with-workspace:787 | optional_agent_tool | skills/paperclip/references/api-reference.md:787 |
+| skill:skills/paperclip/references/api-reference.md:option-b-two-calls-project-first-then-workspace:806 | optional_agent_tool | skills/paperclip/references/api-reference.md:806 |
+| skill:skills/paperclip/references/api-reference.md:governance-and-approvals:835 | optional_agent_tool | skills/paperclip/references/api-reference.md:835 |
+| skill:skills/paperclip/references/api-reference.md:requesting-a-hire-management-only:839 | optional_agent_tool | skills/paperclip/references/api-reference.md:839 |
+| skill:skills/paperclip/references/api-reference.md:ceo-strategy-approval:859 | optional_agent_tool | skills/paperclip/references/api-reference.md:859 |
+| skill:skills/paperclip/references/api-reference.md:issue-thread-confirmations:868 | always_agent_tool | skills/paperclip/references/api-reference.md:868 |
+| skill:skills/paperclip/references/api-reference.md:checkbox-confirmations:926 | always_agent_tool | skills/paperclip/references/api-reference.md:926 |
+| skill:skills/paperclip/references/api-reference.md:item-verdict-requests:1041 | optional_agent_tool | skills/paperclip/references/api-reference.md:1041 |
+| skill:skills/paperclip/references/api-reference.md:checking-approval-status:1151 | optional_agent_tool | skills/paperclip/references/api-reference.md:1151 |
+| skill:skills/paperclip/references/api-reference.md:approval-follow-up-requesting-agent:1157 | always_agent_tool | skills/paperclip/references/api-reference.md:1157 |
+| skill:skills/paperclip/references/api-reference.md:issue-lifecycle:1175 | always_agent_tool | skills/paperclip/references/api-reference.md:1175 |
+| skill:skills/paperclip/references/api-reference.md:error-handling:1205 | control_plane_owned | skills/paperclip/references/api-reference.md:1205 |
+| skill:skills/paperclip/references/api-reference.md:full-api-reference:1219 | optional_agent_tool | skills/paperclip/references/api-reference.md:1219 |
+| skill:skills/paperclip/references/api-reference.md:agents:1221 | optional_agent_tool | skills/paperclip/references/api-reference.md:1221 |
+| skill:skills/paperclip/references/api-reference.md:issues-tasks:1242 | optional_agent_tool | skills/paperclip/references/api-reference.md:1242 |
+| skill:skills/paperclip/references/api-reference.md:companies-projects-goals:1282 | optional_agent_tool | skills/paperclip/references/api-reference.md:1282 |
+| skill:skills/paperclip/references/api-reference.md:routines:1306 | optional_agent_tool | skills/paperclip/references/api-reference.md:1306 |
+| skill:skills/paperclip/references/api-reference.md:approvals-costs-activity-dashboard:1322 | optional_agent_tool | skills/paperclip/references/api-reference.md:1322 |
+| skill:skills/paperclip/references/api-reference.md:secrets:1344 | optional_agent_tool | skills/paperclip/references/api-reference.md:1344 |
+| skill:skills/paperclip/references/api-reference.md:agent-secret-proposals:1357 | optional_agent_tool | skills/paperclip/references/api-reference.md:1357 |
+| skill:skills/paperclip/references/api-reference.md:agent-secret-access:1457 | optional_agent_tool | skills/paperclip/references/api-reference.md:1457 |
+| skill:skills/paperclip/references/api-reference.md:common-mistakes:1497 | optional_agent_tool | skills/paperclip/references/api-reference.md:1497 |
+
+## Legacy MCP Alias Index
+
+This is a compatibility/traceability index, not a tool catalog. “Inherited disposition” is shown only to make the normative target easy to audit.
+
+| Legacy MCP name | Folded into normative row | Inherited disposition | Source anchor |
+| --- | --- | --- | --- |
+| paperclipMe | eval:hb-inbox-lite-01 | control_plane_owned | packages/mcp-server/src/tools.ts:290 |
+| paperclipInboxLite | eval:hb-inbox-lite-01 | control_plane_owned | packages/mcp-server/src/tools.ts:296 |
+| paperclipListAgents | eval:rf-api-mgr-heartbeat-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:302 |
+| paperclipListSkills | eval:rf-cskill-audit-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:308 |
+| paperclipGetAgent | eval:rf-api-mgr-heartbeat-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:314 |
+| paperclipListIssues | eval:se-q-filters-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:323 |
+| paperclipGetIssue | eval:se-get-issue-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:338 |
+| paperclipGetHeartbeatContext | eval:hb-context-01 | control_plane_owned | packages/mcp-server/src/tools.ts:344 |
+| paperclipListComments | eval:se-get-issue-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:353 |
+| paperclipGetComment | eval:hb-wake-comment-01 | control_plane_owned | packages/mcp-server/src/tools.ts:366 |
+| paperclipListIssueApprovals | eval:ap-board-approval-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:373 |
+| paperclipListDocuments | eval:dp-base-revision-01 | always_agent_tool | packages/mcp-server/src/tools.ts:379 |
+| paperclipGetDocument | eval:dp-base-revision-01 | always_agent_tool | packages/mcp-server/src/tools.ts:385 |
+| paperclipListDocumentRevisions | eval:dp-base-revision-01 | always_agent_tool | packages/mcp-server/src/tools.ts:392 |
+| paperclipListProjects | eval:rf-wf-project-setup-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:402 |
+| paperclipGetProject | eval:rf-wf-project-setup-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:408 |
+| paperclipGetIssueWorkspaceRuntime | eval:rf-iws-start-url-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:417 |
+| paperclipControlIssueWorkspaceServices | eval:rf-iws-start-url-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:423 |
+| paperclipWaitForIssueWorkspaceService | eval:rf-iws-target-restart-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:440 |
+| paperclipListGoals | eval:su-parent-goal-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:466 |
+| paperclipGetGoal | eval:su-parent-goal-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:472 |
+| paperclipListApprovals | eval:ap-approval-wake-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:478 |
+| paperclipCreateApproval | eval:ap-board-approval-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:487 |
+| paperclipGetApproval | eval:ap-approval-wake-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:496 |
+| paperclipGetApprovalIssues | eval:ap-approval-wake-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:502 |
+| paperclipListApprovalComments | eval:ap-approval-deny-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:508 |
+| paperclipCreateIssue | eval:su-parent-goal-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:514 |
+| paperclipUpdateIssue | eval:st-done-comment-01 | always_agent_tool | packages/mcp-server/src/tools.ts:521 |
+| paperclipCheckoutIssue | eval:co-body-contract-01 | control_plane_owned | packages/mcp-server/src/tools.ts:528 |
+| paperclipReleaseIssue | eval:er-release-01 | control_plane_owned | packages/mcp-server/src/tools.ts:540 |
+| paperclipAddComment | eval:cm-multiline-01 | always_agent_tool | packages/mcp-server/src/tools.ts:546 |
+| paperclipSuggestTasks | eval:ix-suggest-tasks-01 | always_agent_tool | packages/mcp-server/src/tools.ts:553 |
+| paperclipAskUserQuestions | eval:ix-questions-01 | always_agent_tool | packages/mcp-server/src/tools.ts:565 |
+| paperclipRequestConfirmation | eval:ix-confirmation-plan-01 | always_agent_tool | packages/mcp-server/src/tools.ts:577 |
+| paperclipRequestCheckboxConfirmation | eval:ix-checkbox-01 | always_agent_tool | packages/mcp-server/src/tools.ts:589 |
+| paperclipUpsertIssueDocument | eval:dp-plan-doc-01 | always_agent_tool | packages/mcp-server/src/tools.ts:601 |
+| paperclipRestoreIssueDocumentRevision | eval:dp-base-revision-01 | always_agent_tool | packages/mcp-server/src/tools.ts:612 |
+| paperclipLinkIssueApproval | eval:ap-board-approval-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:627 |
+| paperclipUnlinkIssueApproval | eval:ap-board-approval-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:636 |
+| paperclipApprovalDecision | eval:ap-approval-wake-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:646 |
+| paperclipAddApprovalComment | eval:ap-approval-deny-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:668 |
+| paperclipApiRequest | eval:rf-api-404-report-01 | optional_agent_tool | packages/mcp-server/src/tools.ts:677 |
diff --git a/packages/paperclip-runner/docs/capability-disposition.md b/packages/paperclip-runner/docs/capability-disposition.md
new file mode 100644
index 0000000000..62f281797b
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-disposition.md
@@ -0,0 +1,84 @@
+# Capability Capability Disposition
+
+This page explains how Capability classifies every Paperclip capability. The
+classification itself lives in a generated file,
+[the Capability contract](capability-contract.md), which
+carries a "DO NOT EDIT" header and is rewritten by `generate:capability-inventory`.
+Read this page to understand what the generated table means; read the table for
+the authoritative rows.
+
+## Normative sources
+
+Only two sources are normative:
+
+1. The Paperclip skill and its seven references (`SKILL.md` plus
+ `references/*.md`), contributing **152 headings**.
+2. The Paperclip Evals corpus, contributing **106 cases across 16 groups**.
+
+Together these produce **258 normative rows**. The legacy Paperclip MCP tool
+surface (**41 tools**) is not a production capability surface; each MCP name is
+folded one-to-one into a normative eval row as a traceability alias and inherits
+that row's disposition. The contract prints the alias index only so the
+normative target is easy to audit.
+
+## The three dispositions
+
+Every capability is classified as exactly one of:
+
+- **`control_plane_owned`** — the control plane performs this; no agent tool
+ exists for it. Checkout, inbox resolution, blocker diagnostics, and status
+ arbitration are control-plane-owned. In the explorer these appear as
+ control-plane actions labelled "no agent tool exists for this," never as a
+ callable tool.
+- **`always_agent_tool`** — the agent always has this as a semantic tool. Adding
+ a comment, writing a document, opening an interaction, and registering a
+ deliverable are always-agent-tool capabilities.
+- **`optional_agent_tool`** — the agent may have this tool, but only when a
+ grant unlocks it. Absent the grant, the capability is not exposed and calling
+ it is denied. See [authorization and exposure](capability-authorization-and-exposure.md).
+
+The disposition is what makes the boundary legible: it states, per capability,
+whether an agent can act, must wait for the control plane, or needs a grant
+first.
+
+## Reading the generated contract
+
+The contract has three parts:
+
+- **Baseline counts** — the heading, case, row, and alias totals above, plus the
+ per-group case table.
+- **Skill / reference rows** — one row per heading, its primary disposition, and
+ its `file:line` source anchor.
+- **Legacy MCP alias index** — each MCP name, the normative row it folds into,
+ the inherited disposition, and its `packages/mcp-server/src/tools.ts` anchor.
+
+## Regenerating and checking
+
+Generation reads the live in-repo skill/reference sources, the legacy MCP tool
+source, and the Paperclip Evals corpus, so it **requires** the external eval
+repository (via `PAPERCLIP_EVALS_ROOT` or a known local path) and is not part of
+the offline path. Checking and testing read only the checked-in derivatives
+under `spec/capability/` and need no external repository.
+
+```sh
+# Rewrite every generated file. Requires the external Paperclip Evals corpus.
+pnpm --filter @paperclipai/paperclip-runner generate:capability-inventory
+
+# Validate counts, uniqueness, normative dispositions, one-to-one MCP folds,
+# required fields, and generated-file drift. Offline; no external eval repo.
+pnpm --filter @paperclipai/paperclip-runner check:capability-inventory
+
+# Prove the validator rejects an independent MCP classification and rejects
+# missing, duplicate, or unknown MCP folds. Offline.
+pnpm --filter @paperclipai/paperclip-runner test:capability-inventory
+```
+
+`check:capability-inventory` diffs the checked-in generated files against what the
+live in-repo sources imply and fails on any drift or stale anchor, so the
+contract cannot silently fall out of sync.
+
+## Related
+
+- [Semantic tool catalog](capability-semantic-tools.md)
+- [Authorization and exposure](capability-authorization-and-exposure.md)
+- [Eval conformance](capability-eval-conformance.md)
diff --git a/packages/paperclip-runner/docs/capability-eval-conformance.md b/packages/paperclip-runner/docs/capability-eval-conformance.md
new file mode 100644
index 0000000000..7e85de9665
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-eval-conformance.md
@@ -0,0 +1,90 @@
+# Capability Eval-Derived Conformance
+
+Capability turns the Paperclip Evals corpus into an executable, offline
+conformance suite. The suite derives its cases from the checked-in Capability
+traceability derivative (`spec/capability/eval-traceability.yaml`) — it does
+**not** clone or read an external eval repository, and it starts only the
+in-process mock control plane. It hard-fails unless the derivative declares
+schema version 2, exactly 106 rows, exactly 16 groups, and unique case IDs.
+
+Sources: `src/conformance/capability-eval-suite.ts` and its test
+`src/conformance/capability-eval-suite.test.ts`; the reporter
+`scripts/run-capability-eval-suite.mjs`.
+
+## The corpus
+
+- **106 cases across 16 groups.** Per-group counts: hb 5, co 6, st 8, cm 6,
+ se 4, su 4, bl 5, dp 3, ix 9, ap 6, ar 4, er 9, rf 22, mh 4, rs 3, wk 8.
+- Each case declares an actor role, a task mode, an input scenario, and the
+ expected outcome, all bound to a row in
+ [the capability contract](capability-contract.md).
+
+## Assertion classes
+
+Every case belongs to one assertion class, and each class checks a different
+kind of invariant:
+
+- **`agent_tool_contract`** — a semantic tool call produces the expected typed
+ effect and state change.
+- **`authorization_policy`** — a capability is exposed, denied, or unlocked by a
+ grant exactly as its disposition requires.
+- **`control_plane_invariant`** — a control-plane-owned action happens without
+ any agent tool, and no tool can perform it.
+- **`combined_multi_hop`** — a sequence of operations across turns produces the
+ expected cumulative state and respects forbidden-operation rules.
+- **`restraint_no_call`** — the correct behavior is to make **no** call; the
+ case passes only if the agent deliberately does nothing further.
+
+## Fake-agent matrix and bounded Codex sample
+
+The suite executes each case with a deterministic fake agent whose plan is
+fixed by a fixture seed, so repeat runs are byte-identical. Optional-tool rows
+are run twice — once with the unlocking grants (must be allowed and must mutate
+state) and once ungranted (must be absent or denied with no state change);
+control-plane-owned rows remain absent in every configuration; and
+`restraint_no_call` rows must produce an empty state diff.
+
+The fake-agent surface is 14 always-agent tools plus 4 optional tools unlocked
+by four seed grants (`discovery:tasks:read`, `discovery:agents:read`,
+`delegation:tasks:create`, `governance:approvals:request`) — **18 operations**.
+The suite binds that surface through both the fake-agent and Codex bindings and
+asserts the two operation lists are byte-identical (**18/18**).
+
+A **bounded Codex binding sample** picks one representative case from nine
+groups (`hb`, `dp`, `bl`, `ap`, `ar`, `ix`, `mh`, `rs`, `wk`) and checks that
+`checkout_task` is absent from the Codex surface while every other sampled
+operation is present. It is an offline parity check; no real Codex or network is
+contacted, and the browser explorer holds no credential.
+
+## Running it
+
+```sh
+# Run the 106-case suite in-process (one vitest file drives all cases).
+pnpm --filter @paperclipai/paperclip-runner test:capability-evals
+
+# Build the public surface and write the parity report with per-group counts,
+# assertion classes, the fake-agent matrix, the bounded Codex sample, and the
+# semantic-operation execution counts.
+pnpm --filter @paperclipai/paperclip-runner report:capability-evals
+```
+
+The reporter writes `.paperclip-local/evidence/capability/eval-parity-report.{json,md}`.
+
+Run the bounded provider conformance matrix separately. It creates exactly one
+real Codex turn for each of the 16 checked-in eval groups while retaining the
+in-process mock control plane:
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner report:capability-live-evals
+```
+Each failure carries its case ID, assertion class, semantic operation,
+authorization decision, and final state diff. The report is generated on demand
+and is not committed; delete it before running `docs:validate` (it carries no
+OKF frontmatter). See the
+[verification commands reference](capability-verification-commands.md).
+
+## Related
+
+- [Capability disposition](capability-disposition.md)
+- [Semantic tool catalog](capability-semantic-tools.md)
+- [Scenario explorer](capability-scenario-explorer.md)
diff --git a/packages/paperclip-runner/docs/capability-execution-modes.md b/packages/paperclip-runner/docs/capability-execution-modes.md
new file mode 100644
index 0000000000..b4fe546520
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-execution-modes.md
@@ -0,0 +1,87 @@
+# Capability execution modes and identity
+
+Capability runs the same semantic tool catalog, authorization engine, and mock
+`ControlPlanePort` under two agent modes. Whichever mode is active, the
+surface always names three separate actors so a reader never mistakes one for
+another.
+
+## Three actors, always separate
+
+| Actor | What it is in Capability | What it is **not** |
+| --- | --- | --- |
+| **Real Codex** | A real Codex app-server session driving the turn loop. The agent has no Paperclip skill; it sees only the semantic tools the scenario exposes. | Not a scripted stand-in, and not a Paperclip-aware agent. |
+| **Real runnerd** | The real package-local `paperclip-runnerd` binary. It owns the Codex child process group and proxies newline-delimited JSON-RPC over stdio. | Not an in-process fake and not the production Paperclip runtime. |
+| **Mock Paperclip** | The deterministic in-process `ControlPlanePort` adapter. It holds every issue, comment, document, interaction, approval, and audit record as mock state. | Not the Paperclip control plane, database, or API. No request leaves the process for a Paperclip service. |
+
+The live surface renders a `Real Codex`, `Real runnerd`, and `Mock Paperclip`
+marker at all times. Mock records carry an `MCK-` identifier prefix so a mock
+issue is never confused with a real `PAP-` issue.
+
+## Two agent modes
+
+### Scripted (deterministic) — fake mode
+
+The scripted driver replays a recorded conversation over the same mock core.
+It needs no provider credential, no runnerd process, and no network after
+`pnpm install`. It is the mode used for:
+
+- the 106-case conformance suite and its bounded parity report;
+- the byte-stable screenshot matrix;
+- replay of a recorded session.
+
+Because it is fully deterministic, two runs of the same route produce
+byte-identical output. **A scripted (`mode=fake`) artifact cannot satisfy a
+live acceptance criterion** — it proves the contract, not a real provider turn.
+
+### Codex (bounded) — live mode
+
+Live mode starts a real `paperclip-runnerd` process and a real Codex
+app-server session. The package server — never the browser — owns the Codex
+process, the session, the tool loop, and provider authentication. The browser
+receives no provider, runner, mock-control-plane, or real-control-plane
+credential.
+
+Live mode requires a locally authenticated Codex installation. When that relay
+is not present, the UI keeps Codex disabled with a named reason and scripted
+mode stays available. Live mode is the only mode eligible for the final
+Revision 3/4 acceptance criteria that require a real provider turn.
+
+## Mode-independent controls
+
+These behave the same in either mode because they operate on the mock core and
+the runner session, not on the agent:
+
+- **Reset** restores the original clean mock seed under new run/session
+ authority and rotates or clears the old session authority. It does not
+ affect another browser session's state.
+- **Stop** cancels only the active turn and reaps the runnerd process group;
+ the session and its transcript survive. Cleanup evidence shows no abandoned
+ child process and no active session afterward.
+- **Replay** reproduces a recorded canonical timeline. It is always scripted
+ and always labelled as fake-derived, even inside a session that also ran a
+ live turn.
+- **Refresh / reconnect** restores the same durable session, pending
+ interaction, transcript, and mock state. Reconnect starts a fresh runnerd and
+ Codex app-server and resumes the persisted provider thread; the session does
+ not restart because the browser reconnected.
+
+## Two entry points, one live mode
+
+Live mode is reachable from both primary surfaces:
+
+- the preset scenario explorer, with `mode=live` on a chosen scenario;
+- the [clean-room chat](capability-clean-room-chat.md) at `#/chat`, which has no
+ other mode. It starts blank on a freshly minted mock tenant, pins `mode=live`
+ in the route, and reports a failure to start real Codex rather than falling
+ back to a fixture or a recording.
+
+## Eligibility summary
+
+| Evidence | Eligible for |
+| --- | --- |
+| Scripted / `mode=fake` run | 106-case conformance, replay determinism, screenshot matrix, contract proofs |
+| Live Codex run on the final build | Revision 3/4 final acceptance criteria that require a real provider turn |
+| Revision 2 preview (build `da0d32d74a`, historical URL) | Historical comparison only — **not** eligible for final acceptance |
+
+See the [live runnerd and Codex loop](capability-live-runnerd-codex.md) reference
+for the session API and verification commands behind each row above.
diff --git a/packages/paperclip-runner/docs/capability-future-binding-boundary.md b/packages/paperclip-runner/docs/capability-future-binding-boundary.md
new file mode 100644
index 0000000000..48d5a685e2
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-future-binding-boundary.md
@@ -0,0 +1,51 @@
+# Capability Future Binding Boundary (future upload integration / ACPX)
+
+Capability is a package-local model. It does not integrate the runner into
+Paperclip, and it must not. This page states exactly what Capability defers to
+future upload integration and what clean seam it preserves so the deferral stays cheap.
+
+## What Capability does not do
+
+Nothing in Capability contacts a real Paperclip control plane, database, ACPX
+session, or provider credential. The 106-case conformance suite and the browser
+explorer run entirely against the in-process
+[mock ControlPlanePort](capability-mock-control-plane-port.md) and checked-in
+fixtures. The [forbidden-imports boundary](architecture.md) rejects any import
+that crosses into `server/`, `ui/`, `cli/`, `@paperclipai/db`, or other
+Paperclip workspace internals.
+
+## The preserved seam
+
+The dependency arrow always points from an implementation toward a contract.
+The package owns two contracts a future consumer implements:
+
+- **`ControlPlanePort`** — the narrow surface through which a runner opens a
+ run, appends ordered events, records semantic operations, and submits a
+ terminal result. Capability injects the *mock* adapter behind this port; future upload integration
+ injects a real one. The catalog, authorization engine, and conformance suite
+ bind to the port, not to any adapter.
+- **`NativeSessionBackend`** — the normalized session surface for a future
+ control-plane consumer.
+
+Because the port is the only coupling point, future upload integration replaces the adapter
+without touching the tool catalog, the authorization rules, or the eval-derived
+conformance suite. The package remains independently buildable, testable, and
+runnable against the mock adapter after the real one exists.
+
+## What future upload integration (ACPX) will bind
+
+future upload integration binds a real Paperclip `ControlPlanePort` implementation behind the same
+seam so the semantic tools and authorization engine act against a live control
+plane instead of the mock. That work is out of scope here and requires separate
+CTO approval at the Capability checkpoint (`TASK-16908`). Until then:
+
+- ACPX is future upload integration, not Capability.
+- No Capability documentation claims future upload integration capability.
+- Real integration is a separately reviewed phase, consistent with the
+ [future integration rule](architecture.md) recorded for every prior phase.
+
+## Related
+
+- [Architecture and dependency boundary](architecture.md)
+- [Mock ControlPlanePort](capability-mock-control-plane-port.md)
+- [Capability disposition](capability-disposition.md)
diff --git a/packages/paperclip-runner/docs/capability-issue-thread-ui.md b/packages/paperclip-runner/docs/capability-issue-thread-ui.md
new file mode 100644
index 0000000000..dc1bf9c683
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-issue-thread-ui.md
@@ -0,0 +1,224 @@
+# Capability Paperclip-style issue-thread UI
+
+Capability renders the mock Paperclip issue as a native issue thread. A board user
+reads the thread, answers typed interactions inline, and inspects the evidence
+behind every mock mutation. The implementation follows the binding
+[Capability issue-thread UX contract](design/capability-issue-thread-ux-contract.md).
+
+The same shell hosts a second primary path: the
+[clean-room live chat](capability-clean-room-chat.md) at `#/chat`, which starts a
+blank thread on a freshly minted mock tenant instead of a preset scenario. It
+reuses every surface described below — header, thread, composer, and Evidence
+panel — and differs only in what the session is seeded with and in refusing any
+non-live mode.
+
+## Authority boundary
+
+```text
+browser -> package server -> CapabilityLiveSession -> paperclip-runnerd -> codex
+ (projection) CapabilitySemanticDispatcher -> ControlPlanePort mock
+```
+
+The browser holds no authority. It renders one shape,
+`CapabilityIssueThreadSnapshot` (`src/issue-thread/types.ts`), and computes no
+claim, policy decision, state diff, or parity verdict. Two producers emit that
+shape:
+
+- `capabilityIssueThreadFixture(slug)` — deterministic `fake` snapshots used by the
+ screenshot matrix and the browser suite.
+- `projectCapabilityIssueThread({ snapshot })` — the live projection. It runs in the
+ package server and rearranges durable records only: the live session's
+ transcript, evidence entries, semantic authorization records, and the
+ serialized mock state.
+
+The one browser-initiated mock mutation is an interaction response. It posts to
+`POST /api/capability/ui/interaction`; `CapabilityLiveSession.resolveInteraction` stores
+the typed response in the mock control plane **before** resuming the same Codex
+thread, so the card only leaves `submitting` on server acknowledgement.
+
+No provider, runner, or control-plane credential reaches the page. Redacted
+fields render as `••• redacted` with the redaction rule name, mock issues use
+the reserved `MCK-` prefix, and no real Paperclip URL is ever rendered.
+
+## What the browser is allowed to see
+
+The projection is an internal shape. What ships is the published DTO,
+`toCapabilityPublicThreadView` (`src/issue-thread/public-view.ts`), which copies the
+view field by field — so a field added to the projection, or to any record it
+passes through, cannot reach a browser until it is listed there. Every response
+path uses it: interim stream frames, the settled payload, and reconnect or
+replay replies alike.
+
+Three narrowings stack, so no single omission opens a disclosure path:
+
+1. **At record time.** `redactCapabilityEvidenceData`
+ (`src/live/evidence-redaction.ts`) is the only way an evidence entry enters
+ a live session. Provider notifications are reduced to a coarse category,
+ provider diagnostics to the fact that one occurred, tool arguments to the
+ catalog-declared field names they used, and tool results to the outcome,
+ revisions, and mock entity refs the UI resolves into cards. Provider thread
+ and session identity, model and token metadata, and raw tool payloads are
+ never retained, so no later reader, frame, or log can republish them.
+2. **In the projection.** `Runner & events` details are composed from the
+ redacted record rather than stringified from it, and `Calls & results` names
+ the operation and its field count instead of echoing arguments.
+3. **In the DTO.** Provider-authored turn and call identifiers are replaced by
+ in-view aliases (`turn-1`, `call-1`) that stay consistent across anchors,
+ evidence refs, and successive frames of one turn, and any value the caller
+ declares withheld is scrubbed from the encoded result.
+
+A streamed turn that fails answers with a code and fixed operator copy; the
+underlying message stays server side, because provider text can quote prompts
+and paths.
+
+## Session capability
+
+Every session-scoped route is bound to a per-browser capability. The server
+mints one on session creation, stores only its SHA-256 with the session record,
+sets it as an `HttpOnly; SameSite=Strict` cookie, and compares it in constant
+time on every read and mutation — message, reconnect, interaction, stop, reset,
+and new chat. A valid session id presented without its capability is answered
+`404`, exactly like an id that never existed, so an unauthorized caller cannot
+tell a live session from a dead one.
+
+Rotation belongs to the actions that start something new — `New chat`, a scenario
+POST, reset. Each revokes the caller's existing bindings before issuing the
+replacement cookie. Reopening a page whose stored id is simply gone mints a
+session under the capability the browser already holds instead, because rotating
+there would make two tabs of one surface revoke each other on every load while
+protecting nothing: one browser is one principal, and cross-browser denial rests
+on the binding rather than on how often the value changes.
+
+The two surfaces use separate cookie names (`paperclip_capability_issue`,
+`paperclip_capability_chat`) because they are separate pages of one origin: a single
+name would make opening the explorer revoke the clean room. A script that drives
+these routes has to behave like one browser — `scripts/capability-cookie-jar.mjs` is
+what the smoke scripts use for that.
+
+## Surfaces
+
+- **Header** — three identity chips (`Real Codex` / `Fake agent` / `Replay`,
+ `Real runnerd` / `In-process runner`, and `Mock Paperclip` in every mode),
+ status, priority, run state, and the Scenario/Replay/Reset/Stop controls.
+ `data-session-mode` carries the mode as data, never as styling.
+- **Thread** — turn groups binding the contract's T1–T11 item types: user
+ messages, model prose, durable progress comments marked
+ `Recorded to mock thread`, collapsed tool strips, interaction cards, document
+ revision cards, deliverables, delegation cards, terminal dispositions, typed
+ denials, and muted system notices.
+- **Composer** — six mutually exclusive states behind `data-composer-state`:
+ `ready`, `sending`, `streaming` (input stays editable to steer; Stop is
+ primary), `waiting`, `reconnecting`, and `disabled`. Drafts survive refresh.
+- **Evidence panel** — eight accordion sections in fixed order: Tools exposed,
+ Calls & results, Authorization, Control plane, Runner & events, State diff,
+ Traceability, Parity. The Tools section groups `Agent tool — always`,
+ `Agent tool — granted` (with its grant), and a separated
+ `Control plane (not exposed to the agent)` list, because what the model
+ *cannot* call is first-class evidence. Every strip, denial, and card deep-links
+ into the matching record, and each record links back to its thread anchor.
+ Live sessions additionally put a six-tab DevTools inspector above these
+ sections: revision timeline, complete browser-safe company state, structural
+ diff, protocol records, runtime, and authority. The inspector can pause live
+ following, export redacted JSON, and fork a retained revision.
+
+The panel is collapsed by default and resizable between 320px and 640px with a
+keyboard-operable splitter. Below 1100px it becomes an overlay sheet that
+Escape dismisses; below 768px the page switches to a `Thread` / `Evidence`
+segmented control, with Stop kept outside the `⋯` menu while a turn is active.
+Closing the panel by either route hands focus back to the visible control that
+owns it.
+
+- **Replay strip** — in `mode=replay` a progress strip pins under the header
+ with `Step back`, `Next turn`, and `Play all`. `?at=` is the single
+ source of truth for the parked ordinal, so the three controls and the deep
+ link all move the same value; `Play all` advances one ordinal every 800 ms
+ and parks itself at the end of the recording.
+
+## Routes
+
+```text
+#/issue/?shot=&panel=&rec=&at=&seg=thread|evidence&mode=live
+```
+
+- `shot` seeds one of the twelve deterministic `fake` states.
+- `mode=live` opts into the package session server; the default is `fake`.
+- `capture=1` freezes animation, caret, and smooth scrolling for screenshots.
+- The root element sets `data-thread-state="settled"` once hydration, fixture
+ load, and auto-scroll finish. Tooling waits for that attribute, never a
+ timeout.
+
+## Commands
+
+```sh
+# Deterministic fake-mode app (no provider process)
+pnpm --filter @paperclipai/paperclip-runner console:issue-thread
+
+# Focused browser suite, including the axe gate on all 12 slugs × 2 viewports
+pnpm --filter @paperclipai/paperclip-runner test:browser:scenarios
+
+# View-model and live-projection unit tests
+pnpm --filter @paperclipai/paperclip-runner exec vitest run src/issue-thread
+
+# Screenshot matrix (12 slugs × 2 viewports) and its byte-stability check
+# Recorded evidence generation is deferred from this release.
+pnpm --filter @paperclipai/paperclip-runner check:capability:ui
+
+# Real runnerd + real Codex through the same HTTP routes the browser uses
+pnpm --filter @paperclipai/paperclip-runner smoke:capability:ui
+# Recorded evidence generation is deferred from this release.
+```
+
+Hosts without the Playwright chromium system libraries can either run
+`pnpm --filter @paperclipai/paperclip-runner verify:rootless` or set
+`PAPERCLIP_RUNNER_CHROMIUM_PATH` to a preinstalled Chromium.
+
+The committed PNGs are pinned to the Chromium build listed in
+`.paperclip-local/evidence/capability/ui/index.md`, so `check:capability:ui` needs that same
+browser. Point `PAPERCLIP_RUNNER_CHROMIUM_PATH` at the recorded browser before
+comparing — and when that path is the agent-browser wrapper, also set
+`PAPERCLIP_CHROMIUM_BIN` to the exact binary, because the wrapper otherwise
+picks the newest installed Playwright Chromium. The issue-thread bundle
+self-hosts its Latin Inter and DejaVu Sans Mono WOFF2 faces plus tiny status-glyph
+subsets, so host fontconfig directories do not participate in capture. The recorder probes the
+package-specific bundled families and refuses to record or compare when either
+face is absent or fails to load; the drift report prints the Chromium version
+and bundled-font probe it recorded with.
+
+## Accessibility
+
+The suite enforces the contract's blocking gate: axe reports zero serious or
+critical WCAG 2.1 A/AA violations on every screenshot route at both viewports.
+Structure is one `h1`, `header`/`main`/`complementary` landmarks, a `form`
+composer, `section` interaction cards labelled by their prompt, and tool strips
+as disclosure buttons with `aria-expanded`. Every state chip pairs color with a
+glyph and text, all actionable controls clear 44×44 CSS px on mobile, and
+`prefers-reduced-motion` disables the pulse dot, banner slide, and smooth
+scrolling.
+
+Focus management (§9.2) is covered by named regressions in the browser suite:
+
+- Opening Evidence moves focus to its heading; closing it — with the `Close`
+ button or with Escape on the overlay sheet — returns focus to the toggle.
+- Resolving an interaction card moves focus to the card's state chip. The
+ controls the user just operated unmount on resolve, so without this the
+ keyboard caret drops to `body` at the moment the card changes.
+- The `waiting` composer's anchor moves focus to the pending card's first
+ control.
+
+## Contract deviations
+
+One deviation is recorded against the Capability contract:
+
+- **§5 expired-family dimming.** The contract asks for a 60% opacity body on
+ `stale_target` and the other expired outcomes. A literal opacity drops that
+ card's text to ~3.2:1 and fails the blocking axe gate in §9.7. The dim is
+ implemented as a recessed surface plus muted text that still clears 4.5:1.
+
+## Determinism notes
+
+Fake-mode fixtures render from authored data with a fixed clock, so two captures
+of a slug from a clean checkout are byte-identical. Live sessions are not
+byte-stable — a real model writes their prose — so live evidence lives in
+`.paperclip-local/evidence/capability/ui-live/` and is excluded from the determinism gate.
+Durable comments in live mode carry the mock control plane's own deterministic
+clock rather than wall time, because that is the timestamp on the mock record.
diff --git a/packages/paperclip-runner/docs/capability-live-runnerd-codex.md b/packages/paperclip-runner/docs/capability-live-runnerd-codex.md
new file mode 100644
index 0000000000..7ba6106c9d
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-live-runnerd-codex.md
@@ -0,0 +1,122 @@
+# Capability live runnerd and Codex loop
+
+> **Reference lab, not the production sandbox topology.** This API remains the
+> runner-lab/session implementation used by UI and recovery tests. Production
+> sandbox execution and live protocol evals use the Rust-owned bridge described
+> in [`../../../doc/plans/2026-08-20-single-daemon-runner-tool-bridge.md`](../../../doc/plans/2026-08-20-single-daemon-runner-tool-bridge.md): external control plane → PRP → one Rust `paperclip-runnerd` → provider. The TypeScript dispatcher below does not run in the sandbox.
+
+Capability binds the provider-neutral semantic catalog to a real package-local
+`paperclip-runnerd` process and a real Codex app-server session. Paperclip data
+remains deterministic mock state behind `ControlPlanePort`; no request reaches
+the Paperclip API.
+
+## Process and authority boundary
+
+The process chain is:
+
+```text
+CapabilityLiveSession -> paperclip-runnerd -> codex app-server
+ -> CapabilitySemanticDispatcher -> ControlPlanePort mock
+```
+
+`paperclip-runnerd` owns the Codex child and proxies newline-delimited JSON-RPC
+over stdio. The transport starts a dedicated Unix process group. Normal close,
+stop/reset cleanup, and fatal protocol errors terminate that group with a
+bounded TERM/KILL sequence so a Codex child is not abandoned.
+
+Only the allowlisted Codex host environment is copied. `PAPERCLIP_*` variables,
+provider credentials other than Codex's server-side home, and credentialed
+proxy URLs are not passed to runnerd or Codex. Model-issued commands still use
+the separate skillless, network-disabled workspace permission profile.
+
+## Stable session API
+
+Production workers import the live entrypoint and bind a durable store to the
+attempt's immutable authority tuple:
+
+```ts
+import {
+ CapabilityLiveSessionService,
+ DurableCapabilityLiveSessionStore,
+} from "@paperclipai/paperclip-runner/live";
+
+const binding = { sessionId, runId, companyId, actorId, taskId };
+const store = new DurableCapabilityLiveSessionStore({ directory, binding });
+const service = new CapabilityLiveSessionService({ store });
+const session = await service.create({ ...binding, attemptId, workingDirectory });
+const turn = await session.sendMessage("Read the mock task and report progress.");
+await session.interrupt(); // only when a turn is active
+await service.stop(session.id);
+```
+
+After a worker terminates, a new worker must construct a new service and use the
+production resume entrypoint. It must not call `create` again:
+
+```ts
+const resumed = await new CapabilityLiveSessionService({ store }).resume({
+ sessionId,
+ attemptId: resumeAttemptId,
+ resumeOf: killedAttemptId,
+});
+await resumed.reconcileActiveTurn();
+```
+
+`resume` first commits the killed attempt as `terminated` and the successor as
+`running`. Only then does it start runnerd, read the checkpointed provider
+thread, and resume that exact thread. A missing or corrupt checkpoint, authority
+binding mismatch, attempt-lineage mismatch, or provider thread/session drift
+fails closed. The killed attempt and its usage remain immutable.
+
+The handoff surface for later tracks is:
+
+- `create(input)` starts runnerd, Codex, one mock run, and one dynamic-tool thread.
+- `sendMessage(text)` supports repeated turns on the same provider thread.
+- `pendingInteractions()` and `resolveInteraction(input)` preserve typed human
+ interactions and return their results to that same thread.
+- `reconnect(sessionId)` closes the old process group, starts a fresh runnerd and
+ Codex app-server, then reads and resumes the persisted provider thread.
+- `restore(sessionId)` recreates mock state, transcript, authority, authorization
+ records, pending interactions, and the provider thread from a stored snapshot.
+- `resume({ sessionId, attemptId, resumeOf })` is the cross-worker production
+ path and records distinct linked attempts before provider recovery.
+- `recordUsage(receipt)` durably commits an attempt-bound, exactly-once provider
+ response receipt before the caller acknowledges that response. Reusing a
+ receipt with different contents fails closed.
+- `reconcileActiveTurn()` interrupts and records the terminal fact for a turn
+ that was active in the checkpoint when its worker terminated.
+- `interrupt(reason)` cancels only the active turn and retains session authority.
+- `service.stop(sessionId)` clears authority and reaps the process group;
+ `reset` also deletes the old snapshot and restores the original clean mock
+ seed under new run/session authority.
+
+`DurableCapabilityLiveSessionStore` writes a checksummed, revisioned checkpoint
+with atomic rename plus file and directory fsync. It persists provider identity,
+mock state and semantic idempotency receipts, attempt lineage, active and
+terminal turn facts, and the usage ledger. The included in-memory store remains
+limited to tests and single-process consumers.
+
+Every snapshot includes bounded transcript/evidence, serialized mock state,
+semantic authorization records, runner/Codex PIDs and exit state, and explicit
+network evidence. Tool calls are admitted only when their thread and turn match
+the active Codex turn. Their typed `CapabilitySemanticToolResult` is serialized into
+the app-server response, allowing Codex to use the resulting state revision in
+its next response.
+
+## Verification
+
+Run the deterministic contract suite:
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner test:scenarios
+```
+
+Run a real runnerd and Codex app-server smoke:
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner trace:live-runner -- --json
+```
+
+The smoke requires an authenticated local Codex installation. It checks a real
+semantic tool mutation, typed-result response, same-thread second turn, process
+ownership/cleanup, cleared authority, and zero Paperclip network/child-env
+exposure.
diff --git a/packages/paperclip-runner/docs/capability-mock-control-plane-port.md b/packages/paperclip-runner/docs/capability-mock-control-plane-port.md
new file mode 100644
index 0000000000..4f4a5ef895
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-mock-control-plane-port.md
@@ -0,0 +1,94 @@
+# Capability Mock ControlPlanePort
+
+`CapabilityMockControlPlaneAdapter` is a deterministic, serializable, in-memory
+model of Paperclip's control-plane **semantics** — not its transport. It imports
+no server route, service, database, or provider binding, and it holds no
+credential.
+
+Sources: `src/mock-core/capability-mock-control-plane-adapter.ts`,
+`src/mock-core/capability-control-plane-types.ts`. Shared contract:
+`src/conformance/control-plane-port.ts`.
+
+## Entity domains
+
+The adapter models ten renderable entity domains, which are exactly the domains
+the [scenario explorer](capability-scenario-explorer.md) diffs:
+
+`tasks`, `comments`, `documents`, `interactions`, `approvals`, `artifacts`
+(artifacts and work products), `blockers`, `workspace` (workspace services),
+`budget`, and `run`.
+
+The fixture state also carries supporting records — company, actors, wakes,
+audit, decisions, idempotency, faults, counters, and a fixture clock — that back
+behavior but are not diff domains.
+
+## Operations
+
+- **Lifecycle:** `start()` / `stop()` only change local object state.
+- **Runs:** `openRun()` / `openFixtureRun()` return a run context with atomic
+ checkout; `context(runId)` reads it back.
+- **Events:** `appendEvent()` accepts both canonical PRP events and native run
+ events with deduplication, replay-conflict detection, and contiguous-sequence
+ tracking; `replayEvents()` replays from an exclusive cursor.
+- **Commands:** `applyCommand(envelope)` executes a semantic command with
+ idempotency and an optimistic `expectedRevision`.
+- **Sessions:** `loadSessionCheckpoint()` / `checkpointSession()`.
+- **Terminal:** `completeRun()` reconciles a terminal disposition — `done`,
+ `blocked`, `needs_review`, or `yielded` — and fans out blocker wakes.
+- **Introspection:** `snapshot()`, `decisionRecords()`, `serialize()`, and
+ static `restore()`.
+
+The eighteen semantic command kinds (`report_progress`, `write_document`,
+`request_human_input`, `resolve_human_input`, `register_deliverable`,
+`set_dependencies`, `create_task`, `request_approval`, `decide_approval`,
+`comment_on_approval`, `control_workspace_service`, `record_budget_usage`,
+`finish_task`, `block_task`, `request_review`, `release_task`, `schedule_wake`)
+are the mock side of the [semantic tool catalog](capability-semantic-tools.md).
+
+## Determinism
+
+- **Fixture clock.** Every timestamp derives from `clock.epochMs + tick*1000`
+ and the tick increments on each read. The default epoch is
+ `2026-08-09T00:00:00Z`. There is no RNG anywhere.
+- **Deterministic IDs.** Every generated ID is `fixture--` from a
+ monotonic per-kind counter.
+- **Immutability.** Inputs and outputs are `structuredClone`d and every returned
+ value is deep-frozen. Canonical JSON with sorted keys backs every dedupe and
+ replay comparison.
+- **Serializable.** `serialize()` / `restore()` round-trip under the schema tag
+ `paperclip.capability.mock-state.v1`.
+- **Scripted faults.** A fault rule injects a bounded number of
+ `retryable_error` or `lost_ack` effects; a lost-ack command commits once and
+ returns its stored result on retry.
+
+## Boundary
+
+The adapter is a pure state machine: no network, no database, no provider SDK,
+no credentials. Secrets appearing in wake payloads are redacted against
+`/authorization|credential|api.?key|secret|token/i`; the adapter test seeds
+`apiKey` and `Bearer` values and asserts they never surface. See
+[authorization and exposure](capability-authorization-and-exposure.md).
+
+## Shared conformance
+
+The adapter satisfies the same `runControlPlanePortConformance` contract used by
+earlier phases: it rejects open-binding violations, recovers from source gaps,
+treats duplicates idempotently, and rejects event-id, sequence, replay-binding,
+and result mutations (failing closed with `native_event_replay_conflict`).
+
+## Running the tests
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner exec vitest run \
+ src/conformance/control-plane-port.test.ts \
+ src/mock-core/capability-mock-control-plane-adapter.test.ts
+```
+
+Nine tests, no npm alias of their own — they also run inside
+`pnpm --filter @paperclipai/paperclip-runner test`.
+
+## Related
+
+- [Semantic tool catalog](capability-semantic-tools.md)
+- [Authorization and exposure](capability-authorization-and-exposure.md)
+- [Future binding boundary](capability-future-binding-boundary.md)
diff --git a/packages/paperclip-runner/docs/capability-scenario-explorer.md b/packages/paperclip-runner/docs/capability-scenario-explorer.md
new file mode 100644
index 0000000000..44b4c01ab4
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-scenario-explorer.md
@@ -0,0 +1,84 @@
+# Capability Browser Scenario Explorer
+
+The scenario explorer is a read-only browser surface over all 106 conformance
+cases. It runs each scenario in the page against the
+[mock control plane](capability-mock-control-plane-port.md) through the
+[semantic tool runtime](capability-semantic-tools.md) and renders the resulting
+run artifact. It never re-judges parity, never leaves its own origin, and holds
+no credential.
+
+Sources: `src/scenarios/` (scenario index, runner, parity, state diff) and
+`examples/scenario-explorer/`. Interaction contract:
+[Capability scenario explorer UX](design/capability-scenario-explorer-ux.md).
+
+## The run artifact
+
+Running a scenario produces one immutable artifact carrying:
+
+- **Tool exposure** — which semantic tools were available, and for optional
+ tools, the grant that unlocked each one.
+- **Control-plane actions** — control-plane-owned steps (for example checkout),
+ each labelled "no agent tool exists for this."
+- **Authorization records** — allow and typed-deny decisions; a denial carries
+ the missing claim and no protected task data.
+- **State diff** — an immutable diff over the ten entity domains, with unchanged
+ domains collapsed.
+- **Parity verdict** — the runtime's own verdict, plus Capability's per-case
+ result carried through in a separately labelled block. The explorer displays
+ these; it does not recompute them.
+
+## Read-only and frozen SDK
+
+The explorer imports the frozen **0.1.2** public SDK through the package-local
+alias `@paperclip-runner-local/capability`, which is deliberately not the published
+package name. It adds no SDK export and changes no published surface. Fake mode
+runs entirely in the page against checked-in fixtures and renders fixture time
+only, so two loads of the same route produce identical settled DOM.
+
+## What the explorer shows
+
+- **Home** — 16 group facets whose counts sum to 106, corpus stats, and example
+ deep links.
+- **Picker** — a listbox with live filter chips, disabled zero-count values, and
+ a clear-filters control.
+- **Transcript** — both channels (agent-visible and control-plane).
+- **Inspector** — four tabs: context (exposure and grants), authorization
+ (allow/deny), state diff, and traceability.
+
+## Boundary
+
+- No network request leaves the explorer's own origin.
+- `localStorage` holds no run artifact, grant, or fixture payload.
+- Codex mode is a disabled option with a stated reason; wiring it to the SDK
+ relay is deferred, and the browser holds no provider credential either way.
+- Secrets are redacted before display; redaction chips name the rule and never
+ the value. See [authorization and exposure](capability-authorization-and-exposure.md).
+
+## Accessibility
+
+The 7F browser suite asserts: listbox arrow-key navigation with
+`aria-activedescendant`, WAI-ARIA tab arrow-key activation with exactly one
+tablist, every interactive control named, a polite live region announcing the
+settled verdict, three landmarks, a single `h1`, and — at 390px — one segment at
+a time with zero horizontal overflow.
+
+## Running it
+
+```sh
+# Open the explorer on 127.0.0.1:4183.
+pnpm --filter @paperclipai/paperclip-runner demo:scenarios
+
+# Scenario runtime, explorer components, and route determinism (49 tests).
+pnpm --filter @paperclipai/paperclip-runner test:scenarios
+
+# Browser IA, determinism, evidence routes, boundary, a11y, responsive (25).
+pnpm --filter @paperclipai/paperclip-runner test:browser:scenarios
+
+# Deterministic 24-image acceptance set (12 routes x 2 viewports).
+# Recorded evidence generation is deferred from this release.
+```
+
+## Related
+
+- [Eval conformance](capability-eval-conformance.md)
+- [Verification commands](capability-verification-commands.md)
diff --git a/packages/paperclip-runner/docs/capability-semantic-catalog.md b/packages/paperclip-runner/docs/capability-semantic-catalog.md
new file mode 100644
index 0000000000..dd58332337
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-semantic-catalog.md
@@ -0,0 +1,79 @@
+# Capability semantic catalog and authorization
+
+Capability adds a transport-neutral tool boundary over the deterministic mock
+`ControlPlanePort`. It does not contain Paperclip REST routes, authentication
+headers, credentials, ACPX code, or a real control-plane binding.
+
+## Public boundary
+
+Import the catalog, policy, dispatcher, and binding helpers from the package
+root:
+
+```ts
+import {
+ CapabilitySemanticDispatcher,
+ createCapabilityProviderNeutralBinding,
+} from "@paperclipai/paperclip-runner";
+
+const dispatcher = new CapabilitySemanticDispatcher(mockPort, {
+ scenario: {
+ id: "dependency-manager",
+ claims: ["dependencies:write"],
+ },
+ explicitClaims: ["dependencies:write"],
+});
+
+const exposed = dispatcher.listTools(runId);
+const result = await dispatcher.dispatch({
+ runId,
+ callId: "tool-call-1",
+ operationId: "set_dependencies",
+ input: {
+ idempotencyKey: "dependency-change-1",
+ blockedByTaskIds: ["task-2"],
+ },
+});
+```
+
+`createCapabilityProviderNeutralBinding("fake")` and
+`createCapabilityProviderNeutralBinding("live_codex")` return identical contract
+arrays. Only the binding label differs. Track 7E can translate those definitions
+to Codex tool definitions and pass calls to the dispatcher without changing
+operation IDs, schemas, claims, or result shapes.
+
+## Authorization order
+
+Exposure and invocation both evaluate the actor, active-task ownership and
+mode, scenario allow/deny rules, role restrictions, run claims, and explicit
+claims. Optional operations require their descriptor claim to be present in the
+intersection authorized for the run. The dispatcher repeats the evaluation
+immediately before every mock read or command; a tool that was exposed earlier
+can still be denied after a claim or ownership change.
+
+Unauthorized optional tools are absent from `listTools`. Direct calls receive a
+`paperclip.semantic-denial.v1` result. The mock command boundary independently
+checks command claims and task ownership, so bypassing catalog exposure does not
+bypass authorization.
+
+The generic API escape hatch is an optional `skill_test` descriptor and is
+disabled by default. When a scenario and explicit claim enable it, only two
+read-only package-local paths are accepted: `/mock/state/revision` and
+`/mock/task`.
+
+## Protected data
+
+Tool schemas contain no credential fields. Inputs containing protected keys or
+credential-shaped values are rejected before mutation. Read results, denials,
+and immutable semantic authorization records pass through the same recursive
+redactor. Mock actor discovery also omits capability and budget internals.
+
+Generate and verify the checked-in contracts with:
+
+```sh
+pnpm --dir packages/paperclip-runner generate:semantic-contracts
+pnpm --dir packages/paperclip-runner check:semantic-contracts
+```
+
+The generated contract is
+`generated/capability/semantic-tool-contracts.json`. Capability traceability remains
+under `generated/capability/` and is checked with `check:capability-contract`.
diff --git a/packages/paperclip-runner/docs/capability-semantic-tools.md b/packages/paperclip-runner/docs/capability-semantic-tools.md
new file mode 100644
index 0000000000..9c3af0bbda
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-semantic-tools.md
@@ -0,0 +1,98 @@
+# Capability Semantic Tool Catalog
+
+The semantic tool catalog is the transport-neutral surface an agent may call. It
+is a frozen set of JSON-schema descriptors — no HTTP method, path, or provider
+detail. Provider-specific shapes are produced only by *bindings*, so every
+provider sees the same operation set.
+
+Each protocol action is single-sourced in its own module under
+`src/protocol-actions/`. That module owns the action's policy metadata,
+documentation, examples, and its live and scenario JSON-schema presentations.
+`src/catalog/canonical-operations.ts`, `src/semantic-tools/catalog.ts`, and
+`src/tools/capability-semantic-tool-catalog.ts` are compatibility projections;
+do not add action definitions to them. See
+[catalog reconciliation](../spec/capability/catalog-reconciliation.md).
+
+Sources: `src/protocol-actions/`, `src/catalog/canonical-operations.ts`,
+`src/tools/capability-semantic-tool-catalog.ts`,
+`src/tools/capability-semantic-tool-types.ts`, `src/tools/capability-tool-bindings.ts`,
+barrel `src/tools/index.ts`.
+
+## Two dispositions, plus a separate control-plane list
+
+A tool descriptor's `disposition` is either **`always_agent_tool`** or
+**`optional_agent_tool`**. Control-plane-owned operations are a separate frozen
+list, not a disposition — they are never exposed as tools. See
+[capability disposition](capability-disposition.md) and
+[authorization and exposure](capability-authorization-and-exposure.md).
+
+The catalog holds **37 tools**: 14 always-agent tools and 23 optional tools
+across 10 groups.
+
+
+
+### Always-agent tools (14)
+
+`get_task_context`, `get_task_history`, `list_documents`, `read_document`,
+`list_document_revisions`, `report_progress`, `answer_status_question`,
+`finish_task`, `block_task`, `request_review`, `write_document`,
+`request_human_input`, `register_deliverable`, `inspect_operation_result`.
+
+### Optional tools (23), by group
+
+| Group | Tools |
+| --- | --- |
+| discovery | `search_tasks`, `list_agents`, `list_projects`, `list_goals` |
+| delegation_dependencies | `create_task`, `set_dependencies` |
+| governance | `list_approvals`, `request_approval`, `decide_approval`, `comment_on_approval` |
+| cases | `list_cases`, `upsert_case` |
+| workspace_runtime | `get_workspace_runtime`, `control_workspace_service` |
+| routines | `list_routines`, `manage_routine` |
+| company_skills | `list_company_skills`, `sync_company_skills` |
+| secrets | `list_secret_metadata`, `read_secret_value` |
+| portability_admin | `export_company`, `administer_company` |
+| test_escape_hatch | `generic_api_request` |
+
+## Descriptor shape
+
+Each descriptor carries `operationId`, `version` (1), `title`, `description`,
+input/output JSON schemas, `disposition`, an optional `optionalGroup`,
+`requiredClaims`, and optionally `allowedRoles`, `taskModes`, a `sideEffectClass`
+(`read`, `task_write`, `company_write`, `governance`, `workspace_control`,
+`secret_read`, `admin`, `test_escape_hatch`), an `idempotency` level, `redaction`
+rules, and an abstract `mockCommandMapping`.
+
+The `mockCommandMapping` is one of `context_read`, `snapshot_read`,
+`semantic_command`, `operation_result`, or `mock_extension` — describing *what*
+the tool does against the mock, never *how* a transport would carry it.
+
+## Transport neutrality
+
+Bindings, not descriptors, produce provider shapes, both derived from the same
+`visibleTools.tools` array:
+
+- `CapabilityFakeAgentToolBinding` emits `{operationId, description, inputSchema,
+ outputSchema}`.
+- `CapabilityCodexToolBinding` emits `{type: "function", name, description, strict:
+ true, parameters}`.
+
+Because both bindings derive from one array, the fake-agent and Codex operation
+surfaces are byte-identical. The [conformance suite](capability-eval-conformance.md)
+asserts this parity (the 18/18 fake-agent/Codex operation matrix).
+
+## Running the tests
+
+```sh
+pnpm --filter @paperclipai/paperclip-runner exec vitest run \
+ src/tools/capability-semantic-tools.test.ts
+```
+
+The "catalog" describe asserts a unique, versioned, provider-neutral catalog and
+that every optional group is present.
+
+## Related
+
+- [Authorization and exposure](capability-authorization-and-exposure.md)
+- [Mock ControlPlanePort](capability-mock-control-plane-port.md)
+- [Eval conformance](capability-eval-conformance.md)
diff --git a/packages/paperclip-runner/docs/capability-verification-commands.md b/packages/paperclip-runner/docs/capability-verification-commands.md
new file mode 100644
index 0000000000..8a58ff0e68
--- /dev/null
+++ b/packages/paperclip-runner/docs/capability-verification-commands.md
@@ -0,0 +1,50 @@
+# Capability Verification Commands
+
+Every command here runs from the repository root, is offline and deterministic,
+starts no Paperclip service, and holds no credential. All are prefixed
+`pnpm --filter @paperclipai/paperclip-runner`.
+
+| Surface | Command | Expected result |
+| --- | --- | --- |
+| Capability contract completeness and drift | `check:capability-inventory` | `Capability inventory completeness and generated-output checks passed.` |
+| Contract validator negatives | `test:capability-inventory` | 4 tests pass |
+| Mock control plane + shared port | `exec vitest run src/conformance/control-plane-port.test.ts src/mock-core/capability-mock-control-plane-adapter.test.ts` | 9 tests pass |
+| Semantic tools + authorization/redaction | `exec vitest run src/tools/capability-semantic-tools.test.ts` | 9 tests pass |
+| 106-case conformance | `test:capability-evals` | 1 file (all 106 cases) passes |
+| Fake-agent matrix + bounded Codex + parity report | `report:capability-evals` | `Capability eval conformance passed: 106 cases across 16 groups.` |
+| Scenario runtime + explorer + clean room + routes | `test:scenarios` | 159 tests pass |
+| Browser IA, accessibility, determinism, boundary | `test:browser:scenarios` | Playwright suites pass (60 in the issue-thread/clean-room suite) |
+| Screenshot acceptance set | deferred recorded-evidence campaign | 24 deterministic images |
+| Documentation links | `docs:validate` | all local links resolve |
+
+## Live commands (not offline)
+
+The clean-room chat has no fake or replay path, so its end-to-end proof needs a
+Rust toolchain and a locally authenticated Codex. These are the only Capability
+commands that start a provider.
+
+| Surface | Command | Expected result |
+| --- | --- | --- |
+| Real Codex through real runnerd on a fresh mock tenant | `smoke:capability:cleanroom` | every assertion `true`; two `MCK-` identifiers, one per chat |
+| Preset issue thread against real Codex | `smoke:capability:ui` | every assertion `true` |
+| Clean-room screenshots | deferred recorded-evidence campaign | 7 images; intentionally not byte-stable |
+
+## Notes
+
+- **No Rust toolchain is needed** for any Capability command. The full
+ `verify` target still builds the Rust workspace, but nothing above does.
+- **Determinism.** Fake-mode runs render fixture time only, so repeat runs and
+ repeat screenshots are byte-identical.
+- **The parity report is generated on demand.** `report:capability-evals` writes
+ `.paperclip-local/evidence/capability/eval-parity-report.{json,md}`, which are not
+ committed or scanned by `docs:validate`.
+- **Browser libraries.** On minimal or rootless hosts, extract Playwright's
+ browser libraries first; the package's `verify:rootless` target shows the
+ pattern.
+
+## Related
+
+- [Clean-start tutorial](tutorials/capability-scenario-explorer.md)
+- [Eval conformance](capability-eval-conformance.md)
+- [Scenario explorer](capability-scenario-explorer.md)
+- [Clean-room live chat](capability-clean-room-chat.md)
diff --git a/packages/paperclip-runner/docs/codex-driver.md b/packages/paperclip-runner/docs/codex-driver.md
new file mode 100644
index 0000000000..9e767b8a0a
--- /dev/null
+++ b/packages/paperclip-runner/docs/codex-driver.md
@@ -0,0 +1,168 @@
+# Codex Skillless Codex Driver
+
+## Scope
+
+Codex implements a direct Codex app-server v2 driver behind the package's
+existing `HarnessDriver` contract. The driver, mock core, example CLI, tests,
+and evidence stay inside `packages/paperclip-runner/`. They do not import or
+change Paperclip server, UI, database, or production control-plane behavior.
+
+The app-server process is local to the execution environment and uses newline
+delimited JSON-RPC over stdio. It is not exposed as a network service.
+
+## Identity mapping
+
+| Runner identity | Codex source | Persistence rule |
+| --- | --- | --- |
+| run ID | mock-core input | Never replaced during recovery. |
+| normalized session ID | controller-owned mock-core input | A distinct identity, independent of the run and provider IDs, that stays stable across transport/process recovery. |
+| driver session ID | `thread.id` | Resumed by exact ID. A different returned ID fails recovery. |
+| provider session ID | `thread.sessionId` | Kept separately from the driver thread ID. |
+| turn ID | `turn.id` | Required by steer and interrupt preconditions. |
+| item ID | `item.id`, request ID, or deterministic turn/kind key | Preserved on lifecycle and delta events. |
+| source event ID | runner instance + run + source sequence | Source sequence continues from the persisted snapshot. |
+
+The persisted session snapshot records run, normalized session, driver
+session, provider session, the exact active turn, committed semantic-result
+content/call binding, observed terminal-turn fingerprints, and the last source
+sequence. Recovery starts a new local app-server transport, reads the exact
+persisted thread to validate identity and working directory, resumes that
+thread, and then reads it again for reconciliation. Reconciliation considers
+only the persisted active turn: it retains that turn when active, terminalizes
+that turn when terminal, and fails recoverably when it is missing or a
+different active turn appears. Historical terminal turns never substitute for
+the persisted active turn, and a terminal already in the durable snapshot is
+not emitted again.
+
+## App-server operations
+
+| Driver operation | App-server method | Degradation |
+| --- | --- | --- |
+| initialize | `initialize`, then `initialized` | Startup fails visibly. |
+| create | `thread/start` | Required. |
+| resume | `thread/resume` | `recovered: false` with a redacted reason. |
+| read | `thread/read` | Explicit `HarnessCapabilityUnavailableError`. |
+| start turn | `turn/start` | Required. |
+| steer | `turn/steer` with `expectedTurnId` | Explicit unsupported diagnostic; no stdin fallback. |
+| interrupt | `turn/interrupt` | Explicit unsupported diagnostic; session is not killed. |
+| usage | `thread/tokenUsage/updated` | Returns the last snapshot or explicit unsupported error. |
+| reconcile | `thread/read` plus `session.reconciled` | Disabled when read is unavailable. |
+
+Capability flags are descriptive and executable. Unsupported operations emit
+canonical `harness.diagnostic` events with secret-redacted detail. No
+harness-specific branch is required in the mock core.
+
+## Skillless context boundary
+
+The model receives one text input containing `paperclip.skillless_task.v1`:
+
+- objective;
+- completion-contract revision and criteria;
+- task constraints; and
+- the expected canonical result schema name.
+
+The thread config explicitly disables automatic skill and app instruction
+blocks. Codex's built-in collaboration instructions are enabled by default so
+interactive runs receive native commentary and tool preambles; a driver caller
+may explicitly disable them for a specialized deterministic fixture. This
+does not enable skills, apps, plugins, memories, or extra model-input kinds.
+The model input accepts only text, never a Codex `skill` input. The driver
+captures the returned instruction-source list and requires it to be empty for
+the skillless assertion.
+
+The trusted app-server process has an allowlisted environment. It retains host
+`HOME` and `CODEX_HOME` only so the provider can authenticate. Model-issued
+commands have a separate boundary: an empty-by-default environment with no
+`HOME` or `CODEX_HOME`, no network, and a named Codex
+permission profile requesting read-only minimal runtime files, no host-home or
+Codex-home access, and write access to the assigned workspace. The driver
+refuses filesystem-root workspaces, workspaces containing host `HOME`, and any
+workspace overlapping host `CODEX_HOME`.
+
+The returned sandbox facts remain authoritative. Codex 0.132.0 may inject a
+provider-managed writable root such as `~/.codex/memories` after a first run,
+even with `features.memories=false` and an explicit Codex-home deny. That makes
+the Codex-home directory discoverable in a warmed environment, so Codex does
+not claim whole-directory unreadability. Its authenticated proof instead
+requires each readable `auth.json`/`config.toml` file and an unrelated host
+secret to remain unreadable and unwritable, while recording any injected root
+in `context.sandbox.legacyPolicy`.
+
+Paperclip bearer values, `OPENAI_API_KEY`, arbitrary skill paths, and other
+inherited variables are not passed. Diagnostics redact bearer/basic
+credentials, credentialed proxy URLs, secret query parameters, sensitive JSON
+keys, and common key assignments.
+
+The context snapshot records configuration and environment **key names**, not
+secret values.
+
+## Semantic completion
+
+The provider-facing structured-output schema covers `done` and `needs_review`.
+It uses the strict OpenAI shape: every object rejects additional properties and
+the constant schema field includes both `type: "string"` and `const`.
+
+Two dynamic semantic tools are registered when supported:
+
+- `paperclip_finish` accepts `done` or `needs_review`;
+- `paperclip_block` accepts `blocked` and requires a blocker owner, action,
+ reason, and scope.
+
+Both normalize through the canonical `paperclip.run_result.v1` validator. The
+first valid result is proposed. Any canonically identical result retry is
+idempotent even when the provider assigned a new call ID; the original call
+binding remains persisted for audit, while changed content is rejected. Tool
+calls and provider notifications must name the exact opened thread and active turn. Missing,
+pre-turn, cross-thread, cross-turn, and post-terminal bindings fail the provider
+session closed. Canonically identical terminal replays are no-ops; conflicting
+terminal facts are rejected. Process exit or prose alone never implies
+completion.
+
+The provider cannot commit controller state. It emits `run.result.proposed` and
+a provider turn terminal; the mock core validates the proposal against its
+task envelope, emits `run.result.accepted` or `run.result.rejected`, and alone
+emits `run.terminal`.
+
+## Canonical event mapping
+
+- thread lifecycle -> `session.started`, `session.resumed`,
+ `session.reconciled`;
+- turn lifecycle -> `turn.submitted`, `turn.accepted`, `turn.started`, and one
+ terminal turn event;
+- messages, reasoning, plans, commands, file changes, dynamic tools, and diffs
+ -> `item.started`, `item.delta`, `item.completed`;
+- model selection -> a completed `model` item;
+- app-server decisions -> `runtime_request.created` and
+ `runtime_request.resolved` with redacted detail;
+- token snapshots -> completed `usage` items;
+- semantic verification rows -> completed `verification` items;
+- provider completion -> at most one `run.result.proposed` and one turn
+ terminal;
+- controller decision -> one `run.result.accepted` or `run.result.rejected`,
+ followed by one `run.terminal`.
+
+The JSON-RPC transport limits each input line, pending client requests,
+in-flight server requests, queued notification count and bytes, diagnostic
+lines, and retained provider payloads. Malformed or oversized messages close
+the transport and reject pending work.
+
+The existing Replay reducer consumes the live stream. Replay crosses a
+serialized JSONL boundary that validates byte and event counts, line size,
+schema, run/session binding, unique source event IDs, continuous per-source
+sequence, and exactly one final run terminal before reducing. The Codex
+tracer requires byte-equivalent live and replay snapshots.
+
+## Runnable example
+
+`trace:codex` starts a real local `codex app-server` session through the mock
+core. Its safe task creates `hello.txt` with network disabled. The evidence
+recorder additionally probes all readable host Codex credential/config files
+and an unrelated host secret, requiring reads and writes to be denied while
+workspace output and app-server authentication still succeed. It gates output
+reads on an accepted `done` result and reports missing files by name rather
+than surfacing a raw filesystem `ENOENT`. See the
+[Codex tutorial](tutorials/codex.md).
+
+This phase changes no browser surface, so no new browser screenshot applies.
+The canonical events are proved through the existing reducer/replay path and
+JSON trace evidence.
diff --git a/packages/paperclip-runner/docs/design/capability-issue-thread-ux-contract.md b/packages/paperclip-runner/docs/design/capability-issue-thread-ux-contract.md
new file mode 100644
index 0000000000..18566ca816
--- /dev/null
+++ b/packages/paperclip-runner/docs/design/capability-issue-thread-ux-contract.md
@@ -0,0 +1,429 @@
+# Capability — Issue-Thread UX Contract
+
+Status: **binding** for track 7G (Paperclip-style Web UI) and reviewable by tracks 7E/7K.
+Source of authority: Capability plan (TASK-16897 `plan` document), Revision 3 §"Page information
+architecture", §"Native issue-thread interactions", §"Live execution contract", and Revision 4
+§"Final evidence eligibility". Capability vocabulary: Capability generated contract
+(`generated/capability/capabilities.yaml`, `mcp-tool-map.yaml`, `eval-traceability.yaml`,
+`contract-schema.json` — TASK-16939).
+
+If a corner of this contract is underspecified, 7G comments on TASK-16940 and waits for a
+revision. 7G does not diverge silently. Deviations are either fixed in the implementation or
+written back into this document as an explicit revision.
+
+---
+
+## 0. Vocabulary (normative)
+
+All UI copy and all implementation identifiers use the 7A terms. The UI never invents
+capability language.
+
+- **Dispositions** (closed enum): `control_plane_owned`, `always_agent_tool`,
+ `optional_agent_tool`.
+- **Always semantic operations** (thread-visible verbs): `get_task_context`,
+ `report_progress`, `answer_status_question`, `finish_task`, `block_task`, `request_review`,
+ `write_document` (+ revision reads), `request_human_input` (kinds `questions`,
+ `confirmation`, `checkbox`, `suggest_tasks`, `item_verdicts`), `register_deliverable`.
+- **Optional operation families**: discovery (`scoped_discovery`, `search_tasks`,
+ `list_agents`, …), delegation/dependencies (`create_task`, `delegate_task`,
+ `set_dependencies`, `create_blocked_task`), approvals (`request_approval`,
+ `decide_approval`, …), cases, workspace runtime, routines, company skills, secrets, admin,
+ escape hatch. Grants render exactly as 7A writes them (e.g. `rf:read_or_write`).
+- **Mock-state kinds**: `operation_result`, `runtime_decision_record`, `active_task_context`.
+- **Eval keys**: case `id` (e.g. `ix-confirmation-plan-01`), `group` (16 groups),
+ `fixtureProfile` (e.g. `hb-baseline`), `browserEvidenceRecipe` (e.g. `hb/hb-context-01`),
+ `expectedSemanticOperations`, `forbiddenOperations`.
+- **Modes**: `live` (real Codex via real runnerd), `fake` (deterministic fake agent),
+ `replay` (recorded canonical events). A surface never shows an unlabeled mode.
+
+Human-readable labels (UI copy) for the dispositions are fixed: `Agent tool — always`,
+`Agent tool — granted`, `Control plane`. The machine term appears in the debug panel and in
+tooltips, never invented synonyms ("system tool", "built-in", etc. are forbidden).
+
+## 1. Identity and mode banner (always visible)
+
+The page must answer, at all times and on every viewport, four questions: who is the agent,
+what runs it, what control plane it talks to, and whether this session is live.
+
+1. **Three identity chips**, rendered in the issue header and never hidden by scroll
+ (header is sticky):
+ - `Real Codex` — cyan chip, only when the active session mode is `live`. In `fake` mode the
+ chip reads `Fake agent`; in `replay` it reads `Replay`.
+ - `Real runnerd` — neutral chip with a live process dot (pulse while a runnerd session is
+ attached; static gray when detached). Shown in `live` mode only; `fake`/`replay` show
+ `In-process runner`.
+ - `Mock Paperclip` — amber chip, shown in **all** modes. Tooltip: "All issue records are
+ mock. No real Paperclip API is reachable."
+2. **Mode is data, not styling**: the root element carries `data-session-mode="live|fake|replay"`
+ and the chips render from the server-reported session record. A session artifact with
+ `mode=fake` must be visually distinguishable from `live` in every screenshot (Revision 4
+ evidence eligibility).
+3. **Mock identifier scheme**: mock issues use a reserved prefix that cannot be confused with
+ a real company (fixture default: `MCK-`, e.g. `MCK-31`). The UI never renders a
+ `/PAP/...`-style link to a real control plane; mock entity links navigate inside the
+ explorer only.
+4. **No credential surface**: no header, panel, tooltip, error, or diff may contain a token,
+ key, or `Authorization` value. Redacted fields render as `••• redacted` with the redaction
+ rule name from the authorization record.
+
+## 2. Page information architecture
+
+### 2.1 Desktop (reference viewport 1440×900)
+
+```
+┌──────────────────────────────────────────────────────────────┬───────────────┐
+│ Issue header (sticky): │ │
+│ MCK-31 · title · StatusBadge · PriorityIcon · assignee │ Debug panel │
+│ [Fake agent|Real Codex] [Real runnerd] [Mock Paperclip] │ (collapsed │
+│ run state · Scenario ▾ · Replay · Reset · Stop │ by default; │
+├──────────────────────────────────────────────────────────────┤ resizable │
+│ Thread column (centered, max-width 760px, single column): │ 320–640px) │
+│ turn groups: user message → agent activity → responses │ │
+│ comment / interaction / document / deliverable / │ │
+│ disposition cards in chronological order │ │
+│ system notices (wakes, reconciliation) as one-line rows │ │
+├──────────────────────────────────────────────────────────────┤ │
+│ Composer (pinned bottom of thread column) │ │
+└──────────────────────────────────────────────────────────────┴───────────────┘
+```
+
+- The **thread is the primary readable surface**. Default panel state is **collapsed**; the
+ user's open/width choice persists per browser (`localStorage`), and deep links may force it
+ (§10). The page never leads with raw protocol events.
+- Thread column measure: max-width 760px, centered in the remaining space; `text-sm` body
+ scale per the Paperclip type ramp.
+- Debug panel: right side, resizable 320–640px with a keyboard-operable splitter
+ (`role="separator"`, `aria-valuenow`), collapse toggle in the header (`Evidence` button with
+ open/closed state).
+- Below 1100px viewport width the panel switches from side-by-side to an overlay sheet from
+ the right (same content, same tab order), so the thread column never drops below ~600px.
+
+### 2.2 Mobile (reference viewport 390×844)
+
+- One column. Sticky condensed header: row 1 = `MCK-31` + StatusBadge + overflow menu `⋯`
+ (Scenario, Replay, Reset inside the menu; **Stop stays outside the menu** whenever a turn is
+ active). Row 2 = the three identity chips, wrapping to a second line if needed — chips never
+ cause horizontal page scroll.
+- A **segmented control** with exactly two segments — `Thread` and `Evidence` — sits under the
+ header (pattern validated in the Scenario chat mobile review; a drawer was rejected there and
+ stays rejected). Thread is default. Badge on `Evidence` shows the current turn's
+ authorization-denial count when nonzero.
+- Composer is fixed to the bottom of the `Thread` segment, above the keyboard inset
+ (`env(safe-area-inset-bottom)`).
+- **No horizontal page scroll at 390px** (`document.scrollingElement.scrollWidth <=
+ clientWidth` is an automated acceptance check). `pre`/code/diff blocks wrap or scroll inside
+ their own container only.
+- Touch targets ≥ 44×44 CSS px for every actionable element, including chip tooltips
+ (tap-to-toggle on touch), accordion headers, and interaction-card controls.
+
+## 3. Thread item taxonomy
+
+Every thread item renders from a **mock-core record or canonical event** — the browser holds
+no state authority, computes no claim/policy/parity result, and never mutates mock state
+outside an interaction response (§5). Items in chronological order, grouped by turn:
+
+| # | Item | Source | Anatomy |
+|---|------|--------|---------|
+| T1 | **User message** | thread record | Right-aligned bubble style is **not** used; Paperclip comment card with author "You (board user)", timestamp, markdown body. |
+| T2 | **Agent response** | model output items | Comment card, author = agent identity chip (`Fake agent` / `Real Codex`), streaming state per §6. Model prose only — never confused with durable records (see T3). |
+| T3 | **Durable progress comment** | `report_progress` / `answer_status_question` `operation_result` | Distinct comment card with a `Recorded to mock thread` marker (filled corner tag + tooltip naming the semantic operation). This is the visual boundary between ephemeral model text (T2) and durable mock records. |
+| T4 | **Tool activity strip** | semantic call + typed result | One line per call inside the turn group: status glyph (`✓ ok`, `✕ denied`, `⏳ running`), operation id (`write_document`), one-line human summary, `›` expander. Expanded: request args (redacted per rules), typed result, and a `View in Evidence` link that opens the debug panel pre-filtered to that call. Strips are collapsed by default; a turn shows at most 3 strips + `N more…` expander (progressive disclosure). |
+| T5 | **Interaction card** | `request_human_input` record | §5. Rendered at its chronological position. |
+| T6 | **Document card** | `write_document` result | Document key + title, revision chain (`r3 → r4`), author, `View diff` (opens Evidence → State), and stale marker when a later revision exists. |
+| T7 | **Deliverable card** | `register_deliverable` result | File/ref name, kind (attachment bytes / external ref / workspace file), size, registered-by, download affordance for attachment-backed deliverables. |
+| T8 | **Dependency / delegation card** | optional-op results (`create_task`, `set_dependencies`, …) | Created mock child issues with identifiers and blocker edges (`MCK-32 blocks MCK-31`). |
+| T9 | **Disposition card** | `finish_task` / `block_task` / `request_review` result | Terminal banner card: new status via StatusBadge, explanation body, named blocker owner where applicable. After a terminal disposition the composer enters `disabled` (§6). |
+| T10 | **Denial notice** | typed policy denial (`operation_result` with deny) | Inline red-bordered strip variant of T4: `✕ create_task — denied: missing grant su:read_or_write`. Shows deny reason **from the authorization record verbatim**; never leaks protected state or credentials. |
+| T11 | **System notice** | `runtime_decision_record` (wake, checkout, reconciliation, budget stop, session events) | One-line, muted, icon + text (e.g. `⚙ Wake: issue_blockers_resolved → turn 3 started`). Progressive disclosure into Evidence → Control plane. Never a card; system notices must read quieter than work content. |
+
+Turn grouping: each turn renders a hairline group header `Turn N · · tool calls ·
+` binding T2/T4/T5-T10 items produced within it. Thread auto-follows the newest item
+only while the user is at the bottom; a `Jump to latest` pill appears when scrolled up
+(≥ 300px) or when new items arrive off-screen.
+
+## 4. Composer contract
+
+States (mutually exclusive; `data-composer-state` attribute is the test/screenshot hook):
+
+| State | Trigger | Visual | Controls |
+|-------|---------|--------|----------|
+| `ready` | session attached, no active turn, issue non-terminal | normal input, `Send` primary | Send (Cmd/Ctrl+Enter), attach disabled in Capability |
+| `sending` | message posted, turn not yet streaming | input cleared, inline spinner on Send | Send disabled |
+| `streaming` | active turn | input stays **editable** (steer), primary button becomes `Stop` (destructive-outline), helper text `Codex is working — send to steer, or stop the turn.` | Send = steer (queued as next user input), Stop |
+| `waiting` | pending interaction card requires the user | input disabled, helper `Answer the pending request above to continue.` with an anchor link that scrolls to and focuses the pending card | none |
+| `reconnecting` | transport lost | input disabled, helper `Reconnecting… your session is preserved.` | Retry now |
+| `disabled` | terminal disposition, replay mode, or budget stop | input disabled with reason line (`Issue is done`, `Replay is read-only`, `Budget limit reached`) | New scenario / Reset |
+
+Draft text survives refresh (localStorage per session id). Stop never discards the transcript;
+a stopped turn's partial output stays in the thread with a `Stopped by user` marker on the
+turn header.
+
+## 5. Interaction cards (native lifecycle)
+
+All five `request_human_input` kinds render as native cards using the typed payload's prompt,
+options, labels, and validation — never as markdown asking the user to type an answer.
+
+Kinds and their answer controls:
+
+- `questions` (ask_user_questions): typed form — per-question control (radio/select/short
+ text), one submit.
+- `confirmation` (request_confirmation): target summary + Accept / Reject buttons; reject
+ reason textarea when the payload requires it. **Revision-bound**: the card shows the target
+ (`plan · r4`) and links the exact revision.
+- `checkbox` (request_checkbox_confirmation): checkbox list with min/max enforcement,
+ default-selected ids, Accept label / Reject label from payload.
+- `suggest_tasks`: proposed-task list (title + description); board accepts a subset; accepted
+ tasks appear as T8 cards afterwards.
+- `item_verdicts` (request_item_verdicts): per-item Approve / Reject / Defer segmented
+ buttons; reason required per `requireReasonOn`; supports partial submit (submitted items
+ lock, remaining stay editable, card stays `pending`).
+
+**Response authority path (normative):** the card's submit posts the typed response to the
+package server; the **mock control plane stores the response before the runner receives it**,
+and only then does the same Codex session resume with the typed result. The card's UI state
+moves `pending → submitting → resolved` only on server acknowledgment (no optimistic
+resolution). This is the *only* browser-initiated mock mutation; there is no other write path
+from the page.
+
+**State matrix (all states are required and visually distinct):**
+
+| State | Store shape | Card treatment |
+|-------|------------|----------------|
+| `pending` | status `pending` | Accent left border (violet), controls enabled, `Waiting for you` chip, focus lands on first control when the card is the reason the composer is `waiting`. |
+| `submitting` | in flight | Controls disabled, inline spinner. |
+| `accepted` | status `accepted` | Green check chip `Accepted`, chosen values summarized inline, controls collapse to read-only summary. |
+| `answered` | status `answered` | Same as accepted with `Answered` chip (questions / verdicts complete). |
+| `rejected` | status `rejected` | Red chip `Changes requested`, reason quoted in the card. |
+| `stale_target` | status `expired`, `result.outcome=stale_target` | Gray chip `Stale — plan moved to r5`, link to the superseding revision, controls removed, body dimmed (see revision 2). |
+| `superseded_by_comment` | status `expired`, `result.outcome=superseded_by_comment` | Gray chip `Superseded by a later comment`, link to that comment. |
+| `expired` | status `expired`, no outcome | Gray chip `Expired `. |
+| `withdrawn` | status `cancelled`, `result.outcome=withdrawn` | Gray chip `Withdrawn`, optional reason. |
+| `issue_closed` | status `cancelled`/`expired`, `result.outcome=issue_closed` | Gray chip `Issue closed`. |
+
+Resolved/expired cards are durable history: they stay in the thread at their chronological
+position, are keyboard-reachable, and expose `View request evidence` linking the debug panel
+to the related request, policy decision, mock mutation, wake, and resume events (the plan's
+required debug linkage). Pending cards survive refresh and reconnect (§6) — rehydrated from
+mock state, not from browser memory.
+
+## 6. Session lifecycle behaviors
+
+- **Refresh (F5)**: full restore from server state — same session, same turn ids, transcript,
+ pending interaction cards, composer state, and mode chips. The session **never restarts
+ because the browser reconnected**. Scroll restores to latest; a `Restored session` system
+ notice (T11) is *not* emitted (silent restore) — the evidence panel's session record is the
+ proof.
+- **Reconnect**: on transport drop the composer enters `reconnecting`, a slim amber banner
+ pins under the header (`Connection lost — retrying (attempt n)`), and streaming indicators
+ freeze with a `paused` glyph. On reconnect, missed canonical events replay by ordinal (no
+ duplicates, no gaps) and the banner resolves to a 3s `Reconnected` confirmation.
+ `prefers-reduced-motion` replaces the banner slide with opacity.
+- **Stop**: header Stop and composer Stop are the same action — bounded cancel of the active
+ turn. Post-state: partial output retained + `Stopped by user` turn marker; session and
+ pending interactions unaffected; composer returns to `ready`.
+- **Reset**: destructive — always behind a confirm dialog (`Reset scenario? This clears the
+ mock state and starts a clean session. The transcript will be lost.`; confirm button
+ `Reset scenario`, destructive style; cancel is default focus). Reset re-seeds the fixture,
+ rotates/clears session authority (Revision 4), and lands on a clean thread with a fresh
+ `Turn 0` seeded context. Reset affects only the current browser session's scenario instance.
+- **Replay**: mode `replay` re-renders a recorded run from canonical events. Composer
+ `disabled` (`Replay is read-only`), identity chip row shows `Replay` + `Mock Paperclip`,
+ and a top progress strip allows step/next-turn/play-all with a deterministic `?at=`
+ deep-link parameter. Replay of a `fake` recording must still be labeled as fake-derived
+ (chip `Replay · fake source`) so replay evidence can never satisfy a live criterion.
+- **Stop/Reset/Replay/Scenario controls** live in the header on desktop; on mobile
+ Replay/Reset/Scenario collapse into `⋯`, Stop stays exposed while a turn is active (§2.2).
+
+## 7. Side debug panel (Evidence)
+
+Named **Evidence** in UI copy. Content scope: a **turn selector** at the top (`Turn N ▾`,
+default = latest; `All turns` option for the state and parity sections). Below it, eight
+accordion sections in this order (accordion, not tabs — >7 categories, and multiple sections
+must be open simultaneously for review):
+
+1. **Tools exposed** — the turn's visible tool list grouped `Agent tool — always`, then
+ `Agent tool — granted` (each with its grant, e.g. `rf:read_or_write`), then a separated
+ muted list `Control plane (not exposed to the agent)` naming `control_plane_owned`
+ operations relevant to the fixture. Negative evidence is first-class: the control-plane
+ list exists precisely to show what the model *cannot* call.
+2. **Calls & results** — chronological semantic calls: raw Codex request → dispatched command
+ → typed result/denial, with operation id, version, and redaction annotations.
+3. **Authorization** — one record per decision: operation, claims considered, allow/deny +
+ reason, redactions applied, resulting state change ref. Deny records use the same red
+ accent as T10.
+4. **Control plane** — `runtime_decision_record` stream: checkout/lock, wake scheduling,
+ budget, idempotency/retry, reconciliation, session lifecycle.
+5. **Runner & events** — canonical PRP events and runnerd/Codex process diagnostics (session
+ id, thread id, process state, cleanup evidence).
+6. **State diff** — before/after per entity class (tasks, comments, documents, interactions,
+ approvals, artifacts, blockers, workspace, budget, run). Per-turn by default; `All turns`
+ shows fixture-seed → current. Rendered from immutable snapshots; the browser never
+ computes a diff from its own bookkeeping.
+7. **Traceability** — the fixture's 7A anchors: capability rows (id + `sourceAnchor`), eval
+ case id/group/`browserEvidenceRecipe`, `expectedSemanticOperations`,
+ `forbiddenOperations`, `requiredCapabilityGrants`.
+8. **Parity** — assertion list with verdict chips (`pass` green / `fail` red /
+ `intentional gap` gray + note), summarized as `n/m` in the section header.
+
+Cross-linking contract: every T4 strip, T10 denial, and interaction card deep-links into the
+matching Evidence record (`View in Evidence`), and every Evidence record links back to its
+thread anchor. Deep-link target = section + record id, e.g. `?panel=authorization&rec=`.
+
+Mobile: the same eight sections render inside the `Evidence` segment, full-width accordions,
+turn selector pinned under the segmented control.
+
+## 8. Visual language
+
+Follow the Paperclip design language without importing the product `ui/` package:
+
+- Dark theme default, OKLCH neutral grays; semantic tokens only (background/card/muted/
+ accent/destructive/border/ring equivalents defined package-locally). No raw hex in
+ components.
+- Type ramp: page title `text-xl font-bold`; card titles `text-sm font-medium`; body
+ `text-sm`; metadata `text-xs text-muted-foreground`; identifiers and operation ids
+ `font-mono text-xs`.
+- Status/priority renders with StatusBadge/StatusIcon-equivalent components using the
+ product's status hue table (todo blue, in_progress indigo, in_review violet, done green,
+ blocked red, backlog/cancelled gray).
+- Radii ≤ `rounded-xl`; shadows ≤ `shadow-sm`; density = product issue page, not a marketing
+ layout.
+- Interaction cards use the product interaction-card anatomy (title row + prompt + controls +
+ state chip) so the mock thread reads as a Paperclip issue thread (Jakob's Law is the point
+ of this phase's demo).
+
+## 9. Accessibility acceptance (blocking)
+
+7G is not acceptable until all of these pass; 7K re-verifies them clean-room:
+
+1. **Keyboard tour** (documented, testable): Tab order = header controls → thread (each card
+ is a focusable group; Enter expands) → composer → Evidence toggle → panel. The splitter is
+ arrow-key resizable. Pending interaction controls are reachable without pointer; Escape
+ closes the mobile `⋯` menu and the desktop overlay sheet.
+2. **Focus management**: opening Evidence moves focus to its heading; resolving a card moves
+ focus to the card's state chip; `waiting` composer's anchor link moves focus to the pending
+ card's first control. Focus ring = 3px ring token, never suppressed.
+3. **Live regions**: streaming agent text in `aria-live="polite"` chunk announcements (throttled
+ ≥ 2s); turn completion, denial notices, and interaction resolution announce via a single
+ polite status region; reconnect banner is `role="status"`, Stop confirmation `role="alert"`.
+4. **Structure**: one `h1` (issue title), landmarks `header/main/complementary` (Evidence),
+ `form` for composer; interaction cards are `section`s labeled by their prompt; tool strips
+ are disclosure buttons with `aria-expanded`.
+5. **Color independence**: every state chip pairs color with a glyph + text (`✓ Accepted`,
+ `✕ Denied`, `⏳`); parity verdicts likewise. Contrast ≥ 4.5:1 for text, ≥ 3:1 for UI
+ glyphs, verified in dark theme.
+6. **Reduced motion**: `prefers-reduced-motion` disables pulse dots, banner slides, streaming
+ shimmer, and smooth scrolling (instant jumps).
+7. **Automated gate**: axe (or equivalent) run against every screenshot route in §10 with
+ zero serious/critical violations, executed in CI alongside screenshot capture.
+
+## 10. Deterministic screenshot contract
+
+### 10.1 Route scheme
+
+- Base route: `#/issue/` (e.g. `#/issue/hb-baseline`). Scenario/fixture ids
+ come from 7A `fixtureProfile`; per-case evidence uses the 7A `browserEvidenceRecipe` path
+ (`/`) as the canonical evidence id.
+- Screenshot state param: `?shot=` seeds the named deterministic state below in `fake`
+ mode with: fixed clock (all timestamps render from fixture time), animations/caret/pulse
+ disabled, network idle, fonts loaded.
+- Panel/deep-link params: `?panel=[&rec=]`, `?at=` (replay),
+ `?seg=thread|evidence` (mobile segment).
+- Settle signal: the root element sets `data-thread-state="settled"` when hydration, fixture
+ load, and auto-scroll are complete. Capture tooling waits for it — never for timeouts.
+
+### 10.2 Required matrix (12 slugs × 2 viewports = 24 PNGs)
+
+Viewports: desktop `1440×900`, mobile `390×844`. Output path:
+`.paperclip-local/evidence/capability/ui/--.png` (package-local). Every capture
+also asserts `scrollWidth <= clientWidth` on the scrolling element at 390×844.
+
+| Slug | Seeded state | Must be visible |
+|------|--------------|-----------------|
+| `thread-baseline` | settled 3-turn fake run on `hb-baseline` | header w/ 3 identity chips + mode; T1/T2/T3/T4 items; turn headers; composer `ready` |
+| `turn-streaming` | mid-turn stream | streaming indicator, composer `streaming` w/ Stop, editable steer input |
+| `interaction-question-pending` | `ix-questions-01` pending card | typed question form, `Waiting for you` chip, composer `waiting` w/ anchor helper |
+| `interaction-confirmation-pending` | `ix-confirmation-plan-01` pending | revision-bound target (`plan · r4`) on card, Accept/Reject, reject-reason affordance |
+| `interaction-resolved-mixed` | history incl. accepted + rejected + `stale_target` + `superseded_by_comment` | four visually distinct resolved/expired treatments per §5 |
+| `denial-optional-tool` | denied `create_task` (missing `su:read_or_write`) | T10 denial strip w/ verbatim deny reason; Evidence badge increment |
+| `document-revision` | `dp-plan-doc-01` after `write_document` | T6 card w/ revision chain `r3 → r4` + View diff |
+| `deliverable-registered` | `ar-upload-before-done-01` | T7 deliverable card w/ kind + registered-by |
+| `disposition-terminal` | `st-done-comment-01` finished | T9 terminal card, StatusBadge `done`, composer `disabled` w/ reason |
+| `debug-panel-open` | baseline + `?panel=authorization` | desktop: panel open at 384px w/ 8 sections, authorization records; mobile: `Evidence` segment active |
+| `reconnect-banner` | forced transport drop | amber reconnect banner, composer `reconnecting`, frozen stream glyph |
+| `replay-mode` | replay of recorded fake run `?at=12` | `Replay · fake source` chip, read-only composer, progress strip |
+
+Determinism rule: two captures of the same slug/viewport from a clean checkout must be
+pixel-identical (the Scenario chat byte-identical bar). Anything time-, random-, or
+locale-dependent renders from fixture data.
+
+### 10.3 Evidence naming
+
+Per-eval-case evidence (7F/7K scope) reuses `browserEvidenceRecipe` verbatim:
+`.paperclip-local/evidence/capability/cases//--.png`. The §10.2 matrix is
+the UI acceptance set; case evidence is additive and follows the same settle/determinism
+rules.
+
+## 11. Authority and safety rules (UI-side restatement)
+
+- The browser renders mock-core records, snapshots, and canonical events. It computes no
+ claim, policy, diff, or parity result client-side. UI-side state math is a defect.
+- The only browser-initiated mock mutation is an interaction response (§5). Composer messages
+ go to the runner session, not to mock state.
+- No provider, runner, or control-plane credential ever reaches the browser; redactions render
+ by rule name. Real Paperclip URLs/API paths never appear.
+- Policy and state authority live in the package server + mock `ControlPlanePort`; refresh
+ and reconnect re-derive everything from them.
+
+## 12. 7G handoff checklist
+
+Implementation acceptance (UXDesigner review) requires:
+
+1. All §3 item types and all §5 interaction states implemented and reachable via fixtures.
+2. All §4 composer states with `data-composer-state` hooks.
+3. §6 behaviors demonstrated: refresh restore, reconnect replay, stop, reset confirm, replay
+ read-only.
+4. §7 Evidence panel with all eight sections, turn selector, and bidirectional deep links.
+5. §10 matrix: 24 deterministic PNGs recorded at the named routes, byte-stable across two
+ clean runs, plus the 390px no-horizontal-scroll assertion per capture.
+6. §9 accessibility gate green (axe + documented keyboard tour).
+7. Screenshot review posted to the 7G issue for UXDesigner acceptance before 7G closes.
+
+Questions or gaps → comment on TASK-16940.
+
+---
+
+## Revisions
+
+### Revision 2 — expired-family dimming (2026-08-10, written back by 7G)
+
+Revision 1 specified a literal 60% opacity on the body of `stale_target` and the
+other expired-family cards (§5). Measured against the card surface in the dark
+theme, that renders the card's text at ~3.16:1, which fails §9.5's 4.5:1 bar and
+the blocking axe gate in §9.7 — the two rules cannot both hold.
+
+§9 wins because it is the blocking gate. "Dimmed" is now specified as a
+**recessed treatment**: the card drops to the sunken surface token and its
+secondary text drops to the muted-foreground token, both of which clear 4.5:1.
+The gray state chip, removed controls, and neutral left border are unchanged, so
+the card still reads as history at a glance.
+
+Implemented in `devtools/issue-thread/src/issue-thread.css` and covered by the
+axe gate on the `interaction-resolved-mixed` slug at both viewports.
+
+### Revision 3 — `interaction-resolved-mixed` capture framing (2026-08-10, written by the contract owner during the 7G gate)
+
+Revision 1's §10.2 row required four resolved/expired card treatments visible
+in the `interaction-resolved-mixed` frame. Two of this contract's own rules
+make that impossible in one capture: §10.1's settle signal ends with
+auto-scroll to the latest thread item, and the §5/§8 card anatomy makes four
+resolved cards ~1100px tall — taller than either viewport. The seeded history
+(answered + accepted + rejected + `stale_target` + `superseded_by_comment`)
+cannot fit one auto-scrolled frame.
+
+The row now reads: the seeded state must contain **all five** treatments
+(answered, accepted, rejected, `stale_target`, `superseded_by_comment`), the
+frame shows the latest-scrolled portion with the expired-family chips
+(`Stale …`, `Superseded …`) fully visible, and the visual distinctness of all
+five treatments is asserted by the browser suite plus a scrolled review pass
+at the gate. Splitting the slug into two frames was rejected because the
+24-PNG matrix count is referenced by the 7F/7K evidence and the §9.7 axe gate.
diff --git a/packages/paperclip-runner/docs/design/capability-scenario-explorer-ux.md b/packages/paperclip-runner/docs/design/capability-scenario-explorer-ux.md
new file mode 100644
index 0000000000..525fc065b6
--- /dev/null
+++ b/packages/paperclip-runner/docs/design/capability-scenario-explorer-ux.md
@@ -0,0 +1,549 @@
+# Capability Interaction Map — Scenario Explorer and Screenshot Acceptance Spec
+
+Status: **approved UX contract for 7F implementation** (TASK-16899, 2026-08-09).
+Owner: UXDesigner. Implementer: 7F (browser scenario explorer), integrating
+7C mock adapter output and 7E parity output.
+Companion records: [Capability contract](../capability-contract.md),
+`spec/capability/eval-traceability.yaml`, the Capability plan (TASK-16897),
+[Live console interaction map](live-console-interaction-map.md),
+[SDK component decisions](sdk-component-decisions.md).
+(annotated desktop + mobile renders in `.paperclip-local/evidence/capability/`).
+
+This map is the interaction contract for the Capability browser scenario
+explorer. The explorer is a **read surface over records the runtime already
+produced**. It renders scenario metadata from the generated 7A traceability
+manifest, run artifacts from the 7C mock control-plane adapter, and parity
+results from the 7E conformance suite. The UI never owns state, policy,
+credentials, or execution decisions: it does not compute tool exposure, does
+not evaluate claims, does not apply redaction, and holds no Paperclip or
+provider credential. Every allow/deny, redaction, and state transition shown
+on screen must arrive as a record emitted by the mock core.
+
+## 0. Naming, vocabulary, and canonical labels
+
+- Product name in the UI: **Scenario Explorer** (header: “Capability scenario
+ explorer — mock control plane”).
+- UI copy uses the canonical term **task** (`DESIGN.md` principle 7). Eval
+ case IDs, fixture IDs, route paths, JSON payloads, and source anchors are
+ machine values and stay verbatim (`issues`, `i-0102`, `MCK-102`) in
+ monospace — never “translated.”
+- The three dispositions render with exactly these labels, everywhere:
+ **Control plane** (`control_plane_owned`), **Always tool**
+ (`always_agent_tool`), **Optional tool** (`optional_agent_tool`). The
+ traceability panel and all badges must use the 7A contract values as their
+ source; no UI-local re-classification.
+- The five assertion classes render as: **Invariant**
+ (`control_plane_invariant`), **Tool contract** (`agent_tool_contract`),
+ **Authorization** (`authorization_policy`), **Multi-hop**
+ (`combined_multi_hop`), **Restraint** (`restraint_no_call`).
+- Parity statuses: **Pass**, **Fail**, **Intentional gap**, **Not run**.
+ Each has an icon + text label; color alone never encodes status (WCAG
+ color-independence).
+
+## 1. Shell and information architecture
+
+Package-local Vite entry `examples/scenario-explorer/` with script
+`demo:scenarios` (default port 4183; 4182 belongs to the Standalone demo). It runs
+from static assets plus checked-in fixtures with **no Paperclip services and
+no network dependency** in fake-agent mode. Reuse the frozen `0.1.2` SDK
+surface (`@paperclipai/paperclip-runner/react` + `./styles.css`) through its
+five approved extension points; do not fork the token layer.
+
+Desktop layout (≥ 64rem), one React tree (SDK finding — never render two
+trees):
+
+- **Left rail (`--pcr-rail-width`, 17rem):** scenario picker — search,
+ facet filters, case list grouped by eval group.
+- **Center column (fluid, max `--pcr-center-width`, 48rem):** run view —
+ the chronological scenario transcript and the run header (verdict banner,
+ mode toggle, run/replay controls). Primary surface, strongest hierarchy.
+- **Right rail (`--pcr-inspector-width`, 24rem):** inspector with four tabs
+ — **Context · Authorization · State diff · Parity**. Open by default on
+ desktop (unlike the 4b protocol inspector: here the inspector carries
+ first-class acceptance evidence, not diagnostics).
+
+Mobile (< 64rem): single column behind a segmented control
+(`Scenarios · Run · Inspect`), same pattern as the Live console console. The
+segmented control is sticky at the top; the Run segment is the default when
+a case is selected via route. Touch targets ≥ `--pcr-touch-target` (2.75rem).
+
+Density: diagnostic-console dense, like 4b. All spacing, color, type, radius,
+motion from `--pcr-*` tokens only (`DESIGN.md` principles 2–4; `pnpm
+check:token-gates` applies).
+
+## 2. Scenario picker and filters (left rail)
+
+**Source of truth:** the generated scenario index derived from
+`spec/capability/eval-traceability.yaml` (all 106 cases, 16 groups) plus the
+latest parity results artifact from 7E. The picker consumes these; it never
+re-derives group membership or dispositions.
+
+Contents, top to bottom:
+
+1. **Search input** — substring match on case ID and title. Mono rendering
+ of matched IDs. Debounce ≤ 150ms (Doherty).
+2. **Facet filters** (collapsible group, `--pcr-facet-height` baseline):
+ - **Group** — the 16 eval groups (`hb 5 · co 6 · st 8 · cm 6 · se 4 ·
+ su 4 · bl 5 · dp 3 · ix 9 · ap 6 · ar 4 · er 9 · rf 22 · mh 4 · rs 3 ·
+ wk 8`), each with its case count.
+ - **Case** — direct case-ID picker (combobox) for jump-to-case.
+ - **Disposition** — Control plane / Always tool / Optional tool, from
+ `primaryDisposition`.
+ - **Role** — actor role of the fixture (`ic`, `manager`, `reviewer`,
+ `board_user`, …) from the scenario index (§9 data contract).
+ - **Claim** — required grants from `requiredGrants` (e.g.
+ `approval:read`), rendered mono.
+ - **Parity status** — Pass / Fail / Intentional gap / Not run, from the
+ parity artifact.
+ Facets are AND-combined across facets, OR-combined within a facet. Every
+ facet value shows its live result count; zero-count values stay visible
+ but disabled (recognition over recall — the vocabulary stays learnable).
+ An active-filter chip row with per-chip remove and one **Clear filters**
+ affordance sits above the list (forgiveness).
+3. **Case list** — grouped by eval group with sticky group headers. Each row:
+ case ID (mono, `--pcr-font-size-xs`), title (one line, truncated with
+ full text in a tooltip), disposition badge, parity status dot+icon.
+ Selected row uses `--pcr-accent-surface` with a `--pcr-accent` left rule.
+
+Keyboard: the list is a single-select listbox (`role="listbox"`,
+`aria-activedescendant`); Up/Down move, type-ahead jumps, Enter selects and
+moves focus to the run header. Facets are disclosure buttons +
+checkbox groups. Filter state mirrors into the route (§7) so any filtered
+view is linkable; filter state is never stored as protocol state.
+
+Empty results: “No scenarios match these filters.” + Clear filters button —
+never a bare void.
+
+## 3. Run view (center column)
+
+### 3.1 Run header
+
+- Case ID (mono) + title, group badge, disposition badge, assertion-class
+ badges.
+- **Verdict banner** (SDK `Banner`): overall parity verdict for the last
+ completed run — Pass (success surface), Fail (danger surface), Not run
+ (muted). The banner names the assertion counts (“12 assertions · 12 pass”)
+ and deep-links the Parity tab.
+- **Mode control:** `Fake agent` (default) / `Codex (bounded)` segmented
+ control. Fake mode is always available offline. Codex mode appears only
+ when the local relay is reachable; otherwise the option renders disabled
+ with the reason (“provider relay not running — see tutorial §4”). The
+ browser never holds a provider credential; Codex traffic goes through the
+ same server-side relay boundary proven in SDK.
+- **Run / Re-run** button and replay controls (SDK `replay-controls`).
+ Deterministic fixtures mean re-run in fake mode reproduces the identical
+ timeline; state diffs and parity always describe the displayed run.
+
+### 3.2 Scenario transcript
+
+**Source of truth:** the ordered run artifact timeline emitted by the mock
+core (§9). Items render in artifact order; the UI never reorders, merges, or
+re-times entries.
+
+The transcript interleaves two visually distinct channels (Similarity +
+Common Region — a stranger must tell them apart in two seconds):
+
+1. **Agent channel** — model turns and semantic tool calls/results. Aligned
+ with the standard conversation layout (SDK `conversation`/`message`/
+ `tool-item`). A semantic tool call renders collapsed by default with a
+ one-line summary: operation ID (mono) + disposition badge + one-phrase
+ outcome (“`finish_task` · Always tool · committed”). Expansion is the
+ item-body / request-detail extension point (§8) and shows: operation ID
+ and version, argument JSON, result JSON, idempotency behavior, claims the
+ call required, and redaction chips (§5) — exactly the descriptor fields
+ from the 7D catalog, never re-labeled.
+2. **Control-plane channel** — actions the runtime performed with **no
+ tool** (checkout, wake routing, budget stops, reconciliation, blocker
+ wake scheduling…). These render full-width on a `--pcr-muted` surface
+ with a “Control plane” badge and a gutter glyph distinct from tool items,
+ and are indented from the agent lane. Copy states the actor plainly:
+ “Control plane checked out task MCK-102 — no agent tool exists for this.”
+ Each entry expands to the decision record (inputs considered, resulting
+ state change, audit reference).
+
+Failed/denied entries render in place on `--pcr-danger-surface` with the
+exact typed denial (§5) — failures are part of the record, never toast-only.
+
+Restraint/no-call scenarios (`rs-*`, `er-*` restraint cases) must remain
+legible when the correct behavior is *absence*: after the final agent turn
+the transcript appends a system note “No further operations — N forbidden
+operations, none invoked” which deep-links the Parity tab’s forbidden list.
+Restraint evidence must never depend on an empty screen.
+
+PRP events stay behind a per-item “Protocol events” disclosure (same nested
+pattern as SDK `debugEvents`) — the semantic view is primary; raw
+protocol stays inspectable but secondary (progressive disclosure).
+
+Auto-follow, jump-to-latest, streaming affordances, and reduced-motion rules
+are inherited verbatim from the Live console map §1 — do not redesign them.
+
+## 4. Inspector tabs (right rail)
+
+All four tabs are SDK `Tabs`; the active tab mirrors into the route (§7).
+
+### 4.1 Context
+
+Answers “who ran, under what authority, with which tools” before any
+transcript reading (mental model first):
+
+- **Actor block:** actor role, task mode, fixture ID (mono), wake payload
+ summary (reason + trigger), budget posture.
+- **Capability grants:** the scenario’s `requiredGrants` plus every grant
+ the fixture actually issued, rendered as mono chips.
+- **Tool exposure list** — the acceptance-critical view. Three labeled
+ sections, populated **only** from the mock core’s exposure record:
+ 1. **Always tools (N)** — always-catalog operations visible this run;
+ 2. **Optional tools (N)** — visible optional operations, each with the
+ grant that unlocked it (“`decide_approval` — via `approval:decide`”);
+ 3. **Control plane — no tool (N)** — control-plane-owned capabilities
+ relevant to this scenario, explicitly listed so reviewers can verify
+ absence (negative evidence must be visible, not implied).
+- **Traceability panel:** source anchor (file:line, mono), skill elements
+ (file · section · lines), evidence IDs, legacy MCP mapping row if any —
+ verbatim from the 7A contract.
+
+### 4.2 Authorization
+
+Chronological table of every authorization record the mock core emitted:
+operation (mono), claims considered, decision (**Allow** / **Deny** with
+icon + text), reason string, redactions applied, resulting state change
+reference. Row expansion shows the full record JSON. Denied rows use
+`--pcr-danger-surface`; a count chip on the tab surfaces denials without
+opening it (“Authorization · 1 deny”). Selecting a row scrolls/flashes the
+corresponding transcript entry (uniform connectedness across panels).
+
+### 4.3 State diff
+
+Before/after immutable snapshot diff for the displayed run, grouped by the
+ten entity domains in the plan: **tasks, comments, documents, interactions,
+approvals, artifacts, blockers, workspace, budget, run**. Each domain is a
+collapsible section headed by a change summary (“tasks · 1 changed”,
+“comments · 2 added”); unchanged domains collapse to a single muted row
+(“unchanged”) so change carries the visual weight (Von Restorff). Rows show
+entity ID (mono) + field-level changes as `field: before → after`, with
+added/removed/changed markers as icon + label, not color alone. The diff is
+computed by the runtime/test layer and delivered as data; the UI performs no
+diffing of its own. A “final state” toggle shows the complete after-snapshot
+JSON per domain behind a disclosure.
+
+### 4.4 Parity
+
+The eval acceptance surface:
+
+- **Verdict header** repeating the banner verdict.
+- **Assertion list:** one row per conformance assertion — assertion class
+ badge, human-readable expectation, Pass/Fail, and on fail an expected vs
+ observed block (mono, side-by-side at desktop, stacked at mobile).
+- **Forbidden operations:** each `forbiddenSemantics` entry with “never
+ invoked ✓” or the violating transcript reference.
+- **Intentional gaps:** entries marked as intentional exclusions render
+ under their own heading with the recorded rationale — visually distinct
+ from Fail (a gap is a decision, not a defect).
+- **Legacy baseline note:** the case’s skill-present/skill-absent reference
+ note, verbatim, muted — labeled “reference baseline, not the pass
+ condition.”
+
+## 5. Credentials, redaction, and denial rendering
+
+- Fake mode performs zero network I/O; Codex mode talks only to the local
+ relay; no route, storage key, or bundle string may contain a Paperclip or
+ provider credential. `localStorage` may hold UI preferences only
+ (inspector tab, filter collapse state) — never run artifacts, grants, or
+ anything from a fixture.
+- Redacted values arrive pre-redacted from the mock core. The UI renders a
+ redaction chip: `•••` + “redacted” label + the redaction-rule name in a
+ tooltip. The UI must not receive-and-hide secrets; if a raw secret ever
+ reaches the browser payload, that is a 7C/7D defect, and the explorer
+ additionally fails closed by rendering nothing for fields flagged
+ redacted.
+- A **denied** tool call renders the typed policy denial: operation, the
+ missing claim, and the denial reason — and never any protected state. An
+ **absent** tool simply does not appear in exposure lists; the Context tab’s
+ control-plane section is the affordance proving absence is intentional.
+- The generic escape hatch (`paperclipApiRequest` successor), when a test
+ scenario grants it, renders with a persistent `--pcr-warning-surface`
+ “test-only escape hatch” banner on every call card (plan §4: visually
+ flagged).
+
+## 6. Accessibility, responsive, empty/error states
+
+Accessibility (WCAG POUR; inherits all Live console/5 baseline rules):
+
+- Landmarks: `nav` (picker), `main` (run view), `complementary`
+ (inspector); one `h1`; group headers are `h2`/`h3` in order.
+- Full keyboard operation: picker listbox (§2); `F6`/documented shortcut
+ cycles the three regions; tabs follow the WAI-ARIA tabs pattern
+ (arrow-key activation); every disclosure is a real `button`. Visible
+ focus ring via `--pcr-ring` everywhere.
+- Live regions: run progress announces via `aria-live="polite"`
+ (“Scenario run settled — verdict pass”); denial entries announce
+ assertively once.
+- Reduced motion: with `prefers-reduced-motion`, streaming reveal, banner
+ transitions, and scroll-flash connect cues render instantly/statically
+ (SDK rule).
+- Contrast from the existing token pairs only; parity/authorization
+ statuses always icon + text.
+
+Responsive: content-driven single breakpoint at 64rem (§1). At mobile the
+inspector tabs become full-width stacked sections inside the Inspect
+segment; expected-vs-observed blocks stack; the transcript keeps every
+descendant intrinsically shrinkable (`min-w-0` discipline) so no horizontal
+scrollbar appears at 390px (SDK finding).
+
+States that must be designed, not defaulted:
+
+| State | Surface | Rendering |
+| --- | --- | --- |
+| No scenario selected | Center + inspector | Explorer intro card: corpus stats (106 cases · 16 groups), “Pick a scenario” pointer at the rail, three example deep links. No dead panels. |
+| Scenario selected, not run | Run view | Header + context render from the index; transcript area shows a primed empty state (“Run to produce the deterministic timeline”) with the Run button duplicated inline (Fitts). Diff/Parity tabs show “No run yet.” |
+| Run in progress | All | Progress in the header (indeterminate, reduced-motion safe); transcript streams; tabs show live counts; controls disable with reasons. |
+| Run failed (harness error) | Run view | Danger banner with the exact diagnostic + “Re-run”; partial timeline stays visible and labeled partial. Never silently reset. |
+| Parity fail | Banner/Parity | Fail is a first-class rendered state (danger banner + failing assertion rows) — the explorer’s job is showing it, not avoiding it. |
+| Denied/absent tool | Transcript/Context | Per §5. |
+| Missing/invalid fixture or artifact | Shell | Error card naming the artifact path and the regenerate command (`pnpm generate:capability-inventory` / 7E command). No blank screen. |
+| Codex relay unavailable | Mode control | Disabled option + reason; fake mode remains fully usable. |
+
+## 7. Deterministic screenshot routes
+
+Hash routing (static hosting, no server rewrites):
+
+```
+#/ explorer home (picker + intro)
+#/case/ scenario selected, not run
+#/case/?run=fake deterministic fake run, auto-executed
+#/case/