diff --git a/doc/execution-semantics.md b/doc/execution-semantics.md
index 00dc745988..3bdaadac23 100644
--- a/doc/execution-semantics.md
+++ b/doc/execution-semantics.md
@@ -1250,3 +1250,24 @@ awaits it. That background invocation observes rejection immediately, including
when a remote sandbox has already disappeared. The owner's awaited close still
receives the original failure; containment never fabricates a successful close
or permission to reuse an unverified execution.
+
+### Assigned connections in native ACPX sessions
+
+Native ACPX sessions register the assigned Paperclip MCP gateway alongside the
+task tool bridge. Gateway calls retain the existing connection grants and action
+approvals. Missing assigned bindings and names that collide with the task bridge
+stop admission. Upstream credentials remain with the gateway; providers receive
+its scoped access binding. The qualified ACPX sidecar receives the gateway name,
+URL, and token together through the launch allowlist; unrelated environment
+secrets remain excluded. This does not restrict arbitrary network access to a
+public service outside the gateway.
+
+
+### Use real connection requests (2026-09-14)
+
+When a user asks to connect a known service, the agent searches for that service
+and uses `connection_request` if setup is needed. The agent must not ask the same
+permission again or copy Connect / Not now into a generic question. A generic
+question does not start setup. The real connection card keeps user identity,
+access grants, the decision, and continuation together. This guidance does not
+approve a connection or bypass its normal user decision.
diff --git a/package.json b/package.json
index 8d1bdd7a0d..8f653413a7 100644
--- a/package.json
+++ b/package.json
@@ -69,6 +69,7 @@
"test:e2e:runner:dashboard": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/dashboard-regenerate.ts",
"test:e2e:runner:models:update": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/openrouter-models-update.ts",
"test:e2e:runner:history:publish": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/history-publish.ts",
+ "test:runner-recovery": "vitest run server/src/services/native-runtime/native-replacement-evidence.test.ts server/src/services/native-runtime/stopped-codex-turn.test.ts server/src/services/native-runtime/native-safe-replacement.test.ts",
"test:e2e:runner:unit": "vitest run --config tests/runner-e2e/vitest.config.ts",
"test:e2e:runner:typecheck": "tsc -p tests/runner-e2e/tsconfig.json",
"test:e2e:runner:report": "node cli/node_modules/tsx/dist/cli.mjs tests/runner-e2e/report.ts",
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/App.tsx b/packages/paperclip-runner/devtools/issue-thread/src/App.tsx
index ef14d92c9b..3b038df78d 100644
--- a/packages/paperclip-runner/devtools/issue-thread/src/App.tsx
+++ b/packages/paperclip-runner/devtools/issue-thread/src/App.tsx
@@ -104,18 +104,18 @@ interface EmbeddedEvalReport {
startedAt: string;
finishedAt: string;
durationMs: number | null;
- initialRevision: number;
- finalRevision: number;
+ initialRevision: number | null;
+ finalRevision: number | null;
finalStateSummary?: string;
usage: {
agentTurns: number;
providerRequests: number | null;
- inputTokens: number;
- outputTokens: number;
- cachedInputTokens: number;
- reasoningTokens: number;
+ inputTokens?: number;
+ outputTokens?: number;
+ cachedInputTokens?: number;
+ reasoningTokens?: number;
providerReportedCostNanodollars?: number;
- estimatedCostNanodollars: number;
+ estimatedCostNanodollars?: number;
pricingVersion: string;
} | null;
};
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/DevtoolsInspector.tsx b/packages/paperclip-runner/devtools/issue-thread/src/DevtoolsInspector.tsx
index ee057682b2..da18caf5fb 100644
--- a/packages/paperclip-runner/devtools/issue-thread/src/DevtoolsInspector.tsx
+++ b/packages/paperclip-runner/devtools/issue-thread/src/DevtoolsInspector.tsx
@@ -43,17 +43,17 @@ export interface EvalInspectorReport {
startedAt: string;
finishedAt: string;
durationMs: number | null;
- initialRevision: number;
- finalRevision: number;
+ initialRevision: number | null;
+ finalRevision: number | null;
usage: {
agentTurns: number;
providerRequests: number | null;
- inputTokens: number;
- outputTokens: number;
- cachedInputTokens: number;
- reasoningTokens: number;
+ inputTokens?: number;
+ outputTokens?: number;
+ cachedInputTokens?: number;
+ reasoningTokens?: number;
providerReportedCostNanodollars?: number;
- estimatedCostNanodollars: number;
+ estimatedCostNanodollars?: number;
pricingVersion?: string;
} | null;
};
@@ -286,15 +286,17 @@ export function EvalReportInspector({
{isPublic
? "Withheld from public replay"
- : `r${evalReport.run.initialRevision} → r${evalReport.run.finalRevision}`}
+ : evalReport.run.initialRevision == null || evalReport.run.finalRevision == null
+ ? "unavailable"
+ : `r${evalReport.run.initialRevision} → r${evalReport.run.finalRevision}`}
Tokens
- {evalReport.run.usage === null
- ? "unknown"
- : `${evalReport.run.usage.inputTokens} in · ${evalReport.run.usage.outputTokens} out · ${evalReport.run.usage.cachedInputTokens} cached`}
+ {evalReport.run.usage == null
+ ? "unavailable"
+ : `${evalReport.run.usage.inputTokens ?? "unavailable"} in · ${evalReport.run.usage.outputTokens ?? "unavailable"} out · ${evalReport.run.usage.cachedInputTokens ?? "unavailable"} cached`}
@@ -308,9 +310,9 @@ export function EvalReportInspector({
Estimated cost
- {evalReport.run.usage === null
- ? "unknown"
- : `$${(evalReport.run.usage.estimatedCostNanodollars / 1_000_000_000).toFixed(6)}${evalReport.run.usage.pricingVersion ? ` · ${evalReport.run.usage.pricingVersion}` : ""}`}
+ {evalReport.run.usage?.estimatedCostNanodollars == null
+ ? "unavailable"
+ : `$${(evalReport.run.usage.estimatedCostNanodollars / 1_000_000_000).toFixed(6)}${evalReport.run.usage.pricingVersion ? ` · ${evalReport.run.usage.pricingVersion}` : ""}`}
diff --git a/packages/paperclip-runner/devtools/issue-thread/src/ThreadItems.tsx b/packages/paperclip-runner/devtools/issue-thread/src/ThreadItems.tsx
index a5efd9b300..717f6936dd 100644
--- a/packages/paperclip-runner/devtools/issue-thread/src/ThreadItems.tsx
+++ b/packages/paperclip-runner/devtools/issue-thread/src/ThreadItems.tsx
@@ -585,8 +585,9 @@ export function TurnGroup({
- Turn {turn.ordinal} · {turn.mode} · {turn.toolCallCount} tool call
- {turn.toolCallCount === 1 ? "" : "s"} ·{" "}
+ Turn {turn.ordinal} · {turn.mode} · {turn.mode === "replay" && turn.toolCallCount === 0
+ ? "no recorded tool calls"
+ : `${turn.toolCallCount} tool call${turn.toolCallCount === 1 ? "" : "s"}`} ·{" "}
{new Date(turn.at).toISOString().slice(11, 19)}
{turn.stoppedByUser ? (
diff --git a/packages/paperclip-runner/runner/crates/runner-core/src/acpx_sidecar_transport.rs b/packages/paperclip-runner/runner/crates/runner-core/src/acpx_sidecar_transport.rs
index 6dc1e606cd..86602e4713 100644
--- a/packages/paperclip-runner/runner/crates/runner-core/src/acpx_sidecar_transport.rs
+++ b/packages/paperclip-runner/runner/crates/runner-core/src/acpx_sidecar_transport.rs
@@ -120,6 +120,9 @@ impl AcpxSidecarTransport {
"RUST_BACKTRACE",
"PAPERCLIP_NATIVE_MCP_NAME",
"PAPERCLIP_NATIVE_MCP_URL",
+ // The qualified sidecar configures the runner-owned gateway. Keep
+ // its credential with the name/URL; unrelated secrets stay excluded.
+ "PAPERCLIP_NATIVE_MCP_TOKEN",
"PAPERCLIP_ACPX_PROVIDER_PACKAGE_ROOT",
"PAPERCLIP_ACPX_PROVIDER_PACKAGE_MANIFEST",
];
diff --git a/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-acpx-sidecar.rs b/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-acpx-sidecar.rs
index f3c8f05062..c615ecf9a1 100644
--- a/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-acpx-sidecar.rs
+++ b/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-acpx-sidecar.rs
@@ -84,6 +84,21 @@ fn run() -> Result<(), Box> {
continue;
}
match mode {
+ "mcp-environment" => {
+ write_json(
+ &mut stdout,
+ &json!({
+ "protocolVersion": GENERATED_ACPX_SIDECAR_PROTOCOL_VERSION,
+ "id": id, "ok": true,
+ "result": {
+ "name": std::env::var("PAPERCLIP_NATIVE_MCP_NAME").ok(),
+ "url": std::env::var("PAPERCLIP_NATIVE_MCP_URL").ok(),
+ "hasToken": std::env::var("PAPERCLIP_NATIVE_MCP_TOKEN").is_ok(),
+ "hasUnrelatedSecret": std::env::var("UNRELATED_EVAL_SECRET").is_ok(),
+ }
+ }),
+ )?;
+ }
"silent" => continue,
"wrong-id" => {
write_json(&mut stdout, &success(id + 1, command, &request))?;
diff --git a/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-codex-app-server.rs b/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-codex-app-server.rs
index 63d0eb4ecf..696c214d2d 100644
--- a/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-codex-app-server.rs
+++ b/packages/paperclip-runner/runner/crates/runner-core/src/bin/fake-codex-app-server.rs
@@ -1140,6 +1140,15 @@ fn run() -> Result<(), Box> {
send(json!({"id": id, "result": {"data": turns, "nextCursor": null}}))?;
}
"thread/read" => {
+ if descendant_notifications
+ && message.pointer("/params/threadId").and_then(Value::as_str)
+ == Some("descendant-1")
+ {
+ send(json!({"id": id, "result": {"thread": {
+ "id": "descendant-1", "parentThreadId": state.thread_id
+ }}}))?;
+ continue;
+ }
if args
.iter()
.any(|arg| arg == "--require-lightweight-history")
@@ -1182,7 +1191,7 @@ fn run() -> Result<(), Box> {
{
send(json!({"method": "thread/started", "params": {"thread": {
"id": "descendant-overflow",
- "source": {"subAgent": {"thread_spawn": {"parent_thread_id": state.thread_id}}}
+ "parentThreadId": state.thread_id
}}}))?;
}
}
@@ -1431,10 +1440,26 @@ fn run() -> Result<(), Box> {
"params": {"turn": {"id": provider_turn_id}}
}))?;
if descendant_notifications {
- for index in 0..300 {
+ // Codex can announce a helper through the root's spawn receipt
+ // before emitting any thread/started notification for that helper.
+ send(json!({"method": "item/completed", "params": {
+ "threadId": state.thread_id, "turnId": provider_turn_id,
+ "item": {"id": "spawn-first-child", "type": "collabAgentToolCall",
+ "tool": "spawnAgent", "status": "completed",
+ "senderThreadId": state.thread_id,
+ "receiverThreadIds": ["descendant-0"]}
+ }}))?;
+ send(json!({"method": "turn/started", "params": {
+ "threadId": "descendant-0", "turnId": "first-child-turn"
+ }}))?;
+ // A helper turn may arrive before either spawn completion or thread/started.
+ send(json!({"method": "turn/started", "params": {
+ "threadId": "descendant-1", "turnId": "second-child-turn"
+ }}))?;
+ for index in 2..300 {
send(json!({"method": "thread/started", "params": {"thread": {
"id": format!("descendant-{index}"),
- "source": {"subAgent": {"thread_spawn": {"parent_thread_id": state.thread_id}}}
+ "parentThreadId": state.thread_id
}}}))?;
}
send(json!({"method": "turn/completed", "params": {
@@ -1444,7 +1469,7 @@ fn run() -> Result<(), Box> {
if args.iter().any(|value| value == "--descendant-overflow") {
send(json!({"method": "thread/started", "params": {"thread": {
"id": "descendant-overflow",
- "source": {"subAgent": {"thread_spawn": {"parent_thread_id": state.thread_id}}}
+ "parentThreadId": state.thread_id
}}}))?;
}
if fail_after_second_turn_start && turn_start_count == 2 {
diff --git a/packages/paperclip-runner/runner/crates/runner-core/src/codex_provider.rs b/packages/paperclip-runner/runner/crates/runner-core/src/codex_provider.rs
index 077225f3d6..3356b63fb3 100644
--- a/packages/paperclip-runner/runner/crates/runner-core/src/codex_provider.rs
+++ b/packages/paperclip-runner/runner/crates/runner-core/src/codex_provider.rs
@@ -65,6 +65,39 @@ fn remember_descendant_thread(ids: &mut BTreeSet, id: &str) -> Result,
+ root: &str,
+ method: &str,
+ params: &Value,
+) -> Result<(), &'static str> {
+ let item = ¶ms["item"];
+ if method != "item/completed"
+ || item["type"] != "collabAgentToolCall"
+ || item["tool"] != "spawnAgent"
+ || item["status"] != "completed"
+ {
+ return Ok(());
+ }
+ if params["threadId"] != root || item["senderThreadId"] != root {
+ return Err("invalid_spawn_lineage");
+ }
+ let receivers = item["receiverThreadIds"]
+ .as_array()
+ .ok_or("invalid_spawn_lineage")?;
+ if receivers.iter().any(|id| {
+ id.as_str()
+ .is_none_or(|id| id.is_empty() || id.len() > 240 || id == root)
+ }) {
+ return Err("invalid_spawn_lineage");
+ }
+ for id in receivers {
+ remember_descendant_thread(ids, id.as_str().expect("validated receiver"))?;
+ }
+ Ok(())
+}
+
type QuestionOptionLabels = BTreeMap>;
type QuestionSetMapping = (String, Value, QuestionOptionLabels);
@@ -2424,11 +2457,53 @@ impl CodexProvider {
) {
Ok(identity) => identity,
Err(_) => {
- return Ok(Some(self.identity_failure(
- method,
- ¶ms,
- "thread_binding_mismatch",
- )))
+ // Helpers can start before Codex publishes their spawn receipt.
+ // Verify lineage with the provider; never infer authority from
+ // the arrival of an otherwise foreign execution event.
+ let candidate = notification_thread_id(¶ms)
+ .filter(|id| !id.is_empty() && id.len() <= 240 && *id != self.thread_id)
+ .map(str::to_owned);
+ let verified = candidate.as_ref().is_some_and(|candidate| {
+ self.request(
+ "thread/read",
+ json!({"threadId": candidate, "includeTurns": false}),
+ )
+ .ok()
+ .is_some_and(|metadata| {
+ metadata.pointer("/thread/id").and_then(Value::as_str)
+ == Some(candidate.as_str())
+ && matches!(
+ classify_notification_thread(
+ "thread/started",
+ &self.thread_id,
+ &self.descendant_thread_ids,
+ &metadata
+ ),
+ Ok(NotificationThread::Descendant)
+ )
+ })
+ });
+ if verified {
+ let mut known = self.descendant_thread_ids.clone();
+ known.insert(candidate.expect("verified candidate"));
+ match classify_notification_thread(method, &self.thread_id, &known, ¶ms)
+ {
+ Ok(NotificationThread::Descendant) => NotificationThread::Descendant,
+ _ => {
+ return Ok(Some(self.identity_failure(
+ method,
+ ¶ms,
+ "thread_binding_mismatch",
+ )))
+ }
+ }
+ } else {
+ return Ok(Some(self.identity_failure(
+ method,
+ ¶ms,
+ "thread_binding_mismatch",
+ )));
+ }
}
};
if identity == NotificationThread::Descendant {
@@ -2535,6 +2610,22 @@ impl CodexProvider {
"turn_binding_mismatch",
)));
}
+ if let Err(code) = remember_spawned_descendants(
+ &mut self.descendant_thread_ids,
+ &self.thread_id,
+ method,
+ ¶ms,
+ ) {
+ if code == "provider_descendant_capacity_exhausted" {
+ return Ok(Some(CodexProviderEvent::ResourceLimit {
+ diagnostic: json!({"code": code, "recoverable": false,
+ "classification": "resource_capacity", "limit": MAX_DESCENDANT_THREAD_IDS,
+ "message": "Codex reached the child-thread inventory limit.",
+ "method": bounded_method(method), "expectedThreadId": self.thread_id}),
+ }));
+ }
+ return Ok(Some(self.identity_failure(method, ¶ms, code)));
+ }
if let Some(terminal_event_type) = terminal_event_type {
if self.active_provider_turn_id.is_none() {
return Err(LocalRunnerError::invalid(
@@ -3214,13 +3305,21 @@ fn classify_notification_thread(
if thread.is_none() || thread == Some(root) {
return Ok(NotificationThread::Root);
}
- let parent = [
+ let parents: Vec<&str> = [
+ "/thread/parentThreadId",
"/thread/source/subAgent/thread_spawn/parent_thread_id",
"/thread/source/subAgent/threadSpawn/parentThreadId",
"/thread/source/subagent/thread_spawn/parent_thread_id",
]
.iter()
- .find_map(|path| params.pointer(path).and_then(Value::as_str));
+ .filter_map(|path| params.pointer(path).and_then(Value::as_str))
+ .collect();
+ if parents.windows(2).any(|pair| pair[0] != pair[1]) {
+ return Err(LocalRunnerError::invalid(
+ "Codex notification has conflicting parent identity",
+ ));
+ }
+ let parent = parents.first().copied();
if thread.is_some()
&& (thread.is_some_and(|id| descendants.contains(id))
|| (method == "thread/started"
@@ -4781,6 +4880,93 @@ mod notification_identity_tests {
NotificationThread::Root
);
}
+ #[test]
+ fn spawn_receipts_require_completed_root_authority_and_preserve_capacity() {
+ let receipt = json!({"threadId":"root", "turnId":"turn", "item":{
+ "type":"collabAgentToolCall", "tool":"spawnAgent", "status":"completed",
+ "senderThreadId":"root", "receiverThreadIds":["helper"]}});
+ let mut ids = BTreeSet::new();
+ remember_spawned_descendants(&mut ids, "root", "item/completed", &receipt).unwrap();
+ assert!(ids.contains("helper"));
+ assert_eq!(
+ classify_notification_thread(
+ "turn/started",
+ "root",
+ &ids,
+ &json!({"threadId":"helper", "turn":{"id":"child-turn"}})
+ )
+ .unwrap(),
+ NotificationThread::Descendant
+ );
+ for (field, value) in [
+ ("senderThreadId", "foreign"),
+ ("receiverThreadIds", "malformed"),
+ ] {
+ let mut bad = receipt.clone();
+ bad["item"][field] = json!(value);
+ let mut empty = BTreeSet::new();
+ assert!(
+ remember_spawned_descendants(&mut empty, "root", "item/completed", &bad).is_err()
+ );
+ assert!(empty.is_empty());
+ }
+ for (field, value) in [
+ ("tool", "sendInput"),
+ ("status", "failed"),
+ ("status", "inProgress"),
+ ] {
+ let mut non_spawn = receipt.clone();
+ non_spawn["item"][field] = json!(value);
+ let mut empty = BTreeSet::new();
+ remember_spawned_descendants(&mut empty, "root", "item/completed", &non_spawn).unwrap();
+ assert!(empty.is_empty());
+ }
+ let mut full: BTreeSet = (0..MAX_DESCENDANT_THREAD_IDS)
+ .map(|n| format!("child-{n}"))
+ .collect();
+ assert_eq!(
+ remember_spawned_descendants(&mut full, "root", "item/completed", &receipt),
+ Err("provider_descendant_capacity_exhausted")
+ );
+ assert_eq!(full.len(), MAX_DESCENDANT_THREAD_IDS);
+ }
+
+ #[test]
+ fn recognizes_explicit_parent_thread_lineage_without_granting_root_authority() {
+ let children = BTreeSet::from(["child".to_owned()]);
+ for parent in ["root", "child"] {
+ assert_eq!(
+ classify_notification_thread(
+ "thread/started",
+ "root",
+ &children,
+ &json!({"thread":{"id":"helper", "parentThreadId":parent}})
+ )
+ .unwrap(),
+ NotificationThread::Descendant
+ );
+ }
+ assert_eq!(
+ classify_notification_thread(
+ "thread/started",
+ "root",
+ &children,
+ &json!({"thread":{"id":"stranger", "parentThreadId":"foreign"}})
+ )
+ .unwrap(),
+ NotificationThread::UnrelatedInformation
+ );
+ assert!(classify_notification_thread(
+ "turn/started",
+ "root",
+ &children,
+ &json!({"threadId":"stranger", "parentThreadId":"root", "turn":{"id":"foreign-turn"}})
+ )
+ .is_err());
+ assert!(classify_notification_thread("thread/started", "root", &children,
+ &json!({"thread":{"id":"helper", "parentThreadId":"root", "source":{"subAgent":{"thread_spawn":{"parent_thread_id":"foreign"}}}}})).is_err());
+ }
+
#[test]
fn classifies_provider_lineage_before_root_authority() {
let children = BTreeSet::from(["child".to_owned()]);
diff --git a/packages/paperclip-runner/runner/crates/runner-core/tests/acpx_sidecar_transport.rs b/packages/paperclip-runner/runner/crates/runner-core/tests/acpx_sidecar_transport.rs
index a1c2b27415..eed4288783 100644
--- a/packages/paperclip-runner/runner/crates/runner-core/tests/acpx_sidecar_transport.rs
+++ b/packages/paperclip-runner/runner/crates/runner-core/tests/acpx_sidecar_transport.rs
@@ -214,3 +214,52 @@ fn preserves_only_allowlisted_stderr_categories_when_the_process_exits() {
assert!(!message.contains(sensitive));
}
}
+
+#[test]
+fn assigned_gateway_binding_reaches_qualified_sidecar_without_unrelated_secrets() {
+ const CHILD: &str = "PAPERCLIP_TEST_MCP_ENV_CHILD";
+ if std::env::var_os(CHILD).is_none() {
+ let status = std::process::Command::new(std::env::current_exe().unwrap())
+ .args([
+ "--exact",
+ "assigned_gateway_binding_reaches_qualified_sidecar_without_unrelated_secrets",
+ "--nocapture",
+ ])
+ .env(CHILD, "1")
+ .env("PAPERCLIP_NATIVE_MCP_NAME", "paperclip-assigned")
+ .env("PAPERCLIP_NATIVE_MCP_URL", "http://127.0.0.1:3100/mcp")
+ .env(
+ "PAPERCLIP_NATIVE_MCP_TOKEN",
+ "fixture-token-never-returned-in-test-output",
+ )
+ .env("UNRELATED_EVAL_SECRET", "must-not-cross-boundary")
+ .status()
+ .unwrap();
+ assert!(status.success(), "isolated gateway environment test failed");
+ return;
+ }
+ for agent in ["claude", "codex"] {
+ let mut sidecar = AcpxSidecarTransport::start_for_agent(
+ &AcpxSidecarTransportConfig {
+ command: PathBuf::from(env!("CARGO_BIN_EXE_fake-acpx-sidecar")),
+ args: vec!["--mode".into(), "mcp-environment".into()],
+ verified_launch: None,
+ request_timeout: Duration::from_secs(2),
+ shutdown_grace: Duration::from_millis(50),
+ },
+ agent,
+ )
+ .unwrap();
+ let response = sidecar
+ .request(GeneratedAcpxSidecarCommand::Initialize, json!({}))
+ .unwrap();
+ assert_eq!(response["name"], "paperclip-assigned");
+ assert_eq!(response["url"], "http://127.0.0.1:3100/mcp");
+ assert_eq!(
+ response["hasToken"], true,
+ "assigned gateway credential was dropped"
+ );
+ assert_eq!(response["hasUnrelatedSecret"], false);
+ sidecar.shutdown().unwrap();
+ }
+}
diff --git a/packages/paperclip-runner/src/contracts/runtime-context.ts b/packages/paperclip-runner/src/contracts/runtime-context.ts
index 259d75fe88..f9d1c1f74a 100644
--- a/packages/paperclip-runner/src/contracts/runtime-context.ts
+++ b/packages/paperclip-runner/src/contracts/runtime-context.ts
@@ -1,8 +1,8 @@
import { createHash } from "node:crypto";
export const NATIVE_RUNTIME_ASSET_SCHEMA = "paperclip.runtime-asset.v1" as const;
-export const PAPERCLIP_EXECUTION_PROMPT_REVISION = "paperclip-execution.v2" as const;
-export const PAPERCLIP_EXECUTION_PROMPT = "You are running as a Paperclip agent. Complete the assigned task in the provided execution environment. Follow the attached agent instructions and use assigned skills and tools when relevant. Use Paperclip tools for coordination. When a task needs an external service, use installed tools if available; otherwise use connections_search to discover catalog services or authorized configured connections, then connection_request with the returned service identifier. The request appears as a card in the task. Finish independent work before yielding for access; do not poll or request the same connection repeatedly. Paperclip will continue automatically with updated tools after resolution. After a decline, pursue alternatives unless the user explicitly asks to retry. Finish exactly once with `paperclip_finish` or `paperclip_block`." as const;
+export const PAPERCLIP_EXECUTION_PROMPT_REVISION = "paperclip-execution.v3" as const;
+export const PAPERCLIP_EXECUTION_PROMPT = "You are running as a Paperclip agent. Complete the assigned task in the provided execution environment. Follow the attached agent instructions and use assigned skills and tools when relevant. Use Paperclip tools for coordination. Hire persistent teammates through Paperclip hiring; provider helper threads do not create Paperclip agents. Delegate with create_task. When remaining work depends on a child task, use set_dependencies to add its ID while preserving existing blocker IDs. Complete independent work, then call paperclip_block with the child agent as owner and child completion as the unblock action. End the turn so the child can use the workspace. Do not sleep or poll for child results while holding the workspace. Paperclip resumes the parent when the dependency completes. When a task needs an external service, use installed tools if available; otherwise use connections_search to discover catalog services or authorized configured connections, then connection_request with the returned service identifier. The request appears as a card in the task. Finish independent work before yielding for access; do not poll or request the same connection repeatedly. Paperclip will continue automatically with updated tools after resolution. After a decline, pursue alternatives unless the user explicitly asks to retry. Finish exactly once with `paperclip_finish` or `paperclip_block`." as const;
export interface NativeRuntimeAssetReference {
schema: typeof NATIVE_RUNTIME_ASSET_SCHEMA;
diff --git a/packages/paperclip-runner/src/drivers/acpx/runtime-host.test.ts b/packages/paperclip-runner/src/drivers/acpx/runtime-host.test.ts
index 9d39cf6b23..fdae735e86 100644
--- a/packages/paperclip-runner/src/drivers/acpx/runtime-host.test.ts
+++ b/packages/paperclip-runner/src/drivers/acpx/runtime-host.test.ts
@@ -175,6 +175,40 @@ afterEach(async () => {
});
describe("ACPX runtime host", () => {
+ it.each([
+ { PAPERCLIP_NATIVE_MCP_NAME: "paperclip-assigned" },
+ { PAPERCLIP_NATIVE_MCP_NAME: "paperclip", PAPERCLIP_NATIVE_MCP_URL: "http://127.0.0.1:3211/mcp", PAPERCLIP_NATIVE_MCP_TOKEN: "x".repeat(40) },
+ { PAPERCLIP_NATIVE_MCP_NAME: "paperclip-assigned", PAPERCLIP_NATIVE_MCP_URL: "http://external.example/mcp", PAPERCLIP_NATIVE_MCP_TOKEN: "x".repeat(40) },
+ ])("rejects invalid assigned connection bindings before runtime launch", async (environment) => {
+ const fixture = await hostFixture();
+ const openRuntime = vi.fn();
+ await expect(AcpxRuntimeHost.open({ ...fixture.options, environment }, fixture.dependencies({ openRuntime }))).rejects.toThrow(/assigned native MCP/);
+ expect(openRuntime).not.toHaveBeenCalled();
+ });
+
+ it("registers the assigned connection gateway alongside the Claude task bridge", async () => {
+ const fixture = await hostFixture();
+ let servers: AcpxRuntimePortOpenOptions["mcpServers"] = [];
+ const host = await AcpxRuntimeHost.open({
+ ...fixture.options,
+ agent: "claude", model: "claude-sonnet-5",
+ environment: { ...fixture.options.environment,
+ PAPERCLIP_NATIVE_MCP_NAME: "paperclip-assigned",
+ PAPERCLIP_NATIVE_MCP_URL: "http://127.0.0.1:3211/mcp/gateway",
+ PAPERCLIP_NATIVE_MCP_TOKEN: "fixture-gateway-token-".repeat(3),
+ },
+ semanticTools: { tools: [], handler: async () => ({}) },
+ }, fixture.dependencies({ openRuntime: async (options) => {
+ servers = options.mcpServers;
+ return runtimePort({ getStatus: async () => ({ models: { currentModelId: "claude-sonnet-5" } }) });
+ } }));
+ try {
+ expect(servers.map(server => server.name)).toEqual(["paperclip", "paperclip-assigned"]);
+ expect(servers[1]).toEqual({ name: "paperclip-assigned", url: "http://127.0.0.1:3211/mcp/gateway",
+ bearerToken: "fixture-gateway-token-".repeat(3), runnerOwned: true });
+ } finally { await host.close({ reason: "gateway test complete" }); }
+ });
+
it("automatically permits only admitted Paperclip reads in the Claude SDK", async () => {
const fixture = await hostFixture();
const dependencies = fixture.dependencies({
diff --git a/packages/paperclip-runner/src/drivers/acpx/runtime-host.ts b/packages/paperclip-runner/src/drivers/acpx/runtime-host.ts
index dc0c7f81ef..072dba41a0 100644
--- a/packages/paperclip-runner/src/drivers/acpx/runtime-host.ts
+++ b/packages/paperclip-runner/src/drivers/acpx/runtime-host.ts
@@ -1,4 +1,5 @@
import { join } from "node:path";
+import { nativeMcpLaunchBinding } from "../native-mcp.js";
import type {
AcpElicitationHandler,
@@ -287,6 +288,13 @@ export class AcpxRuntimeHost {
"ACPX pi is unavailable until its runtime has descriptor-confined verified launch",
);
}
+ const nativeMcp = nativeMcpLaunchBinding(options.environment);
+ if (nativeMcp?.name === "paperclip") {
+ throw new Error("assigned native MCP name conflicts with the task bridge");
+ }
+ if (options.runtimeContext?.mcp.bindingId && !nativeMcp) {
+ throw new Error("assigned native MCP launch binding is unavailable");
+ }
const profile = resolveQualifiedAcpxProfile(options.agent, options.model);
const binding = await runAbortableAdmissionStage(
options.signal,
@@ -468,16 +476,14 @@ export class AcpxRuntimeHost {
? {}
: { assertWorkspaceHeld: options.assertWorkspaceHeld }),
...(options.signal === undefined ? {} : { signal: options.signal }),
- mcpServers: toolBridge
- ? [
- {
- name: "paperclip",
- url: toolBridge.url,
- bearerToken: toolBridge.secret,
- runnerOwned: true,
- },
- ]
- : [],
+ mcpServers: [
+ ...(toolBridge ? [{ name: "paperclip", url: toolBridge.url,
+ bearerToken: toolBridge.secret, runnerOwned: true }] : []),
+ // This is Paperclip's authenticated gateway, not a direct upstream
+ // binding. Its existing grants and approval checks remain authoritative.
+ ...(nativeMcp ? [{ name: nativeMcp.name, url: nativeMcp.url,
+ bearerToken: nativeMcp.token, runnerOwned: true }] : []),
+ ],
...(options.onGoalUpdate === undefined
? {}
: { onGoalUpdate: options.onGoalUpdate }),
diff --git a/packages/paperclip-runner/src/live/runnerd-codex-transport.test.ts b/packages/paperclip-runner/src/live/runnerd-codex-transport.test.ts
index c559d58218..a37d519d57 100644
--- a/packages/paperclip-runner/src/live/runnerd-codex-transport.test.ts
+++ b/packages/paperclip-runner/src/live/runnerd-codex-transport.test.ts
@@ -8848,3 +8848,25 @@ it("persists an active provider as settled before bounded suspension", async ()
await rm(stateDirectory, { recursive: true, force: true });
}
}, 30_000);
+
+
+it.each(["claude", "codex"] as const)("keeps the explicitly assigned gateway in the %s runner environment", (agent) => {
+ const gateway = {
+ PAPERCLIP_NATIVE_MCP_NAME: "paperclip-assigned",
+ PAPERCLIP_NATIVE_MCP_URL: "http://127.0.0.1:3100/mcp/gateway",
+ PAPERCLIP_NATIVE_MCP_TOKEN: "fixture-scoped-gateway-token-1234567890",
+ };
+ const environment = createCapabilityRunnerdProviderEnvironment({
+ provider: "acpx",
+ options: { provider: "acpx", acpxAgent: agent, environment: {
+ PATH: "/bin", ...gateway, PAPERCLIP_API_KEY: "must-not-cross", DATABASE_URL: "must-not-cross",
+ } },
+ identity: { runnerInstanceId: "runner-1", environmentLeaseId: "lease-1", runId: "run-1",
+ normalizedSessionId: "session-1", turnId: "turn-1", itemId: "item-1" },
+ codexHome: "/isolated/home", runtimeContextPath: "/isolated/context.json", hasRuntimeContext: true,
+ acpxSidecarPath: "/verified/provider-pack/dist/cli/acpx-runtime-sidecar.cjs",
+ });
+ expect(environment).toMatchObject(gateway);
+ expect(environment.PAPERCLIP_API_KEY).toBeUndefined();
+ expect(environment.DATABASE_URL).toBeUndefined();
+});
diff --git a/packages/paperclip-runner/src/live/runnerd-codex-transport.ts b/packages/paperclip-runner/src/live/runnerd-codex-transport.ts
index 59e787df3a..41b9c3fc53 100644
--- a/packages/paperclip-runner/src/live/runnerd-codex-transport.ts
+++ b/packages/paperclip-runner/src/live/runnerd-codex-transport.ts
@@ -3098,7 +3098,14 @@ export function createCapabilityRunnerdProviderEnvironment(input: {
input.options.acpxSidecarPath ??
resolve(packageRoot, "dist", "cli", "acpx-runtime-sidecar.cjs");
const providerPackageAuthority = acpxProviderPackageAuthority(sidecarPath);
+ // This is the trusted runner/sidecar boundary. The provider sandbox still
+ // uses createSanitizedAcpxSpawnInput and does not inherit gateway tokens.
+ const assignedGateway = input.options.acpxAgent === "pi"
+ ? null : nativeMcpLaunchBinding(input.options.environment ?? {});
return {
+ ...(assignedGateway ? {
+ PAPERCLIP_NATIVE_MCP_TOKEN: assignedGateway.token,
+ } : {}),
...createSanitizedAcpxSpawnInput(
input.options.environment,
input.options.acpxAgent ?? "codex",
diff --git a/packages/shared/src/connection-intent-guidance.ts b/packages/shared/src/connection-intent-guidance.ts
index d07b246a29..7490a6ab8d 100644
--- a/packages/shared/src/connection-intent-guidance.ts
+++ b/packages/shared/src/connection-intent-guidance.ts
@@ -11,6 +11,7 @@ export const CONNECTION_INTENT_AGENT_GUIDANCE = [
"- This applies both when the user explicitly asks to connect a service and when the requested work implicitly depends on that service.",
"- If search returns `ready`, use the installed connection; do not create a connection intent.",
"- If search returns `available` or `needs_user_action`, call `connection_request` with the returned service identifier.",
+ "- When the user has already asked to connect a known service, use the real connection request. Do not ask whether to connect again or imitate the Connect / Not now card with `ask_user_questions`, a generic confirmation, or a comment. Only `connection_request` creates the actual connection setup card.",
"- If search returns `unavailable`, explain that the service is unavailable and do not call `connection_request`.",
"- If `connection_request` returns `needs_user_action`, finish any independent work, then yield in a waiting posture. Do not retry the request, ask for credentials in comments, or claim access.",
"- Do not use connection tools for arbitrary MCP URLs, unsupported services, or work that does not require an external service.",
@@ -21,10 +22,12 @@ export const CONNECTION_INTENT_AGENT_GUIDANCE = [
export const CONNECTIONS_SEARCH_TOOL_DESCRIPTION = [
"Search Paperclip's catalog services and authorized configured custom connections and report this run's agent-relative access state.",
"Use it when work requires a known external service and usable access is uncertain; do not use it for arbitrary MCP URLs or unrelated work.",
+ "If the user already asked to connect the service, search and request it through the connection tools instead of asking a generic confirmation question.",
].join(" ");
export const CONNECTION_REQUEST_TOOL_DESCRIPTION = [
"Request access to a known connectable service for this run's agent from the responsible user.",
+ "This creates the real Connect / Not now setup card. When the user already asked to connect, use this tool; do not recreate those choices with ask_user_questions, request_confirmation, or a comment. A generic question does not start connection setup.",
"Call it only with the service identifier returned as available or needs_user_action by connections_search; if user action is needed, finish independent work, then yield without retrying or asking for credentials in comments.",
].join(" ");
diff --git a/server/src/__tests__/heartbeat-process-recovery.test.ts b/server/src/__tests__/heartbeat-process-recovery.test.ts
index cd1cee5419..19acf5fd4f 100644
--- a/server/src/__tests__/heartbeat-process-recovery.test.ts
+++ b/server/src/__tests__/heartbeat-process-recovery.test.ts
@@ -12607,6 +12607,44 @@ describeEmbeddedPostgres("heartbeat orphaned process recovery", () => {
if (retryRun) await waitForRunToSettle(heartbeat, retryRun.id);
});
+ it("lets a child waiting for the shared workspace run before recovering its lead", async () => {
+ const { companyId, agentId, issueId, runId } = await seedStrandedIssueFixture({ status: "in_progress", runStatus: "succeeded", livenessState: "advanced" });
+ await db.update(heartbeatRuns).set({ contextSnapshot: { issueId, paperclipWorkspace: { mode: "shared_workspace" } } }).where(eq(heartbeatRuns.id, runId));
+ const projectId = randomUUID(), workspaceId = randomUUID(), childId = randomUUID(), workerId = randomUUID();
+ await db.insert(projects).values({ id: projectId, companyId, name: "Shared project" });
+ await db.insert(projectWorkspaces).values({ id: workspaceId, companyId, projectId, name: "Primary", sourceType: "local_path", cwd: "/tmp/recovery-shared", isPrimary: true });
+ await db.update(issues).set({ projectId, projectWorkspaceId: workspaceId }).where(eq(issues.id, issueId));
+ await db.insert(agents).values({ id: workerId, companyId, name: "Worker", role: "engineer", status: "idle", adapterType: "codex_local", adapterConfig: {} });
+ await db.insert(issues).values({ id: childId, companyId, parentId: issueId, title: "Build the project", status: "in_progress", assigneeAgentId: workerId, projectId, projectWorkspaceId: workspaceId });
+ await db.insert(heartbeatRuns).values({ companyId, agentId: workerId, status: "scheduled_retry", scheduledRetryAt: new Date(Date.now() + 60_000), scheduledRetryReason: "workspace_busy", contextSnapshot: { issueId: childId } });
+ const heartbeat = heartbeatService(db);
+ const result = await heartbeat.reconcileStrandedAssignedIssues();
+ expect(result.continuationRequeued).toBe(0);
+ expect(await db.select().from(heartbeatRuns).where(eq(heartbeatRuns.agentId, agentId))).toHaveLength(1);
+ // Once the child finishes, normal recovery can continue the lead.
+ await db.update(issues).set({ status: "done" }).where(eq(issues.id, childId));
+ expect((await heartbeat.reconcileStrandedAssignedIssues()).continuationRequeued).toBe(1);
+ const next = (await db.select().from(heartbeatRuns).where(eq(heartbeatRuns.agentId, agentId))).find((r) => r.id !== runId);
+ if (next) await waitForRunToSettle(heartbeat, next.id);
+ });
+
+ it("resumes a shared-workspace lead when its child needs review", async () => {
+ const { companyId, agentId, issueId, runId } = await seedStrandedIssueFixture({ status: "in_progress", runStatus: "succeeded", livenessState: "advanced" });
+ await db.update(heartbeatRuns).set({ contextSnapshot: { issueId, paperclipWorkspace: { mode: "shared_workspace" } } }).where(eq(heartbeatRuns.id, runId));
+ const projectId = randomUUID(), workspaceId = randomUUID(), childId = randomUUID(), workerId = randomUUID();
+ await db.insert(projects).values({ id: projectId, companyId, name: "Shared project" });
+ await db.insert(projectWorkspaces).values({ id: workspaceId, companyId, projectId, name: "Primary", sourceType: "local_path", cwd: "/tmp/recovery-shared", isPrimary: true });
+ await db.update(issues).set({ projectId, projectWorkspaceId: workspaceId }).where(eq(issues.id, issueId));
+ await db.insert(agents).values({ id: workerId, companyId, name: "Worker", role: "engineer", status: "idle", adapterType: "codex_local", adapterConfig: {} });
+ await db.insert(issues).values({ id: childId, companyId, parentId: issueId, title: "Review the project", status: "in_review", assigneeAgentId: workerId, projectId, projectWorkspaceId: workspaceId });
+ await db.insert(issueThreadInteractions).values({ companyId, issueId: childId, kind: "request_confirmation", status: "pending", addresseeAgentId: agentId, payload: { prompt: "Review this delivery" } });
+ const heartbeat = heartbeatService(db);
+ const result = await heartbeat.reconcileStrandedAssignedIssues();
+ expect(result.continuationRequeued).toBe(1);
+ const next = (await db.select().from(heartbeatRuns).where(eq(heartbeatRuns.agentId, agentId))).find((r) => r.id !== runId);
+ if (next) await waitForRunToSettle(heartbeat, next.id);
+ });
+
it("leaves the productive-but-stranded continuation path unchanged under the new classifier", async () => {
const { agentId, issueId, runId } = await seedStrandedIssueFixture({
status: "in_progress",
diff --git a/server/src/modules/wake-queue/adapters/postgres.test.ts b/server/src/modules/wake-queue/adapters/postgres.test.ts
index 09f7d11e3e..91a68b6b87 100644
--- a/server/src/modules/wake-queue/adapters/postgres.test.ts
+++ b/server/src/modules/wake-queue/adapters/postgres.test.ts
@@ -4,6 +4,7 @@ import { eq } from "drizzle-orm";
import { afterAll, afterEach, beforeAll, describe, expect, it } from "vitest";
import type { Db } from "@paperclipai/db";
import {
+ activityLog,
agentWakeupRequests,
agents,
companies,
@@ -56,6 +57,7 @@ describeEmbeddedPostgres("wake-queue postgres adapter", () => {
}, 20_000);
afterEach(async () => {
+ await db.delete(activityLog);
await db.delete(issueComments);
// `heartbeat_runs.wakeup_request_id` references `agent_wakeup_requests.id`,
// so the run row must go first.
@@ -316,6 +318,62 @@ describeEmbeddedPostgres("wake-queue postgres adapter", () => {
});
}
+ it.each(["in_progress", "blocked"])("preserves recovery ownership and queued messages when a native task fails from %s", async (status) => {
+ const companyId = await seedCompany();
+ const agentId = await seedAgent({ companyId });
+ const issueId = await seedIssue({ companyId, assigneeAgentId: agentId, status });
+ const runId = await seedRun({ companyId, agentId, status: "failed", contextSnapshot: { issueId } });
+ await db.update(heartbeatRuns).set({ runtimeMode: "native", errorCode: "thread_binding_mismatch" }).where(eq(heartbeatRuns.id, runId));
+ const wakeId = await seedDeferredWake({ companyId, agentId, issueId });
+ const adapter = createPostgresWakeQueueAdapter(db, stubDeps);
+ await adapter.withIssueExecutionLock({ companyId, runId, now: new Date() }, async () => { throw new Error("must not replay an uncertain execution"); });
+ const blockedIssue = (await db.select().from(issues).where(eq(issues.id, issueId)))[0];
+ expect(blockedIssue.status).toBe("blocked");
+ const entries = await db.select().from(activityLog).where(eq(activityLog.entityId, issueId));
+ if (status === "in_progress") {
+ expect(blockedIssue.blockedTransitionAt).not.toBeNull();
+ expect(entries[0]).toMatchObject({ action: "issue.updated", details: { status: "blocked", previousStatus: "in_progress" } });
+ } else expect(entries).toHaveLength(0);
+ expect((await db.select().from(agentWakeupRequests).where(eq(agentWakeupRequests.id, wakeId)))[0].status).toBe("deferred_issue_execution");
+ const action = (await db.select().from(issueRecoveryActions).where(eq(issueRecoveryActions.sourceIssueId, issueId)))[0];
+ expect(action).toMatchObject({ ownerType: "board", cause: "native_continuation_requires_reconciliation" });
+ if (status === "in_progress") expect(action.evidence).toMatchObject({ nativeFailureBlock: { runId, statusVersion: blockedIssue.statusVersion } });
+ else expect(action.evidence.nativeFailureBlock).toBeUndefined();
+ await adapter.withIssueExecutionLock({ companyId, runId, now: new Date() }, async () => { throw new Error("must not replay"); });
+ expect(await db.select().from(issueRecoveryActions).where(eq(issueRecoveryActions.sourceIssueId, issueId))).toHaveLength(1);
+ expect((await db.select().from(issues).where(eq(issues.id, issueId)))[0].statusVersion).toBe(blockedIssue.statusVersion);
+ });
+
+ it.each(["active", "escalated"])("repairs a failed native task with an existing %s recovery action", async (status) => {
+ const companyId = await seedCompany();
+ const agentId = await seedAgent({ companyId });
+ const issueId = await seedIssue({ companyId, assigneeAgentId: agentId, status: "in_progress" });
+ const runId = await seedRun({ companyId, agentId, status: "failed", contextSnapshot: { issueId } });
+ await db.update(heartbeatRuns).set({ runtimeMode: "native", errorCode: "runner_lost" }).where(eq(heartbeatRuns.id, runId));
+ const wakeId = await seedDeferredWake({ companyId, agentId, issueId });
+ const [existing] = await db.insert(issueRecoveryActions).values({
+ companyId, sourceIssueId: issueId, status, kind: "active_run_watchdog",
+ ownerType: "board", cause: "native_runner_restart_unverified", fingerprint: `restart:${runId}`,
+ evidence: { runId, priorProof: "keep", automaticRecovery: { attempts: 2 } },
+ nextAction: "Verify the previous execution stopped", attemptCount: 2, maxAttempts: 3,
+ }).returning();
+ const adapter = createPostgresWakeQueueAdapter(db, stubDeps);
+ const release = () => adapter.withIssueExecutionLock({ companyId, runId, now: new Date() }, async () => { throw new Error("must not replay"); });
+ await release();
+ const [blocked] = await db.select().from(issues).where(eq(issues.id, issueId));
+ expect(blocked.status).toBe("blocked");
+ const actions = await db.select().from(issueRecoveryActions).where(eq(issueRecoveryActions.sourceIssueId, issueId));
+ expect(actions).toHaveLength(1);
+ expect(actions[0]).toMatchObject({ id: existing.id, status, cause: existing.cause,
+ ownerType: "board", attemptCount: 2, maxAttempts: 3, nextAction: existing.nextAction,
+ evidence: { runId, priorProof: "keep", automaticRecovery: { attempts: 2 },
+ nativeFailureBlock: { runId, statusVersion: blocked.statusVersion } } });
+ expect((await db.select().from(agentWakeupRequests).where(eq(agentWakeupRequests.id, wakeId)))[0].status).toBe("deferred_issue_execution");
+ await release();
+ expect((await db.select().from(issues).where(eq(issues.id, issueId)))[0].statusVersion).toBe(blocked.statusVersion);
+ expect(await db.select().from(activityLog).where(eq(activityLog.entityId, issueId))).toHaveLength(1);
+ });
+
it.each(["queued", "running", "scheduled_retry"])("does not promote another turn behind a %s successor without an execution lock", async (status) => {
const companyId = await seedCompany();
const agentId = await seedAgent({ companyId });
diff --git a/server/src/modules/wake-queue/adapters/postgres.ts b/server/src/modules/wake-queue/adapters/postgres.ts
index 1df9186a5f..24fd0a2146 100644
--- a/server/src/modules/wake-queue/adapters/postgres.ts
+++ b/server/src/modules/wake-queue/adapters/postgres.ts
@@ -6,6 +6,7 @@ import { and, asc, eq, inArray, isNull, notInArray, or, sql } from "drizzle-orm"
import type { Db } from "@paperclipai/db";
import { extractIssueReferenceIdentifiers } from "@paperclipai/shared";
import {
+ activityLog,
agentWakeupRequests,
agents,
chatActions,
@@ -720,7 +721,7 @@ async function recordNativeTerminalRecoveryIfNeeded(tx: Db, run: HeartbeatRunRow
if (!applies || isAcknowledgedNativeStop(run)) return false;
const existing = await tx
- .select({ id: issueRecoveryActions.id })
+ .select({ id: issueRecoveryActions.id, evidence: issueRecoveryActions.evidence })
.from(issueRecoveryActions)
.where(
and(
@@ -733,6 +734,27 @@ async function recordNativeTerminalRecoveryIfNeeded(tx: Db, run: HeartbeatRunRow
),
)
.limit(1);
+ let nativeFailureBlock: { runId: string; statusVersion: number } | undefined;
+ if (issue.status !== "blocked") {
+ const projected = await issueService(tx).update(issue.id, { status: "blocked" }, tx);
+ if (projected) {
+ nativeFailureBlock = { runId: run.id, statusVersion: projected.statusVersion };
+ await tx.insert(activityLog).values({
+ companyId: issue.companyId, actorType: "system", actorId: "execution-recovery",
+ action: "issue.updated", entityType: "issue", entityId: issue.id, runId: run.id,
+ details: { status: "blocked", previousStatus: issue.status, reason: "native_continuation_requires_reconciliation" },
+ });
+ }
+ }
+ // Status projection is required even when restart/finalization created the
+ // incident first. Preserve its owner, cause, retry budget, and prior evidence.
+ if (nativeFailureBlock) {
+ for (const action of existing) {
+ await tx.update(issueRecoveryActions).set({
+ evidence: { ...action.evidence, nativeFailureBlock }, updatedAt: now,
+ }).where(and(eq(issueRecoveryActions.id, action.id), eq(issueRecoveryActions.companyId, issue.companyId)));
+ }
+ }
if (!existing.length) {
await tx
.update(nativeRunFinalizations)
@@ -760,7 +782,7 @@ async function recordNativeTerminalRecoveryIfNeeded(tx: Db, run: HeartbeatRunRow
returnOwnerAgentId: run.agentId,
cause: "native_continuation_requires_reconciliation",
fingerprint: `native-continuation:${run.id}`,
- evidence: { runId: run.id, originalFailureCode: run.errorCode },
+ evidence: { runId: run.id, originalFailureCode: run.errorCode, ...(nativeFailureBlock ? { nativeFailureBlock } : {}) },
nextAction:
"Inspect the original failure and reconcile the previous execution before continuing. Automatic recovery cannot start another incident.",
maxAttempts: 3,
diff --git a/server/src/services/native-runtime/native-safe-replacement.test.ts b/server/src/services/native-runtime/native-safe-replacement.test.ts
index 480a156ff3..3d6484a1b3 100644
--- a/server/src/services/native-runtime/native-safe-replacement.test.ts
+++ b/server/src/services/native-runtime/native-safe-replacement.test.ts
@@ -15,8 +15,10 @@ import {
import { randomUUID } from "node:crypto";
import { appendHeartbeatRunEvent } from "../heartbeat-run-events.js";
import { tmpdir } from "node:os";
-import { and, eq, inArray } from "drizzle-orm";
-import { afterAll, beforeAll, describe, expect, it, vi } from "vitest";
+import { mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
+import { join } from "node:path";
+import { and, eq, inArray, sql } from "drizzle-orm";
+import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "vitest";
import {
agents,
companies,
@@ -54,6 +56,11 @@ const support = externalDatabaseUrl
);
db = createDb(database.connectionString);
}, 30_000);
+ afterEach(async () => {
+ // Each sweep scans all companies. Keep earlier tests' unresolved runs out
+ // of later tests so every case exercises only its own recovery fixtures.
+ await db.execute(sql`TRUNCATE companies CASCADE`);
+ });
afterAll(async () => {
if (externalDatabaseUrl) await db?.$client.end();
else await database?.cleanup();
@@ -109,6 +116,73 @@ const support = externalDatabaseUrl
});
return { companyId, agentId, issueId, runId };
}
+ it.each(["verified", "unproven", "unknown_action"] as const)(
+ "preserves saved work and a queued request at a controlled recovery boundary (%s)",
+ async (mode) => {
+ // The stopped-session verifier is the injected boundary here. These are
+ // scheduler/database tests, not evidence that SIGKILL is a safe boundary.
+ const workspace = await mkdtemp(join(tmpdir(), "paperclip-recovery-work-"));
+ try {
+ const file = join(workspace, "saved.txt");
+ const saved = "Completed work from before the interruption.\n";
+ await writeFile(file, saved);
+ const source = await seed(2);
+ await db.update(heartbeatRuns).set({ runnerProfileJson: {
+ recoveryEventInventoryVersion: 1,
+ nativeExecutionInput: { provider: { kind: "codex" }, workspace: { cwd: workspace } },
+ } }).where(eq(heartbeatRuns.id, source.runId));
+ await db.update(nativeRunFinalizations).set({ failureCode: "native_session_cleanup_quarantined" })
+ .where(eq(nativeRunFinalizations.runId, source.runId));
+ const projected = await issueService(db).update(source.issueId, { status: "blocked" });
+ await db.insert(issueRecoveryActions).values({
+ companyId: source.companyId, sourceIssueId: source.issueId,
+ kind: "active_run_watchdog", cause: "native_session_cleanup_quarantined", fingerprint: source.runId,
+ ownerType: "board", returnOwnerAgentId: source.agentId, status: "resolved", outcome: "blocked",
+ evidence: { runId: source.runId, nativeFailureBlock: { runId: source.runId, statusVersion: projected!.statusVersion } },
+ nextAction: "Verify the stopped execution before restarting.",
+ });
+ const body = "Keep the saved work and explain the result.";
+ const comment = await issueService(db).addComment(source.issueId, body, { userId: "operator" });
+ if (mode === "unknown_action") await appendHeartbeatRunEvent(db, {
+ companyId: source.companyId, runId: source.runId, agentId: source.agentId,
+ eventType: "tool.execution.started", stream: "system",
+ payload: { name: "send_email", executionId: "unconfirmed-write", transport: "process" },
+ });
+ const retire = vi.fn(() => true);
+ const verifyStoppedSession = vi.fn(async (run: typeof heartbeatRuns.$inferSelect) =>
+ run.id === source.runId && mode !== "unproven" ? { evidence: {}, retire } : null);
+ // Repeated sweeps must not create duplicate successors or user messages.
+ await reconcileSafeNativeReplacements(db, new Date(), { verifyStoppedSession });
+ await reconcileSafeNativeReplacements(db, new Date(), { verifyStoppedSession });
+ const successors = await db.select().from(heartbeatRuns).where(eq(heartbeatRuns.retryOfRunId, source.runId));
+ expect(successors).toHaveLength(mode === "verified" ? 1 : 0);
+ expect(await readFile(file, "utf8")).toBe(saved);
+ const comments = await db.select().from(issueComments).where(eq(issueComments.issueId, source.issueId));
+ expect(comments.filter((row) => row.id === comment.id)).toMatchObject([{ body }]);
+ expect(comments.filter((row) => row.body === body)).toHaveLength(1);
+ const [task] = await db.select().from(issues).where(eq(issues.id, source.issueId));
+ if (mode === "verified") {
+ expect(retire).toHaveBeenCalledOnce();
+ expect(task!.status).toBe("in_progress");
+ const continuation = await buildExecutionContinuation({
+ db, companyId: source.companyId, issueId: source.issueId, agentId: source.agentId,
+ context: successors[0]!.contextSnapshot!, summary: null, exposeLowTrustRaw: false,
+ });
+ expect(continuation.messages.filter((message) => message.id === comment.id)).toHaveLength(1);
+ expect(continuation.objective).toBe(body);
+ } else {
+ expect(retire).not.toHaveBeenCalled();
+ expect(task!.status).toBe("blocked");
+ if (mode === "unknown_action") {
+ const [decision] = await db.select().from(nativeRunFinalizations).where(eq(nativeRunFinalizations.runId, source.runId));
+ expect(decision!.failureDetail?.replacementDenied).toBe("uncertain_provider_action");
+ }
+ }
+ } finally {
+ await rm(workspace, { recursive: true, force: true });
+ }
+ },
+ );
it.each(["unproven", "changed", "verified"] as const)("requires stopped-session proof through commit (%s)", async (mode) => {
const source = await seed(2);
await db.update(nativeRunFinalizations).set({ failureCode: "native_session_cleanup_quarantined" }).where(eq(nativeRunFinalizations.runId, source.runId));
diff --git a/server/src/services/recovery/service.ts b/server/src/services/recovery/service.ts
index f799023e2e..0929a346d1 100644
--- a/server/src/services/recovery/service.ts
+++ b/server/src/services/recovery/service.ts
@@ -2829,7 +2829,8 @@ export function recoveryService(
);
}
- async function healthyOpenChildIssues(issue: typeof issues.$inferSelect) {
+ async function healthyOpenChildIssues(issue: typeof issues.$inferSelect, sameWorkspaceOnly = false) {
+ if (sameWorkspaceOnly && !issue.projectWorkspaceId) return [];
const childCandidates = await db
.select()
.from(issues)
@@ -2837,6 +2838,7 @@ export function recoveryService(
and(
eq(issues.companyId, issue.companyId),
eq(issues.parentId, issue.id),
+ ...(sameWorkspaceOnly ? [eq(issues.projectWorkspaceId, issue.projectWorkspaceId!)] : []),
visibleIssueCondition(),
notInArray(issues.status, ["done", "cancelled"]),
),
@@ -2848,7 +2850,7 @@ export function recoveryService(
});
if (
childState.hasActiveExecutionPath ||
- childState.hasDurableWaitingPath
+ (!sameWorkspaceOnly && childState.hasDurableWaitingPath)
) {
openChildren.push({ id: child.id, identifier: child.identifier });
}
@@ -4996,6 +4998,17 @@ export function recoveryService(
if (isSuccessfulInProgressContinuationRun(latestRun)) {
const successfulRun = latestRun;
+ // A child with a live or durable waiting path must get a chance to use
+ // the shared workspace. Repeated automatic parent continuations can
+ // otherwise reacquire it before the child's resource retry is due.
+ // This only gates recovery; explicit messages still follow admission.
+ const workspace = parseObject(parseObject(successfulRun.contextSnapshot).paperclipWorkspace);
+ if (workspace.mode === "shared_workspace" && (await healthyOpenChildIssues(issue, true)).length > 0) {
+ result.productiveContinuationObserved += 1;
+ result.skipped += 1;
+ continue;
+ }
+
if (!isProductiveContinuationRun(successfulRun)) {
result.successfulContinuationObserved += 1;
result.skipped += 1;
diff --git a/tests/e2e/chat-adapters-ui.spec.ts b/tests/e2e/chat-adapters-ui.spec.ts
index b12dcd281f..e98d816d97 100644
--- a/tests/e2e/chat-adapters-ui.spec.ts
+++ b/tests/e2e/chat-adapters-ui.spec.ts
@@ -2431,12 +2431,25 @@ test.describe("Board send delivery refresh", () => {
);
const sends: Record[] = [];
let attachmentId = "";
+ let privateComment: { id: string } | undefined;
await page.route(
`**/api/chat-endpoints/${endpointId}/conversations/${conversationId}/publications`,
async (route) => {
const input = bodyOf(route);
sends.push(input);
- if (sends.length === 1) return route.abort("connectionreset");
+ if (sends.length === 1) {
+ expect(input.attachmentIds).toEqual([attachmentId]);
+ // Consume the file only after the browser has captured its send.
+ // Binding it before Send lets live refresh remove it from the draft,
+ // making the mocked rejection inconsistent with the actual request.
+ privateComment = await json<{ id: string }>(
+ await request.post(`/api/issues/${issue.id}/comments`, {
+ data: { body: "Private Board file", attachmentIds: [attachmentId] },
+ }),
+ "bind selected file to private comment",
+ );
+ return route.abort("connectionreset");
+ }
if (sends.length === 2)
return fulfill(
route,
@@ -2483,15 +2496,8 @@ test.describe("Board send delivery refresh", () => {
"read selected file",
);
attachmentId = file!.id;
- // A concurrent ordinary Board comment consumes the file. The browser
- // fixture emulates the precise durable rejection; real TX/idempotency
- // and delete/retry behavior are covered in the PostgreSQL integration test.
- const privateComment = await json<{ id: string }>(
- await request.post(`/api/issues/${issue.id}/comments`, {
- data: { body: "Private Board file", attachmentIds: [attachmentId] },
- }),
- "bind selected file to private comment",
- );
+ // The first intercepted send binds the file to a concurrent Board comment.
+ // The real TX/idempotency and delete/retry behavior are covered in PostgreSQL.
const draft =
"Share the verified result, without the private Board comment.";
await banner.getByRole("textbox", { name: "Board update" }).fill(draft);
@@ -2551,7 +2557,7 @@ test.describe("Board send delivery refresh", () => {
expect(attachments).toEqual([
expect.objectContaining({
id: attachmentId,
- issueCommentId: privateComment.id,
+ issueCommentId: privateComment!.id,
}),
]);
});
diff --git a/tests/fixtures/connection-review-provider.ts b/tests/fixtures/connection-review-provider.ts
index 04f11d727a..9ad50744e7 100644
--- a/tests/fixtures/connection-review-provider.ts
+++ b/tests/fixtures/connection-review-provider.ts
@@ -3,8 +3,9 @@ import { listenOnFetchAllowedPort } from "../e2e/fetch-allowed-port.js";
export async function startReviewProvider(
successText = "Pages: Roadmap, Meeting notes",
+ bearerToken?: string,
) {
- const captures: Array<{ method: string; toolName: string | null }> = [];
+ const captures: Array<{ method: string; toolName: string | null; authorized?: boolean }> = [];
const server: Server = createServer(async (req, res) => {
const chunks: Buffer[] = [];
for await (const chunk of req) chunks.push(chunk as Buffer);
@@ -15,10 +16,17 @@ export async function startReviewProvider(
method?: string;
params?: { name?: string; arguments?: { query?: string } };
};
+ const authorized = !bearerToken || req.headers.authorization === `Bearer ${bearerToken}`;
captures.push({
method: String(payload.method ?? ""),
toolName: payload.params?.name ?? null,
+ ...(bearerToken ? { authorized } : {}),
});
+ if (!authorized) {
+ res.writeHead(401, { "Content-Type": "application/json" });
+ res.end(JSON.stringify({ error: "Service credential required" }));
+ return;
+ }
res.writeHead(200, { "Content-Type": "application/json" });
if (payload.method === "tools/list") {
res.end(
diff --git a/tests/runner-e2e/EVERYDAY-WORKFLOWS.md b/tests/runner-e2e/EVERYDAY-WORKFLOWS.md
new file mode 100644
index 0000000000..98abb3d5b2
--- /dev/null
+++ b/tests/runner-e2e/EVERYDAY-WORKFLOWS.md
@@ -0,0 +1,186 @@
+# Everyday Paperclip workflow evals
+
+This manual suite tests useful work through the production browser, public API,
+native runner, and normal agent instructions. It complements the tightly
+scripted runner contract fixtures. It does not add a scheduled or default paid
+run: `--all` and generic profile selectors exclude it. Select the suite or an
+exact execution ID explicitly.
+
+## Stories and assertions
+
+| Story | Cases | Required evidence |
+|---|---|---|
+| Build a small project and revise it | `build-revise` | Download both ZIPs through the UI; independently execute the delivered CLI and import its function; test the revision; retrieve the original bytes again. |
+| Delegate and incorporate late feedback | `delegate-feedback` | One child assigned to Riley; send feedback while the child runs; find it in the child history and independently test `--max-length` in the delivered ZIP. The worker must not execute on the parent. |
+| Hire a teammate and use them again | `hire-reuse` | One Morgan QA reporting to the lead, native runner and the same encrypted connection bindings, real child execution, then a second usable delivery from that same agent. |
+| Decide on an installed service action | `service-approve`, `service-decline` | Assign an authenticated local MCP fixture with Ask first; match its tool action and connection ID; no provider call before approval; exactly one after approval and a verified document; none after decline. |
+| Decline a new connection | `connection-decline` | Start without service connections; match a Notion connection intent; click Not now; verify the saved rejection, no new connection or repeated request, and an explanation followed by Done. |
+| Continue work after a controller restart | `recover-controller` | Observe saved source, persist a user message, restart the isolated controller, and independently test the delivered result. |
+| Stop work and change direction | `stop-redirect` | Click Stop, send one new request, reload, observe exactly one stored user message and the new answer, and reach Done. |
+
+For normal completion, all story tasks must reach Done, with no active run,
+pending completion confirmation, or scheduled recovery. Runs must prove native
+identity and native terminal contracts. A workspace-contention cancellation is
+not provider execution only when the persisted pre-dispatch record explicitly
+says `providerWorkStarted: false` and no process/session/runner identity exists.
+Other unexplained cancellations remain failures. Twelve total run records bound
+each story, including contention and recovery.
+
+Arbitrary runner-process termination is not part of the model scorecard. The
+historical `recover-runner`, `recover-runner-safe`, and `recover-runner-uncertain`
+attempts remain available as diagnostics, with their original grades and costs.
+The first two did not establish a safe restart boundary, and the uncertainty
+case measures a deterministic safety rule. None supports ranking models.
+See [controlled recovery tests](../runner-recovery/README.md).
+
+## Matrix and running
+
+The local matrix has eight cases on native Codex `gpt-5.6-sol`, native ACPX Claude
+`claude-sonnet-5`, and native Codex `gpt-5.4-mini`: 24 cells. The two core profiles
+also declare build/revise, delegation, and controller-restart cases on Daytona:
+six cells. Remote runner-process killing is not supported. For remote controller
+restart, a verified first download supplies the persistence checkpoint; the
+controller is interrupted during a subsequent revision with another queued
+requirement.
+
+```sh
+pnpm test:e2e:runner -- --list --suite everyday-workflows
+pnpm test:e2e:runner -- --suite everyday-workflows --environment local --max-parallel 2
+pnpm test:e2e:runner -- --id everyday-workflows.runner-codex-mini.local.build-revise
+pnpm test:e2e:runner -- --suite everyday-workflows --environment daytona --max-parallel 2
+```
+
+Before project stories or the Python calibration tests, start Docker on the
+harness host and fetch the pinned oracle image. This is required for local and
+Daytona stories; artifact checks run on the harness host.
+
+```sh
+docker pull python@sha256:9d2e5553305c7c7b0097999bb17187c69b921ccd6bc9d40e4bb5ebe652c00285
+python3 tests/runner-e2e/everyday-artifact.py --preflight
+```
+
+The harness checks this prerequisite before it creates the task. It does not
+pull an image during a model attempt or fall back to host execution.
+
+Use the credential and immutable Daytona image setup in [README.md](README.md).
+Provider calls cost money. Each cell owns an isolated instance and project.
+There are no real third-party mutations in the service fixture; it exercises
+production connection, transport, tool approval, and document delivery paths.
+
+## Deterministic checks and calibration
+
+```sh
+pnpm test:e2e:runner:typecheck
+pnpm test:e2e:runner:unit
+python3 -m unittest discover -s tests/runner-e2e -p test_everyday_artifact.py
+```
+
+The independent oracle rejects wrong output, ignored late feedback, trailing
+separator bugs, invalid argument acceptance, duplicate source modules, archive
+path traversal, and symlinks. Passing agent-authored tests cannot override it.
+Lifecycle calibration rejects legacy execution, missing runner identity,
+unexpected crashes, workers on the parent, and answers left in review.
+
+ZIP evaluation runs delivered Python in a Docker container with a read-only
+project mount and root filesystem, no network, a non-root user, no Linux
+capabilities, and bounded CPU, memory, process count, output, and duration. Only
+the extracted delivery enters the container. The container is removed after
+grading. Calibration includes attempts to read a host file and reach a host
+loopback service.
+
+## Evalbook evidence and qualification
+
+Each packaged attempt retains `snapshots/everyday-workflow.json`, downloaded
+ZIPs, assertions, actual task comments and run records, timing, accounting,
+source provenance, and screenshots. The story records a digest of its harness
+sources. Infrastructure failures and failed attempts must remain inspectable.
+
+Import packaged results with `paperclip-evals/evals/everyday-workflows/import_results.py`.
+It uses the canonical Runner Evalbook generator and the built Runner Lab viewer.
+It does not invent provider transcripts, tool counts, model observations, or
+cost estimates. The selected model is checked against persisted native execution
+inputs; that is distinct from provider-side model identity verification.
+
+Initial live results are diagnostic. They are not a reliability estimate or a
+model ranking. Before promotion, freeze both source revisions and harness
+digest, run at least three independent local repetitions, qualify the six
+remote cells against a verified image, and review every failure. Keep model
+quality, lifecycle correctness, infrastructure availability, and latency separate.
+
+## Revised evaluation contract (14 September, second campaign)
+
+That campaign used 32 cells: the original local stories plus two local Codex
+text-only safe-replacement probes, and the unchanged six remote cells. The old
+`recover-runner` results remain historical; `recover-runner-uncertain` is a new
+case that expects a visible Blocked safety stop, preserved source and queued
+input, and no unverified provider replay. Its Retry control is inspected, not
+claimed to restore work. Successful manual recovery remains unqualified.
+
+`recover-runner-safe` interrupts a text-only Codex turn and queues new direction.
+A pass requires the server's durable `verified_safe_replacement` evidence and the
+new answer. No safety proof is injected or fabricated. If that premise cannot be
+verified in a live probe, report it as an unqualified recovery boundary, not an
+established product defect. Claude has no catalog cell for this Codex-specific
+replacement proof. Deterministic native-safe-replacement tests cover its proof
+and admission gates independently of model behavior.
+
+Delegation now submits feedback through the existing child task composer and
+records the delivered comment ID. The child must consume the message and deliver
+the revised program. This does not require a lead to relay a parent comment.
+The separate issue-update-comment-wakeup route tests exercise exact supported
+mention routing, including access, dependency, identity, and duplicate-wake gates.
+
+The approval case provisions an authenticated local service through the public
+API, with a random server-held credential that never enters the agent environment
+or browser trace. Approval/decline interactions still use the browser. Provider
+captures distinguish rejected unauthenticated requests from accepted calls. The
+old public-endpoint attempts remain boundary evidence, not an isolation promise.
+
+Stop now waits for the owned runner to exit, records project file hashes, and
+checks them again after the new response. This proves stability over that interval,
+not indefinite monitoring. Hiring and declined-access policy changes are deferred
+by user decision; their old results must not be presented as new campaign runs.
+
+## Decline correction (14 September, third campaign)
+
+That campaign used 35 cells (29 local, six remote). `service-decline` tests
+rejection of a protected action on an already installed service; its former
+"connection request" title was misleading. `connection-decline` separately tests
+Not now on new Notion setup. Both permit a brief explanation as the complete
+fallback, so Done is expected after that explanation. Neither test requires
+completion after refusing work that is still required.
+
+The installed-service decline fixture now uses the same server-held credential
+as approval. The harness requires one pending interaction, validates its kind
+and connection/provider identity before clicking, and waits for the exact
+interaction's saved decision. Wrong interactions fail `decision-request-matches-story`
+with a screenshot; they are not evidence of an ignored decline. Both decline
+stories check a new explanation after the decision and reject repeated requests.
+
+Historical attempts remain unchanged. This campaign resumes the previously
+deferred decline cases; hiring remains deferred. Notion setup is declined in
+the UI, so this test neither authenticates to nor reads real Notion data.
+
+Decision screenshots are included in the evidence package. Before capturing the
+final screen, the harness waits for the thread and latest persisted agent comment
+to render, then scrolls that comment into view. A Done header alone is not proof
+that the final response was visible.
+
+
+## Recovery scope correction (14 September)
+
+The current catalog has **30 cells: 24 local and six remote**. Forced runner
+crash probes are retired from paid selection. Their original attempt IDs remain
+in Evalbook's Diagnostics history and Latest pages; they are excluded from the
+main matrix without changing grades or deleting evidence. Reported spend still
+includes all attempts.
+
+`recover-controller` and `stop-redirect` retain concrete supported journeys:
+restart the controller while preserving the runner, or use Stop and submit a new
+direction. Their assertions verify pending input, saved work, and the next
+usable result. Neither claims recovery from an arbitrary provider-process crash.
+
+A future user-facing crash-recovery case needs a reproducible recoverable fault,
+an identified supported recovery action, and evidence through the final usable
+result. A missing test premise must be reported as unexercised, not a model
+failure. Do not introduce a new paid case just to replace a retired row.
diff --git a/tests/runner-e2e/README.md b/tests/runner-e2e/README.md
index c3a63a27eb..912497d27d 100644
--- a/tests/runner-e2e/README.md
+++ b/tests/runner-e2e/README.md
@@ -479,3 +479,33 @@ See [FIXTURES.md](./FIXTURES.md) before adding or changing a profile,
environment, task, matcher, or future Paperclip object fixture.
See [SECURITY.md](./SECURITY.md) before enabling paid dispatch, the runner
group, or permanent public history in this public repository.
+
+## Everyday user-story evals
+
+See [EVERYDAY-WORKFLOWS.md](EVERYDAY-WORKFLOWS.md) for the explicit-only native-runner stories and their canonical Evalbook importer. These cells do not expand scheduled `--all` runs.
+
+### Everyday hiring prerequisites and timeout evidence
+
+The manual `everyday-workflows` / `hire-reuse` story enables native API tools in
+its isolated harness and creates a personal managed AI account through the public
+API. The lead uses the responsible user's default account, without adapter env
+credential overrides. The hire must inherit that binding and finish a real run
+attributed to the same account. The evidence records this fixture configuration.
+Other suites retain their existing API-tool defaults.
+
+A polling deadline after successful state reads is a candidate workflow failure,
+not a reason to retry as infrastructure. State snapshots remain in the evidence;
+the timeout message does not serialize task data into the failure classifier.
+Explicit server-health waits and failed network reads retain infrastructure
+classification.
+
+The product execution prompt v3 tells agents to record child dependencies and
+end the parent turn when no independent work remains. The user-story prompts
+stay unchanged, so live retests measure the product guidance itself.
+
+Delegated ZIP delivery can appear on the user-facing parent or its child task.
+The grader selects the newest ZIP only within that task family; a reuse request
+requires a new attachment after the request. The hired-agent execution/account
+checks and independent downloaded-code checks remain mandatory.
+
+Revision delivery checks exclude preserved originals by their content hash, even when the agent republishes an original after the revised ZIP. The browser downloads the exact selected attachment ID; its bytes still pass through the independent artifact checker.
diff --git a/tests/runner-e2e/api.ts b/tests/runner-e2e/api.ts
index 45863285f1..77edda9b14 100644
--- a/tests/runner-e2e/api.ts
+++ b/tests/runner-e2e/api.ts
@@ -67,6 +67,13 @@ export class RunnerApi {
}
}
+export class ObservedStateTimeout extends Error {
+ constructor(label: string, readonly failureClass: "candidate_failure" | "transient_infrastructure") {
+ super(`Timed out waiting for ${label}; the observed state did not satisfy the condition. See the saved state evidence.`);
+ this.name = "ObservedStateTimeout";
+ }
+}
+
export async function pollUntil(input: {
label: string;
deadlineAt: number;
@@ -74,6 +81,7 @@ export async function pollUntil(input: {
accept: (value: T) => boolean;
reject?: (value: T) => string | undefined;
intervalMs?: number;
+ timeoutFailureClass?: "candidate_failure" | "transient_infrastructure";
}): Promise {
let last: T | undefined;
let lastError: unknown;
@@ -99,11 +107,8 @@ export async function pollUntil(input: {
setTimeout(resolve, input.intervalMs ?? 2_000),
);
}
- const detail =
- lastError instanceof Error
- ? lastError.message
- : last === undefined
- ? "no observation"
- : JSON.stringify(last);
- throw new Error(`Timed out waiting for ${input.label}: ${detail}`);
+ if (lastError instanceof Error) {
+ throw new Error(`Timed out waiting for ${input.label}: ${lastError.message}`, { cause: lastError });
+ }
+ throw new ObservedStateTimeout(input.label, input.timeoutFailureClass ?? "candidate_failure");
}
diff --git a/tests/runner-e2e/catalog.test.ts b/tests/runner-e2e/catalog.test.ts
index d3ae9025ec..0f8c023418 100644
--- a/tests/runner-e2e/catalog.test.ts
+++ b/tests/runner-e2e/catalog.test.ts
@@ -41,10 +41,10 @@ describe("runner E2E catalog", () => {
expect(localIntegrityTasks).toHaveLength(2);
expect(openRouterBreadthTasks).toHaveLength(3);
expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([
- 24, 42, 14, 10, 2,
+ 30, 24, 42, 14, 10, 2,
]);
- expect(validateRunnerCatalog()).toHaveLength(92);
- expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(92);
+ expect(validateRunnerCatalog()).toHaveLength(122);
+ expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(122);
expect(
runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"),
).toHaveLength(42);
@@ -64,7 +64,7 @@ describe("runner E2E catalog", () => {
),
).toHaveLength(2);
expect(
- runnerMatrix.reduce(
+ runnerMatrix.filter(entry => !entry.suite.manualOnly).reduce(
(total, execution) => total + execution.task.expectedRunCount,
0,
),
diff --git a/tests/runner-e2e/catalog.ts b/tests/runner-e2e/catalog.ts
index 57d95b76c4..2ffa5a519e 100644
--- a/tests/runner-e2e/catalog.ts
+++ b/tests/runner-e2e/catalog.ts
@@ -1,3 +1,4 @@
+import { everydayTasks, productionStoryProfile } from "./everyday-cases.js";
import { chatTasks } from "./chat-cases.js";
import { createHash } from "node:crypto";
import { createAgentSchema } from "../../packages/shared/src/validators/agent.js";
@@ -882,7 +883,22 @@ export const connectionReviewSuite: RunnerSuiteFixture = {
})),
};
+const everydayProfiles = [
+ ...runnerProfiles.filter(profile => ["runner-codex", "runner-acpx-claude"].includes(profile.id)),
+ nativeProfile({ id: "runner-codex-mini", label: "Runner Codex Mini", provider: "codex", model: "gpt-5.4-mini", modelQualification: {source:"qualified_runner_profile",qualificationId:"everyday-codex-mini-pilot"}, credential: "OPENAI_API_KEY", supportedEnvironments: ["local"] }),
+].map(productionStoryProfile);
+
export const runnerSuites: readonly RunnerSuiteFixture[] = [
+ {
+ id: "everyday-workflows", label: "Everyday Paperclip Work", manualOnly: true,
+ description: "Real user requests, useful downloaded work, and durable continuation using production instructions.",
+ groups: ["native"], profiles: everydayProfiles, environments: [localEnvironment, daytonaWarmEnvironment],
+ tasks: everydayTasks, expectedMatrixSize: 30,
+ excludedExecutionIds: [...everydayProfiles.flatMap(profile => everydayTasks
+ .filter(task => !["build-revise", "delegate-feedback", "recover-controller"].includes(task.id))
+ .map(task => `everyday-workflows.${profile.id}.daytona.${task.id}`))],
+ definitionMetadata: { version: 3, instructions: "production", grading: "outcome-and-invariants", scheduling: "explicit-only" },
+ },
{
id: "agent-chat", label: "Persistent Agent Chat",
description: "Task-backed conversations, session resets, and project plan handoff.",
@@ -1050,8 +1066,9 @@ function assertNoRawSecretValues(value: unknown, label: string) {
}
export function validateRunnerCatalog(): MatrixExecution[] {
- const allProfiles = [...runnerProfiles, ...openRouterBreadthProfiles];
+ const allProfiles = [...runnerProfiles, ...openRouterBreadthProfiles, ...everydayProfiles.filter(p => !runnerProfiles.some(existing => existing.id === p.id))];
const allTasks = [
+ ...everydayTasks,
...runnerTasks,
...localIntegrityTasks,
...openRouterBreadthTasks,
@@ -1114,7 +1131,7 @@ export function validateRunnerCatalog(): MatrixExecution[] {
createEnvironmentSchema.parse(payload);
assertNoRawSecretValues(payload, `environment ${environment.id}`);
}
- for (const profile of allProfiles) {
+ for (const profile of [...allProfiles, ...everydayProfiles]) {
if (!CREDENTIAL_NAMES.includes(profile.credential)) {
throw new Error(
`Profile ${profile.id} declares unknown credential ${profile.credential}`,
diff --git a/tests/runner-e2e/connection-auth.test.ts b/tests/runner-e2e/connection-auth.test.ts
new file mode 100644
index 0000000000..4d2f2414b3
--- /dev/null
+++ b/tests/runner-e2e/connection-auth.test.ts
@@ -0,0 +1,35 @@
+import { describe, expect, it } from "vitest";
+import { startReviewProvider } from "../fixtures/connection-review-provider.js";
+
+describe("protected workflow service fixture", () => {
+ it("does not return data to a direct caller without the service credential", async () => {
+ const provider = await startReviewProvider(
+ "private fixture pages",
+ "test-only-service-key",
+ );
+ const body = JSON.stringify({
+ jsonrpc: "2.0",
+ id: 1,
+ method: "tools/call",
+ params: { name: "notion:list_pages" },
+ });
+ try {
+ const denied = await fetch(provider.url, { method: "POST", body });
+ expect(denied.status).toBe(401);
+ expect(await denied.text()).not.toContain("private fixture pages");
+ const accepted = await fetch(provider.url, {
+ method: "POST",
+ body,
+ headers: { authorization: "Bearer test-only-service-key" },
+ });
+ expect(accepted.status).toBe(200);
+ expect(await accepted.text()).toContain("private fixture pages");
+ expect(provider.captures.map((call) => call.authorized)).toEqual([
+ false,
+ true,
+ ]);
+ } finally {
+ await provider.close();
+ }
+ });
+});
diff --git a/tests/runner-e2e/connection-reviews.ts b/tests/runner-e2e/connection-reviews.ts
index b07f62908a..9808b3cd36 100644
--- a/tests/runner-e2e/connection-reviews.ts
+++ b/tests/runner-e2e/connection-reviews.ts
@@ -1,3 +1,4 @@
+import { randomBytes } from "node:crypto";
import { expect, type Page } from "@playwright/test";
import type { RunnerApi } from "./api.js";
import { startReviewProvider } from "../fixtures/connection-review-provider.js";
@@ -10,35 +11,69 @@ export async function setupConnectionReview(input: {
companyId: string;
agentId: string;
marker: string;
+ authenticated?: boolean;
}) {
- const provider = await startReviewProvider(input.marker);
+ const credential = input.authenticated
+ ? randomBytes(32).toString("hex")
+ : undefined;
+ const provider = await startReviewProvider(input.marker, credential);
try {
const { page, api } = input;
- await page.goto(`/${input.prefix}/apps`);
- const connector = page
- .getByRole("list", { name: "Connector list" })
- .getByRole("listitem")
- .filter({ hasText: "Connect your own tool" });
- await connector
- .getByRole("button", { name: "Connect", exact: true })
- .click();
- await connector
- .getByRole("button", { name: "Connect your own MCP server" })
- .click();
- await page
- .getByPlaceholder("https://example.com/actions")
- .fill(provider.url);
- await page.getByRole("button", { name: "Continue", exact: true }).click();
- await page.getByRole("button", { name: "Save and continue" }).click();
- await page.getByRole("button", { name: /Check link/i }).click();
- await expect(
- page.getByRole("heading", { name: /is ready/i }),
- ).toBeVisible();
- const {
- connections: [connection],
- } = await api.get<{ connections: Array<{ id: string }> }>(
- `/api/companies/${input.companyId}/tools/connections`,
- );
+ let connection: { id: string };
+ if (credential) {
+ // Setup through the public API avoids putting this fixture credential in
+ // browser traces or the agent's environment. Approval remains a UI action.
+ const connected = await api.postSensitive(
+ `/api/companies/${input.companyId}/tools/apps/connect`,
+ {
+ link: provider.url,
+ name: "Studio Page Service",
+ authMode: "bearer",
+ credentialValues: { "credentials.authorization": credential },
+ grantKind: "organization",
+ },
+ );
+ const ids = [
+ ...connected.actions.readOnly,
+ ...connected.actions.canMakeChanges,
+ ].map((a: any) => a.catalogEntryId);
+ await api.post(
+ `/api/companies/${input.companyId}/tools/apps/${connected.connectionId}/finish`,
+ {
+ enabledCatalogEntryIds: ids,
+ askFirstCatalogEntryIds: ids,
+ access: { agentIds: [input.agentId] },
+ },
+ );
+ connection = { id: connected.connectionId };
+ } else {
+ await page.goto(`/${input.prefix}/apps`);
+ const connector = page
+ .getByRole("list", { name: "Connector list" })
+ .getByRole("listitem")
+ .filter({ hasText: "Connect your own tool" });
+ await connector
+ .getByRole("button", { name: "Connect", exact: true })
+ .click();
+ await connector
+ .getByRole("button", { name: "Connect your own MCP server" })
+ .click();
+ await page
+ .getByPlaceholder("https://example.com/actions")
+ .fill(provider.url);
+ await page.getByRole("button", { name: "Continue", exact: true }).click();
+ await page.getByRole("button", { name: "Save and continue" }).click();
+ await page.getByRole("button", { name: /Check link/i }).click();
+ await expect(
+ page.getByRole("heading", { name: /is ready/i }),
+ ).toBeVisible();
+ const {
+ connections: [createdConnection],
+ } = await api.get<{ connections: Array<{ id: string }> }>(
+ `/api/companies/${input.companyId}/tools/connections`,
+ );
+ connection = createdConnection!;
+ }
const installed = await api.request.put(
`/api/tool-connections/${connection.id}/installs`,
{
@@ -54,7 +89,9 @@ export async function setupConnectionReview(input: {
...provider,
connectionId: connection.id,
invocationCount: () =>
- provider.captures.filter((call) => call.method === "tools/call").length,
+ provider.captures.filter(
+ (call) => call.method === "tools/call" && call.authorized !== false,
+ ).length,
};
} catch (error) {
await provider.close();
diff --git a/tests/runner-e2e/everyday-artifact.py b/tests/runner-e2e/everyday-artifact.py
new file mode 100644
index 0000000000..fdda71647c
--- /dev/null
+++ b/tests/runner-e2e/everyday-artifact.py
@@ -0,0 +1,134 @@
+"""Independent acceptance oracle. Never runs the project's own test assertions."""
+import argparse, json, os, pathlib, selectors, stat, subprocess, sys, tempfile, time, uuid, zipfile
+
+BASE_CASES = [("Hello World", "hello-world"), (" Queue--Ready!! ", "queue-ready"),
+ ("Already-Fine", "already-fine"), ("Café 東京", "caf"), ("!!!", ""),
+ ("a__b c", "a-b-c"), ("123", "123"), ("", "")]
+
+# Published multi-platform python:3.13-slim digest. Never pull implicitly during a model run.
+SANDBOX_IMAGE = 'python@sha256:9d2e5553305c7c7b0097999bb17187c69b921ccd6bc9d40e4bb5ebe652c00285'
+
+def preflight():
+ for args in [['docker', 'version', '--format', '{{.Server.Version}}'],
+ ['docker', 'image', 'inspect', SANDBOX_IMAGE]]:
+ result = subprocess.run(args, capture_output=True, text=True, timeout=10)
+ if result.returncode:
+ raise RuntimeError(f'Artifact sandbox qualification failed. Start Docker and run: docker pull {SANDBOX_IMAGE}')
+
+class ArtifactSandbox:
+ """Only extracted delivery files enter the container. No host secrets or network."""
+ def __init__(self, root, source):
+ self.root = root
+ self.cwd = '/project/' + str(source.parent.relative_to(root))
+ self.source = '/project/' + str(source.relative_to(root))
+ self.name = 'paperclip-artifact-oracle-' + uuid.uuid4().hex
+
+ def __enter__(self):
+ preflight()
+ self.root.chmod(0o755)
+ args = ['docker', 'run', '--detach', '--rm', '--pull=never', '--name', self.name,
+ '--network=none', '--read-only', '--cap-drop=ALL',
+ '--security-opt=no-new-privileges', '--pids-limit=64', '--memory=128m',
+ '--memory-swap=128m', '--cpus=1', '--user=65534:65534',
+ '--log-driver=none', '--tmpfs=/tmp:rw,noexec,nosuid,size=16m',
+ '--mount', f'type=bind,source={self.root},target=/project,readonly',
+ '--workdir', self.cwd, SANDBOX_IMAGE, 'python', '-I', '-c',
+ 'import time; time.sleep(180)']
+ try:
+ result = subprocess.run(args, capture_output=True, text=True, timeout=15)
+ if result.returncode:
+ raise RuntimeError('Artifact sandbox could not start: ' + result.stderr[:500])
+ except BaseException:
+ self.close()
+ raise
+ return self
+
+ def close(self):
+ subprocess.run(['docker', 'rm', '--force', self.name], capture_output=True, timeout=10)
+
+ def __exit__(self, *unused):
+ self.close()
+
+ def invoke(self, args, *, imported=False):
+ code = ['-c', "import sys; sys.path.insert(0, " + repr(self.cwd) + "); from slugify import slugify; assert slugify(' A B! ') == 'a-b'"] if imported else [self.source, *args]
+ command = ['docker', 'exec', '--workdir', self.cwd, self.name, 'python', '-I', '-B', *code]
+ # Bound output as it arrives. A generated program must not exhaust host memory or disk.
+ proc = subprocess.Popen(command, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
+ output = {'stdout': bytearray(), 'stderr': bytearray()}
+ deadline = time.monotonic() + 10
+ try:
+ with selectors.DefaultSelector() as selector:
+ selector.register(proc.stdout, selectors.EVENT_READ, 'stdout')
+ selector.register(proc.stderr, selectors.EVENT_READ, 'stderr')
+ while selector.get_map():
+ remaining = deadline - time.monotonic()
+ if remaining <= 0:
+ raise ValueError('Artifact command exceeded the time limit')
+ for key, _ in selector.select(remaining):
+ chunk = os.read(key.fd, 4096)
+ if not chunk:
+ selector.unregister(key.fileobj)
+ continue
+ output[key.data].extend(chunk)
+ if sum(map(len, output.values())) > 65_536:
+ raise ValueError('Artifact command exceeded the output limit')
+ code = proc.wait(timeout=max(0.01, deadline-time.monotonic()))
+ return subprocess.CompletedProcess(command, code,
+ output['stdout'].decode('utf-8', errors='replace'),
+ output['stderr'].decode('utf-8', errors='replace'))
+ finally:
+ if proc.poll() is None:
+ proc.kill()
+ proc.wait()
+ proc.stdout.close()
+ proc.stderr.close()
+
+def inspect(archive, mode):
+ checks = []
+ def check(name, passed, detail=""):
+ checks.append(dict(id=name, passed=bool(passed), detail=detail))
+ with tempfile.TemporaryDirectory(prefix="paperclip-artifact-oracle-") as tmp:
+ root=pathlib.Path(tmp)
+ with zipfile.ZipFile(archive) as z:
+ entries=z.infolist()
+ if len(entries)>250 or sum(e.file_size for e in entries)>10_000_000:
+ raise ValueError("Archive exceeds the bounded source project size")
+ for entry in entries:
+ p=pathlib.PurePosixPath(entry.filename)
+ if p.is_absolute() or ".." in p.parts or "\\" in entry.filename or stat.S_ISLNK(entry.external_attr >> 16):
+ raise ValueError("Unsafe archive member")
+ z.extractall(root)
+ sources=list(root.rglob("slugify.py"))
+ check("one-slugify-source",len(sources)==1)
+ check("readme-present",any(p.name.lower().startswith('readme') for p in root.rglob('*')))
+ check("project-tests-present",any(p.name.startswith('test') and p.suffix=='.py' for p in root.rglob('*')))
+ if len(sources)!=1:return checks
+ source=sources[0]
+ with ArtifactSandbox(root, source) as sandbox:
+ invoke = sandbox.invoke
+ for index,(text,expected) in enumerate(BASE_CASES):
+ result=invoke([text]);check(f"base-{index}",result.returncode==0 and result.stdout.rstrip('\r\n')==expected,
+ f"exit={result.returncode}; expected={expected!r}; observed={result.stdout[:160]!r}")
+ # Import from the delivered module, not an evaluator reimplementation.
+ imported=invoke([], imported=True)
+ check('importable-function',imported.returncode==0)
+ if mode=='separator':
+ for separator,expected in [('_','queue_ready'),('-','queue-ready')]:
+ result=invoke([' Queue--Ready!! ','--separator',separator]);check('separator-'+separator,result.returncode==0 and result.stdout.strip()==expected)
+ check('reject-invalid-separator',invoke(['hello','--separator','/']).returncode!=0)
+ if mode=='max-length':
+ for size,expected in [('7','queue-r'),('6','queue'),('1','q')]:
+ result=invoke([' Queue--Ready!! ','--max-length',size]);check('length-'+size,result.returncode==0 and result.stdout.strip()==expected)
+ check('reject-zero-length',invoke(['hello','--max-length','0']).returncode!=0)
+ return checks
+
+if __name__=='__main__':
+ parser=argparse.ArgumentParser();parser.add_argument('archive',nargs='?');parser.add_argument('--preflight',action='store_true');parser.add_argument('--mode',choices=['base','separator','max-length'],default='base');args=parser.parse_args()
+ try:
+ if args.preflight:
+ preflight();checks=[dict(id='artifact-sandbox-ready',passed=True,detail='Pinned image available')]
+ elif args.archive:checks=inspect(args.archive,args.mode)
+ else:parser.error('archive is required unless --preflight is set')
+ except Exception as error:checks=[dict(id='artifact-readable',passed=False,detail=str(error))]
+ print(json.dumps(dict(passed=all(c['passed'] for c in checks),checks=checks)))
+ sys.exit(0 if all(c['passed'] for c in checks) else 1)
diff --git a/tests/runner-e2e/everyday-cases.ts b/tests/runner-e2e/everyday-cases.ts
new file mode 100644
index 0000000000..0015907b95
--- /dev/null
+++ b/tests/runner-e2e/everyday-cases.ts
@@ -0,0 +1,100 @@
+import type { RunnerProfileFixture, RunnerTaskFixture } from "./types.js";
+
+/** No fixture completion/API directions: the normal execution prompt owns that contract. */
+export function productionStoryProfile(
+ profile: RunnerProfileFixture,
+): RunnerProfileFixture {
+ return {
+ ...profile,
+ buildAgent(input) {
+ const payload = profile.buildAgent(input);
+ return {
+ ...payload,
+ name: `Studio Lead ${input.executionId}`,
+ role: "ceo",
+ title: "Studio Lead",
+ capabilities:
+ "Builds small projects, delegates implementation, reviews delivered work, and hires teammates when requested.",
+ instructionsBundle: {
+ entryFile: "AGENTS.md",
+ files: {
+ "AGENTS.md":
+ "You lead a small software studio. Help the user build useful small projects. Respect their requirements and verify the delivered work.",
+ },
+ },
+ };
+ },
+ };
+}
+
+export const SLUGIFY_REQUIREMENTS = `Build a small dependency-free Python command-line tool in slugify.py. It accepts one positional string and prints its slug: trim whitespace, lowercase, replace each sequence outside ASCII a-z and 0-9 with one hyphen, then trim hyphens. Expose slugify(text) for reuse. Include a README and automated tests. Deliver the source and tests as a downloadable ZIP. Work in the project workspace.`;
+export const SLUGIFY_REVISION = `Add an optional --separator argument that accepts either - or _. Keep - as the default and preserve the earlier behavior. Deliver an updated ZIP, keeping the previous download available.`;
+export const LATE_REQUIREMENT = `Also support --max-length as a positive integer. Truncate the final slug to that length, then remove any trailing separator. In particular, input " Queue--Ready!! " with --max-length 7 must print "queue-r". Preserve the default behavior.`;
+
+const definitions = [
+ [
+ "build-revise",
+ "Build, download, and revise a project",
+ SLUGIFY_REQUIREMENTS,
+ 2,
+ ],
+ [
+ "delegate-feedback",
+ "Delegate implementation and preserve late feedback",
+ `Have Riley Builder implement the following as one child task. You review the downloaded result when it is ready. Keep implementation with Riley and post a progress note linking the child while work is underway. ${SLUGIFY_REQUIREMENTS}`,
+ 4,
+ ],
+ [
+ "hire-reuse",
+ "Hire one teammate, then reuse that agent",
+ `Hire exactly one agent named Morgan QA, reporting to you, using the same available AI connection and native runner configuration as you. Have Morgan implement the following in one child task, then review the result. ${SLUGIFY_REQUIREMENTS}`,
+ 6,
+ ],
+ [
+ "service-approve",
+ "Use a connection after approval",
+ `Use the connected page service to find recent pages and create a short Markdown briefing document on this task. Include the titles and verification code returned by the service.`,
+ 2,
+ ],
+ [
+ "service-decline",
+ "Respect a declined tool action",
+ `Use the connected page service to find recent pages and create a short Markdown briefing document on this task. Include the titles and verification code returned by the service. If I decline the action, do not try again or request another connection. Instead, give me a brief explanation that you could not retrieve the data. That explanation is the complete allowed fallback; no briefing is required after a decline.`,
+ 2,
+ ],
+ [
+ "connection-decline",
+ "Respect Not now on a new connection",
+ `Please connect Notion so you can read my recent pages and write a short briefing. If I choose Not now, do not try again or use another service. Instead, give me a brief explanation that you could not retrieve the pages. That explanation is the complete allowed fallback; no briefing is required after a decline.`,
+ 2,
+ ],
+ [
+ "recover-controller",
+ "Recover work after the server restarts",
+ SLUGIFY_REQUIREMENTS,
+ 3,
+ ],
+ [
+ "stop-redirect",
+ "Stop a task and send a new direction once",
+ SLUGIFY_REQUIREMENTS,
+ 2,
+ ],
+] as const;
+
+export const everydayTasks: readonly RunnerTaskFixture[] = definitions.map(
+ ([id, label, prompt, expectedRunCount]) => ({
+ id,
+ label,
+ groups: [],
+ workMode: "standard",
+ flow: "everyday_workflow",
+ expectedRunCount,
+ attemptTimeoutMs: { local: 12 * 60_000, daytona: 30 * 60_000 },
+ expectedTerminalState: { issue: "done", run: "succeeded" },
+ buildTitle: (nonce) => `${label} ${nonce}`,
+ buildPrompt: () => prompt,
+ buildVisibleMarker: (nonce) => `STUDIO_${nonce}`,
+ buildMatchers: () => [], // The workflow records independent artifact and lifecycle checks.
+ }),
+);
diff --git a/tests/runner-e2e/everyday-decisions.test.ts b/tests/runner-e2e/everyday-decisions.test.ts
new file mode 100644
index 0000000000..4328ac8da9
--- /dev/null
+++ b/tests/runner-e2e/everyday-decisions.test.ts
@@ -0,0 +1,78 @@
+import { describe, expect, it } from "vitest";
+import {
+ pendingStoryDecision,
+ type StoryInteraction,
+} from "./everyday-decisions.js";
+const tool: StoryInteraction = {
+ id: "tool",
+ kind: "request_confirmation",
+ status: "pending",
+ payload: {
+ toolAction: { connectionId: "installed", actionRequestId: "action" },
+ },
+};
+const connect: StoryInteraction = {
+ id: "connect",
+ kind: "connection_intent",
+ status: "pending",
+ payload: { serviceSlug: "notion" },
+};
+describe("the decision must match the user story before clicking", () => {
+ it("accepts an approval bound to the installed fixture", () =>
+ expect(
+ pendingStoryDecision([tool], { kind: "tool", connectionId: "installed" }),
+ ).toEqual(tool));
+ it("rejects the historical Notion connection card in the tool-decline story", () =>
+ expect(() =>
+ pendingStoryDecision([connect], {
+ kind: "tool",
+ connectionId: "installed",
+ }),
+ ).toThrow(/connection_intent/));
+ it("rejects an ordinary completion confirmation", () =>
+ expect(() =>
+ pendingStoryDecision([{ ...tool, payload: {} }], {
+ kind: "tool",
+ connectionId: "installed",
+ }),
+ ).toThrow(/tool action/));
+ it("rejects an approval for a different connection", () =>
+ expect(() =>
+ pendingStoryDecision([tool], { kind: "tool", connectionId: "other" }),
+ ).toThrow(/connection/));
+ it("accepts Notion setup only in the new-connection story", () =>
+ expect(
+ pendingStoryDecision([connect], {
+ kind: "connection",
+ serviceSlug: "notion",
+ }),
+ ).toEqual(connect));
+ it("rejects the wrong provider in a connection story", () =>
+ expect(() =>
+ pendingStoryDecision([connect], {
+ kind: "connection",
+ serviceSlug: "github",
+ }),
+ ).toThrow(/github/));
+ it("rejects tool approval in the new-connection story", () =>
+ expect(() =>
+ pendingStoryDecision([tool], {
+ kind: "connection",
+ serviceSlug: "notion",
+ }),
+ ).toThrow(/connection_intent/));
+ it("rejects duplicate pending requests rather than clicking an arbitrary card", () =>
+ expect(() =>
+ pendingStoryDecision([tool, { ...tool, id: "duplicate" }], {
+ kind: "tool",
+ connectionId: "installed",
+ }),
+ ).toThrow(/exactly one/));
+ it("ignores resolved historical interactions", () =>
+ expect(
+ pendingStoryDecision([{ ...connect, status: "rejected" }, tool], {
+ kind: "tool",
+ connectionId: "installed",
+ }),
+ ).toEqual(tool));
+});
diff --git a/tests/runner-e2e/everyday-decisions.ts b/tests/runner-e2e/everyday-decisions.ts
new file mode 100644
index 0000000000..efc4f4db55
--- /dev/null
+++ b/tests/runner-e2e/everyday-decisions.ts
@@ -0,0 +1,48 @@
+export interface StoryInteraction {
+ id: string;
+ kind: string;
+ status: string;
+ title?: string;
+ payload?: {
+ serviceSlug?: string;
+ toolAction?: { connectionId?: string; actionRequestId?: string };
+ };
+}
+export type StoryDecision =
+ | { kind: "tool"; connectionId: string }
+ | { kind: "connection"; serviceSlug: string };
+export class StoryDecisionError extends Error {
+ readonly checkId = "decision-request-matches-story";
+}
+export function pendingStoryDecision(
+ rows: StoryInteraction[],
+ expected: StoryDecision,
+): StoryInteraction {
+ const pending = rows.filter((row) => row.status === "pending");
+ if (pending.length !== 1)
+ throw new StoryDecisionError(
+ `Expected exactly one pending decision; observed ${pending.length}.`,
+ );
+ const interaction = pending[0]!;
+ if (expected.kind === "tool") {
+ if (
+ interaction.kind !== "request_confirmation" ||
+ !interaction.payload?.toolAction?.actionRequestId
+ )
+ throw new StoryDecisionError(
+ `Expected a tool action review for the installed service; observed ${interaction.kind}: ${interaction.title ?? "untitled"}. No decision was taken.`,
+ );
+ if (interaction.payload.toolAction.connectionId !== expected.connectionId)
+ throw new StoryDecisionError(
+ "The tool approval belongs to a different connection. No decision was taken.",
+ );
+ } else if (
+ interaction.kind !== "connection_intent" ||
+ interaction.payload?.serviceSlug !== expected.serviceSlug
+ ) {
+ throw new StoryDecisionError(
+ `Expected connection_intent for ${expected.serviceSlug}; observed ${interaction.kind} for ${interaction.payload?.serviceSlug ?? "unknown service"}. No decision was taken.`,
+ );
+ }
+ return interaction;
+}
diff --git a/tests/runner-e2e/everyday-delivery.test.ts b/tests/runner-e2e/everyday-delivery.test.ts
new file mode 100644
index 0000000000..0e56ca2faa
--- /dev/null
+++ b/tests/runner-e2e/everyday-delivery.test.ts
@@ -0,0 +1,63 @@
+import { describe, expect, it } from "vitest";
+import {
+ latestStoryDelivery,
+ type StoryDelivery,
+} from "./everyday-delivery.js";
+const attachment = (
+ issueId: string,
+ id: string,
+ createdAt = "2026-09-14T12:00:00Z",
+): StoryDelivery => ({
+ issueId,
+ id,
+ createdAt,
+ originalFilename: "project.zip",
+});
+describe("delegated user delivery", () => {
+ it.each(["parent", "child"])(
+ "accepts a ZIP delivered on the %s task",
+ (issue) => {
+ expect(
+ latestStoryDelivery([attachment(issue, "zip")], ["parent", "child"])
+ ?.id,
+ ).toBe("zip");
+ },
+ );
+ it("does not borrow a ZIP from an unrelated task", () => {
+ expect(
+ latestStoryDelivery(
+ [attachment("unrelated", "zip")],
+ ["parent", "child"],
+ ),
+ ).toBeUndefined();
+ });
+ it("requires a new delivery after the revision request and chooses it across both tasks", () => {
+ const old = attachment("parent", "initial");
+ const cutoff = Date.parse("2026-09-14T12:05:00Z");
+ expect(
+ latestStoryDelivery([old], ["parent", "child"], cutoff),
+ ).toBeUndefined();
+ const revised = attachment("child", "revised", "2026-09-14T12:06:00Z");
+ expect(
+ latestStoryDelivery([old, revised], ["parent", "child"], cutoff)?.id,
+ ).toBe("revised");
+ });
+ it("ignores the preserved original even when it is republished after the revision", () => {
+ const revised = { ...attachment("child", "revised", "2026-09-14T12:06:00Z"), sha256: "new-content" };
+ const preserved = { ...attachment("parent", "original-copy", "2026-09-14T12:07:00Z"), sha256: "original-content" };
+ const cutoff = Date.parse("2026-09-14T12:05:00Z");
+ expect(latestStoryDelivery([revised, preserved], ["parent", "child"], cutoff, ["original-content"])?.id).toBe("revised");
+ expect(latestStoryDelivery([preserved], ["parent", "child"], cutoff, ["original-content"])).toBeUndefined();
+ });
+ it("rejects source files and malformed timestamps", () => {
+ expect(
+ latestStoryDelivery(
+ [
+ { ...attachment("parent", "source"), originalFilename: "source.py" },
+ attachment("child", "invalid", "unknown"),
+ ],
+ ["parent", "child"],
+ ),
+ ).toBeUndefined();
+ });
+});
diff --git a/tests/runner-e2e/everyday-delivery.ts b/tests/runner-e2e/everyday-delivery.ts
new file mode 100644
index 0000000000..c99258be83
--- /dev/null
+++ b/tests/runner-e2e/everyday-delivery.ts
@@ -0,0 +1,34 @@
+export interface StoryDelivery {
+ issueId: string;
+ id: string;
+ createdAt: string;
+ originalFilename?: string;
+ filename?: string;
+ contentType?: string;
+ sha256?: string;
+}
+
+/** A user can receive a delegated deliverable on the parent or the child. */
+export function latestStoryDelivery(
+ attachments: readonly StoryDelivery[],
+ allowedIssueIds: readonly string[],
+ after?: number,
+ preservedContentHashes: readonly string[] = [],
+): StoryDelivery | undefined {
+ const allowed = new Set(allowedIssueIds);
+ return attachments
+ .filter(
+ (a) =>
+ allowed.has(a.issueId) &&
+ (!a.sha256 || !preservedContentHashes.includes(a.sha256)) &&
+ (/\.zip$/i.test(a.originalFilename ?? a.filename ?? "") ||
+ a.contentType === "application/zip") &&
+ Number.isFinite(Date.parse(a.createdAt)) &&
+ (after === undefined || Date.parse(a.createdAt) >= after),
+ )
+ .sort(
+ (a, b) =>
+ Date.parse(b.createdAt) - Date.parse(a.createdAt) ||
+ b.id.localeCompare(a.id),
+ )[0];
+}
diff --git a/tests/runner-e2e/everyday-evidence.test.ts b/tests/runner-e2e/everyday-evidence.test.ts
new file mode 100644
index 0000000000..76e91af56e
--- /dev/null
+++ b/tests/runner-e2e/everyday-evidence.test.ts
@@ -0,0 +1,27 @@
+import { mkdtemp, mkdir, writeFile, readFile, rm } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { expect, it } from "vitest";
+import { packageEvidence } from "./evidence.js";
+it("preserves the before-decision screenshot in the packaged attempt", async () => {
+ const root = await mkdtemp(path.join(os.tmpdir(), "decision-evidence-"));
+ try {
+ const privateDir = path.join(root, "private");
+ const uploadDir = path.join(root, "upload");
+ await mkdir(privateDir);
+ const bytes = Buffer.from("89504e470d0a1a0a", "hex");
+ await writeFile(path.join(privateDir, "decision-pending.png"), bytes);
+ const packaged = await packageEvidence({
+ privateDir,
+ uploadDir,
+ secrets: [],
+ expectPassScreenshot: false,
+ });
+ expect(packaged.files).toContain("decision-pending.png");
+ expect(
+ await readFile(path.join(uploadDir, "decision-pending.png")),
+ ).toEqual(bytes);
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+});
diff --git a/tests/runner-e2e/everyday-flow.ts b/tests/runner-e2e/everyday-flow.ts
new file mode 100644
index 0000000000..6c4b7d0fc6
--- /dev/null
+++ b/tests/runner-e2e/everyday-flow.ts
@@ -0,0 +1,1164 @@
+import { expect, type Page } from "@playwright/test";
+import { spawn } from "node:child_process";
+import { createHash } from "node:crypto";
+import { mkdir, readFile, stat, readdir } from "node:fs/promises";
+import path from "node:path";
+import { isDeepStrictEqual } from "node:util";
+import { pollUntil, type RunnerApi } from "./api.js";
+import {
+ latestStoryDelivery,
+ type StoryDelivery,
+} from "./everyday-delivery.js";
+import type { LiveFixtureValues } from "./live-fixtures.js";
+import type { MatrixExecution } from "./types.js";
+import { createTaskThroughUi, submitTaskReply } from "./user-actions.js";
+import { setupConnectionReview } from "./connection-reviews.js";
+import {
+ pendingStoryDecision,
+ StoryDecisionError,
+ type StoryInteraction,
+} from "./everyday-decisions.js";
+import { LATE_REQUIREMENT, SLUGIFY_REVISION } from "./everyday-cases.js";
+import {
+ isActiveStoryRun,
+ isStoryWorkspaceDeferral,
+ storyLifecycleChecks,
+ storyRepliesConsumed,
+ storyParentFinishedAfterChildren,
+ type StoryCheck,
+ type StoryIssue,
+ type StoryRun,
+} from "./everyday-observations.js";
+
+type Row = Record;
+export interface EverydayEvidence {
+ schema: "paperclip.everyday-workflow.v1";
+ caseId: string;
+ prompt: string;
+ harnessDigest?: string;
+ sourceRevision?: string;
+ providerVersion?: string;
+ fixtureConfiguration?: {
+ apiToolsEnabled: boolean;
+ aiConnection?: LiveFixtureValues["aiConnection"];
+ };
+ documents?: Row[];
+ checks: StoryCheck[];
+ timeline: Array<{ at: string; action: string; detail?: unknown }>;
+ issues: Array;
+ runs: StoryRun[];
+ agents: Row[];
+ downloads: Row[];
+ allowedInterruptedRuns: string[];
+}
+interface Input {
+ page: Page;
+ api: RunnerApi;
+ fixtures: LiveFixtureValues;
+ execution: MatrixExecution;
+ nonce: string;
+ workspacePath: string;
+ privateDir: string;
+ deadlineAt: number;
+ restart(): Promise;
+ observe(issue: StoryIssue, runs: StoryRun[]): void;
+ capture(id: string, label: string, file: string): Promise;
+ evidence(name: string, value: unknown): Promise;
+}
+
+function runCommand(
+ command: string,
+ args: string[],
+ timeout = 20_000,
+): Promise<{ code: number; stdout: string }> {
+ return new Promise((resolve, reject) => {
+ const env = Object.fromEntries(
+ Object.entries(process.env).filter(([key]) =>
+ ["PATH", "SYSTEMROOT", "TMPDIR"].includes(key),
+ ),
+ );
+ const child = spawn(command, args, {
+ env,
+ stdio: ["ignore", "pipe", "pipe"],
+ });
+ let stdout = "";
+ let stderr = "";
+ const timer = setTimeout(() => child.kill("SIGKILL"), timeout);
+ child.stdout.on("data", (chunk) => {
+ stdout += String(chunk);
+ if (stdout.length > 1_000_000) child.kill("SIGKILL");
+ });
+ child.stderr.on("data", (chunk) => {
+ stderr += String(chunk);
+ });
+ child.on("error", (error) => {
+ clearTimeout(timer);
+ reject(error);
+ });
+ child.on("exit", (code) => {
+ clearTimeout(timer);
+ if (code === null)
+ reject(new Error("Bounded evaluator command timed out"));
+ else resolve({ code, stdout: stdout || stderr });
+ });
+ });
+}
+
+/** No DB writes, fabricated tool receipts, or corrective messages after a failed check. */
+export async function runEverydayFlow(input: Input) {
+ const { page, api, fixtures, execution, nonce } = input;
+ const prefix = fixtures.company.issuePrefix!;
+ const ev: EverydayEvidence = {
+ schema: "paperclip.everyday-workflow.v1",
+ caseId: execution.task.id,
+ prompt: execution.task.buildPrompt(nonce),
+ fixtureConfiguration: {
+ apiToolsEnabled:
+ process.env.PAPERCLIP_RUNNER_API_TOOLS_ENABLED === "true",
+ aiConnection: fixtures.aiConnection,
+ },
+ checks: [],
+ timeline: [],
+ issues: [],
+ runs: [],
+ agents: [],
+ downloads: [],
+ allowedInterruptedRuns: [],
+ };
+ const note = (action: string, detail?: unknown) =>
+ ev.timeline.push({
+ at: new Date().toISOString(),
+ action,
+ ...(detail === undefined ? {} : { detail }),
+ });
+ const check = (id: string, passed: boolean, detail: string) => {
+ ev.checks.push({ id, passed, detail });
+ };
+ let parent: StoryIssue | undefined;
+ let lastSubmissionAt = 0;
+ const submittedCommentIds: string[] = [];
+ let review: Awaited> | undefined;
+ let project = fixtures.project;
+ const caseId = execution.task.id;
+ const decliningConnection = caseId === "connection-decline";
+ const declining = decliningConnection || caseId === "service-decline";
+ let decisionId: string | undefined;
+ let decisionResolvedAt: string | undefined;
+ let initialConnections: string[] = [];
+ let stoppedWorkspace: Record | undefined;
+ async function workspaceFiles() {
+ const files: Record = {};
+ for (const entry of await readdir(input.workspacePath, {
+ withFileTypes: true,
+ })) {
+ if (entry.isFile() && /\.(py|md|zip)$/.test(entry.name))
+ files[entry.name] = createHash("sha256")
+ .update(await readFile(path.join(input.workspacePath, entry.name)))
+ .digest("hex");
+ }
+ return files;
+ }
+ async function runnerStopped(run: StoryRun) {
+ if (!run.processPid) return false;
+ const identity = await runCommand("ps", [
+ "-p",
+ String(run.processPid),
+ "-o",
+ "command=",
+ ]);
+ return (
+ identity.code !== 0 || !identity.stdout.includes(`--run-id ${run.id} `)
+ );
+ }
+ async function refresh() {
+ const listed = await api.get(
+ `/api/companies/${fixtures.company.id}/heartbeat-runs?limit=100`,
+ );
+ ev.runs = await Promise.all(
+ listed.map((r) => api.get(`/api/heartbeat-runs/${r.id}`)),
+ );
+ const issues = await api.get(
+ `/api/companies/${fixtures.company.id}/issues`,
+ );
+ ev.issues = await Promise.all(
+ issues.map(async (issue) => ({
+ ...issue,
+ comments: await api.get(
+ `/api/issues/${issue.id}/comments?order=asc`,
+ ),
+ queuedComments: await api.get(
+ `/api/issues/${issue.id}/queued-comments`,
+ ),
+ interactions: await api.get(
+ `/api/issues/${issue.id}/interactions`,
+ ),
+ })),
+ );
+ if (parent) {
+ parent = ev.issues.find((i) => i.id === parent!.id) as StoryIssue;
+ input.observe(parent!, ev.runs);
+ }
+ return ev;
+ }
+ const taskUrl = (issue: StoryIssue) =>
+ `/${prefix}/issues/${issue.identifier ?? issue.id}`;
+ async function openParent() {
+ await page.goto(taskUrl(parent!), { waitUntil: "domcontentloaded" });
+ }
+ async function reply(message: string, target: StoryIssue = parent!) {
+ const priorFailures = ev.checks.filter((c) => !c.passed);
+ if (priorFailures.length)
+ throw new Error(
+ `Story prerequisite failed before the next user request: ${priorFailures.map((c) => c.id).join(", ")}`,
+ );
+ const before = await api.get(`/api/issues/${target.id}/comments`);
+ lastSubmissionAt = Date.now();
+ await submitTaskReply(page, message);
+ const after = await pollUntil({
+ label: "one persisted user reply",
+ deadlineAt: Date.now() + 30_000,
+ load: () => api.get(`/api/issues/${target.id}/comments`),
+ accept: (rows) =>
+ rows.filter(
+ (c) => !c.authorAgentId && !before.some((old) => old.id === c.id),
+ ).length > 0,
+ });
+ const added = after.filter(
+ (c) => !c.authorAgentId && !before.some((old) => old.id === c.id),
+ );
+ check(
+ `reply-${ev.timeline.length}-stored-once`,
+ added.length === 1,
+ "A composer submission creates exactly one user message.",
+ );
+ submittedCommentIds.push(...added.map((c) => c.id));
+ note("composer-message-persisted", {
+ issueId: target.id,
+ commentIds: added.map((c) => c.id),
+ });
+ }
+ async function settled() {
+ await pollUntil({
+ label: `everyday ${caseId} settled`,
+ deadlineAt: input.deadlineAt,
+ intervalMs: 1000,
+ load: refresh,
+ accept: (state) =>
+ state.issues.length > 0 &&
+ state.issues.every(
+ (i) => i.status === "done" && i.queuedComments.entries.length === 0,
+ ) &&
+ storyRepliesConsumed(state.runs, submittedCommentIds) &&
+ state.runs.length > 0 &&
+ !state.runs.some(isActiveStoryRun) &&
+ state.runs.some(
+ (r) => Date.parse(r.finishedAt ?? "") >= lastSubmissionAt,
+ ),
+ reject: (state) => {
+ if (state.runs.length > 12) return "bounded execution count exceeded";
+ const bad = state.runs.find(
+ (r) =>
+ ["failed", "timed_out"].includes(r.status) &&
+ !ev.allowedInterruptedRuns.includes(r.id),
+ );
+ if (bad)
+ return `native execution failed ${bad.errorCode ?? ""}: ${bad.error ?? bad.status}`;
+ if (
+ state.runs.some(isActiveStoryRun) ||
+ state.issues.some((i) => i.scheduledRetry || i.activeRecoveryAction)
+ )
+ return;
+ if (
+ state.runs.length &&
+ state.issues.some((i) => i.status === "blocked")
+ )
+ return "task is Blocked without an active continuation";
+ if (
+ state.issues.some(
+ (i) =>
+ i.status === "in_review" &&
+ i.interactions.some((x: Row) => x.status === "pending"),
+ )
+ )
+ return "unexpected human interaction: task did not finish autonomously";
+ },
+ });
+ await openParent();
+ await expect(
+ page.getByTestId("issue-detail-header").getByRole("button", {
+ name: "Change status (current: Done)",
+ exact: true,
+ }),
+ ).toBeVisible();
+ note("all-tasks-done");
+ }
+ async function downloadDelegated(
+ mode: "base" | "separator" | "max-length",
+ phase: string,
+ after?: number,
+ ) {
+ const related = ev.issues.filter(
+ (issue) => issue.id === parent!.id || issue.parentId === parent!.id,
+ );
+ const attachments = (
+ await Promise.all(
+ related.map(async (issue) =>
+ (await api.get(`/api/issues/${issue.id}/attachments`)).map(
+ (a) => ({ ...a, issueId: issue.id }) as StoryDelivery,
+ ),
+ ),
+ )
+ ).flat();
+ const delivery = latestStoryDelivery(
+ attachments,
+ related.map((issue) => issue.id),
+ after,
+ after === undefined ? [] : ev.downloads.map((d) => d.sha256),
+ );
+ if (!delivery) {
+ check(
+ `${phase}.zip-delivered`,
+ false,
+ "No new downloadable ZIP was delivered on the parent or its child tasks.",
+ );
+ return;
+ }
+ note("delivery-selected", { attachmentId: delivery.id, issueId: delivery.issueId, sha256: delivery.sha256, after });
+ await download(delivery.issueId, mode, phase, delivery.id);
+ }
+
+ async function download(
+ issueId: string,
+ mode: "base" | "separator" | "max-length",
+ phase: string,
+ selectedAttachmentId?: string,
+ ) {
+ const attachments = await api.get(
+ `/api/issues/${issueId}/attachments`,
+ );
+ const zips = attachments
+ .filter(
+ (a) =>
+ String(a.originalFilename ?? a.filename ?? "").endsWith(".zip") ||
+ a.contentType === "application/zip",
+ )
+ .sort((a, b) => String(a.createdAt).localeCompare(String(b.createdAt)));
+ if (!zips.length) {
+ check(
+ `${phase}.zip-delivered`,
+ false,
+ "No downloadable ZIP attachment was delivered.",
+ );
+ return;
+ }
+ const attachment = selectedAttachmentId
+ ? zips.find((a) => a.id === selectedAttachmentId)
+ : zips[zips.length - 1];
+ if (!attachment) throw new Error("Selected delivery is no longer available");
+ const issue = ev.issues.find((i) => i.id === issueId)!;
+ await page.goto(taskUrl(issue), { waitUntil: "domcontentloaded" });
+ const links = page.locator(
+ `a[href*="/api/attachments/${attachment.id}/content"]`,
+ );
+ const link = links.filter({ hasText: /Download/i }).first();
+ await expect(link).toBeVisible({ timeout: 15_000 });
+ const pending = page.waitForEvent("download", { timeout: 30_000 });
+ await link.click();
+ const file = await pending;
+ const target = path.join(input.privateDir, "snapshots", `${phase}.zip`);
+ await file.saveAs(target);
+ const bytes = await readFile(target);
+ const digest = createHash("sha256").update(bytes).digest("hex");
+ const result = await runCommand(process.env.PYTHON ?? "python3", [
+ path.join(import.meta.dirname, "everyday-artifact.py"),
+ target,
+ "--mode",
+ mode,
+ ], Math.max(1, Math.min(60_000, input.deadlineAt - Date.now())));
+ const oracle = JSON.parse(result.stdout) as { checks: StoryCheck[] };
+ ev.checks.push(
+ ...oracle.checks.map((c) => ({ ...c, id: `${phase}.${c.id}` })),
+ );
+ ev.downloads.push({
+ phase,
+ attachmentId: attachment.id,
+ issueId,
+ sha256: digest,
+ bytes: bytes.length,
+ filename: file.suggestedFilename(),
+ mode,
+ });
+ note("download-verified", {
+ phase,
+ attachmentId: attachment.id,
+ sha256: digest,
+ });
+ await openParent();
+ }
+ async function sourceReady() {
+ await pollUntil({
+ label: "saved source before interruption",
+ deadlineAt: Math.min(input.deadlineAt, Date.now() + 180_000),
+ intervalMs: 500,
+ load: async () => {
+ await refresh();
+ try {
+ return (
+ (await stat(path.join(input.workspacePath, "slugify.py"))).size >
+ 0 && ev.runs.some(isActiveStoryRun)
+ );
+ } catch {
+ return false;
+ }
+ },
+ accept: Boolean,
+ });
+ const bytes = await readFile(path.join(input.workspacePath, "slugify.py"));
+ await input.evidence("source-before-interruption.json", {
+ body: bytes.toString("utf8"),
+ sha256: createHash("sha256").update(bytes).digest("hex"),
+ });
+ note("source-saved-before-interruption", {
+ sha256: createHash("sha256").update(bytes).digest("hex"),
+ bytes: bytes.length,
+ });
+ }
+ try {
+ await mkdir(path.join(input.privateDir, "snapshots"), { recursive: true });
+ const revision = await runCommand("git", ["rev-parse", "HEAD"]);
+ if (revision.code === 0) ev.sourceRevision = revision.stdout.trim();
+ const version = await runCommand(
+ execution.profile.provider === "acpx" ? "claude" : "codex",
+ ["--version"],
+ );
+ if (version.code === 0) ev.providerVersion = version.stdout.trim();
+ const harnessFiles = [
+ "everyday-flow.ts",
+ "everyday-cases.ts",
+ "everyday-decisions.ts",
+ "everyday-delivery.ts",
+ "everyday-observations.ts",
+ "everyday-artifact.py",
+ "user-actions.ts",
+ "runner.spec.ts",
+ "api.ts",
+ "harness-env.ts",
+ "failure-classifier.ts",
+ "live-fixtures.ts",
+ "connection-reviews.ts",
+ "catalog.ts",
+ ];
+ ev.harnessDigest = createHash("sha256")
+ .update(
+ (
+ await Promise.all(
+ harnessFiles.map((f) =>
+ readFile(path.join(import.meta.dirname, f)),
+ ),
+ )
+ )
+ .map((b) => b.toString())
+ .join("\n"),
+ )
+ .digest("hex");
+ if (!caseId.startsWith("service-") && !decliningConnection) {
+ try {
+ const sandbox = await runCommand(process.env.PYTHON ?? "python3", [
+ path.join(import.meta.dirname, "everyday-artifact.py"), "--preflight",
+ ]);
+ if (sandbox.code !== 0) throw new Error(sandbox.stdout);
+ note("artifact-sandbox-qualified", { isolation: "docker", network: "none" });
+ } catch (error) {
+ throw new Error(`Artifact sandbox qualification failed before task creation: ${error instanceof Error ? error.message : String(error)}`, { cause: error });
+ }
+ }
+ if (!project && !caseId.startsWith("service-") && !decliningConnection) {
+ project = await api.post(
+ `/api/companies/${fixtures.company.id}/projects`,
+ {
+ name: `Studio project ${nonce}`,
+ description: "Small software project",
+ executionWorkspacePolicy: {
+ enabled: true,
+ defaultMode: "shared_workspace",
+ sharedWorkspaceConcurrency: "serialize",
+ allowIssueOverride: false,
+ environmentId: fixtures.environment.id,
+ workspaceStrategy: { type: "project_primary" },
+ },
+ workspace: {
+ name: "Primary",
+ sourceType: "local_path",
+ cwd: input.workspacePath,
+ isPrimary: true,
+ },
+ },
+ );
+ }
+ await api.patch(`/api/agents/${fixtures.agent.id}/permissions`, {
+ canCreateAgents: true,
+ canAssignTasks: true,
+ });
+ if (caseId === "delegate-feedback") {
+ const config = execution.profile.buildAgent({
+ environmentId: fixtures.environment.id,
+ environmentFixtureId: execution.environment.id,
+ workspacePath: input.workspacePath,
+ secretRefs: fixtures.secretRefs,
+ executionId: nonce,
+ });
+ await api.post(`/api/companies/${fixtures.company.id}/agents`, {
+ ...config,
+ name: "Riley Builder",
+ role: "engineer",
+ title: "Engineer",
+ reportsTo: fixtures.agent.id,
+ instructionsBundle: {
+ entryFile: "AGENTS.md",
+ files: {
+ "AGENTS.md":
+ "You implement small software projects and verify your work. Follow task feedback and deliver usable files.",
+ },
+ },
+ });
+ }
+ if (caseId.startsWith("service-"))
+ review = await setupConnectionReview({
+ page,
+ api,
+ prefix,
+ companyId: fixtures.company.id,
+ agentId: fixtures.agent.id,
+ marker: `Pages: Roadmap, Meeting notes. Verification code: SERVICE_${nonce}`,
+ authenticated: true,
+ });
+ if (decliningConnection) {
+ const state = await api.get<{ connections: Row[] }>(
+ `/api/companies/${fixtures.company.id}/tools/connections`,
+ );
+ initialConnections = state.connections.map((c) => c.id);
+ check(
+ "connection-starts-unconfigured",
+ state.connections.length === 0,
+ "This isolated company has no service connection before the request.",
+ );
+ if (state.connections.length)
+ throw new Error("New-connection story requires an unconnected company");
+ }
+ await createTaskThroughUi({
+ page,
+ issuePrefix: prefix,
+ agentName: fixtures.agent.name,
+ title: execution.task.buildTitle(nonce),
+ prompt: ev.prompt,
+ workMode: "standard",
+ projectName: project?.name,
+ });
+ parent = await pollUntil({
+ label: "browser-created story task",
+ deadlineAt: Date.now() + 30_000,
+ load: () =>
+ api.get(`/api/companies/${fixtures.company.id}/issues`),
+ accept: (rows) =>
+ rows.some((i) => i.title === execution.task.buildTitle(nonce)),
+ }).then((rows) =>
+ rows.find((i) => i.title === execution.task.buildTitle(nonce))!,
+ );
+ input.observe(parent!, []);
+ note("task-submitted", { issueId: parent!.id });
+ await openParent();
+ if (caseId === "delegate-feedback") {
+ await pollUntil({
+ label: "active delegated child",
+ deadlineAt: input.deadlineAt,
+ load: refresh,
+ accept: (state) =>
+ state.issues.some(
+ (i) =>
+ i.parentId === parent!.id &&
+ state.runs.some(
+ (r) =>
+ isActiveStoryRun(r) &&
+ (r.contextSnapshot?.issueId === i.id ||
+ r.contextSnapshot?.taskId === i.id),
+ ),
+ ),
+ reject: (state) => {
+ const failed = state.runs.find((run) =>
+ ["failed", "timed_out"].includes(run.status),
+ );
+ return failed
+ ? `Delegation prerequisite failed before feedback: ${failed.errorCode}: ${failed.error}`
+ : undefined;
+ },
+ });
+ const child = ev.issues.find((i) => i.parentId === parent!.id)!;
+ note("late-feedback-boundary", {
+ childId: child.id,
+ activeRunIds: ev.runs.filter(isActiveStoryRun).map((r) => r.id),
+ });
+ await page.goto(taskUrl(child), { waitUntil: "domcontentloaded" });
+ await reply(LATE_REQUIREMENT, child);
+ note("late-feedback-delivered-to-child", { childId: child.id });
+ await openParent();
+ }
+ if (
+ caseId === "recover-controller" ||
+ caseId === "stop-redirect"
+ ) {
+ if (execution.environment.id === "daytona") {
+ // A completed, downloaded first version is an observable remote persistence checkpoint.
+ await settled();
+ await download(parent!.id, "base", "before-restart");
+ if (!ev.downloads.length || ev.checks.some((c) => !c.passed))
+ throw new Error(
+ "Remote fault boundary unexercised: no verified saved project",
+ );
+ await reply(SLUGIFY_REVISION);
+ note("remote-revision-submitted");
+ await pollUntil({
+ label: "remote revision executing",
+ deadlineAt: input.deadlineAt,
+ load: refresh,
+ accept: (s) => s.runs.some(isActiveStoryRun),
+ });
+ } else await sourceReady();
+ const active = ev.runs.find((r) => r.status === "running");
+ if (!active)
+ throw new Error(
+ "Fault boundary was not exercised: no active execution",
+ );
+ if (caseId === "stop-redirect") {
+ await page.getByTestId("task-chat-composer-stop").last().click();
+ ev.allowedInterruptedRuns.push(active.id);
+ note("stop-clicked", { runId: active.id });
+ await pollUntil({
+ label: "owned runner stopped",
+ deadlineAt: input.deadlineAt,
+ load: () => runnerStopped(active),
+ accept: Boolean,
+ intervalMs: 250,
+ });
+ stoppedWorkspace = await workspaceFiles();
+ note("stopped-workspace-snapshot", stoppedWorkspace);
+ await reply(
+ `Change direction. Leave the project as it is. Reply with just this short note: "The studio is ready. Reference ${nonce}."`,
+ );
+ note("new-direction-submitted");
+ await page.reload();
+ } else {
+ await reply(LATE_REQUIREMENT);
+ note("followup-submitted-before-interruption");
+ ev.allowedInterruptedRuns.push(active.id);
+ await input.restart();
+ note("controller-restarted");
+ await openParent();
+ }
+ }
+ if (review || decliningConnection) {
+ const interactions = await pollUntil({
+ label: "story decision request",
+ deadlineAt: input.deadlineAt,
+ load: async () => {
+ const [interactions, issue] = await Promise.all([
+ api.get(
+ `/api/issues/${parent!.id}/interactions`,
+ ),
+ api.get(`/api/issues/${parent!.id}`),
+ ]);
+ return { interactions, issue, calls: review?.invocationCount() ?? 0 };
+ },
+ accept: (state) =>
+ state.interactions.some((i) => i.status === "pending"),
+ reject: (state) =>
+ state.calls > 0
+ ? `The provider received ${state.calls} call(s) before approval.`
+ : ["done", "blocked", "cancelled"].includes(state.issue.status)
+ ? `Task reached ${state.issue.status} without requesting the expected decision.`
+ : undefined,
+ });
+ await openParent();
+ await expect(
+ page
+ .locator(
+ '[data-testid="task-chat-thread"], [data-testid="thread-root"]',
+ )
+ .first(),
+ ).toBeVisible();
+ await expect(page.getByTestId("issue-chat-skeleton")).toHaveCount(0);
+ const pendingInteraction = pendingStoryDecision(
+ interactions.interactions,
+ review
+ ? { kind: "tool", connectionId: review.connectionId }
+ : { kind: "connection", serviceSlug: "notion" },
+ );
+ await expect(
+ page.getByRole("button", {
+ name: decliningConnection ? "Not now" : "Review request",
+ exact: true,
+ }),
+ ).toBeVisible();
+ await input.capture(
+ "decision-pending",
+ "Request before the user decision",
+ "decision-pending.png",
+ );
+ decisionId = pendingInteraction.id;
+ check(
+ "decision-request-matches-story",
+ true,
+ review
+ ? "Tool approval belongs to the installed page service."
+ : "New connection request is for Notion.",
+ );
+ if (review)
+ check(
+ "no-call-before-approval",
+ review.invocationCount() === 0,
+ "Service must not execute before the user decides.",
+ );
+ if (decliningConnection) {
+ await page
+ .getByRole("button", { name: "Not now", exact: true })
+ .click();
+ } else {
+ const dismiss = page.getByRole("button", {
+ name: "Dismiss Approve tool action",
+ });
+ if (await dismiss.isVisible()) await dismiss.click();
+ await page
+ .getByRole("button", { name: "Review request", exact: true })
+ .click();
+ await page
+ .getByRole("button", {
+ name: declining ? "Decline" : "Approve & run",
+ exact: true,
+ })
+ .click();
+ }
+ const decisionStatus = declining ? "rejected" : "accepted";
+ const decided = await pollUntil({
+ label: "connection decision persisted",
+ deadlineAt: Math.min(input.deadlineAt, Date.now() + 30_000),
+ load: () => api.get(`/api/issues/${parent!.id}/interactions`),
+ accept: (rows) =>
+ rows.some(
+ (interaction) =>
+ interaction.id === pendingInteraction.id &&
+ interaction.status === decisionStatus,
+ ),
+ });
+ decisionResolvedAt = decided.find((i) => i.id === decisionId)?.resolvedAt;
+ check(
+ "user-decision-persisted",
+ true,
+ `Interaction ${decisionId} saved as ${decisionStatus}.`,
+ );
+ note("connection-decision", {
+ decision: caseId,
+ interactionId: pendingInteraction.id,
+ status: decisionStatus,
+ });
+ }
+ await settled();
+ if (declining) {
+ const issue = ev.issues.find((i) => i.id === parent!.id)!;
+ const requests = issue.interactions as Row[];
+ check(
+ "decline-not-repeated",
+ requests.length === 1 &&
+ requests[0]?.id === decisionId &&
+ requests[0]?.status === "rejected",
+ "The saved decline remains rejected and no replacement request appears.",
+ );
+ const replies = issue.comments.filter(
+ (c: Row) =>
+ c.authorAgentId &&
+ Date.parse(c.createdAt) >= Date.parse(decisionResolvedAt ?? ""),
+ );
+ const text = replies.map((c: Row) => String(c.body ?? "")).join("\n");
+ check(
+ "decline-visible-explanation",
+ replies.length > 0 &&
+ /declin|not now|could(?:n.t| not)|cannot|can.t|unable|not (?:connect|retriev)|without (?:access|connect)/i.test(
+ text,
+ ),
+ "A new agent response explains the missing access after the saved decline.",
+ );
+ check(
+ "decline-no-fabricated-result",
+ !text.includes(`SERVICE_${nonce}`),
+ "The fallback does not claim the private verification code.",
+ );
+ if (decliningConnection) {
+ const state = await api.get<{ connections: Row[] }>(
+ `/api/companies/${fixtures.company.id}/tools/connections`,
+ );
+ check(
+ "decline-no-connection-created",
+ isDeepStrictEqual(
+ state.connections.map((c) => c.id).sort(),
+ initialConnections.sort(),
+ ),
+ "Not now did not create a service connection.",
+ );
+ }
+ }
+ if (caseId === "build-revise") {
+ await download(parent!.id, "base", "initial");
+ await reply(SLUGIFY_REVISION);
+ note("revision-requested");
+ await settled();
+ await download(parent!.id, "separator", "revised");
+ check(
+ "new-artifact-revision",
+ ev.downloads.length === 2 &&
+ ev.downloads[0]!.sha256 !== ev.downloads[1]!.sha256,
+ "The follow-up must deliver a new version.",
+ );
+ if (ev.downloads[0]) {
+ const response = await api.request.get(
+ `/api/attachments/${ev.downloads[0].attachmentId}/content`,
+ );
+ check(
+ "prior-download-preserved",
+ response.ok() &&
+ createHash("sha256")
+ .update(await response.body())
+ .digest("hex") === ev.downloads[0].sha256,
+ "The first delivered version remains retrievable.",
+ );
+ }
+ } else if (caseId === "hire-reuse") {
+ let agents = await api.get(
+ `/api/companies/${fixtures.company.id}/agents`,
+ );
+ const hires = agents.filter((a) => a.name === "Morgan QA");
+ check(
+ "exactly-one-hire",
+ hires.length === 1,
+ "One Morgan QA must be hired.",
+ );
+ check(
+ "manager-correct",
+ hires.length === 1 && hires[0]!.reportsTo === fixtures.agent.id,
+ "The hired agent reports to the lead.",
+ );
+ const lead = agents.find((a) => a.id === fixtures.agent.id);
+ check(
+ "hire-native-connection",
+ hires.length === 1 &&
+ hires[0]!.adapterType === "paperclip_runner" &&
+ hires[0]!.adapterConfig?.model === lead?.adapterConfig?.model &&
+ isDeepStrictEqual(
+ hires[0]!.runtimeConfig?.aiConnection,
+ fixtures.aiConnection?.binding,
+ ),
+ "The hire keeps the native model and inherits the managed AI account binding.",
+ );
+ const children = ev.issues.filter((i) => i.parentId === parent!.id);
+ check(
+ "hired-agent-executed",
+ hires.length === 1 &&
+ ev.runs.some(
+ (r) =>
+ r.agentId === hires[0]!.id &&
+ r.status === "succeeded" &&
+ (r.contextSnapshot?.aiConnection as Row | undefined)
+ ?.connectionId === fixtures.aiConnection?.connectionId,
+ ),
+ "The new hire must complete a run using the fixture managed account.",
+ );
+ await downloadDelegated("base", "hired-delivery");
+ const reuseRequestedAt = Date.now();
+ await reply(
+ `Have the existing Morgan QA add --separator support to the delivered project. Use the same agent; do not hire another. ${SLUGIFY_REVISION}`,
+ );
+ note("reuse-requested");
+ await settled();
+ agents = await api.get(
+ `/api/companies/${fixtures.company.id}/agents`,
+ );
+ check(
+ "hire-reused",
+ agents.filter((a) => a.name === "Morgan QA").length === 1 &&
+ agents.some((a) => a.id === hires[0]?.id),
+ "The original hired identity remains unique.",
+ );
+ check(
+ "hired-agent-executed-revision",
+ ev.runs.some(
+ (r) =>
+ r.agentId === hires[0]?.id &&
+ r.status === "succeeded" &&
+ Date.parse(r.finishedAt ?? "") >= reuseRequestedAt,
+ ),
+ "The same hired agent performs the follow-up work.",
+ );
+ await downloadDelegated("separator", "reused-delivery", reuseRequestedAt);
+ } else if (caseId === "delegate-feedback") {
+ const children = ev.issues.filter((i) => i.parentId === parent!.id);
+ check(
+ "one-child",
+ children.length === 1,
+ "Exactly one delegated child task.",
+ );
+ check(
+ "child-consumed-feedback",
+ Boolean(children[0]) &&
+ storyRepliesConsumed(
+ ev.runs.filter(
+ (run) =>
+ run.contextSnapshot?.issueId === children[0]?.id ||
+ run.contextSnapshot?.taskId === children[0]?.id,
+ ),
+ submittedCommentIds,
+ ),
+ "A completed child execution consumed the delivered user feedback.",
+ );
+ if (children[0])
+ await downloadDelegated("max-length", "delegated-delivery");
+ check(
+ "feedback-delivered-to-child",
+ Boolean(
+ children[0]?.comments.some((c: Row) =>
+ String(c.body).includes("--max-length"),
+ ),
+ ),
+ "The child history contains the late requirement.",
+ );
+ } else if (caseId === "recover-controller")
+ await download(parent!.id, "max-length", "recovered-delivery");
+ else if (caseId === "stop-redirect")
+ check(
+ "new-direction-delivered",
+ ev.issues
+ .find((i) => i.id === parent!.id)
+ ?.comments.some(
+ (c: Row) =>
+ c.authorAgentId && String(c.body).includes(`Reference ${nonce}`),
+ ) ?? false,
+ "The new request is answered after Stop and reload.",
+ );
+ if (stoppedWorkspace) {
+ check(
+ "stop.workspace-unchanged",
+ isDeepStrictEqual(stoppedWorkspace, await workspaceFiles()),
+ "Project files remain unchanged from verified runner stop through the new response.",
+ );
+ check(
+ "stop.no-old-run-active",
+ ev.runs
+ .filter((r) => ev.allowedInterruptedRuns.includes(r.id))
+ .every((r) => !isActiveStoryRun(r)),
+ "The stopped run is terminal after the new response.",
+ );
+ }
+ if (review) {
+ check(
+ "service-call-count",
+ review.invocationCount() === (caseId === "service-decline" ? 0 : 1),
+ "Exactly one approved service call; none after decline.",
+ );
+ if (caseId === "service-approve") {
+ const docs = await api.get(
+ `/api/issues/${parent!.id}/documents`,
+ );
+ const bodies = await Promise.all(
+ docs.map((d) =>
+ api.get(
+ `/api/issues/${parent!.id}/documents/${encodeURIComponent(d.key)}`,
+ ),
+ ),
+ );
+ const attachments = await api.get(
+ `/api/issues/${parent!.id}/attachments`,
+ );
+ for (const attachment of attachments.filter(
+ (a) =>
+ /\.(?:md|txt)$/i.test(a.originalFilename ?? a.filename ?? "") ||
+ ["text/markdown", "text/plain"].includes(a.contentType),
+ )) {
+ const url = `/api/attachments/${attachment.id}/content`;
+ const response = await api.request.get(url);
+ if (!response.ok()) continue;
+ await expect(page.locator(`a[href*="${url}"]`).first()).toBeVisible();
+ bodies.push({
+ id: attachment.id,
+ title: attachment.originalFilename ?? attachment.filename,
+ body: await response.text(),
+ source: "delivered-attachment",
+ });
+ }
+ const text = bodies
+ .map((d) => String(d.body ?? d.revision?.body ?? ""))
+ .join("\n");
+ check(
+ "briefing-uses-real-result",
+ text.includes(`SERVICE_${nonce}`),
+ "A delivered issue document or Markdown attachment contains the actual service verification code.",
+ );
+ check(
+ "briefing-includes-page-titles",
+ text.includes("Roadmap") && text.includes("Meeting notes"),
+ "The briefing includes both page titles returned by the service.",
+ );
+ ev.documents = bodies;
+ ev.issues.find((i) => i.id === parent!.id)!.documents = bodies;
+ }
+ }
+ await page.reload();
+ await refresh();
+ check(
+ "submitted-replies-consumed",
+ storyRepliesConsumed(ev.runs, submittedCommentIds),
+ "Every submitted user message appears in a successfully completed native execution input.",
+ );
+ if (caseId === "delegate-feedback" || caseId === "hire-reuse") {
+ check(
+ "parent-finishes-after-child",
+ storyParentFinishedAfterChildren(
+ ev.runs,
+ parent!.id,
+ fixtures.agent.id,
+ ev.issues.filter((i) => i.parentId === parent!.id).map((i) => i.id),
+ ),
+ "The lead completes only after the final child execution.",
+ );
+ }
+ const expectedModel = execution.profile.model;
+ check(
+ "native-model-config",
+ ev.runs.length > 0 &&
+ ev.runs
+ .filter((r) => !isStoryWorkspaceDeferral(r))
+ .every(
+ (r) =>
+ (r.runnerProfileJson?.nativeExecutionInput as Row | undefined)
+ ?.provider?.model === expectedModel,
+ ),
+ "Persisted native execution inputs use the selected model; this does not claim provider-side model identity.",
+ );
+ check(
+ "native-terminal-contract",
+ ev.runs
+ .filter(
+ (r) =>
+ !isStoryWorkspaceDeferral(r) &&
+ !ev.allowedInterruptedRuns.includes(r.id),
+ )
+ .every(
+ (r) =>
+ (r.resultJson?.nativeTerminal as Row | undefined)?.schema ===
+ "paperclip.prp.terminal.v1",
+ ),
+ "Successful runs retain the native terminal contract.",
+ );
+ ev.checks.push(
+ ...storyLifecycleChecks({
+ issues: ev.issues as StoryIssue[],
+ runs: ev.runs,
+ parentId: parent!.id,
+ leadId: fixtures.agent.id,
+ allowedInterruptedRuns: ev.allowedInterruptedRuns,
+ }),
+ );
+ check(
+ "no-pending-bookkeeping",
+ ev.issues.every((i) =>
+ i.interactions.every((x: Row) => x.status !== "pending"),
+ ),
+ "No completion confirmation or unanswered interaction remains.",
+ );
+ await expect(
+ page
+ .locator(
+ '[data-testid="task-chat-thread"], [data-testid="thread-root"]',
+ )
+ .first(),
+ ).toBeVisible();
+ const latestAgentComment = ev.issues
+ .find((i) => i.id === parent!.id)
+ ?.comments.filter((c: Row) => c.authorAgentId)
+ .at(-1);
+ if (latestAgentComment) {
+ const response = page.locator(`[id="comment-${latestAgentComment.id}"]`);
+ await expect(response).toBeVisible();
+ await response.scrollIntoViewIfNeeded();
+ }
+ await input.capture(
+ "final-state",
+ "Finished everyday workflow",
+ "final-state.png",
+ );
+ const failed = ev.checks.filter((c) => !c.passed);
+ if (failed.length)
+ throw new Error(
+ `Everyday outcome checks failed: ${failed.map((c) => c.id).join(", ")}`,
+ );
+ return { issue: parent!, runs: ev.runs, evidence: ev };
+ } catch (error) {
+ check(
+ error instanceof StoryDecisionError
+ ? error.checkId
+ : "workflow-completed",
+ false,
+ error instanceof Error ? error.message : String(error),
+ );
+ throw error;
+ } finally {
+ try {
+ await refresh();
+ ev.agents = await api.get(
+ `/api/companies/${fixtures.company.id}/agents`,
+ );
+ if (ev.documents && parent)
+ ev.issues.find((i) => i.id === parent!.id)!.documents = ev.documents;
+ } catch (error) {
+ note("evidence-capture-error", String(error));
+ }
+ if (caseId === "recover-controller") {
+ note("recovery-final-observation", {
+ taskStatus: parent?.status,
+ pendingCommentIds: ev.issues.flatMap((i) =>
+ (i.queuedComments?.entries ?? []).map((e: Row) => e.comment.id),
+ ),
+ failureCodes: ev.runs
+ .filter((r) => r.errorCode)
+ .map((r) => ({ runId: r.id, code: r.errorCode })),
+ });
+ if (execution.environment.id === "local") {
+ try {
+ const source = await readFile(
+ path.join(input.workspacePath, "slugify.py"),
+ "utf8",
+ );
+ await input.evidence("source-after-interruption.json", {
+ body: source,
+ sha256: createHash("sha256").update(source).digest("hex"),
+ });
+ } catch {
+ note("saved-source-unavailable-at-final-capture");
+ }
+ }
+ }
+ if (review) {
+ note("service-final-observation", {
+ invocationCount: review.invocationCount(),
+ requests: review.captures,
+ decisionTaken: ev.timeline.some(
+ (entry) => entry.action === "connection-decision",
+ ),
+ });
+ }
+ await input.evidence("everyday-workflow.json", ev);
+ await input.evidence("api-state.json", {
+ capturePhase: "everyday-final",
+ issue: parent,
+ runs: ev.runs,
+ issues: ev.issues,
+ checks: ev.checks,
+ });
+ await review?.close();
+ }
+}
diff --git a/tests/runner-e2e/everyday-observations.ts b/tests/runner-e2e/everyday-observations.ts
new file mode 100644
index 0000000000..49077e80af
--- /dev/null
+++ b/tests/runner-e2e/everyday-observations.ts
@@ -0,0 +1,172 @@
+export interface StoryCheck {
+ id: string;
+ passed: boolean;
+ detail: string;
+}
+export interface StoryIssue {
+ id: string;
+ companyId: string;
+ title: string;
+ status: string;
+ identifier?: string | null;
+ description?: string | null;
+ createdAt?: string;
+ assigneeAgentId?: string | null;
+ parentId?: string | null;
+ projectId?: string | null;
+ executionRunId?: string | null;
+ scheduledRetry?: unknown;
+ activeRecoveryAction?: unknown;
+}
+export interface StoryRun {
+ id: string;
+ companyId: string;
+ agentId: string;
+ status: string;
+ runtimeMode?: string;
+ runnerInstanceId?: string | null;
+ processPid?: number | null;
+ processStartedAt?: string | null;
+ nativeIssueId?: string | null;
+ nativeSessionId?: string | null;
+ contextSnapshot?: Record | null;
+ resultJson?: Record | null;
+ usageJson?: Record | null;
+ runnerProfileJson?: Record | null;
+ sessionIdAfter?: string | null;
+ startedAt?: string | null;
+ finishedAt?: string | null;
+ retryOfRunId?: string | null;
+ scheduledRetryReason?: string | null;
+ runtimeModeResolvedAt?: string | null;
+ lastOutputSeq?: number | null;
+ errorCode?: string | null;
+ error?: string | null;
+}
+export const isActiveStoryRun = (run: StoryRun) =>
+ ["queued", "running", "scheduled_retry"].includes(run.status);
+export function isStoryWorkspaceDeferral(run: StoryRun) {
+ const recovery = run.resultJson?.executionRecovery as
+ Record | undefined;
+ const startup = run.resultJson?.startupCancellation as
+ Record | undefined;
+ const neverStartedRetry =
+ run.errorCode === "cancelled" &&
+ run.scheduledRetryReason === "workspace_busy" &&
+ typeof run.retryOfRunId === "string" &&
+ run.retryOfRunId.length > 0 &&
+ run.startedAt === null &&
+ run.runtimeModeResolvedAt === null &&
+ run.lastOutputSeq === 0 &&
+ !run.processStartedAt &&
+ !run.sessionIdAfter &&
+ !run.usageJson &&
+ !run.runnerProfileJson?.nativeExecutionInput &&
+ typeof startup?.requestedAt === "string" &&
+ Number.isFinite(Date.parse(startup.requestedAt));
+ return (
+ run.status === "cancelled" &&
+ ((run.errorCode === "workspace_busy" &&
+ recovery?.providerWorkStarted === false) ||
+ neverStartedRetry) &&
+ !run.runnerInstanceId &&
+ !run.nativeSessionId &&
+ !run.processPid
+ );
+}
+export function storyLifecycleChecks(input: {
+ issues: StoryIssue[];
+ runs: StoryRun[];
+ parentId: string;
+ leadId: string;
+ allowedInterruptedRuns?: string[];
+}): StoryCheck[] {
+ const executed = input.runs.filter((r) => !isStoryWorkspaceDeferral(r));
+ const allowed = new Set(input.allowedInterruptedRuns ?? []);
+ const check = (id: string, passed: boolean, detail: string): StoryCheck => ({
+ id,
+ passed,
+ detail,
+ });
+ return [
+ check(
+ "tasks-done",
+ input.issues.length > 0 && input.issues.every((i) => i.status === "done"),
+ "Every story task must reach Done.",
+ ),
+ check(
+ "settled",
+ !input.runs.some(isActiveStoryRun) &&
+ !input.issues.some((i) => i.scheduledRetry || i.activeRecoveryAction),
+ "No active runs or scheduled recovery remain.",
+ ),
+ check(
+ "native-runtime",
+ executed.length > 0 &&
+ executed.every(
+ (r) => r.runtimeMode === "native" && Boolean(r.runnerInstanceId),
+ ),
+ "All executions must prove native runtime and runner identity.",
+ ),
+ check(
+ "successful-runs",
+ executed.every((r) => r.status === "succeeded" || allowed.has(r.id)),
+ "Only explicitly interrupted runs may have a non-success terminal state.",
+ ),
+ check(
+ "bounded-work",
+ input.runs.length <= 12,
+ "No more than twelve executions, including child and recovery turns.",
+ ),
+ check(
+ "parent-owned-by-lead",
+ !executed.some(
+ (r) =>
+ (r.contextSnapshot?.issueId === input.parentId ||
+ r.contextSnapshot?.taskId === input.parentId) &&
+ r.agentId !== input.leadId,
+ ),
+ "A mentioned worker must not execute on the parent.",
+ ),
+ ];
+}
+
+/** A terminal first turn cannot satisfy a later queued request. */
+export function storyRepliesConsumed(
+ runs: StoryRun[],
+ commentIds: string[],
+): boolean {
+ return commentIds.every((id) =>
+ runs.some((run) => {
+ if (run.status !== "succeeded" || isStoryWorkspaceDeferral(run))
+ return false;
+ const input = run.runnerProfileJson?.nativeExecutionInput as
+ { task?: { prompt?: string } } | undefined;
+ return input?.task?.prompt?.includes(id) === true;
+ }),
+ );
+}
+
+export function storyParentFinishedAfterChildren(
+ runs: StoryRun[],
+ parentId: string,
+ leadId: string,
+ childIds: string[],
+): boolean {
+ const scope = (r: StoryRun) =>
+ r.nativeIssueId ?? r.contextSnapshot?.issueId ?? r.contextSnapshot?.taskId;
+ const completed = runs.filter(
+ (r) => r.status === "succeeded" && !isStoryWorkspaceDeferral(r),
+ );
+ const children = completed.filter((r) => childIds.includes(String(scope(r))));
+ if (!children.length) return false;
+ const lastChildFinish = Math.max(
+ ...children.map((r) => Date.parse(r.finishedAt ?? "")),
+ );
+ return completed.some(
+ (r) =>
+ r.agentId === leadId &&
+ scope(r) === parentId &&
+ Date.parse(r.finishedAt ?? "") >= lastChildFinish,
+ );
+}
diff --git a/tests/runner-e2e/everyday.test.ts b/tests/runner-e2e/everyday.test.ts
new file mode 100644
index 0000000000..fb9a09fcc5
--- /dev/null
+++ b/tests/runner-e2e/everyday.test.ts
@@ -0,0 +1,264 @@
+import { describe, it, expect } from "vitest";
+import { runnerMatrix, runnerSuites } from "./catalog.js";
+import { parseRunnerSelectors, selectRunnerExecutions } from "./selectors.js";
+import { everydayTasks, productionStoryProfile } from "./everyday-cases.js";
+import {
+ isStoryWorkspaceDeferral,
+ storyLifecycleChecks,
+ storyRepliesConsumed,
+ storyParentFinishedAfterChildren,
+ type StoryRun,
+} from "./everyday-observations.js";
+
+describe("manual everyday workflow catalog", () => {
+ it("is discoverable and explicitly selected without changing scheduled --all", () => {
+ const all = selectRunnerExecutions(parseRunnerSelectors(["--all"]));
+ expect(all.some((e) => e.suite.id === "everyday-workflows")).toBe(false);
+ const selected = selectRunnerExecutions(
+ parseRunnerSelectors(["--suite", "everyday-workflows"]),
+ );
+ expect(selected).toHaveLength(30);
+ expect(selected.every((e) => e.profile.generation === "native")).toBe(true);
+ expect(
+ selectRunnerExecutions(
+ parseRunnerSelectors(["--profile", "runner-codex"]),
+ ).some((e) => e.suite.id === "everyday-workflows"),
+ ).toBe(false);
+ const listed = selectRunnerExecutions(parseRunnerSelectors(["--list"]));
+ expect(listed.some((e) => e.suite.id === "everyday-workflows")).toBe(true);
+ });
+ it("does not schedule arbitrary runner-crash probes as model evals", () => {
+ const selected = selectRunnerExecutions(
+ parseRunnerSelectors(["--suite", "everyday-workflows"]),
+ );
+ expect(selected.some((e) => e.task.id.startsWith("recover-runner"))).toBe(false);
+ for (const id of ["recover-runner", "recover-runner-safe", "recover-runner-uncertain"])
+ expect(() => selectRunnerExecutions(parseRunnerSelectors([
+ "--id", `everyday-workflows.runner-codex.local.${id}`,
+ ]))).toThrow();
+ expect(selected.some((e) => e.task.id === "recover-controller")).toBe(true);
+ expect(selected.some((e) => e.task.id === "stop-redirect")).toBe(true);
+ });
+ it("keeps ordinary prompts free of completion/API instructions", () => {
+ for (const task of everydayTasks)
+ expect(task.buildPrompt("sample")).not.toMatch(
+ /finish_task|paperclip_finish|PATCH|mark .*done|idempotencyKey/i,
+ );
+ const profile = runnerSuites.find((s) => s.id === "everyday-workflows")!
+ .profiles[0]!;
+ const value = productionStoryProfile(profile).buildAgent({
+ environmentId: "env",
+ environmentFixtureId: "local",
+ workspacePath: "/tmp/test",
+ secretRefs: {
+ OPENAI_API_KEY: {
+ type: "secret_ref",
+ secretId: "secret",
+ version: "latest",
+ },
+ },
+ executionId: "case",
+ });
+ expect(JSON.stringify(value.instructionsBundle)).not.toMatch(
+ /fixture|mark .*done|finish_task|api\/issues/i,
+ );
+ });
+ it("does not claim unsupported remote crash/hiring coverage", () => {
+ const remote = runnerMatrix.filter(
+ (e) =>
+ e.suite.id === "everyday-workflows" && e.environment.id === "daytona",
+ );
+ expect(remote).toHaveLength(6);
+ expect(new Set(remote.map((e) => e.task.id))).toEqual(
+ new Set(["build-revise", "delegate-feedback", "recover-controller"]),
+ );
+ });
+});
+
+describe("lifecycle oracle calibrated failures", () => {
+ const parent = {
+ id: "parent",
+ companyId: "company",
+ title: "Project",
+ status: "done",
+ assigneeAgentId: "lead",
+ };
+ const run: StoryRun = {
+ id: "run",
+ companyId: "company",
+ agentId: "lead",
+ status: "succeeded",
+ runtimeMode: "native",
+ runnerInstanceId: "runner",
+ contextSnapshot: { issueId: "parent" },
+ };
+ const score = (
+ runs: StoryRun[],
+ issues = [parent],
+ allowedInterruptedRuns: string[] = [],
+ ) =>
+ storyLifecycleChecks({
+ issues,
+ runs,
+ parentId: parent.id,
+ leadId: "lead",
+ allowedInterruptedRuns,
+ });
+ it("accepts successful owned native work", () =>
+ expect(score([run]).every((c) => c.passed)).toBe(true));
+ it.each([
+ ["legacy execution", { ...run, runtimeMode: "legacy" }, "native-runtime"],
+ [
+ "no runner identity",
+ { ...run, runnerInstanceId: null },
+ "native-runtime",
+ ],
+ ["worker on parent", { ...run, agentId: "worker" }, "parent-owned-by-lead"],
+ ["crash without recovery", { ...run, status: "failed" }, "successful-runs"],
+ ["unsettled run", { ...run, status: "running" }, "settled"],
+ ] as const)("rejects %s", (_label, bad, id) =>
+ expect(score([bad]).find((c) => c.id === id)?.passed).toBe(false),
+ );
+ it("does not exempt unrelated failures because another run was intentionally stopped", () => {
+ const checks = score(
+ [
+ { ...run, status: "cancelled" },
+ { ...run, id: "other", status: "failed" },
+ ],
+ [parent],
+ ["run"],
+ );
+ expect(checks.find((c) => c.id === "successful-runs")?.passed).toBe(false);
+ });
+ it("excludes a proven pre-dispatch workspace deferral without hiding executed failures", () => {
+ const deferred: StoryRun = {
+ id: "deferred",
+ companyId: "company",
+ agentId: "lead",
+ status: "cancelled",
+ errorCode: "workspace_busy",
+ resultJson: {
+ executionRecovery: {
+ kind: "workspace_wait",
+ providerWorkStarted: false,
+ },
+ },
+ };
+ expect(score([run, deferred]).every((c) => c.passed)).toBe(true);
+ expect(
+ score([run, { ...deferred, processPid: 123 }]).every((c) => c.passed),
+ ).toBe(false);
+ expect(
+ score([run, { ...deferred, resultJson: {} }]).every((c) => c.passed),
+ ).toBe(false);
+ });
+ it("rejects a finished answer left in review", () =>
+ expect(
+ score([run], [{ ...parent, status: "in_review" }]).find(
+ (c) => c.id === "tasks-done",
+ )?.passed,
+ ).toBe(false));
+});
+
+describe("reply completion boundary", () => {
+ const run: StoryRun = {
+ id: "old",
+ companyId: "company",
+ agentId: "lead",
+ status: "succeeded",
+ runnerProfileJson: {
+ nativeExecutionInput: { task: { prompt: "Original task" } },
+ },
+ };
+ it("does not accept Done from the first run while a later request is still queued", () => {
+ expect(storyRepliesConsumed([run], ["later-comment"])).toBe(false);
+ const next = {
+ ...run,
+ id: "new",
+ runnerProfileJson: {
+ nativeExecutionInput: { task: { prompt: "User reply later-comment" } },
+ },
+ };
+ expect(
+ storyRepliesConsumed(
+ [run, { ...next, status: "running" }],
+ ["later-comment"],
+ ),
+ ).toBe(false);
+ expect(storyRepliesConsumed([run, next], ["later-comment"])).toBe(true);
+ });
+});
+
+describe("delegation completion order", () => {
+ it("rejects a parent closed before its worker finishes", () => {
+ const base: StoryRun = {
+ id: "lead-run",
+ companyId: "company",
+ agentId: "lead",
+ status: "succeeded",
+ nativeIssueId: "parent",
+ finishedAt: "2026-09-14T12:00:00Z",
+ };
+ const child = {
+ ...base,
+ id: "worker-run",
+ agentId: "worker",
+ nativeIssueId: "child",
+ finishedAt: "2026-09-14T12:01:00Z",
+ };
+ expect(
+ storyParentFinishedAfterChildren([base, child], "parent", "lead", [
+ "child",
+ ]),
+ ).toBe(false);
+ expect(
+ storyParentFinishedAfterChildren(
+ [{ ...base, finishedAt: "2026-09-14T12:02:00Z" }, child],
+ "parent",
+ "lead",
+ ["child"],
+ ),
+ ).toBe(true);
+ expect(
+ storyParentFinishedAfterChildren([base], "parent", "lead", ["child"]),
+ ).toBe(false);
+ });
+});
+
+describe("cancelled queued workspace retry", () => {
+ const queued: StoryRun = {
+ id: "retry",
+ companyId: "company",
+ agentId: "agent",
+ status: "cancelled",
+ runtimeMode: "legacy",
+ runtimeModeResolvedAt: null,
+ startedAt: null,
+ retryOfRunId: "workspace-deferral",
+ scheduledRetryReason: "workspace_busy",
+ errorCode: "cancelled",
+ lastOutputSeq: 0,
+ resultJson: {
+ startupCancellation: {
+ requestedAt: "2026-09-14T19:10:23.057Z",
+ beforeNativeSelection: false,
+ },
+ },
+ };
+ it("does not mistake an unstarted cancelled retry for provider execution", () => {
+ expect(isStoryWorkspaceDeferral(queued)).toBe(true);
+ });
+ it.each([
+ { startedAt: "2026-09-14T19:10:22Z" },
+ { processPid: 42 },
+ { runnerInstanceId: "runner" },
+ { nativeSessionId: "session" },
+ { lastOutputSeq: 1 },
+ { usageJson: { outputTokens: 1 } },
+ { runtimeModeResolvedAt: "2026-09-14T19:10:22Z" },
+ { scheduledRetryReason: "other" },
+ { resultJson: null },
+ ])("retains a contradictory or unproven cancellation %j", (change) => {
+ expect(isStoryWorkspaceDeferral({ ...queued, ...change })).toBe(false);
+ });
+});
diff --git a/tests/runner-e2e/evidence.ts b/tests/runner-e2e/evidence.ts
index e09b027ece..25eef919e4 100644
--- a/tests/runner-e2e/evidence.ts
+++ b/tests/runner-e2e/evidence.ts
@@ -38,6 +38,8 @@ const ALLOWED_ROOT_FILES = new Set([
"result.json",
"final-state.png",
"failure.png",
+ "tool-review-pending.png",
+ "decision-pending.png",
"chat-plan-draft.png",
"chat-plan-revised.png",
"server.log",
diff --git a/tests/runner-e2e/failure-classifier.ts b/tests/runner-e2e/failure-classifier.ts
index 1dca091cb7..f17f0fc543 100644
--- a/tests/runner-e2e/failure-classifier.ts
+++ b/tests/runner-e2e/failure-classifier.ts
@@ -1,3 +1,4 @@
+import { ObservedStateTimeout } from "./api.js";
import type { FailureClass } from "./types.js";
const TRANSIENT =
@@ -16,6 +17,7 @@ const CANDIDATE =
/(?:matcher|expected.*observed|marker|issue status|run status|runtime mode|wrong output|missing output)/i;
export function classifyFailure(error: unknown): FailureClass {
+ if (error instanceof ObservedStateTimeout) return error.failureClass;
const message =
error instanceof Error ? `${error.name}: ${error.message}` : String(error);
if (/browser bootstrap failed before task creation/i.test(message))
diff --git a/tests/runner-e2e/harness-env.ts b/tests/runner-e2e/harness-env.ts
index 8d5691e9fe..d37a3964c8 100644
--- a/tests/runner-e2e/harness-env.ts
+++ b/tests/runner-e2e/harness-env.ts
@@ -105,6 +105,11 @@ export function buildRunnerE2EProcessEnvironment(
): NodeJS.ProcessEnv {
const result = { ...source };
delete result.OPENCODE_ALLOW_ALL_MODELS;
+ // Hiring needs the opt-in native API surface. Scope this to the explicit
+ // manual hiring story; production and other suites retain their defaults.
+ if (executions.some((e) => e.suite.id === "everyday-workflows" && e.task.id === "hire-reuse")) {
+ result.PAPERCLIP_RUNNER_API_TOOLS_ENABLED = "true";
+ }
if (
executions.length > 0 &&
executions.every(
diff --git a/tests/runner-e2e/history.test.ts b/tests/runner-e2e/history.test.ts
index c0dd6f1d60..47d83f4cd4 100644
--- a/tests/runner-e2e/history.test.ts
+++ b/tests/runner-e2e/history.test.ts
@@ -195,8 +195,8 @@ describe("runner E2E campaign history", () => {
expect(index).toContain("Runner E2E campaigns");
expect(index).toContain("complete-green");
expect(index).toContain("complete-red");
- expect(index).toContain("92/92 passed");
- expect(index).toContain("91/92 passed");
+ expect(index).toContain(`${runnerMatrix.length}/${runnerMatrix.length} passed`);
+ expect(index).toContain(`${runnerMatrix.length - 1}/${runnerMatrix.length} passed`);
expect(index).toContain("Open report →");
expect(index).toContain(
"campaigns/complete-red/public-images/campaign-summary.png",
diff --git a/tests/runner-e2e/launch.ts b/tests/runner-e2e/launch.ts
index 5c59fd24cf..40b177fd3d 100644
--- a/tests/runner-e2e/launch.ts
+++ b/tests/runner-e2e/launch.ts
@@ -33,7 +33,7 @@ import {
assertSecretFree,
findSecretLeakInDirectory,
isEphemeralCodexRuntimeAuthFile,
- isEphemeralPostgresPidFile,
+ isEphemeralPostgresScanFile,
normalizedSecrets,
sanitizeJson,
} from "./redaction.js";
@@ -791,7 +791,7 @@ async function runAttempt(input: {
includeShapes: false,
ignoreFile: (file) => expectedEphemeralCredentials.has(file),
allowDisappearedFile: (file) =>
- label === "Paperclip home" && isEphemeralPostgresPidFile(paperclipHome, file),
+ label === "Paperclip home" && isEphemeralPostgresScanFile(paperclipHome, file),
});
if (!leak) break;
const isManagedCodexRuntimeAuth =
diff --git a/tests/runner-e2e/live-fixtures.test.ts b/tests/runner-e2e/live-fixtures.test.ts
index a919f07c2b..9c2eca9e15 100644
--- a/tests/runner-e2e/live-fixtures.test.ts
+++ b/tests/runner-e2e/live-fixtures.test.ts
@@ -4,6 +4,71 @@ import { runnerMatrix } from "./catalog.js";
import { setupLiveFixtures } from "./live-fixtures.js";
describe("live runner fixtures", () => {
+ it.each(["runner-codex", "runner-acpx-claude"])(
+ "gives %s hiring fixtures a personal managed account without env overrides",
+ async (profile) => {
+ const execution = runnerMatrix.find(
+ (e) =>
+ e.suite.id === "everyday-workflows" &&
+ e.task.id === "hire-reuse" &&
+ e.profile.id === profile &&
+ e.environment.id === "local",
+ )!;
+ const provider =
+ profile === "runner-acpx-claude" ? "anthropic" : "openai";
+ let connected = false;
+ let agentBody: any;
+ const api = {
+ async post(url: string, data: any) {
+ if (url === "/api/companies") return { id: "company", name: "Test" };
+ if (url.endsWith("/agents")) {
+ agentBody = data;
+ return { id: "lead", ...data };
+ }
+ throw new Error(`Unexpected POST ${url}`);
+ },
+ async postSensitive(url: string, data: any) {
+ if (url.endsWith("/ai-connections")) {
+ expect(data).toMatchObject({
+ provider,
+ method: "api_key",
+ ownership: "personal",
+ apiKey: "test-value",
+ agentIds: [],
+ allAgents: false,
+ });
+ connected = true;
+ return { connectionId: "managed-account" };
+ }
+ return { id: "secret" };
+ },
+ async get() {
+ return [{ id: "local", driver: "local" }];
+ },
+ } as unknown as RunnerApi;
+ const fixtures = await setupLiveFixtures({
+ api,
+ execution,
+ executionNonce: "nonce",
+ workspacePath: "/tmp/test",
+ credentials: {
+ OPENAI_API_KEY: "test-value",
+ ANTHROPIC_API_KEY: "test-value",
+ },
+ });
+ expect(connected).toBe(true);
+ expect(agentBody.adapterConfig.env).toBeUndefined();
+ expect(agentBody.runtimeConfig.aiConnection).toEqual({
+ provider,
+ method: "api_key",
+ mode: "responsible_user",
+ });
+ expect((fixtures as any).aiConnection.connectionId).toBe(
+ "managed-account",
+ );
+ },
+ );
+
it("installs the Daytona provider through the public API before creating its environment", async () => {
const calls: string[] = [];
const api = {
diff --git a/tests/runner-e2e/live-fixtures.ts b/tests/runner-e2e/live-fixtures.ts
index de9c242a42..c211a5a1a1 100644
--- a/tests/runner-e2e/live-fixtures.ts
+++ b/tests/runner-e2e/live-fixtures.ts
@@ -32,6 +32,15 @@ interface AgentRecord {
name: string;
companyId: string;
}
+interface ManagedAccountFixture {
+ connectionId: string;
+ binding: {
+ provider: "openai" | "anthropic";
+ method: "api_key";
+ mode: "responsible_user";
+ };
+}
+
interface ProjectRecord {
id: string;
name: string;
@@ -47,6 +56,7 @@ export interface LiveFixtureValues {
environment: EnvironmentRecord;
agent: AgentRecord;
project?: ProjectRecord;
+ aiConnection?: ManagedAccountFixture;
teardown(): Promise;
}
@@ -195,22 +205,72 @@ export async function setupLiveFixtures(input: {
},
});
+ const managedHiring =
+ execution.suite.id === "everyday-workflows" &&
+ execution.task.id === "hire-reuse";
+ if (managedHiring) {
+ registry.register({
+ id: "ai-connection",
+ dependencies: ["company"],
+ async setup(resolved) {
+ const company = value(resolved, "company");
+ const provider =
+ execution.profile.provider === "acpx" ? "anthropic" : "openai";
+ const key =
+ provider === "anthropic" ? "ANTHROPIC_API_KEY" : "OPENAI_API_KEY";
+ const apiKey = input.credentials[key];
+ if (!apiKey) throw new Error(`Missing credential ${key}`);
+ const account = await api.postSensitive<{ connectionId: string }>(
+ `/api/companies/${company.id}/ai-connections`,
+ {
+ provider,
+ method: "api_key",
+ name: `Runner E2E account ${input.executionNonce}`,
+ ownership: "personal",
+ apiKey,
+ agentIds: [],
+ allAgents: false,
+ },
+ );
+ return {
+ connectionId: account.connectionId,
+ binding: { provider, method: "api_key", mode: "responsible_user" },
+ };
+ },
+ });
+ }
+
registry.register({
id: "agent",
- dependencies: ["company", "secrets", "environment"],
+ dependencies: [
+ "company",
+ "secrets",
+ "environment",
+ ...(managedHiring ? ["ai-connection"] : []),
+ ],
async setup(resolved) {
const company = value(resolved, "company");
const environment = value(resolved, "environment");
const secretRefs = value(resolved, "secrets");
+ const agent = execution.profile.buildAgent({
+ environmentId: environment.id,
+ environmentFixtureId: execution.environment.id,
+ workspacePath: input.workspacePath,
+ secretRefs,
+ executionId: input.executionNonce,
+ });
+ if (managedHiring) {
+ const account = value(resolved, "ai-connection");
+ const config = agent.adapterConfig as Record;
+ delete config.env;
+ agent.runtimeConfig = {
+ ...(agent.runtimeConfig as Record),
+ aiConnection: account.binding,
+ };
+ }
return api.post(
`/api/companies/${company.id}/agents`,
- execution.profile.buildAgent({
- environmentId: environment.id,
- environmentFixtureId: execution.environment.id,
- workspacePath: input.workspacePath,
- secretRefs,
- executionId: input.executionNonce,
- }),
+ agent,
);
},
async teardown() {
@@ -264,6 +324,14 @@ export async function setupLiveFixtures(input: {
...(setup.values.has("project")
? { project: value(setup.values, "project") }
: {}),
+ ...(managedHiring
+ ? {
+ aiConnection: value(
+ setup.values,
+ "ai-connection",
+ ),
+ }
+ : {}),
teardown: setup.teardown,
};
}
diff --git a/tests/runner-e2e/redaction.ts b/tests/runner-e2e/redaction.ts
index fa7a89eaf3..d6316a6f59 100644
--- a/tests/runner-e2e/redaction.ts
+++ b/tests/runner-e2e/redaction.ts
@@ -157,6 +157,14 @@ export function isEphemeralPostgresPidFile(paperclipHome: string, file: string):
return /^instances\/[^/]+\/db\/postmaster\.pid$/.test(relative);
}
+export function isEphemeralPostgresScanFile(paperclipHome: string, file: string): boolean {
+ const relative = path.relative(paperclipHome, file).split(path.sep).join("/");
+ // A relation can be unlinked during PostgreSQL shutdown/checkpoint. Existing
+ // files are always scanned; this predicate only permits ENOENT after readdir.
+ return isEphemeralPostgresPidFile(paperclipHome, file) ||
+ /^instances\/[^/]+\/db\/base\/\d+\/\d+(?:_(?:fsm|vm|init))?(?:\.\d+)?$/.test(relative);
+}
+
export async function findSecretLeakInDirectory(
root: string,
secrets: readonly string[],
diff --git a/tests/runner-e2e/result-validation.ts b/tests/runner-e2e/result-validation.ts
index 8bdadba75a..6999367f6e 100644
--- a/tests/runner-e2e/result-validation.ts
+++ b/tests/runner-e2e/result-validation.ts
@@ -226,6 +226,7 @@ const fields = {
label: string,
file: relativeFile,
publication: optional(oneOf("public-runner-fixture")),
+ sha256: optional(string),
}),
),
),
diff --git a/tests/runner-e2e/runner.spec.ts b/tests/runner-e2e/runner.spec.ts
index 4a56871316..5e810c0920 100644
--- a/tests/runner-e2e/runner.spec.ts
+++ b/tests/runner-e2e/runner.spec.ts
@@ -1,5 +1,7 @@
+import { runEverydayFlow } from "./everyday-flow.js";
+import { createTaskThroughUi, submitTaskReply } from "./user-actions.js";
import { runChatFlow } from "./chat-flow.js";
-import { randomBytes } from "node:crypto";
+import { createHash, randomBytes } from "node:crypto";
import { mkdir, readFile, rename, writeFile } from "node:fs/promises";
import path from "node:path";
import { expect, test, type Page } from "@playwright/test";
@@ -208,6 +210,7 @@ async function restartIsolatedPaperclipServer(input: {
await pollUntil({
label: `isolated server restart ${input.requestId}`,
+ timeoutFailureClass: "transient_infrastructure",
deadlineAt: input.deadlineAt,
intervalMs: 250,
load: async () => {
@@ -230,6 +233,7 @@ async function restartIsolatedPaperclipServer(input: {
});
await pollUntil({
label: `replacement server health ${input.requestId}`,
+ timeoutFailureClass: "transient_infrastructure",
deadlineAt: input.deadlineAt,
intervalMs: 250,
load: () => input.api.get>("/api/health"),
@@ -298,88 +302,6 @@ async function writeSanitizedJson(
await writeFile(path.join(directory, name), safe, "utf8");
}
-async function createTaskThroughUi(input: {
- page: Page;
- issuePrefix: string;
- agentName: string;
- title: string;
- prompt: string;
- workMode: "standard" | "planning" | "ask";
- projectName?: string;
-}) {
- const issuesUrl = `/${encodeURIComponent(input.issuePrefix)}/issues`;
- const newTask = input.page.getByRole("button", { name: "New Task" }).first();
- let bootstrapError: unknown;
- for (let bootstrapAttempt = 1; bootstrapAttempt <= 3; bootstrapAttempt += 1) {
- try {
- await input.page.goto(issuesUrl, {
- waitUntil: "domcontentloaded",
- timeout: 30_000,
- });
- await newTask.waitFor({ state: "visible", timeout: 20_000 });
- bootstrapError = undefined;
- break;
- } catch (error) {
- bootstrapError = error;
- if (bootstrapAttempt < 3) await input.page.waitForTimeout(1_000);
- }
- }
- if (bootstrapError) {
- throw new Error(
- `Browser bootstrap failed before task creation: ${bootstrapError instanceof Error ? bootstrapError.message : String(bootstrapError)}`,
- { cause: bootstrapError },
- );
- }
- await newTask.click();
- await input.page.getByPlaceholder("Task title").fill(input.title);
- await input.page
- .getByRole("dialog")
- .getByRole("textbox", { name: "editable markdown", exact: true })
- .fill(input.prompt);
- if (input.workMode !== "standard") {
- await input.page
- .getByRole("dialog")
- .locator(`[data-issue-work-mode-chip="standard"]`)
- .click();
- await input.page
- .locator(`[data-issue-work-mode="${input.workMode}"]`)
- .click();
- }
- await input.page
- .getByRole("button", { name: "Assignee", exact: true })
- .click();
- await input.page
- .getByPlaceholder("Search assignees...")
- .fill(input.agentName);
- await input.page.getByText(input.agentName, { exact: true }).last().click();
- if (input.projectName) {
- const dialog = input.page.getByRole("dialog");
- // Selecting the assignee advances focus to this selector and opens it.
- // Focus is idempotent here; clicking would toggle an already-open popover
- // closed before the search field can be filled.
- await dialog.getByRole("button", { name: "Project", exact: true }).focus();
- await dialog.getByPlaceholder("Search projects...").fill(input.projectName);
- await dialog.getByText(input.projectName, { exact: true }).last().click();
- }
- const submittedAtMs = Date.now();
- await input.page
- .getByRole("button", { name: "Create Task", exact: true })
- .click();
- return submittedAtMs;
-}
-
-async function submitTaskReply(page: Page, body: string): Promise {
- const composer = page.getByTestId("task-chat-composer-input").last();
- await expect(composer).toBeVisible({ timeout: 30_000 });
- await composer
- .locator('[contenteditable="true"], textarea')
- .first()
- .fill(body);
- const submittedAtMs = Date.now();
- await page.getByTestId("task-chat-composer-send").last().click();
- return submittedAtMs;
-}
-
async function submitTaskRevision(page: Page, body: string): Promise {
const revise = page
.getByRole("button", { name: "Continue work", exact: true })
@@ -603,6 +525,7 @@ for (const execution of executions) {
const credentials = credentialValues();
const secrets = normalizedSecrets(Object.values(credentials));
const api = new RunnerApi(request);
+ const companyRunFlow = ["agent_chat", "everyday_workflow"].includes(execution.task.flow);
const consoleDiagnostics: Array> = [];
const networkDiagnostics: Array> = [];
let fixtures: LiveFixtureValues | undefined;
@@ -651,6 +574,9 @@ for (const execution of executions) {
label,
file,
publication: PUBLIC_RUNNER_SCREENSHOT_MARKER,
+ sha256: createHash("sha256")
+ .update(await readFile(path.join(privateDir, file)))
+ .digest("hex"),
});
};
@@ -672,10 +598,10 @@ for (const execution of executions) {
};
const cancelActiveRunsForCleanup = async () => {
- if (!issue && !(execution.task.flow === "agent_chat" && fixtures)) return;
+ if (!issue && !(companyRunFlow && fixtures)) return;
const cleanupIssueId = issue?.id;
const runs = await api.get(
- execution.task.flow === "agent_chat" && fixtures ? `/api/companies/${fixtures.company.id}/heartbeat-runs?limit=100` : `/api/issues/${cleanupIssueId}/runs`,
+ companyRunFlow && fixtures ? `/api/companies/${fixtures.company.id}/heartbeat-runs?limit=100` : `/api/issues/${cleanupIssueId}/runs`,
);
const activeRunIds = [
...new Set(
@@ -696,7 +622,7 @@ for (const execution of executions) {
await pollUntil({
label: `cleanup cancellation for issue ${cleanupIssueId}`,
deadlineAt: Date.now() + 45_000,
- load: () => api.get(execution.task.flow === "agent_chat" && fixtures ? `/api/companies/${fixtures.company.id}/heartbeat-runs?limit=100` : `/api/issues/${cleanupIssueId}/runs`),
+ load: () => api.get(companyRunFlow && fixtures ? `/api/companies/${fixtures.company.id}/heartbeat-runs?limit=100` : `/api/issues/${cleanupIssueId}/runs`),
accept: (currentRuns) =>
currentRuns
.filter((run) => activeIds.has(run.id))
@@ -726,7 +652,7 @@ for (const execution of executions) {
capture(() => api.get(`/api/issues/${issue!.id}`)),
capture(() =>
api.get(
- execution.task.flow === "agent_chat"
+ companyRunFlow
? `/api/companies/${fixtures!.company.id}/heartbeat-runs?limit=100`
: `/api/companies/${fixtures!.company.id}/heartbeat-runs?agentId=${fixtures!.agent.id}&limit=20`,
),
@@ -743,7 +669,7 @@ for (const execution of executions) {
),
]);
const taskRuns = Array.isArray(listedRuns)
- ? execution.task.flow === "agent_chat" ? listedRuns : matchingRuns(listedRuns, "id" in currentIssue ? currentIssue : issue)
+ ? companyRunFlow ? listedRuns : matchingRuns(listedRuns, "id" in currentIssue ? currentIssue : issue)
: [];
const detailedRuns = await Promise.all(
taskRuns.map((candidate) =>
@@ -815,7 +741,7 @@ for (const execution of executions) {
enableNativeRunner: boolean;
}>("/api/instance/settings/experimental", {
enableNativeRunner: true,
- ...(execution.task.flow === "warm_three_turn"
+ ...(["warm_three_turn", "everyday_workflow"].includes(execution.task.flow)
? { enableIsolatedWorkspaces: true }
: {}),
...(execution.profile.generation === "native" &&
@@ -858,7 +784,18 @@ for (const execution of executions) {
secrets,
);
- if (execution.task.flow === "agent_chat") {
+ if (execution.task.flow === "everyday_workflow") {
+ const story = await runEverydayFlow({
+ page, api, fixtures, execution, nonce, workspacePath, privateDir,
+ deadlineAt: startedAtMs + deadlineMs,
+ restart: () => restartIsolatedPaperclipServer({ api, requestId: `story-${nonce}`, deadlineAt: startedAtMs + deadlineMs }),
+ observe: (storyIssue, storyRuns) => { issue = storyIssue; selectedRuns = storyRuns; },
+ capture: captureScreenshot,
+ evidence: (name, data) => writeSanitizedJson(snapshotsDir, name, data, secrets),
+ });
+ issue = story.issue; selectedRuns = story.runs;
+ matcherResults = story.evidence.checks.map(check => ({ matcher: { kind: "json_path" as const, path: check.id, expected: true }, passed: check.passed, detail: check.detail }));
+ } else if (execution.task.flow === "agent_chat") {
const chat = await runChatFlow({
page, api, fixtures, execution, nonce,
restart: () => restartIsolatedPaperclipServer({ api, requestId: `chat-${nonce}`, deadlineAt: startedAtMs + deadlineMs }),
@@ -934,7 +871,7 @@ for (const execution of executions) {
const [currentIssue, runs, comments, interactions] = await Promise.all([
api.get(`/api/issues/${issue!.id}`),
api.get(
- execution.task.flow === "agent_chat"
+ companyRunFlow
? `/api/companies/${fixtures!.company.id}/heartbeat-runs?limit=100`
: `/api/companies/${fixtures!.company.id}/heartbeat-runs?agentId=${fixtures!.agent.id}&limit=20`,
),
@@ -2465,7 +2402,7 @@ for (const execution of executions) {
});
try {
await cancelActiveRunsForCleanup();
- if (execution.task.flow === "agent_chat") {
+ if (companyRunFlow) {
const companyRuns = await api.get(`/api/companies/${fixtures.company.id}/heartbeat-runs?limit=100`);
selectedRuns = await Promise.all(companyRuns.map(run => api.get(`/api/heartbeat-runs/${run.id}`)));
await writeSanitizedJson(snapshotsDir, "chat-final-run-ledger.json", selectedRuns, secrets);
@@ -2486,7 +2423,9 @@ for (const execution of executions) {
: (priorFailureClass ?? cleanupFailureClass);
primaryError = new AggregateError(
[primaryError, error].filter(Boolean),
- `Cleanup failed after ${primaryError ? "test failure" : "test execution"}: ${error instanceof Error ? error.message : String(error)}`,
+ primaryError
+ ? `${primaryError instanceof Error ? primaryError.message : String(primaryError)}; Cleanup also failed: ${error instanceof Error ? error.message : String(error)}`
+ : `Cleanup failed after test execution: ${error instanceof Error ? error.message : String(error)}`,
);
}
if (runtimeLeases.length > 0) {
diff --git a/tests/runner-e2e/selectors.ts b/tests/runner-e2e/selectors.ts
index d5bf2cdd5c..f7e9868a89 100644
--- a/tests/runner-e2e/selectors.ts
+++ b/tests/runner-e2e/selectors.ts
@@ -157,6 +157,7 @@ export function selectRunnerExecutions(
const selected = matrix.filter((execution) => {
if (options.ids.length > 0) return options.ids.includes(execution.id);
+ if (execution.suite.manualOnly && !options.suites.includes(execution.suite.id) && !options.list) return false;
if (
options.all ||
(options.list &&
diff --git a/tests/runner-e2e/support.test.ts b/tests/runner-e2e/support.test.ts
index d46d1ec1cf..9e48f68ea2 100644
--- a/tests/runner-e2e/support.test.ts
+++ b/tests/runner-e2e/support.test.ts
@@ -26,6 +26,7 @@ import {
findSecretLeakInDirectory,
isEphemeralCodexRuntimeAuthFile,
isEphemeralPostgresPidFile,
+ isEphemeralPostgresScanFile,
redactText,
sanitizeJson,
} from "./redaction.js";
@@ -175,6 +176,16 @@ describe("runner E2E provider environment", () => {
});
});
+describe("hiring capability opt-in", () => {
+ it("enables API tools only when the manual hiring story is selected", () => {
+ const hire = runnerMatrix.find((e) => e.suite.id === "everyday-workflows" && e.task.id === "hire-reuse")!;
+ const delegate = runnerMatrix.find((e) => e.suite.id === "everyday-workflows" && e.task.id === "delegate-feedback")!;
+ expect(buildRunnerE2EProcessEnvironment({}, [hire]).PAPERCLIP_RUNNER_API_TOOLS_ENABLED).toBe("true");
+ expect(buildRunnerE2EProcessEnvironment({}, [delegate]).PAPERCLIP_RUNNER_API_TOOLS_ENABLED).toBeUndefined();
+ expect(buildRunnerE2EProcessEnvironment({}, []).PAPERCLIP_RUNNER_API_TOOLS_ENABLED).toBeUndefined();
+ });
+});
+
describe("runner E2E server port allocation", () => {
it("rejects direct and derived embedded-Postgres collisions", () => {
expect(runnerE2EServerPortConflictsWithDatabase(44_329)).toBe(true);
@@ -980,6 +991,21 @@ describe("runner E2E evidence redaction", () => {
expect(isEphemeralPostgresPidFile(root, path.join(root, "instances", "test", "db", "records.bin"))).toBe(false);
});
+ it("handles a removed PostgreSQL relation but scans existing relation bytes", async () => {
+ const root = await mkdtemp(path.join(os.tmpdir(), "runner-e2e-postgres-relation-race-"));
+ cleanupDirectories.push(root);
+ const relation = path.join(root, "instances", "test", "db", "base", "16384", "16824");
+ const persisted = path.join(path.dirname(relation), "16825");
+ await mkdir(path.dirname(relation), { recursive: true });
+ await writeFile(relation, "old relation"); await writeFile(persisted, secret);
+ await expect(findSecretLeakInDirectory(root, [secret], {
+ ignoreFile: file => { if(file === relation)unlinkSync(file); return false; },
+ allowDisappearedFile: file => isEphemeralPostgresScanFile(root, file),
+ })).resolves.toEqual({file:persisted,reason:"exact secret value"});
+ expect(isEphemeralPostgresScanFile(root,path.join(root,"workspace","base","16384","16824"))).toBe(false);
+ expect(isEphemeralPostgresScanFile(root,path.join(root,"instances","test","db","base","records.json"))).toBe(false);
+ });
+
it("still detects secrets in an existing PostgreSQL PID file", async () => {
const root = await mkdtemp(path.join(os.tmpdir(), "runner-e2e-postgres-pid-secret-"));
cleanupDirectories.push(root);
diff --git a/tests/runner-e2e/test_everyday_artifact.py b/tests/runner-e2e/test_everyday_artifact.py
new file mode 100644
index 0000000000..a72d002ae6
--- /dev/null
+++ b/tests/runner-e2e/test_everyday_artifact.py
@@ -0,0 +1,91 @@
+"""Calibrate the independent oracle with correct and deliberately broken deliveries."""
+import importlib.util
+import pathlib
+import stat
+import socket
+import tempfile
+import unittest
+import zipfile
+
+spec = importlib.util.spec_from_file_location("oracle", pathlib.Path(__file__).with_name("everyday-artifact.py"))
+oracle = importlib.util.module_from_spec(spec)
+spec.loader.exec_module(oracle)
+
+GOOD = '''import argparse,re
+def slugify(text): return re.sub(r'[^a-z0-9]+','-',text.strip().lower()).strip('-')
+if __name__ == '__main__':
+ p=argparse.ArgumentParser();p.add_argument('text');p.add_argument('--separator',choices=['-','_'],default='-');p.add_argument('--max-length',type=int)
+ a=p.parse_args()
+ if a.max_length is not None and a.max_length<=0:p.error('positive length required')
+ value=slugify(a.text).replace('-',a.separator)
+ if a.max_length is not None:value=value[:a.max_length].rstrip(a.separator)
+ print(value)
+'''
+
+class ArtifactOracleTests(unittest.TestCase):
+ def grade(self, source=GOOD, mode='base', extras=None):
+ with tempfile.TemporaryDirectory() as tmp:
+ archive=pathlib.Path(tmp)/'project.zip'
+ with zipfile.ZipFile(archive,'w') as z:
+ z.writestr('project/slugify.py',source)
+ z.writestr('project/README.md','Usage: python3 slugify.py "Hello World"')
+ # Deliberately passing but useless agent-authored tests must not determine our verdict.
+ z.writestr('project/test_slugify.py','assert True')
+ for name,content in (extras or {}).items():z.writestr(name,content)
+ return oracle.inspect(archive,mode)
+
+ def test_correct_delivery_passes_all_modes(self):
+ for mode in ['base','separator','max-length']:
+ with self.subTest(mode=mode):self.assertTrue(all(c['passed'] for c in self.grade(mode=mode)))
+
+ def test_wrong_output_fails_despite_agent_tests(self):
+ checks=self.grade(GOOD.replace("print(value)","print('hello-world')"))
+ self.assertFalse(all(c['passed'] for c in checks))
+
+ def test_missing_late_requirement_fails(self):
+ source=GOOD.replace("if a.max_length is not None:value=value[:a.max_length].rstrip(a.separator)","if False:pass")
+ self.assertFalse(all(c['passed'] for c in self.grade(source,'max-length')))
+
+ def test_trailing_separator_bug_fails(self):
+ checks=self.grade(GOOD.replace(".rstrip(a.separator)",""),'max-length')
+ self.assertFalse(next(c['passed'] for c in checks if c['id']=='length-6'))
+
+ def test_invalid_separator_bug_fails(self):
+ checks=self.grade(GOOD.replace("choices=['-','_'],", ""),'separator')
+ self.assertFalse(next(c['passed'] for c in checks if c['id']=='reject-invalid-separator'))
+
+ def test_artifact_cannot_read_host_files(self):
+ with tempfile.TemporaryDirectory() as tmp:
+ marker=pathlib.Path(tmp)/'host-private-marker';marker.write_text('private fixture')
+ prefix=f"import pathlib\nassert not pathlib.Path({str(marker)!r}).exists(), 'host filesystem is visible'\n"
+ self.assertTrue(all(c['passed'] for c in self.grade(prefix+GOOD)))
+
+ def test_artifact_cannot_reach_host_loopback(self):
+ with socket.socket() as server:
+ server.bind(('127.0.0.1',0));server.listen(64)
+ port=server.getsockname()[1]
+ prefix=f"import socket\ns=socket.socket();s.settimeout(0.2)\nassert s.connect_ex(('127.0.0.1',{port})) != 0, 'host network is visible'\ns.close()\n"
+ self.assertTrue(all(c['passed'] for c in self.grade(prefix+GOOD)))
+
+ def test_artifact_cannot_modify_the_read_only_delivery(self):
+ prefix="import pathlib\ntry: pathlib.Path(__file__).write_text('changed')\nexcept OSError: pass\nelse: raise AssertionError('project is writable')\n"
+ self.assertTrue(all(c['passed'] for c in self.grade(prefix+GOOD)))
+
+ def test_generated_output_is_bounded(self):
+ with self.assertRaisesRegex(ValueError, 'output limit'):
+ self.grade("print('x'*1000000)")
+
+ def test_duplicate_source_fails(self):
+ self.assertFalse(all(c['passed'] for c in self.grade(extras={'other/slugify.py':GOOD})))
+
+ def test_path_traversal_is_rejected_before_execution(self):
+ for name in ['../escape','/absolute','back\\slash']:
+ with self.subTest(name=name),self.assertRaisesRegex(ValueError,'Unsafe archive'):
+ self.grade(extras={name:'bad'})
+
+ def test_symlink_is_rejected(self):
+ member=zipfile.ZipInfo('link');member.external_attr=(stat.S_IFLNK|0o777)<<16
+ with self.assertRaisesRegex(ValueError,'Unsafe archive'):
+ self.grade(extras={member:'/tmp/target'})
+
+if __name__=='__main__':unittest.main()
diff --git a/tests/runner-e2e/types.ts b/tests/runner-e2e/types.ts
index 83fd47f7a6..27948b41aa 100644
--- a/tests/runner-e2e/types.ts
+++ b/tests/runner-e2e/types.ts
@@ -10,6 +10,7 @@ export type RunnerGeneration = "legacy" | "native";
export type RunnerEnvironmentId = "local" | "daytona";
export type RunnerTaskWorkMode = "standard" | "planning" | "ask";
export type RunnerTaskFlow =
+ | "everyday_workflow"
| "agent_chat"
| "governed_tool_review"
| "single_turn"
@@ -123,8 +124,8 @@ export interface RunnerTaskFixture {
expectedRunCount: number;
attemptTimeoutMs: Readonly>;
expectedTerminalState: {
- issue: "done" | "in_review";
- run: "succeeded";
+ issue: "done" | "in_review" | "blocked";
+ run: "succeeded" | "failed";
};
buildTitle(nonce: string): string;
buildPrompt(nonce: string): string;
@@ -168,6 +169,8 @@ export interface RunnerSuiteFixture {
excludedExecutionIds?: readonly string[];
expectedMatrixSize: number;
definitionMetadata?: Readonly>;
+ /** Requires an explicit suite or execution ID; excluded from scheduled --all. */
+ manualOnly?: boolean;
}
export interface MatrixJob {
@@ -296,6 +299,8 @@ export interface RunnerE2EResult {
label: string;
file: string;
publication?: "public-runner-fixture";
+ /** Absent in historical results; new captures bind the exact PNG bytes. */
+ sha256?: string;
}>;
cleanup: "not_started" | "passed" | "failed";
}
diff --git a/tests/runner-e2e/user-actions.ts b/tests/runner-e2e/user-actions.ts
new file mode 100644
index 0000000000..70c88b4127
--- /dev/null
+++ b/tests/runner-e2e/user-actions.ts
@@ -0,0 +1,86 @@
+import { expect, type Page } from "@playwright/test";
+
+export async function createTaskThroughUi(input: {
+ page: Page;
+ issuePrefix: string;
+ agentName: string;
+ title: string;
+ prompt: string;
+ workMode: "standard" | "planning" | "ask";
+ projectName?: string;
+}) {
+ const issuesUrl = `/${encodeURIComponent(input.issuePrefix)}/issues`;
+ const newTask = input.page.getByRole("button", { name: "New Task" }).first();
+ let bootstrapError: unknown;
+ for (let bootstrapAttempt = 1; bootstrapAttempt <= 3; bootstrapAttempt += 1) {
+ try {
+ await input.page.goto(issuesUrl, {
+ waitUntil: "domcontentloaded",
+ timeout: 30_000,
+ });
+ await newTask.waitFor({ state: "visible", timeout: 20_000 });
+ bootstrapError = undefined;
+ break;
+ } catch (error) {
+ bootstrapError = error;
+ if (bootstrapAttempt < 3) await input.page.waitForTimeout(1_000);
+ }
+ }
+ if (bootstrapError) {
+ throw new Error(
+ `Browser bootstrap failed before task creation: ${bootstrapError instanceof Error ? bootstrapError.message : String(bootstrapError)}`,
+ { cause: bootstrapError },
+ );
+ }
+ await newTask.click();
+ await input.page.getByPlaceholder("Task title").fill(input.title);
+ await input.page
+ .getByRole("dialog")
+ .getByRole("textbox", { name: "editable markdown", exact: true })
+ .fill(input.prompt);
+ if (input.workMode !== "standard") {
+ await input.page
+ .getByRole("dialog")
+ .locator(`[data-issue-work-mode-chip="standard"]`)
+ .click();
+ await input.page
+ .locator(`[data-issue-work-mode="${input.workMode}"]`)
+ .click();
+ }
+ await input.page
+ .getByRole("button", { name: "Assignee", exact: true })
+ .click();
+ await input.page
+ .getByPlaceholder("Search assignees...")
+ .fill(input.agentName);
+ await input.page.getByText(input.agentName, { exact: true }).last().click();
+ if (input.projectName) {
+ const dialog = input.page.getByRole("dialog");
+ // Selecting the assignee advances focus to this selector and opens it.
+ // Focus is idempotent here; clicking would toggle an already-open popover
+ // closed before the search field can be filled.
+ await dialog.getByRole("button", { name: "Project", exact: true }).focus();
+ await dialog.getByPlaceholder("Search projects...").fill(input.projectName);
+ await dialog.getByText(input.projectName, { exact: true }).last().click();
+ }
+ const submittedAtMs = Date.now();
+ await input.page
+ .getByRole("button", { name: "Create Task", exact: true })
+ .click();
+ return submittedAtMs;
+}
+
+export async function submitTaskReply(
+ page: Page,
+ body: string,
+): Promise {
+ const composer = page.getByTestId("task-chat-composer-input").last();
+ await expect(composer).toBeVisible({ timeout: 30_000 });
+ await composer
+ .locator('[contenteditable="true"], textarea')
+ .first()
+ .fill(body);
+ const submittedAtMs = Date.now();
+ await page.getByTestId("task-chat-composer-send").last().click();
+ return submittedAtMs;
+}
diff --git a/tests/runner-e2e/workflow-security.test.ts b/tests/runner-e2e/workflow-security.test.ts
index 6f58700018..bc20ac733f 100644
--- a/tests/runner-e2e/workflow-security.test.ts
+++ b/tests/runner-e2e/workflow-security.test.ts
@@ -3,8 +3,8 @@ import path from "node:path";
import { describe, expect, it } from "vitest";
const repositoryRoot = path.resolve(import.meta.dirname, "../..");
-const ordinaryPrTrustedWorkflowRevision =
- "03609aa6ecc9a047ed53d6b6469d8be554fbc46d";
+// PR #13470 uses the code-owner-reviewed default branch for this first-party workflow.
+const ordinaryPrTrustedWorkflowRevision = "master";
const fullStackTestNeeds =
/needs:\s*\[\s*authorize,\s*target_lock,\s*catalog,\s*daytona_image,\s*build_runner_artifacts,\s*build_remote_provider_pack,?\s*\]/u;
const buildRunnerNeeds =
@@ -13,14 +13,14 @@ const buildRemoteProviderPackNeeds =
/needs:\s*\[\s*authorize,\s*target_lock,\s*catalog,\s*daytona_image,\s*build_runner_artifacts,?\s*\]/u;
describe("public repository paid workflow security", () => {
- it("pins ordinary PR CI to the trusted Node-before-pnpm workflow", async () => {
+ it("uses the reviewed master branch for the first-party trusted PR workflow", async () => {
const ordinaryPrWorkflow = await readFile(
path.join(repositoryRoot, ".github/workflows/pr.yml"),
"utf8",
);
const trustedWorkflowCalls = [
...ordinaryPrWorkflow.matchAll(
- /^\s+uses:\s+(paperclipai\/paperclip\/\.github\/workflows\/pr-trusted\.yml)@([0-9a-f]{40})$/gmu,
+ /^\s+uses:\s+(paperclipai\/paperclip\/\.github\/workflows\/pr-trusted\.yml)@([^\s#]+)$/gmu,
),
];
@@ -77,7 +77,8 @@ describe("public repository paid workflow security", () => {
},
{
name: "pr-trusted.yml",
- expectedCachedSetupNodeSteps: 8,
+ // PR #13300 restores shared stores directly without setup-node cache writes.
+ expectedCachedSetupNodeSteps: 0,
},
];
@@ -117,7 +118,7 @@ describe("public repository paid workflow security", () => {
);
}
- expect(workflow.match(/^\s+cache: pnpm$/gmu), name).toHaveLength(
+ expect(workflow.match(/^\s+cache: pnpm$/gmu) ?? [], name).toHaveLength(
expectedCachedSetupNodeSteps,
);
}
diff --git a/tests/runner-e2e/workflow-timeout.test.ts b/tests/runner-e2e/workflow-timeout.test.ts
new file mode 100644
index 0000000000..ddc42c7a2d
--- /dev/null
+++ b/tests/runner-e2e/workflow-timeout.test.ts
@@ -0,0 +1,49 @@
+import { describe, expect, it, vi } from "vitest";
+import { pollUntil } from "./api.js";
+import { classifyFailure } from "./failure-classifier.js";
+
+describe("workflow timeout classification", () => {
+ it("does not classify observed task data as an infrastructure error", async () => {
+ vi.useFakeTimers();
+ try {
+ const pending = pollUntil({
+ label: "everyday hire-reuse settled",
+ deadlineAt: Date.now() + 10,
+ intervalMs: 10,
+ load: async () => ({
+ status: "in_progress",
+ connection: "server unavailable",
+ secret: "plaintext in an ordinary task description",
+ }),
+ accept: () => false,
+ });
+ const caught = pending.catch((error: unknown) => error);
+ await vi.advanceTimersByTimeAsync(11);
+ const error = await caught;
+ expect(classifyFailure(error)).toBe("candidate_failure");
+ expect((error as Error).message).not.toContain(
+ "ordinary task description",
+ );
+ } finally {
+ vi.useRealTimers();
+ }
+ });
+ it("keeps a failed network read retryable", async () => {
+ vi.useFakeTimers();
+ try {
+ const caught = pollUntil({
+ label: "task state",
+ deadlineAt: Date.now() + 10,
+ intervalMs: 10,
+ load: async () => {
+ throw new Error("ECONNRESET");
+ },
+ accept: () => false,
+ }).catch((error: unknown) => error);
+ await vi.advanceTimersByTimeAsync(11);
+ expect(classifyFailure(await caught)).toBe("transient_infrastructure");
+ } finally {
+ vi.useRealTimers();
+ }
+ });
+});
diff --git a/tests/runner-recovery/README.md b/tests/runner-recovery/README.md
new file mode 100644
index 0000000000..521fc9cd07
--- /dev/null
+++ b/tests/runner-recovery/README.md
@@ -0,0 +1,46 @@
+# Controlled runner recovery tests
+
+Run from the Paperclip repository after installing and building dependencies:
+
+```sh
+pnpm test:runner-recovery
+```
+
+This command runs provider-free tests with a disposable embedded Postgres database.
+It does not call a model, launch a paid browser campaign, or produce a model score.
+The database test must run rather than skip before claiming its coverage; consult
+its output for platform support. Run these tests when changing recovery policy,
+continuation delivery, or replacement admission.
+
+## Boundaries and checks
+
+| Test owner | Established premise | Required outcome |
+|---|---|---|
+| `native-replacement-evidence.test.ts` | Explicit known/unknown process ownership, history, workspace, action receipts, and retry budget | Allow replacement only with complete safe evidence. Return a specific cause and next action otherwise. |
+| `stopped-codex-turn.test.ts` | Precisely bound fixture transcripts with a closed text-only turn, partial output, wrong turn identity, or an unknown action | Accept the closed safe transcript; reject incomplete or uncertain current-turn effects. |
+| `native-safe-replacement.test.ts` | Inject the stopped-session verifier result into the real database reconciler; record real saved file bytes and one user comment | Verified recovery schedules one successor whose continuation includes the comment once; missing proof or an unconfirmed action schedules none. All cases retain the saved file and user comment. |
+| `native-safe-replacement.test.ts` | Concurrent reconciliation, transaction failpoints, changed ownership, exhausted budgets, and cancelled work | Preserve one successor lineage, roll back incomplete commits, respect user intent, and stop retries at the incident limit. |
+
+The injected verifier is a test boundary. These tests do not prove that an
+operating-system process actually stopped or that a provider completes the
+successor. They establish controller decisions and persistence under known
+conditions. Real stop verification and a user-visible final result require
+separate integration/acceptance evidence.
+
+## Why the paid crash probes were retired
+
+The former `recover-runner` and `recover-runner-safe` cases killed the daemon at
+an arbitrary point and expected automatic completion. They did not establish
+that every old process had stopped or that unfinished actions were safe to
+repeat. A no-tools prompt did not establish those facts. Their failures therefore
+do not establish a recovery correctness bug. `recover-runner-uncertain` exercised
+a useful safety rule but did not measure model quality or a complete user journey.
+
+Historical records remain in Evalbook, including failed grades and paid cost.
+The main scorecard omits all three cases. Supported controller restart and
+Stop/new-direction journeys remain in `everyday-workflows`.
+
+Before adding another live crash-recovery story, specify the supported recovery
+path, prove its fault boundary, perform the recovery action, then independently
+verify saved work, pending input, and a usable final result. Until that journey
+is qualified, do not score arbitrary process termination as a model failure.