mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-06 21:05:21 +02:00
Fix native Claude question bridging and runner behavioral regressions
Validate completion revisions in-turn, clarify text questions and approved handoffs, and make continuation and approval-overlap fixtures deterministic. Add native Claude question round-trip coverage. Co-Authored-By: Paperclip <noreply@paperclip.ing>
This commit is contained in:
1 parent
86d1992ac4
commit
cd34aaee0c
40 files changed
+563
-206
No files matched your search
@@ -179,23 +179,23 @@ The skill/reference inventory and eval cases are the only normative behavior sou
|
||||
| skill:skills/paperclip/references/api-reference.md:requesting-a-hire-management-only:861 | optional_agent_tool | skills/paperclip/references/api-reference.md:861 |
|
||||
| skill:skills/paperclip/references/api-reference.md:ceo-strategy-approval:893 | optional_agent_tool | skills/paperclip/references/api-reference.md:893 |
|
||||
| skill:skills/paperclip/references/api-reference.md:questions-and-waiting-for-human-input:902 | always_agent_tool | skills/paperclip/references/api-reference.md:902 |
|
||||
| skill:skills/paperclip/references/api-reference.md:issue-thread-confirmations:993 | always_agent_tool | skills/paperclip/references/api-reference.md:993 |
|
||||
| skill:skills/paperclip/references/api-reference.md:checkbox-confirmations:1051 | always_agent_tool | skills/paperclip/references/api-reference.md:1051 |
|
||||
| skill:skills/paperclip/references/api-reference.md:item-verdict-requests:1166 | optional_agent_tool | skills/paperclip/references/api-reference.md:1166 |
|
||||
| skill:skills/paperclip/references/api-reference.md:checking-approval-status:1276 | optional_agent_tool | skills/paperclip/references/api-reference.md:1276 |
|
||||
| skill:skills/paperclip/references/api-reference.md:approval-follow-up-requesting-agent:1282 | always_agent_tool | skills/paperclip/references/api-reference.md:1282 |
|
||||
| skill:skills/paperclip/references/api-reference.md:issue-lifecycle:1300 | always_agent_tool | skills/paperclip/references/api-reference.md:1300 |
|
||||
| skill:skills/paperclip/references/api-reference.md:error-handling:1330 | control_plane_owned | skills/paperclip/references/api-reference.md:1330 |
|
||||
| skill:skills/paperclip/references/api-reference.md:full-api-reference:1344 | optional_agent_tool | skills/paperclip/references/api-reference.md:1344 |
|
||||
| skill:skills/paperclip/references/api-reference.md:agents:1346 | optional_agent_tool | skills/paperclip/references/api-reference.md:1346 |
|
||||
| skill:skills/paperclip/references/api-reference.md:issues-tasks:1367 | optional_agent_tool | skills/paperclip/references/api-reference.md:1367 |
|
||||
| skill:skills/paperclip/references/api-reference.md:companies-projects-goals:1407 | optional_agent_tool | skills/paperclip/references/api-reference.md:1407 |
|
||||
| skill:skills/paperclip/references/api-reference.md:routines:1431 | optional_agent_tool | skills/paperclip/references/api-reference.md:1431 |
|
||||
| skill:skills/paperclip/references/api-reference.md:approvals-costs-activity-dashboard:1447 | optional_agent_tool | skills/paperclip/references/api-reference.md:1447 |
|
||||
| skill:skills/paperclip/references/api-reference.md:secrets:1469 | optional_agent_tool | skills/paperclip/references/api-reference.md:1469 |
|
||||
| skill:skills/paperclip/references/api-reference.md:agent-secret-proposals:1482 | optional_agent_tool | skills/paperclip/references/api-reference.md:1482 |
|
||||
| skill:skills/paperclip/references/api-reference.md:agent-secret-access:1582 | optional_agent_tool | skills/paperclip/references/api-reference.md:1582 |
|
||||
| skill:skills/paperclip/references/api-reference.md:common-mistakes:1622 | optional_agent_tool | skills/paperclip/references/api-reference.md:1622 |
|
||||
| skill:skills/paperclip/references/api-reference.md:issue-thread-confirmations:999 | always_agent_tool | skills/paperclip/references/api-reference.md:999 |
|
||||
| skill:skills/paperclip/references/api-reference.md:checkbox-confirmations:1057 | always_agent_tool | skills/paperclip/references/api-reference.md:1057 |
|
||||
| skill:skills/paperclip/references/api-reference.md:item-verdict-requests:1172 | optional_agent_tool | skills/paperclip/references/api-reference.md:1172 |
|
||||
| skill:skills/paperclip/references/api-reference.md:checking-approval-status:1282 | optional_agent_tool | skills/paperclip/references/api-reference.md:1282 |
|
||||
| skill:skills/paperclip/references/api-reference.md:approval-follow-up-requesting-agent:1288 | always_agent_tool | skills/paperclip/references/api-reference.md:1288 |
|
||||
| skill:skills/paperclip/references/api-reference.md:issue-lifecycle:1306 | always_agent_tool | skills/paperclip/references/api-reference.md:1306 |
|
||||
| skill:skills/paperclip/references/api-reference.md:error-handling:1336 | control_plane_owned | skills/paperclip/references/api-reference.md:1336 |
|
||||
| skill:skills/paperclip/references/api-reference.md:full-api-reference:1350 | optional_agent_tool | skills/paperclip/references/api-reference.md:1350 |
|
||||
| skill:skills/paperclip/references/api-reference.md:agents:1352 | optional_agent_tool | skills/paperclip/references/api-reference.md:1352 |
|
||||
| skill:skills/paperclip/references/api-reference.md:issues-tasks:1373 | optional_agent_tool | skills/paperclip/references/api-reference.md:1373 |
|
||||
| skill:skills/paperclip/references/api-reference.md:companies-projects-goals:1413 | optional_agent_tool | skills/paperclip/references/api-reference.md:1413 |
|
||||
| skill:skills/paperclip/references/api-reference.md:routines:1437 | optional_agent_tool | skills/paperclip/references/api-reference.md:1437 |
|
||||
| skill:skills/paperclip/references/api-reference.md:approvals-costs-activity-dashboard:1453 | optional_agent_tool | skills/paperclip/references/api-reference.md:1453 |
|
||||
| skill:skills/paperclip/references/api-reference.md:secrets:1475 | optional_agent_tool | skills/paperclip/references/api-reference.md:1475 |
|
||||
| skill:skills/paperclip/references/api-reference.md:agent-secret-proposals:1488 | optional_agent_tool | skills/paperclip/references/api-reference.md:1488 |
|
||||
| skill:skills/paperclip/references/api-reference.md:agent-secret-access:1588 | optional_agent_tool | skills/paperclip/references/api-reference.md:1588 |
|
||||
| skill:skills/paperclip/references/api-reference.md:common-mistakes:1628 | optional_agent_tool | skills/paperclip/references/api-reference.md:1628 |
|
||||
|
||||
## Legacy MCP Alias Index
|
||||
|
||||
|
||||
@@ -740,162 +740,162 @@
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:993",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:999",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L993:issue-thread-confirmations",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L999:issue-thread-confirmations",
|
||||
"heading": "Issue-thread confirmations",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1051",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1057",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1051:checkbox-confirmations",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1057:checkbox-confirmations",
|
||||
"heading": "Checkbox confirmations",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1166",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1172",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1166:item-verdict-requests",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1172:item-verdict-requests",
|
||||
"heading": "Item verdict requests",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1276",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1276:checking-approval-status",
|
||||
"heading": "Checking approval status",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1282",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1282:approval-follow-up-requesting-agent",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1282:checking-approval-status",
|
||||
"heading": "Checking approval status",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1288",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1288:approval-follow-up-requesting-agent",
|
||||
"heading": "Approval follow-up (requesting agent)",
|
||||
"primaryDisposition": "control_plane_owned",
|
||||
"semanticOperation": "runtime_reconciliation",
|
||||
"expectedMockState": "runtime_decision_record"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1300",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1306",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1300:issue-lifecycle",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1306:issue-lifecycle",
|
||||
"heading": "Issue Lifecycle",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1330",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1336",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1330:error-handling",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1336:error-handling",
|
||||
"heading": "Error Handling",
|
||||
"primaryDisposition": "control_plane_owned",
|
||||
"semanticOperation": "runtime_reconciliation",
|
||||
"expectedMockState": "runtime_decision_record"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1344",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1350",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1344:full-api-reference",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1350:full-api-reference",
|
||||
"heading": "Full API Reference",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1346",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1352",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1346:agents",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1352:agents",
|
||||
"heading": "Agents",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1367",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1373",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1367:issues-tasks",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1373:issues-tasks",
|
||||
"heading": "Issues (Tasks)",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1407",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1413",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1407:companies-projects-goals",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1413:companies-projects-goals",
|
||||
"heading": "Companies, Projects, Goals",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1431",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1437",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1431:routines",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1437:routines",
|
||||
"heading": "Routines",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1447",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1453",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1447:approvals-costs-activity-dashboard",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1453:approvals-costs-activity-dashboard",
|
||||
"heading": "Approvals, Costs, Activity, Dashboard",
|
||||
"primaryDisposition": "control_plane_owned",
|
||||
"semanticOperation": "runtime_reconciliation",
|
||||
"expectedMockState": "runtime_decision_record"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1469",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1475",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1469:secrets",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1475:secrets",
|
||||
"heading": "Secrets",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1482",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1488",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1482:agent-secret-proposals",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1488:agent-secret-proposals",
|
||||
"heading": "Agent secret proposals",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1535",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1541",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1535:re-bind-an-existing-secret-under-a-new-path-no-secret-id",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1541:re-bind-an-existing-secret-under-a-new-path-no-secret-id",
|
||||
"heading": "Re-bind an existing secret under a new path (no secret ID)",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1582",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1588",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1582:agent-secret-access",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1588:agent-secret-access",
|
||||
"heading": "Agent secret access",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
"expectedMockState": "operation_result"
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1622",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:1628",
|
||||
"kind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1622:common-mistakes",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md#L1628:common-mistakes",
|
||||
"heading": "Common Mistakes",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
"semanticOperation": "scoped_discovery",
|
||||
|
||||
@@ -5,6 +5,6 @@ Generated by `scripts/generate-capability-contract.mjs`; do not edit generated f
|
||||
- Skill/reference headings: 156
|
||||
- Legacy MCP tools: 42
|
||||
- Eval cases: 106 across 16 groups
|
||||
- Deterministic content SHA-256: `5ffebd5f684e312c07cc87723aa25864e85fc3f320e24341ed678852eef5c638`
|
||||
- Deterministic content SHA-256: `b42cfa2dbf4d314914ca18c77829f54e06b585892e38b1c4857b583989367a03`
|
||||
|
||||
Every row has exactly one primary disposition, a source anchor, a semantic operation, and a mock-state expectation.
|
||||
@@ -1858,7 +1858,7 @@
|
||||
"type": "string"
|
||||
},
|
||||
"initialPlan": {
|
||||
"description": "Relevant markdown plan to persist on the new task before it starts.",
|
||||
"description": "Remaining execution steps to persist as the task plan. Exclude completed planning, approval, and handoff steps; cite the source plan revision and approval. A copied plan is not a new approval gate.",
|
||||
"maxLength": 20000,
|
||||
"type": [
|
||||
"string",
|
||||
|
||||
@@ -2084,9 +2084,9 @@
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:issue-thread-confirmations:993",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:issue-thread-confirmations:999",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:993",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:999",
|
||||
"title": "Issue-thread confirmations",
|
||||
"expectedSemantics": "Skill guidance headed “Issue-thread confirmations”.",
|
||||
"primaryDisposition": "always_agent_tool",
|
||||
@@ -2095,13 +2095,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:993"
|
||||
"skill:skills/paperclip/references/api-reference.md:999"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:checkbox-confirmations:1051",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:checkbox-confirmations:1057",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1051",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1057",
|
||||
"title": "Checkbox confirmations",
|
||||
"expectedSemantics": "Skill guidance headed “Checkbox confirmations”.",
|
||||
"primaryDisposition": "always_agent_tool",
|
||||
@@ -2110,13 +2110,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1051"
|
||||
"skill:skills/paperclip/references/api-reference.md:1057"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:item-verdict-requests:1166",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:item-verdict-requests:1172",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1166",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1172",
|
||||
"title": "Item verdict requests",
|
||||
"expectedSemantics": "Skill guidance headed “Item verdict requests”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2125,13 +2125,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1166"
|
||||
"skill:skills/paperclip/references/api-reference.md:1172"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:checking-approval-status:1276",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:checking-approval-status:1282",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1276",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1282",
|
||||
"title": "Checking approval status",
|
||||
"expectedSemantics": "Skill guidance headed “Checking approval status”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2139,29 +2139,29 @@
|
||||
"assertionClasses": [
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1276"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:approval-follow-up-requesting-agent:1282",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1282",
|
||||
"title": "Approval follow-up (requesting agent)",
|
||||
"expectedSemantics": "Skill guidance headed “Approval follow-up (requesting agent)”.",
|
||||
"primaryDisposition": "always_agent_tool",
|
||||
"requiredGrants": [],
|
||||
"assertionClasses": [
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1282"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:issue-lifecycle:1300",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:approval-follow-up-requesting-agent:1288",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1300",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1288",
|
||||
"title": "Approval follow-up (requesting agent)",
|
||||
"expectedSemantics": "Skill guidance headed “Approval follow-up (requesting agent)”.",
|
||||
"primaryDisposition": "always_agent_tool",
|
||||
"requiredGrants": [],
|
||||
"assertionClasses": [
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1288"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:issue-lifecycle:1306",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1306",
|
||||
"title": "Issue Lifecycle",
|
||||
"expectedSemantics": "Skill guidance headed “Issue Lifecycle”.",
|
||||
"primaryDisposition": "always_agent_tool",
|
||||
@@ -2170,13 +2170,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1300"
|
||||
"skill:skills/paperclip/references/api-reference.md:1306"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:error-handling:1330",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:error-handling:1336",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1330",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1336",
|
||||
"title": "Error Handling",
|
||||
"expectedSemantics": "Skill guidance headed “Error Handling”.",
|
||||
"primaryDisposition": "control_plane_owned",
|
||||
@@ -2185,13 +2185,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1330"
|
||||
"skill:skills/paperclip/references/api-reference.md:1336"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:full-api-reference:1344",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:full-api-reference:1350",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1344",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1350",
|
||||
"title": "Full API Reference",
|
||||
"expectedSemantics": "Skill guidance headed “Full API Reference”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2200,13 +2200,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1344"
|
||||
"skill:skills/paperclip/references/api-reference.md:1350"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:agents:1346",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:agents:1352",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1346",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1352",
|
||||
"title": "Agents",
|
||||
"expectedSemantics": "Skill guidance headed “Agents”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2215,13 +2215,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1346"
|
||||
"skill:skills/paperclip/references/api-reference.md:1352"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:issues-tasks:1367",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:issues-tasks:1373",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1367",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1373",
|
||||
"title": "Issues (Tasks)",
|
||||
"expectedSemantics": "Skill guidance headed “Issues (Tasks)”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2230,13 +2230,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1367"
|
||||
"skill:skills/paperclip/references/api-reference.md:1373"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:companies-projects-goals:1407",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:companies-projects-goals:1413",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1407",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1413",
|
||||
"title": "Companies, Projects, Goals",
|
||||
"expectedSemantics": "Skill guidance headed “Companies, Projects, Goals”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2245,13 +2245,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1407"
|
||||
"skill:skills/paperclip/references/api-reference.md:1413"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:routines:1431",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:routines:1437",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1431",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1437",
|
||||
"title": "Routines",
|
||||
"expectedSemantics": "Skill guidance headed “Routines”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2260,13 +2260,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1431"
|
||||
"skill:skills/paperclip/references/api-reference.md:1437"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:approvals-costs-activity-dashboard:1447",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:approvals-costs-activity-dashboard:1453",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1447",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1453",
|
||||
"title": "Approvals, Costs, Activity, Dashboard",
|
||||
"expectedSemantics": "Skill guidance headed “Approvals, Costs, Activity, Dashboard”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2275,13 +2275,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1447"
|
||||
"skill:skills/paperclip/references/api-reference.md:1453"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:secrets:1469",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:secrets:1475",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1469",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1475",
|
||||
"title": "Secrets",
|
||||
"expectedSemantics": "Skill guidance headed “Secrets”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2290,13 +2290,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1469"
|
||||
"skill:skills/paperclip/references/api-reference.md:1475"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:agent-secret-proposals:1482",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:agent-secret-proposals:1488",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1482",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1488",
|
||||
"title": "Agent secret proposals",
|
||||
"expectedSemantics": "Skill guidance headed “Agent secret proposals”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2305,13 +2305,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1482"
|
||||
"skill:skills/paperclip/references/api-reference.md:1488"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:agent-secret-access:1582",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:agent-secret-access:1588",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1582",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1588",
|
||||
"title": "Agent secret access",
|
||||
"expectedSemantics": "Skill guidance headed “Agent secret access”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2320,13 +2320,13 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1582"
|
||||
"skill:skills/paperclip/references/api-reference.md:1588"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:common-mistakes:1622",
|
||||
"id": "skill:skills/paperclip/references/api-reference.md:common-mistakes:1628",
|
||||
"sourceKind": "skill_heading",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1622",
|
||||
"sourceAnchor": "skills/paperclip/references/api-reference.md:1628",
|
||||
"title": "Common Mistakes",
|
||||
"expectedSemantics": "Skill guidance headed “Common Mistakes”.",
|
||||
"primaryDisposition": "optional_agent_tool",
|
||||
@@ -2335,7 +2335,7 @@
|
||||
"control_plane_invariant"
|
||||
],
|
||||
"evidenceIds": [
|
||||
"skill:skills/paperclip/references/api-reference.md:1622"
|
||||
"skill:skills/paperclip/references/api-reference.md:1628"
|
||||
]
|
||||
}
|
||||
]
|
||||
|
||||
@@ -478,7 +478,7 @@ const descriptors: readonly PaperclipSemanticActionDescriptor[] = [
|
||||
...idempotency,
|
||||
title: text("Task title.", 500),
|
||||
projectId: nullableText("Project identifier for the new task."),
|
||||
initialPlan: nullableText("Relevant markdown plan to persist on the new task before it starts."),
|
||||
initialPlan: nullableText("Remaining execution steps to persist as the task plan. Exclude completed planning, approval, and handoff steps; cite the source plan revision and approval. A copied plan is not a new approval gate."),
|
||||
description: nullableText("Child task description."),
|
||||
assigneeActorId: nullableText("Optional actor assignee.", 200),
|
||||
priority: { enum: ["critical", "high", "medium", "low"] },
|
||||
|
||||
@@ -38,7 +38,7 @@ const completionClaimSchema = {
|
||||
additionalProperties: false,
|
||||
required: ["contractRevision", "objectiveSatisfied", "criteria", "remainingWork"],
|
||||
properties: {
|
||||
contractRevision: { type: "string", minLength: 1 },
|
||||
contractRevision: { type: "string", minLength: 1, description: "Use the current turn completion.revision (or completionContract.revision on the first turn), never a previous turn’s revision. On stale-revision feedback, reassess the current request and correct the report without repeating completed work." },
|
||||
objectiveSatisfied: { type: "boolean" },
|
||||
criteria: {
|
||||
type: "array",
|
||||
|
||||
@@ -78,7 +78,7 @@ export function runtimeRequestKind(method: string): HarnessRuntimeRequestKind |
|
||||
) {
|
||||
return "user_input";
|
||||
}
|
||||
if (method === "mcpServer/elicitation/request") return "elicitation";
|
||||
if (method === "mcpServer/elicitation/request" || method === "elicitation/create") return "elicitation";
|
||||
return null;
|
||||
}
|
||||
|
||||
@@ -89,6 +89,7 @@ export function runtimeRequestKind(method: string): HarnessRuntimeRequestKind |
|
||||
* degrading back to the legacy textarea presentation.
|
||||
*/
|
||||
export function hasCodexQuestionForm(method: string, params: Record<string, unknown>): boolean {
|
||||
if (method === "elicitation/create") return "questionSet" in params;
|
||||
if (method === "item/tool/requestUserInput" || method === "tool/requestUserInput") {
|
||||
return "questions" in params;
|
||||
}
|
||||
@@ -266,6 +267,7 @@ export function normalizeCodexQuestionSet(
|
||||
params: Record<string, unknown>,
|
||||
responseContext: CodexQuestionResponseContext,
|
||||
): PaperclipQuestionSet | null {
|
||||
if (method === "elicitation/create") return parsePaperclipQuestionSet(params.questionSet);
|
||||
if (method === "item/tool/requestUserInput" || method === "tool/requestUserInput") {
|
||||
if (!Array.isArray(params.questions) || params.questions.length === 0) return null;
|
||||
if (params.questions.length > 64) throw new Error("Codex question form exceeds 64 questions");
|
||||
@@ -550,6 +552,9 @@ export function runtimeRequestResponse(
|
||||
resolution: HarnessRuntimeRequestResolution,
|
||||
responseContext: CodexQuestionResponseContext,
|
||||
): Record<string, unknown> {
|
||||
// The durable transport sends this canonical resolution to the ACPX sidecar,
|
||||
// which owns conversion back to the original provider form values.
|
||||
if (request.method === "elicitation/create") return structuredClone(resolution);
|
||||
if (
|
||||
request.requestKind === "command_approval" ||
|
||||
request.requestKind === "file_approval"
|
||||
|
||||
@@ -292,7 +292,11 @@ async function handleServerRequestBody(
|
||||
prompt: runtimeRequestPrompt(requestKind, request.params),
|
||||
details: record(redactCodexValue(boundedCodexValue(request.params))),
|
||||
...(input !== null ? { input } : {}),
|
||||
origin: {
|
||||
origin: request.method === "elicitation/create" ? {
|
||||
adapter: "acpx-runtime-sidecar",
|
||||
provider: text(record(request.params.origin).provider, "acpx"),
|
||||
method: request.method,
|
||||
} : {
|
||||
adapter: "codex-app-server",
|
||||
provider: "codex",
|
||||
method: request.method,
|
||||
|
||||
@@ -207,7 +207,7 @@ export function safeCodexRequestResponse(
|
||||
if (method === "item/permissions/requestApproval") {
|
||||
return { permissions: {}, scope: "turn" };
|
||||
}
|
||||
if (method === "mcpServer/elicitation/request") {
|
||||
if (method === "mcpServer/elicitation/request" || method === "elicitation/create") {
|
||||
return { action, content: null, _meta: null };
|
||||
}
|
||||
if (
|
||||
|
||||
@@ -0,0 +1,33 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { bridgedCodexQuestionParams } from "./runnerd-codex-transport.js";
|
||||
import { normalizeAcpFormElicitation } from "../drivers/acpx/acp-question-adapter.js";
|
||||
import { createCodexQuestionResponseContext, normalizeCodexQuestionSet, runtimeRequestKind, runtimeRequestResponse } from "../drivers/codex/codex-question-adapter.js";
|
||||
|
||||
describe("ACPX questions through the shared runner transport", () => {
|
||||
it("keeps Claude's form and answer identities intact through the round trip", () => {
|
||||
const form = normalizeAcpFormElicitation({ mode: "form", message: "Interview", requestedSchema: {
|
||||
type: "object", required: ["organization", "goal"], properties: {
|
||||
organization: { type: "string", description: "What does your organization do?" },
|
||||
goal: { type: "string", description: "What should we achieve?" },
|
||||
timing: { type: "string", oneOf: [{ const: "today", title: "Today" }, { const: "later", title: "Later" }] },
|
||||
},
|
||||
} })!;
|
||||
const origin = { adapter: "acpx-runtime-sidecar", provider: "claude", method: "elicitation/create" };
|
||||
const params = bridgedCodexQuestionParams({ requestId: "question-1", input: form.questionSet, origin }, origin.method, "session", "turn")!;
|
||||
expect(runtimeRequestKind(origin.method)).toBe("elicitation");
|
||||
const context = createCodexQuestionResponseContext();
|
||||
const shown = normalizeCodexQuestionSet(origin.method, params, context)!;
|
||||
expect(shown).toEqual(form.questionSet);
|
||||
const [org, goal, timing] = shown.questions;
|
||||
const response = { schema: "paperclip.question_response.v1" as const, answers: {
|
||||
[org!.id]: { text: "Garden club" }, [goal!.id]: { text: "Welcome note" },
|
||||
[timing!.id]: { selectedOptionIds: [timing!.options![1]!.id] },
|
||||
} };
|
||||
expect(runtimeRequestResponse({ requestId: "question-1", requestKind: "elicitation", method: origin.method,
|
||||
turnId: "turn", itemId: "item", status: "pending", prompt: "Interview", input: shown }, { action: "submit", response }, context)).toEqual({ action: "submit", response });
|
||||
expect(form.accept(response)).toEqual({ action: "accept", content: { organization: "Garden club", goal: "Welcome note", timing: "later" } });
|
||||
});
|
||||
it("does not admit an invalid canonical question set", () => {
|
||||
expect(() => normalizeCodexQuestionSet("elicitation/create", { questionSet: { schema: "bad", questions: [] } }, createCodexQuestionResponseContext())).toThrow();
|
||||
});
|
||||
});
|
||||
@@ -863,7 +863,7 @@ async function awaitAdoptedRunnerAuthentication(input: {
|
||||
}
|
||||
}
|
||||
|
||||
function bridgedCodexQuestionParams(
|
||||
export function bridgedCodexQuestionParams(
|
||||
request: Record<string, unknown>,
|
||||
method: string,
|
||||
threadId: string,
|
||||
@@ -884,6 +884,12 @@ function bridgedCodexQuestionParams(
|
||||
? request.itemId
|
||||
: String(request.requestId ?? "runtime-input"),
|
||||
};
|
||||
// ACPX has already normalized and bound these IDs in Rust. Reconstructing a
|
||||
// Codex form here would change option IDs and break the answer's return path.
|
||||
if (method === "elicitation/create") {
|
||||
return { ...common, questionSet, origin: request.origin,
|
||||
message: questionSet.description ?? questionSet.title ?? "A tool needs your input" };
|
||||
}
|
||||
if (method === "mcpServer/elicitation/request") {
|
||||
const required: string[] = [];
|
||||
const properties = Object.fromEntries(
|
||||
@@ -5889,7 +5895,8 @@ class DurablePrpCodexTransport implements CodexAppServerTransport {
|
||||
params &&
|
||||
(method === "item/tool/requestUserInput" ||
|
||||
method === "tool/requestUserInput" ||
|
||||
method === "mcpServer/elicitation/request") &&
|
||||
method === "mcpServer/elicitation/request" ||
|
||||
method === "elicitation/create") &&
|
||||
!this.#bridgedRuntimeInputs.has(requestId)
|
||||
) {
|
||||
this.#bridgedRuntimeInputs.set(requestId, {
|
||||
|
||||
@@ -23,7 +23,7 @@ Work in this order.
|
||||
|
||||
1. Take the path the user picked.
|
||||
|
||||
- `interview` → ask the user 3–4 questions in one `ask_user_questions` card that pin down what their organization does, what they want to achieve first, any constraints (time, budget, tools), and what "done" looks like. Don't guess; ask. Don't post anything else before the card. The answers lead to the plan-and-team path in step 2.
|
||||
- `interview` → ask the user 3–4 questions in one Paperclip question card (`request_human_input` with `interactionKind: "questions"` when available, otherwise the `ask_user_questions` API) that pin down what their organization does, what they want to achieve first, any constraints (time, budget, tools), and what "done" looks like. Don't guess; ask. Don't post anything else before the card. The answers lead to the plan-and-team path in step 2.
|
||||
|
||||
- `task` → the text they typed is the task. If it is clear enough to propose on, go straight to step 2. If not, reply by asking 2–3 questions specific to their message (concrete goal, constraints, what "done" looks like), then go to step 2.
|
||||
|
||||
@@ -31,6 +31,7 @@ Work in this order.
|
||||
|
||||
2. Propose, then wait for acceptance.
|
||||
|
||||
- Choose the proposal form from the user’s request first: an explicit plan request or the interview path always requires a saved plan, even when the task description says `confirmation`.
|
||||
- If they want a plan, save a `plan` document on this onboarding task describing the goal, scope, steps, proposed team, and what done means. Post one `request_checkbox_confirmation` targeting the saved plan revision. A card or thread message alone is not a saved plan. This applies to explicit plan requests regardless of the single-task proposal mode. Proposing a team does not authorize hiring it.
|
||||
- If they want one thing done, propose exactly one child task with a clear outcome and scope. Ask them to accept it before creating the child. Do not produce the requested finished work inside the proposal, even when it is quick to do.
|
||||
- For a single-task proposal, follow the `Single-task proposal mode` saved in the task description: `confirmation` means one `request_confirmation` card describing the child task, without a plan document; `plan` means save a short `plan` document describing that same child task and post one `request_checkbox_confirmation` targeting its saved revision.
|
||||
|
||||
@@ -85,7 +85,7 @@ When the user asks to approve a plan before handoff, publish the plan and create
|
||||
|
||||
Before handing off work, inspect available projects and repositories. Every task you create from this chat must belong to a suitable project. Reuse an appropriate existing project; otherwise use create_project. Consider all relevant available repositories and pass repositoryIds for one or multiple repositories when the work spans them. For existing GitHub repositories you can access that are absent from the catalog, pass their HTTPS repositoryUrls; this registers them with the project without creating remote GitHub repositories. You may combine known IDs and URLs and attach multiple repositories. The direct HTTP equivalent is POST /api/companies/{companyId}/projects with name, repositoryIds and/or repositoryUrls arrays, and an idempotencyKey. Include all selected repositories in that creation; do not combine these arrays with workspace. Never invent repository IDs or substitute inaccessible repositories. Ask when the choice is materially ambiguous or required access is missing. Non-code projects may need no repository.
|
||||
|
||||
Create ordinary assigned tasks, never subtasks of this conversation. Give each task a clear outcome, context, acceptance criteria, project, and appropriate assignee. Use create_task with initialPlan to copy the relevant plan into the new task before execution starts. If using the HTTP API directly, POST /api/companies/{companyId}/issues with projectId, assigneeAgentId, status: "todo", initialPlan containing the relevant plan Markdown, and an idempotencyKey; omit parentId. Putting a plan in description does not create the task's plan document. Verify the new task's plan document before claiming the handoff is complete. Preserve the original plan here. When splitting work, include the relevant part of the plan in each task. Create and link each task before claiming it exists.
|
||||
Create ordinary assigned tasks, never subtasks of this conversation. Give each task a clear outcome, context, acceptance criteria, project, and appropriate assignee. Use create_task with initialPlan to copy the relevant plan into the new task before execution starts. If using the HTTP API directly, POST /api/companies/{companyId}/issues with projectId, assigneeAgentId, status: "todo", initialPlan containing the relevant plan Markdown, and an idempotencyKey; omit parentId. Putting a plan in description does not create the task's plan document. Verify the new task's plan document before claiming the handoff is complete. Preserve the original plan here. Carry forward only the remaining execution steps, not completed planning, approval, project creation, or task creation steps. Include the source conversation ID, approved plan revision ID, and accepted interaction ID so the worker can verify the recorded approval. State the approved scope and what is already done; do not claim the new plan document has its own approval. The worker should execute that authorized scope, and ask again only if scope changes or another applicable gate requires it. When splitting work, include the relevant part of the plan in each task. Create and link each task before claiming it exists.
|
||||
|
||||
Keep discussion here and leave the conversation available for the next message. Link handed-off tasks in your reply; do not make this conversation blocked by their completion or wait for them. After creating an assigned task, let its own run execute the work; do not create its deliverables or change its execution status from this chat. Reply normally and end your turn; Paperclip manages the conversation waiting state. Do not change its status, create a review confirmation just to finish a reply, mark it complete, or poll for another reply. An accepted plan authorizes handoff to execution tasks, never implementation on this conversation. Honor normal approvals. Ask mode is non-mutating. Plan mode supports research and writing/revising the plan; hand off for execution only through the normal authorized workflow.`;
|
||||
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import { beforeAll, afterAll, describe, expect, it } from "vitest";
|
||||
import { eq } from "drizzle-orm";
|
||||
import { createDb, companies, agents, issues, heartbeatRuns, documents, documentRevisions, issueDocuments, issueThreadInteractions } from "@paperclipai/db";
|
||||
import { getEmbeddedPostgresTestSupport, startEmbeddedPostgresTestDatabase } from "../../__tests__/helpers/embedded-postgres.js";
|
||||
import { handoffPlanContext } from "./handoff-plan-context.js";
|
||||
|
||||
const support = await getEmbeddedPostgresTestSupport();
|
||||
(support.supported ? describe : describe.skip)("handoff approval evidence", () => {
|
||||
let temporary: Awaited<ReturnType<typeof startEmbeddedPostgresTestDatabase>>;
|
||||
let db: ReturnType<typeof createDb>;
|
||||
beforeAll(async () => { temporary = await startEmbeddedPostgresTestDatabase("handoff-plan-"); db = createDb(temporary.connectionString); }, 20_000);
|
||||
afterAll(async () => { await temporary?.cleanup(); });
|
||||
async function seed() {
|
||||
const companyId = randomUUID(), agentId = randomUUID(), sourceId = randomUUID(), runId = randomUUID();
|
||||
const documentId = randomUUID(), revisionId = randomUUID(), interactionId = randomUUID();
|
||||
await db.insert(companies).values({ id: companyId, name: "Handoff", issuePrefix: companyId.slice(0, 8) });
|
||||
await db.insert(agents).values({ id: agentId, companyId, name: "Planner", adapterType: "paperclip_runner" });
|
||||
await db.insert(issues).values({ id: sourceId, companyId, title: "Source chat", conversationAgentId: agentId, assigneeAgentId: agentId, conversationUserId: "operator", conversationState: "active" });
|
||||
await db.insert(heartbeatRuns).values({ id: runId, companyId, agentId, nativeIssueId: sourceId, status: "succeeded" });
|
||||
await db.insert(documents).values({ id: documentId, companyId, latestBody: "A newer unapproved plan" });
|
||||
await db.insert(documentRevisions).values({ id: revisionId, documentId, companyId, revisionNumber: 1, body: "Write the approved note." });
|
||||
await db.insert(issueDocuments).values({ companyId, issueId: sourceId, documentId, key: "plan" });
|
||||
await db.insert(issueThreadInteractions).values({ id: interactionId, companyId, issueId: sourceId, kind: "request_confirmation", status: "accepted", resolvedAt: new Date("2026-09-01"), payload: { version: 1, prompt: "Approve the plan", target: { type: "issue_document", key: "plan", revisionId } } });
|
||||
const [task] = await db.insert(issues).values({ companyId, title: "Execute", originRunId: runId, createdAt: new Date("2026-09-02") }).returning();
|
||||
return { task, companyId, sourceId, runId, revisionId, interactionId, documentId };
|
||||
}
|
||||
it("returns the exact accepted source revision, never a newer unapproved body or new-document approval", async () => {
|
||||
const f = await seed();
|
||||
const result = await handoffPlanContext(db, f.task);
|
||||
expect(result).toMatchObject({ sourceIssueId: f.sourceId, revisionId: f.revisionId, interactionId: f.interactionId, markdown: "Write the approved note." });
|
||||
expect(result?.guidance).toContain("not approval of changes");
|
||||
expect(result?.guidance).toContain("Preserve other applicable gates");
|
||||
});
|
||||
it.each(["pending", "later", "other-company", "other-task", "unrelated-document", "no-origin"])("does not infer authority from %s evidence", async (kind) => {
|
||||
const f = await seed();
|
||||
if (kind === "pending") await db.update(issueThreadInteractions).set({ status: "pending" }).where(eq(issueThreadInteractions.id, f.interactionId));
|
||||
if (kind === "later") await db.update(issueThreadInteractions).set({ resolvedAt: new Date("2026-09-03") }).where(eq(issueThreadInteractions.id, f.interactionId));
|
||||
if (kind === "other-company") f.task.companyId = randomUUID();
|
||||
if (kind === "other-task") await db.update(issues).set({ conversationAgentId: null, conversationUserId: null, conversationState: null }).where(eq(issues.id, f.sourceId));
|
||||
if (kind === "unrelated-document") await db.update(issueDocuments).set({ key: "unrelated" }).where(eq(issueDocuments.documentId, f.documentId));
|
||||
if (kind === "no-origin") f.task.originRunId = null;
|
||||
expect(await handoffPlanContext(db, f.task)).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,33 @@
|
||||
import { and, desc, eq, lte } from "drizzle-orm";
|
||||
import { documentRevisions, heartbeatRuns, issueDocuments, issues, issueThreadInteractions, type Db } from "@paperclipai/db";
|
||||
|
||||
/** Approval evidence belongs to the source conversation and exact revision.
|
||||
* It informs scope; it does not approve the new task's document or waive gates. */
|
||||
export async function handoffPlanContext(db: Db, task: typeof issues.$inferSelect) {
|
||||
if (!task.originRunId || task.conversationAgentId) return null;
|
||||
const [sourceRun] = await db.select().from(heartbeatRuns).where(and(
|
||||
eq(heartbeatRuns.id, task.originRunId), eq(heartbeatRuns.companyId, task.companyId),
|
||||
));
|
||||
const sourceId = sourceRun?.nativeIssueId ?? sourceRun?.contextSnapshot?.issueId;
|
||||
if (typeof sourceId !== "string" || sourceId === task.id) return null;
|
||||
const [source] = await db.select().from(issues).where(and(eq(issues.id, sourceId), eq(issues.companyId, task.companyId)));
|
||||
if (!source?.conversationAgentId) return null;
|
||||
const accepted = await db.select().from(issueThreadInteractions).where(and(
|
||||
eq(issueThreadInteractions.companyId, task.companyId), eq(issueThreadInteractions.issueId, source.id),
|
||||
eq(issueThreadInteractions.kind, "request_confirmation"), eq(issueThreadInteractions.status, "accepted"),
|
||||
lte(issueThreadInteractions.resolvedAt, task.createdAt),
|
||||
)).orderBy(desc(issueThreadInteractions.resolvedAt));
|
||||
for (const interaction of accepted) {
|
||||
const target = (interaction.payload as { target?: { type?: string; key?: string; revisionId?: string; issueId?: string } }).target;
|
||||
if (target?.type !== "issue_document" || target.key !== "plan" || !target.revisionId ||
|
||||
(target.issueId && target.issueId !== source.id)) continue;
|
||||
const [revision] = await db.select({ markdown: documentRevisions.body, revisionId: documentRevisions.id })
|
||||
.from(documentRevisions).innerJoin(issueDocuments, and(
|
||||
eq(issueDocuments.documentId, documentRevisions.documentId), eq(issueDocuments.companyId, task.companyId),
|
||||
eq(issueDocuments.issueId, source.id), eq(issueDocuments.key, "plan"),
|
||||
)).where(and(eq(documentRevisions.id, target.revisionId), eq(documentRevisions.companyId, task.companyId)));
|
||||
if (revision) return { sourceIssueId: source.id, interactionId: interaction.id, ...revision,
|
||||
guidance: "This source plan was accepted before this task was created. Execute the assigned scope within that plan; planning and handoff steps already completed in the source are not new work. This is not approval of changes to scope or of a new task document. Preserve other applicable gates." };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
@@ -8,6 +8,7 @@ import {
|
||||
approvals,
|
||||
agents,
|
||||
heartbeatRuns,
|
||||
completionContracts,
|
||||
issueApprovals,
|
||||
issueThreadInteractions,
|
||||
issues,
|
||||
@@ -55,6 +56,21 @@ export async function nativeCompletionFeedback(
|
||||
? "Review blocker recorded. Paperclip will preserve the task and record the reviewer recovery action."
|
||||
: "Review report accepted. The recorded review decision controls task completion; this report cannot override it.";
|
||||
}
|
||||
// Bind feedback to this run, not the first contract from a reused session or
|
||||
// an unrelated newer run. Reject before admitting the result so the provider
|
||||
// can correct the report in the same turn.
|
||||
if (run.completionContractId) {
|
||||
const contract = await db.select().from(completionContracts).where(and(
|
||||
eq(completionContracts.id, run.completionContractId),
|
||||
eq(completionContracts.companyId, run.companyId),
|
||||
eq(completionContracts.issueId, issue.id),
|
||||
)).then((rows) => rows[0]);
|
||||
if (!contract) throw new Error("Completion report's bound contract no longer exists.");
|
||||
const current = contract.contractJson as { revision?: string; criteria?: Array<{ id: string }> };
|
||||
if (result.completionClaim.contractRevision !== current.revision) {
|
||||
throw new Error(`Stale completionClaim.contractRevision. This turn requires ${JSON.stringify(current.revision)} with criterion IDs ${JSON.stringify(current.criteria?.map((c) => c.id) ?? [])}. Reassess the current request and resubmit your report with that revision; do not repeat completed work.`);
|
||||
}
|
||||
}
|
||||
const signals = normalizePrpResultSignals(result);
|
||||
if (
|
||||
result.reportedWorkDisposition === "done" &&
|
||||
|
||||
@@ -199,12 +199,17 @@ describeEmbeddedPostgres("native question bridge", () => {
|
||||
};
|
||||
}
|
||||
|
||||
it("materializes, validates, and durably resumes a provider-neutral question response", async () => {
|
||||
it.each(["codex", "claude"])("materializes, validates, and durably resumes a %s question response", async (provider) => {
|
||||
await seed();
|
||||
const interaction = await projectNativeRuntimeRequest({
|
||||
db,
|
||||
binding: binding(),
|
||||
event: runtimeRequestEvent(),
|
||||
event: { ...runtimeRequestEvent(), payload: {
|
||||
request: { ...(runtimeRequestEvent().payload.request as Record<string, unknown>),
|
||||
origin: { adapter: provider === "claude" ? "acpx-runtime-sidecar" : "codex-app-server", provider,
|
||||
method: provider === "claude" ? "elicitation/create" : "item/tool/requestUserInput" },
|
||||
},
|
||||
} },
|
||||
});
|
||||
|
||||
expect(interaction).toMatchObject({
|
||||
|
||||
@@ -1053,7 +1053,7 @@ describe("PaperclipControlPlanePort conformance", () => {
|
||||
backendKind: "mock",
|
||||
sourceInstanceId: runnerInstanceId,
|
||||
});
|
||||
const result = { ...structuredClone(CONTROL_PLANE_CONFORMANCE_RESULT), reportedWorkDisposition: "needs_review" as const, attentionRequests: [{ kind: "approval" as const, summary: "Approve publication", ownerClass: "human" as const }, { kind: "review" as const, summary: "Review release notes", ownerClass: "agent" as const, targetAgentId: reviewerAgentId }] };
|
||||
const result = { ...structuredClone(CONTROL_PLANE_CONFORMANCE_RESULT), completionClaim: { ...CONTROL_PLANE_CONFORMANCE_RESULT.completionClaim, contractRevision: "phase6-v1" }, reportedWorkDisposition: "needs_review" as const, attentionRequests: [{ kind: "approval" as const, summary: "Approve publication", ownerClass: "human" as const }, { kind: "review" as const, summary: "Review release notes", ownerClass: "agent" as const, targetAgentId: reviewerAgentId }] };
|
||||
await expect(nativeCompletionFeedback(db, runId, { ...result, attentionRequests: [] }))
|
||||
.rejects.toThrow("needs_review requires");
|
||||
await expect(nativeCompletionFeedback(db, runId, {
|
||||
@@ -1062,6 +1062,10 @@ describe("PaperclipControlPlanePort conformance", () => {
|
||||
await expect(nativeCompletionFeedback(db, runId, {
|
||||
...result, attentionRequests: [{ kind: "review", summary: "Review work", ownerClass: "agent", targetAgentId: "99999999-9999-4999-8999-999999999999" }],
|
||||
})).rejects.toThrow("not available in this company");
|
||||
await expect(nativeCompletionFeedback(db, runId, {
|
||||
...result,
|
||||
completionClaim: { ...result.completionClaim, contractRevision: "stale-first-turn" },
|
||||
})).rejects.toThrow(/contractRevision.*phase6-v1/);
|
||||
await expect(nativeCompletionFeedback(db, runId, result)).resolves.toContain("Completion report accepted");
|
||||
await port.completeRun({
|
||||
result,
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { handoffPlanContext } from "./handoff-plan-context.js";
|
||||
import { callCreateSkillTool } from "../skill-tools.js";
|
||||
import { callProjectTool } from "../project-tools.js";
|
||||
import { isConnectorTool, executeConnectorTool, type ConnectorAssignment } from "../connector-runtime.js";
|
||||
@@ -378,6 +379,7 @@ export class PaperclipRunnerToolAuthority {
|
||||
},
|
||||
connectionGuidance: CONNECTION_INTENT_AGENT_GUIDANCE,
|
||||
acceptedPlan: await this.#acceptedPlan(context.run.contextSnapshot),
|
||||
sourcePlanApproval: await handoffPlanContext(this.db, context.issue),
|
||||
childReviewOutcomes: await childReviewOutcomes(this.db, this.binding.companyId, this.binding.issueId),
|
||||
...(this.binding.nativeReview ? {
|
||||
assignedReview: (await getNativeReviewAssignment(this.db, {
|
||||
@@ -1349,7 +1351,7 @@ export class PaperclipRunnerToolAuthority {
|
||||
}
|
||||
: null;
|
||||
if (targetRevisionId !== null && suppliedPayload.target === undefined && inferredPlanningTarget === null) {
|
||||
throw new Error("paperclip_runner_interaction_target_incomplete");
|
||||
throw new Error('paperclip_runner_interaction_target_incomplete: targetRevisionId also requires payload.target = { type: "issue_document", key: "plan", revisionId: targetRevisionId }. Use the actual document key. If the source plan already authorized this scope, do not request approval again merely because the execution task has a new plan document.');
|
||||
}
|
||||
const normalizedPayload = inferredPlanningTarget !== null
|
||||
? { ...suppliedPayload, target: inferredPlanningTarget }
|
||||
|
||||
@@ -687,3 +687,7 @@ Results are ranked by relevance: title matches first, then identifier, descripti
|
||||
For detailed API tables, JSON response schemas, worked examples (IC and Manager heartbeats), governance/approvals, cross-team delegation rules, error codes, issue lifecycle diagram, and the common mistakes table, read: `skills/paperclip/references/api-reference.md`
|
||||
|
||||
Again, rule #1 is: never ask a human to do what an agent could do. Try harder. Try again. Ask another agent to help. Keep working until the goal is fully accomplished.
|
||||
|
||||
**Asking a free-text question.**
|
||||
|
||||
For an open answer, use a text field, not invented choices. Copy the complete **Text answer** interaction example in [references/api-reference.md](references/api-reference.md#questions-and-waiting-for-human-input): it includes both the canonical `payload.questionSet` (`answerMode: "text"`) and required storage fields. The storage fallback alone renders the wrong control.
|
||||
@@ -903,32 +903,9 @@ POST /api/companies/{companyId}/approvals
|
||||
|
||||
Ask only when missing input materially blocks the request. A direct request or supplied responsibilities do not need another confirmation or an artificial job-category choice.
|
||||
|
||||
Use `ask_user_questions` for a short question card. Each `payload.questions` entry requires `id`, `prompt`, `selectionMode`, and options with `id` and `label`. Choice questions must offer at least two distinct, meaningful choices; use the canonical text presentation below for open-ended questions. Do not send `question`/`type: "text"` or an empty options array in a `payload.questions` entry. Set `resolverPolicy: "human_only"` when the answer must come from the user.
|
||||
Choose the input control from the answer you need: use a **text field** for a name, description, constraint, or other open answer; use choices only for an actual decision with at least two meaningful alternatives. Do not turn an open question into invented categories.
|
||||
|
||||
```json
|
||||
POST /api/issues/{issueId}/interactions
|
||||
{
|
||||
"kind": "ask_user_questions",
|
||||
"idempotencyKey": "questions:{issueId}:responsibility:v1",
|
||||
"title": "Hire responsibility",
|
||||
"resolverPolicy": "human_only",
|
||||
"continuationPolicy": "wake_assignee",
|
||||
"payload": {
|
||||
"version": 1,
|
||||
"questions": [{
|
||||
"id": "responsibility",
|
||||
"prompt": "What should the new agent be responsible for?",
|
||||
"selectionMode": "single",
|
||||
"required": true,
|
||||
"allowOther": true,
|
||||
"options": [
|
||||
{ "id": "research", "label": "Research", "description": "Find and summarize information." },
|
||||
{ "id": "writing", "label": "Writing", "description": "Draft and edit content." }
|
||||
]
|
||||
}]
|
||||
}
|
||||
}
|
||||
```
|
||||
**Text answer (copy this complete payload)**
|
||||
|
||||
For an open-ended answer, render a text field using `payload.questionSet` with `answerMode: "text"`, no options, and no `customAnswer`. The REST API still requires matching `payload.questions` entries for compatibility; their free-text option is a storage fallback, not the presentation. Keep question IDs and prompts identical in both fields. Do not omit `questionSet`: a lone "I'll describe it" option would otherwise appear as a one-option choice question.
|
||||
|
||||
@@ -962,6 +939,35 @@ POST /api/issues/{issueId}/interactions
|
||||
}
|
||||
```
|
||||
|
||||
**Multiple choice**
|
||||
|
||||
Use `ask_user_questions` for a short question card. Each `payload.questions` entry requires `id`, `prompt`, `selectionMode`, and options with `id` and `label`. Choice questions must offer at least two distinct, meaningful choices; use the canonical text presentation above for open-ended questions. Do not send `question`/`type: "text"` or an empty options array in a `payload.questions` entry. Set `resolverPolicy: "human_only"` when the answer must come from the user.
|
||||
|
||||
```json
|
||||
POST /api/issues/{issueId}/interactions
|
||||
{
|
||||
"kind": "ask_user_questions",
|
||||
"idempotencyKey": "questions:{issueId}:responsibility:v1",
|
||||
"title": "Hire responsibility",
|
||||
"resolverPolicy": "human_only",
|
||||
"continuationPolicy": "wake_assignee",
|
||||
"payload": {
|
||||
"version": 1,
|
||||
"questions": [{
|
||||
"id": "responsibility",
|
||||
"prompt": "What should the new agent be responsible for?",
|
||||
"selectionMode": "single",
|
||||
"required": true,
|
||||
"allowOther": true,
|
||||
"options": [
|
||||
{ "id": "research", "label": "Research", "description": "Find and summarize information." },
|
||||
{ "id": "writing", "label": "Writing", "description": "Draft and edit content." }
|
||||
]
|
||||
}]
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
After verifying the interaction was saved and is pending, record the waiting state:
|
||||
|
||||
```json
|
||||
|
||||
@@ -17,8 +17,9 @@ The vocabulary is: a **campaign** is one workflow invocation against one SHA; a
|
||||
environments × cases; an **execution/cell** is one parallel job; and an
|
||||
**attempt** is one isolated harness run, including an infrastructure retry.
|
||||
|
||||
The browser creates and assigns the task. The harness does not call a private
|
||||
runner hook or write fixtures directly to the database.
|
||||
The browser creates and assigns the task; fixtures use public APIs. The
|
||||
`accept-while-running` case additionally holds the committed card’s creation
|
||||
response in the test server until browser acceptance, to exercise real overlap.
|
||||
|
||||
The launcher always sets `PAPERCLIP_ANNOUNCEMENTS_ENABLED=false` for its isolated
|
||||
instances so announcement panels do not obscure screenshot evidence. No shell
|
||||
@@ -167,7 +168,7 @@ Both suites save and restore experimental settings. Browser E2E always starts a
|
||||
throwaway instance; never point the authenticated suite at the running demo.
|
||||
Missing provider credentials fail paid preflight and are not passing coverage.
|
||||
|
||||
The default `--all` selection is 166 cells (143 local and 23 Daytona) and 362
|
||||
The default `--all` selection is 167 cells (144 local and 23 Daytona) and 363
|
||||
expected paid agent turns. The explicit-only everyday suite adds 35 catalog cells
|
||||
and is excluded from `--all`. Follow-up steps remain ordered within their cell; all other
|
||||
cells are independent. Narrow selectors are strongly recommended while
|
||||
@@ -454,7 +455,7 @@ Set `RUNNER_E2E_AWS_ENABLED=true` to route paid cells to the repository-scoped
|
||||
ephemeral AWS RunsOn fleet selected by
|
||||
`runs-on/fleet=paperclip-public-pr-x64/env=public-ci`. Any other value uses the
|
||||
proven GitHub-hosted `ubuntu-latest` target. Set `RUNNER_E2E_MAX_PARALLEL` to an
|
||||
integer from 1–100 on AWS (default 100). The 166-cell default selection takes more than
|
||||
integer from 1–100 on AWS (default 100). The 167-cell default selection takes more than
|
||||
one wave at that limit; use suite selectors for smaller campaigns. The fallback runner retains its 1–57 limit and
|
||||
default of 32. Multi-turn steps are sequential inside their cell while
|
||||
independent cells overlap. Artifacts and merged HTML/JUnit/normalized reports
|
||||
@@ -723,7 +724,7 @@ resolver projections. This is a regression sample, not an exhaustive injection
|
||||
or authorization evaluation.
|
||||
|
||||
The native-only `question-tool-documentation` case adds two cells (Runner Codex
|
||||
and Runner ACPX Claude), for 22 continuation cells total. It asks for a clickable
|
||||
and Runner ACPX Claude), for 23 continuation cells total. It asks for a clickable
|
||||
Morning/Afternoon question, followed by an open text question, then a saved note
|
||||
using both real answers. The user prompt contains no tool names or payload recipes.
|
||||
Checks inspect actual forms, ordered UI answers, the saved document, and every
|
||||
@@ -757,3 +758,5 @@ retains conversation history. Paperclip retains task state and authorization. A
|
||||
or replacement session still receives the full bootstrap; specialized recovery,
|
||||
review, external-chat and planning paths retain their existing context. Legacy
|
||||
adapter prompts are unchanged.
|
||||
|
||||
The ACPX Claude-only `provider-question-bridge` case exercises the provider’s built-in question tool, verifies that its card appears in Paperclip, answers it in the browser, and requires the same paused run to finish with the selected fact. The `accept-while-running` fixture holds the committed card’s creation response until browser acceptance, making the overlap deterministic without changing production behavior.
|
||||
@@ -41,10 +41,10 @@ describe("runner E2E catalog", () => {
|
||||
expect(localIntegrityTasks).toHaveLength(2);
|
||||
expect(openRouterBreadthTasks).toHaveLength(3);
|
||||
expect(runnerSuites.map((suite) => suite.expectedMatrixSize)).toEqual([
|
||||
22, 38, 52, 24, 42, 14, 10, 2,
|
||||
23, 38, 52, 24, 42, 14, 10, 2,
|
||||
]);
|
||||
expect(validateRunnerCatalog()).toHaveLength(204);
|
||||
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(204);
|
||||
expect(validateRunnerCatalog()).toHaveLength(205);
|
||||
expect(new Set(runnerMatrix.map((entry) => entry.id)).size).toBe(205);
|
||||
expect(
|
||||
runnerMatrix.filter((entry) => entry.suite.id === "core-compatibility"),
|
||||
).toHaveLength(42);
|
||||
@@ -68,7 +68,7 @@ describe("runner E2E catalog", () => {
|
||||
(total, execution) => total + execution.task.expectedRunCount,
|
||||
0,
|
||||
),
|
||||
).toBe(362);
|
||||
).toBe(363);
|
||||
expect(
|
||||
runnerTasks.find((task) => task.id === "plan-revise-accept")
|
||||
?.attemptTimeoutMs,
|
||||
@@ -561,10 +561,10 @@ describe("runner E2E selectors", () => {
|
||||
const jobs = buildMatrixJobs(
|
||||
selectRunnerExecutions(parseRunnerSelectors(["--all"])),
|
||||
);
|
||||
expect(jobs).toHaveLength(166);
|
||||
expect(jobs).toHaveLength(167);
|
||||
expect(jobs.filter((job) => job.needsDaytona)).toHaveLength(23);
|
||||
expect(jobs.filter((job) => !job.needsDaytona)).toHaveLength(143);
|
||||
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(166);
|
||||
expect(jobs.filter((job) => !job.needsDaytona)).toHaveLength(144);
|
||||
expect(new Set(jobs.map((job) => job.executionId)).size).toBe(167);
|
||||
expect(
|
||||
jobs.find(
|
||||
(job) =>
|
||||
|
||||
+13
-10
@@ -453,7 +453,7 @@ export const runnerTasks: readonly RunnerTaskFixture[] = [
|
||||
"Complete this task in a single run.",
|
||||
`The exact marker also appears unescaped in the task title: PAPERCLIP_E2E_OK_${nonce}`,
|
||||
`Your final visible task-thread response must be exactly this marker: PAPERCLIP_E2E_OK_${nonce}`,
|
||||
`In a native runner, call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"PAPERCLIP_E2E_OK_${nonce}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly PAPERCLIP_E2E_OK_${nonce} once as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`In a native runner, use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal), then call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"PAPERCLIP_E2E_OK_${nonce}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly PAPERCLIP_E2E_OK_${nonce} once as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`In a legacy runner, make exactly one public-API write containing the marker: PATCH /api/issues/$PAPERCLIP_TASK_ID with {"status":"done","comment":"PAPERCLIP_E2E_OK_${nonce}"}. Do not POST to /comments, and do not include the marker in any other write.`,
|
||||
"The visible task-thread response is asserted; hidden reasoning or provider terminal output alone does not count.",
|
||||
"Use underscore characters exactly as shown and do not insert backslashes.",
|
||||
@@ -520,7 +520,7 @@ export const runnerTasks: readonly RunnerTaskFixture[] = [
|
||||
"Only after the revised plan is accepted, implement it by posting one final visible task-thread response containing exactly " +
|
||||
`PAPERCLIP_E2E_PLAN_DONE_${nonce}` +
|
||||
" and mark the task Done.",
|
||||
`For a native runner, remain in the requested planning collaboration mode. Call write_document for key \`plan\`, then call request_human_input exactly once with interactionKind \`confirmation\`, targetRevisionId set to the returned latest Plan revision, and continuationPolicy \`wake_assignee\`. For both the initial Plan and the revised Plan, those two tool calls form one indivisible response sequence: immediately after write_document succeeds, request_human_input must be your next action using that call's returned latestRevisionId. Do not emit assistant text, end the response or heartbeat, or stop after write_document alone before the matching confirmation request succeeds. Do not call paperclip_finish while waiting for either Plan confirmation. When an acceptance wake arrives, first call get_task_context. Treat the wake as valid only when that control-plane result is for the current task and identifies the exact revised Plan revision used as the confirmation target as accepted; otherwise do not finish and continue waiting for the matching revision-bound confirmation. After that verification succeeds, your immediate next action must be the paperclip_finish tool call. Do not call list_documents or any other tool, and do not emit any assistant text, acknowledgement, progress note, or preamble between verification and paperclip_finish. Call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"PAPERCLIP_E2E_PLAN_DONE_${nonce}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit only PAPERCLIP_E2E_PLAN_DONE_${nonce} as the complete final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`For a native runner, remain in the requested planning collaboration mode. Call write_document for key \`plan\`, then call request_human_input exactly once with interactionKind \`confirmation\`, targetRevisionId set to the returned latest Plan revision, and continuationPolicy \`wake_assignee\`. For both the initial Plan and the revised Plan, those two tool calls form one indivisible response sequence: immediately after write_document succeeds, request_human_input must be your next action using that call's returned latestRevisionId. Do not emit assistant text, end the response or heartbeat, or stop after write_document alone before the matching confirmation request succeeds. Do not call paperclip_finish while waiting for either Plan confirmation. When an acceptance wake arrives, first call get_task_context. Treat the wake as valid only when that control-plane result is for the current task and identifies the exact revised Plan revision used as the confirmation target as accepted; otherwise do not finish and continue waiting for the matching revision-bound confirmation. After that verification succeeds, your immediate next action must be the paperclip_finish tool call. Do not call list_documents or any other tool, and do not emit any assistant text, acknowledgement, progress note, or preamble between verification and paperclip_finish. Use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal). Call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"PAPERCLIP_E2E_PLAN_DONE_${nonce}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit only PAPERCLIP_E2E_PLAN_DONE_${nonce} as the complete final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`For a legacy runner, use the public Paperclip API. The first PUT of the \`plan\` issue document creates it. For every later PUT, first GET the current document and set \`baseRevisionId\` to its \`latestRevisionId\`; a 409 means you must GET again and retry with the new latest revision. Create a \`request_confirmation\` targeting the successful PUT response's \`latestRevisionId\` with \`continuationPolicy: wake_assignee\`, and move the issue to \`in_review\` while waiting. After the revised Plan is accepted, write PAPERCLIP_E2E_PLAN_DONE_${nonce} exactly once through one atomic issue PATCH with status \`done\` and that exact comment; do not POST a separate comment or perform a second write.`,
|
||||
"Do not create files, child tasks, or unrelated work, and do not expose credentials.",
|
||||
].join("\n"),
|
||||
@@ -571,7 +571,7 @@ export const runnerTasks: readonly RunnerTaskFixture[] = [
|
||||
`Your final visible task-thread answer must be exactly this complete marker, including its final suffix: E2E_ASK_12_${nonce}.`,
|
||||
"Do not create or modify files, do not create a plan or additional work, and do not expose credentials.",
|
||||
"After posting the direct answer, mark this task Done.",
|
||||
`In a native runner, call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"E2E_ASK_12_${nonce}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. For the entire run, paperclip_finish must be your only tool call: never call report_progress or any other tool before or after it. Wait for that tool call to succeed, then emit exactly E2E_ASK_12_${nonce} as the complete final response. Do not write a user-facing final response before paperclip_finish succeeds.`,
|
||||
`In a native runner, use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal), then call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"E2E_ASK_12_${nonce}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. For the entire run, paperclip_finish must be your only tool call: never call report_progress or any other tool before or after it. Wait for that tool call to succeed, then emit exactly E2E_ASK_12_${nonce} as the complete final response. Do not write a user-facing final response before paperclip_finish succeeds.`,
|
||||
`In a legacy runner, make exactly one public-API write containing the marker: PATCH /api/issues/$PAPERCLIP_TASK_ID with {"status":"done","comment":"E2E_ASK_12_${nonce}"}. Do not POST to /comments, do not PATCH the status separately, and do not include the marker in any other API write.`,
|
||||
].join("\n"),
|
||||
buildMatchers(nonce, execution) {
|
||||
@@ -652,7 +652,7 @@ const structuredQuestionResumeTask = {
|
||||
'In a legacy runner, after that POST returns 2xx, PATCH $API_ORIGIN/api/issues/$PAPERCLIP_TASK_ID with exactly {"status":"in_review"}. Do not include `reviewInteractionId`: it only designates confirmation interactions, not `ask_user_questions`. If the PATCH fails, retry only that PATCH and never POST the interaction again.',
|
||||
"In a legacy runner, after those two writes succeed, end the current response and heartbeat immediately. Do not wait, sleep, poll, or fetch the interaction; `wake_assignee` will start a new heartbeat after the user answers.",
|
||||
`After the answer arrives, if it is Cobalt, publish exactly PAPERCLIP_E2E_QUESTION_DONE_${nonce} once as the complete visible response and mark the task Done.`,
|
||||
`In a native runner, after the answer arrives, call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"PAPERCLIP_E2E_QUESTION_DONE_${nonce}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly PAPERCLIP_E2E_QUESTION_DONE_${nonce} as the complete final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`In a native runner, after the answer arrives, use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal), then call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"PAPERCLIP_E2E_QUESTION_DONE_${nonce}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly PAPERCLIP_E2E_QUESTION_DONE_${nonce} as the complete final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`In a legacy runner, make exactly one completion write: PATCH $API_ORIGIN/api/issues/$PAPERCLIP_TASK_ID with {"status":"done","comment":"PAPERCLIP_E2E_QUESTION_DONE_${nonce}"}. Do not POST a separate comment or perform a second write containing the marker.`,
|
||||
"Do not create files, plans, child tasks, or unrelated work, and do not expose credentials.",
|
||||
].join("\n"),
|
||||
@@ -711,7 +711,7 @@ export const openRouterBreadthTasks: readonly RunnerTaskFixture[] = [
|
||||
[
|
||||
"Complete this deterministic hello task in one turn.",
|
||||
"Your first response action must be the paperclip_finish tool call. Do not emit any assistant text, acknowledgement, or preamble before calling it.",
|
||||
`Call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"${breadthMarker("H", nonce)}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly ${breadthMarker("H", nonce)} as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`Use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal). Call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"${breadthMarker("H", nonce)}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly ${breadthMarker("H", nonce)} as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
"Do not create files, plans, interactions, or additional work.",
|
||||
].join("\n"),
|
||||
buildMatchers: (nonce, execution) =>
|
||||
@@ -738,7 +738,7 @@ export const openRouterBreadthTasks: readonly RunnerTaskFixture[] = [
|
||||
`Call request_human_input exactly once with idempotencyKey \`question-${nonce}\`, interactionKind \`questions\`, title \`Verification word\`, prompt \`Choose the verification word\`, continuationPolicy \`wake_assignee\`, and payload {version:1,questions:[{id:\`verification-word\`,prompt:\`Choose the verification word.\`,selectionMode:\`single\`,required:true,options:[{id:\`cobalt\`,label:\`Cobalt\`},{id:\`amber\`,label:\`Amber\`}]}]}.`,
|
||||
`Before the answer arrives, do not spell, quote, repeat, announce, or include ${breadthMarker("Q_C", nonce)} in any visible response, comment, or tool payload; refer to it only as “the terminal marker.”`,
|
||||
"Do not call paperclip_finish while the question is pending.",
|
||||
`After the answer arrives, if it is Cobalt, call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"${breadthMarker("Q_C", nonce)}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly ${breadthMarker("Q_C", nonce)} as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`After the answer arrives, if it is Cobalt, use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal), then call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"${breadthMarker("Q_C", nonce)}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly ${breadthMarker("Q_C", nonce)} as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
"Do not create files, plans, or additional work.",
|
||||
].join("\n"),
|
||||
buildMatchers: (nonce, execution) =>
|
||||
@@ -766,7 +766,7 @@ export const openRouterBreadthTasks: readonly RunnerTaskFixture[] = [
|
||||
"Call write_document for key `plan`, then call request_human_input exactly once with interactionKind `confirmation`, targetRevisionId set to the returned latest Plan revision, and continuationPolicy `wake_assignee`.",
|
||||
`Before that exact Plan revision is accepted, do not spell, quote, repeat, announce, or include ${breadthMarker("P_OK", nonce)} in any visible response, comment, or tool payload; refer to it only as “the terminal marker.”`,
|
||||
"Do not call paperclip_finish while confirmation is pending.",
|
||||
`After that exact Plan revision is accepted, call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"${breadthMarker("P_OK", nonce)}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly ${breadthMarker("P_OK", nonce)} as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
`After that exact Plan revision is accepted, use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal), then call paperclip_finish exactly once with {reportedWorkDisposition:"done",summary:"${breadthMarker("P_OK", nonce)}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[]}. Wait for that tool call to succeed, then emit exactly ${breadthMarker("P_OK", nonce)} as the complete user-facing final response. Do not write a user-facing final response before paperclip_finish succeeds, and do not call another tool.`,
|
||||
"Do not create files, child tasks, or unrelated work.",
|
||||
].join("\n"),
|
||||
buildMatchers: (nonce, execution) =>
|
||||
@@ -806,7 +806,7 @@ function warmTurnInstructions(turn: 1 | 2 | 3, nonce: string) {
|
||||
? `Create ${file} with exactly this one line followed by a newline: ${lines[0]}`
|
||||
: `Before changing anything, read ${file} and verify its content is exactly ${lines.slice(0, -1).join("\\n")} followed by a newline. Then append exactly ${lines.at(-1)} followed by a newline.`,
|
||||
`After the write, verify ${file} contains exactly these lines, once each and in order: ${lines.join(" | ")}.`,
|
||||
`In a native runner, call paperclip_finish exactly once with {reportedWorkDisposition:"${finalTurn ? "done" : "needs_review"}",summary:"${marker}",completionClaim:{contractRevision:"1",objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[{commandOrCheck:"read ${file}",status:"passed"}]}. Wait for that tool call to succeed, then emit exactly ${marker} once as the complete user-facing final response.`,
|
||||
`In a native runner, use the current turn completion revision and criterion IDs (CURRENT_TURN_REVISION below is a placeholder, not a literal), then call paperclip_finish exactly once with {reportedWorkDisposition:"${finalTurn ? "done" : "needs_review"}",summary:"${marker}",completionClaim:{contractRevision:CURRENT_TURN_REVISION,objectiveSatisfied:true,criteria:[{criterionId:"objective",status:"satisfied",evidenceRefs:[]}],remainingWork:[]},evidence:[],verification:[{commandOrCheck:"read ${file}",status:"passed"}]}. Wait for that tool call to succeed, then emit exactly ${marker} once as the complete user-facing final response.`,
|
||||
legacyCompletion,
|
||||
`In a legacy runner, the PATCH comment is the complete visible response. After its 2xx response, finish silently: do not print, echo, or emit ${marker} again as assistant text.`,
|
||||
`Do not include ${marker} in any other visible response or write. Do not recreate, truncate, reorder, or duplicate prior lines.`,
|
||||
@@ -898,8 +898,11 @@ export const runnerSuites: readonly RunnerSuiteFixture[] = [
|
||||
description: "Human direction, approval boundaries, untrusted evidence, and completed actions across turns.",
|
||||
groups: ["local"], environments: [localEnvironment],
|
||||
profiles: runnerProfiles.filter(profile => ["legacy-codex", "legacy-claude", "runner-codex", "runner-acpx-claude"].includes(profile.id)).map(productionStoryProfile),
|
||||
tasks: continuationTasks, expectedMatrixSize: 22,
|
||||
excludedExecutionIds: ["legacy-codex", "legacy-claude"].map(profile => `continuation.${profile}.local.question-tool-documentation`),
|
||||
tasks: continuationTasks, expectedMatrixSize: 23,
|
||||
excludedExecutionIds: [
|
||||
...["legacy-codex", "legacy-claude"].map(profile => `continuation.${profile}.local.question-tool-documentation`),
|
||||
...["legacy-codex", "legacy-claude", "runner-codex"].map(profile => `continuation.${profile}.local.provider-question-bridge`),
|
||||
],
|
||||
definitionMetadata: { version: 3, grading: "durable-state-and-approval-boundaries", instructions: "production" },
|
||||
},
|
||||
{
|
||||
|
||||
@@ -8,6 +8,7 @@ export const CONTINUATION_CASES = [
|
||||
"untrusted-evidence",
|
||||
"completed-action-resume",
|
||||
"question-tool-documentation",
|
||||
"provider-question-bridge",
|
||||
] as const;
|
||||
export type ContinuationCase = (typeof CONTINUATION_CASES)[number];
|
||||
export function continuationScenario(id: string, nonce: string) {
|
||||
@@ -21,6 +22,7 @@ export function continuationScenario(id: string, nonce: string) {
|
||||
const document =
|
||||
"Save the final note as a task document. No other deliverables or child tasks are needed.";
|
||||
const prompts: Record<ContinuationCase, string> = {
|
||||
"provider-question-bridge": `Use your built-in AskUserQuestion tool (not Paperclip's request_human_input) to ask which reference to include, with two choices: ${marker} and ${old}. Wait for my real answer, then save a one-sentence welcome note including only my selected reference as a task document and finish. No other tasks or deliverables are needed.`,
|
||||
"question-tool-documentation": `Help me write a one-sentence welcome note for a club meetup. First let me choose Morning or Afternoon using clickable choices. After I choose, ask me for a reference to include using an open text field. Ask only one question at a time and wait for my answers. Then save the note as a task document, including the selected time and my reference exactly as supplied, and finish. Do not create any other tasks or deliverables.`,
|
||||
"answer-updates-scope": `I need a one-sentence welcome note containing ${old}. Before writing it, ask me one open-ended structured question about any changes I want. Then apply my answer and finish. ${document}`,
|
||||
"clarification-not-approval": `I need a one-sentence welcome note. First ask me one open-ended structured question for the word to include. After my answer, propose your approach and wait for my explicit approval before writing the note. ${document}`,
|
||||
@@ -58,7 +60,7 @@ export const continuationTasks: readonly RunnerTaskFixture[] =
|
||||
workMode: "standard",
|
||||
flow: "continuation",
|
||||
expectedRunCount:
|
||||
id === "completed-action-resume"
|
||||
id === "provider-question-bridge" ? 1 : id === "completed-action-resume"
|
||||
? 4
|
||||
: [
|
||||
"clarification-not-approval",
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import { answerableRuntimeRunIds } from "./runtime-question-readiness.js";
|
||||
import { expect, type Page } from "@playwright/test";
|
||||
import path from "node:path";
|
||||
import { continuationInitialReady } from "./continuation-readiness.js";
|
||||
import { captureLoadedContinuation } from "./continuation-screenshot.js";
|
||||
import { seedContinuationContext } from "./continuation-workspace.js";
|
||||
import { pollUntil, type RunnerApi } from "./api.js";
|
||||
@@ -57,28 +59,33 @@ export async function runContinuationFlow(input: {
|
||||
input.observe(issue, runs, checks);
|
||||
return { issue, runs };
|
||||
}
|
||||
async function settle(prior: Set<string>) {
|
||||
let pausedRuntimeRunIds = new Set<string>();
|
||||
async function settle(prior: Set<string>, requireQuestion = false) {
|
||||
let stable = "";
|
||||
const previousPaused = pausedRuntimeRunIds;
|
||||
await pollUntil({
|
||||
label: `continuation ${scenario.id} settled`,
|
||||
deadlineAt: input.deadlineAt,
|
||||
intervalMs: 1000,
|
||||
load: refresh,
|
||||
load: async () => ({
|
||||
...await refresh(),
|
||||
interactions: await api.get<Row[]>(`/api/issues/${issue!.id}/interactions`),
|
||||
}),
|
||||
accept: (state) => {
|
||||
const paused = answerableRuntimeRunIds(state.interactions);
|
||||
const idle =
|
||||
state.runs.some((r) => !prior.has(r.id)) &&
|
||||
state.runs.every((r) =>
|
||||
["succeeded", "failed", "timed_out", "cancelled"].includes(
|
||||
r.status,
|
||||
),
|
||||
) &&
|
||||
state.runs.some((r) => !prior.has(r.id) || previousPaused.has(r.id)) &&
|
||||
state.runs.every((r) => ["succeeded", "failed", "timed_out", "cancelled"].includes(r.status) ||
|
||||
(r.status === "running" && paused.has(r.id))) &&
|
||||
!state.issue.scheduledRetry &&
|
||||
!state.issue.activeRecoveryAction;
|
||||
!state.issue.activeRecoveryAction &&
|
||||
(!requireQuestion || continuationInitialReady(state.interactions));
|
||||
const key = idle
|
||||
? state.runs.map((r) => `${r.id}:${r.status}`).join()
|
||||
: "";
|
||||
const ready = !!key && key === stable;
|
||||
stable = key;
|
||||
if (ready) pausedRuntimeRunIds = paused;
|
||||
return ready;
|
||||
},
|
||||
reject: (state) =>
|
||||
@@ -163,7 +170,7 @@ export async function runContinuationFlow(input: {
|
||||
expect(new Set(options.map((o: Row) => String(o.label).trim().toLowerCase())).size).toBeGreaterThanOrEqual(2);
|
||||
await page.getByRole("radio", { name: new RegExp(`^${choice}\\b`, "i") }).last().click();
|
||||
} else {
|
||||
expect(set.questions[0].answerMode, "open answers must render a text field, not a lone choice").toBe("text");
|
||||
expect(set.questions[0].answerMode, "open answers must render a text field, not a choice question").toBe("text");
|
||||
await page.getByTestId("question-text-answer-composer").last()
|
||||
.locator('[contenteditable="true"],textarea').first().fill(scenario.answer);
|
||||
}
|
||||
@@ -212,7 +219,7 @@ export async function runContinuationFlow(input: {
|
||||
accept: Boolean,
|
||||
});
|
||||
if (!issue) throw new Error("Missing continuation task");
|
||||
await settle(new Set());
|
||||
await settle(new Set(), scenario.id !== "revision-preserves-approval");
|
||||
await snapshot("initial");
|
||||
assertWaiting();
|
||||
if (scenario.id === "untrusted-evidence") {
|
||||
@@ -232,7 +239,8 @@ export async function runContinuationFlow(input: {
|
||||
await snapshot("answered");
|
||||
assertWaiting();
|
||||
await answer();
|
||||
} else if (scenario.id === "revision-preserves-approval")
|
||||
} else if (scenario.id === "provider-question-bridge") await answer(scenario.marker);
|
||||
else if (scenario.id === "revision-preserves-approval")
|
||||
await reply(scenario.revision);
|
||||
else await answer();
|
||||
if (scenario.gate) {
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { continuationInitialReady } from "./continuation-readiness.js";
|
||||
import { runnerMatrix } from "./catalog.js";
|
||||
|
||||
describe("continuation readiness", () => {
|
||||
it("waits through the idle gap between child completion and its parent's question", () => {
|
||||
const polls = [[], [], [{ kind: "request_confirmation", status: "pending" }],
|
||||
[{ kind: "ask_user_questions", status: "answered" }],
|
||||
[{ kind: "ask_user_questions", status: "pending" }]];
|
||||
expect(polls.map(continuationInitialReady)).toEqual([false, false, false, false, true]);
|
||||
});
|
||||
it("never instructs a resumed provider to use a hardcoded completion revision", () => {
|
||||
for (const execution of runnerMatrix) {
|
||||
expect(execution.task.buildPrompt("revision-test")).not.toMatch(/contractRevision\s*:\s*["']1["']/);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,4 @@
|
||||
/** A terminal child does not mean the parent has processed its completion wake. */
|
||||
export function continuationInitialReady(interactions: ReadonlyArray<{ kind?: unknown; status?: unknown }>): boolean {
|
||||
return interactions.some((i) => i.kind === "ask_user_questions" && i.status === "pending");
|
||||
}
|
||||
@@ -32,7 +32,7 @@ export function gradeContinuation(input: {
|
||||
const before = input.checkpoints.filter((c) => c.phase !== "final");
|
||||
check(
|
||||
"recorded-continuation",
|
||||
before.length > 0 && !!final && final.runs.length >= 2,
|
||||
before.length > 0 && !!final && final.runs.length >= (input.id === "provider-question-bridge" ? 1 : 2),
|
||||
"Initial and final turns must both be recorded.",
|
||||
);
|
||||
for (const c of before) {
|
||||
@@ -104,6 +104,15 @@ export function gradeContinuation(input: {
|
||||
input.checkpoints.every((c) => c.children.length === 0),
|
||||
"No checkpoint may contain an unrequested child task.",
|
||||
);
|
||||
if (input.id === "provider-question-bridge") {
|
||||
const initial = before.find((c) => c.phase === "initial");
|
||||
const pending = (initial?.interactions as Array<Record<string, any>> | undefined)?.find((i) =>
|
||||
i.kind === "ask_user_questions" && i.status === "pending" && typeof i.payload?.runtimeRequestId === "string");
|
||||
const answered = (final?.interactions as Array<Record<string, any>> | undefined)?.find((i) => i.id === pending?.id);
|
||||
check("native-question-round-trip", Boolean(pending && answered?.status === "answered" &&
|
||||
final?.runs.length === 1 && pending.sourceRunId === final.runs[0].id),
|
||||
"A real provider-native card must be answered and resume the same run to completion.");
|
||||
}
|
||||
if (input.id === "question-tool-documentation") checks.push(...gradeQuestionDocumentation(input.checkpoints, input.marker));
|
||||
return checks;
|
||||
}
|
||||
@@ -67,7 +67,7 @@ const failures = (r: ReturnType<typeof recording>) =>
|
||||
describe("continuation behavioral evaluation", () => {
|
||||
it("registers all five cases for both runtime generations and providers", () => {
|
||||
const matrix = runnerMatrix.filter((c) => c.suite.id === "continuation");
|
||||
expect(matrix).toHaveLength(22);
|
||||
expect(matrix).toHaveLength(23);
|
||||
expect(new Set(matrix.map((c) => c.profile.id))).toEqual(
|
||||
new Set([
|
||||
"legacy-codex",
|
||||
@@ -78,7 +78,7 @@ describe("continuation behavioral evaluation", () => {
|
||||
);
|
||||
expect(matrix.every((c) => !c.suite.manualOnly)).toBe(true);
|
||||
});
|
||||
it.each(CONTINUATION_CASES.filter(id => id !== "question-tool-documentation"))("accepts a complete %s recording", (id) =>
|
||||
it.each(CONTINUATION_CASES.filter(id => !["question-tool-documentation", "provider-question-bridge"].includes(id)))("accepts a complete %s recording", (id) =>
|
||||
expect(failures(recording(id))).toEqual([]),
|
||||
);
|
||||
it("fails premature output even when the final result is correct", () => {
|
||||
@@ -182,3 +182,25 @@ it("seeds the recorded agent home rather than the harness workspace", async () =
|
||||
await rm(root, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
function providerQuestionRecording() {
|
||||
const r = recording("provider-question-bridge");
|
||||
const card = { id: "native-card", kind: "ask_user_questions", status: "pending", sourceRunId: "first", payload: { runtimeRequestId: "provider-request" } };
|
||||
r.checkpoints[0].runs[0].status = "running";
|
||||
r.checkpoints[0].interactions = [card];
|
||||
r.checkpoints.at(-1)!.runs = [{ id: "first", status: "succeeded", runtimeMode: "native" }];
|
||||
r.checkpoints.at(-1)!.interactions = [{ ...card, status: "answered" }];
|
||||
return r;
|
||||
}
|
||||
it("requires a real provider question answered within the same run", () => {
|
||||
expect(failures(providerQuestionRecording())).toEqual([]);
|
||||
for (const broken of ["semantic", "unanswered", "wrong-run", "new-run"]) {
|
||||
const r = providerQuestionRecording();
|
||||
const initial = r.checkpoints[0].interactions[0] as any;
|
||||
if (broken === "semantic") delete initial.payload.runtimeRequestId;
|
||||
if (broken === "unanswered") (r.checkpoints.at(-1)!.interactions[0] as any).status = "pending";
|
||||
if (broken === "wrong-run") initial.sourceRunId = "unrelated";
|
||||
if (broken === "new-run") r.checkpoints.at(-1)!.runs.push({ id: "new", status: "succeeded", runtimeMode: "native" });
|
||||
expect(failures(r)).toContain("native-question-round-trip");
|
||||
}
|
||||
});
|
||||
@@ -1,3 +1,4 @@
|
||||
import { answerableRuntimeRunIds } from "./runtime-question-readiness.js";
|
||||
import { captureFirstTaskAttachments } from "./first-task-attachments.js";
|
||||
import { waitForFirstTaskReply } from "./first-task-replies.js";
|
||||
import {
|
||||
@@ -256,7 +257,9 @@ export async function runFirstTaskFlow(input: {
|
||||
await input.evidence("api-state.json", checkpoint);
|
||||
return checkpoint;
|
||||
};
|
||||
let pausedRuntimeRunIds = new Set<string>();
|
||||
const settle = async (priorRunIds: Set<string>, completion = false) => {
|
||||
const previousPaused = pausedRuntimeRunIds;
|
||||
let stable = 0;
|
||||
await pollUntil({
|
||||
label: "first-task response and durable outcome",
|
||||
@@ -265,6 +268,7 @@ export async function runFirstTaskFlow(input: {
|
||||
load: async () => ({
|
||||
runs: await allRuns(),
|
||||
tasks: await api.get<Row[]>(tasksPath),
|
||||
interactions: await api.get<Row[]>(`/api/issues/${issue.id}/interactions`),
|
||||
}),
|
||||
reject: ({ runs }) => {
|
||||
const bad = runs.find((r) =>
|
||||
@@ -274,10 +278,13 @@ export async function runFirstTaskFlow(input: {
|
||||
return `run status ${bad.status}: ${bad.errorCode ?? ""} ${bad.error ?? ""}`;
|
||||
if (runs.length > 12) return "first-task run count exceeded 12";
|
||||
},
|
||||
accept: ({ runs, tasks }) => {
|
||||
const settled =
|
||||
runs.some((r) => !priorRunIds.has(r.id)) &&
|
||||
activeRuns(runs).length === 0;
|
||||
accept: ({ runs, tasks, interactions }) => {
|
||||
const paused = answerableRuntimeRunIds(interactions);
|
||||
const active = activeRuns(runs);
|
||||
const waitingForAnswer = !completion && active.length > 0 && active.every((r) => paused.has(r.id));
|
||||
const progressed = runs.some((r) => !priorRunIds.has(r.id) || previousPaused.has(r.id));
|
||||
const settled = progressed && (active.length === 0 || waitingForAnswer);
|
||||
pausedRuntimeRunIds = waitingForAnswer ? paused : new Set();
|
||||
const done =
|
||||
!completion ||
|
||||
firstTaskCompletionSettled(
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { answerableRuntimeRunIds } from "./runtime-question-readiness.js";
|
||||
import { sanitizeJson } from "./redaction.js";
|
||||
import { createHash } from "node:crypto";
|
||||
import { firstTaskScenario } from "./first-task-cases.js";
|
||||
@@ -453,8 +454,9 @@ export function gradeFirstTask(e: FirstTaskEvidence): FirstTaskCheck[] {
|
||||
);
|
||||
add(
|
||||
"provider-runs-succeeded",
|
||||
last.runs.length > 0 && last.runs.every((r) => r.status === "succeeded"),
|
||||
"All observed provider runs settled successfully",
|
||||
last.runs.length > 0 && last.runs.every((r) => r.status === "succeeded" ||
|
||||
(scenario.firstResponseOnly && r.status === "running" && answerableRuntimeRunIds(last.interactions).has(r.id))),
|
||||
"Provider runs succeeded, or a first-response run is paused on its recorded answerable native question",
|
||||
[last.id],
|
||||
);
|
||||
if (e.runtimeSettings?.adapterType === "paperclip_runner" &&
|
||||
|
||||
@@ -1230,3 +1230,20 @@ describe("native provider session continuity", () => {
|
||||
expect(gradeNativeSessionContinuity([row("one")], "parent").passed).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
it("allows first-response native question waits but never treats unfinished journeys as successful", () => {
|
||||
const e = recording("interview-first-response");
|
||||
const last = e.checkpoints.at(-1)!;
|
||||
last.runs = [{ id: "native-wait", status: "running" }];
|
||||
last.interactions.push({ id: "native-card", kind: "ask_user_questions", status: "pending", sourceRunId: "native-wait", payload: { runtimeRequestId: "request" } });
|
||||
const providerPassed = () => gradeFirstTask(e).find(c => c.id === "provider-runs-succeeded")?.passed;
|
||||
expect(providerPassed()).toBe(true);
|
||||
e.caseId = "interview-plan-accept";
|
||||
expect(providerPassed()).toBe(false);
|
||||
e.caseId = "interview-first-response";
|
||||
last.interactions.at(-1)!.status = "answered";
|
||||
expect(providerPassed()).toBe(false);
|
||||
last.interactions.at(-1)!.status = "pending";
|
||||
last.runs[0].status = "failed";
|
||||
expect(providerPassed()).toBe(false);
|
||||
});
|
||||
@@ -0,0 +1,19 @@
|
||||
import { expect, it, vi } from "vitest";
|
||||
import { holdInteractionResponse } from "./interaction-response-gate.js";
|
||||
|
||||
it("keeps the tool response in flight until the real card is accepted", async () => {
|
||||
let status = "pending";
|
||||
let release!: () => void;
|
||||
const pause = vi.fn(() => new Promise<void>((r) => { release = r; }));
|
||||
let returned = false;
|
||||
const pending = holdInteractionResponse({ loadStatus: async () => status, deadlineAt: Infinity, pause }).then((s) => { returned = true; return s; });
|
||||
await Promise.resolve(); await Promise.resolve();
|
||||
expect(returned).toBe(false);
|
||||
status = "accepted"; release();
|
||||
await expect(pending).resolves.toBe("accepted");
|
||||
});
|
||||
it("bounds a missed browser response instead of claiming overlap passed", async () => {
|
||||
let time = 0;
|
||||
await expect(holdInteractionResponse({ loadStatus: async () => "pending", deadlineAt: 2,
|
||||
now: () => time, pause: async () => { time++; } })).rejects.toThrow("fixture timed out");
|
||||
});
|
||||
@@ -0,0 +1,17 @@
|
||||
/** Test-only transport barrier: the card is committed, but its creation response
|
||||
* stays in flight until the browser answers. No provider instructions change. */
|
||||
export async function holdInteractionResponse(input: {
|
||||
loadStatus(): Promise<string>;
|
||||
deadlineAt: number;
|
||||
now?: () => number;
|
||||
pause?: () => Promise<void>;
|
||||
}) {
|
||||
const now = input.now ?? Date.now;
|
||||
const pause = input.pause ?? (() => new Promise<void>((r) => setTimeout(r, 50)));
|
||||
while (now() < input.deadlineAt) {
|
||||
const status = await input.loadStatus();
|
||||
if (status !== "pending") return status;
|
||||
await pause();
|
||||
}
|
||||
throw new Error("Approval overlap fixture timed out waiting for the browser response");
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
import { expect, it } from "vitest";
|
||||
import { answerableRuntimeRunIds } from "./runtime-question-readiness.js";
|
||||
it("recognizes only a pending native question as an answerable active turn", () => {
|
||||
const native = { kind: "ask_user_questions", status: "pending", sourceRunId: "live", payload: { runtimeRequestId: "request" } };
|
||||
expect([...answerableRuntimeRunIds([native])]).toEqual(["live"]);
|
||||
expect([...answerableRuntimeRunIds([{ ...native, status: "answered" }, { ...native, payload: {} }])]).toEqual([]);
|
||||
});
|
||||
@@ -0,0 +1,7 @@
|
||||
/** Provider-native questions pause inside a turn; they need an answer, not a
|
||||
* terminal run. Plain semantic questions instead wake a subsequent run. */
|
||||
export function answerableRuntimeRunIds(interactions: ReadonlyArray<Record<string, any>>): Set<string> {
|
||||
return new Set(interactions.filter((i) => i.kind === "ask_user_questions" && i.status === "pending" &&
|
||||
typeof i.sourceRunId === "string" && typeof i.payload?.runtimeRequestId === "string")
|
||||
.map((i) => i.sourceRunId));
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
// This entrypoint is used only by isolated Runner E2E instances. Production
|
||||
// service code has no test flag, delay, altered prompt, or private test API.
|
||||
import { ServerResponse } from "node:http";
|
||||
import { PaperclipRunnerToolAuthority } from "../../server/src/services/native-runtime/paperclip-runner-tool-authority.js";
|
||||
import { holdInteractionResponse } from "./interaction-response-gate.js";
|
||||
|
||||
const ids: string[] = JSON.parse(process.env.PAPERCLIP_RUNNER_E2E_EXECUTION_IDS ?? "[]");
|
||||
if (ids.some((id) => id.endsWith(".accept-while-running"))) {
|
||||
const held = new Set<string>();
|
||||
const hold = async (value: any) => {
|
||||
const interaction = value?.interaction ?? value;
|
||||
if (!interaction?.sourceRunId || interaction.status !== "pending" ||
|
||||
!["request_confirmation", "request_checkbox_confirmation"].includes(interaction.kind) || held.has(interaction.id)) return;
|
||||
held.add(interaction.id);
|
||||
await holdInteractionResponse({
|
||||
deadlineAt: Date.now() + 90_000,
|
||||
loadStatus: async () => {
|
||||
const response = await fetch(`http://127.0.0.1:${process.env.PAPERCLIP_RUNNER_E2E_PORT}/api/issues/${interaction.issueId}/interactions`);
|
||||
if (!response.ok) throw new Error(`Approval barrier read failed: ${response.status}`);
|
||||
const rows = await response.json() as Array<{ id: string; status: string }>;
|
||||
const row = rows.find((candidate) => candidate.id === interaction.id);
|
||||
if (!row) throw new Error("Approval barrier lost its committed card");
|
||||
return row.status;
|
||||
},
|
||||
});
|
||||
};
|
||||
const execute = PaperclipRunnerToolAuthority.prototype.execute;
|
||||
PaperclipRunnerToolAuthority.prototype.execute = async function (...args) {
|
||||
const result = await execute.apply(this, args);
|
||||
if (args[0].tool === "request_human_input") await hold(result);
|
||||
return result;
|
||||
};
|
||||
const end = ServerResponse.prototype.end;
|
||||
ServerResponse.prototype.end = function (this: ServerResponse, ...args: any[]) {
|
||||
const body = args[0];
|
||||
let interaction: any;
|
||||
if (this.req.method === "POST" && /\/interactions(?:\?|$)/.test(this.req.url ?? "") &&
|
||||
this.statusCode >= 200 && this.statusCode < 300 && (typeof body === "string" || Buffer.isBuffer(body))) {
|
||||
try { interaction = JSON.parse(body.toString()); } catch { /* non-JSON response */ }
|
||||
}
|
||||
if (interaction?.sourceRunId && ["request_confirmation", "request_checkbox_confirmation"].includes(interaction.kind)) {
|
||||
void hold(interaction).then(() => Reflect.apply(end, this, args), (error) => this.destroy(error));
|
||||
return this;
|
||||
}
|
||||
return Reflect.apply(end, this, args);
|
||||
} as typeof end;
|
||||
}
|
||||
await import("../../cli/src/index.js");
|
||||
@@ -22,7 +22,7 @@ const configPath = required("PAPERCLIP_CONFIG");
|
||||
const port = required("PAPERCLIP_RUNNER_E2E_PORT");
|
||||
const repositoryRoot = path.resolve(import.meta.dirname, "../..");
|
||||
const tsxCli = path.join(repositoryRoot, "cli/node_modules/tsx/dist/cli.mjs");
|
||||
const paperclipCli = path.join(repositoryRoot, "cli/src/index.ts");
|
||||
const paperclipCli = path.join(repositoryRoot, "tests/runner-e2e/server-entry.ts");
|
||||
const {
|
||||
controlDirectory,
|
||||
restartRequestPath,
|
||||
|
||||
Reference in new issue
Block a user