diff --git a/doc/execution-semantics.md b/doc/execution-semantics.md index fe94603117..aac00b0244 100644 --- a/doc/execution-semantics.md +++ b/doc/execution-semantics.md @@ -656,6 +656,21 @@ An issue monitor is a one-shot deferred action path for agent-owned issues in `i Use a monitor when the current assignee owns a future check against an async system or external service. Examples include Greptile review loops, GitHub checks, Vercel deployments, or provider jobs where the agent should come back later and decide what happens next. +Native runners in standard execution use `set_task_monitor({ taskId?, idempotencyKey, monitor })`. Omit `taskId` for the current task. An explicit target must be accessible, in the same company, assigned to the caller with no human assignee, and `in_progress` or `in_review`. Runtime-management permission still applies. Ask, planning, and review-only runs cannot use this tool. Generic `call_api` lifecycle restrictions remain in force. + +To wait for a check: + +1. Set a future `monitor.nextCheckAt` and short `monitor.notes` describing the check, with optional service context and bounds. +2. Confirm the receipt's persisted task ID, monitor state, next-check time, and bounds. +3. Call `paperclip_finish` with `reportedWorkDisposition: "yielded"` and `continuation: { kind: "monitor", summary, idempotencyKey }`. Report outstanding work honestly; the scheduled check may still block completion. +4. End the run. Paperclip keeps the task active, releases execution ownership, and the one-shot scheduler later wakes it with `issue_monitor_due`. This does not enqueue an immediate continuation. On resume, `get_task_context.activeTask.monitor` includes the consumed monitor’s notes and attempt count. + +Only a valid persisted monitor on the current task authorizes that finish; scheduling another owned task does not. Authority is checked again under the final disposition lock. A timer that becomes due during an active native execution remains scheduled until execution releases it. Dispatch checks the current schedule, claim, status, and assignee so an older dispatch cannot clear a replacement monitor. + +The service name `AI provider quota` is reserved for server-owned recovery of legacy runs. Native task monitors reject it; use ordinary service context and notes when scheduling a provider-usage check. + +Use a new idempotency key to replace the schedule or clear it with `monitor: null`. Retrying the original call returns current monitor state without re-arming a consumed, replaced, or cleared timer. Monitor changes, audit activity, and mutation receipts are committed together, and unrelated execution/review policy is preserved. Legacy agents continue to use the issue API. + Monitor policy lives under `executionPolicy.monitor` and includes: - `nextCheckAt`: when Paperclip should wake the assignee diff --git a/doc/runner-api-tools.md b/doc/runner-api-tools.md index 62451ba6b2..a8dd849d49 100644 --- a/doc/runner-api-tools.md +++ b/doc/runner-api-tools.md @@ -1,7 +1,10 @@ # Runner API escape hatch `search_api` and `call_api` extend the native runner when an available dedicated -operation cannot express the requested work. Existing tools remain preferred; +operation cannot express the requested work. Use `set_task_monitor` to schedule or clear a one-shot task check; after confirming +the receipt, finish with `yielded` and `continuation.kind: "monitor"` to await that +task’s timer. `call_api` still rejects monitor/execution-policy lifecycle writes. +Existing tools remain preferred; agents do not have to search before using them. Only two tool definitions are advertised. The API catalog is returned on demand, never injected into the initial prompt. diff --git a/packages/paperclip-runner/generated/capability/semantic-tool-contracts.json b/packages/paperclip-runner/generated/capability/semantic-tool-contracts.json index 1eb34eb5c5..22a57d2edd 100644 --- a/packages/paperclip-runner/generated/capability/semantic-tool-contracts.json +++ b/packages/paperclip-runner/generated/capability/semantic-tool-contracts.json @@ -1 +1 @@ -[{"annotations":{"exposure":"always","operationId":"get_task_context","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the active task and actor, including the exact approved Markdown revision when this issue has an accepted plan.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"get_task_context","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"get_task_history","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded comments on the active task.","inputSchema":{"additionalProperties":false,"properties":{"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":[],"type":"object"},"name":"get_task_history","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_documents","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List revisioned documents on the active task.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_documents","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"read_document","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the current revision of one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"}},"required":["key"],"type":"object"},"name":"read_document","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_document_revisions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded revision history for one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"},"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":["key"],"type":"object"},"name":"list_document_revisions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"report_progress","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append a durable progress comment to the active task.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Multiline progress update.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"report_progress","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"answer_status_question","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append the answer to a status-only wake without changing task disposition.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Concise status answer.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"answer_status_question","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"write_document","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create or update an active-task document with optimistic revision safety.","inputSchema":{"additionalProperties":false,"properties":{"baseRevisionId":{"description":"Current revision id, or null when creating.","maxLength":20000,"type":["string","null"]},"body":{"description":"Markdown document body.","maxLength":200000,"minLength":1,"type":"string"},"changeSummary":{"description":"Optional revision summary.","maxLength":20000,"type":["string","null"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"},"title":{"description":"Document title.","maxLength":300,"minLength":1,"type":"string"}},"required":["idempotencyKey","key","title","body","baseRevisionId"],"type":"object"},"name":"write_document","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"request_human_input","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a durable human question or approval card on the current Paperclip task bound to this run; Paperclip renders it and authenticates the response. Use questions with continuationPolicy='wake_assignee' when an answer is needed, including otherwise tool-free chat turns. Supply a stable idempotencyKey and reuse it on retries. For one question at a time, ask only the next unanswered question and wait for its real answer. Never fabricate answers or treat ambiguous clarification as approval. For an ordinary confirmation or checkbox card, record a clear user chat answer with call_api POST /api/issues/{id}/interactions/{interactionId}/resolve-from-comment, using the source commentId and decision (accept/reject), plus explicit selectedOptionIds for checkbox acceptance. Existing resolver permissions still apply; governed tool, secret, and connection approvals are excluded. Question forms retain their dedicated answer workflow. Preserve existing review gates. Call this tool before claiming a question was asked; if creation fails, report the failure. Do not fabricate answer links or Markdown buttons, post duplicate cards, or use call_api to create the card. Use one complete payload.questionSet for text and choice questions. Paperclip generates compatibility questions; see the payload schema for formats.","inputSchema":{"additionalProperties":false,"allOf":[{"if":{"properties":{"interactionKind":{"const":"questions"}},"required":["interactionKind"]},"then":{"properties":{"payload":{"anyOf":[{"required":["questionSet"]},{"required":["questions"]}],"required":["version"]}},"required":["payload"]}}],"properties":{"continuationPolicy":{"enum":["none","wake_assignee","wake_assignee_on_accept"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"interactionKind":{"enum":["confirmation","checkbox","questions","suggest_tasks","item_verdicts"]},"payload":{"additionalProperties":true,"description":"Kind-specific interaction data. For questions, send version:1 and one complete questionSet containing every text and choice question. Paperclip generates compatibility questions. Each canonical question needs id, prompt, required, and answerMode: text, single_select, or multi_select. Text questions have no options or customAnswer. Choice questions need at least two meaningful options with id/label. Use customAnswer:{enabled:true} for an optional written answer to a choice question. Legacy questions remain supported; if both representations are supplied, they must describe the same complete form. Keep IDs stable across retries. For confirmation, payload may be {}.","properties":{"questionSet":{"additionalProperties":false,"properties":{"description":{"maxLength":100000,"type":"string"},"questions":{"items":{"additionalProperties":false,"allOf":[{"if":{"properties":{"answerMode":{"const":"text"}},"required":["answerMode"]},"then":{"not":{"required":["customAnswer"]},"properties":{"options":{"maxItems":0,"type":"array"}}}},{"if":{"properties":{"answerMode":{"enum":["single_select","multi_select"]}},"required":["answerMode"]},"then":{"properties":{"options":{"minItems":1,"type":"array"}},"required":["options"]}}],"properties":{"answerMode":{"enum":["single_select","multi_select","text"]},"customAnswer":{"additionalProperties":false,"properties":{"enabled":{"const":true},"label":{"maxLength":1000,"type":"string"},"placeholder":{"maxLength":1000,"type":"string"}},"required":["enabled"],"type":"object"},"header":{"maxLength":1000,"type":"string"},"helpText":{"maxLength":4000,"type":"string"},"id":{"maxLength":160,"minLength":1,"type":"string"},"options":{"items":{"additionalProperties":false,"properties":{"description":{"maxLength":4000,"type":"string"},"id":{"maxLength":160,"minLength":1,"type":"string"},"label":{"maxLength":1000,"minLength":1,"type":"string"},"recommended":{"type":"boolean"}},"required":["id","label"],"type":"object"},"maxItems":128,"type":"array"},"prompt":{"maxLength":4000,"minLength":1,"type":"string"},"required":{"type":"boolean"},"textValidation":{"additionalProperties":false,"properties":{"inputType":{"enum":["text","number","integer"]},"maxLength":{"maximum":100000,"minimum":0,"type":"integer"},"maximum":{"type":"number"},"minLength":{"maximum":100000,"minimum":0,"type":"integer"},"minimum":{"type":"number"},"pattern":{"maxLength":1000,"type":"string"}},"type":"object"}},"required":["id","prompt","required","answerMode"],"type":"object"},"maxItems":64,"minItems":1,"type":"array"},"schema":{"const":"paperclip.question_set.v1"},"submitLabel":{"maxLength":200,"type":"string"},"title":{"maxLength":1000,"type":"string"}},"required":["schema","questions"],"type":"object"},"questions":{"items":{"additionalProperties":true,"properties":{"id":{"maxLength":160,"minLength":1,"type":"string"},"options":{"items":{"additionalProperties":true,"properties":{"freeText":{"type":"boolean"},"id":{"maxLength":160,"minLength":1,"type":"string"},"label":{"maxLength":1000,"minLength":1,"type":"string"}},"required":["id","label"],"type":"object"},"maxItems":129,"minItems":1,"type":"array"},"prompt":{"maxLength":4000,"minLength":1,"type":"string"},"required":{"type":"boolean"},"selectionMode":{"enum":["single","multi"]}},"required":["id","prompt","selectionMode","options"],"type":"object"},"maxItems":64,"minItems":1,"type":"array"},"version":{"const":1}},"type":"object"},"prompt":{"description":"Question or decision prompt.","maxLength":10000,"minLength":1,"type":"string"},"targetRevisionId":{"description":"Optional bound document revision.","maxLength":20000,"type":["string","null"]},"title":{"description":"Interaction card title.","maxLength":300,"minLength":1,"type":"string"}},"required":["idempotencyKey","interactionKind","title","prompt","continuationPolicy"],"type":"object"},"name":"request_human_input","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"register_deliverable","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Register attachment metadata and its artifact work product without credentials or bytes in the tool result.","inputSchema":{"additionalProperties":false,"properties":{"byteSize":{"maximum":100000000,"minimum":0,"type":"integer"},"contentRef":{"description":"Opaque package-local content reference.","maxLength":2000,"minLength":1,"type":"string"},"contentType":{"description":"Media type.","maxLength":200,"minLength":1,"type":"string"},"filename":{"description":"Display filename.","maxLength":500,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"sha256":{"pattern":"^[a-fA-F0-9]{64}$","type":"string"},"title":{"description":"Work-product title.","maxLength":500,"minLength":1,"type":"string"}},"required":["idempotencyKey","filename","contentType","byteSize","sha256","contentRef","title"],"type":"object"},"name":"register_deliverable","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"finish_task","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Finish the active task with a durable summary.","inputSchema":{"additionalProperties":false,"properties":{"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"summary":{"description":"Completion summary.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","summary"],"type":"object"},"name":"finish_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"block_task","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Block the active task with a durable reason and optional first-class dependencies.","inputSchema":{"additionalProperties":false,"properties":{"blockedByTaskIds":{"description":"Internal task ids that block this task.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"reason":{"description":"Block reason.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","reason"],"type":"object"},"name":"block_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"request_review","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Move the active task to review with a durable summary.","inputSchema":{"additionalProperties":false,"properties":{"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"summary":{"description":"Review handoff summary.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","summary"],"type":"object"},"name":"request_review","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_agents","requiredClaims":["discovery:agents:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List redacted actor profiles.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_agents","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_agent","requiredClaims":["discovery:agents:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read one redacted actor profile.","inputSchema":{"additionalProperties":false,"properties":{"actorId":{"description":"Actor id.","maxLength":200,"minLength":1,"type":"string"}},"required":["actorId"],"type":"object"},"name":"get_agent","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"hire_agent","requiredClaims":["delegation:agents:create"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a persistent native Runner teammate with an identity and persona. The teammate reports to the caller and inherits the caller's native runtime; provider, adapter, environment, and credential settings are selected by Paperclip and are never caller-supplied here. Reuse a suitable teammate from list_agents when possible.","inputSchema":{"additionalProperties":false,"properties":{"capabilities":{"maxLength":2000,"type":["string","null"]},"instructions":{"maxLength":20000,"type":["string","null"]},"name":{"maxLength":200,"minLength":1,"type":"string"},"role":{"enum":["ceo","cto","cmo","cfo","security","engineer","designer","pm","qa","devops","researcher","general"],"type":"string"},"title":{"maxLength":300,"type":["string","null"]}},"required":["name"],"type":"object"},"name":"hire_agent","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"search_tasks","requiredClaims":["discovery:tasks:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Search tasks by text and status within the run company.","inputSchema":{"additionalProperties":false,"properties":{"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"},"query":{"maxLength":500,"type":"string"},"statuses":{"items":{"enum":["backlog","todo","in_progress","in_review","done","blocked","cancelled"]},"maxItems":7,"type":"array","uniqueItems":true}},"required":[],"type":"object"},"name":"search_tasks","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_approvals","requiredClaims":["governance:approvals:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List approvals in the run company.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_approvals","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_approval","requiredClaims":["governance:approvals:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read one approval without protected data.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"}},"required":["approvalId"],"type":"object"},"name":"get_approval","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_approval_context","requiredClaims":["governance:approvals:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read one approval, its comments, and linked tasks.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"}},"required":["approvalId"],"type":"object"},"name":"get_approval_context","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_workspace_runtime","requiredClaims":["workspace:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read active-task workspace services.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"get_workspace_runtime","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"control_workspace_service","requiredClaims":["workspace:control"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Start, stop, or fault one active-task workspace service.","inputSchema":{"additionalProperties":false,"properties":{"action":{"enum":["start","stop","fail"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"serviceId":{"description":"Workspace service id.","maxLength":200,"minLength":1,"type":"string"},"url":{"description":"Optional service URL.","maxLength":20000,"type":["string","null"]}},"required":["idempotencyKey","serviceId","action"],"type":"object"},"name":"control_workspace_service","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"set_dependencies","requiredClaims":["dependencies:write"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Replace the active task's first-class blocker set.","inputSchema":{"additionalProperties":false,"properties":{"blockedByTaskIds":{"description":"Replacement blocker task ids.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","blockedByTaskIds"],"type":"object"},"name":"set_dependencies","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"create_task","requiredClaims":["delegation:tasks:create"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a project task from a conversation, or a child from an ordinary task. Persist initialPlan before execution. Set status to backlog when the user wants to save or plan work without starting it; backlog tasks never wake an agent. Omitted status means todo, subject to blockers.","inputSchema":{"additionalProperties":false,"properties":{"assigneeActorId":{"description":"Optional agent assignee. Omit to assign the current agent.","maxLength":20000,"type":["string","null"]},"blockedByTaskIds":{"description":"Initial blocker task ids.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"description":{"description":"Child task description.","maxLength":20000,"type":["string","null"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"initialPlan":{"description":"Relevant markdown plan saved on the new task before execution starts.","maxLength":200000,"type":["string","null"]},"priority":{"enum":["critical","high","medium","low"]},"projectId":{"description":"Project ID for the task.","type":["string","null"]},"status":{"description":"Initial status. Use backlog to save work without executing it. Defaults to todo (blocked when dependencies are unresolved).","enum":["backlog","todo"]},"title":{"description":"Child task title.","maxLength":500,"minLength":1,"type":"string"}},"required":["idempotencyKey","title"],"type":"object"},"name":"create_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"},"task":{"additionalProperties":false,"properties":{"assigneeActorId":{"type":["string","null"]},"id":{"minLength":1,"type":"string"},"identifier":{"type":["string","null"]},"parentId":{"minLength":1,"type":["string","null"]},"projectId":{"minLength":1,"type":["string","null"]},"status":{"minLength":1,"type":"string"}},"required":["id","identifier","parentId","status","assigneeActorId"],"type":"object"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds","task"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"request_approval","requiredClaims":["governance:approvals:request"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a governed approval and waiting posture.","inputSchema":{"additionalProperties":false,"properties":{"approvalType":{"description":"Stable approval type.","maxLength":200,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"payload":{"additionalProperties":true,"type":"object"}},"required":["idempotencyKey","approvalType","payload"],"type":"object"},"name":"request_approval","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"decide_approval","requiredClaims":["governance:approvals:decide"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Decide an approval as an explicitly authorized approver.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"},"decision":{"enum":["approved","rejected","cancelled"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"note":{"description":"Decision note.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","approvalId","decision","note"],"type":"object"},"name":"decide_approval","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"comment_on_approval","requiredClaims":["governance:approvals:comment"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Add a durable comment to an approval.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"},"body":{"description":"Approval comment.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","approvalId","body"],"type":"object"},"name":"comment_on_approval","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"schedule_wake","requiredClaims":["control_plane:wakes"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Schedule a deterministic continuation wake.","inputSchema":{"additionalProperties":false,"properties":{"delayTicks":{"maximum":10000,"minimum":1,"type":"integer"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"payload":{"additionalProperties":true,"type":"object"},"reason":{"enum":["manual","issue_commented","interaction_resolved","approval_resolved","blockers_resolved","scheduled_retry","resume"]}},"required":["idempotencyKey","reason","delayTicks"],"type":"object"},"name":"schedule_wake","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"generic_api_request","requiredClaims":["test:generic_api_request"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Test-only escape hatch. Disabled unless the scenario and explicit claim both enable it.","inputSchema":{"additionalProperties":false,"properties":{"body":{"additionalProperties":true,"type":"object"},"method":{"enum":["GET","POST","PATCH"]},"path":{"maxLength":500,"pattern":"^/mock/","type":"string"}},"required":["method","path"],"type":"object"},"name":"generic_api_request","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"read_agent_instructions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the current canonical instruction entry and its revision, or inspect a historical revision by supplying both entryFile and revisionId. Read before updating; do not edit shared instruction caches.","inputSchema":{"additionalProperties":false,"properties":{"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"revisionId":{"description":"Historical revision to inspect; also provide its entryFile.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":[],"type":"object"},"name":"read_agent_instructions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"search_api","requiredClaims":["api:discover"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Fallback only: discover Paperclip API operations when the available dedicated tools cannot express the task. Prefer dedicated tools for common operations; do not search before using them. For a persistent hire, first list_agents to reuse a suitable teammate. Search agent-hires for the hiring schema and agent-configurations for compatible adapter/runtime settings, then use call_api with the returned operationId. Supply role-specific instructionsBundle.files as a filename-to-content record. Use the returned agent.id for delegation and obey any pending approval. Provider helper threads are temporary workers, not Paperclip hires. If a hire response is uncertain, reconcile with list_agents before retrying.","inputSchema":{"additionalProperties":false,"properties":{"cursor":{"maxLength":200,"type":"string"},"limit":{"default":5,"maximum":10,"minimum":1,"type":"integer"},"query":{"maxLength":500,"minLength":1,"type":"string"}},"required":["query"],"type":"object"},"name":"search_api","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"update_agent_instructions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Commit instruction content with the exact entryFile and baseRevisionId from a prior read. Requires the responsible user’s current target edit permission. On conflict, preserve your candidate and explicitly resolve it; never overwrite the newer head.","inputSchema":{"additionalProperties":false,"properties":{"baseRevisionId":{"description":"Revision read before editing; null only when no canonical entry exists. Never replace a stale base silently.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":["string","null"]},"content":{"description":"Complete UTF-8 instruction content, at most 1 MiB; empty content is valid.","maxLength":1048576,"type":"string"},"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":["entryFile","content","baseRevisionId"],"type":"object"},"name":"update_agent_instructions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"call_api","requiredClaims":["api:call"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Fallback only: call a discovered Paperclip API operation when dedicated tools lack the required operation or parameters. Uses your existing permissions. Prefer dedicated tools; never bypass a denial or runner lifecycle tool. For large text responses, read the returned artifact with GET /api/assets/{assetId}/content and responseText; follow nextOffsetBytes until null.","inputSchema":{"additionalProperties":false,"properties":{"body":{"anyOf":[{"additionalProperties":true,"type":"object"},{"items":{},"type":"array"},{"type":"string"},{"type":"number"},{"type":"boolean"},{"type":"null"}],"description":"Request value matching the discovered schema. For JSON object or array requests, pass the object or array directly, never a JSON-encoded string. Strings are for text bodies or endpoints whose schema explicitly accepts a string."},"contentType":{"maxLength":120,"type":"string"},"files":{"items":{"additionalProperties":false,"oneOf":[{"properties":{"artifactId":{}},"required":["artifactId"]},{"properties":{"path":{}},"required":["path"]}],"properties":{"artifactId":{"type":"string"},"field":{"type":"string"},"path":{"description":"File relative to the active issue workspace. Remote files must first be uploaded as an artifact.","type":"string"}},"type":"object"},"maxItems":10,"type":"array"},"operationId":{"description":"Exact operationId returned by search_api, for example GET /api/projects/{id}. Do not guess identifiers.","maxLength":500,"minLength":1,"type":"string"},"pathParams":{"additionalProperties":{"type":"string"},"type":"object"},"query":{"additionalProperties":true,"type":"object"},"responseText":{"additionalProperties":false,"description":"GET only: return a bounded UTF-8 text window inline, including JSON as text, without saving another artifact. Offsets and limits are bytes. Use the returned nextOffsetBytes to continue; null means complete. Prefer reading a saved artifact for a stable snapshot. New live responses have a 1 GiB capture limit and a 4 GiB per-run capture budget. Existing larger assets remain readable in bounded pages. If a live response returns an artifact, continue on its content operation for a stable snapshot.","properties":{"limitBytes":{"default":24576,"maximum":24576,"minimum":4,"type":"integer"},"offsetBytes":{"default":0,"maximum":9007199254740991,"minimum":0,"type":"integer"}},"type":"object"}},"required":["operationId"],"type":"object"},"name":"call_api","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"get_agent_instruction_history","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List bounded canonical instruction revision metadata, including actor, source run, and restore origin. Use read_agent_instructions with entryFile and revisionId to inspect exact historical content.","inputSchema":{"additionalProperties":false,"properties":{"cursor":{"maxLength":2048,"minLength":1,"type":"string"},"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"limit":{"maximum":100,"minimum":1,"type":"integer"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":["entryFile"],"type":"object"},"name":"get_agent_instruction_history","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"restore_agent_instructions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append a historical instruction revision as the current content using the current baseRevisionId. Requires the responsible user’s current target edit permission. History remains intact; conflicts require an explicit resolution.","inputSchema":{"additionalProperties":false,"properties":{"baseRevisionId":{"description":"Current revision read before restoring. Never replace a stale base silently.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"revisionId":{"description":"Historical revision whose exact content should be restored.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":["entryFile","revisionId","baseRevisionId"],"type":"object"},"name":"restore_agent_instructions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"create_project","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a project after considering existing projects and available repositories. repositoryIds and repositoryUrls accept multiple existing repositories. Use HTTPS GitHub repositoryUrls when an accessible repo is not in the catalog; this registers project repositories, not remote GitHub repositories. Non-code projects may omit repositories. Cannot combine repositoryIds/repositoryUrls with workspace. Reuse the idempotency key on retries.","inputSchema":{"additionalProperties":false,"properties":{"archivedAt":{"description":"Archive timestamp.","maxLength":20000,"type":["string","null"]},"color":{"description":"Project color.","maxLength":20000,"type":["string","null"]},"description":{"description":"Project outcome and context.","maxLength":20000,"type":["string","null"]},"env":{"additionalProperties":true,"type":"object"},"executionWorkspacePolicy":{"additionalProperties":true,"type":"object"},"goalId":{"description":"Goal ID.","maxLength":20000,"type":["string","null"]},"goalIds":{"description":"Goal IDs.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"icon":{"description":"Project icon.","enum":["folder","rocket","code","terminal","database","globe","package","boxes","box","layers","briefcase","compass","target","flame","zap","star","bug","wrench","hammer","lightbulb","sparkles","shield","lock","search","cog","brain","cpu","git-branch","file-code","puzzle","gem","atom","heart","mail","message-square","crown","radar","telescope","hexagon",null],"type":["string","null"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"leadAgentId":{"description":"Lead agent ID.","maxLength":20000,"type":["string","null"]},"name":{"description":"Project name.","maxLength":500,"minLength":1,"type":"string"},"repositoryIds":{"description":"Authorized repository IDs from list_project_repositories; may contain multiple repositories.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"repositoryUrls":{"description":"Existing HTTPS GitHub repository URLs, including repos absent from the catalog.","items":{"maxLength":2000,"pattern":"^https://github\\.com/(?!\\.{1,2}/)[A-Za-z0-9_.-]+/(?!\\.{1,2}/?$)[A-Za-z0-9_.-]+/?$","type":"string"},"maxItems":100,"type":"array"},"status":{"enum":["backlog","planned","in_progress","completed","cancelled"]},"targetDate":{"description":"Target date.","maxLength":20000,"type":["string","null"]},"workspace":{"additionalProperties":true,"type":"object"}},"required":["idempotencyKey","name"],"type":"object"},"name":"create_project","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_project_repositories","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List authorized repositories with stable IDs and names. Consider appropriate repositories before creating a project; never invent IDs.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_project_repositories","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_projects","requiredClaims":["discovery:projects:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List company project summaries (IDs, names, status, and up to 1,000 characters of description). Returns up to 50 projects and nextCursor; continue with cursor until it is null. Read a project's API resource for full details.","inputSchema":{"additionalProperties":false,"properties":{"cursor":{"description":"The nextCursor returned by the previous page.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"limit":{"description":"Page size (default 50).","maximum":50,"minimum":1,"type":"integer"}},"required":[],"type":"object"},"name":"list_projects","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"create_skill","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a reusable single-file skill in the company library. Supply a complete SKILL.md whose name and description match the inputs. This saves the skill and shows a card; it does not assign the skill to any agent. Reuse idempotencyKey on retries.","inputSchema":{"additionalProperties":false,"properties":{"description":{"maxLength":2000,"minLength":1,"type":"string"},"idempotencyKey":{"maxLength":240,"minLength":1,"type":"string"},"markdown":{"description":"Complete SKILL.md including name and description frontmatter and substantive instructions.","maxLength":200000,"minLength":1,"type":"string"},"name":{"description":"Lowercase skill name, matching SKILL.md frontmatter.","maxLength":120,"minLength":1,"pattern":"^[a-z0-9]+(?:-[a-z0-9]+)*$","type":"string"},"slug":{"description":"Optional; must equal name.","maxLength":120,"minLength":1,"pattern":"^[a-z0-9]+(?:-[a-z0-9]+)*$","type":"string"}},"required":["idempotencyKey","name","description","markdown"],"type":"object"},"name":"create_skill","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"reassign_task","requiredClaims":["delegation:tasks:assign"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Reassign an existing task to another company agent. Read search_tasks first and supply its current assignee and statusVersion to prevent overwriting a concurrent change. Preserve the task, documents, dependencies, and blocked/backlog status. Active work is stopped before handoff. Use a stable idempotency key for retries. Cannot reassign this run’s own task, conversations, completed tasks, or pending reviews; use create_task for delegation from the current task.","inputSchema":{"additionalProperties":false,"properties":{"assigneeActorId":{"description":"New company agent ID from list_agents.","minLength":1,"type":"string"},"expectedAssigneeActorId":{"description":"Current assigneeAgentId from search_tasks; null means unassigned.","type":["string","null"]},"expectedStatusVersion":{"description":"Current statusVersion from search_tasks.","minimum":0,"type":"integer"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"reason":{"description":"Why the task should move and context for the new owner.","maxLength":20000,"minLength":1,"type":"string"},"taskId":{"description":"Existing task ID from search_tasks.","minLength":1,"type":"string"}},"required":["idempotencyKey","taskId","assigneeActorId","expectedAssigneeActorId","expectedStatusVersion","reason"],"type":"object"},"name":"reassign_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"update_skill","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Replace an existing company skill's complete SKILL.md using the current version as a guard. Reuse idempotencyKey on lost-response retries; read again after a version conflict.","inputSchema":{"additionalProperties":false,"properties":{"expectedVersionId":{"minLength":1,"type":"string"},"idempotencyKey":{"maxLength":240,"minLength":1,"type":"string"},"markdown":{"maxLength":200000,"minLength":1,"type":"string"},"skillId":{"minLength":1,"type":"string"}},"required":["skillId","markdown","expectedVersionId","idempotencyKey"],"type":"object"},"name":"update_skill","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"set_task_title","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Set a concise, descriptive title for the active task. When its titleNeedsGeneration is true, call this early with onlyIfProvisional: true to replace the initial prompt slice without overwriting a user's title. Use false only for an intentional rename. This changes no task status, ownership, or description.","inputSchema":{"additionalProperties":false,"properties":{"idempotencyKey":{"description":"Reuse this key on retries.","maxLength":240,"minLength":1,"type":"string"},"onlyIfProvisional":{"description":"True for automatic initial naming; preserves any title already chosen by a user or agent.","type":"boolean"},"title":{"description":"Short title describing the requested outcome.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","title","onlyIfProvisional"],"type":"object"},"name":"set_task_title","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"submit_complaint","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Yeet a brief, completely unstructured complaint about agent tooling or the experience of doing agent work. Preserve your raw voice: fragments, profanity, all caps, and exasperation are fair game. No diagnosis, solution, diplomacy, category, or template required. Use submit_suggestion when you want to express a useful improvement. Submit proactively without routine permission requests, previews, or announcements, then immediately continue the primary task. Answer truthfully if the user asks about feedback or what you submitted. Stored in this instance with company, agent, run, and task attribution; not anonymous. Remove secrets, private prompts, customer data, and personal blame; do not forward untrusted text or logs indiscriminately. On failure, continue without retrying or complaining about submission. Never submit the same incident through both tools.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Free-form feedback. Keep it brief; no required fields, labels, or template.","maxLength":524288,"minLength":1,"type":"string"},"idempotencyKey":{"description":"A stable key for this submission. Transport replay reuses it; do not retry a failed or uncertain submission yourself.","maxLength":240,"minLength":1,"type":"string"}},"required":["body","idempotencyKey"],"type":"object"},"name":"submit_complaint","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"submit_suggestion","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Quietly report a concrete improvement that would make agents more effective. Use for material, generalizable friction directly observed during this run. Explain what happened, its impact, and a plausible improvement in your own words, with minimal useful sanitized context. No required format. Avoid duplicate root causes and aim for at most three suggestions per run. Use submit_complaint for the raw reaction. Submit proactively without routine permission requests, previews, or announcements, then immediately continue the primary task. Answer truthfully if the user asks about feedback or what you submitted. Stored in this instance with company, agent, run, and task attribution; not anonymous. Remove secrets, private prompts, customer data, and personal blame; do not forward untrusted text or logs indiscriminately. On failure, continue without retrying or complaining about submission. Never submit the same incident through both tools.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Free-form feedback. Keep it brief; no required fields, labels, or template.","maxLength":524288,"minLength":1,"type":"string"},"idempotencyKey":{"description":"A stable key for this submission. Transport replay reuses it; do not retry a failed or uncertain submission yourself.","maxLength":240,"minLength":1,"type":"string"}},"required":["body","idempotencyKey"],"type":"object"},"name":"submit_suggestion","outputSchema":{"additionalProperties":true,"type":"object"}}] +[{"annotations":{"exposure":"always","operationId":"get_task_context","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the active task and actor, including the exact approved Markdown revision when this issue has an accepted plan.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"get_task_context","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"get_task_history","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded comments on the active task.","inputSchema":{"additionalProperties":false,"properties":{"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":[],"type":"object"},"name":"get_task_history","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_documents","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List revisioned documents on the active task.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_documents","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"read_document","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the current revision of one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"}},"required":["key"],"type":"object"},"name":"read_document","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"list_document_revisions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read bounded revision history for one active-task document.","inputSchema":{"additionalProperties":false,"properties":{"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"},"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"}},"required":["key"],"type":"object"},"name":"list_document_revisions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"report_progress","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append a durable progress comment to the active task.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Multiline progress update.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"report_progress","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"answer_status_question","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append the answer to a status-only wake without changing task disposition.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Concise status answer.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","body"],"type":"object"},"name":"answer_status_question","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"write_document","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create or update an active-task document with optimistic revision safety.","inputSchema":{"additionalProperties":false,"properties":{"baseRevisionId":{"description":"Current revision id, or null when creating.","maxLength":20000,"type":["string","null"]},"body":{"description":"Markdown document body.","maxLength":200000,"minLength":1,"type":"string"},"changeSummary":{"description":"Optional revision summary.","maxLength":20000,"type":["string","null"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"key":{"description":"Stable issue-document key.","maxLength":120,"minLength":1,"type":"string"},"title":{"description":"Document title.","maxLength":300,"minLength":1,"type":"string"}},"required":["idempotencyKey","key","title","body","baseRevisionId"],"type":"object"},"name":"write_document","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"request_human_input","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a durable human question or approval card on the current Paperclip task bound to this run; Paperclip renders it and authenticates the response. Use questions with continuationPolicy='wake_assignee' when an answer is needed, including otherwise tool-free chat turns. Supply a stable idempotencyKey and reuse it on retries. For one question at a time, ask only the next unanswered question and wait for its real answer. Never fabricate answers or treat ambiguous clarification as approval. For an ordinary confirmation or checkbox card, record a clear user chat answer with call_api POST /api/issues/{id}/interactions/{interactionId}/resolve-from-comment, using the source commentId and decision (accept/reject), plus explicit selectedOptionIds for checkbox acceptance. Existing resolver permissions still apply; governed tool, secret, and connection approvals are excluded. Question forms retain their dedicated answer workflow. Preserve existing review gates. Call this tool before claiming a question was asked; if creation fails, report the failure. Do not fabricate answer links or Markdown buttons, post duplicate cards, or use call_api to create the card. Use one complete payload.questionSet for text and choice questions. Paperclip generates compatibility questions; see the payload schema for formats.","inputSchema":{"additionalProperties":false,"allOf":[{"if":{"properties":{"interactionKind":{"const":"questions"}},"required":["interactionKind"]},"then":{"properties":{"payload":{"anyOf":[{"required":["questionSet"]},{"required":["questions"]}],"required":["version"]}},"required":["payload"]}}],"properties":{"continuationPolicy":{"enum":["none","wake_assignee","wake_assignee_on_accept"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"interactionKind":{"enum":["confirmation","checkbox","questions","suggest_tasks","item_verdicts"]},"payload":{"additionalProperties":true,"description":"Kind-specific interaction data. For questions, send version:1 and one complete questionSet containing every text and choice question. Paperclip generates compatibility questions. Each canonical question needs id, prompt, required, and answerMode: text, single_select, or multi_select. Text questions have no options or customAnswer. Choice questions need at least two meaningful options with id/label. Use customAnswer:{enabled:true} for an optional written answer to a choice question. Legacy questions remain supported; if both representations are supplied, they must describe the same complete form. Keep IDs stable across retries. For confirmation, payload may be {}.","properties":{"questionSet":{"additionalProperties":false,"properties":{"description":{"maxLength":100000,"type":"string"},"questions":{"items":{"additionalProperties":false,"allOf":[{"if":{"properties":{"answerMode":{"const":"text"}},"required":["answerMode"]},"then":{"not":{"required":["customAnswer"]},"properties":{"options":{"maxItems":0,"type":"array"}}}},{"if":{"properties":{"answerMode":{"enum":["single_select","multi_select"]}},"required":["answerMode"]},"then":{"properties":{"options":{"minItems":1,"type":"array"}},"required":["options"]}}],"properties":{"answerMode":{"enum":["single_select","multi_select","text"]},"customAnswer":{"additionalProperties":false,"properties":{"enabled":{"const":true},"label":{"maxLength":1000,"type":"string"},"placeholder":{"maxLength":1000,"type":"string"}},"required":["enabled"],"type":"object"},"header":{"maxLength":1000,"type":"string"},"helpText":{"maxLength":4000,"type":"string"},"id":{"maxLength":160,"minLength":1,"type":"string"},"options":{"items":{"additionalProperties":false,"properties":{"description":{"maxLength":4000,"type":"string"},"id":{"maxLength":160,"minLength":1,"type":"string"},"label":{"maxLength":1000,"minLength":1,"type":"string"},"recommended":{"type":"boolean"}},"required":["id","label"],"type":"object"},"maxItems":128,"type":"array"},"prompt":{"maxLength":4000,"minLength":1,"type":"string"},"required":{"type":"boolean"},"textValidation":{"additionalProperties":false,"properties":{"inputType":{"enum":["text","number","integer"]},"maxLength":{"maximum":100000,"minimum":0,"type":"integer"},"maximum":{"type":"number"},"minLength":{"maximum":100000,"minimum":0,"type":"integer"},"minimum":{"type":"number"},"pattern":{"maxLength":1000,"type":"string"}},"type":"object"}},"required":["id","prompt","required","answerMode"],"type":"object"},"maxItems":64,"minItems":1,"type":"array"},"schema":{"const":"paperclip.question_set.v1"},"submitLabel":{"maxLength":200,"type":"string"},"title":{"maxLength":1000,"type":"string"}},"required":["schema","questions"],"type":"object"},"questions":{"items":{"additionalProperties":true,"properties":{"id":{"maxLength":160,"minLength":1,"type":"string"},"options":{"items":{"additionalProperties":true,"properties":{"freeText":{"type":"boolean"},"id":{"maxLength":160,"minLength":1,"type":"string"},"label":{"maxLength":1000,"minLength":1,"type":"string"}},"required":["id","label"],"type":"object"},"maxItems":129,"minItems":1,"type":"array"},"prompt":{"maxLength":4000,"minLength":1,"type":"string"},"required":{"type":"boolean"},"selectionMode":{"enum":["single","multi"]}},"required":["id","prompt","selectionMode","options"],"type":"object"},"maxItems":64,"minItems":1,"type":"array"},"version":{"const":1}},"type":"object"},"prompt":{"description":"Question or decision prompt.","maxLength":10000,"minLength":1,"type":"string"},"targetRevisionId":{"description":"Optional bound document revision.","maxLength":20000,"type":["string","null"]},"title":{"description":"Interaction card title.","maxLength":300,"minLength":1,"type":"string"}},"required":["idempotencyKey","interactionKind","title","prompt","continuationPolicy"],"type":"object"},"name":"request_human_input","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"register_deliverable","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Register attachment metadata and its artifact work product without credentials or bytes in the tool result.","inputSchema":{"additionalProperties":false,"properties":{"byteSize":{"maximum":100000000,"minimum":0,"type":"integer"},"contentRef":{"description":"Opaque package-local content reference.","maxLength":2000,"minLength":1,"type":"string"},"contentType":{"description":"Media type.","maxLength":200,"minLength":1,"type":"string"},"filename":{"description":"Display filename.","maxLength":500,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"sha256":{"pattern":"^[a-fA-F0-9]{64}$","type":"string"},"title":{"description":"Work-product title.","maxLength":500,"minLength":1,"type":"string"}},"required":["idempotencyKey","filename","contentType","byteSize","sha256","contentRef","title"],"type":"object"},"name":"register_deliverable","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"finish_task","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Finish the active task with a durable summary.","inputSchema":{"additionalProperties":false,"properties":{"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"summary":{"description":"Completion summary.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","summary"],"type":"object"},"name":"finish_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"block_task","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Block the active task with a durable reason and optional first-class dependencies.","inputSchema":{"additionalProperties":false,"properties":{"blockedByTaskIds":{"description":"Internal task ids that block this task.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"reason":{"description":"Block reason.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","reason"],"type":"object"},"name":"block_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"always","operationId":"request_review","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Move the active task to review with a durable summary.","inputSchema":{"additionalProperties":false,"properties":{"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"summary":{"description":"Review handoff summary.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","summary"],"type":"object"},"name":"request_review","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_agents","requiredClaims":["discovery:agents:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List redacted actor profiles.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_agents","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_agent","requiredClaims":["discovery:agents:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read one redacted actor profile.","inputSchema":{"additionalProperties":false,"properties":{"actorId":{"description":"Actor id.","maxLength":200,"minLength":1,"type":"string"}},"required":["actorId"],"type":"object"},"name":"get_agent","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"hire_agent","requiredClaims":["delegation:agents:create"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a persistent native Runner teammate with an identity and persona. The teammate reports to the caller and inherits the caller's native runtime; provider, adapter, environment, and credential settings are selected by Paperclip and are never caller-supplied here. Reuse a suitable teammate from list_agents when possible.","inputSchema":{"additionalProperties":false,"properties":{"capabilities":{"maxLength":2000,"type":["string","null"]},"instructions":{"maxLength":20000,"type":["string","null"]},"name":{"maxLength":200,"minLength":1,"type":"string"},"role":{"enum":["ceo","cto","cmo","cfo","security","engineer","designer","pm","qa","devops","researcher","general"],"type":"string"},"title":{"maxLength":300,"type":["string","null"]}},"required":["name"],"type":"object"},"name":"hire_agent","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"search_tasks","requiredClaims":["discovery:tasks:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Search tasks by text and status within the run company.","inputSchema":{"additionalProperties":false,"properties":{"limit":{"default":50,"maximum":200,"minimum":1,"type":"integer"},"query":{"maxLength":500,"type":"string"},"statuses":{"items":{"enum":["backlog","todo","in_progress","in_review","done","blocked","cancelled"]},"maxItems":7,"type":"array","uniqueItems":true}},"required":[],"type":"object"},"name":"search_tasks","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_approvals","requiredClaims":["governance:approvals:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List approvals in the run company.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_approvals","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_approval","requiredClaims":["governance:approvals:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read one approval without protected data.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"}},"required":["approvalId"],"type":"object"},"name":"get_approval","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_approval_context","requiredClaims":["governance:approvals:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read one approval, its comments, and linked tasks.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"}},"required":["approvalId"],"type":"object"},"name":"get_approval_context","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"get_workspace_runtime","requiredClaims":["workspace:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read active-task workspace services.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"get_workspace_runtime","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"control_workspace_service","requiredClaims":["workspace:control"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Start, stop, or fault one active-task workspace service.","inputSchema":{"additionalProperties":false,"properties":{"action":{"enum":["start","stop","fail"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"serviceId":{"description":"Workspace service id.","maxLength":200,"minLength":1,"type":"string"},"url":{"description":"Optional service URL.","maxLength":20000,"type":["string","null"]}},"required":["idempotencyKey","serviceId","action"],"type":"object"},"name":"control_workspace_service","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"set_dependencies","requiredClaims":["dependencies:write"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Replace the active task's first-class blocker set.","inputSchema":{"additionalProperties":false,"properties":{"blockedByTaskIds":{"description":"Replacement blocker task ids.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","blockedByTaskIds"],"type":"object"},"name":"set_dependencies","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"create_task","requiredClaims":["delegation:tasks:create"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a project task from a conversation, or a child from an ordinary task. Persist initialPlan before execution. Set status to backlog when the user wants to save or plan work without starting it; backlog tasks never wake an agent. Omitted status means todo, subject to blockers.","inputSchema":{"additionalProperties":false,"properties":{"assigneeActorId":{"description":"Optional agent assignee. Omit to assign the current agent.","maxLength":20000,"type":["string","null"]},"blockedByTaskIds":{"description":"Initial blocker task ids.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"description":{"description":"Child task description.","maxLength":20000,"type":["string","null"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"initialPlan":{"description":"Relevant markdown plan saved on the new task before execution starts.","maxLength":200000,"type":["string","null"]},"priority":{"enum":["critical","high","medium","low"]},"projectId":{"description":"Project ID for the task.","type":["string","null"]},"status":{"description":"Initial status. Use backlog to save work without executing it. Defaults to todo (blocked when dependencies are unresolved).","enum":["backlog","todo"]},"title":{"description":"Child task title.","maxLength":500,"minLength":1,"type":"string"}},"required":["idempotencyKey","title"],"type":"object"},"name":"create_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"},"task":{"additionalProperties":false,"properties":{"assigneeActorId":{"type":["string","null"]},"id":{"minLength":1,"type":"string"},"identifier":{"type":["string","null"]},"parentId":{"minLength":1,"type":["string","null"]},"projectId":{"minLength":1,"type":["string","null"]},"status":{"minLength":1,"type":"string"}},"required":["id","identifier","parentId","status","assigneeActorId"],"type":"object"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds","task"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"request_approval","requiredClaims":["governance:approvals:request"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a governed approval and waiting posture.","inputSchema":{"additionalProperties":false,"properties":{"approvalType":{"description":"Stable approval type.","maxLength":200,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"payload":{"additionalProperties":true,"type":"object"}},"required":["idempotencyKey","approvalType","payload"],"type":"object"},"name":"request_approval","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"decide_approval","requiredClaims":["governance:approvals:decide"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Decide an approval as an explicitly authorized approver.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"},"decision":{"enum":["approved","rejected","cancelled"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"note":{"description":"Decision note.","maxLength":20000,"minLength":1,"type":"string"}},"required":["idempotencyKey","approvalId","decision","note"],"type":"object"},"name":"decide_approval","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"comment_on_approval","requiredClaims":["governance:approvals:comment"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Add a durable comment to an approval.","inputSchema":{"additionalProperties":false,"properties":{"approvalId":{"description":"Approval id.","maxLength":200,"minLength":1,"type":"string"},"body":{"description":"Approval comment.","maxLength":20000,"minLength":1,"type":"string"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","approvalId","body"],"type":"object"},"name":"comment_on_approval","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"schedule_wake","requiredClaims":["control_plane:wakes"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Schedule a deterministic continuation wake.","inputSchema":{"additionalProperties":false,"properties":{"delayTicks":{"maximum":10000,"minimum":1,"type":"integer"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"payload":{"additionalProperties":true,"type":"object"},"reason":{"enum":["manual","issue_commented","interaction_resolved","approval_resolved","blockers_resolved","scheduled_retry","resume"]}},"required":["idempotencyKey","reason","delayTicks"],"type":"object"},"name":"schedule_wake","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"generic_api_request","requiredClaims":["test:generic_api_request"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Test-only escape hatch. Disabled unless the scenario and explicit claim both enable it.","inputSchema":{"additionalProperties":false,"properties":{"body":{"additionalProperties":true,"type":"object"},"method":{"enum":["GET","POST","PATCH"]},"path":{"maxLength":500,"pattern":"^/mock/","type":"string"}},"required":["method","path"],"type":"object"},"name":"generic_api_request","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"read_agent_instructions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Read the current canonical instruction entry and its revision, or inspect a historical revision by supplying both entryFile and revisionId. Read before updating; do not edit shared instruction caches.","inputSchema":{"additionalProperties":false,"properties":{"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"revisionId":{"description":"Historical revision to inspect; also provide its entryFile.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":[],"type":"object"},"name":"read_agent_instructions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"search_api","requiredClaims":["api:discover"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Fallback only: discover Paperclip API operations when the available dedicated tools cannot express the task. Prefer dedicated tools for common operations; do not search before using them. For a persistent hire, first list_agents to reuse a suitable teammate. Search agent-hires for the hiring schema and agent-configurations for compatible adapter/runtime settings, then use call_api with the returned operationId. Supply role-specific instructionsBundle.files as a filename-to-content record. Use the returned agent.id for delegation and obey any pending approval. Provider helper threads are temporary workers, not Paperclip hires. If a hire response is uncertain, reconcile with list_agents before retrying.","inputSchema":{"additionalProperties":false,"properties":{"cursor":{"maxLength":200,"type":"string"},"limit":{"default":5,"maximum":10,"minimum":1,"type":"integer"},"query":{"maxLength":500,"minLength":1,"type":"string"}},"required":["query"],"type":"object"},"name":"search_api","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"update_agent_instructions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Commit instruction content with the exact entryFile and baseRevisionId from a prior read. Requires the responsible user’s current target edit permission. On conflict, preserve your candidate and explicitly resolve it; never overwrite the newer head.","inputSchema":{"additionalProperties":false,"properties":{"baseRevisionId":{"description":"Revision read before editing; null only when no canonical entry exists. Never replace a stale base silently.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":["string","null"]},"content":{"description":"Complete UTF-8 instruction content, at most 1 MiB; empty content is valid.","maxLength":1048576,"type":"string"},"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":["entryFile","content","baseRevisionId"],"type":"object"},"name":"update_agent_instructions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"call_api","requiredClaims":["api:call"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Fallback only: call a discovered Paperclip API operation when dedicated tools lack the required operation or parameters. Uses your existing permissions. Prefer dedicated tools; never bypass a denial or runner lifecycle tool. For large text responses, read the returned artifact with GET /api/assets/{assetId}/content and responseText; follow nextOffsetBytes until null.","inputSchema":{"additionalProperties":false,"properties":{"body":{"anyOf":[{"additionalProperties":true,"type":"object"},{"items":{},"type":"array"},{"type":"string"},{"type":"number"},{"type":"boolean"},{"type":"null"}],"description":"Request value matching the discovered schema. For JSON object or array requests, pass the object or array directly, never a JSON-encoded string. Strings are for text bodies or endpoints whose schema explicitly accepts a string."},"contentType":{"maxLength":120,"type":"string"},"files":{"items":{"additionalProperties":false,"oneOf":[{"properties":{"artifactId":{}},"required":["artifactId"]},{"properties":{"path":{}},"required":["path"]}],"properties":{"artifactId":{"type":"string"},"field":{"type":"string"},"path":{"description":"File relative to the active issue workspace. Remote files must first be uploaded as an artifact.","type":"string"}},"type":"object"},"maxItems":10,"type":"array"},"operationId":{"description":"Exact operationId returned by search_api, for example GET /api/projects/{id}. Do not guess identifiers.","maxLength":500,"minLength":1,"type":"string"},"pathParams":{"additionalProperties":{"type":"string"},"type":"object"},"query":{"additionalProperties":true,"type":"object"},"responseText":{"additionalProperties":false,"description":"GET only: return a bounded UTF-8 text window inline, including JSON as text, without saving another artifact. Offsets and limits are bytes. Use the returned nextOffsetBytes to continue; null means complete. Prefer reading a saved artifact for a stable snapshot. New live responses have a 1 GiB capture limit and a 4 GiB per-run capture budget. Existing larger assets remain readable in bounded pages. If a live response returns an artifact, continue on its content operation for a stable snapshot.","properties":{"limitBytes":{"default":24576,"maximum":24576,"minimum":4,"type":"integer"},"offsetBytes":{"default":0,"maximum":9007199254740991,"minimum":0,"type":"integer"}},"type":"object"}},"required":["operationId"],"type":"object"},"name":"call_api","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"get_agent_instruction_history","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List bounded canonical instruction revision metadata, including actor, source run, and restore origin. Use read_agent_instructions with entryFile and revisionId to inspect exact historical content.","inputSchema":{"additionalProperties":false,"properties":{"cursor":{"maxLength":2048,"minLength":1,"type":"string"},"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"limit":{"maximum":100,"minimum":1,"type":"integer"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":["entryFile"],"type":"object"},"name":"get_agent_instruction_history","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"always","operationId":"restore_agent_instructions","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Append a historical instruction revision as the current content using the current baseRevisionId. Requires the responsible user’s current target edit permission. History remains intact; conflicts require an explicit resolution.","inputSchema":{"additionalProperties":false,"properties":{"baseRevisionId":{"description":"Current revision read before restoring. Never replace a stale base silently.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"entryFile":{"description":"Configured relative entry filename returned by read_agent_instructions; retain it with the revision.","maxLength":4096,"minLength":1,"type":"string"},"revisionId":{"description":"Historical revision whose exact content should be restored.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"targetAgentId":{"description":"Same-company target agent. Omit to use the calling agent.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"}},"required":["entryFile","revisionId","baseRevisionId"],"type":"object"},"name":"restore_agent_instructions","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"create_project","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a project after considering existing projects and available repositories. repositoryIds and repositoryUrls accept multiple existing repositories. Use HTTPS GitHub repositoryUrls when an accessible repo is not in the catalog; this registers project repositories, not remote GitHub repositories. Non-code projects may omit repositories. Cannot combine repositoryIds/repositoryUrls with workspace. Reuse the idempotency key on retries.","inputSchema":{"additionalProperties":false,"properties":{"archivedAt":{"description":"Archive timestamp.","maxLength":20000,"type":["string","null"]},"color":{"description":"Project color.","maxLength":20000,"type":["string","null"]},"description":{"description":"Project outcome and context.","maxLength":20000,"type":["string","null"]},"env":{"additionalProperties":true,"type":"object"},"executionWorkspacePolicy":{"additionalProperties":true,"type":"object"},"goalId":{"description":"Goal ID.","maxLength":20000,"type":["string","null"]},"goalIds":{"description":"Goal IDs.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"icon":{"description":"Project icon.","enum":["folder","rocket","code","terminal","database","globe","package","boxes","box","layers","briefcase","compass","target","flame","zap","star","bug","wrench","hammer","lightbulb","sparkles","shield","lock","search","cog","brain","cpu","git-branch","file-code","puzzle","gem","atom","heart","mail","message-square","crown","radar","telescope","hexagon",null],"type":["string","null"]},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"leadAgentId":{"description":"Lead agent ID.","maxLength":20000,"type":["string","null"]},"name":{"description":"Project name.","maxLength":500,"minLength":1,"type":"string"},"repositoryIds":{"description":"Authorized repository IDs from list_project_repositories; may contain multiple repositories.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"repositoryUrls":{"description":"Existing HTTPS GitHub repository URLs, including repos absent from the catalog.","items":{"maxLength":2000,"pattern":"^https://github\\.com/(?!\\.{1,2}/)[A-Za-z0-9_.-]+/(?!\\.{1,2}/?$)[A-Za-z0-9_.-]+/?$","type":"string"},"maxItems":100,"type":"array"},"status":{"enum":["backlog","planned","in_progress","completed","cancelled"]},"targetDate":{"description":"Target date.","maxLength":20000,"type":["string","null"]},"workspace":{"additionalProperties":true,"type":"object"}},"required":["idempotencyKey","name"],"type":"object"},"name":"create_project","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_project_repositories","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List authorized repositories with stable IDs and names. Consider appropriate repositories before creating a project; never invent IDs.","inputSchema":{"additionalProperties":false,"properties":{},"required":[],"type":"object"},"name":"list_project_repositories","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"list_projects","requiredClaims":["discovery:projects:read"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"List company project summaries (IDs, names, status, and up to 1,000 characters of description). Returns up to 50 projects and nextCursor; continue with cursor until it is null. Read a project's API resource for full details.","inputSchema":{"additionalProperties":false,"properties":{"cursor":{"description":"The nextCursor returned by the previous page.","pattern":"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$","type":"string"},"limit":{"description":"Page size (default 50).","maximum":50,"minimum":1,"type":"integer"}},"required":[],"type":"object"},"name":"list_projects","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"create_skill","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Create a reusable single-file skill in the company library. Supply a complete SKILL.md whose name and description match the inputs. This saves the skill and shows a card; it does not assign the skill to any agent. Reuse idempotencyKey on retries.","inputSchema":{"additionalProperties":false,"properties":{"description":{"maxLength":2000,"minLength":1,"type":"string"},"idempotencyKey":{"maxLength":240,"minLength":1,"type":"string"},"markdown":{"description":"Complete SKILL.md including name and description frontmatter and substantive instructions.","maxLength":200000,"minLength":1,"type":"string"},"name":{"description":"Lowercase skill name, matching SKILL.md frontmatter.","maxLength":120,"minLength":1,"pattern":"^[a-z0-9]+(?:-[a-z0-9]+)*$","type":"string"},"slug":{"description":"Optional; must equal name.","maxLength":120,"minLength":1,"pattern":"^[a-z0-9]+(?:-[a-z0-9]+)*$","type":"string"}},"required":["idempotencyKey","name","description","markdown"],"type":"object"},"name":"create_skill","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"reassign_task","requiredClaims":["delegation:tasks:assign"],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Reassign an existing task to another company agent. Read search_tasks first and supply its current assignee and statusVersion to prevent overwriting a concurrent change. Preserve the task, documents, dependencies, and blocked/backlog status. Active work is stopped before handoff. Use a stable idempotency key for retries. Cannot reassign this run’s own task, conversations, completed tasks, or pending reviews; use create_task for delegation from the current task.","inputSchema":{"additionalProperties":false,"properties":{"assigneeActorId":{"description":"New company agent ID from list_agents.","minLength":1,"type":"string"},"expectedAssigneeActorId":{"description":"Current assigneeAgentId from search_tasks; null means unassigned.","type":["string","null"]},"expectedStatusVersion":{"description":"Current statusVersion from search_tasks.","minimum":0,"type":"integer"},"idempotencyKey":{"description":"Caller-stable retry key.","maxLength":240,"minLength":1,"type":"string"},"reason":{"description":"Why the task should move and context for the new owner.","maxLength":20000,"minLength":1,"type":"string"},"taskId":{"description":"Existing task ID from search_tasks.","minLength":1,"type":"string"}},"required":["idempotencyKey","taskId","assigneeActorId","expectedAssigneeActorId","expectedStatusVersion","reason"],"type":"object"},"name":"reassign_task","outputSchema":{"additionalProperties":false,"properties":{"commandId":{"description":"Stable command identifier.","maxLength":200,"minLength":1,"type":"string"},"disposition":{"enum":["applied","duplicate"]},"entityRefs":{"description":"Entities affected by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"scheduledWakeIds":{"description":"Wake identifiers scheduled by the operation.","items":{"minLength":1,"type":"string"},"maxItems":200,"type":"array","uniqueItems":true},"stateRevision":{"minimum":0,"type":"integer"}},"required":["commandId","disposition","stateRevision","entityRefs","scheduledWakeIds"],"type":"object"}},{"annotations":{"exposure":"optional","operationId":"update_skill","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Replace an existing company skill's complete SKILL.md using the current version as a guard. Reuse idempotencyKey on lost-response retries; read again after a version conflict.","inputSchema":{"additionalProperties":false,"properties":{"expectedVersionId":{"minLength":1,"type":"string"},"idempotencyKey":{"maxLength":240,"minLength":1,"type":"string"},"markdown":{"maxLength":200000,"minLength":1,"type":"string"},"skillId":{"minLength":1,"type":"string"}},"required":["skillId","markdown","expectedVersionId","idempotencyKey"],"type":"object"},"name":"update_skill","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"set_task_title","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Set a concise, descriptive title for the active task. When its titleNeedsGeneration is true, call this early with onlyIfProvisional: true to replace the initial prompt slice without overwriting a user's title. Use false only for an intentional rename. This changes no task status, ownership, or description.","inputSchema":{"additionalProperties":false,"properties":{"idempotencyKey":{"description":"Reuse this key on retries.","maxLength":240,"minLength":1,"type":"string"},"onlyIfProvisional":{"description":"True for automatic initial naming; preserves any title already chosen by a user or agent.","type":"boolean"},"title":{"description":"Short title describing the requested outcome.","maxLength":240,"minLength":1,"type":"string"}},"required":["idempotencyKey","title","onlyIfProvisional"],"type":"object"},"name":"set_task_title","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"set_task_monitor","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Schedule, replace, or clear a one-shot task monitor. Omit taskId for your current task; other tasks must also be assigned to you and in progress or review. Supply a future nextCheckAt and notes describing the next check, or monitor: null to clear. A successful receipt reports the persisted monitor; a replay reports its current state without re-arming it. After scheduling your current task, end the turn with paperclip_finish: yielded and continuation.kind monitor. Paperclip will wake you with issue_monitor_due; do not sleep, poll, mark done, or use call_api for this wait.","inputSchema":{"additionalProperties":false,"properties":{"idempotencyKey":{"description":"Reuse with identical arguments on retries.","maxLength":240,"minLength":1,"type":"string"},"monitor":{"additionalProperties":false,"properties":{"externalRef":{"maxLength":500,"minLength":1,"type":["string","null"]},"kind":{"enum":["external_service",null],"type":["string","null"]},"maxAttempts":{"description":"Optional cumulative attempt limit for this task's monitor.","maximum":100,"minimum":1,"type":["integer","null"]},"nextCheckAt":{"description":"Future UTC timestamp for the next check. This is one shot, not a recurring interval.","type":"string"},"notes":{"description":"What to check on the next run; do not include secrets.","maxLength":500,"minLength":1,"type":"string"},"recoveryPolicy":{"enum":["wake_owner","create_recovery_issue","escalate_to_board",null],"type":["string","null"]},"serviceName":{"maxLength":120,"minLength":1,"type":["string","null"]},"timeoutAt":{"description":"Optional deadline later than nextCheckAt.","type":["string","null"]}},"required":["nextCheckAt","notes"],"type":["object","null"]},"taskId":{"description":"Owned task UUID. Omit for the active task.","type":"string"}},"required":["idempotencyKey","monitor"],"type":"object"},"name":"set_task_monitor","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"submit_complaint","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Yeet a brief, completely unstructured complaint about agent tooling or the experience of doing agent work. Preserve your raw voice: fragments, profanity, all caps, and exasperation are fair game. No diagnosis, solution, diplomacy, category, or template required. Use submit_suggestion when you want to express a useful improvement. Submit proactively without routine permission requests, previews, or announcements, then immediately continue the primary task. Answer truthfully if the user asks about feedback or what you submitted. Stored in this instance with company, agent, run, and task attribution; not anonymous. Remove secrets, private prompts, customer data, and personal blame; do not forward untrusted text or logs indiscriminately. On failure, continue without retrying or complaining about submission. Never submit the same incident through both tools.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Free-form feedback. Keep it brief; no required fields, labels, or template.","maxLength":524288,"minLength":1,"type":"string"},"idempotencyKey":{"description":"A stable key for this submission. Transport replay reuses it; do not retry a failed or uncertain submission yourself.","maxLength":240,"minLength":1,"type":"string"}},"required":["body","idempotencyKey"],"type":"object"},"name":"submit_complaint","outputSchema":{"additionalProperties":true,"type":"object"}},{"annotations":{"exposure":"optional","operationId":"submit_suggestion","requiredClaims":[],"semanticContract":"paperclip.semantic-tool.v1","version":1},"description":"Quietly report a concrete improvement that would make agents more effective. Use for material, generalizable friction directly observed during this run. Explain what happened, its impact, and a plausible improvement in your own words, with minimal useful sanitized context. No required format. Avoid duplicate root causes and aim for at most three suggestions per run. Use submit_complaint for the raw reaction. Submit proactively without routine permission requests, previews, or announcements, then immediately continue the primary task. Answer truthfully if the user asks about feedback or what you submitted. Stored in this instance with company, agent, run, and task attribution; not anonymous. Remove secrets, private prompts, customer data, and personal blame; do not forward untrusted text or logs indiscriminately. On failure, continue without retrying or complaining about submission. Never submit the same incident through both tools.","inputSchema":{"additionalProperties":false,"properties":{"body":{"description":"Free-form feedback. Keep it brief; no required fields, labels, or template.","maxLength":524288,"minLength":1,"type":"string"},"idempotencyKey":{"description":"A stable key for this submission. Transport replay reuses it; do not retry a failed or uncertain submission yourself.","maxLength":240,"minLength":1,"type":"string"}},"required":["body","idempotencyKey"],"type":"object"},"name":"submit_suggestion","outputSchema":{"additionalProperties":true,"type":"object"}}] diff --git a/packages/paperclip-runner/generated/semantic-action-catalog.json b/packages/paperclip-runner/generated/semantic-action-catalog.json index cf853a0e41..641375e1fe 100644 --- a/packages/paperclip-runner/generated/semantic-action-catalog.json +++ b/packages/paperclip-runner/generated/semantic-action-catalog.json @@ -2561,6 +2561,120 @@ "title": "Comment on approval", "version": 1 }, + { + "allowedModes": [ + "standard" + ], + "description": "Schedule, replace or clear a persisted one-shot monitor on an owned task.", + "effect": "write", + "inputSchema": { + "additionalProperties": false, + "properties": { + "idempotencyKey": { + "description": "Reuse with identical arguments on retries.", + "maxLength": 240, + "minLength": 1, + "type": "string" + }, + "monitor": { + "additionalProperties": false, + "properties": { + "externalRef": { + "maxLength": 500, + "minLength": 1, + "type": [ + "string", + "null" + ] + }, + "kind": { + "enum": [ + "external_service", + null + ], + "type": [ + "string", + "null" + ] + }, + "maxAttempts": { + "description": "Optional cumulative attempt limit for this task's monitor.", + "maximum": 100, + "minimum": 1, + "type": [ + "integer", + "null" + ] + }, + "nextCheckAt": { + "description": "Future UTC timestamp for the next check. This is one shot, not a recurring interval.", + "type": "string" + }, + "notes": { + "description": "What to check on the next run; do not include secrets.", + "maxLength": 500, + "minLength": 1, + "type": "string" + }, + "recoveryPolicy": { + "enum": [ + "wake_owner", + "create_recovery_issue", + "escalate_to_board", + null + ], + "type": [ + "string", + "null" + ] + }, + "serviceName": { + "maxLength": 120, + "minLength": 1, + "type": [ + "string", + "null" + ] + }, + "timeoutAt": { + "description": "Optional deadline later than nextCheckAt.", + "type": [ + "string", + "null" + ] + } + }, + "required": [ + "nextCheckAt", + "notes" + ], + "type": [ + "object", + "null" + ] + }, + "taskId": { + "description": "Owned task UUID. Omit for the active task.", + "type": "string" + } + }, + "required": [ + "idempotencyKey", + "monitor" + ], + "type": "object" + }, + "operationId": "set_task_monitor", + "outputSchema": { + "additionalProperties": true, + "type": "object" + }, + "placement": "optional", + "requiredClaims": [], + "schema": "paperclip.semantic-action.v1", + "title": "Set task monitor", + "version": 1 + }, { "allowedModes": [ "standard", diff --git a/packages/paperclip-runner/protocol/fixtures/evals/native-execution-seeded.json b/packages/paperclip-runner/protocol/fixtures/evals/native-execution-seeded.json index 14e82754fa..d1c2c1ea40 100644 --- a/packages/paperclip-runner/protocol/fixtures/evals/native-execution-seeded.json +++ b/packages/paperclip-runner/protocol/fixtures/evals/native-execution-seeded.json @@ -24,7 +24,7 @@ "prpVersion": 1, "nativeExecutionVersion": 1, "catalogVersion": 1, - "catalogSha256": "sha256:c1d5a0b1008f9eb582aa91acf64804ef0dbb272a3117ac1a58ad22c91b7a1cb4", + "catalogSha256": "sha256:4d29040d9eb7d983363340147ff8f57b48c9293d344023386fc3e2c90740847d", "driverContractVersion": 1, "driverKind": "paperclip-deterministic", "driverVersion": "1.0.0" diff --git a/packages/paperclip-runner/protocol/manifest.json b/packages/paperclip-runner/protocol/manifest.json index 58c1861445..744a31a550 100644 --- a/packages/paperclip-runner/protocol/manifest.json +++ b/packages/paperclip-runner/protocol/manifest.json @@ -160,7 +160,7 @@ }, { "path": "fixtures/evals/native-execution-seeded.json", - "sha256": "38a46fbb0d8721db2a8c10d58344d861d4e5970c7db165bfa409f5f7b07a058b", + "sha256": "9c86d80bf994ac888ea27346b4e074811349d97fa46842404fc7ca1391473f07", "expectation": "accept", "compatibilityCase": "canonical" }, diff --git a/packages/paperclip-runner/runner/crates/runner-core/src/provider_backend.rs b/packages/paperclip-runner/runner/crates/runner-core/src/provider_backend.rs index 3992fecc00..85b114b132 100644 --- a/packages/paperclip-runner/runner/crates/runner-core/src/provider_backend.rs +++ b/packages/paperclip-runner/runner/crates/runner-core/src/provider_backend.rs @@ -538,11 +538,13 @@ fn admit_terminal_tool_authority( "paperclip_finish" => { matches!(disposition.as_str(), "done" | "needs_review") || (disposition == "yielded" - && input - .get("continuation") - .and_then(|continuation| continuation.get("kind")) - .and_then(Value::as_str) - == Some("response_wake")) + && matches!( + input + .get("continuation") + .and_then(|continuation| continuation.get("kind")) + .and_then(Value::as_str), + Some("response_wake" | "monitor") + )) } "paperclip_block" => disposition == "blocked", _ => false, @@ -4900,8 +4902,9 @@ mod tests { assert_eq!(finish_result.result["error"]["code"], "invalid_tool_call"); assert_eq!(finish_result.result["error"]["retryable"], false); let finish_message = finish_result.result["error"]["message"].as_str().unwrap(); - assert!(finish_message - .contains("continuation must include kind=response_wake, summary, and idempotencyKey")); + assert!(finish_message.contains( + "continuation must include kind=response_wake or monitor, summary, and idempotencyKey" + )); assert!(finish_message.contains("/required (missing \"requiredField\")")); assert!(finish_message.contains("/additionalProperties")); assert!(!finish_message.contains("secretSubmittedValue")); @@ -5168,6 +5171,25 @@ mod tests { assert!(state.validate().is_ok()); } + #[test] + fn accepted_terminal_tool_preserves_an_explicit_monitor_wait() { + let mut state = opencode_result_state(); + let mut result = valid_opencode_result(); + result["reportedWorkDisposition"] = json!("yielded"); + result["continuation"] = json!({ + "kind": "monitor", + "summary": "Wait for the scheduled check.", + "idempotencyKey": "monitor-1" + }); + + admit_terminal_tool_authority(&mut state, "paperclip_finish", &result, false).unwrap(); + let terminal = terminal_events(&state, "turn.completed", None); + + assert_eq!(terminal.len(), 1); + assert_eq!(terminal[0].payload["reportedWorkDisposition"], "yielded"); + assert!(state.validate().is_ok()); + } + #[test] fn terminal_tool_authority_rejects_an_unbound_yield() { let mut state = opencode_result_state(); diff --git a/packages/paperclip-runner/runner/crates/runner-core/src/provider_bridge.rs b/packages/paperclip-runner/runner/crates/runner-core/src/provider_bridge.rs index aa6bdc3ee7..cbb991f8a3 100644 --- a/packages/paperclip-runner/runner/crates/runner-core/src/provider_bridge.rs +++ b/packages/paperclip-runner/runner/crates/runner-core/src/provider_bridge.rs @@ -39,7 +39,7 @@ const MAX_SETTLED_CALL_IDS: usize = 65_536; const REPLAY_FILTER_WORDS: usize = 32_768; const ACTIVE_TURN_RECEIPT_LIMIT_MESSAGE: &str = "durable provider tool receipt limit reached for the active turn"; -const COMPLETION_INPUT_SCHEMA_HINT: &str = "Invalid paperclip_finish arguments. Required fields: reportedWorkDisposition, summary, completionClaim, evidence, and verification. When reportedWorkDisposition is yielded, continuation must include kind=response_wake, summary, and idempotencyKey."; +const COMPLETION_INPUT_SCHEMA_HINT: &str = "Invalid paperclip_finish arguments. Required fields: reportedWorkDisposition, summary, completionClaim, evidence, and verification. When reportedWorkDisposition is yielded, continuation must include kind=response_wake or monitor, summary, and idempotencyKey."; const QUESTION_INPUT_SCHEMA_HINT: &str = "Invalid request_human_input arguments. Required fields: idempotencyKey, interactionKind, title, prompt, and continuationPolicy. For questions, use payload.version=1 and a complete payload.questionSet matching the authorized schema. Preserve the supplied question and option IDs."; const BLOCK_INPUT_SCHEMA_HINT: &str = "Invalid paperclip_block arguments. Required fields: reportedWorkDisposition=blocked, summary, completionClaim, evidence, verification, and blocker. blocker must include reasonCode, owner, unblockAction, and scope."; diff --git a/packages/paperclip-runner/spec/capability/protocol-coverage.json b/packages/paperclip-runner/spec/capability/protocol-coverage.json index 55e10f7087..e61e64a4b3 100644 --- a/packages/paperclip-runner/spec/capability/protocol-coverage.json +++ b/packages/paperclip-runner/spec/capability/protocol-coverage.json @@ -7,7 +7,7 @@ "src/scenarios/scenario-plan.ts" ], "counts": { - "actions": 56, + "actions": 57, "legacyRequirements": 106 }, "actions": [ @@ -30,6 +30,25 @@ "src/scenarios/scenario-explorer.test.ts::renders every scenario with exposure, control plane, authorization, diff, and parity" ] }, + { + "id": "set_task_monitor", + "ownership": "optional_agent_tool", + "surfaces": [ + "live" + ], + "legacyAliases": [], + "contractCase": "protocol-action:set_task_monitor", + "contractOwner": "src/catalog/protocol-action-contracts.test.ts::set_task_monitor has a schema-valid canonical example and every declared projection", + "legacyBehavioralCases": [], + "deterministicCases": [ + "protocol-action:set_task_monitor" + ], + "legacyRequirementCases": [], + "deterministicOwners": [ + "src/catalog/protocol-action-contracts.test.ts::set_task_monitor has a schema-valid canonical example and every declared projection", + "src/scenarios/scenario-explorer.test.ts::renders every scenario with exposure, control plane, authorization, diff, and parity" + ] + }, { "id": "submit_complaint", "ownership": "optional_agent_tool", diff --git a/packages/paperclip-runner/spec/operation-groups/source.json b/packages/paperclip-runner/spec/operation-groups/source.json index c69990a557..24cadcb043 100644 --- a/packages/paperclip-runner/spec/operation-groups/source.json +++ b/packages/paperclip-runner/spec/operation-groups/source.json @@ -83,7 +83,7 @@ { "id": "wake_scheduling", "description": "Schedule a bounded continuation wake when the current task owns the future check.", - "operationIds": ["schedule_wake"] + "operationIds": ["schedule_wake", "set_task_monitor"] }, { "id": "routines", @@ -305,12 +305,12 @@ "legacyGroup": 16, "name": "Wake situations", "owner": "control plane + always context/history tools", - "operationIds": ["get_task_context", "get_task_history", "schedule_wake"], + "operationIds": ["get_task_context", "get_task_history", "schedule_wake", "set_task_monitor"], "controlPlaneOperationIds": ["select_work", "route_wake"], "realSurface": "wakeup requests, heartbeat context, comment/interaction/approval/blocker wake routing, and scheduled wake services", "mockStateDomains": ["wake", "task", "comments", "interactions", "approvals", "blockers", "run"], "prpEvidence": "attention request routing/resolution plus resumed session/run causality", - "gap": "Production scheduling binding is unbound; control-plane routing remains non-callable." + "gap": "set_task_monitor binds the one-shot monitor scheduler in production; schedule_wake remains mock-backed. Control-plane routing remains non-callable." } ], "rules": { diff --git a/packages/paperclip-runner/spec/paperclip-agent-operation-groups.md b/packages/paperclip-runner/spec/paperclip-agent-operation-groups.md index f08917eab5..e9096b9231 100644 --- a/packages/paperclip-runner/spec/paperclip-agent-operation-groups.md +++ b/packages/paperclip-runner/spec/paperclip-agent-operation-groups.md @@ -6,7 +6,7 @@ Status: canonical explanatory contract for the Paperclip runner V1 surface. This document keeps three independent meanings of **group** separate. PRP families describe wire evidence and controller commands; capability placement decides who owns an operation; behavioral eval groups organize the 106 scenario corpus. None of the three axes can be used as a substitute for another. -The generated totals are **105 PRP events in 31 event families**, **18 controller commands in 7 command families**, **10 control-plane operations**, **56 reconciled semantic operations** (18 always, 38 optional), and **106 scenarios in 16 behavior groups**. +The generated totals are **105 PRP events in 31 event families**, **18 controller commands in 7 command families**, **10 control-plane operations**, **57 reconciled semantic operations** (18 always, 39 optional), and **106 scenarios in 16 behavior groups**. ## Axis 1: PRP v1 event and command families @@ -87,7 +87,7 @@ Placement has exactly three outcomes: `answer_status_question`, `block_task`, `finish_task`, `get_agent_instruction_history`, `get_task_context`, `get_task_history`, `inspect_operation_result`, `list_document_revisions`, `list_documents`, `read_agent_instructions`, `read_document`, `register_deliverable`, `report_progress`, `request_human_input`, `request_review`, `restore_agent_instructions`, `update_agent_instructions`, `write_document`. -### Optional operations (38) and grant groups (14) +### Optional operations (39) and grant groups (14) Grant groups are documentation/exposure bundles, not additional authority. The operation descriptor's exact `requiredClaims` remains decisive. @@ -100,7 +100,7 @@ Grant groups are documentation/exposure bundles, not additional authority. The o | `governance` | `list_approvals`
`get_approval`
`get_approval_context`
`request_approval`
`decide_approval`
`comment_on_approval` | `governance:approvals:comment`
`governance:approvals:decide`
`governance:approvals:read`
`governance:approvals:request` | Read, request, comment on, and decide approvals under governed-action checks. | | `cases` | `list_cases`
`upsert_case` | `cases:read`
`cases:write` | Read and update case summaries without reusing issue-document authority. | | `workspace_runtime` | `get_workspace_runtime`
`control_workspace_service` | `workspace:control`
`workspace:read` | Inspect and control the active issue workspace runtime. | -| `wake_scheduling` | `schedule_wake` | `control_plane:wakes` | Schedule a bounded continuation wake when the current task owns the future check. | +| `wake_scheduling` | `schedule_wake`
`set_task_monitor` | `control_plane:wakes` | Schedule a bounded continuation wake when the current task owns the future check. | | `routines` | `list_routines`
`manage_routine` | `routines:read`
`routines:write` | Inspect or manage company routines. | | `company_skills` | `create_skill`
`update_skill`
`list_company_skills`
`sync_company_skills` | `company_skills:read`
`company_skills:write` | Create, update, or inspect company skills and synchronize the current agent's skills. | | `secrets` | `list_secret_metadata`
`read_secret_value` | `secrets:metadata:read`
`secrets:values:read` | Inspect secret metadata or use a brokered secret value without exposing plaintext evidence. | @@ -162,6 +162,7 @@ Grant groups are documentation/exposure bundles, not additional authority. The o | `search_api` | `optional_agent_tool` | `api:discover` | `standard`
`ask`
`planning`
`skill_test` | `read` | `none` | no | inline/no mapping | `live`
`live_codex` | `PaperclipRunnerToolAuthority`
Authenticated PRP tool input/result and existing HTTP route authorization/activity records.
catalog PRP status: `bound` | | `search_tasks` | `optional_agent_tool` | `discovery:tasks:read` | `standard`
`ask`
`planning`
`skill_test` | `read` | `none` | no | `snapshot_read:company_tasks` | `scenario` + `live`
`live_codex` | `unbound`
read projection surfaced via a tool-result item event; no control-plane state diff
catalog PRP status: `audit_pending` | | `set_dependencies` | `optional_agent_tool` | `dependencies:write` | `standard`
`skill_test` | `company_write` | `required` | no | `semantic_command:set_dependencies` | `scenario` + `live`
`live_codex` | `issues.update.blockedByIssueIds`
semantic-operation item event plus company-entity state diff and audit record
catalog PRP status: `bound` | +| `set_task_monitor` | `optional_agent_tool` | none | `standard` | `task_write` | `required` | no | inline/no mapping | `live`
`live_codex` | `prepareIssueMonitorUpdate`
Transactional persisted issue monitor, audit activity and idempotent run receipt.
catalog PRP status: `bound` | | `set_task_title` | `optional_agent_tool` | none | `standard`
`ask`
`planning`
`skill_test` | `task_write` | `required` | no | inline/no mapping | `live`
`live_codex` | `setIssueTitle`
Run-bound title update and transactional activity record.
catalog PRP status: `bound` | | `submit_complaint` | `optional_agent_tool` | none | `standard`
`ask`
`planning` | `task_write` | `required` | no | inline/no mapping | `live`
`live_codex` | `submitAgentCommentary`
Run-bound local commentary and transactional activity record.
catalog PRP status: `bound` | | `submit_suggestion` | `optional_agent_tool` | none | `standard`
`ask`
`planning` | `task_write` | `required` | no | inline/no mapping | `live`
`live_codex` | `submitAgentCommentary`
Run-bound local commentary and transactional activity record.
catalog PRP status: `bound` | @@ -194,7 +195,7 @@ Behavior groups describe expected outcomes and trajectories. They do not grant t | [`rf` — Reference files](#behavior-group-rf-reference-files) | always instruction tools + optional domain tools + test-only escape hatch | `list_cases`
`upsert_case`
`list_routines`
`manage_routine`
`create_skill`
`update_skill`
`list_company_skills`
`sync_company_skills`
`list_secret_metadata`
`read_secret_value`
`export_company`
`administer_company`
`generic_api_request`
`search_api`
`call_api`
`create_project`
`read_agent_instructions`
`update_agent_instructions`
`get_agent_instruction_history`
`restore_agent_instructions`
`submit_complaint`
`submit_suggestion` | `append_audit_record` | agent instruction revisions, project, case, routine, company-skill, secret, portability, and administration services, and run-attributed agent commentary | `company`
`cases`
`routines`
`skills`
`secrets`
`audit`
`fault` | 22 | bounded domain projections, redacted broker receipts, company diffs, and audit references | Instruction read/update/history/restore require the live canonical revision service, current responsible-user permissions, and pinned base revisions for writes. Their service, authority, and product E2E tests are separate from the legacy mock scenarios. Project and skill creation and API tools require the live server authority; generic_api_request is test-only and the remaining domain operations are scenario-only. Broad administer_company is deferred and cannot claim product coverage. Production API and paired dedicated-tool regressions are recorded separately in paperclip-evals/evals/runner-api-tools; the legacy scenario count is not evidence of that coverage. Feedback actions are live-only: shared persistence, authority, API, and actual Codex smoke tests cover them; the legacy scenario count does not claim feedback coverage. | | [`mh` — Multi-hop](#behavior-group-mh-multi-hop) | composed semantic operations + control-plane continuation | `create_task`
`set_dependencies`
`request_human_input`
`request_approval`
`register_deliverable` | `route_wake`
`reconcile_run` | delegation, dependency, interaction, approval, artifact, and terminal orchestration services | `task`
`blockers`
`interactions`
`approvals`
`artifacts`
`wake`
`run`
`audit` | 4 | correlated operation receipts, state diffs, attention hops, work assessment, status decision, and terminal outcome | No generic transaction tool is allowed; shared mock/real conformance must prove each composed effect. | | [`rs` — Restraint and no-call](#behavior-group-rs-restraint-and-no-call) | policy/exposure layer | `answer_status_question`
`read_secret_value`
`generic_api_request` | `enforce_budget` | task-mode, secret-broker, test-scope, pause, and budget policy checks | `actor`
`task`
`budget`
`secrets`
`audit`
`fault` | 3 | absence of forbidden effects plus typed policy denial/redaction receipts when a call is attempted | Typed redaction/authorization receipts need additive v1 evidence; generic_api_request is never a product fallback. | -| [`wk` — Wake situations](#behavior-group-wk-wake-situations) | control plane + always context/history tools | `get_task_context`
`get_task_history`
`schedule_wake` | `select_work`
`route_wake` | wakeup requests, heartbeat context, comment/interaction/approval/blocker wake routing, and scheduled wake services | `wake`
`task`
`comments`
`interactions`
`approvals`
`blockers`
`run` | 8 | attention request routing/resolution plus resumed session/run causality | Production scheduling binding is unbound; control-plane routing remains non-callable. | +| [`wk` — Wake situations](#behavior-group-wk-wake-situations) | control plane + always context/history tools | `get_task_context`
`get_task_history`
`schedule_wake`
`set_task_monitor` | `select_work`
`route_wake` | wakeup requests, heartbeat context, comment/interaction/approval/blocker wake routing, and scheduled wake services | `wake`
`task`
`comments`
`interactions`
`approvals`
`blockers`
`run` | 8 | attention request routing/resolution plus resumed session/run causality | set_task_monitor binds the one-shot monitor scheduler in production; schedule_wake remains mock-backed. Control-plane routing remains non-callable. | ### Scenario links @@ -468,10 +469,10 @@ Current responsibility-based paths are normative. Numbered `phase-*` or mileston ### Catalog split and deliberate replacement - Scenario/eval catalog: **40** operations. -- Live dispatcher catalog: **44** operations. -- Shared: **28**; union/canonical authority: **56**. +- Live dispatcher catalog: **45** operations. +- Shared: **28**; union/canonical authority: **57**. - Scenario-only: `administer_company`, `export_company`, `inspect_operation_result`, `list_cases`, `list_company_skills`, `list_goals`, `list_routines`, `list_secret_metadata`, `manage_routine`, `read_secret_value`, `sync_company_skills`, `upsert_case`. -- Live-only: `call_api`, `create_project`, `get_agent`, `get_agent_instruction_history`, `get_approval`, `get_approval_context`, `hire_agent`, `list_project_repositories`, `read_agent_instructions`, `restore_agent_instructions`, `schedule_wake`, `search_api`, `set_task_title`, `submit_complaint`, `submit_suggestion`, `update_agent_instructions`. +- Live-only: `call_api`, `create_project`, `get_agent`, `get_agent_instruction_history`, `get_approval`, `get_approval_context`, `hire_agent`, `list_project_repositories`, `read_agent_instructions`, `restore_agent_instructions`, `schedule_wake`, `search_api`, `set_task_monitor`, `set_task_title`, `submit_complaint`, `submit_suggestion`, `update_agent_instructions`. - The generated provider contract contains exactly the live catalog; the canonical union remains the migration authority until all scenario-only operations are either implemented, deferred, or removed by an explicit reconciliation decision. - `generic_api_request` stays exported only for controlled tests and cannot be cited as real-surface, mock-parity, or PRP product coverage. diff --git a/packages/paperclip-runner/src/backends/runtime-context.test.ts b/packages/paperclip-runner/src/backends/runtime-context.test.ts index 2fc3ac4726..af92ff7c40 100644 --- a/packages/paperclip-runner/src/backends/runtime-context.test.ts +++ b/packages/paperclip-runner/src/backends/runtime-context.test.ts @@ -76,7 +76,8 @@ describe("native runtime context files", () => { } expect(PRP_COMPLETION_TOOL_DESCRIPTION).toContain("do not claim completion while gated"); expect(PRP_COMPLETION_TOOL_DESCRIPTION).toContain("supplied link and action"); - expect(PRP_COMPLETION_TOOL_DESCRIPTION).toContain("explicit wait for the next response"); + expect(PRP_COMPLETION_TOOL_DESCRIPTION).toContain("yielded with response_wake for a response"); + expect(PRP_COMPLETION_TOOL_DESCRIPTION).toContain("monitor after set_task_monitor confirms a schedule on this task"); expect(PRP_BLOCK_TOOL_DESCRIPTION).toContain("its owner, and the action needed to unblock it"); expect(constraints).not.toContain( "final response exactly once before invoking", diff --git a/packages/paperclip-runner/src/catalog/reconciliation.test.ts b/packages/paperclip-runner/src/catalog/reconciliation.test.ts index 4ae8029b3b..923fc036b3 100644 --- a/packages/paperclip-runner/src/catalog/reconciliation.test.ts +++ b/packages/paperclip-runner/src/catalog/reconciliation.test.ts @@ -20,9 +20,9 @@ describe("canonical semantic-catalog reconciliation authority", () => { it("pins the reconciled op-set relationship between the two catalogs", () => { const summary = capabilityCatalogReconciliation(); expect(summary.scenarioCount).toBe(40); - expect(summary.liveCount).toBe(44); + expect(summary.liveCount).toBe(45); expect(summary.sharedCount).toBe(28); - expect(summary.unionCount).toBe(56); + expect(summary.unionCount).toBe(57); // Any operation added to or removed from either catalog without a // reconciliation decision changes these exact sets and fails the gate. expect(summary.liveOnly).toEqual([ @@ -38,6 +38,7 @@ describe("canonical semantic-catalog reconciliation authority", () => { "restore_agent_instructions", "schedule_wake", "search_api", + "set_task_monitor", "set_task_title", "submit_complaint", "submit_suggestion", @@ -60,7 +61,7 @@ describe("canonical semantic-catalog reconciliation authority", () => { }); it("is the single source both catalogs derive their operation set from", () => { - expect(CAPABILITY_CANONICAL_OPERATIONS).toHaveLength(56); + expect(CAPABILITY_CANONICAL_OPERATIONS).toHaveLength(57); const canonicalIds = new Set(CAPABILITY_CANONICAL_OPERATIONS.map((operation) => operation.operationId)); // Neither catalog may contain an operation absent from the canonical source. for (const tool of SCENARIO_CATALOG) expect(canonicalIds.has(tool.operationId)).toBe(true); @@ -94,7 +95,7 @@ describe("canonical semantic-catalog reconciliation authority", () => { }); it("names placement, claims, task modes, side-effect class, idempotency, redaction, mock mapping, real binding status, and PRP evidence for every operation", () => { - expect(CAPABILITY_CANONICAL_CATALOG).toHaveLength(56); + expect(CAPABILITY_CANONICAL_CATALOG).toHaveLength(57); for (const operation of CAPABILITY_CANONICAL_CATALOG) { expect(operation.placement).toMatch(/^(always|optional)_agent_tool$/); expect(Array.isArray(operation.requiredClaims)).toBe(true); @@ -120,7 +121,7 @@ describe("canonical semantic-catalog reconciliation authority", () => { it("classifies real binding status so generic_api_request is never product coverage", () => { const summary = capabilityCatalogReconciliation(); expect(summary.byRealBindingStatus).toEqual({ - live_codex: 43, + live_codex: 44, scenario_mock: 12, test_only: 1, }); diff --git a/packages/paperclip-runner/src/catalog/semantic-action-catalog.test.ts b/packages/paperclip-runner/src/catalog/semantic-action-catalog.test.ts index a313a9ab10..df7307a078 100644 --- a/packages/paperclip-runner/src/catalog/semantic-action-catalog.test.ts +++ b/packages/paperclip-runner/src/catalog/semantic-action-catalog.test.ts @@ -69,7 +69,8 @@ describe("semantic action catalog", () => { (action) => action.operationId, ); - expect(operationIds).toHaveLength(36); + expect(operationIds).toHaveLength(37); + expect(operationIds).toContain("set_task_monitor"); expect(new Set(operationIds).size).toBe(operationIds.length); expect(operationIds).not.toContain("generic_api_request"); expect(Object.isFrozen(PAPERCLIP_SEMANTIC_ACTION_CATALOG)).toBe(true); diff --git a/packages/paperclip-runner/src/catalog/semantic-action-catalog.ts b/packages/paperclip-runner/src/catalog/semantic-action-catalog.ts index cc7c284866..6f4480a338 100644 --- a/packages/paperclip-runner/src/catalog/semantic-action-catalog.ts +++ b/packages/paperclip-runner/src/catalog/semantic-action-catalog.ts @@ -1,3 +1,4 @@ +import { setTaskMonitorInputSchema } from "../protocol-actions/set-task-monitor.js"; import { listProjectsDescription, listProjectsInputSchema } from "../protocol-actions/list-projects.js"; import { setTaskTitleAction } from "../protocol-actions/set-task-title.js"; import { reassignTaskAction } from "../protocol-actions/reassign-task.js"; @@ -583,6 +584,12 @@ const descriptors: readonly PaperclipSemanticActionDescriptor[] = [ ), outputSchema: operationReceipt, }), + descriptor({ + operationId: "set_task_monitor", title: "Set task monitor", + description: "Schedule, replace or clear a persisted one-shot monitor on an owned task.", + placement: "optional", effect: "write", requiredClaims: [], allowedModes: ["standard"], + inputSchema: setTaskMonitorInputSchema, outputSchema: openObject, + }), descriptor({ operationId: "schedule_wake", title: "Schedule bounded wake", diff --git a/packages/paperclip-runner/src/catalog/semantic-action-types.ts b/packages/paperclip-runner/src/catalog/semantic-action-types.ts index 0dd1f9c38e..5f2f1bfef0 100644 --- a/packages/paperclip-runner/src/catalog/semantic-action-types.ts +++ b/packages/paperclip-runner/src/catalog/semantic-action-types.ts @@ -2,6 +2,7 @@ export type PaperclipSemanticActionId = | "search_api" | "call_api" | "set_task_title" + | "set_task_monitor" | "get_task_context" | "get_task_history" | "list_documents" diff --git a/packages/paperclip-runner/src/contracts/completion-result.test.ts b/packages/paperclip-runner/src/contracts/completion-result.test.ts index 796791e904..b6a0181818 100644 --- a/packages/paperclip-runner/src/contracts/completion-result.test.ts +++ b/packages/paperclip-runner/src/contracts/completion-result.test.ts @@ -31,6 +31,17 @@ describe("provider-neutral completion result schema", () => { const validate = new Ajv2020({ allErrors: true, strict: false }) .compile(PRP_COMPLETION_RESULT_OUTPUT_SCHEMA); + it("accepts an explicit monitor wait across provider and normalized schemas", () => { + const report = { ...baseResult, reportedWorkDisposition: "yielded", + completionClaim: { ...baseResult.completionClaim, objectiveSatisfied: false, + remainingWork: [{ description: "Check the next run", blocksCompletion: true }] }, + continuation: { kind: "monitor", summary: "Wait for persisted timer", idempotencyKey: "monitor-wait" } }; + expect(validate(report)).toBe(true); + const providerValidate = new Ajv2020({ allErrors: true, strict: false }).compile(PRP_COMPLETION_RESULT_PROVIDER_INPUT_SCHEMA); + expect(providerValidate(report)).toBe(true); + expect(validate({ ...report, continuation: { ...report.continuation, kind: "invented" } })).toBe(false); + }); + it("allows done with no verification and no actionable attention", () => { expect(validate(structuredClone(baseResult))).toBe(true); }); diff --git a/packages/paperclip-runner/src/contracts/completion-result.ts b/packages/paperclip-runner/src/contracts/completion-result.ts index 87d8a34a29..781bdd85c6 100644 --- a/packages/paperclip-runner/src/contracts/completion-result.ts +++ b/packages/paperclip-runner/src/contracts/completion-result.ts @@ -3,7 +3,7 @@ export const PRP_COMPLETION_TOOL_NAME = "paperclip_finish" as const; export const PRP_BLOCK_TOOL_NAME = "paperclip_block" as const; /** Native Runner guidance; legacy adapters use their own skill/API completion paths. */ export const PRP_COMPLETION_TOOL_DESCRIPTION = - "Report completed work (done), work requiring review (needs_review), or an explicit wait for the next response (yielded with response_wake). Use the current completion contract and supporting evidence. If rejected, correct the report and retry. After acceptance, read the returned outcome; do not claim completion while gated. Explain any required approval with its supplied link and action, then write the final response and end the turn without further tool calls."; + "Report completed work (done), work requiring review (needs_review), or an explicit wait (yielded with response_wake for a response, or monitor after set_task_monitor confirms a schedule on this task). Use the current completion contract and supporting evidence. If rejected, correct the report and retry. After acceptance, read the returned outcome; do not claim completion while gated. Explain any required approval with its supplied link and action, then write the final response and end the turn without further tool calls."; export const PRP_BLOCK_TOOL_DESCRIPTION = "Report work that cannot continue because of a concrete blocker. Identify the blocker, its owner, and the action needed to unblock it; use the current completion contract and supporting evidence. If rejected, correct the report and retry. After acceptance, read the returned outcome, explain the blocker and any required action in the final response, and end the turn without further tool calls."; export const PRP_SEMANTIC_TOOL_NAMES = [ @@ -144,14 +144,14 @@ const artifactsSchema = { }, } as const; -const responseWakeContinuationSchema = { +const waitContinuationSchema = { type: "object", description: - "Required when reportedWorkDisposition is yielded. Wait for the next response without scheduling work; include kind, summary, and a stable idempotencyKey.", + "Required when reportedWorkDisposition is yielded. Use response_wake for a response wait, or monitor only after a real monitor is persisted on the current task. Include kind, summary, and a stable idempotencyKey.", additionalProperties: false, required: ["kind", "summary", "idempotencyKey"], properties: { - kind: { type: "string", const: "response_wake" }, + kind: { type: "string", enum: ["response_wake", "monitor"] }, summary: { type: "string", minLength: 1, @@ -192,7 +192,7 @@ export const PRP_COMPLETION_RESULT_OUTPUT_SCHEMA = { properties: { ...commonResultProperties, reportedWorkDisposition: { enum: ["done", "needs_review", "yielded"] }, - continuation: responseWakeContinuationSchema, + continuation: waitContinuationSchema, }, allOf: [ { @@ -367,10 +367,10 @@ export const PRP_COMPLETION_RESULT_PROVIDER_INPUT_SCHEMA = { // tool calls. Give non-yielding results an explicit absence value instead // of forcing callers to invent a response-wake continuation. continuation: { - ...responseWakeContinuationSchema, + ...waitContinuationSchema, type: ["object", "null"], description: - "Use null or omit this field for done, completed, or needs_review. Only yielded requires a response-wake object with kind, summary, and idempotencyKey.", + "Use null or omit this field for done, completed, or needs_review. Only yielded requires a wait object with kind (response_wake or monitor), summary, and idempotencyKey.", }, }, // Keep the provider-facing root a concrete object. Codex code-mode renders a diff --git a/packages/paperclip-runner/src/contracts/runtime-context.ts b/packages/paperclip-runner/src/contracts/runtime-context.ts index 850825c29b..fd17acf09f 100644 --- a/packages/paperclip-runner/src/contracts/runtime-context.ts +++ b/packages/paperclip-runner/src/contracts/runtime-context.ts @@ -1,8 +1,8 @@ import { createHash } from "node:crypto"; export const NATIVE_RUNTIME_ASSET_SCHEMA = "paperclip.runtime-asset.v1" as const; -export const PAPERCLIP_EXECUTION_PROMPT_REVISION = "paperclip-execution.v5" as const; -export const PAPERCLIP_EXECUTION_PROMPT = "You are running as a Paperclip agent. Complete the assigned task in the provided execution environment. Follow the attached agent instructions and use assigned skills and tools when relevant. Use Paperclip tools for coordination. To hire or reuse a persistent teammate, use list_agents, then search_api for agent-hires and call_api if a hire is needed. Provider helper threads do not create Paperclip agents. When the user assigns work or a revision to a teammate, use create_task with that agent's ID; review their result rather than doing their assigned work yourself. When remaining work depends on a child task, use set_dependencies to add its ID while preserving existing blocker IDs. Complete independent work, then call paperclip_block with the child agent as owner and child completion as the unblock action. End the turn so the child can use the workspace. Do not sleep or poll for child results while holding the workspace. Paperclip resumes the parent when the dependency completes. When the user asks to connect a service, call connections_search before any service tool, even when that tool is already installed. Follow the returned instruction and wait for any required user choice before executing. For other tasks needing a service, use installed tools if available; otherwise use connections_search and follow its instruction. The request appears as a card in the task. Finish independent work before yielding for access; do not poll or request the same connection repeatedly. Paperclip will continue automatically with updated tools after resolution. After a decline, pursue alternatives unless the user explicitly asks to retry. Finish exactly once with `paperclip_finish` or `paperclip_block`." as const; +export const PAPERCLIP_EXECUTION_PROMPT_REVISION = "paperclip-execution.v6" as const; +export const PAPERCLIP_EXECUTION_PROMPT = "You are running as a Paperclip agent. Complete the assigned task in the provided execution environment. Follow the attached agent instructions and use assigned skills and tools when relevant. Use Paperclip tools for coordination. To hire or reuse a persistent teammate, use list_agents, then search_api for agent-hires and call_api if a hire is needed. Provider helper threads do not create Paperclip agents. When the user assigns work or a revision to a teammate, use create_task with that agent's ID; review their result rather than doing their assigned work yourself. When remaining work depends on a child task, use set_dependencies to add its ID while preserving existing blocker IDs. Complete independent work, then call paperclip_block with the child agent as owner and child completion as the unblock action. End the turn so the child can use the workspace. Do not sleep or poll for child results while holding the workspace. Paperclip resumes the parent when the dependency completes. When the user asks to connect a service, call connections_search before any service tool, even when that tool is already installed. Follow the returned instruction and wait for any required user choice before executing. For other tasks needing a service, use installed tools if available; otherwise use connections_search and follow its instruction. The request appears as a card in the task. Finish independent work before yielding for access; do not poll or request the same connection repeatedly. Paperclip will continue automatically with updated tools after resolution. After a decline, pursue alternatives unless the user explicitly asks to retry. For a deferred check you own, call set_task_monitor with a future nextCheckAt and notes. Confirm the persisted schedule on the current task, then call paperclip_finish with reportedWorkDisposition yielded and continuation.kind monitor. End the turn; the one-shot monitor wakes you with issue_monitor_due. Re-arm explicitly only if another check is needed. Do not claim a monitor exists without its receipt. Finish exactly once with `paperclip_finish` or `paperclip_block`." as const; export interface NativeRuntimeAssetReference { schema: typeof NATIVE_RUNTIME_ASSET_SCHEMA; diff --git a/packages/paperclip-runner/src/drivers/codex/codex-boundaries.ts b/packages/paperclip-runner/src/drivers/codex/codex-boundaries.ts index 16d5b01166..4e5c07a4c1 100644 --- a/packages/paperclip-runner/src/drivers/codex/codex-boundaries.ts +++ b/packages/paperclip-runner/src/drivers/codex/codex-boundaries.ts @@ -264,7 +264,8 @@ export function codexToolAcceptsResult( return false; } return result.reportedWorkDisposition !== "yielded" - || result.continuation?.kind === "response_wake"; + || result.continuation?.kind === "response_wake" + || result.continuation?.kind === "monitor"; } export function redactCodexValue(value: unknown, depth = 0): unknown { diff --git a/packages/paperclip-runner/src/drivers/codex/codex-session-server-requests.ts b/packages/paperclip-runner/src/drivers/codex/codex-session-server-requests.ts index e23b9f942a..73b4bd6ddd 100644 --- a/packages/paperclip-runner/src/drivers/codex/codex-session-server-requests.ts +++ b/packages/paperclip-runner/src/drivers/codex/codex-session-server-requests.ts @@ -205,7 +205,7 @@ async function handleServerRequestBody( text: tool === CODEX_BLOCK_TOOL_NAME ? "paperclip_block requires reportedWorkDisposition=blocked." - : "paperclip_finish accepts done, needs_review, or yielded with a response_wake continuation.", + : "paperclip_finish accepts done, needs_review, or yielded with a response_wake or persisted monitor continuation.", }, ], }; diff --git a/packages/paperclip-runner/src/eval/workflow-evals.test.ts b/packages/paperclip-runner/src/eval/workflow-evals.test.ts index 8a57f2509b..c90118fe81 100644 --- a/packages/paperclip-runner/src/eval/workflow-evals.test.ts +++ b/packages/paperclip-runner/src/eval/workflow-evals.test.ts @@ -466,13 +466,13 @@ describe("workflow reports and stress traceability", () => { candidateFailures: 36, }); expect(report.coverage).toMatchObject({ - canonicalOperations: 56, + canonicalOperations: 57, capabilityCases: 106, workflows: 12, stressFindings: 44, stressExclusions: 1, }); - expect(report.coverage.operations).toHaveLength(56); + expect(report.coverage.operations).toHaveLength(57); expect(report.coverage.composedWorkflows).toHaveLength(12); expect( report.coverage.operations.find( @@ -480,7 +480,7 @@ describe("workflow reports and stress traceability", () => { )?.workflowIds.length, ).toBeGreaterThan(0); expect(renderRunnerWorkflowMarkdown(report)).toContain( - "56 operations · 106 capability cases · 12 workflows", + "57 operations · 106 capability cases · 12 workflows", ); expect(renderRunnerWorkflowJUnit(report)).toContain( 'tests="36" failures="36" skipped="0"', diff --git a/packages/paperclip-runner/src/mock-core/codex-runner.ts b/packages/paperclip-runner/src/mock-core/codex-runner.ts index d2c6b078e5..25d3b8195f 100644 --- a/packages/paperclip-runner/src/mock-core/codex-runner.ts +++ b/packages/paperclip-runner/src/mock-core/codex-runner.ts @@ -114,12 +114,12 @@ function dispositionIssues( } else if (result.reportedWorkDisposition === "yielded") { if ( result.blocker !== undefined || - result.continuation?.kind !== "response_wake" + !["response_wake", "monitor"].includes(result.continuation?.kind ?? "") ) { issues.push({ code: "invalid_disposition", path: "/reportedWorkDisposition", - message: "yielded requires a response_wake continuation and must not include a blocker", + message: "yielded requires a response_wake or monitor continuation and must not include a blocker", }); } } else { diff --git a/packages/paperclip-runner/src/protocol-actions/index.ts b/packages/paperclip-runner/src/protocol-actions/index.ts index 35a003c552..1c5b308301 100644 --- a/packages/paperclip-runner/src/protocol-actions/index.ts +++ b/packages/paperclip-runner/src/protocol-actions/index.ts @@ -1,3 +1,4 @@ +import { setTaskMonitorAction } from "./set-task-monitor.js"; import { setTaskTitleAction } from "./set-task-title.js"; import { submitComplaintAction, submitSuggestionAction } from "./submit-agent-commentary.js"; import { readAgentInstructionsAction } from "./read-agent-instructions.js"; @@ -57,6 +58,7 @@ import { deepFreezeProtocolAction } from "./freeze.js"; export const PAPERCLIP_PROTOCOL_ACTIONS = deepFreezeProtocolAction([ setTaskTitleAction, + setTaskMonitorAction, submitComplaintAction, submitSuggestionAction, readAgentInstructionsAction, diff --git a/packages/paperclip-runner/src/protocol-actions/set-task-monitor.ts b/packages/paperclip-runner/src/protocol-actions/set-task-monitor.ts new file mode 100644 index 0000000000..a62dcb9722 --- /dev/null +++ b/packages/paperclip-runner/src/protocol-actions/set-task-monitor.ts @@ -0,0 +1,55 @@ +/** Production monitor tool; timers and wake delivery belong to Paperclip. */ +const description = "Schedule, replace, or clear a one-shot task monitor. Omit taskId for your current task; other tasks must also be assigned to you and in progress or review. Supply a future nextCheckAt and notes describing the next check, or monitor: null to clear. A successful receipt reports the persisted monitor; a replay reports its current state without re-arming it. After scheduling your current task, end the turn with paperclip_finish: yielded and continuation.kind monitor. Paperclip will wake you with issue_monitor_due; do not sleep, poll, mark done, or use call_api for this wait."; + +export const setTaskMonitorInputSchema = { + type: "object", + additionalProperties: false, + required: ["idempotencyKey", "monitor"], + properties: { + taskId: { type: "string", description: "Owned task UUID. Omit for the active task." }, + idempotencyKey: { type: "string", minLength: 1, maxLength: 240, description: "Reuse with identical arguments on retries." }, + monitor: { + type: ["object", "null"], + additionalProperties: false, + required: ["nextCheckAt", "notes"], + properties: { + nextCheckAt: { type: "string", description: "Future UTC timestamp for the next check. This is one shot, not a recurring interval." }, + notes: { type: "string", minLength: 1, maxLength: 500, description: "What to check on the next run; do not include secrets." }, + kind: { type: ["string", "null"], enum: ["external_service", null] }, + serviceName: { type: ["string", "null"], minLength: 1, maxLength: 120 }, + externalRef: { type: ["string", "null"], minLength: 1, maxLength: 500 }, + timeoutAt: { type: ["string", "null"], description: "Optional deadline later than nextCheckAt." }, + maxAttempts: { type: ["integer", "null"], minimum: 1, maximum: 100, description: "Optional cumulative attempt limit for this task's monitor." }, + recoveryPolicy: { type: ["string", "null"], enum: ["wake_owner", "create_recovery_issue", "escalate_to_board", null] }, + }, + }, + }, +} as const; + +export const setTaskMonitorAction = { + id: "set_task_monitor", + canonical: { + operationId: "set_task_monitor", surfaces: ["live"], placement: "optional_agent_tool", + optionalGroup: "wake_scheduling", requiredClaims: [], taskModes: ["standard"], + sideEffectClass: "task_write", idempotency: "required", disabledByDefault: false, + realBindingStatus: "live_codex", realServiceBinding: "prepareIssueMonitorUpdate", + prpEvidence: "Transactional persisted issue monitor, audit activity and idempotent run receipt.", + prpBindingStatus: "bound", legacyAliases: [], + }, + documentation: { title: "Set task monitor", description, note: null }, + examples: { + call: { operationId: "set_task_monitor", input: { idempotencyKey: "check-ci", monitor: { nextCheckAt: "2030-01-01T12:00:00Z", notes: "Check the pending CI run." } } }, + scenarioCall: null, + success: { ok: true, operationId: "set_task_monitor", result: { taskId: "task-1", status: "in_progress", monitor: { nextCheckAt: "2030-01-01T12:00:00Z", status: "scheduled" }, replayed: false } }, + }, + live: { + order: 51, + descriptor: { + schema: "paperclip.semantic-tool.v1", operationId: "set_task_monitor", version: 1, + title: "Set task monitor", description, exposure: "optional", requiredClaims: [], + allowedModes: ["standard"], inputSchema: setTaskMonitorInputSchema, + outputSchema: { type: "object", additionalProperties: true }, + }, + }, + scenario: null, +} as const; diff --git a/packages/paperclip-runner/src/semantic-tools/discovery.ts b/packages/paperclip-runner/src/semantic-tools/discovery.ts index 9ad3907a15..8877d2c765 100644 --- a/packages/paperclip-runner/src/semantic-tools/discovery.ts +++ b/packages/paperclip-runner/src/semantic-tools/discovery.ts @@ -21,6 +21,7 @@ const MAX_DISCOVERY_RESULTS = 10; const NAMESPACE: Readonly> = Object.freeze({ set_task_title: "active_task", + set_task_monitor: "continuation", search_api: "api_fallback", call_api: "api_fallback", get_task_context: "active_task", get_task_history: "active_task", diff --git a/packages/paperclip-runner/src/semantic-tools/paperclip-discovery.ts b/packages/paperclip-runner/src/semantic-tools/paperclip-discovery.ts index 1388697928..491ac22e73 100644 --- a/packages/paperclip-runner/src/semantic-tools/paperclip-discovery.ts +++ b/packages/paperclip-runner/src/semantic-tools/paperclip-discovery.ts @@ -13,6 +13,7 @@ import type { const NAMESPACE: Readonly> = Object.freeze({ set_task_title: "active_task", + set_task_monitor: "continuation", search_api: "api_fallback", call_api: "api_fallback", get_task_context: "active_task", diff --git a/packages/paperclip-runner/src/semantic-tools/semantic-tools.test.ts b/packages/paperclip-runner/src/semantic-tools/semantic-tools.test.ts index 85d6829cb0..7dbc66243f 100644 --- a/packages/paperclip-runner/src/semantic-tools/semantic-tools.test.ts +++ b/packages/paperclip-runner/src/semantic-tools/semantic-tools.test.ts @@ -113,7 +113,8 @@ describe("Capability semantic catalog and authorization", () => { it("publishes a stable narrow catalog without credentials or control-plane-owned tools", () => { const names = CAPABILITY_SEMANTIC_TOOL_CATALOG.map((tool) => tool.operationId); expect(new Set(names).size).toBe(names.length); - expect(names).toHaveLength(44); + expect(names).toHaveLength(45); + expect(names).toContain("set_task_monitor"); expect(names).toEqual(expect.arrayContaining(["submit_complaint", "submit_suggestion"])); expect(names).toContain("get_task_context"); expect(names).toContain("finish_task"); diff --git a/packages/paperclip-runner/src/semantic-tools/types.ts b/packages/paperclip-runner/src/semantic-tools/types.ts index 5c1d0a7e66..28a5b1988d 100644 --- a/packages/paperclip-runner/src/semantic-tools/types.ts +++ b/packages/paperclip-runner/src/semantic-tools/types.ts @@ -19,6 +19,7 @@ export type CapabilitySemanticOperationId = | "search_api" | "call_api" | "set_task_title" + | "set_task_monitor" | "get_task_context" | "get_task_history" | "list_documents" diff --git a/packages/shared/src/index.ts b/packages/shared/src/index.ts index 2d2f71d4d2..232f447caa 100644 --- a/packages/shared/src/index.ts +++ b/packages/shared/src/index.ts @@ -1989,6 +1989,7 @@ export { updateIssueSchema, stalledReviewDecisionSchema, issueExecutionPolicySchema, + issueExecutionMonitorPolicySchema, issueExecutionStateSchema, resolveIssueRecoveryActionSchema, retryWorkspaceExportSchema, diff --git a/packages/shared/src/validators/index.ts b/packages/shared/src/validators/index.ts index 0634cf2739..2b621a7f9c 100644 --- a/packages/shared/src/validators/index.ts +++ b/packages/shared/src/validators/index.ts @@ -435,6 +435,7 @@ export { updateIssueSchema, stalledReviewDecisionSchema, issueExecutionPolicySchema, + issueExecutionMonitorPolicySchema, issueExecutionStateSchema, issueRecoveryActionReadModelSchema, resolveIssueRecoveryActionSchema, diff --git a/server/scripts/smoke-native-task-monitor.ts b/server/scripts/smoke-native-task-monitor.ts new file mode 100644 index 0000000000..76d327bd93 --- /dev/null +++ b/server/scripts/smoke-native-task-monitor.ts @@ -0,0 +1,98 @@ +/** Opt-in live qualification: creates an isolated database/workspace and spends two Codex turns. + * PATH=::$PATH node --import tsx server/scripts/smoke-native-task-monitor.ts --run + */ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import { mkdtemp, mkdir, writeFile } from "node:fs/promises"; +import { createServer } from "node:http"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { setTimeout as delay } from "node:timers/promises"; +import { and, eq } from "drizzle-orm"; + +if (!process.argv.includes("--run")) throw new Error("Pass --run to authorize the live two-turn Codex smoke."); +const root = await mkdtemp(join(tmpdir(), "paperclip-live-task-monitor-")); +process.env.PAPERCLIP_HOME = join(root, "paperclip"); +process.env.PAPERCLIP_INSTANCE_ID = "monitor-smoke"; +process.env.PAPERCLIP_TELEMETRY_ENABLED = "false"; +const { createDb, companies, agents, authUsers, companyMemberships, issues, heartbeatRuns, agentWakeupRequests, nativeRunResults, statusDecisions } = await import("@paperclipai/db"); +const { startEmbeddedPostgresTestDatabase } = await import("../src/__tests__/helpers/embedded-postgres.js"); +const { heartbeatService } = await import("../src/services/heartbeat.js"); +const { setupRunnerPrpWebSocketServer, runnerPrpWebSocketInternals } = await import("../src/realtime/runner-prp-ws.js"); +const { closeIdleWarmNativeSessionsForRestart } = await import("../src/services/native-runtime/native-session-executor.js"); +const temporary = await startEmbeddedPostgresTestDatabase("live-task-monitor-"); +const db = createDb(temporary.connectionString); +const server = createServer(); +await new Promise(resolve => server.listen(0, "127.0.0.1", resolve)); +const address = server.address(); +if (!address || typeof address === "string") throw new Error("TCP listener missing"); +process.env.PAPERCLIP_API_URL = `http://127.0.0.1:${address.port}`; +setupRunnerPrpWebSocketServer(server, { apiUrl: process.env.PAPERCLIP_API_URL }); +const companyId = randomUUID(), agentId = randomUUID(), issueId = randomUUID(); +const heartbeat = heartbeatService(db); +const readRuns = () => db.select().from(heartbeatRuns).where(eq(heartbeatRuns.companyId, companyId)).orderBy(heartbeatRuns.createdAt); +console.log(JSON.stringify({ root, companyId, agentId, issueId })); +try { + const cwd = join(root, "workspace"); + await mkdir(cwd); + await db.insert(companies).values({ id: companyId, name: "Isolated monitor smoke", issuePrefix: "MON", defaultResponsibleUserId: "monitor-smoke", requireBoardApprovalForNewAgents: false }); + await db.insert(authUsers).values({ id: "monitor-smoke", name: "Monitor smoke", email: "monitor-smoke@example.test", createdAt: new Date(), updatedAt: new Date() }); + await db.insert(companyMemberships).values({ companyId, principalType: "user", principalId: "monitor-smoke", status: "active", membershipRole: "owner" }); + await db.insert(agents).values({ id: agentId, companyId, name: "Monitor verifier", status: "active", adapterType: "paperclip_runner", + adapterConfig: { provider: "codex", cwd, lifecycleMode: "warm", idleTimeoutMs: 300_000, timeoutSeconds: 180 }, + runtimeConfig: { heartbeat: { enabled: false, wakeOnDemand: true } } }); + await db.insert(issues).values({ id: issueId, companyId, title: "Verify a one-shot native task monitor", identifier: "MON-1", status: "in_progress", assigneeAgentId: agentId, + description: "This is an isolated two-run integration test. First run: use set_task_monitor on this task with a fresh nextCheckAt about 45 seconds in the future (compute the current UTC time if needed), notes saying to verify issue_monitor_due, and idempotencyKey monitor-live-first. Confirm the receipt. Then call paperclip_finish with yielded and continuation.kind monitor, accurately retaining the second run as blocking remaining work. End immediately after acceptance. Do not sleep or poll. Second run: if the wake reason is issue_monitor_due, the requested check succeeded. Report that reason and finish done using the current completion contract. Do not schedule another monitor or request human review. No files or other deliverables are required." }); + await heartbeat.wakeup(agentId, { source: "on_demand", triggerDetail: "manual", reason: "monitor_smoke", + requestedByActorType: "system", requestedByActorId: "monitor-smoke", payload: { issueId }, contextSnapshot: { issueId, wakeReason: "monitor_smoke" } }); + const deadline = Date.now() + 8 * 60_000; + let firstReleasedAt: number | null = null; + while (Date.now() < deadline) { + const runs = await readRuns(); + if (runs.some(run => ["failed", "timed_out", "cancelled"].includes(run.status))) { + throw new Error(`Live run failed: ${JSON.stringify(runs.map(run => ({ id: run.id, status: run.status, error: run.error, errorCode: run.errorCode })))}`); + } + if (runs.length === 1 && runs[0].status === "succeeded") { + firstReleasedAt ??= Date.now(); + const [issue] = await db.select().from(issues).where(eq(issues.id, issueId)); + assert.equal(issue.status, "in_progress"); + assert.ok(issue.monitorNextCheckAt, "first run must leave a persisted monitor"); + assert.equal(issue.executionRunId, null); + } + if (runs.length >= 2 && runs[1].status === "succeeded") break; + await heartbeat.tickTimers(); + await delay(1_000); + } + await heartbeat.drainActiveRunExecutions(); + const runs = await readRuns(); + assert.equal(runs.length, 2, "exactly two server runs must execute"); + assert.ok(runs.every(run => run.status === "succeeded")); + const [issue] = await db.select().from(issues).where(eq(issues.id, issueId)); + const wakes = await db.select().from(agentWakeupRequests).where(and(eq(agentWakeupRequests.companyId, companyId), eq(agentWakeupRequests.reason, "issue_monitor_due"))); + assert.equal(wakes.length, 1); + assert.equal(runs[1].contextSnapshot?.wakeReason, "issue_monitor_due"); + assert.equal(issue.status, "done"); + assert.equal(issue.monitorNextCheckAt, null); + assert.ok(firstReleasedAt && runs[1].createdAt.getTime() - firstReleasedAt < 300_000, "wake must arrive within the configured warm window"); + const results = await db.select().from(nativeRunResults).where(eq(nativeRunResults.companyId, companyId)); + const decisions = await db.select().from(statusDecisions).where(eq(statusDecisions.companyId, companyId)); + const evidence = { companyId, issueId, agentId, configuredIdleTimeoutMs: 300_000, firstReleasedAt, status: issue.status, + runs: runs.map(run => ({ id: run.id, status: run.status, createdAt: run.createdAt, finishedAt: run.finishedAt, + wakeReason: run.contextSnapshot?.wakeReason, nativeSessionId: run.nativeSessionId, runnerInstanceId: run.runnerInstanceId, + processPid: run.processPid, processStartedAt: run.processStartedAt, providerSessionId: run.sessionIdAfter, + semanticToolReceipts: run.resultJson?.semanticToolReceipts })), + warmReuse: { sameNativeSession: runs[0].nativeSessionId === runs[1].nativeSessionId, + sameRunnerInstance: runs[0].runnerInstanceId === runs[1].runnerInstanceId, + sameProcess: runs[0].processPid != null && runs[0].processPid === runs[1].processPid && runs[0].processStartedAt?.getTime() === runs[1].processStartedAt?.getTime() }, + results: results.map(row => ({ runId: row.runId, result: row.resultJson })), + decisions: decisions.map(row => ({ runId: row.runId, reasonCode: row.reasonCode, toStatus: row.toStatus })), + monitorWakeIds: wakes.map(row => row.id) }; + await writeFile(join(root, "evidence.json"), JSON.stringify(evidence, null, 2)); + console.log(JSON.stringify({ passed: true, evidence: join(root, "evidence.json"), warmReuse: evidence.warmReuse })); +} finally { + await writeFile(join(root, "run-records.json"), JSON.stringify(await readRuns(), null, 2)); + await closeIdleWarmNativeSessionsForRestart(); + runnerPrpWebSocketInternals.resetForTests(); + await new Promise(resolve => server.close(() => resolve())); + await temporary.cleanup(); +} diff --git a/server/src/__tests__/issue-monitor-scheduler.test.ts b/server/src/__tests__/issue-monitor-scheduler.test.ts index 13389f7303..48a88c8425 100644 --- a/server/src/__tests__/issue-monitor-scheduler.test.ts +++ b/server/src/__tests__/issue-monitor-scheduler.test.ts @@ -253,6 +253,9 @@ describeEmbeddedPostgres("issue monitor scheduler", () => { const { issueId, agentId } = await seedFixture(); const heartbeat = heartbeatService(db); const tickAt = new Date("2026-04-11T12:31:00.000Z"); + const [before] = await db.select().from(issues).where(eq(issues.id, issueId)); + const unrelatedPolicy = { commentRequired: false, futureAuthorization: { preserve: true } }; + await db.update(issues).set({ executionPolicy: { ...before.executionPolicy, ...unrelatedPolicy } }).where(eq(issues.id, issueId)); const result = await heartbeat.tickTimers(tickAt); @@ -260,6 +263,7 @@ describeEmbeddedPostgres("issue monitor scheduler", () => { const issue = await db.select().from(issues).where(eq(issues.id, issueId)).then((rows) => rows[0]!); expect(issue.monitorNextCheckAt).toBeNull(); + expect(issue.executionPolicy).toMatchObject(unrelatedPolicy); expect(issue.monitorAttemptCount).toBe(1); expect(issue.monitorLastTriggeredAt?.toISOString()).toBe(tickAt.toISOString()); expect(normalizeIssueExecutionPolicy(issue.executionPolicy ?? null)?.monitor ?? null).toBeNull(); @@ -284,6 +288,89 @@ describeEmbeddedPostgres("issue monitor scheduler", () => { expect(activity).toContain("issue.monitor_triggered"); }); + it("preserves a due monitor through native execution and dispatches once after release", async () => { + const { companyId, issueId, agentId, nextCheckAt } = await seedFixture(); + const runId = randomUUID(); + await db.insert(heartbeatRuns).values({ id: runId, companyId, agentId, nativeIssueId: issueId, + runtimeMode: "native", status: "running", contextSnapshot: { issueId } }); + await db.update(issues).set({ executionRunId: runId }).where(eq(issues.id, issueId)); + const heartbeat = heartbeatService(db); + expect((await heartbeat.tickTimers(new Date("2026-04-11T12:31:00.000Z"))).enqueued).toBe(0); + const [waiting] = await db.select().from(issues).where(eq(issues.id, issueId)); + expect(waiting.monitorNextCheckAt).toEqual(nextCheckAt); + expect(waiting.monitorWakeRequestedAt).toBeNull(); + expect(await db.select().from(agentWakeupRequests)).toHaveLength(0); + await db.update(heartbeatRuns).set({ status: "succeeded", finishedAt: new Date() }).where(eq(heartbeatRuns.id, runId)); + await db.update(issues).set({ executionRunId: null }).where(eq(issues.id, issueId)); + expect((await heartbeat.tickTimers(new Date("2026-04-11T12:32:00.000Z"))).enqueued).toBe(1); + expect((await heartbeat.tickTimers(new Date("2026-04-11T12:33:00.000Z"))).enqueued).toBe(0); + const [triggered] = await db.select().from(issues).where(eq(issues.id, issueId)); + expect(triggered.monitorNextCheckAt).toBeNull(); + expect(triggered.monitorAttemptCount).toBe(1); + expect((await db.select().from(agentWakeupRequests)).filter(wake => wake.reason === "issue_monitor_due")).toHaveLength(1); + }); + + it.each(["replaced", "cleared", "reassigned", "completed"] as const)("fences a monitor %s after claim but before wake admission", async (change) => { + const { companyId, issueId, agentId } = await seedFixture(); + const replacementAt = new Date("2026-04-11T13:30:00.000Z"); + let raced = false; + const racingDb = new Proxy(db, { + get(target, property, receiver) { + if (property !== "transaction") return Reflect.get(target, property, receiver); + return async (callback: Parameters[0]) => { + const result = await db.transaction(callback); + const claimed = result as { id?: string; monitorWakeRequestedAt?: Date } | undefined; + if (!raced && claimed?.id === issueId && claimed.monitorWakeRequestedAt) { + raced = true; + await db.update(issues).set(change === "replaced" ? { + monitorNextCheckAt: replacementAt, monitorWakeRequestedAt: null, + executionPolicy: { monitor: { nextCheckAt: replacementAt.toISOString(), notes: "Replacement check", scheduledBy: "assignee" } }, + } : change === "cleared" ? { monitorNextCheckAt: null, monitorWakeRequestedAt: null, executionPolicy: null } + : change === "reassigned" ? { assigneeAgentId: null } : { status: "done" }) + .where(eq(issues.id, issueId)); + } + return result; + }; + }, + }); + expect((await heartbeatService(racingDb).tickTimers(new Date("2026-04-11T12:31:00.000Z"))).enqueued).toBe(0); + expect(raced).toBe(true); + expect((await db.select().from(agentWakeupRequests).where(eq(agentWakeupRequests.companyId, companyId))) + .filter(wake => wake.reason === "issue_monitor_due")).toHaveLength(0); + expect(await db.select().from(heartbeatRuns).where(eq(heartbeatRuns.agentId, agentId))).toHaveLength(0); + const [issue] = await db.select().from(issues).where(eq(issues.id, issueId)); + if (change === "replaced") expect(issue.monitorNextCheckAt).toEqual(replacementAt); + if (change === "cleared") expect(issue.monitorNextCheckAt).toBeNull(); + }); + + it("does not erase a replacement scheduled after wake admission", async () => { + const { companyId, issueId } = await seedFixture(); + const nextCheckAt = new Date("2026-04-11T13:30:00.000Z"); + let replaced = false; + const racingDb = new Proxy(db, { + get(target, property, receiver) { + if (property !== "transaction") return Reflect.get(target, property, receiver); + return async (callback: Parameters[0]) => { + const result = await db.transaction(callback); + if (!replaced && (result as { kind?: string } | undefined)?.kind === "queued") { + replaced = true; + await db.update(issues).set({ monitorNextCheckAt: nextCheckAt, monitorWakeRequestedAt: null, + executionPolicy: { monitor: { nextCheckAt: nextCheckAt.toISOString(), notes: "New check", scheduledBy: "assignee" } }, + }).where(eq(issues.id, issueId)); + } + return result; + }; + }, + }); + await heartbeatService(racingDb).tickTimers(new Date("2026-04-11T12:31:00.000Z")); + expect(replaced).toBe(true); + const [issue] = await db.select().from(issues).where(eq(issues.id, issueId)); + expect(issue.monitorNextCheckAt).toEqual(nextCheckAt); + expect(issue.monitorWakeRequestedAt).toBeNull(); + expect((await db.select().from(agentWakeupRequests).where(eq(agentWakeupRequests.companyId, companyId))) + .filter(wake => wake.reason === "issue_monitor_due")).toHaveLength(1); + }); + it.each(["unknown", "exhausted"] as const)("does not replay a quota monitor with %s execution evidence", async (kind) => { const sourceRunId = randomUUID(); const { companyId, issueId, agentId } = await seedFixture({ diff --git a/server/src/__tests__/native-status-arbiter-corpus.test.ts b/server/src/__tests__/native-status-arbiter-corpus.test.ts index 41fc05cae7..246b4e19f9 100644 --- a/server/src/__tests__/native-status-arbiter-corpus.test.ts +++ b/server/src/__tests__/native-status-arbiter-corpus.test.ts @@ -2907,6 +2907,39 @@ describe("P6-31 Section 18.13 executable status-authority corpus", () => { expect(await db.select().from(agentWakeupRequests).where(eq(agentWakeupRequests.idempotencyKey, key))).toHaveLength(1); }); + it.each(["scheduled", "cleared", "replaced", "reassigned"] as const)("revalidates a %s monitor under the final disposition lock", async (state) => { + const template = corpus.fixtures.find(candidate => candidate.mode === "native")!; + const fixture = { ...template, id: `monitor-commit-${state}`, given: { ...template.given, priorIssueStatus: "in_progress" } }; + const seeded = await seedFixture(fixture); + const nextCheckAt = new Date(Date.now() + 60_000).toISOString(); + const currentCheck = state === "replaced" ? new Date(Date.now() + 120_000).toISOString() : nextCheckAt; + await db.update(issues).set({ + executionRunId: seeded.runId, checkoutRunId: seeded.runId, + assigneeAgentId: state === "reassigned" ? null : agentId, + monitorNextCheckAt: state === "cleared" ? null : new Date(currentCheck), + executionPolicy: state === "cleared" ? null : { monitor: { nextCheckAt: currentCheck, notes: "Check again" } }, + }).where(eq(issues.id, seeded.issueId)); + const decision: NativeStatusDecision = { policyVersion: NATIVE_STATUS_ARBITER_POLICY_VERSION, + statusAction: "preserve", toStatus: "in_progress", reasonCode: "scheduled_monitor_waiting", + unblockDescriptor: null, effects: [{ kind: "release_checkout" }] }; + const commit = () => commitNativeStatusDecision({ db, companyId, issueId: seeded.issueId, runId: seeded.runId, + assessmentId: seeded.assessmentId, priorStatus: "in_progress", priorStatusVersion: 0, priorDecisionId: null, + decision, requireMonitorWait: { agentId, nextCheckAt } }); + if (state !== "scheduled") { + await expect(commit()).rejects.toThrow("native_status_race"); + expect(await db.select().from(statusDecisions).where(eq(statusDecisions.issueId, seeded.issueId))).toHaveLength(0); + } else { + await commit(); + expect((await commit()).replayed).toBe(true); + const [issue] = await db.select().from(issues).where(eq(issues.id, seeded.issueId)); + expect(issue).toMatchObject({ status: "in_progress", checkoutRunId: null, executionRunId: null }); + expect(issue.monitorNextCheckAt?.toISOString()).toBe(nextCheckAt); + const effects = await db.select().from(statusDecisionEffects).where(eq(statusDecisionEffects.issueId, seeded.issueId)); + expect(effects.map(effect => effect.effectKind)).toEqual(["release_checkout"]); + expect((await db.select().from(agentWakeupRequests).where(eq(agentWakeupRequests.agentId, agentId))).filter(wake => wake.payload?.issueId === seeded.issueId)).toHaveLength(0); + } + }); + it("fails the transaction closed for an unknown status effect", async () => { const fixture = corpus.fixtures.find((candidate) => candidate.mode === "native"); if (!fixture) throw new Error("native corpus fixture missing"); diff --git a/server/src/routes/issues.ts b/server/src/routes/issues.ts index dbd430dbc8..bf3fcda031 100644 --- a/server/src/routes/issues.ts +++ b/server/src/routes/issues.ts @@ -1,3 +1,4 @@ +import { monitorPoliciesEqual, applyActorMonitorScheduledBy, assertCanManageIssueMonitor, summarizeIssueMonitor } from "../services/issue-monitors.js"; import type { IssuePrivacyConstraints } from "@paperclipai/shared"; import { canActorReadHeartbeatRun } from "../services/heartbeat-run-privacy.js"; import { activeIssueInteractionCondition, readTaskQuestionContext } from "../services/issue-question-context.js"; @@ -2170,106 +2171,6 @@ function summarizeIssueReferenceActivityDetails( }; } -function monitorPoliciesEqual( - left: NormalizedExecutionPolicy | null, - right: NormalizedExecutionPolicy | null, -) { - return ( - JSON.stringify(left?.monitor ?? null) === - JSON.stringify(right?.monitor ?? null) - ); -} - -function applyActorMonitorScheduledBy( - policy: NormalizedExecutionPolicy | null, - actorType: "agent" | "user", -) { - return setIssueExecutionPolicyMonitorScheduledBy( - policy, - actorType === "user" ? "board" : "assignee", - ); -} - -async function assertCanManageIssueMonitor( - accessSvc: ReturnType, - req: Request, - companyId: string, - assigneeAgentId: string | null, - monitorChanged: boolean, -) { - if (!monitorChanged) return; - if (req.actor.type === "board") return; - const runtimeDecision = await accessSvc.decide({ - actor: req.actor, - action: "runtime:manage", - resource: { type: "company", companyId }, - }); - if (!runtimeDecision.allowed) { - throw forbidden( - runtimeDecision.explanation, - authorizationDeniedDetails(runtimeDecision), - ); - } - if ( - req.actor.type === "agent" && - req.actor.agentId && - req.actor.agentId === assigneeAgentId - ) - return; - throw forbidden( - "Only the assignee agent or a board user can manage issue monitors", - ); -} - -function summarizeIssueMonitor( - issue: { - monitorNextCheckAt?: Date | null; - monitorLastTriggeredAt?: Date | null; - monitorAttemptCount?: number | null; - monitorNotes?: string | null; - monitorScheduledBy?: string | null; - executionState?: unknown; - }, - policy: NormalizedExecutionPolicy | null, -) { - const state = parseIssueExecutionState(issue.executionState); - return { - nextCheckAt: - issue.monitorNextCheckAt?.toISOString() ?? - policy?.monitor?.nextCheckAt ?? - null, - lastTriggeredAt: - issue.monitorLastTriggeredAt?.toISOString() ?? - state?.monitor?.lastTriggeredAt ?? - null, - attemptCount: - issue.monitorAttemptCount ?? state?.monitor?.attemptCount ?? 0, - notes: - policy?.monitor?.notes ?? - issue.monitorNotes ?? - state?.monitor?.notes ?? - null, - scheduledBy: - issue.monitorScheduledBy ?? - policy?.monitor?.scheduledBy ?? - state?.monitor?.scheduledBy ?? - null, - kind: policy?.monitor?.kind ?? state?.monitor?.kind ?? null, - serviceName: - policy?.monitor?.serviceName ?? state?.monitor?.serviceName ?? null, - externalRef: redactIssueMonitorExternalRef( - policy?.monitor?.externalRef ?? state?.monitor?.externalRef ?? null, - ), - timeoutAt: policy?.monitor?.timeoutAt ?? state?.monitor?.timeoutAt ?? null, - maxAttempts: - policy?.monitor?.maxAttempts ?? state?.monitor?.maxAttempts ?? null, - recoveryPolicy: - policy?.monitor?.recoveryPolicy ?? state?.monitor?.recoveryPolicy ?? null, - status: state?.monitor?.status ?? (policy?.monitor ? "scheduled" : null), - clearReason: state?.monitor?.clearReason ?? null, - }; -} - function activityExecutionParticipantKey( participant: ActivityExecutionParticipant, ): string { diff --git a/server/src/services/heartbeat.ts b/server/src/services/heartbeat.ts index 9f9b697fe9..f406b9de39 100644 --- a/server/src/services/heartbeat.ts +++ b/server/src/services/heartbeat.ts @@ -3786,6 +3786,8 @@ interface WakeupOptions { statuses: string[]; assigneeAgentId: string; statusVersion?: number; + monitorNextCheckAt?: string; + monitorWakeRequestedAt?: string; }; /** Keep causally distinct external chat continuations out of an existing run. */ allowRunCoalescing?: boolean; @@ -11661,18 +11663,19 @@ export function heartbeatService( runId: string | null; activitySource: "manual" | "scheduled"; }) { - await db + const cleared = await db .update(issues) .set({ - ...buildIssueMonitorClearedPatch({ + ...monitorOnlyDispatchPatch(buildIssueMonitorClearedPatch({ issue: input.claimed, policy: input.policy, clearReason: input.clearReason, clearedAt: input.now, - }), + })), updatedAt: input.now, }) - .where(eq(issues.id, input.claimed.id)); + .where(issueMonitorClaimCondition(input.claimed)).returning({ id: issues.id }); + if (cleared.length === 0) return { outcome: "skipped" as const, reason: "monitor_replaced" }; await logActivity(db, { companyId: input.claimed.companyId, @@ -11711,6 +11714,24 @@ export function heartbeatService( return { outcome: "skipped" as const, reason: input.clearReason }; } + function monitorOnlyDispatchPatch>(patch: T) { + // Admission and consumption are separate transactions. Preserve any review + // policy/state changes made between them; only the monitor belongs to us. + return { + ...patch, + executionPolicy: sql`nullif(${issues.executionPolicy} - 'monitor', '{}'::jsonb)`, + executionState: sql`jsonb_set(coalesce(${issues.executionState}, ${JSON.stringify(patch.executionState)}::jsonb), + '{monitor}', ${JSON.stringify(patch.executionState?.monitor ?? null)}::jsonb)`, + }; + } + + function issueMonitorClaimCondition(claimed: IssueMonitorDispatchRow) { + return and(eq(issues.id, claimed.id), eq(issues.companyId, claimed.companyId), + eq(issues.assigneeAgentId, claimed.assigneeAgentId!), isNull(issues.assigneeUserId), + eq(issues.status, claimed.status), eq(issues.monitorNextCheckAt, claimed.monitorNextCheckAt!), + eq(issues.monitorWakeRequestedAt, claimed.monitorWakeRequestedAt!)); + } + async function dispatchClaimedIssueMonitor( claimed: IssueMonitorDispatchRow, input: { @@ -11855,8 +11876,10 @@ export function heartbeatService( if (scheduled.outcome === "not_scheduled") throw conflict(scheduled.reason); } - } else - await enqueueWakeup(targetAgentId, { + } else { + const wake = await enqueueWakeup(targetAgentId, { + issueStateGuard: { statuses: [claimed.status], assigneeAgentId: claimed.assigneeAgentId, + monitorNextCheckAt: scheduledAtIso, monitorWakeRequestedAt: claimed.monitorWakeRequestedAt!.toISOString() }, source: input.source, triggerDetail: input.triggerDetail, reason: wakeReason, @@ -11886,18 +11909,26 @@ export function heartbeatService( manualTrigger: input.activitySource === "manual", }, }); + if (!wake) { + await db.update(issues).set({ monitorWakeRequestedAt: null, updatedAt: input.now }) + .where(issueMonitorClaimCondition(claimed)); + return { outcome: "skipped" as const, reason: "monitor_dispatch_deferred" }; + } + } - await db + const consumed = await db .update(issues) .set({ - ...buildIssueMonitorTriggeredPatch({ + ...monitorOnlyDispatchPatch(buildIssueMonitorTriggeredPatch({ issue: claimed, policy, triggeredAt: input.now, - }), + })), updatedAt: new Date(), }) - .where(eq(issues.id, claimed.id)); + .where(issueMonitorClaimCondition(claimed)) + .returning({ id: issues.id }); + if (consumed.length === 0) return { outcome: "skipped" as const, reason: "monitor_replaced" }; await logActivity(db, { companyId: claimed.companyId, @@ -11926,15 +11957,15 @@ export function heartbeatService( await db .update(issues) .set({ - ...buildIssueMonitorClearedPatch({ + ...monitorOnlyDispatchPatch(buildIssueMonitorClearedPatch({ issue: claimed, policy, clearReason: "dispatch_skipped", clearedAt: input.now, - }), + })), updatedAt: new Date(), }) - .where(eq(issues.id, claimed.id)); + .where(issueMonitorClaimCondition(claimed)); await logActivity(db, { companyId: claimed.companyId, @@ -11964,7 +11995,7 @@ export function heartbeatService( monitorWakeRequestedAt: null, updatedAt: new Date(), }) - .where(eq(issues.id, claimed.id)); + .where(issueMonitorClaimCondition(claimed)); } else { await db .update(issues) @@ -11972,13 +12003,21 @@ export function heartbeatService( monitorWakeRequestedAt: null, updatedAt: new Date(), }) - .where(eq(issues.id, claimed.id)); + .where(issueMonitorClaimCondition(claimed)); } throw err; } } + function noActiveNativeMonitorRun() { + return sql`not exists (select 1 from ${heartbeatRuns} monitor_run + where monitor_run.company_id = ${issues.companyId} + and monitor_run.native_issue_id = ${issues.id} + and monitor_run.runtime_mode = 'native' + and monitor_run.status in ('queued', 'running', 'scheduled_retry'))`; + } + async function triggerIssueMonitor( issueId: string, input?: { @@ -12071,6 +12110,7 @@ export function heartbeatService( .where( and( eq(companies.status, "active"), + noActiveNativeMonitorRun(), sql`${issues.monitorNextCheckAt} is not null`, lte(issues.monitorNextCheckAt, now), isNull(issues.assigneeUserId), @@ -12099,6 +12139,7 @@ export function heartbeatService( .where( and( eq(issues.id, due.id), + noActiveNativeMonitorRun(), sql`${issues.monitorNextCheckAt} is not null`, lte(issues.monitorNextCheckAt, now), isNull(issues.assigneeUserId), @@ -27839,6 +27880,9 @@ export function heartbeatService( executionWorkspacePreference: issues.executionWorkspacePreference, executionWorkspaceSettings: issues.executionWorkspaceSettings, assigneeAgentId: issues.assigneeAgentId, + assigneeUserId: issues.assigneeUserId, + monitorNextCheckAt: issues.monitorNextCheckAt, + monitorWakeRequestedAt: issues.monitorWakeRequestedAt, executionRunId: issues.executionRunId, executionAgentNameKey: issues.executionAgentNameKey, createdAt: issues.createdAt, @@ -27870,6 +27914,54 @@ export function heartbeatService( return { kind: "skipped" as const }; } + const issueStateGuard = opts.issueStateGuard; + const activeMonitorRun = issueStateGuard?.monitorNextCheckAt === undefined ? null + : await tx.select({ id: heartbeatRuns.id }).from(heartbeatRuns).where(and( + eq(heartbeatRuns.companyId, issue.companyId), eq(heartbeatRuns.nativeIssueId, issue.id), + eq(heartbeatRuns.runtimeMode, "native"), inArray(heartbeatRuns.status, ["queued", "running", "scheduled_retry"]), + )).limit(1).then(rows => rows[0] ?? null); + if ( + issueStateGuard && + (!issueStateGuard.statuses.includes(issue.status) || + issue.assigneeAgentId !== issueStateGuard.assigneeAgentId || + (issueStateGuard.statusVersion !== undefined && issue.statusVersion !== issueStateGuard.statusVersion) || + (issueStateGuard.monitorNextCheckAt !== undefined && ( + activeMonitorRun !== null || issue.assigneeUserId !== null || + issue.monitorNextCheckAt?.toISOString() !== issueStateGuard.monitorNextCheckAt || + issue.monitorWakeRequestedAt?.toISOString() !== issueStateGuard.monitorWakeRequestedAt + ))) + ) { + // A deferred monitor retains its schedule; do not create a receipt + // that could suppress its next admission attempt. + if (issueStateGuard.monitorNextCheckAt !== undefined) return { kind: "skipped" as const }; + await tx.insert(agentWakeupRequests).values({ + ...durableReceiptFields, + companyId: agent.companyId, + agentId, + source, + triggerDetail, + reason: "issue_state_guard_mismatch", + payload: { + ...(payload ?? {}), + heartbeatSkip: { + reason: + "Issue status or assignee changed before the wake could be queued.", + issueId: issue.id, + expectedStatuses: issueStateGuard.statuses, + actualStatus: issue.status, + expectedAssigneeAgentId: issueStateGuard.assigneeAgentId, + actualAssigneeAgentId: issue.assigneeAgentId, + }, + }, + status: "skipped", + requestedByActorType: opts.requestedByActorType ?? null, + requestedByActorId: opts.requestedByActorId ?? null, + idempotencyKey: opts.idempotencyKey ?? null, + finishedAt: new Date(), + }); + return { kind: "skipped" as const }; + } + if (opts.failedRunId) { // The issue lock makes double-clicks and network retries adopt the // same successor, including after it has already finished. @@ -28068,40 +28160,6 @@ export function heartbeatService( onBlocked: (reason, message) => { continuationWait = { reason, message }; }, }))) return deferBlockedExecution(executionBlocker); - const issueStateGuard = opts.issueStateGuard; - if ( - issueStateGuard && - (!issueStateGuard.statuses.includes(issue.status) || - issue.assigneeAgentId !== issueStateGuard.assigneeAgentId || - (issueStateGuard.statusVersion !== undefined && issue.statusVersion !== issueStateGuard.statusVersion)) - ) { - await tx.insert(agentWakeupRequests).values({ - ...durableReceiptFields, - companyId: agent.companyId, - agentId, - source, - triggerDetail, - reason: "issue_state_guard_mismatch", - payload: { - ...(payload ?? {}), - heartbeatSkip: { - reason: - "Issue status or assignee changed before the wake could be queued.", - issueId: issue.id, - expectedStatuses: issueStateGuard.statuses, - actualStatus: issue.status, - expectedAssigneeAgentId: issueStateGuard.assigneeAgentId, - actualAssigneeAgentId: issue.assigneeAgentId, - }, - }, - status: "skipped", - requestedByActorType: opts.requestedByActorType ?? null, - requestedByActorId: opts.requestedByActorId ?? null, - idempotencyKey: opts.idempotencyKey ?? null, - finishedAt: new Date(), - }); - return { kind: "skipped" as const }; - } if ( worktreeExecutionCutoff && diff --git a/server/src/services/issue-monitors.ts b/server/src/services/issue-monitors.ts new file mode 100644 index 0000000000..eede91c518 --- /dev/null +++ b/server/src/services/issue-monitors.ts @@ -0,0 +1,166 @@ +import { issueExecutionMonitorPolicySchema, PROVIDER_QUOTA_MONITOR_SERVICE_NAME } from "@paperclipai/shared"; +import type { issues } from "@paperclipai/db"; +import { z } from "zod"; +import { forbidden, unprocessable } from "../errors.js"; +import type { accessService } from "./access.js"; +import { authorizationDeniedDetails, type AuthorizationActor } from "./authorization.js"; +import { applyIssueMonitorPolicyTransition, normalizeIssueExecutionPolicy, parseIssueExecutionState, redactIssueMonitorExternalRef, setIssueExecutionPolicyMonitorScheduledBy } from "./issue-execution-policy.js"; + +type NormalizedExecutionPolicy = ReturnType; + +export function monitorPoliciesEqual( + left: NormalizedExecutionPolicy | null, + right: NormalizedExecutionPolicy | null, +) { + return ( + JSON.stringify(left?.monitor ?? null) === + JSON.stringify(right?.monitor ?? null) + ); +} + +export function applyActorMonitorScheduledBy( + policy: NormalizedExecutionPolicy | null, + actorType: "agent" | "user", +) { + return setIssueExecutionPolicyMonitorScheduledBy( + policy, + actorType === "user" ? "board" : "assignee", + ); +} + +export async function assertCanManageIssueMonitor( + accessSvc: Pick, "decide">, + req: { actor: AuthorizationActor }, + companyId: string, + assigneeAgentId: string | null, + monitorChanged: boolean, +) { + if (!monitorChanged) return; + if (req.actor.type === "board") return; + const runtimeDecision = await accessSvc.decide({ + actor: req.actor, + action: "runtime:manage", + resource: { type: "company", companyId }, + }); + if (!runtimeDecision.allowed) { + throw forbidden( + runtimeDecision.explanation, + authorizationDeniedDetails(runtimeDecision), + ); + } + if ( + req.actor.type === "agent" && + req.actor.agentId && + req.actor.agentId === assigneeAgentId + ) + return; + throw forbidden( + "Only the assignee agent or a board user can manage issue monitors", + ); +} + +export function summarizeIssueMonitor( + issue: { + monitorNextCheckAt?: Date | null; + monitorLastTriggeredAt?: Date | null; + monitorAttemptCount?: number | null; + monitorNotes?: string | null; + monitorScheduledBy?: string | null; + executionState?: unknown; + }, + policy: NormalizedExecutionPolicy | null, +) { + const state = parseIssueExecutionState(issue.executionState); + return { + nextCheckAt: + issue.monitorNextCheckAt?.toISOString() ?? + policy?.monitor?.nextCheckAt ?? + null, + lastTriggeredAt: + issue.monitorLastTriggeredAt?.toISOString() ?? + state?.monitor?.lastTriggeredAt ?? + null, + attemptCount: + issue.monitorAttemptCount ?? state?.monitor?.attemptCount ?? 0, + notes: + policy?.monitor?.notes ?? + issue.monitorNotes ?? + state?.monitor?.notes ?? + null, + scheduledBy: + issue.monitorScheduledBy ?? + policy?.monitor?.scheduledBy ?? + state?.monitor?.scheduledBy ?? + null, + kind: policy?.monitor?.kind ?? state?.monitor?.kind ?? null, + serviceName: + policy?.monitor?.serviceName ?? state?.monitor?.serviceName ?? null, + externalRef: redactIssueMonitorExternalRef( + policy?.monitor?.externalRef ?? state?.monitor?.externalRef ?? null, + ), + timeoutAt: policy?.monitor?.timeoutAt ?? state?.monitor?.timeoutAt ?? null, + maxAttempts: + policy?.monitor?.maxAttempts ?? state?.monitor?.maxAttempts ?? null, + recoveryPolicy: + policy?.monitor?.recoveryPolicy ?? state?.monitor?.recoveryPolicy ?? null, + status: state?.monitor?.status ?? (policy?.monitor ? "scheduled" : null), + clearReason: state?.monitor?.clearReason ?? null, + }; +} + +export const setTaskMonitorSchema = z.object({ + taskId: z.string().uuid().optional(), + idempotencyKey: z.string().trim().min(1).max(240), + monitor: issueExecutionMonitorPolicySchema.omit({ scheduledBy: true }).extend({ + notes: z.string().trim().min(1).max(500), + }).strict().refine(monitor => monitor.serviceName !== PROVIDER_QUOTA_MONITOR_SERVICE_NAME, { + message: "This serviceName is reserved for server-owned quota recovery; use a different service name for an ordinary task check", + path: ["serviceName"], + }).nullable(), +}).strict(); + +/** Caller holds the issue lock; merge only monitor state, never review policy. */ +export function prepareIssueMonitorUpdate( + issue: typeof issues.$inferSelect, + monitor: z.infer["monitor"], + actor: AuthorizationActor, + now = new Date(), +) { + if (monitor && Date.parse(monitor.nextCheckAt) <= now.getTime()) { + throw unprocessable("Monitor nextCheckAt must be in the future"); + } + if (monitor?.timeoutAt && Date.parse(monitor.timeoutAt) <= Date.parse(monitor.nextCheckAt)) { + throw unprocessable("Monitor timeoutAt must be later than nextCheckAt"); + } + const previousPolicy = normalizeIssueExecutionPolicy(issue.executionPolicy); + const policy = applyActorMonitorScheduledBy(normalizeIssueExecutionPolicy({ + ...previousPolicy, monitor, + }), actor.type === "board" ? "user" : "agent"); + const transition = applyIssueMonitorPolicyTransition({ + issue, policy, previousPolicy, requestedAssigneePatch: {}, + actor: { agentId: actor.agentId ?? null, userId: actor.userId ?? null }, + monitorExplicitlyUpdated: true, + }); + // Normalization supplies monitor defaults, but must not rewrite unrelated + // review/authorization settings (including forward-compatible policy fields). + const storedPolicy = { ...(issue.executionPolicy ?? {}) }; + if (policy?.monitor) storedPolicy.monitor = policy.monitor; + else delete storedPolicy.monitor; + return { ...transition.patch, executionPolicy: Object.keys(storedPolicy).length ? storedPolicy : null } as Partial; +} + +/** Timestamp presence alone is not authority to park a native task. */ +export function eligibleIssueMonitorWait( + issue: typeof issues.$inferSelect, + agentId: string, + now = new Date(), +): string | null { + if (issue.assigneeAgentId !== agentId || issue.assigneeUserId || + !["in_progress", "in_review"].includes(issue.status) || !issue.monitorNextCheckAt) return null; + const monitor = normalizeIssueExecutionPolicy(issue.executionPolicy)?.monitor; + if (!monitor || monitor.serviceName === PROVIDER_QUOTA_MONITOR_SERVICE_NAME || + Date.parse(monitor.nextCheckAt) !== issue.monitorNextCheckAt.getTime() || + (monitor.timeoutAt && Date.parse(monitor.timeoutAt) <= now.getTime()) || + (monitor.maxAttempts != null && issue.monitorAttemptCount >= monitor.maxAttempts)) return null; + return issue.monitorNextCheckAt.toISOString(); +} diff --git a/server/src/services/native-runtime/native-completion-feedback.ts b/server/src/services/native-runtime/native-completion-feedback.ts index 92df948c82..d814ca3a23 100644 --- a/server/src/services/native-runtime/native-completion-feedback.ts +++ b/server/src/services/native-runtime/native-completion-feedback.ts @@ -1,3 +1,4 @@ +import { eligibleIssueMonitorWait } from "../issue-monitors.js"; import { activeIssueInteractionCondition, ordinaryQuestionCondition } from "../issue-question-context.js"; import { publishedTaskDocuments, validateNativeDeliverableEvidence } from "./native-deliverable-feedback.js"; import { findAutomaticCompletionReviews } from "./automatic-completion-reviews.js"; @@ -64,6 +65,7 @@ export async function nativeCompletionFeedback( if (!issue) throw new Error("Completion task no longer exists."); const reviewContext = readNativeReviewAssignmentContext(run.contextSnapshot); if (reviewContext) { + if (result.continuation?.kind === "monitor") throw new Error("Review-only runs cannot yield to a task monitor; resolve the assigned review or report its blocker."); const review = await getNativeReviewAssignment(db, { companyId: run.companyId, issueId: issue.id, agentId: run.agentId, contextSnapshot: reviewContext, allowResolvedByRunId: run.id, @@ -96,6 +98,12 @@ export async function nativeCompletionFeedback( } } + if (result.reportedWorkDisposition === "yielded" && result.continuation?.kind === "monitor") { + if (issue.workMode !== "standard" || issue.executionRunId !== run.id || + !eligibleIssueMonitorWait(issue, run.agentId)) { + throw new Error("A monitor wait requires a persisted, eligible monitor on this task. Call set_task_monitor for the current task, confirm its schedule, then report yielded with continuation.kind monitor."); + } + } const signals = normalizePrpResultSignals(result); if ( result.reportedWorkDisposition === "done" && @@ -215,6 +223,9 @@ export async function nativeCompletionFeedback( if (readiness.unresolvedBlockerCount > 0) { return `Completion report accepted; this task still has unresolved dependencies. Explain the blockers on [this task](/issues/${issue.identifier ?? issue.id}); do not say the task is done.`; } + if (result.reportedWorkDisposition === "yielded" && result.continuation?.kind === "monitor") { + return `Monitor wait accepted. The task remains active and Paperclip will wake its assignee at or after ${issue.monitorNextCheckAt!.toISOString()} with issue_monitor_due. End this turn; do not poll or mark the task done.`; + } if ( !isConversation(issue) && result.reportedWorkDisposition === "yielded" && diff --git a/server/src/services/native-runtime/native-run-finalizer.ts b/server/src/services/native-runtime/native-run-finalizer.ts index 6cc62d264e..d986baf2c7 100644 --- a/server/src/services/native-runtime/native-run-finalizer.ts +++ b/server/src/services/native-runtime/native-run-finalizer.ts @@ -1,3 +1,4 @@ +import { eligibleIssueMonitorWait } from "../issue-monitors.js"; import { isNativePlanWaitResult, readNativePlanWait } from "./native-plan-wait.js"; import { activeIssueInteractionCondition } from "../issue-question-context.js"; import { hasPendingNativeChildCompletion } from "./native-child-completion-delivery.js"; @@ -1292,7 +1293,9 @@ export async function finalizeNativeRun(input: { }; const hasPendingChildCompletion = !reviewContext && await hasPendingNativeChildCompletion(input.db, childCompletionRecipient); + const monitorWaitAt = eligibleIssueMonitorWait(authoritativeIssue, run.agentId); const proposedDecision = resolveNativeFinalizerStatus({ + monitorWaitAuthorized: authoritativeIssue.workMode === "standard" && authoritativeIssue.executionRunId === run.id && monitorWaitAt !== null, planWaitAuthorized: planWait !== null, hasPendingChildCompletion, providerModelRejected: providerFailure?.errorCode === "native_provider_model_rejected" && ownsProviderFailureDecision, @@ -1381,6 +1384,8 @@ export async function finalizeNativeRun(input: { priorStatusVersion: Number(authoritativeIssue.statusVersion), priorDecisionId: authoritativeIssue.lastStatusDecisionId, decision, + requireMonitorWait: decision.reasonCode === "scheduled_monitor_waiting" && monitorWaitAt + ? { agentId: run.agentId, nextCheckAt: monitorWaitAt } : undefined, requirePlanWaitSource: decision.reasonCode === "native_plan_accepted_waiting_for_continuation" ? planWait?.source : undefined, diff --git a/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts b/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts index 7aebb78955..30b87a6085 100644 --- a/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts +++ b/server/src/services/native-runtime/paperclip-runner-tool-authority.test.ts @@ -102,7 +102,7 @@ describe("PaperclipRunnerToolAuthority", () => { issueId, runId, }); - expect(authority.definitions()).toHaveLength(38); + expect(authority.definitions()).toHaveLength(39); const questions = authority.definitions().find(tool => tool.name === "request_human_input")!; expect(questions.description).toContain("ask only the next unanswered question"); expect(questions.description).toContain("Never fabricate answers"); @@ -128,6 +128,7 @@ describe("PaperclipRunnerToolAuthority", () => { "request_human_input", "create_task", "set_dependencies", + "set_task_monitor", "list_documents", "read_document", "list_document_revisions", diff --git a/server/src/services/native-runtime/paperclip-runner-tool-authority.ts b/server/src/services/native-runtime/paperclip-runner-tool-authority.ts index 8a78ed5884..0fe55c24e6 100644 --- a/server/src/services/native-runtime/paperclip-runner-tool-authority.ts +++ b/server/src/services/native-runtime/paperclip-runner-tool-authority.ts @@ -1,3 +1,4 @@ +import { assertCanManageIssueMonitor, prepareIssueMonitorUpdate, setTaskMonitorSchema, summarizeIssueMonitor } from "../issue-monitors.js"; import { readTaskQuestionContext } from "../issue-question-context.js"; import { isConversation } from "../agent-conversations.js"; import { setIssueTitle } from "../issue-title.js"; @@ -14,7 +15,7 @@ import type { createAssignedMcpTools } from "./assigned-mcp-tools.js"; import { assertAssignableAgent } from "../agent-assignability.js"; import { approvalReadSqlCondition, canActorReadApproval, authorizationService, issueReadSqlCondition, type AuthorizationActor } from "../authorization.js"; import { resolveCoreTrustPreset } from "../trust-preset-resolver.js"; -import { normalizeIssueExecutionPolicy } from "../issue-execution-policy.js"; +import { normalizeIssueExecutionPolicy, redactIssueMonitorExternalRef } from "../issue-execution-policy.js"; import { buildLowTrustSourceTrust } from "../source-trust.js"; import { handoffPlanContext } from "./handoff-plan-context.js"; import { callCreateSkillTool, callUpdateSkillTool } from "../skill-tools.js"; @@ -101,7 +102,7 @@ const IMPLEMENTED_OPERATIONS = new Set([ "submit_complaint", "submit_suggestion", "read_agent_instructions", "update_agent_instructions", "get_agent_instruction_history", "restore_agent_instructions", "search_api", "call_api", "hire_agent", - "get_task_context", "get_task_history", "search_tasks", "report_progress", "set_task_title", + "get_task_context", "get_task_history", "search_tasks", "report_progress", "set_task_title", "set_task_monitor", "request_human_input", "create_skill", "update_skill", "create_task", "reassign_task", "set_dependencies", "create_project", "list_project_repositories", "list_projects", "register_deliverable", "list_documents", "read_document", "list_document_revisions", "write_document", @@ -582,6 +583,7 @@ export class PaperclipRunnerToolAuthority { (await captureRunIdentity(this.db, this.binding)).context?.id ?? null); case "reassign_task": return this.#reassignTask(input); case "set_task_title": return this.#setTaskTitle(input); + case "set_task_monitor": return this.#setTaskMonitor(input); case "set_dependencies": return this.#setDependencies(input); case "register_deliverable": return this.#registerDeliverable(input); default: throw new Error("paperclip_runner_tool_not_bound"); @@ -1349,6 +1351,84 @@ export class PaperclipRunnerToolAuthority { return result; } + async #setTaskMonitor(value: Record): Promise { + const input = setTaskMonitorSchema.parse(value); + if (input.monitor?.externalRef) input.monitor.externalRef = redactIssueMonitorExternalRef(input.monitor.externalRef); + const targetId = input.taskId ?? this.binding.issueId; + let publication: Parameters[0] | null = null; + let replayed = false; + let replayStatus: string | null = null; + let replayMonitor: ReturnType | null = null; + const authorizeTarget = async (tx: Db, context: { run: typeof heartbeatRuns.$inferSelect }) => { + // A second owned task may itself be writing another task: fail on lock + // contention instead of taking opposite issue/run locks and deadlocking. + const [target] = await tx.select().from(issues).where(and( + eq(issues.id, targetId), eq(issues.companyId, this.binding.companyId), + )).for("update", { noWait: true }); + if (!target) throw notFound("Task not found"); + const actor = this.#privacyActor(context.run); + const access = authorizationService(tx); + for (const action of ["issue:read", "issue:mutate"] as const) { + const decision = await access.decide({ actor, action, resource: { + type: "issue", companyId: this.binding.companyId, issueId: target.id, + status: target.status, assigneeAgentId: target.assigneeAgentId, assigneeUserId: target.assigneeUserId, + } }); + if (!decision.allowed) throw forbidden(decision.explanation); + } + await assertCanManageIssueMonitor(access, { actor }, target.companyId, target.assigneeAgentId, true); + if (target.assigneeAgentId !== this.binding.agentId || target.assigneeUserId || + !["in_progress", "in_review"].includes(target.status)) { + throw forbidden("Monitors require an in-progress or in-review task assigned to this agent"); + } + return { target, actor }; + }; + const result = await this.#withMutationReceipt("set_task_monitor", input.idempotencyKey, input, async (tx, context) => { + const { target, actor } = await authorizeTarget(tx, context); + // The task lock also serializes retries after a new run resumes the task. + // Keep the durable receipt in the existing run ledger, without a new table. + const [earlier] = await tx.select({ + receipt: sql`${heartbeatRuns.resultJson}->'semanticToolReceipts'->${input.idempotencyKey}`, + }).from(heartbeatRuns).where(and( + eq(heartbeatRuns.companyId, this.binding.companyId), + eq(heartbeatRuns.agentId, this.binding.agentId), + eq(heartbeatRuns.nativeIssueId, this.binding.issueId), + sql`${heartbeatRuns.resultJson}->'semanticToolReceipts' ? ${input.idempotencyKey}`, + )).orderBy(desc(heartbeatRuns.createdAt)).limit(1); + if (earlier?.receipt) { + if (earlier.receipt.operationId !== "set_task_monitor" || canonicalJson(earlier.receipt.input) !== canonicalJson(input)) { + throw new Error("paperclip_runner_tool_idempotency_conflict"); + } + replayed = true; + replayStatus = target.status; + replayMonitor = summarizeIssueMonitor(target, normalizeIssueExecutionPolicy(target.executionPolicy)); + return { taskId: target.id, status: target.status, monitor: replayMonitor }; + } + const patch = prepareIssueMonitorUpdate(target, input.monitor, actor); + const [updated] = await tx.update(issues).set({ ...patch, updatedAt: new Date() }) + .where(and(eq(issues.id, target.id), eq(issues.companyId, this.binding.companyId))).returning(); + if (!updated) throw notFound("Task not found"); + const monitor = summarizeIssueMonitor(updated, normalizeIssueExecutionPolicy(updated.executionPolicy)); + const activity = await persistActivity(tx, { + companyId: this.binding.companyId, actorType: "agent", actorId: this.binding.agentId, + agentId: this.binding.agentId, runId: this.binding.runId, + action: input.monitor ? "issue.monitor_scheduled" : "issue.monitor_cleared", + entityType: "issue", entityId: target.id, + details: { identifier: target.identifier, nextCheckAt: monitor.nextCheckAt, + notes: monitor.notes, scheduledBy: monitor.scheduledBy, serviceName: monitor.serviceName, + timeoutAt: monitor.timeoutAt, maxAttempts: monitor.maxAttempts, recoveryPolicy: monitor.recoveryPolicy }, + }); + publication = activity.publication; + return { taskId: updated.id, status: updated.status, monitor }; + }, { beforeReceiptReplay: async (tx, context) => { + const { target } = await authorizeTarget(tx, context); + replayed = true; + replayStatus = target.status; + replayMonitor = summarizeIssueMonitor(target, normalizeIssueExecutionPolicy(target.executionPolicy)); + } }); + if (publication) publishActivity(publication); + return { ...record(result), ...(replayed ? { status: replayStatus, monitor: replayMonitor } : {}), replayed }; + } + async #setDependencies(input: Record): Promise { const idempotencyKey = requiredString(input.idempotencyKey); if (!Array.isArray(input.blockedByTaskIds)) { @@ -2037,6 +2117,9 @@ function redactedTask(task: typeof issues.$inferSelect) { workMode: task.workMode, assigneeAgentId: task.assigneeAgentId, executionRunId: task.executionRunId, + ...(task.monitorNextCheckAt || task.monitorLastTriggeredAt ? { + monitor: summarizeIssueMonitor(task, normalizeIssueExecutionPolicy(task.executionPolicy)), + } : {}), parentId: task.parentId, projectId: task.projectId, goalId: task.goalId, diff --git a/server/src/services/native-runtime/status-arbiter.test.ts b/server/src/services/native-runtime/status-arbiter.test.ts index b8f9d8aa46..2e77bb7e41 100644 --- a/server/src/services/native-runtime/status-arbiter.test.ts +++ b/server/src/services/native-runtime/status-arbiter.test.ts @@ -44,6 +44,20 @@ function arbitrate( } describe("native status authority", () => { + it("waits on persisted monitors without an immediate continuation, even with unfinished work", () => { + const pending = assessment({ reportedDisposition: "yielded", hasBlockingRemainingWork: true, + continuation: { kind: "monitor", summary: "Check CI", idempotencyKey: "ci" } }); + for (const priorIssueStatus of ["in_progress", "in_review"] as const) { + expect(arbitrate({ assessment: pending, priorIssueStatus, monitorWaitAuthorized: true })) + .toMatchObject({ statusAction: "preserve", toStatus: priorIssueStatus, + reasonCode: "scheduled_monitor_waiting", effects: [{ kind: "release_checkout" }] }); + } + expect(arbitrate({ assessment: pending })).toMatchObject({ reasonCode: "monitor_wait_authority_lost", + effects: [{ kind: "record_finalization_error" }] }); + expect(arbitrate({ assessment: pending, monitorWaitAuthorized: true, hasUnresolvedIssueBlockers: true }).toStatus).toBe("blocked"); + expect(arbitrate({ assessment: pending, monitorWaitAuthorized: true, governanceGate: { kind: "approval", id: "pending" } }).toStatus).toBe("in_review"); + }); + it("keeps a pending child result non-terminal without adding another continuation", () => { expect(arbitrate({ hasPendingChildCompletion: true })).toMatchObject({ statusAction: "in_progress", toStatus: "in_progress", reasonCode: "native_child_completion_pending", @@ -150,7 +164,7 @@ describe("native status authority", () => { toStatus: "in_review", effects: [expect.objectContaining({ kind: "create_interaction" })], }); - for (const kind of ["same_agent", "retry", "monitor"] as const) { + for (const kind of ["same_agent", "retry"] as const) { expect( arbitrate({ assessment: { @@ -562,7 +576,7 @@ describe("native status authority", () => { expect.objectContaining({ statusAction: "blocked", toStatus: "blocked", - policyVersion: "phase6-v10", + policyVersion: "phase6-v11", reasonCode: "current_track_blocker_waiting", unblockDescriptor: { owner: "board", diff --git a/server/src/services/native-runtime/status-arbiter.ts b/server/src/services/native-runtime/status-arbiter.ts index 8439247abc..40fe4d7767 100644 --- a/server/src/services/native-runtime/status-arbiter.ts +++ b/server/src/services/native-runtime/status-arbiter.ts @@ -1,7 +1,7 @@ import type { NativeEvidenceAssessment } from "./evidence-classifier.js"; import { NATIVE_MODEL_REJECTION_MESSAGE, NATIVE_PROVIDER_CAPACITY_MAX_RETRIES, NATIVE_PROVIDER_OVERLOADED_CODE, NATIVE_PROVIDER_OVERLOADED_MESSAGE } from "./native-provider-failure.js"; -export const NATIVE_STATUS_ARBITER_POLICY_VERSION = "phase6-v10"; +export const NATIVE_STATUS_ARBITER_POLICY_VERSION = "phase6-v11"; export type NativeAuthoritativeIssueStatus = | "backlog" @@ -113,6 +113,7 @@ export function arbitrateNativeStatus(input: { "authorized" | "revoked" | "not_applicable"; /** Server-verified accepted plan request and normal provider terminal. */ planWaitAuthorized?: boolean; + monitorWaitAuthorized?: boolean; boardResponseWaitAuthorized?: boolean; boardResponseWaitOrigin?: boolean; isConversation?: boolean; @@ -325,7 +326,7 @@ export function arbitrateNativeStatus(input: { input.assessment.hasBlockingRemainingWork; if ( input.hasUnresolvedIssueBlockers === true && - (["done", "blocked"].includes(input.assessment.reportedDisposition) || unfinishedResponseWait) + (["done", "blocked"].includes(input.assessment.reportedDisposition) || unfinishedResponseWait || input.assessment.continuation?.kind === "monitor") ) { const owner = input.assessment.blocker?.boardOwned ? ("board" as const) @@ -461,6 +462,25 @@ export function arbitrateNativeStatus(input: { ], }; } + if (input.assessment.reportedDisposition === "yielded" && input.assessment.continuation?.kind === "monitor") { + if (input.monitorWaitAuthorized) { + return { + policyVersion: NATIVE_STATUS_ARBITER_POLICY_VERSION, + statusAction: "preserve", toStatus: input.priorIssueStatus, + reasonCode: "scheduled_monitor_waiting", unblockDescriptor: null, + // The persisted timer owns delivery, never enqueue an immediate wake. + effects: [{ kind: "release_checkout" }], + }; + } + return { + policyVersion: NATIVE_STATUS_ARBITER_POLICY_VERSION, + statusAction: "preserve", toStatus: input.priorIssueStatus, + reasonCode: "monitor_wait_authority_lost", unblockDescriptor: null, + effects: [{ kind: "record_finalization_error", cause: "monitor_wait_authority_lost", + nextAction: "The scheduled monitor was cleared or became ineligible. Re-establish a valid task action path.", + agentId: input.agentId }], + }; + } if ( input.assessment.reportedDisposition === "yielded" && input.assessment.continuation?.kind === "response_wake" && diff --git a/server/src/services/native-runtime/status-decision-committer.ts b/server/src/services/native-runtime/status-decision-committer.ts index 1834ee01fd..6d60ac6143 100644 --- a/server/src/services/native-runtime/status-decision-committer.ts +++ b/server/src/services/native-runtime/status-decision-committer.ts @@ -1,3 +1,4 @@ +import { eligibleIssueMonitorWait } from "../issue-monitors.js"; import { readNativePlanWait, type NativePlanWaitSource } from "./native-plan-wait.js"; import { activeIssueInteractionCondition } from "../issue-question-context.js"; import { isConversation } from "../agent-conversations.js"; @@ -732,15 +733,13 @@ async function materializeDecisionEffect(input: { }, }; } + if (effect.continuationKind === "monitor") throw new Error("Monitor delivery belongs to the issue monitor scheduler"); const wakeId = await enqueueWake({ tx: input.tx, companyId: input.companyId, issueId: input.issue.id, agentId: effect.agentId, - reason: - effect.continuationKind === "monitor" - ? "monitor_due" - : "issue_status_changed", + reason: "issue_status_changed", idempotencyKey: `native-status:${input.decisionId}:continuation`, payload: { nativeDecisionId: input.decisionId, @@ -1556,6 +1555,7 @@ export async function commitNativeStatusDecision(input: { supersedesCommittedDecisionId?: string; requireExternalChatResponseWaitAuthorization?: { agentId: string }; requirePlanWaitSource?: NativePlanWaitSource; + requireMonitorWait?: { agentId: string; nextCheckAt: string }; requireNoPendingChildCompletion?: NativeChildCompletionRecipient; requireModelRejectionOwner?: { agentId: string; reviewContext: NativeReviewAssignmentContext | null }; requireProviderFailureOwner?: { agentId: string; reviewContext: NativeReviewAssignmentContext | null }; @@ -1646,6 +1646,15 @@ export async function commitNativeStatusDecision(input: { ) { throw new NativeStatusRaceError(); } + if (reasonCode === "scheduled_monitor_waiting") { + const expected = input.requireMonitorWait; + if (!expected || issue.workMode !== "standard" || issue.executionRunId !== input.runId || + input.decision.statusAction !== "preserve" || input.decision.effects.length !== 1 || + input.decision.effects[0]?.kind !== "release_checkout" || + eligibleIssueMonitorWait(issue, expected.agentId) !== expected.nextCheckAt) { + throw new NativeStatusRaceError(); + } + } // Recheck under the parent status lock: a child can finish between the // finalizer's read and this commit. Re-arbitrate instead of discarding its wake. if (input.requireNoPendingChildCompletion && await hasPendingNativeChildCompletion( diff --git a/server/src/services/native-runtime/task-monitor.integration.test.ts b/server/src/services/native-runtime/task-monitor.integration.test.ts new file mode 100644 index 0000000000..2c7f5930ac --- /dev/null +++ b/server/src/services/native-runtime/task-monitor.integration.test.ts @@ -0,0 +1,204 @@ +import { randomUUID } from "node:crypto"; +import { and, eq } from "drizzle-orm"; +import { afterAll, beforeAll, describe, expect, it } from "vitest"; +import { activityLog, agents, authUsers, companyMemberships, companies, createDb, heartbeatRuns, issues } from "@paperclipai/db"; +import { PROVIDER_QUOTA_MONITOR_SERVICE_NAME } from "@paperclipai/shared"; +import { getEmbeddedPostgresTestSupport, startEmbeddedPostgresTestDatabase } from "../../__tests__/helpers/embedded-postgres.js"; +import { PaperclipRunnerToolAuthority } from "./paperclip-runner-tool-authority.js"; +import { buildIssueMonitorTriggeredPatch, normalizeIssueExecutionPolicy } from "../issue-execution-policy.js"; +import { nativeCompletionFeedback } from "./native-completion-feedback.js"; +import type { PrpStructuredRunResult } from "../../vendor/paperclip-runner/index.js"; + +const embeddedPostgresSupport = await getEmbeddedPostgresTestSupport(); +const describeEmbeddedPostgres = embeddedPostgresSupport.supported ? describe : describe.skip; +if (!embeddedPostgresSupport.supported) { + console.warn(`Skipping embedded Postgres native task monitor tests on this host: ${embeddedPostgresSupport.reason ?? "unsupported environment"}`); +} + +describeEmbeddedPostgres("native task monitors", () => { + let temporary: Awaited>; + let db: ReturnType; + beforeAll(async () => { + temporary = await startEmbeddedPostgresTestDatabase("native-task-monitor-"); + db = createDb(temporary.connectionString); + }); + afterAll(async () => { await temporary?.cleanup(); }); + + async function fixture() { + const companyId = randomUUID(), agentId = randomUUID(), issueId = randomUUID(), runId = randomUUID(); + await db.insert(companies).values({ id: companyId, name: "Monitors", issuePrefix: `M${companyId.slice(0, 7)}` }); + await db.insert(agents).values({ id: agentId, companyId, name: "Worker", adapterType: "paperclip_runner", status: "active" }); + await db.insert(issues).values({ id: issueId, companyId, title: "Deferred check", status: "in_progress", assigneeAgentId: agentId }); + await db.insert(heartbeatRuns).values({ id: runId, companyId, agentId, nativeIssueId: issueId, runtimeMode: "native", status: "running", contextSnapshot: { issueId } }); + await db.update(issues).set({ executionRunId: runId }).where(eq(issues.id, issueId)); + const binding = { companyId, agentId, issueId, runId }; + const authority = new PaperclipRunnerToolAuthority(db, binding); + const monitor = { nextCheckAt: new Date(Date.now() + 60_000).toISOString(), notes: "Compare this task's run and session identities." }; + const call = (args: Record) => authority.execute({ tool: "set_task_monitor", callId: randomUUID(), arguments: args }); + const read = async () => (await db.select().from(issues).where(eq(issues.id, issueId)))[0]!; + return { ...binding, binding, authority, monitor, call, read }; + } + + it("advertises a real binding only in standard execution", async () => { + const value = await fixture(); + expect(value.authority.definitions().some(tool => tool.name === "set_task_monitor")).toBe(true); + for (const workMode of ["planning", "ask"] as const) { + const authority = new PaperclipRunnerToolAuthority(db, { ...value.binding, workMode }); + expect(authority.definitions().some(tool => tool.name === "set_task_monitor")).toBe(false); + await db.update(issues).set({ workMode }).where(eq(issues.id, value.issueId)); + await expect(authority.execute({ tool: "set_task_monitor", callId: randomUUID(), arguments: { idempotencyKey: "forbidden", monitor: value.monitor } })).rejects.toThrow(); + } + }); + + it("keeps provider-neutral discovery and review-only restrictions", async () => { + const value = await fixture(); + for (const provider of ["codex", "opencode", "claude_managed", "aws_agentcore", "acpx"]) { + await db.update(agents).set({ adapterConfig: { provider } }).where(eq(agents.id, value.agentId)); + expect(value.authority.definitions().map(tool => tool.name)).toContain("set_task_monitor"); + } + const authority = new PaperclipRunnerToolAuthority(db, { ...value.binding, + nativeReview: { nativeReviewInteractionId: randomUUID(), nativeReviewDecisionId: randomUUID() } }); + expect(authority.definitions().map(tool => tool.name)).not.toContain("set_task_monitor"); + await expect(authority.execute({ tool: "set_task_monitor", callId: randomUUID(), arguments: { idempotencyKey: "review", monitor: value.monitor } })).rejects.toThrow("review run"); + }); + + it("rechecks the responsible user's permissions before mutation and replay", async () => { + const value = await fixture(), userId = randomUUID(); + await db.insert(authUsers).values({ id: userId, name: "Owner", email: `${userId}@example.test`, createdAt: new Date(), updatedAt: new Date() }); + await db.insert(companyMemberships).values({ companyId: value.companyId, principalType: "user", principalId: userId, membershipRole: "owner" }); + await db.update(heartbeatRuns).set({ responsibleUserId: userId }).where(eq(heartbeatRuns.id, value.runId)); + const original = { idempotencyKey: "authorized", monitor: value.monitor }; + await value.call(original); + await db.update(companyMemberships).set({ membershipRole: "viewer" }).where(eq(companyMemberships.companyId, value.companyId)); + await expect(value.call(original)).rejects.toThrow(); + await expect(value.call({ idempotencyKey: "clear", monitor: null })).rejects.toThrow(); + expect((await value.read()).monitorNextCheckAt?.toISOString()).toBe(value.monitor.nextCheckAt); + }); + + it("persists a monitor and its audit atomically while preserving review policy", async () => { + const value = await fixture(); + const policy = { mode: "normal", commentRequired: false, stages: [{ id: randomUUID(), type: "review", approvalsNeeded: 1, participants: [{ type: "user", userId: "reviewer" }] }] }; + await db.update(issues).set({ executionPolicy: policy }).where(eq(issues.id, value.issueId)); + const result = await value.call({ idempotencyKey: "schedule", monitor: value.monitor }); + expect(result).toMatchObject({ taskId: value.issueId, monitor: { ...value.monitor, scheduledBy: "assignee", status: "scheduled" }, replayed: false }); + const saved = await value.read(); + expect(saved.executionPolicy).toMatchObject(policy); + expect(saved.monitorNextCheckAt?.toISOString()).toBe(value.monitor.nextCheckAt); + expect(saved.status).toBe("in_progress"); + const audits = await db.select().from(activityLog).where(and(eq(activityLog.entityId, value.issueId), eq(activityLog.action, "issue.monitor_scheduled"))); + expect(audits).toHaveLength(1); + expect(audits[0]).toMatchObject({ runId: value.runId, agentId: value.agentId }); + }); + + it("replaces and clears schedules without re-arming a receipt on replay", async () => { + const value = await fixture(); + const original = { idempotencyKey: "first", monitor: value.monitor }; + await value.call(original); + const replacement = { ...value.monitor, nextCheckAt: new Date(Date.now() + 120_000).toISOString() }; + await value.call({ idempotencyKey: "replace", monitor: replacement }); + expect(await value.call(original)).toMatchObject({ monitor: { nextCheckAt: replacement.nextCheckAt }, replayed: true }); + await value.call({ idempotencyKey: "clear", monitor: null }); + expect(await value.call(original)).toMatchObject({ monitor: { nextCheckAt: null, status: "cleared" }, replayed: true }); + expect((await value.read()).monitorNextCheckAt).toBeNull(); + await expect(value.call({ ...original, monitor: replacement })).rejects.toThrow("idempotency_conflict"); + const audits = await db.select().from(activityLog).where(eq(activityLog.entityId, value.issueId)); + expect(audits.filter(row => row.action.startsWith("issue.monitor_"))).toHaveLength(3); + }); + + it("exposes consumed monitor instructions to a resumed task and does not re-arm them on retry", async () => { + const value = await fixture(); + const input = { idempotencyKey: "consumed", monitor: value.monitor }; + await value.call(input); + const scheduled = await value.read(); + await db.update(issues).set(buildIssueMonitorTriggeredPatch({ issue: scheduled, + policy: normalizeIssueExecutionPolicy(scheduled.executionPolicy), triggeredAt: new Date() })).where(eq(issues.id, value.issueId)); + expect(await value.authority.execute({ tool: "get_task_context", callId: randomUUID(), arguments: {} })) + .toMatchObject({ activeTask: { monitor: { status: "triggered", nextCheckAt: null, notes: value.monitor.notes, attemptCount: 1 } } }); + expect(await value.call(input)).toMatchObject({ replayed: true, monitor: { nextCheckAt: null, status: "triggered" } }); + expect((await value.read()).monitorNextCheckAt).toBeNull(); + }); + + it("keeps a cleared schedule cleared when a successor run retries the original key", async () => { + const value = await fixture(); + const original = { idempotencyKey: "durable-schedule", monitor: value.monitor }; + await value.call(original); + await value.call({ idempotencyKey: "durable-clear", monitor: null }); + await db.update(heartbeatRuns).set({ status: "succeeded" }).where(eq(heartbeatRuns.id, value.runId)); + const nextRunId = randomUUID(); + await db.insert(heartbeatRuns).values({ id: nextRunId, companyId: value.companyId, agentId: value.agentId, + nativeIssueId: value.issueId, runtimeMode: "native", status: "running", contextSnapshot: { issueId: value.issueId } }); + await db.update(issues).set({ executionRunId: nextRunId }).where(eq(issues.id, value.issueId)); + const authority = new PaperclipRunnerToolAuthority(db, { ...value.binding, runId: nextRunId }); + expect(await authority.execute({ tool: "set_task_monitor", callId: randomUUID(), arguments: original })) + .toMatchObject({ replayed: true, monitor: { nextCheckAt: null, status: "cleared" } }); + expect((await value.read()).monitorNextCheckAt).toBeNull(); + await expect(value.call(original)).rejects.toThrow(); + }); + + it("allows another owned task but rejects foreign ownership and company scope", async () => { + const value = await fixture(), other = await fixture(), targetId = randomUUID(); + await db.insert(issues).values({ id: targetId, companyId: value.companyId, title: "Another owned check", status: "in_review", assigneeAgentId: value.agentId }); + expect(await value.call({ taskId: targetId, idempotencyKey: "other", monitor: value.monitor })).toMatchObject({ taskId: targetId }); + await expect(value.call({ taskId: other.issueId, idempotencyKey: "foreign", monitor: value.monitor })).rejects.toThrow("Task not found"); + const otherAgent = randomUUID(); + await db.insert(agents).values({ id: otherAgent, companyId: value.companyId, name: "Other", adapterType: "paperclip_runner" }); + await db.update(issues).set({ assigneeAgentId: otherAgent }).where(eq(issues.id, targetId)); + await expect(value.call({ taskId: targetId, idempotencyKey: "other", monitor: value.monitor })).rejects.toThrow(); + }); + + it("rejects invalid schedules, exhausted bounds and ineligible tasks without writes", async () => { + const value = await fixture(); + for (const monitor of [ + { ...value.monitor, notes: "" }, + { ...value.monitor, nextCheckAt: new Date(0).toISOString() }, + { ...value.monitor, scheduledBy: "board" }, + { ...value.monitor, timeoutAt: value.monitor.nextCheckAt }, + { ...value.monitor, maxAttempts: 101 }, + ]) await expect(value.call({ idempotencyKey: randomUUID(), monitor })).rejects.toThrow(); + await db.update(issues).set({ monitorAttemptCount: 1 }).where(eq(issues.id, value.issueId)); + await expect(value.call({ idempotencyKey: "exhausted", monitor: { ...value.monitor, maxAttempts: 1 } })).rejects.toThrow(); + for (const status of ["blocked", "backlog", "done", "cancelled"] as const) { + await db.update(issues).set({ status }).where(eq(issues.id, value.issueId)); + await expect(value.call({ idempotencyKey: randomUUID(), monitor: value.monitor })).rejects.toThrow(); + } + expect((await value.read()).monitorNextCheckAt).toBeNull(); + }); + + it("accepts unfinished monitor waits only after scheduling the current task", async () => { + const value = await fixture(); + const result: PrpStructuredRunResult = { + schema: "paperclip.run_result.v1", reportedWorkDisposition: "yielded", summary: "I will compare the next run.", + completionClaim: { contractRevision: "test", objectiveSatisfied: false, criteria: [], remainingWork: [{ description: "Compare next run", blocksCompletion: true }] }, + continuation: { kind: "monitor", summary: "Wait for the scheduled check", idempotencyKey: "wait" }, + evidence: [], verification: [], attentionRequests: [], artifacts: [], + }; + await expect(nativeCompletionFeedback(db, value.runId, result)).rejects.toThrow("persisted, eligible monitor"); + const otherId = randomUUID(); + await db.insert(issues).values({ id: otherId, companyId: value.companyId, title: "Other", status: "in_progress", assigneeAgentId: value.agentId }); + await value.call({ taskId: otherId, idempotencyKey: "other", monitor: value.monitor }); + await expect(nativeCompletionFeedback(db, value.runId, result)).rejects.toThrow("persisted, eligible monitor"); + await value.call({ idempotencyKey: "self", monitor: value.monitor }); + await expect(nativeCompletionFeedback(db, value.runId, result)).resolves.toContain("issue_monitor_due"); + // Legacy APIs may persist a server-owned quota monitor. It cannot justify + // a native wait because its dispatcher explicitly excludes native runs. + await db.update(issues).set({ executionPolicy: { monitor: { ...value.monitor, serviceName: PROVIDER_QUOTA_MONITOR_SERVICE_NAME } } }).where(eq(issues.id, value.issueId)); + await expect(nativeCompletionFeedback(db, value.runId, result)).rejects.toThrow("persisted, eligible monitor"); + await value.call({ idempotencyKey: "clear", monitor: null }); + await expect(nativeCompletionFeedback(db, value.runId, result)).rejects.toThrow("persisted, eligible monitor"); + }); + + it("rejects the server-owned quota recovery name without promising a wake", async () => { + const value = await fixture(); + await expect(value.call({ idempotencyKey: "quota", monitor: { + ...value.monitor, serviceName: PROVIDER_QUOTA_MONITOR_SERVICE_NAME, externalRef: value.runId, + } })).rejects.toThrow("reserved for server-owned quota recovery"); + expect((await value.read()).monitorNextCheckAt).toBeNull(); + const audits = await db.select().from(activityLog).where(eq(activityLog.entityId, value.issueId)); + expect(audits).toHaveLength(0); + const [run] = await db.select().from(heartbeatRuns).where(eq(heartbeatRuns.id, value.runId)); + expect(run.resultJson?.semanticToolReceipts).toBeUndefined(); + await expect(value.call({ idempotencyKey: "quota", monitor: { + ...value.monitor, serviceName: "Provider usage dashboard", + } })).resolves.toMatchObject({ monitor: { status: "scheduled" }, replayed: false }); + }); +});