mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-09 16:35:27 +02:00
## Thinking Path > - Paperclip is the open source app people use to manage AI agents for work > - The server admits agent work through heartbeat scheduling and execution paths > - Operators need to stop new work before maintenance or a graceful shutdown > - A process restart alone does not provide a reusable admission control primitive > - This pull request adds an instance API that holds new task admission and reports process quiescence > - The benefit is a small, auditable control that lets operators wait for active work without a restart ## Linked Issues or Issue Description **Problem or motivation** Operators cannot hold new task admission without restarting the Paperclip process. A restart can interrupt maintenance flows and does not provide a status signal for active work. **Proposed solution** Add `GET /instance/task-drain`, `POST /instance/task-drain`, and `DELETE /instance/task-drain`. The server keeps the drain state in process memory, applies it to every scheduling suppression path, supports an optional TTL up to 24 hours, and reports active wake and run counts. **Alternatives considered** A timer would clear the drain after its TTL, but it could keep the Node.js event loop open during shutdown. A database row would add storage and query work for process-local state. The implementation uses lazy expiry and process memory instead. **Roadmap alignment** The change supports the roadmap goal for enforced outcomes and safe recovery actions. It does not duplicate a listed roadmap item. **Additional context** This is a server and shared-package change. It adds no user interface and no database migration. ## What Changed - Add process-local task-drain state with lazy TTL expiry. - Add task-drain admission suppression to the shared heartbeat resolver. - Add instance routes to read, start, and stop a task drain. - Add validation for positive TTL values and the shared 24-hour maximum. - Add activity records for drain mutations and tests for status, access control, validation, and suppression. ## Verification - Run `pnpm exec vitest run --project @paperclipai/server server/src/__tests__/heartbeat-task-drain.test.ts server/src/__tests__/instance-settings-routes.test.ts server/src/__tests__/heartbeat-scheduling-suppression.test.ts`. - Run `pnpm --filter @paperclipai/shared exec tsc --noEmit`. - Run `pnpm --filter @paperclipai/server exec tsc --noEmit` and compare its known pre-existing errors with the base commit. - Confirm that pull request CI reaches a terminal green state. ## Risks The drain state exists only in process memory, so a restart clears it. This behavior matches the process-local design. A drain without a TTL remains active until an operator calls the delete route. The status route reads in-memory activity sets and does not query stale database rows. ## Model Used OpenAI Codex, GPT-5, extended reasoning with tool use and code execution. The exact runtime context window is not exposed by the execution environment. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [x] All Paperclip CI gates are green - [x] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip <noreply@paperclip.ing>
162 lines
7.1 KiB
TypeScript
162 lines
7.1 KiB
TypeScript
import { z } from "zod";
|
|
import { DEFAULT_FEEDBACK_DATA_SHARING_PREFERENCE } from "../types/feedback.js";
|
|
import {
|
|
DAILY_RETENTION_PRESETS,
|
|
WEEKLY_RETENTION_PRESETS,
|
|
MONTHLY_RETENTION_PRESETS,
|
|
DEFAULT_BACKUP_RETENTION,
|
|
DEFAULT_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS,
|
|
MAX_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS,
|
|
MIN_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS,
|
|
} from "../types/instance.js";
|
|
import { feedbackDataSharingPreferenceSchema } from "./feedback.js";
|
|
import { shapeWithoutDefaults } from "./partial.js";
|
|
|
|
function presetSchema<T extends readonly number[]>(presets: T, label: string) {
|
|
return z.number().refine(
|
|
(v): v is T[number] => (presets as readonly number[]).includes(v),
|
|
{ message: `${label} must be one of: ${presets.join(", ")}` },
|
|
);
|
|
}
|
|
|
|
export const backupRetentionPolicySchema = z.object({
|
|
dailyDays: presetSchema(DAILY_RETENTION_PRESETS, "dailyDays").default(DEFAULT_BACKUP_RETENTION.dailyDays),
|
|
weeklyWeeks: presetSchema(WEEKLY_RETENTION_PRESETS, "weeklyWeeks").default(DEFAULT_BACKUP_RETENTION.weeklyWeeks),
|
|
monthlyMonths: presetSchema(MONTHLY_RETENTION_PRESETS, "monthlyMonths").default(DEFAULT_BACKUP_RETENTION.monthlyMonths),
|
|
});
|
|
|
|
export const instanceGeneralSettingsSchema = z.object({
|
|
censorUsernameInLogs: z.boolean().default(false),
|
|
keyboardShortcuts: z.boolean().default(false),
|
|
feedbackDataSharingPreference: feedbackDataSharingPreferenceSchema.default(
|
|
DEFAULT_FEEDBACK_DATA_SHARING_PREFERENCE,
|
|
),
|
|
backupRetention: backupRetentionPolicySchema.default(DEFAULT_BACKUP_RETENTION),
|
|
// Execution policy. Absent/"any" = unrestricted; "kubernetes" forces the
|
|
// Kubernetes sandbox provider and denies local/ssh execution (cloud_tenant).
|
|
executionMode: z.enum(["kubernetes", "any"]).optional(),
|
|
}).strict();
|
|
|
|
export const patchInstanceGeneralSettingsSchema = z
|
|
.object(shapeWithoutDefaults(instanceGeneralSettingsSchema.shape))
|
|
.partial()
|
|
.strict();
|
|
|
|
export const instanceExperimentalSettingsSchema = z.object({
|
|
enableEnvironments: z.boolean().default(false),
|
|
enableNativeRunner: z.boolean().default(false),
|
|
enableManagedSandboxOnly: z.boolean().default(false),
|
|
enableIsolatedWorkspaces: z.boolean().default(false),
|
|
enableStreamlinedLeftNavigation: z.boolean().default(true),
|
|
enableApps: z.boolean().default(false),
|
|
enablePipelines: z.boolean().default(false),
|
|
enableCases: z.boolean().default(false),
|
|
enableConferenceRoomChat: z.boolean().default(false),
|
|
enableClassicTaskInterface: z.boolean().default(false),
|
|
enableTaskWatchdogs: z.boolean().default(false),
|
|
enableIssuePlanDecompositions: z.boolean().default(false),
|
|
enableExperimentalFileViewer: z.boolean().default(false),
|
|
enableExternalObjects: z.boolean().default(false),
|
|
enableSmokeLab: z.boolean().default(false),
|
|
enableBuiltInAgents: z.boolean().default(false),
|
|
enableBetaSkills: z.boolean().default(false),
|
|
enableSummaries: z.boolean().default(false),
|
|
enableStatusCards: z.boolean().default(false),
|
|
enableDecisions: z.boolean().default(false),
|
|
enableGoalsSidebarLink: z.boolean().default(false),
|
|
enableServerInfoDebugView: z.boolean().default(false),
|
|
enableSimplifiedEnglishInteractions: z.boolean().default(false),
|
|
autoRestartDevServerWhenIdle: z.boolean().default(false),
|
|
enableIssueGraphLivenessAutoRecovery: z.boolean().default(false),
|
|
enableWorkspaceBranchReconcileForward: z.boolean().default(true),
|
|
enableWorkspaceDirtyQuarantineRepair: z.boolean().default(true),
|
|
enableOwnerInstanceAdmin: z.boolean().default(false),
|
|
// Kill switch for the sandbox duplex command-stream bridge. Default off. When
|
|
// off the host keeps the file bridge for every run with no manifest change and
|
|
// no redeploy. The host reads this per run before it selects the transport.
|
|
enableSandboxDuplexBridge: z.boolean().default(false),
|
|
enableWorktreeRunExecution: z.boolean().default(false),
|
|
worktreeRunExecutionActivatedAt: z.string().datetime().nullable().default(null),
|
|
worktreeRunExecutionActivationInstanceId: z.string().min(1).nullable().default(null),
|
|
issueGraphLivenessAutoRecoveryLookbackHours: z
|
|
.number()
|
|
.int()
|
|
.min(MIN_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS)
|
|
.max(MAX_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS)
|
|
.default(DEFAULT_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS),
|
|
}).strict();
|
|
|
|
export const patchInstanceExperimentalSettingsSchema = z
|
|
.object(
|
|
shapeWithoutDefaults(
|
|
instanceExperimentalSettingsSchema
|
|
.omit({
|
|
worktreeRunExecutionActivatedAt: true,
|
|
worktreeRunExecutionActivationInstanceId: true,
|
|
})
|
|
.shape,
|
|
),
|
|
)
|
|
.partial()
|
|
.strip();
|
|
|
|
export const managedSettingMetadataSchema = z.object({
|
|
managed: z.literal(true),
|
|
managedBy: z.literal("paperclip-cloud"),
|
|
}).strict();
|
|
|
|
// Response shape of the experimental settings endpoints: on cloud-managed
|
|
// instances every overlaid key is listed in `managedKeys`; self-hosted
|
|
// responses omit the field entirely.
|
|
export const instanceExperimentalSettingsWithManagedSchema = instanceExperimentalSettingsSchema.extend({
|
|
managedKeys: z.record(z.string(), managedSettingMetadataSchema).optional(),
|
|
}).strict();
|
|
|
|
export const patchInstanceSettingsSchema = z.object({
|
|
defaultEnvironmentId: z.string().guid().nullable().optional(),
|
|
}).strict();
|
|
|
|
export const issueGraphLivenessAutoRecoveryRequestSchema = z.object({
|
|
lookbackHours: z
|
|
.number()
|
|
.int()
|
|
.min(MIN_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS)
|
|
.max(MAX_ISSUE_GRAPH_LIVENESS_AUTO_RECOVERY_LOOKBACK_HOURS)
|
|
.optional(),
|
|
}).strict();
|
|
|
|
// The longest time a task drain can run before it expires on its own. A
|
|
// caller can send a shorter `ttlMs`, but not a longer one — the request must
|
|
// fail instead of the server silently clamping the value.
|
|
export const MAX_TASK_DRAIN_TTL_MS = 24 * 60 * 60 * 1000;
|
|
|
|
export const startTaskDrainRequestSchema = z.object({
|
|
ttlMs: z.number().int().positive().max(MAX_TASK_DRAIN_TTL_MS).nullable().optional(),
|
|
}).strict();
|
|
|
|
export type InstanceGeneralSettings = z.infer<typeof instanceGeneralSettingsSchema>;
|
|
// The patch schema removes each default so an absent key stays absent. Declare
|
|
// the type from the full settings type, so every field keeps its precise type.
|
|
export type PatchInstanceGeneralSettings = Partial<InstanceGeneralSettings>;
|
|
export type InstanceExperimentalSettings = z.infer<typeof instanceExperimentalSettingsSchema>;
|
|
export type PatchInstanceExperimentalSettings = Partial<
|
|
Omit<
|
|
InstanceExperimentalSettings,
|
|
"worktreeRunExecutionActivatedAt" | "worktreeRunExecutionActivationInstanceId"
|
|
>
|
|
>;
|
|
export type PatchInstanceSettings = z.infer<typeof patchInstanceSettingsSchema>;
|
|
export type IssueGraphLivenessAutoRecoveryRequest = z.infer<
|
|
typeof issueGraphLivenessAutoRecoveryRequestSchema
|
|
>;
|
|
export type StartTaskDrainRequest = z.infer<typeof startTaskDrainRequestSchema>;
|
|
|
|
export const instanceSettingsSchema = z.object({
|
|
id: z.string().guid(),
|
|
defaultEnvironmentId: z.string().guid().nullable(),
|
|
general: instanceGeneralSettingsSchema,
|
|
experimental: instanceExperimentalSettingsWithManagedSchema,
|
|
createdAt: z.union([z.date(), z.string().datetime()]),
|
|
updatedAt: z.union([z.date(), z.string().datetime()]),
|
|
}).strict();
|