mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-09 06:15:21 +02:00
<!-- ASD-STE100 --> ## Thinking Path > - Paperclip is the open source app people use to manage AI agents for work > - The server keeps fleet health with periodic sweeps and shows a dashboard with run activity > - The Paperclip instance became slow again after the first round of recovery-sweep indexes landed > - Live profiling found four steady-state hot paths that read much more data than they use > - This pull request bounds the dashboard recursion, adds the missing taskKey index, and narrows two wide reads > - The benefit is a large drop in constant database load and a responsive server ## Linked Issues or Issue Description **Describe the bug** The server becomes slow while agents work. Live query sampling shows four hot paths: 1. The dashboard run-activity recursive CTE reads every run a company ever had on each call. One call takes 2.85 seconds. The UI calls it after almost every fleet event through the dashboard and sidebar-badges routes. 2. The productivity-review sweep runs each 30 seconds. Its run-scope filter is `issueId OR taskId OR taskKey` on the run context JSONB. No index exists for `taskKey`. The planner must detoast every run snapshot for the agent. One query takes 444 ms and the sweep makes one for each of ~152 candidate issues. 3. The attention failed-run section selects the full `context_snapshot` for every run newer than the oldest exhausted run. That fetch moves 29 MB for each feed build. 4. The retention sweep pages the attention feed with a cursor. Each page makes a full feed rebuild. **Expected behavior** Periodic sweeps and dashboard queries read only the data they use, and use indexes. **Actual behavior** The database stays saturated. Users see a slow server. ## What Changed - `server/src/services/dashboard.ts`: bound both arms of the `recovered_runs` recursive CTE to the chart window. A retry is always newer than the run it retries, so the bound cannot change visible chart data. Live time went from 2,852 ms to 54 ms. - `packages/db/src/migrations/0210_heartbeat_context_taskkey_index.sql`: add the `taskKey` expression index that completes the issueId/taskId/taskKey trio. With all three, the planner uses a BitmapOr. Live time for the productivity run-scope query went from 444 ms to 1.9 ms. - `packages/db/src/schema/heartbeat_runs.ts`: mirror the new index in the Drizzle schema. - `server/src/services/productivity-review.ts`: select only the seven run fields the evidence code reads. Before, the query pulled full rows with `result_json` (up to 43 kB per row, 100 rows per issue). - `server/src/services/attention.ts`: project `issueId`/`taskId` text fields instead of the full `context_snapshot` in the failed-run newer-runs query (29 MB per feed build before). - `server/src/index.ts`: the retention sweep now builds the attention feed once per company with `all: true` instead of one full rebuild per cursor page. - `packages/db/src/heartbeat-context-snapshot-index-migration.test.ts`: cover the new index and re-run migration 0210 statements to prove idempotency. ## Verification - `pnpm --filter @paperclipai/db typecheck` (includes migration numbering and safety checks) — pass. - `npx tsc --noEmit` in `server/` — pass. - `npx vitest run packages/db/src/heartbeat-context-snapshot-index-migration.test.ts` — pass (embedded Postgres, full migration chain, planner assertions, idempotent re-run of 0209 and 0210). - `npx vitest run` on attention, dashboard, productivity-review, decision-retention, issue-blocker-attention, and issue-review-attention test files — 72/72 pass. - Live EXPLAIN ANALYZE before/after numbers are in the What Changed list. ## Risks - Migration 0210 builds one btree index without CONCURRENTLY inside the transactional migration runner. The table is not in the large-table bucket. The 0209 twin built in seconds on a 100k-row live table. - The CTE bound excludes retry ancestors that are older than the chart window. Those rows are not visible to the chart query, so chart output does not change. - The attention projection changes JSONB scalar handling in one edge case: a non-string `issueId`/`taskId` value now casts to text instead of reading as absent. These keys are always strings in practice. - The retention sweep now holds one full feed in memory per company. The cursor loop already accumulated all pages into one array, so peak memory is unchanged. ## Model Used Claude Fable 5 (`claude-fable-5`, Anthropic, Mythos-class tier, extended thinking + tool use) via Paperclip agent runtime. - [x] I searched existing PRs and issues and this change is not a duplicate. Co-authored-by: Claude Fable 5 <noreply@anthropic.com>
214 lines
7.9 KiB
TypeScript
214 lines
7.9 KiB
TypeScript
import { and, eq, gte, sql } from "drizzle-orm";
|
|
import type { Db } from "@paperclipai/db";
|
|
import { agents, approvals, companies, costEvents, heartbeatRuns, issues } from "@paperclipai/db";
|
|
import { notFound } from "../errors.js";
|
|
import { budgetService } from "./budgets.js";
|
|
import { visibleIssueCondition } from "./issue-visibility.js";
|
|
|
|
const DASHBOARD_RUN_ACTIVITY_DAYS = 14;
|
|
|
|
function formatUtcDateKey(date: Date): string {
|
|
return date.toISOString().slice(0, 10);
|
|
}
|
|
|
|
export function getUtcMonthStart(date: Date): Date {
|
|
return new Date(Date.UTC(date.getUTCFullYear(), date.getUTCMonth(), 1));
|
|
}
|
|
|
|
function getRecentUtcDateKeys(now: Date, days: number): string[] {
|
|
const todayUtc = Date.UTC(now.getUTCFullYear(), now.getUTCMonth(), now.getUTCDate());
|
|
return Array.from({ length: days }, (_, index) => {
|
|
const dayOffset = index - (days - 1);
|
|
return formatUtcDateKey(new Date(todayUtc + dayOffset * 24 * 60 * 60 * 1000));
|
|
});
|
|
}
|
|
|
|
export function dashboardService(db: Db) {
|
|
const budgets = budgetService(db);
|
|
return {
|
|
summary: async (companyId: string) => {
|
|
const company = await db
|
|
.select()
|
|
.from(companies)
|
|
.where(eq(companies.id, companyId))
|
|
.then((rows) => rows[0] ?? null);
|
|
|
|
if (!company) throw notFound("Company not found");
|
|
|
|
const agentRows = await db
|
|
.select({ status: agents.status, count: sql<number>`count(*)` })
|
|
.from(agents)
|
|
.where(eq(agents.companyId, companyId))
|
|
.groupBy(agents.status);
|
|
|
|
const taskRows = await db
|
|
.select({ status: issues.status, count: sql<number>`count(*)` })
|
|
.from(issues)
|
|
.where(and(eq(issues.companyId, companyId), visibleIssueCondition()))
|
|
.groupBy(issues.status);
|
|
|
|
const pendingApprovals = await db
|
|
.select({ count: sql<number>`count(*)` })
|
|
.from(approvals)
|
|
.where(and(eq(approvals.companyId, companyId), eq(approvals.status, "pending")))
|
|
.then((rows) => Number(rows[0]?.count ?? 0));
|
|
|
|
const agentCounts: Record<string, number> = {
|
|
active: 0,
|
|
running: 0,
|
|
paused: 0,
|
|
error: 0,
|
|
};
|
|
for (const row of agentRows) {
|
|
const count = Number(row.count);
|
|
// "idle" agents are operational — count them as active
|
|
const bucket = row.status === "idle" ? "active" : row.status;
|
|
agentCounts[bucket] = (agentCounts[bucket] ?? 0) + count;
|
|
}
|
|
|
|
const taskCounts: Record<string, number> = {
|
|
open: 0,
|
|
inProgress: 0,
|
|
blocked: 0,
|
|
done: 0,
|
|
};
|
|
for (const row of taskRows) {
|
|
const count = Number(row.count);
|
|
if (row.status === "in_progress") taskCounts.inProgress += count;
|
|
if (row.status === "blocked") taskCounts.blocked += count;
|
|
if (row.status === "done") taskCounts.done += count;
|
|
if (row.status !== "done" && row.status !== "cancelled") taskCounts.open += count;
|
|
}
|
|
|
|
const now = new Date();
|
|
const monthStart = getUtcMonthStart(now);
|
|
const runActivityDays = getRecentUtcDateKeys(now, DASHBOARD_RUN_ACTIVITY_DAYS);
|
|
const runActivityStart = new Date(`${runActivityDays[0]}T00:00:00.000Z`);
|
|
const [{ monthSpend }] = await db
|
|
.select({
|
|
monthSpend: sql<number>`coalesce(sum(${costEvents.costCents}), 0)::double precision`,
|
|
})
|
|
.from(costEvents)
|
|
.where(
|
|
and(
|
|
eq(costEvents.companyId, companyId),
|
|
gte(costEvents.occurredAt, monthStart),
|
|
),
|
|
);
|
|
|
|
const monthSpendCents = Number(monthSpend);
|
|
// Per-day run breakdown. A run is "recovered" when its retry chain later
|
|
// succeeded (recovered_runs = all ancestors of a succeeded retry), so a
|
|
// restart-killed run whose retry succeeded is pulled out of the headline
|
|
// failed count. error_code is carried through so a failure spike can be
|
|
// attributed to an error class (e.g. process_lost, provider_quota).
|
|
// Both recursive arms are bounded to the chart window: a retry is always
|
|
// created after the run it retries, so ancestors of an out-of-window
|
|
// child are themselves out of window and invisible to the membership
|
|
// test below. Unbounded, the seed walks every run the company ever had.
|
|
const runActivityRows = (await db.execute(sql`
|
|
WITH RECURSIVE recovered_runs(id) AS (
|
|
SELECT parent.id
|
|
FROM ${heartbeatRuns} AS child
|
|
JOIN ${heartbeatRuns} AS parent ON parent.id = child.retry_of_run_id
|
|
WHERE child.company_id = ${companyId}
|
|
AND child.status = 'succeeded'
|
|
AND child.created_at >= ${runActivityStart.toISOString()}::timestamptz
|
|
UNION
|
|
SELECT parent.id
|
|
FROM recovered_runs rr
|
|
JOIN ${heartbeatRuns} AS child ON child.id = rr.id
|
|
JOIN ${heartbeatRuns} AS parent ON parent.id = child.retry_of_run_id
|
|
WHERE child.created_at >= ${runActivityStart.toISOString()}::timestamptz
|
|
)
|
|
SELECT
|
|
to_char(run.created_at AT TIME ZONE 'UTC', 'YYYY-MM-DD') AS date,
|
|
run.status AS status,
|
|
run.error_code AS error_code,
|
|
(run.id IN (SELECT id FROM recovered_runs)) AS recovered,
|
|
count(*)::double precision AS count
|
|
FROM ${heartbeatRuns} AS run
|
|
WHERE run.company_id = ${companyId}
|
|
AND run.created_at >= ${runActivityStart.toISOString()}::timestamptz
|
|
GROUP BY date, run.status, run.error_code, recovered
|
|
`)) as unknown as Iterable<{
|
|
date: string;
|
|
status: string;
|
|
error_code: string | null;
|
|
recovered: boolean | string;
|
|
count: number | string;
|
|
}>;
|
|
|
|
const runActivity = new Map(
|
|
runActivityDays.map((date) => [
|
|
date,
|
|
{
|
|
date,
|
|
succeeded: 0,
|
|
failed: 0,
|
|
recovered: 0,
|
|
other: 0,
|
|
total: 0,
|
|
failedByErrorCode: {} as Record<string, number>,
|
|
},
|
|
]),
|
|
);
|
|
for (const row of runActivityRows) {
|
|
const bucket = runActivity.get(String(row.date));
|
|
if (!bucket) continue;
|
|
const count = Number(row.count);
|
|
const status = String(row.status);
|
|
// Postgres booleans can arrive as JS boolean or "t"/"true" depending on driver.
|
|
const recovered = row.recovered === true || row.recovered === "t" || row.recovered === "true";
|
|
if (status === "succeeded") {
|
|
bucket.succeeded += count;
|
|
} else if (status === "failed" || status === "timed_out") {
|
|
if (recovered) {
|
|
bucket.recovered += count;
|
|
} else {
|
|
bucket.failed += count;
|
|
const code =
|
|
typeof row.error_code === "string" && row.error_code.length > 0
|
|
? row.error_code
|
|
: "unknown";
|
|
bucket.failedByErrorCode[code] = (bucket.failedByErrorCode[code] ?? 0) + count;
|
|
}
|
|
} else {
|
|
bucket.other += count;
|
|
}
|
|
bucket.total += count;
|
|
}
|
|
|
|
const utilization =
|
|
company.budgetMonthlyCents > 0
|
|
? (monthSpendCents / company.budgetMonthlyCents) * 100
|
|
: 0;
|
|
const budgetOverview = await budgets.overview(companyId);
|
|
|
|
return {
|
|
companyId,
|
|
agents: {
|
|
active: agentCounts.active,
|
|
running: agentCounts.running,
|
|
paused: agentCounts.paused,
|
|
error: agentCounts.error,
|
|
},
|
|
tasks: taskCounts,
|
|
costs: {
|
|
monthSpendCents,
|
|
monthBudgetCents: company.budgetMonthlyCents,
|
|
monthUtilizationPercent: Number(utilization.toFixed(2)),
|
|
},
|
|
pendingApprovals,
|
|
budgets: {
|
|
activeIncidents: budgetOverview.activeIncidents.length,
|
|
pendingApprovals: budgetOverview.pendingApprovalCount,
|
|
pausedAgents: budgetOverview.pausedAgentCount,
|
|
pausedProjects: budgetOverview.pausedProjectCount,
|
|
},
|
|
runActivity: Array.from(runActivity.values()),
|
|
};
|
|
},
|
|
};
|
|
}
|