mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-06 10:48:12 +02:00
## Thinking Path > - Paperclip is the open source app that people use to manage AI agents for work. > - The runner E2E system verifies agent profiles across supported execution environments. > - Its generated report is the main surface for inspecting those results and their visual evidence. > - Expanded matcher content could change the matrix column widths and make comparisons difficult. > - Screenshots also required extra navigation, and the report did not have current-result search and filters. > - This pull request stabilizes the matrix layout and makes visual evidence directly browsable. > - The benefit is faster inspection of retained test evidence without another paid matrix run. ## Linked Issues or Issue Description **What happened?** The runner E2E report changed matrix column widths when a matcher table expanded. The report also made screenshot comparison and current-result discovery slower than necessary. **Expected behavior** The matrix columns must remain stable. Each retained screenshot must appear as a thumbnail. The gallery must support keyboard navigation and show the relevant execution metadata. The report must support client-side search and filters. **Steps to reproduce** 1. Open a runner E2E matrix report that contains retained screenshots. 2. Expand the matcher details in a matrix cell. 3. Observe the matrix column movement in the old report. **Paperclip version or commit** `8430bd897` **Deployment mode** Generated static runner E2E report. ## What Changed - Keep matrix and matcher table widths stable when details expand. - Show retained screenshot thumbnails in each test card. - Add a full-screen evidence gallery with mouse, keyboard, and swipe navigation. - Show agent, environment, runtime, status, duration, token, and matcher data in the gallery header. - Add client-side search and profile, environment, suite, and status filters below the report section tabs. - Keep the filters in normal document flow while the report tabs remain sticky. - Add report generator assertions for the new layout and controls. - Add Playwright coverage for filtering, stable matcher expansion, and filtered gallery navigation. ## Verification - `pnpm test:e2e:runner:typecheck` - `pnpm exec vitest run --config tests/runner-e2e/vitest.config.ts tests/runner-e2e/report.test.ts` - `PAPERCLIP_PLAYWRIGHT_CHANNEL=chrome pnpm exec playwright test --config tests/e2e/playwright.config.ts tests/e2e/runner-e2e-dashboard.spec.ts` - `node --test ./scripts/__tests__/e2e-shard.test.mjs` - `pnpm check:token-gates` - `pnpm -r typecheck` - `pnpm test:run` - `pnpm build` - Regenerated the report from GitHub Actions run `33963318820` without rerunning the matrix. - Verified 121 retained thumbnails across 66 results in a local browser. - Verified that expanded matchers keep matrix widths at 260, 486, and 486 pixels in a 1280-pixel viewport. - Verified search, filters, modal metadata, and arrow-key gallery navigation. ## Risks Low risk. This change only modifies the static runner E2E report generator and its tests. It does not change runner execution or retained evidence data. > This work is focused report polish. It does not duplicate planned core work in `ROADMAP.md`. ## Model Used OpenAI Codex with `gpt-5.6-sol`. Codex desktop managed the context window. The model used reasoning, filesystem tools, code execution, GitHub CLI access, and in-app browser verification. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [ ] All Paperclip CI gates are green - [ ] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip <noreply@paperclip.ing>
165 lines
5.1 KiB
TypeScript
165 lines
5.1 KiB
TypeScript
import { expect, test } from "@playwright/test";
|
|
|
|
import { runnerExecutionById } from "../runner-e2e/catalog.js";
|
|
import {
|
|
renderRunnerE2EDashboard,
|
|
type RunnerDashboardEntry,
|
|
} from "../runner-e2e/dashboard.js";
|
|
import type { MatrixExecution, RunnerE2EResult } from "../runner-e2e/types.js";
|
|
|
|
const executionIds = [
|
|
"core-compatibility.legacy-codex.local.message-marker",
|
|
"core-compatibility.legacy-codex.daytona.message-marker",
|
|
"core-compatibility.legacy-claude.local.message-marker",
|
|
"local-session-integrity.runner-acpx-codex.local.structured-question-restart-resume",
|
|
] as const;
|
|
|
|
function resultFor(
|
|
execution: MatrixExecution,
|
|
status: RunnerE2EResult["status"],
|
|
): RunnerDashboardEntry {
|
|
const result: RunnerE2EResult = {
|
|
schema: "paperclip.runner-e2e.result/v1",
|
|
executionId: execution.id,
|
|
suiteId: execution.suite.id,
|
|
attempt: 1,
|
|
status,
|
|
profileId: execution.profile.id,
|
|
environmentId: execution.environment.id,
|
|
caseId: execution.task.id,
|
|
provider: execution.profile.provider,
|
|
model: execution.profile.model,
|
|
runtimeMode: execution.profile.expectedRuntimeMode,
|
|
startedAt: "2026-09-05T12:00:00.000Z",
|
|
finishedAt: "2026-09-05T12:00:12.000Z",
|
|
durationMs: 12_000,
|
|
runIds: [`run-${execution.id}`],
|
|
usage: {
|
|
inputTokens: 1_250,
|
|
outputTokens: 75,
|
|
cachedInputTokens: 500,
|
|
},
|
|
matcherResults: [
|
|
{
|
|
matcher: { kind: "message_contains", expected: "PAPERCLIP_E2E_OK" },
|
|
passed: status === "passed",
|
|
detail: status === "passed" ? "matched" : "marker missing",
|
|
},
|
|
],
|
|
screenshots: [
|
|
{
|
|
id: "final-state",
|
|
label: "Final visible task state",
|
|
file: "final-state.png",
|
|
},
|
|
],
|
|
cleanup: "passed",
|
|
};
|
|
|
|
return {
|
|
result,
|
|
valid: status === "passed",
|
|
errors: status === "passed" ? [] : ["fixture failure"],
|
|
evidenceBaseHref: `evidence/${execution.id}/attempt-1`,
|
|
evidenceFiles: ["final-state.png"],
|
|
};
|
|
}
|
|
|
|
test("filters matrix results and pages through the filtered screenshot gallery", async ({
|
|
page,
|
|
}) => {
|
|
const catalog = executionIds.map(runnerExecutionById);
|
|
const entries = catalog.map((execution, index) =>
|
|
resultFor(execution, index === 1 ? "failed" : "passed"),
|
|
);
|
|
await page.setContent(
|
|
renderRunnerE2EDashboard({
|
|
title: "Runner E2E interaction fixture",
|
|
generatedAt: "2026-09-05T12:01:00.000Z",
|
|
expected: executionIds,
|
|
catalog,
|
|
entries,
|
|
}),
|
|
);
|
|
|
|
const filters = page
|
|
.locator("[data-report-query]")
|
|
.locator("..")
|
|
.locator("..");
|
|
const tabs = page.locator(".suite-nav");
|
|
await expect(page.locator("[data-report-result]")).toHaveText(
|
|
"4 of 4 tests shown",
|
|
);
|
|
await expect(page.locator("[data-gallery-item]")).toHaveCount(4);
|
|
expect(
|
|
await filters.evaluate((element) => getComputedStyle(element).position),
|
|
).toBe("static");
|
|
expect(
|
|
await tabs.evaluate((element) =>
|
|
element.nextElementSibling?.classList.contains("report-filters"),
|
|
),
|
|
).toBe(true);
|
|
|
|
const firstTable = page.locator(".matrix").first();
|
|
const widthsBefore = await firstTable
|
|
.locator("thead th")
|
|
.evaluateAll((cells) =>
|
|
cells.map((cell) => cell.getBoundingClientRect().width),
|
|
);
|
|
await firstTable.locator("details summary").first().click();
|
|
const widthsAfter = await firstTable
|
|
.locator("thead th")
|
|
.evaluateAll((cells) =>
|
|
cells.map((cell) => cell.getBoundingClientRect().width),
|
|
);
|
|
expect(widthsAfter).toEqual(widthsBefore);
|
|
|
|
await page.locator("[data-report-query]").fill("Legacy Claude");
|
|
await expect(page.locator("[data-report-result]")).toHaveText(
|
|
"1 of 4 tests shown",
|
|
);
|
|
await expect(page.locator("[data-gallery-open]")).toHaveText(
|
|
"View gallery · 1",
|
|
);
|
|
|
|
await page.locator("[data-report-reset]").click();
|
|
await expect(page.locator("[data-report-query]")).toBeFocused();
|
|
await page
|
|
.locator("select[data-report-profile]")
|
|
.selectOption("legacy-codex");
|
|
await expect(page.locator("[data-report-result]")).toHaveText(
|
|
"2 of 4 tests shown",
|
|
);
|
|
await expect(page.locator("[data-gallery-open]")).toHaveText(
|
|
"View gallery · 2",
|
|
);
|
|
|
|
await page.locator("[data-gallery-open]").click();
|
|
const dialog = page.locator("[data-gallery-dialog]");
|
|
await expect(dialog).toBeVisible();
|
|
await expect(dialog.locator("[data-gallery-profile]")).toHaveText(
|
|
"Legacy Codex",
|
|
);
|
|
await expect(dialog.locator("[data-gallery-environment]")).toHaveText(
|
|
"Isolated local",
|
|
);
|
|
await expect(dialog.locator("[data-gallery-position]")).toHaveText(
|
|
"Image 1 of 2",
|
|
);
|
|
|
|
await page.keyboard.press("ArrowRight");
|
|
await expect(dialog.locator("[data-gallery-environment]")).toHaveText(
|
|
"Daytona sandbox",
|
|
);
|
|
await expect(dialog.locator("[data-gallery-status]")).toHaveText("failed");
|
|
await expect(dialog.locator("[data-gallery-position]")).toHaveText(
|
|
"Image 2 of 2",
|
|
);
|
|
|
|
await page.keyboard.press("Escape");
|
|
await expect(dialog).not.toBeVisible();
|
|
await page.locator("[data-report-reset]").click();
|
|
await page.keyboard.press("/");
|
|
await expect(page.locator("[data-report-query]")).toBeFocused();
|
|
});
|