mirror of
https://github.com/paperclipai/paperclip.git
synced 2026-10-07 16:11:46 +02:00
## Thinking Path > - Paperclip manages AI agents and their work. > - Native Runner agents report completion through finish and block tools. > - The providers receive different descriptions for those tools. > - Completion guidance belongs with the tools that enforce the result. > - This pull request shares the descriptions and refreshes retained catalogs. > - A separate native suite checks completion and blocking on production defaults. > - Legacy agents retain their separate skill and API paths. ## Linked Issues or Issue Description Refs: #14920, #14948, #14985. **Current behavior** Native Codex and MCP bridges describe finish and block differently. Retained provider sessions can keep old descriptions. **Proposed behavior** Native providers receive the same finish and block descriptions. The descriptions cover report selection, validation feedback, returned outcomes, approval gates and final-answer timing. Retained native sessions refresh from v13 to v14. **Reason and benefit** Put the completion procedure next to its native tool. Preserve stock base instructions, schemas, permissions and terminal semantics. This PR now stands alone on master. It contains no reduced manual, shared prompt or operational-skill changes from #14948. ## What Changed - Add canonical native finish and block descriptions. Use them in direct Codex and both native MCP bridges. - Advance the native tool contract to v14. Cover old-v13 refresh without replacing task identity or prior history. - Check authenticated tool catalogs, provider start/resume frames and serialized daemon catalogs. - Add an independent, explicit-only native completion suite. Preserve the original assigned-skill durable-document journey. Pair it with a concrete whole-task blocker across Codex, ACPX Claude and OpenCode. - Verify the actual public production default bundle and budgets before execution. Require independent durable disposition, native result/terminal receipts and observable provider-final ordering. - Correct the blocker browser oracle to accept the requested explanation. Keep exact owner/action/scope checks. Calibrate positive, missing and contradictory replies. - Preserve only actual `tool_call` terminal names (`paperclip_finish` / `paperclip_block`) in the native compatibility run-log projection. Require the same named call ID through its finishing result; retain all other redaction boundaries. - Admit verified hosted shallow checkout/build hydration and bind the selected runnerd to exact source/archive/binary provenance. Hosted cells truthfully reuse the existing trusted build; local admission executes Rust calibration. Forward only public source/run identifiers through both launcher preflight subprocess paths. - Enforce single attempts in the launcher for opted-in fixtures. Keep ordinary retry policy unchanged. Run exact-source, credential-free admission before credential loading. ## Verification - Frozen candidate: `d6e59e4712a3158ab4cd7d58deff1389b4578c21`, based on master `59c07ede72dc08b8aba149a01cc11e0b7a204621`; historical descriptions: `e74ed61a69fbdd8b3a8f15dd6456bc3140246e33`. Exactly the five original native production files and six unit tests differ. Both carry identical corrected fixtures, strict named finishing-call grader, closed compatibility carrier and admission. Defaults, profiles/models/auth/permissions and manifest bytes match. - Actual launcher `prepareNativeCompletionPreflight` → `verifyNativeCompletionPreflight` admission passes on both exact refs with zero providers: candidate 132 / historical 127 selected TypeScript assertions, 128 Node calibrations and one Rust normalization calibration each; E2E typecheck, manifest checks, selected binary provenance and six-cell discovery pass. Each has 257 explicitly skipped unrelated assertions, not coverage. The credential-free environment calibration exercises both real prepare/verify subprocess options with public hosted identifiers and rejects credential/ambient overrides. Complete actual launcher prepare→verify also passes on both frozen refs with explicitly synthetic hosted metadata/verified archives, separately labeled as calibration rather than a trusted GitHub run. Exact framed provenance parsing and mock source identity are calibrated without relaxing the real verifier. - [Complete matched qualification report](https://github.com/paperclipai/paperclip/blob/532066620b88e8731a5211fbe1cbc48ce8c7dd1a/doc/plans/2026-10-02-native-completion-master-qualification.md), [immutable manifest](https://github.com/paperclipai/paperclip/blob/532066620b88e8731a5211fbe1cbc48ce8c7dd1a/doc/plans/2026-10-02-native-completion-calibrated-manifest.json) and [closed retained audit/hashes](https://github.com/paperclipai/paperclip/blob/532066620b88e8731a5211fbe1cbc48ce8c7dd1a/doc/plans/2026-10-02-native-completion-calibrated-results/comparison.json) are inspectable. All six candidate cells pass; historical descriptions pass five. Paired outcomes: **zero new failures, one new pass (Codex blocker), five unchanged passes, zero pending pairs**. [Candidate campaign](https://github.com/paperclipai/paperclip/actions/runs/37098728980) and [historical campaign](https://github.com/paperclipai/paperclip/actions/runs/37098815696) each execute six original attempt-1 native runs, with no campaign retry and successful cleanup. Their trusted workflow revision is `215586d127e97c9301d86e769a39a15c13298ca2`, separate from measured source. [Candidate public HTML](https://d1p6rlowie26tp.cloudfront.net/runner-e2e/campaigns/gha-37098728980-1/index.html) and [historical public HTML](https://d1p6rlowie26tp.cloudfront.net/runner-e2e/campaigns/gha-37098815696-1/index.html) retain declared screenshots. - Independent candidate evidence agrees with all original grades: 51 strict native checks, 12 served-default/budget checks and 21 original skill/document checks pass. The historical Codex blocker saves the correct whole-task blocker but omits the required marker from its actual provider final and identical saved reply. This is not semantic-summary fallback. Its original browser/matcher failure stays retained; the additional native snapshot/grade and workspace before/after digest were never written and are not fabricated by the separate API/PRP audit. Historical Codex completion has one failed finish followed by success within the same native run; the public receipt records no failure reason. All twelve runs and their usage remain counted. Reported model-cost subtotals are $0.00421482 historical/$0.00437391 candidate; Codex/Claude zero entries have unknown billing type, actual invoices are unverified and hosted execution cost is unmetered. One matched trial supports no extra failure within these six cases, not broad statistical or coding-quality equivalence. - Initial hosted `e18c2cf9` / `459455ac` and subsequent `0a9c5a7` / `00a761b` cohorts each stopped before providers in all twelve cells. The latter failed a mocked-receipt unit test under ambient hosted metadata; all source/build proofs passed. [All twelve later setup receipts](https://github.com/paperclipai/paperclip/blob/402ee94c52273ad58de355ae9a7d562dd22f8101/doc/plans/2026-10-02-native-completion-qualified-hosted-setup.json) are retained. [Exact failed setup receipts](https://github.com/paperclipai/paperclip/blob/27653eb1a8f8ce839776d760f4563f672e5a706c/doc/plans/2026-10-02-native-completion-master-hosted-setup.json) and the original manifest remain intact. Local sandbox-denied loopback and stale anchor-expectation attempts are retained separately; unchanged appropriate assertions were corrected/admitted before paid dispatch. Old anonymous OpenCode streams are not assigned inferred tool names or retroactively passed. - Full provider-free E2E support previously passed 927 tests in 67 files. Exact-head d6 normal CI run `37098409915`, attempt 1 passes full repository typecheck/build/tests, Runner Rust/static checks, all browser shards/aggregate and canary: 52 check-runs pass, four intentional skips, Snyk passes. Fresh Greptile check `111132956342` is 5/5 with zero unresolved threads. Source-specific deterministic tests do not substitute for the bounded live comparison. - Earlier native source `9138f570c341c251a5727c32d6615ce238bc8e03` is archived. Its [complete reduced-manual-context report](https://github.com/paperclipai/paperclip/blob/9138f570c341c251a5727c32d6615ce238bc8e03/doc/plans/2026-10-02-native-completion-live-comparison.md) remains intact, including original failures, grader limits and provider-free replay. It is not current-master-context qualification. ## Risks Changed tool text can change model behavior. The completed six-pair qualification shows no extra failing outcomes in this bounded trial; other tasks and repeated-run variance remain unmeasured. Observable final ordering does not prove provider feedback consumption. Public evidence can fail closed if a provider does not expose the required result sequence. This slice does not remove native fixed prompts or measure general coding quality. No database, schema, permission or legacy completion changes occur. ## Model Used OpenAI Codex, GPT-6 family, with code inspection, execution and tool use. The exact deployment ID and context-window size are not exposed in this session. They are unavailable rather than inferred from the model menu. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [x] All Paperclip CI gates are green - [x] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip <noreply@paperclip.ing>
439 lines
12 KiB
TypeScript
439 lines
12 KiB
TypeScript
export const CREDENTIAL_NAMES = [
|
|
"OPENAI_API_KEY",
|
|
"ANTHROPIC_API_KEY",
|
|
"OPENROUTER_API_KEY",
|
|
"KIMI_MODEL_API_KEY",
|
|
"XAI_API_KEY",
|
|
"GROK_AUTH_JSON",
|
|
"DAYTONA_API_KEY",
|
|
"CURSOR_AUTH_TOKEN",
|
|
"COPILOT_GITHUB_TOKEN",
|
|
] as const;
|
|
|
|
export type CredentialName = (typeof CREDENTIAL_NAMES)[number];
|
|
export type RunnerGeneration = "legacy" | "native";
|
|
export type RunnerEnvironmentId = "local" | "daytona";
|
|
export type RunnerTaskWorkMode = "standard" | "planning" | "ask";
|
|
export type RunnerTaskFlow =
|
|
| "blocker_guidance"
|
|
| "everyday_workflow"
|
|
| "context_integrity"
|
|
|
|
| "continuation_accounting"
|
|
| "continuation"
|
|
| "first_task"
|
|
| "agent_chat"
|
|
| "governed_tool_review"
|
|
| "single_turn"
|
|
| "plan_revision_acceptance"
|
|
| "question_resume_completion"
|
|
| "plan_approval_completion"
|
|
| "warm_three_turn"
|
|
| "instruction_persistence";
|
|
|
|
export interface SecretReference {
|
|
type: "secret_ref";
|
|
secretId: string;
|
|
version: "latest";
|
|
}
|
|
|
|
export type SecretReferenceMap = Partial<
|
|
Record<CredentialName, SecretReference>
|
|
>;
|
|
|
|
export interface AgentFixtureBuildInput {
|
|
environmentId: string;
|
|
environmentFixtureId: RunnerEnvironmentId;
|
|
workspacePath: string;
|
|
secretRefs: SecretReferenceMap;
|
|
executionId: string;
|
|
}
|
|
|
|
export interface EnvironmentFixtureBuildInput {
|
|
secretRefs: SecretReferenceMap;
|
|
daytonaImage?: string;
|
|
executionId: string;
|
|
}
|
|
|
|
export interface RunnerProfileFixture {
|
|
id: string;
|
|
label: string;
|
|
generation: RunnerGeneration;
|
|
groups: readonly string[];
|
|
adapterType: string;
|
|
provider: string;
|
|
model: string;
|
|
modelQualification: {
|
|
source:
|
|
| "adapter_constant"
|
|
| "qualified_runner_profile"
|
|
| "candidate_runner_profile"
|
|
| "openrouter_rankings_snapshot";
|
|
qualificationId: string;
|
|
};
|
|
qualificationCandidate?: "cursor" | "copilot" | "pi";
|
|
ranking?: {
|
|
rank: number;
|
|
canonicalModelId: string;
|
|
snapshotId: string;
|
|
capturedAt: string;
|
|
sourceUrl: string;
|
|
};
|
|
credential: Exclude<CredentialName, "DAYTONA_API_KEY">;
|
|
supportedEnvironments: readonly RunnerEnvironmentId[];
|
|
expectedRuntimeMode: RunnerGeneration;
|
|
expectedRuntimeMetadata: {
|
|
adapterType: string;
|
|
provider: string;
|
|
};
|
|
buildAgent(input: AgentFixtureBuildInput): Record<string, unknown>;
|
|
}
|
|
|
|
export interface EnvironmentFixture {
|
|
id: RunnerEnvironmentId;
|
|
/** Distinguishes materially different configurations that share a provider ID. */
|
|
configurationKey?: string;
|
|
label: string;
|
|
groups: readonly string[];
|
|
driver: "local" | "sandbox";
|
|
provider: "local" | "daytona";
|
|
credential?: "DAYTONA_API_KEY";
|
|
lifecycle: {
|
|
setup: "instance_managed" | "create_via_api";
|
|
probe: "run_context_via_api";
|
|
cleanup: "instance_shutdown" | "delete_via_api_and_destroy_leases";
|
|
};
|
|
expectedExecutionTarget: {
|
|
kind: "local" | "remote";
|
|
transport?: "sandbox";
|
|
};
|
|
buildEnvironment(
|
|
input: EnvironmentFixtureBuildInput,
|
|
): Record<string, unknown>;
|
|
}
|
|
|
|
export type Matcher =
|
|
| { kind: "message_exact"; expected: string }
|
|
| { kind: "message_contains"; expected: string }
|
|
| { kind: "message_occurrences"; expected: string; count: number }
|
|
| { kind: "message_regex"; pattern: string; flags?: string }
|
|
| { kind: "message_ordered"; expected: readonly string[] }
|
|
| { kind: "issue_status"; expected: string }
|
|
| { kind: "run_status"; expected: string }
|
|
| { kind: "runtime_mode"; expected: RunnerGeneration }
|
|
| { kind: "environment"; expected: RunnerEnvironmentId }
|
|
| { kind: "file_exists"; path: string }
|
|
| { kind: "file_exact"; path: string; expected: string }
|
|
| { kind: "file_contains"; path: string; expected: string }
|
|
| { kind: "artifact_exists"; name: string; mimeType?: string }
|
|
| { kind: "json_path"; path: string; expected: unknown }
|
|
| { kind: "json_schema"; schema: Record<string, unknown> };
|
|
|
|
export interface RunnerTaskFixture {
|
|
id: string;
|
|
label: string;
|
|
groups: readonly string[];
|
|
workMode: RunnerTaskWorkMode;
|
|
flow: RunnerTaskFlow;
|
|
expectedRunCount: number;
|
|
/** Optional lower bound; expectedRunCount remains the maximum/cost estimate. */
|
|
minimumExpectedRunCount?: number;
|
|
/** Admit only the first attempt, including provider or infrastructure failures. */
|
|
automaticRetryPolicy?: "single_attempt";
|
|
attemptTimeoutMs: Readonly<Record<RunnerEnvironmentId, number>>;
|
|
expectedTerminalState: {
|
|
issue: "done" | "in_review" | "blocked" | "in_progress";
|
|
run: "succeeded" | "failed" | "cancelled";
|
|
};
|
|
buildTitle(nonce: string): string;
|
|
buildPrompt(nonce: string): string;
|
|
buildVisibleMarker(nonce: string): string;
|
|
buildRevisionRequest?(nonce: string): string;
|
|
buildFollowupMessages?(nonce: string): readonly [string, string];
|
|
turnTimeoutMs?: number;
|
|
buildQuestionAnswer?(nonce: string): {
|
|
optionLabel: string;
|
|
expectedMarker: string;
|
|
};
|
|
/** Restart the isolated Paperclip server after the waiting turn settles. */
|
|
restartServerBeforeQuestionAnswer?: boolean;
|
|
toolReviewDecision?: "approve" | "decline" | "always" | "restart";
|
|
buildPlanMarkers?(nonce: string): {
|
|
draft: string;
|
|
revised: string;
|
|
};
|
|
buildMatchers(nonce: string, execution: MatrixExecution): readonly Matcher[];
|
|
}
|
|
|
|
export interface MatrixExecution {
|
|
id: string;
|
|
suite: RunnerSuiteFixture;
|
|
suiteDefinitionHash: string;
|
|
profile: RunnerProfileFixture;
|
|
environment: EnvironmentFixture;
|
|
task: RunnerTaskFixture;
|
|
groups: readonly string[];
|
|
requiredCredentials: readonly CredentialName[];
|
|
}
|
|
|
|
export interface RunnerSuiteFixture {
|
|
id: string;
|
|
label: string;
|
|
description: string;
|
|
groups: readonly string[];
|
|
profiles: readonly RunnerProfileFixture[];
|
|
environments: readonly EnvironmentFixture[];
|
|
tasks: readonly RunnerTaskFixture[];
|
|
excludedExecutionIds?: readonly string[];
|
|
expectedMatrixSize: number;
|
|
definitionMetadata?: Readonly<Record<string, unknown>>;
|
|
/** Requires an explicit suite or execution ID; excluded from scheduled --all. */
|
|
manualOnly?: boolean;
|
|
}
|
|
|
|
export interface MatrixJob {
|
|
executionId: string;
|
|
suiteId: string;
|
|
profileId: string;
|
|
credentialName: Exclude<CredentialName, "DAYTONA_API_KEY">;
|
|
environmentId: RunnerEnvironmentId;
|
|
caseId: string;
|
|
timeoutMinutes: number;
|
|
needsDaytona: boolean;
|
|
}
|
|
|
|
export type FailureClass =
|
|
| "candidate_failure"
|
|
| "provider_variance"
|
|
| "transient_infrastructure"
|
|
| "permanent_infrastructure"
|
|
| "secret_leak"
|
|
| "cleanup_failure";
|
|
|
|
export type RunnerE2ECostStatus =
|
|
| "reported"
|
|
| "estimated"
|
|
| "partial"
|
|
| "unpriced"
|
|
| "unavailable"
|
|
| "not_metered";
|
|
|
|
export interface RunnerE2ERuntimeUsage {
|
|
provider: RunnerEnvironmentId;
|
|
/** Sum of the selected Paperclip heartbeat-run spans. */
|
|
agentRunDurationMs: number;
|
|
/** Sum of provider lease windows when the environment exposes leases. */
|
|
leaseDurationMs: number | null;
|
|
leaseCount: number;
|
|
cpuCores?: number;
|
|
memoryGiB?: number;
|
|
diskGiB?: number;
|
|
estimatedListCostUsd?: number;
|
|
costStatus: "estimated" | "unavailable" | "not_metered";
|
|
costSource:
|
|
| "daytona_public_list_price"
|
|
| "provider_cost_unavailable"
|
|
| "local_not_metered";
|
|
pricingAsOf?: string;
|
|
pricingUrl?: string;
|
|
}
|
|
|
|
export interface RunnerE2EBillingSummary {
|
|
llm: {
|
|
runCount: number;
|
|
runsWithTokenUsage: number;
|
|
runsWithReportedCost: number;
|
|
inputTokens: number;
|
|
outputTokens: number;
|
|
cachedInputTokens: number;
|
|
totalTokens: number;
|
|
reportedCostUsd: number;
|
|
costStatus: Exclude<RunnerE2ECostStatus, "estimated" | "not_metered">;
|
|
};
|
|
runtime: RunnerE2ERuntimeUsage;
|
|
/** Provider-reported model spend only; never includes unknown/unpriced runs. */
|
|
reportedCostUsd: number;
|
|
/** Public-list-price estimate for metered execution infrastructure. */
|
|
estimatedRuntimeCostUsd: number;
|
|
/** Separately recorded post-processing judge usage; absent when not judged. */
|
|
judge?: { inputTokens: number | null; outputTokens: number | null; estimatedCostUsd: number | null; reservedCostUsd: number };
|
|
/** Reported model subtotal plus runtime and judge list-price estimates. */
|
|
observedAndEstimatedCostUsd: number | null;
|
|
complete: boolean;
|
|
}
|
|
|
|
export interface RunnerE2EResult {
|
|
schema: "paperclip.runner-e2e.result/v1" | "paperclip.runner-e2e.result/v2";
|
|
executionId: string;
|
|
suiteId?: string;
|
|
suiteDefinitionHash?: string;
|
|
source?: {
|
|
sha: string | null;
|
|
ref: string | null;
|
|
workflowRunUrl: string | null;
|
|
};
|
|
rankingSnapshot?: {
|
|
snapshotId: string;
|
|
capturedAt: string;
|
|
sourceUrl: string;
|
|
rank: number;
|
|
canonicalModelId: string;
|
|
};
|
|
attempt: number;
|
|
status: "passed" | "failed";
|
|
failureClass?: FailureClass;
|
|
error?: string;
|
|
profileId: string;
|
|
environmentId: RunnerEnvironmentId;
|
|
caseId: string;
|
|
provider: string;
|
|
model: string;
|
|
runtimeMode: RunnerGeneration;
|
|
issueId?: string;
|
|
issueIdentifier?: string | null;
|
|
runIds?: string[];
|
|
turnTimings?: Array<{
|
|
turn: number;
|
|
submittedAt: string;
|
|
runStartedAt: string | null;
|
|
runFinishedAt: string | null;
|
|
schedulerLatencyMs: number | null;
|
|
runDurationMs: number | null;
|
|
responseLatencyMs: number | null;
|
|
runId: string;
|
|
leaseAcquisitionOutcome: "created" | "resumed" | "replacement" | "unknown";
|
|
}>;
|
|
startedAt: string;
|
|
finishedAt: string;
|
|
durationMs: number;
|
|
usage?: Record<string, unknown> | null;
|
|
runtimeUsage?: RunnerE2ERuntimeUsage;
|
|
billing?: RunnerE2EBillingSummary;
|
|
matcherResults?: Array<{
|
|
matcher: Matcher;
|
|
passed: boolean;
|
|
detail: string;
|
|
}>;
|
|
screenshots?: Array<{
|
|
id: string;
|
|
label: string;
|
|
file: string;
|
|
publication?: "public-runner-fixture";
|
|
/** Absent in historical results; new captures bind the exact PNG bytes. */
|
|
sha256?: string;
|
|
}>;
|
|
firstTask?: import("./first-task-scoring.js").FirstTaskEvidence;
|
|
completionQuality?: import("./completion-quality.js").CompletionQualityRecord[];
|
|
firstTaskQuality?: import("./first-task-quality.js").FirstTaskQuality;
|
|
cleanup: "not_started" | "passed" | "failed";
|
|
}
|
|
|
|
export interface RunnerE2ESuiteSummary {
|
|
suiteId: string;
|
|
suiteDefinitionHash: string;
|
|
expected: number;
|
|
selected: number;
|
|
executed: number;
|
|
passed: number;
|
|
failed: number;
|
|
incomplete?: number;
|
|
retries: number;
|
|
cleanupPassed: boolean;
|
|
complete: boolean;
|
|
durationMs: number;
|
|
billing: RunnerE2EAggregateBillingSummary;
|
|
}
|
|
|
|
export interface RunnerE2EJudgeBillingSummary {
|
|
attempts: number;
|
|
inputTokens: number;
|
|
outputTokens: number;
|
|
estimatedCostUsd: number | null;
|
|
reservedCostUsd: number;
|
|
attemptsWithUnknownUsage: number;
|
|
}
|
|
|
|
export interface RunnerE2EAggregateBillingSummary {
|
|
judge?: RunnerE2EJudgeBillingSummary;
|
|
testCount: number;
|
|
agentRunDurationMs: number;
|
|
leaseDurationMs: number;
|
|
llm: RunnerE2EBillingSummary["llm"];
|
|
reportedLlmCostUsd: number;
|
|
estimatedRuntimeCostUsd: number;
|
|
observedAndEstimatedCostUsd: number | null;
|
|
testsWithCompleteBilling: number;
|
|
}
|
|
|
|
export interface RunnerE2ECampaign {
|
|
schema: "paperclip.runner-e2e.campaign/v2";
|
|
campaignId: string;
|
|
generatedAt: string;
|
|
source: {
|
|
sha: string | null;
|
|
ref: string | null;
|
|
workflowRunUrl: string | null;
|
|
eventName: string | null;
|
|
};
|
|
expected: string[];
|
|
complete: boolean;
|
|
selected: number;
|
|
executed: number;
|
|
passed: number;
|
|
failed: number;
|
|
incomplete?: number;
|
|
retries: number;
|
|
cleanupPassed: boolean;
|
|
rankingSnapshots: Array<{
|
|
snapshotId: string;
|
|
capturedAt: string;
|
|
sourceUrl: string;
|
|
}>;
|
|
billing: RunnerE2EAggregateBillingSummary;
|
|
suites: RunnerE2ESuiteSummary[];
|
|
results: RunnerE2EResult[];
|
|
}
|
|
|
|
export interface RunnerE2EHistoryExecution {
|
|
executionId: string;
|
|
suiteId: string;
|
|
profileId: string;
|
|
environmentId: RunnerEnvironmentId;
|
|
caseId: string;
|
|
provider: string;
|
|
model: string;
|
|
status: "passed" | "failed" | "incomplete";
|
|
durationMs: number;
|
|
attempt: number;
|
|
cleanup: RunnerE2EResult["cleanup"];
|
|
billing: RunnerE2EBillingSummary;
|
|
}
|
|
|
|
export interface RunnerE2EHistoryCampaign {
|
|
campaignId: string;
|
|
generatedAt: string;
|
|
source: RunnerE2ECampaign["source"];
|
|
complete: boolean;
|
|
selected: number;
|
|
executed: number;
|
|
passed: number;
|
|
failed: number;
|
|
incomplete?: number;
|
|
retries: number;
|
|
cleanupPassed: boolean;
|
|
publicUrl: string;
|
|
billing: RunnerE2EAggregateBillingSummary;
|
|
suites: RunnerE2ESuiteSummary[];
|
|
executions: RunnerE2EHistoryExecution[];
|
|
}
|
|
|
|
export interface RunnerE2EHistoryIndex {
|
|
schema: "paperclip.runner-e2e.history/v1";
|
|
updatedAt: string;
|
|
latestCampaignId: string | null;
|
|
latestGreenCampaignId: string | null;
|
|
latestBySuite: Record<string, string>;
|
|
latestGreenBySuite: Record<string, string>;
|
|
campaigns: RunnerE2EHistoryCampaign[];
|
|
}
|