From cdf04a33fa04b2dad1236740d9fc86fb52c8e51b Mon Sep 17 00:00:00 2001 From: Devin Foley Date: Tue, 22 Sep 2026 16:56:00 -0700 Subject: [PATCH] feat(adapters): refresh current coding models and reasoning controls (#13829) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ## Thinking Path > - Paperclip is the open source app people use to manage AI agents for work. > - Its adapters supply model catalogs and reasoning controls to agent setup. > - Several provider releases are missing from the fallback catalogs. > - Some newer models also have effort levels that the UI does not offer. > - Operators need the exact supported IDs and controls when discovery is unavailable. > - This pull request updates the existing adapters from current provider documentation. > - Operators can select current coding models without entering custom IDs. ## Linked Issues or Issue Description **What existing behavior does this improve?** Model selection and reasoning controls across the existing coding-agent adapters. **Subsystem affected** Claude, Codex, Grok, Gemini, Cursor, Kimi, and OpenCode adapters; model discovery tests; agent creation and editing. **Current behavior** The catalogs omit Opus 5.5, GPT-6 Sol/Luna, Grok 4.7/4.6/4.5, current Gemini Flash models, and several Cursor/Kimi choices. Bedrock has obsolete IDs. The UI omits supported effort levels and saves Grok effort under a key the runtime does not read. **Proposed behavior** Offer verified current model IDs and model-specific efforts. Remove retired Gemini 2.0 choices. Keep configured defaults and saved model IDs. Keep runtime discovery for account-specific choices. **Reason and benefit** Catch up with provider releases through September 22, 2026. Correct the picker and runtime controls together. **Breaking changes** No database or API change. Gemini 2.0 options leave the picker after their June 1 shutdown. Existing saved IDs remain unchanged. Corrected Bedrock catalog IDs do not rewrite saved configuration. **Additional context** Supersedes the separate GPT-6 Sol PR #13830. Fable 5.1 was already merged in #12730, and GPT-6 Astra in #12851. The Grok 4.6/4.5 proposal #11324 was closed and parked by its author. This change retains the default-sentinel fix from #12062. Related discovery proposals #13127 and #13565 do not supply these catalog and effort updates. Searches found no open PR for the additional model IDs. See [the dated audit](https://github.com/paperclipai/paperclip/blob/feat/claude-opus-5-5/doc/adapter-model-audit-2026-09-22.md) for exact scope, primary sources, runtime observations, and account-specific limits. This updates existing adapters and does not duplicate planned core work. ## What Changed - Add Opus 5.5 for direct Claude and Bedrock, with a Claude Code 2.1.280 gate. Correct and extend Bedrock model IDs. - Add GPT-6 Sol/Luna and Fast mode. Offer Ultra for Astra/Sol and GPT-5.6 Sol/Terra, and Max for both Luna generations. - Add Grok 4.7/4.6/4.5, expose supported Extra High effort, and save Grok edits under `reasoningEffort`. - Add Gemini Flash 3.8/3.7/3.6/3.5, Flash Lite 3.5/3.1, and 3 Flash Preview. Remove retired 2.0 choices. - Add the current documented Cursor fallback models, including Fable 5.1, Composer 2.5, and Muse Spark 1.3. - Refresh OpenCode fallback IDs used in remote environments from its installed provider registry. - Add Kimi K3 256K. Update the existing coding alias to K2.8 Preview and enable its CLI effort settings. - Use model-specific Claude/Grok efforts in creation and editing. Clear unsupported effort when switching models. - Add catalog, CLI/ACP forwarding, compatibility, and UI persistence coverage. Record the audit and sources. ## Verification - Latest head `6e63c9ef53b54ba869cd4fb431a8570bebe289f4`: 53 CI checks passed, 2 skipped. This includes full workspace typecheck, build, and all test shards. Greptile is 5/5 with zero unresolved threads. GitHub reports no merge conflicts. - 340 focused tests passed across adapter metadata, CLI/ACP arguments, Claude version checks, Kimi effort, Grok execution, server model discovery, and UI effort selection/persistence. - `pnpm --filter @paperclipai/adapter-claude-local --filter @paperclipai/adapter-codex-local --filter @paperclipai/adapter-grok-local --filter @paperclipai/adapter-gemini-local --filter @paperclipai/adapter-kimi-local --filter @paperclipai/adapter-cursor-local --filter @paperclipai/adapter-opencode-local typecheck` — passed. The same filters with `build` passed. - `pnpm check:token-gates` and `git diff --check` — passed. - Full workspace and UI typechecks were attempted locally. They stop on existing missing `three` dependencies in `packages/shared/src/cliplab`. - Full `pnpm test:run` and `pnpm build` were not run locally. Worktree creation exhausted disk space, so a clean dependency install is not feasible on this host. Focused checks reuse existing dependencies. CI supplies full workspace verification. - No provider inference was run. Account-specific runtime model lists were inspected where available. - Manual check: select the new models in agent setup and editing. Confirm Luna has Max but no Ultra, Grok 4.7 has Extra High, and Fable 5.1 has Extra High/Max. Save Grok effort and confirm `adapterConfig.reasoningEffort` contains the selection. ## Risks - Catalog presence does not grant account access. Older CLIs and restricted accounts can reject a model. Opus 5.5 has an explicit upgrade check. - Higher effort can increase cost and latency. Existing agent defaults are unchanged. - Cursor fallback IDs come from public model documentation; the local account exposed no live catalog. Runtime discovery still adds account-specific variants. - Kimi effort remains supported only on its explicit CLI engine. This does not add effort support to its default ACP engine. - Saved obsolete Bedrock or retired Gemini IDs are not migrated automatically. ## Model Used OpenAI GPT-6 through Codex, with reasoning, tool use, and code execution. The exact deployment ID and context window are not exposed to this session. ## Checklist - [x] I have included a thinking path that traces from project context to this change - [x] I have specified the model used (with version and capability details) - [x] I have checked ROADMAP.md and confirmed this PR does not duplicate planned core work - [x] I have searched GitHub for duplicate or related PRs and linked them above - [x] I have either (a) linked existing issues with `Fixes: #` / `Closes #` / `Refs #` OR (b) described the issue in-PR following the relevant issue template - [x] I have not referenced internal/instance-local Paperclip issues or links (only public GitHub `#NNN` / `github.com/paperclipai/paperclip` URLs) - [x] My branch name describes the change (e.g. `docs/...`, `fix/...`) and contains no internal Paperclip ticket id or instance-derived details - [x] I have run tests locally and they pass - [x] I have added or updated tests where applicable - [x] I have updated relevant documentation to reflect my changes - [x] I have considered and documented any risks above - [x] All Paperclip CI gates are green - [x] Greptile is 5/5 with no open P2s, recommendations, or follow-ups - [x] I will address all Greptile and reviewer comments before requesting merge --------- Co-authored-by: Paperclip --- doc/adapter-model-audit-2026-09-22.md | 81 +++++++++++++++++++ .../adapters/claude-local/src/index.test.ts | 16 +++- packages/adapters/claude-local/src/index.ts | 14 +++- .../src/server/cli-capabilities.ts | 9 +-- .../src/server/execute.remote.test.ts | 23 +++--- .../claude-local/src/server/execute.ts | 2 +- .../claude-local/src/server/models.ts | 9 ++- .../src/server/test.probe.test.ts | 10 ++- .../src/server/test.remote.test.ts | 12 ++- .../adapters/claude-local/src/server/test.ts | 6 +- .../adapters/codex-local/src/index.test.ts | 19 ++++- packages/adapters/codex-local/src/index.ts | 34 ++++++-- .../codex-local/src/server/acp.test.ts | 10 +-- .../codex-local/src/server/codex-args.test.ts | 14 ++-- .../codex-local/src/ui/build-config.test.ts | 10 +-- packages/adapters/cursor-local/src/index.ts | 12 +++ packages/adapters/gemini-local/src/index.ts | 9 ++- packages/adapters/grok-local/src/index.ts | 11 ++- .../grok-local/src/server/execute.test.ts | 13 +++ packages/adapters/kimi-local/src/index.ts | 11 ++- .../kimi-local/src/server/execute.test.ts | 6 +- packages/adapters/opencode-local/src/index.ts | 12 +++ server/src/__tests__/adapter-models.test.ts | 49 +++++++++-- .../AgentConfigForm.render.test.tsx | 17 ++++ ui/src/components/AgentConfigForm.tsx | 15 +++- ui/src/components/agent-config-primitives.tsx | 2 +- ui/src/lib/agent-setup-fields.test.ts | 21 +++++ ui/src/lib/agent-setup-fields.ts | 6 +- ui/src/lib/codex-reasoning-effort.test.ts | 12 ++- 29 files changed, 379 insertions(+), 86 deletions(-) create mode 100644 doc/adapter-model-audit-2026-09-22.md create mode 100644 ui/src/lib/agent-setup-fields.test.ts diff --git a/doc/adapter-model-audit-2026-09-22.md b/doc/adapter-model-audit-2026-09-22.md new file mode 100644 index 0000000000..26fe550ce6 --- /dev/null +++ b/doc/adapter-model-audit-2026-09-22.md @@ -0,0 +1,81 @@ +# Adapter model audit: September 22, 2026 + +This audit covers model selection in the existing coding-agent adapters. Catalog +entries identify models; provider accounts and installed CLIs determine access. +No agent defaults or saved model selections are migrated. + +| Adapter | Changes from the audit | +| --- | --- | +| Claude Code | Add Opus 5.5 and require CLI 2.1.280. Fable 5.1, Fable 5, Sonnet 5, and Mythos 5 were already listed. Expose the documented model-specific effort levels in creation and editing. | +| Claude on Bedrock | Add Opus 5.5, Opus 5, Sonnet 5, Opus 4.7, and Sonnet 4.6. Correct the obsolete `-v1` suffix on Opus 4.8 and Fable 5. Apply the Opus 5.5 version check to Bedrock IDs too. | +| Codex and the Codex runner catalog | Add GPT-6 Sol and Luna, including Fast mode. Astra was already listed. Expose efforts through Ultra for Astra, Sol, and GPT-5.6 Sol/Terra; cap both Luna generations at Max. | +| Grok Build | Add Grok 4.7, 4.6, and 4.5. Offer Extra High for 4.7/4.6. Save edited effort as `reasoningEffort`, which the runtime consumes. Keep `grok-build` as the sentinel that lets the CLI choose its default. | +| Gemini CLI | Add Flash 3.8, 3.7, 3.6, 3.5, Flash Lite 3.5/3.1, and 3 Flash Preview. Remove the retired Gemini 2.0 choices. Keep Auto and the existing 3.1 Pro and 2.5 choices. | +| Cursor | Add the current documented fallback IDs for Composer 2.5, Opus 5.5, Fable 5.1, Sonnet 5, GPT-5.6 Sol/Terra/Luna, Gemini 3.8 Flash, Muse Spark 1.3, and Grok 4.7/4.6/4.5. Runtime model discovery remains available. | +| OpenCode | Refresh the static fallback used by remote environments with GPT-6 and GPT-5.6 families, current Claude models, Gemini 3.8 Flash, and Grok 4.7. | +| Kimi Code | Add K3 256K. Relabel `kimi-for-coding` as K2.8 Preview, which replaced K2.7 under the same ID. Forward CLI effort for K2.8 Preview and both K3 variants. | + +## Sources and verification + +- [Claude Code model configuration](https://code.claude.com/docs/en/model-config) + documents the Opus 5.5 CLI requirement and supported efforts. + [Opus 5.5 specifications](https://platform.claude.com/docs/en/models/opus-5-5/overview) + and [model ID conventions](https://platform.claude.com/docs/en/about-claude/models/model-ids-and-versions) + supply direct and Bedrock IDs. Bedrock entries use the catalog's existing US + inference-profile convention; regional availability remains account-dependent. +- [OpenAI's Codex model guide](https://learn.chatgpt.com/docs/models) lists GPT-6 + Sol and Luna. The installed Codex model metadata independently lists Astra, + Sol, Luna, and all three GPT-5.6 variants with the effort sets used here. + [GPT-6 Luna specifications](https://developers.openai.com/api/docs/models/gpt-6-luna) + also document Fast pricing. Codex's Ultra setting is a CLI capability; it is + not inferred from the API's effort enum. +- [Grok 4.7](https://docs.x.ai/developers/grok-4-7) and + [Grok 4.6](https://docs.x.ai/developers/models/grok-4.6) document their IDs and + four effort levels. The installed `grok models` command had no authenticated + account and returned only its default sentinel. No inference was run. +- [Google's model catalog](https://ai.google.dev/gemini-api/docs/models) supplies + the new Gemini IDs. [The retirement schedule](https://ai.google.dev/gemini-api/docs/deprecations) + records Gemini 2.0 shutdown on June 1, 2026. Gemini 2.5 remains available to + existing users. [Gemini CLI model selection](https://geminicli.com/docs/cli/model/) + accepts an explicit model; login method and entitlement determine availability. +- [Cursor's catalog](https://cursor.com/docs/models-and-pricing) links each + model's exact ID. `cursor-agent --list-models` returned no models for the local + account, so the additions use documented IDs rather than guessed variant + suffixes. Account-specific Fast/thinking variants remain discoverable. +- [Kimi's model configuration](https://www.kimi.com/code/docs/en/kimi-code/models.html) + documents all four current IDs, K2.8's in-place alias upgrade, and effort + support. Kimi effort remains limited to the existing explicit CLI engine; + the default ACP engine's effort mapping is outside this catalog update. + +## Other adapters and restricted models + +OpenCode discovers local models, but remote environment routes use its static +fallback catalog. All twelve added provider-qualified IDs were also present in +the installed OpenCode registry. [OpenCode model configuration](https://opencode.ai/docs/models/) +documents its `provider/model` format and provider registry. + +Pi already discovers models from its runtime or provider registry. +Hermes and OpenClaw accept provider configuration without a curated model list. +Cursor Cloud obtains its account's model list from Cursor. These paths do not +need a static entry per upstream release. Local OpenCode discovery returned +current models, including Muse Spark 1.3 and Nemotron 3.5 Lightning. + +Anthropic describes Mythos 5.1 as invitation-only in the +[Fable 5.1 documentation](https://platform.claude.com/docs/en/models/fable-5-1/overview). +The public current-model catalog does not publish a selectable ID for it. Keep +account discovery and custom IDs available instead of inventing an ID. Grok +4.7 Fast is available in Grok Build and Cursor, but its exact account-specific +CLI variant ID was not exposed locally. It is not a public xAI API model. +Image, video, audio, and embedding models are outside these coding-agent pickers. + +## Prior work checked + +Fable 5.1 was already merged in [#12730](https://github.com/paperclipai/paperclip/pull/12730). +Astra was already merged in [#12851](https://github.com/paperclipai/paperclip/pull/12851). +The Grok 4.6/4.5 proposal [#11324](https://github.com/paperclipai/paperclip/pull/11324) +was closed and parked by its author. This update preserves the newer CLI-default +sentinel behavior from [#12062](https://github.com/paperclipai/paperclip/pull/12062). +Open model-discovery work such as [#13127](https://github.com/paperclipai/paperclip/pull/13127) +and [#13565](https://github.com/paperclipai/paperclip/pull/13565) is separate from +these catalog and effort corrections. No open PR covering the newly added IDs +was found before implementation. diff --git a/packages/adapters/claude-local/src/index.test.ts b/packages/adapters/claude-local/src/index.test.ts index 8add6054b5..ce8bdf5a36 100644 --- a/packages/adapters/claude-local/src/index.test.ts +++ b/packages/adapters/claude-local/src/index.test.ts @@ -1,7 +1,21 @@ import { describe, expect, it } from "vitest"; -import { DEFAULT_CLAUDE_LOCAL_MODEL, resolveClaudeModel } from "./index.js"; +import { claudeLocalReasoningEffortsForModel, DEFAULT_CLAUDE_LOCAL_MODEL, resolveClaudeModel } from "./index.js"; +import { minimumClaudeCliVersionForModel } from "./server/cli-capabilities.js"; describe("Claude model defaults", () => { + it.each(["claude-opus-5-5", "us.anthropic.claude-opus-5-5", "global.anthropic.claude-opus-5-5[1m]"])("requires a current CLI and offers all efforts for %s", (model) => { + expect(minimumClaudeCliVersionForModel(model)).toBe("2.1.280"); + expect(claudeLocalReasoningEffortsForModel(model)).toEqual(["low", "medium", "high", "xhigh", "max"]); + }); + + it("keeps model-specific Claude reasoning limits", () => { + expect(claudeLocalReasoningEffortsForModel("claude-fable-5-1")).toContain("xhigh"); + expect(claudeLocalReasoningEffortsForModel("claude-sonnet-5")).toContain("max"); + expect(claudeLocalReasoningEffortsForModel("claude-sonnet-4-6")).toEqual(["low", "medium", "high", "max"]); + expect(claudeLocalReasoningEffortsForModel("claude-haiku-4-5")).toEqual([]); + expect(claudeLocalReasoningEffortsForModel("custom-model")).toEqual(["low", "medium", "high"]); + }); + it.each([undefined, null, "", " "])("uses Opus 5 for an unset model (%j)", (model) => { expect(DEFAULT_CLAUDE_LOCAL_MODEL).toBe("claude-opus-5"); expect(resolveClaudeModel(model)).toBe("claude-opus-5"); diff --git a/packages/adapters/claude-local/src/index.ts b/packages/adapters/claude-local/src/index.ts index c18d78d1b5..61ba77cabe 100644 --- a/packages/adapters/claude-local/src/index.ts +++ b/packages/adapters/claude-local/src/index.ts @@ -22,6 +22,16 @@ export function resolveClaudeModel( } export const type = "claude_local"; + +export function claudeLocalReasoningEffortsForModel(model: string): readonly string[] { + const id = model.trim().replace(/\[1m\]$/, "").replace(/^(?:(?:us|eu|apac|global)\.)?anthropic\./, ""); + if (/^claude-haiku-/.test(id)) return []; + if (/^claude-(?:opus-5(?:-5)?|opus-4-[78]|sonnet-5|fable-5(?:-1)?)$/.test(id)) { + return ["low", "medium", "high", "xhigh", "max"]; + } + if (/^claude-(?:opus|sonnet)-4-6(?:-v1)?$/.test(id)) return ["low", "medium", "high", "max"]; + return ["low", "medium", "high"]; +} export const label = "Claude Code"; export const SANDBOX_INSTALL_COMMAND = "npm install -g @anthropic-ai/claude-code"; @@ -32,6 +42,7 @@ export const models = [ { id: "claude-fable-5-1", label: "Claude Fable 5.1" }, { id: "claude-fable-5", label: "Claude Fable 5" }, { id: "claude-mythos-5", label: "Claude Mythos 5" }, + { id: "claude-opus-5-5", label: "Claude Opus 5.5" }, { id: "claude-opus-5", label: "Claude Opus 5" }, { id: "claude-opus-4-7", label: "Claude Opus 4.7" }, { id: "claude-opus-4-6", label: "Claude Opus 4.6" }, @@ -49,7 +60,7 @@ Core fields: - cwd (string, optional): default absolute working directory fallback for the agent process (created if missing when possible) - instructionsFilePath (string, optional): absolute path to a markdown instructions file injected at runtime - model (string, optional): Claude model id. Missing or blank defaults to ${DEFAULT_CLAUDE_LOCAL_MODEL} in both CLI and ACP, including existing agents. Explicit model IDs and ANTHROPIC_MODEL overrides are preserved. Bedrock/Vertex without an explicit model retain their provider default. -- effort (string, optional): reasoning effort passed via --effort (low|medium|high) +- effort (string, optional): model-specific reasoning effort passed via --effort (low|medium|high; current Opus, Sonnet 5, and Fable models also support xhigh|max) - chrome (boolean, optional): pass --chrome when running Claude - promptTemplate (string, optional): run prompt template - maxTurnsPerRun (number, optional): max turns for one run @@ -77,6 +88,7 @@ Operational fields: - graceSec (number, optional): SIGTERM grace period in seconds Notes: +- Claude Opus 5.5 uses model ID \`claude-opus-5-5\` and requires Claude Code v2.1.280 or later. - filesystemScope and networkScope are spawn-level confinement and are orthogonal to Claude permission flags. Both require Bubblewrap on the host and explicit engine="cli"; default or explicit ACP is rejected because ACP confinement is not yet supported. networkScope="allowlist" injects HTTP_PROXY/HTTPS_PROXY for the CLI while its private network namespace blocks direct sockets, so every required provider/API hostname must be listed explicitly. - The Claude ACP lane requires Node >=24.11.0 and @agentclientprotocol/claude-agent-acp to be installed with this adapter package. Missing prerequisites fail both default and explicit ACP runs with an actionable setup error; the adapter never switches engines automatically. - For ACP runs, model selection is passed through ANTHROPIC_MODEL at ACP server startup; Paperclip-managed Claude permissions and ephemeral skill materialization are handled by the shared ACP engine. diff --git a/packages/adapters/claude-local/src/server/cli-capabilities.ts b/packages/adapters/claude-local/src/server/cli-capabilities.ts index 5fa3ba7cb5..70da0cd3dc 100644 --- a/packages/adapters/claude-local/src/server/cli-capabilities.ts +++ b/packages/adapters/claude-local/src/server/cli-capabilities.ts @@ -6,11 +6,6 @@ const effortFlagSupportCache = new Map>(); export const CLAUDE_FABLE_5_1_MIN_CLI_VERSION = "2.1.251"; -const CLAUDE_FABLE_5_1_MODEL_IDS = new Set([ - "claude-fable-5-1", - "us.anthropic.claude-fable-5-1", -]); - export function claudeCommandLooksLike(command: string, expected = "claude"): boolean { const base = path.basename(command).toLowerCase(); return base === expected || base === `${expected}.cmd` || base === `${expected}.exe`; @@ -41,7 +36,9 @@ function cacheKeyForTarget(command: string, target: AdapterExecutionTarget | nul } export function minimumClaudeCliVersionForModel(model: string): string | null { - return CLAUDE_FABLE_5_1_MODEL_IDS.has(model.trim()) + const modelId = model.trim().replace(/\[1m\]$/, "").replace(/^(?:(?:us|eu|apac|global)\.)?anthropic\./, ""); + if (modelId === "claude-opus-5-5") return "2.1.280"; + return modelId === "claude-fable-5-1" ? CLAUDE_FABLE_5_1_MIN_CLI_VERSION : null; } diff --git a/packages/adapters/claude-local/src/server/execute.remote.test.ts b/packages/adapters/claude-local/src/server/execute.remote.test.ts index c1e776d20c..ba7728ab7e 100644 --- a/packages/adapters/claude-local/src/server/execute.remote.test.ts +++ b/packages/adapters/claude-local/src/server/execute.remote.test.ts @@ -18,7 +18,7 @@ const { signal: null, timedOut: false, stdout: args.includes("--version") - ? "2.1.251 (Claude Code)\n" + ? "2.1.280 (Claude Code)\n" : [ JSON.stringify({ type: "system", subtype: "init", session_id: "claude-session-1", model: "claude-sonnet" }), JSON.stringify({ type: "assistant", session_id: "claude-session-1", message: { content: [{ type: "text", text: "hello" }] } }), @@ -466,14 +466,14 @@ describe("claude remote execution", () => { return { args: call?.[2] ?? [], result }; } - it("passes the exact configured Fable 5.1 ID as --model on the CLI lane", async () => { + it.each(["claude-fable-5-1", "claude-opus-5-5"])("passes %s as --model on the CLI lane", async (model) => { const { args } = await executeWithModel("paperclip-claude-model-direct-", { - model: "claude-fable-5-1", + model, }); const modelFlag = args.indexOf("--model"); expect(modelFlag).toBeGreaterThanOrEqual(0); - expect(args[modelFlag + 1]).toBe("claude-fable-5-1"); + expect(args[modelFlag + 1]).toBe(model); }); it("passes the Bedrock-native Fable 5.1 ID as --model under Bedrock auth", async () => { @@ -496,27 +496,30 @@ describe("claude remote execution", () => { expect(args).not.toContain("--model"); }); - it("rejects Fable 5.1 before launch when the CLI is older than 2.1.251", async () => { + it.each([ + ["claude-fable-5-1", "2.1.251", "2.1.247"], + ["claude-opus-5-5", "2.1.280", "2.1.279"], + ])("rejects %s before launch below CLI %s", async (model, minimumVersion, detectedVersion) => { runChildProcess.mockResolvedValueOnce({ exitCode: 0, signal: null, timedOut: false, - stdout: "2.1.247 (Claude Code)\n", + stdout: `${detectedVersion} (Claude Code)\n`, stderr: "", pid: 123, startedAt: new Date().toISOString(), }); const { args, result } = await executeWithModel("paperclip-claude-model-old-cli-", { - model: "claude-fable-5-1", + model, }); expect(args).toEqual([]); expect(result.errorCode).toBe("claude_cli_version_incompatible"); - expect(result.errorMessage).toContain("requires Claude Code 2.1.251 or newer"); + expect(result.errorMessage).toContain(`${model} requires Claude Code ${minimumVersion} or newer`); expect(result.resultJson).toMatchObject({ - requiredClaudeCodeVersion: "2.1.251", - detectedClaudeCodeVersion: "2.1.247", + requiredClaudeCodeVersion: minimumVersion, + detectedClaudeCodeVersion: detectedVersion, }); }); diff --git a/packages/adapters/claude-local/src/server/execute.ts b/packages/adapters/claude-local/src/server/execute.ts index 83209568c0..c9387b6e07 100644 --- a/packages/adapters/claude-local/src/server/execute.ts +++ b/packages/adapters/claude-local/src/server/execute.ts @@ -1286,7 +1286,7 @@ export async function execute(ctx: AdapterExecutionContext): Promise { expect(JSON.stringify(spawnedEnv)).not.toContain("caller-proxy"); }); - it("warns without executing when runtime PATH selects a different local Claude executable", async () => { + it.each([ + ["claude-fable-5-1", "2.1.251"], + ["claude-opus-5-5", "2.1.280"], + ])("warns without executing %s when runtime PATH selects a different executable", async (model, minimumVersion) => { const runtimeDir = await mkdtemp(path.join(os.tmpdir(), "paperclip-cli-runtime-path-")); const runtimeClaudePath = path.join(runtimeDir, "claude"); await writeFile(runtimeClaudePath, "#!/bin/sh\nexit 0\n"); await chmod(runtimeClaudePath, 0o755); try { - probeResult.value = { exitCode: 0, stdout: "2.1.251 (Claude Code)\n", stderr: "" }; + probeResult.value = { exitCode: 0, stdout: `${minimumVersion} (Claude Code)\n`, stderr: "" }; const result = await testEnvironment({ companyId: "company-1", @@ -654,7 +657,7 @@ describe("claude CLI local hello probe hardening", () => { config: { engine: "cli", command: "claude", - model: "claude-fable-5-1", + model, env: { PATH: runtimeDir }, }, executionTarget: null, @@ -664,6 +667,7 @@ describe("claude CLI local hello probe hardening", () => { expect(result.status).toBe("warn"); expect(result.checks).toContainEqual(expect.objectContaining({ code: "claude_cli_version_probe_mismatch", + hint: `Ensure the runtime-selected Claude Code is ${minimumVersion} or newer. Execution will verify that exact executable before launch.`, level: "warn", })); expect(runAdapterExecutionTargetProcess).not.toHaveBeenCalled(); diff --git a/packages/adapters/claude-local/src/server/test.remote.test.ts b/packages/adapters/claude-local/src/server/test.remote.test.ts index b8cc0b36d6..9062260e02 100644 --- a/packages/adapters/claude-local/src/server/test.remote.test.ts +++ b/packages/adapters/claude-local/src/server/test.remote.test.ts @@ -141,10 +141,13 @@ describe("claude sandbox auth-missing check", () => { }); describe("claude CLI model compatibility check", () => { - it("fails before the hello probe when Fable 5.1 is configured with an older CLI", async () => { + it.each([ + ["claude-fable-5-1", "2.1.251", "2.1.247"], + ["claude-opus-5-5", "2.1.280", "2.1.279"], + ])("fails before the hello probe for %s below CLI %s", async (model, minimumVersion, detectedVersion) => { probeResult.value = { exitCode: 0, - stdout: "2.1.247 (Claude Code)\n", + stdout: `${detectedVersion} (Claude Code)\n`, stderr: "", }; @@ -154,7 +157,7 @@ describe("claude CLI model compatibility check", () => { config: { engine: "cli", command: "claude", - model: "claude-fable-5-1", + model, }, executionTarget: sandboxTarget, environmentName: "Daytona", @@ -164,7 +167,8 @@ describe("claude CLI model compatibility check", () => { expect(result.checks).toContainEqual(expect.objectContaining({ code: "claude_cli_version_incompatible", level: "error", - detail: "Detected Claude Code 2.1.247.", + message: `${model} requires Claude Code ${minimumVersion} or newer on the CLI lane.`, + detail: `Detected Claude Code ${detectedVersion}.`, })); expect(runAdapterExecutionTargetProcess).toHaveBeenCalledTimes(1); const versionCall = runAdapterExecutionTargetProcess.mock.calls[0] as unknown as [ diff --git a/packages/adapters/claude-local/src/server/test.ts b/packages/adapters/claude-local/src/server/test.ts index 8677ad1761..b7cf5001b4 100644 --- a/packages/adapters/claude-local/src/server/test.ts +++ b/packages/adapters/claude-local/src/server/test.ts @@ -266,9 +266,9 @@ export async function testEnvironment( code: "claude_cli_version_probe_mismatch", level: "warn", message: - "Skipped Fable 5.1 readiness probing because the runtime PATH selects a different Claude executable than the trusted local Test probe.", + `Skipped ${configuredModel} readiness probing because the runtime PATH selects a different Claude executable than the trusted local Test probe.`, hint: - "Ensure the runtime-selected Claude Code is 2.1.251 or newer. Execution will verify that exact executable before launch.", + `Ensure the runtime-selected Claude Code is ${minimumCliVersion} or newer. Execution will verify that exact executable before launch.`, }); } else if (canRunProbe && minimumCliVersion && versionProbeCommand) { const versionProbeEnv = localProbe?.env ?? env; @@ -289,7 +289,7 @@ export async function testEnvironment( checks.push({ code: "claude_cli_version_incompatible", level: "error", - message: `Claude Fable 5.1 requires Claude Code ${minimumCliVersion} or newer on the CLI lane.`, + message: `${configuredModel} requires Claude Code ${minimumCliVersion} or newer on the CLI lane.`, detail: detectedCliVersion ? `Detected Claude Code ${detectedCliVersion}.` : "Could not determine the installed Claude Code version.", diff --git a/packages/adapters/codex-local/src/index.test.ts b/packages/adapters/codex-local/src/index.test.ts index c70529e6da..855a0c06db 100644 --- a/packages/adapters/codex-local/src/index.test.ts +++ b/packages/adapters/codex-local/src/index.test.ts @@ -14,21 +14,24 @@ describe("codex local adapter metadata", () => { // Default to the concrete gpt-5.6-sol slug — Codex ships no metadata for the bare gpt-5.6 // alias, so it must not be advertised or used as the default (it triggers a fallback warning). expect(DEFAULT_CODEX_LOCAL_MODEL).toBe("gpt-5.6-sol"); - expect(modelIds.slice(0, 4)).toEqual([ + expect(modelIds.slice(0, 6)).toEqual([ "gpt-5.6-sol", "gpt-6-astra", + "gpt-6-sol", + "gpt-6-luna", "gpt-5.6-terra", "gpt-5.6-luna", ]); expect(modelIds).not.toContain("gpt-5.6"); expect(isCodexLocalFastModeSupported(DEFAULT_CODEX_LOCAL_MODEL)).toBe(true); expect(isCodexLocalFastModeSupported("gpt-6-astra")).toBe(true); + expect(isCodexLocalFastModeSupported("gpt-6-sol")).toBe(true); expect(modelIds).not.toContain("gpt-5.3-codex"); expect(modelIds).not.toContain("gpt-5.3-codex-spark"); }); - it("uses the reasoning efforts advertised for GPT-6 Astra", () => { - expect(codexLocalReasoningEffortsForModel("gpt-6-astra")).toEqual([ + it.each(["gpt-6-astra", "gpt-6-sol", " gpt-6-sol ", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6"])("uses the reasoning efforts advertised for %s", (model) => { + expect(codexLocalReasoningEffortsForModel(model)).toEqual([ "low", "medium", "high", @@ -36,7 +39,10 @@ describe("codex local adapter metadata", () => { "max", "ultra", ]); - expect(codexLocalReasoningEffortsForModel("gpt-5.6-sol")).toEqual([ + }); + + it.each(["gpt-5.5", "custom-model"])("preserves legacy efforts for %s", (model) => { + expect(codexLocalReasoningEffortsForModel(model)).toEqual([ "minimal", "low", "medium", @@ -45,6 +51,11 @@ describe("codex local adapter metadata", () => { ]); }); + it.each(["gpt-6-luna", "gpt-5.6-luna"])("caps %s at max and supports Fast mode", (model) => { + expect(codexLocalReasoningEffortsForModel(model)).toEqual(["low", "medium", "high", "xhigh", "max"]); + expect(isCodexLocalFastModeSupported(model)).toBe(true); + }); + it("normalizes the legacy bare gpt-5.6 alias to the concrete gpt-5.6-sol slug", () => { expect(normalizeCodexModel("gpt-5.6")).toBe("gpt-5.6-sol"); expect(normalizeCodexModel(" gpt-5.6 ")).toBe("gpt-5.6-sol"); diff --git a/packages/adapters/codex-local/src/index.ts b/packages/adapters/codex-local/src/index.ts index 4bd62ddeff..3cb69dc4ce 100644 --- a/packages/adapters/codex-local/src/index.ts +++ b/packages/adapters/codex-local/src/index.ts @@ -14,6 +14,8 @@ export const DEFAULT_CODEX_LOCAL_MODEL = PAPERCLIP_RUNNER_DEFAULT_MODELS.codex; export const DEFAULT_CODEX_LOCAL_BYPASS_APPROVALS_AND_SANDBOX = true; export const CODEX_LOCAL_FAST_MODE_SUPPORTED_MODELS = [ "gpt-6-astra", + "gpt-6-sol", + "gpt-6-luna", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", @@ -41,18 +43,22 @@ const CODEX_LOCAL_DEFAULT_REASONING_EFFORTS = [ "xhigh", ] as const; -const CODEX_LOCAL_ASTRA_REASONING_EFFORTS = [ +const CODEX_LOCAL_MAX_REASONING_EFFORTS = [ "low", "medium", "high", "xhigh", "max", +] as const; + +const CODEX_LOCAL_ULTRA_REASONING_EFFORTS = [ + ...CODEX_LOCAL_MAX_REASONING_EFFORTS, "ultra", ] as const; export type CodexLocalReasoningEffort = | (typeof CODEX_LOCAL_DEFAULT_REASONING_EFFORTS)[number] - | (typeof CODEX_LOCAL_ASTRA_REASONING_EFFORTS)[number]; + | (typeof CODEX_LOCAL_ULTRA_REASONING_EFFORTS)[number]; export function normalizeCodexModel(model: string | null | undefined): string { const normalizedModel = normalizeModelId(model); @@ -62,9 +68,19 @@ export function normalizeCodexModel(model: string | null | undefined): string { export function codexLocalReasoningEffortsForModel( model: string | null | undefined, ): readonly CodexLocalReasoningEffort[] { - return normalizeCodexModel(model) === "gpt-6-astra" - ? CODEX_LOCAL_ASTRA_REASONING_EFFORTS - : CODEX_LOCAL_DEFAULT_REASONING_EFFORTS; + const normalizedModel = normalizeCodexModel(model); + switch (normalizedModel) { + case "gpt-6-astra": + case "gpt-6-sol": + case "gpt-5.6-sol": + case "gpt-5.6-terra": + return CODEX_LOCAL_ULTRA_REASONING_EFFORTS; + case "gpt-6-luna": + case "gpt-5.6-luna": + return CODEX_LOCAL_MAX_REASONING_EFFORTS; + default: + return CODEX_LOCAL_DEFAULT_REASONING_EFFORTS; + } } export function isCodexLocalKnownModel(model: string | null | undefined): boolean { @@ -94,6 +110,8 @@ export const models = [ // DEFAULT_CODEX_LOCAL_MODEL is gpt-5.6-sol, so it doubles as the first (default) 5.6 entry. { id: DEFAULT_CODEX_LOCAL_MODEL, label: DEFAULT_CODEX_LOCAL_MODEL }, { id: "gpt-6-astra", label: "gpt-6-astra" }, + { id: "gpt-6-sol", label: "gpt-6-sol" }, + { id: "gpt-6-luna", label: "gpt-6-luna" }, { id: "gpt-5.6-terra", label: "gpt-5.6-terra" }, { id: "gpt-5.6-luna", label: "gpt-5.6-luna" }, { id: "gpt-5.4", label: "gpt-5.4" }, @@ -116,10 +134,10 @@ Core fields: - cwd (string, optional): default absolute working directory fallback for the agent process (created if missing when possible) - instructionsFilePath (string, optional): absolute path to a markdown instructions file prepended to stdin prompt at runtime - model (string, optional): Codex model id -- modelReasoningEffort (string, optional): reasoning effort override passed via -c model_reasoning_effort=...; GPT-6 Astra supports low|medium|high|xhigh|max|ultra +- modelReasoningEffort (string, optional): reasoning effort override passed via -c model_reasoning_effort=...; GPT-6 Astra/Sol and GPT-5.6 Sol/Terra support low|medium|high|xhigh|max|ultra; GPT-6 Luna and GPT-5.6 Luna support low|medium|high|xhigh|max - promptTemplate (string, optional): run prompt template - search (boolean, optional): run codex with --search -- fastMode (boolean, optional): enable Codex Fast mode; supported on GPT-6 Astra, GPT-5.6 (sol/terra/luna), GPT-5.5, GPT-5.4 and passed through for manual model IDs +- fastMode (boolean, optional): enable Codex Fast mode; supported on GPT-6 (astra/sol/luna), GPT-5.6 (sol/terra/luna), GPT-5.5, GPT-5.4 and passed through for manual model IDs - dangerouslyBypassApprovalsAndSandbox (boolean, optional): run with bypass flag - command (string, optional): defaults to "codex" - extraArgs (string[], optional): additional CLI args @@ -150,7 +168,7 @@ Notes: - Paperclip injects desired local skills into the effective CODEX_HOME/skills/ directory at execution time so Codex can discover "$paperclip" and related skills without polluting the project working directory. For new and updated agents, Paperclip assigns an isolated managed home at ~/.paperclip/instances//companies//agents//codex-home/skills/; when CODEX_HOME is explicitly overridden in adapter config, that override is used instead. - New and updated codex_local agents persist an empty OPENAI_API_KEY override by default so a host-level OPENAI_API_KEY cannot leak into Codex runs through process inheritance. Explicit CODEX_HOME overrides must not point at the shared company codex-home, $CODEX_HOME, or ~/.codex. - Some model/tool combinations reject certain effort levels (for example minimal with web search enabled). -- Fast mode is supported on GPT-6 Astra, GPT-5.6 (sol/terra/luna), GPT-5.5, GPT-5.4 and manual model IDs. When enabled for those models, Paperclip applies \`service_tier="fast"\` and \`features.fast_mode=true\`. +- Fast mode is supported on GPT-6 (astra/sol/luna), GPT-5.6 (sol/terra/luna), GPT-5.5, GPT-5.4 and manual model IDs. When enabled for those models, Paperclip applies \`service_tier="fast"\` and \`features.fast_mode=true\`. - When Paperclip realizes a workspace/runtime for a run, it injects PAPERCLIP_WORKSPACE_* and PAPERCLIP_RUNTIME_* env vars for agent-side tooling. - The ACP engine keeps its workspace sandbox and enables network access on each turn. Explicit sandbox_workspace_write.network_access overrides in extraArgs (or env.PAPERCLIP_CODEX_ACP_NETWORK_ACCESS="false") disable it; execution-target network denial wins. The bundled ACP patch is needed because upstream mode presets override Codex config.toml on every turn. - The CLI engine defaults to a writable workspace sandbox with network access for unattended work and Paperclip API calls. It does not enable the dangerous bypass flag. Explicit sandbox modes/profiles and network overrides in extraArgs retain their meaning. An execution-target network denial remains enforced. diff --git a/packages/adapters/codex-local/src/server/acp.test.ts b/packages/adapters/codex-local/src/server/acp.test.ts index e98a1881e4..eaa8c3dfe2 100644 --- a/packages/adapters/codex-local/src/server/acp.test.ts +++ b/packages/adapters/codex-local/src/server/acp.test.ts @@ -508,15 +508,15 @@ describe("codex_local ACP lane", () => { }); }); - it("forwards GPT-6 Astra controls to the ACPX Codex target", () => { + it.each([["gpt-6-astra", "ultra"], ["gpt-6-sol", "ultra"], ["gpt-6-luna", "max"], ["gpt-5.6-sol", "ultra"], ["gpt-5.6-terra", "ultra"], ["gpt-5.6-luna", "max"]])("forwards %s controls to the ACPX Codex target", (model, effort) => { expect(buildCodexAcpConfig({ engine: "acp", - model: "gpt-6-astra", - modelReasoningEffort: "ultra", + model, + modelReasoningEffort: effort, fastMode: true, })).toMatchObject({ - model: "gpt-6-astra", - modelReasoningEffort: "ultra", + model, + modelReasoningEffort: effort, fastMode: true, }); }); diff --git a/packages/adapters/codex-local/src/server/codex-args.test.ts b/packages/adapters/codex-local/src/server/codex-args.test.ts index 0cdfa266db..266f987dc8 100644 --- a/packages/adapters/codex-local/src/server/codex-args.test.ts +++ b/packages/adapters/codex-local/src/server/codex-args.test.ts @@ -9,14 +9,14 @@ describe("buildCodexExecArgs", () => { if (resumeSessionId) expect(args.slice(-3)).toEqual(["resume", resumeSessionId, "-"]); }); - it("forwards GPT-6 Astra, its ultra reasoning effort, and fast mode", () => { + it.each([["gpt-6-astra", "ultra"], ["gpt-6-sol", "ultra"], ["gpt-6-luna", "max"], ["gpt-5.6-sol", "ultra"], ["gpt-5.6-terra", "ultra"], ["gpt-5.6-luna", "max"]])("forwards %s, its supported reasoning effort, and fast mode", (model, effort) => { const result = buildCodexExecArgs({ - model: "gpt-6-astra", - modelReasoningEffort: "ultra", + model, + modelReasoningEffort: effort, fastMode: true, }); - expect(result.model).toBe("gpt-6-astra"); + expect(result.model).toBe(model); expect(result.fastModeApplied).toBe(true); expect(result.fastModeIgnoredReason).toBeNull(); expect(result.args).toEqual([ @@ -24,9 +24,9 @@ describe("buildCodexExecArgs", () => { "--json", "--dangerously-bypass-approvals-and-sandbox", "--model", - "gpt-6-astra", + model, "-c", - 'model_reasoning_effort="ultra"', + `model_reasoning_effort="${effort}"`, "-c", 'service_tier="fast"', "-c", @@ -148,7 +148,7 @@ describe("buildCodexExecArgs", () => { expect(result.fastModeRequested).toBe(true); expect(result.fastModeApplied).toBe(false); expect(result.fastModeIgnoredReason).toContain( - "currently only supported on gpt-6-astra, gpt-5.6-sol, gpt-5.6-terra, gpt-5.6-luna, gpt-5.5, gpt-5.4 or manually configured model IDs", + "currently only supported on gpt-6-astra, gpt-6-sol, gpt-6-luna, gpt-5.6-sol, gpt-5.6-terra, gpt-5.6-luna, gpt-5.5, gpt-5.4 or manually configured model IDs", ); expect(result.args).toEqual([ "exec", diff --git a/packages/adapters/codex-local/src/ui/build-config.test.ts b/packages/adapters/codex-local/src/ui/build-config.test.ts index d8736bd94d..b9338a3e5c 100644 --- a/packages/adapters/codex-local/src/ui/build-config.test.ts +++ b/packages/adapters/codex-local/src/ui/build-config.test.ts @@ -63,18 +63,18 @@ describe("buildCodexLocalConfig", () => { }); }); - it("persists the exact GPT-6 Astra model and supported controls", () => { + it.each([["gpt-6-astra", "ultra"], ["gpt-6-sol", "ultra"], ["gpt-6-luna", "max"], ["gpt-5.6-sol", "ultra"], ["gpt-5.6-terra", "ultra"], ["gpt-5.6-luna", "max"]])("persists the exact %s model and supported controls", (model, effort) => { const config = buildCodexLocalConfig( makeValues({ - model: "gpt-6-astra", - thinkingEffort: "ultra", + model, + thinkingEffort: effort, fastMode: true, }), ); expect(config).toMatchObject({ - model: "gpt-6-astra", - modelReasoningEffort: "ultra", + model, + modelReasoningEffort: effort, fastMode: true, }); }); diff --git a/packages/adapters/cursor-local/src/index.ts b/packages/adapters/cursor-local/src/index.ts index 0a78423c4f..a10618c640 100644 --- a/packages/adapters/cursor-local/src/index.ts +++ b/packages/adapters/cursor-local/src/index.ts @@ -14,6 +14,18 @@ export const DEFAULT_CURSOR_LOCAL_MODEL = "auto"; const CURSOR_FALLBACK_MODEL_IDS = [ "auto", + "composer-2.5", + "claude-opus-5-5", + "claude-fable-5-1", + "claude-sonnet-5", + "gpt-5.6-sol", + "gpt-5.6-terra", + "gpt-5.6-luna", + "gemini-3.8-flash", + "muse-spark-1.3", + "grok-4.7", + "grok-4.6", + "grok-4.5", "composer-1.5", "composer-1", "gpt-5.3-codex-low", diff --git a/packages/adapters/gemini-local/src/index.ts b/packages/adapters/gemini-local/src/index.ts index 4ac6b8215d..d71ce497e7 100644 --- a/packages/adapters/gemini-local/src/index.ts +++ b/packages/adapters/gemini-local/src/index.ts @@ -13,11 +13,16 @@ export const models = [ { id: DEFAULT_GEMINI_LOCAL_MODEL, label: "Auto" }, { id: "gemini-3.1-pro-preview", label: "Gemini 3.1 Pro Preview" }, { id: "gemini-3.1-pro-preview-customtools", label: "Gemini 3.1 Pro Preview (Custom Tools)" }, + { id: "gemini-3.8-flash", label: "Gemini 3.8 Flash" }, + { id: "gemini-3.7-flash", label: "Gemini 3.7 Flash" }, + { id: "gemini-3.6-flash", label: "Gemini 3.6 Flash" }, + { id: "gemini-3.5-flash", label: "Gemini 3.5 Flash" }, + { id: "gemini-3.5-flash-lite", label: "Gemini 3.5 Flash Lite" }, + { id: "gemini-3.1-flash-lite", label: "Gemini 3.1 Flash Lite" }, + { id: "gemini-3-flash-preview", label: "Gemini 3 Flash Preview" }, { id: "gemini-2.5-pro", label: "Gemini 2.5 Pro" }, { id: "gemini-2.5-flash", label: "Gemini 2.5 Flash" }, { id: "gemini-2.5-flash-lite", label: "Gemini 2.5 Flash Lite" }, - { id: "gemini-2.0-flash", label: "Gemini 2.0 Flash" }, - { id: "gemini-2.0-flash-lite", label: "Gemini 2.0 Flash Lite" }, ]; export const agentConfigurationDoc = `# gemini_local agent configuration diff --git a/packages/adapters/grok-local/src/index.ts b/packages/adapters/grok-local/src/index.ts index 8c669ad807..f296d57401 100644 --- a/packages/adapters/grok-local/src/index.ts +++ b/packages/adapters/grok-local/src/index.ts @@ -5,8 +5,17 @@ export const DEFAULT_GROK_LOCAL_MODEL = "grok-build"; export const models = [ { id: DEFAULT_GROK_LOCAL_MODEL, label: DEFAULT_GROK_LOCAL_MODEL }, + { id: "grok-4.7", label: "Grok 4.7" }, + { id: "grok-4.6", label: "Grok 4.6" }, + { id: "grok-4.5", label: "Grok 4.5" }, ]; +export function grokLocalReasoningEffortsForModel(model: string): readonly string[] { + return model.trim() === "grok-4.7" || model.trim() === "grok-4.6" + ? ["low", "medium", "high", "xhigh"] + : ["low", "medium", "high"]; +} + export const agentConfigurationDoc = `# grok_local agent configuration Adapter: grok_local @@ -27,7 +36,7 @@ Core fields: - promptTemplate (string, optional): run prompt template - model (string, optional): Grok model id. Defaults to grok-build. - permissionMode (string, optional): Grok permission mode passed via \`--permission-mode\`. Unset by default: Grok >= 1.0 enforces \`dontAsk\` as deny-by-default and it overrides \`--always-approve\`, so unattended runs rely on \`--always-approve\` alone unless you explicitly need a mode -- reasoningEffort (string, optional): Grok reasoning effort passed via \`--reasoning-effort\` +- reasoningEffort (string, optional): Grok reasoning effort (low|medium|high; grok-4.7 and grok-4.6 also accept xhigh) passed via \`--reasoning-effort\` - maxTurns (number, optional): maximum agent turns for the run - command (string, optional): defaults to "grok" - extraArgs (string[], optional): additional CLI args diff --git a/packages/adapters/grok-local/src/server/execute.test.ts b/packages/adapters/grok-local/src/server/execute.test.ts index 5b0356a7cb..7cd6e07fb9 100644 --- a/packages/adapters/grok-local/src/server/execute.test.ts +++ b/packages/adapters/grok-local/src/server/execute.test.ts @@ -160,6 +160,19 @@ async function makeCtx(runId: string, cwd: string): Promise { + it.each(["grok-4.7", "grok-4.6"])("forwards the explicit %s model and xhigh effort", async (model) => { + const root = await makeTempRoot(); + const ctx = await makeCtx("model-selection", root); + ctx.config = { cwd: root, model, reasoningEffort: "xhigh" }; + runProcessMock.mockResolvedValue(makeSuccessfulRunResult()); + + await execute(ctx); + + const args = runProcessMock.mock.calls[0][3] as string[]; + expect(args[args.indexOf("--model") + 1]).toBe(model); + expect(args[args.indexOf("--reasoning-effort") + 1]).toBe("xhigh"); + }); + beforeEach(() => { mocks.state.isRemote = false; mocks.state.prepareRuntimeResult = null; diff --git a/packages/adapters/kimi-local/src/index.ts b/packages/adapters/kimi-local/src/index.ts index 2ad83521ae..6a87c0b9d3 100644 --- a/packages/adapters/kimi-local/src/index.ts +++ b/packages/adapters/kimi-local/src/index.ts @@ -8,9 +8,10 @@ export const SANDBOX_INSTALL_COMMAND = buildSandboxNpmInstallCommand("@moonshot- export const DEFAULT_KIMI_LOCAL_MODEL = "kimi-code/kimi-for-coding"; export const models = [ - { id: DEFAULT_KIMI_LOCAL_MODEL, label: "K2.7 Coding" }, + { id: DEFAULT_KIMI_LOCAL_MODEL, label: "K2.8 Preview" }, { id: "kimi-code/kimi-for-coding-highspeed", label: "K2.7 Coding Highspeed" }, { id: "kimi-code/k3", label: "K3" }, + { id: "kimi-code/k3-256k", label: "K3 (256K)" }, ]; /** @@ -26,7 +27,11 @@ export type KimiEffort = (typeof KIMI_SUPPORTED_EFFORTS)[number]; * Models that advertise `support_efforts` in Kimi's model catalog. Keep in * sync with `models` above; only these accept KIMI_MODEL_THINKING_EFFORT. */ -export const EFFORT_CAPABLE_MODELS = new Set(["kimi-code/k3"]); +export const EFFORT_CAPABLE_MODELS = new Set([ + DEFAULT_KIMI_LOCAL_MODEL, + "kimi-code/k3", + "kimi-code/k3-256k", +]); export function modelSupportsEffort(model: string): boolean { return EFFORT_CAPABLE_MODELS.has(model.trim()); @@ -65,7 +70,7 @@ Core fields: - instructionsFilePath (string, optional): absolute path to a markdown instructions file prepended to the run prompt. Sibling files in the same directory (HEARTBEAT.md, SOUL.md, TOOLS.md) are made readable via --add-dir for local runs. - promptTemplate (string, optional): run prompt template - model (string, optional): Kimi model alias (provider/model). Defaults to kimi-code/kimi-for-coding. -- effort (string, optional): thinking effort (low | medium | high | max). CLI lane only (engine=cli): forwarded as KIMI_MODEL_THINKING_EFFORT for effort-capable models (currently kimi-code/k3); "medium" maps to "high" since Kimi has no medium tier. Ignored for models without support_efforts, and NOT forwarded on the default ACP engine lane (Kimi ACP exposes a separate "thinking" option that is not wired yet) — pin engine=cli when effort control matters. +- effort (string, optional): thinking effort (low | medium | high | max). CLI lane only (engine=cli): forwarded as KIMI_MODEL_THINKING_EFFORT for effort-capable models (K2.8 Preview, K3, and K3 256K); "medium" maps to "high" since Kimi has no medium tier. Ignored for models without support_efforts, and NOT forwarded on the default ACP engine lane (Kimi ACP exposes a separate "thinking" option that is not wired yet) — pin engine=cli when effort control matters. - command (string, optional): defaults to "kimi" - extraArgs (string[], optional): additional CLI args - env (object, optional): KEY=VALUE environment variables diff --git a/packages/adapters/kimi-local/src/server/execute.test.ts b/packages/adapters/kimi-local/src/server/execute.test.ts index f005916d8a..7d4b5aff4e 100644 --- a/packages/adapters/kimi-local/src/server/execute.test.ts +++ b/packages/adapters/kimi-local/src/server/execute.test.ts @@ -333,7 +333,7 @@ describe("kimi_local execute", () => { expect(seenEnv.TERM).toBe("xterm-256color"); }); - it("forwards configured effort as KIMI_MODEL_THINKING_EFFORT for effort-capable models", async () => { + it.each(["kimi-code/k3", "kimi-code/k3-256k", "kimi-code/kimi-for-coding"])("forwards configured effort for %s", async (model) => { const root = await makeTempRoot(); let seenEnv: Record = {}; runProcessMock.mockImplementation(async (_runId, _target, _command, _args, options) => { @@ -341,7 +341,7 @@ describe("kimi_local execute", () => { return { exitCode: 0, signal: null, timedOut: false, stdout: KIMI_STDOUT, stderr: "" }; }); - await execute(makeContext(root, { config: { cwd: root, model: "kimi-code/k3", effort: "high" } })); + await execute(makeContext(root, { config: { cwd: root, model, effort: "high" } })); expect(seenEnv.KIMI_MODEL_THINKING_EFFORT).toBe("high"); }); @@ -368,7 +368,7 @@ describe("kimi_local execute", () => { }); await execute(makeContext(root, { - config: { cwd: root, model: "kimi-code/kimi-for-coding", effort: "high" }, + config: { cwd: root, model: "kimi-code/kimi-for-coding-highspeed", effort: "high" }, })); expect(seenEnv.KIMI_MODEL_THINKING_EFFORT).toBeUndefined(); diff --git a/packages/adapters/opencode-local/src/index.ts b/packages/adapters/opencode-local/src/index.ts index ca121e9125..0c02ab9ba1 100644 --- a/packages/adapters/opencode-local/src/index.ts +++ b/packages/adapters/opencode-local/src/index.ts @@ -53,6 +53,18 @@ export function isValidOpenCodeModelId(value: unknown): value is string { export const models: Array<{ id: string; label: string }> = [ { id: DEFAULT_OPENCODE_LOCAL_MODEL, label: DEFAULT_OPENCODE_LOCAL_MODEL }, + { id: "openai/gpt-6-astra", label: "openai/gpt-6-astra" }, + { id: "openai/gpt-6-sol", label: "openai/gpt-6-sol" }, + { id: "openai/gpt-6-luna", label: "openai/gpt-6-luna" }, + { id: "openai/gpt-5.6-sol", label: "openai/gpt-5.6-sol" }, + { id: "openai/gpt-5.6-terra", label: "openai/gpt-5.6-terra" }, + { id: "openai/gpt-5.6-luna", label: "openai/gpt-5.6-luna" }, + { id: "anthropic/claude-opus-5-5", label: "anthropic/claude-opus-5-5" }, + { id: "anthropic/claude-opus-5", label: "anthropic/claude-opus-5" }, + { id: "anthropic/claude-fable-5-1", label: "anthropic/claude-fable-5-1" }, + { id: "anthropic/claude-sonnet-5", label: "anthropic/claude-sonnet-5" }, + { id: "google/gemini-3.8-flash", label: "google/gemini-3.8-flash" }, + { id: "xai/grok-4.7", label: "xai/grok-4.7" }, { id: "openai/gpt-5.5", label: "openai/gpt-5.5" }, { id: "openai/gpt-5.4", label: "openai/gpt-5.4" }, { id: "openai/gpt-5.4-mini", label: "openai/gpt-5.4-mini" }, diff --git a/server/src/__tests__/adapter-models.test.ts b/server/src/__tests__/adapter-models.test.ts index 3b84c652a6..e35ba1b447 100644 --- a/server/src/__tests__/adapter-models.test.ts +++ b/server/src/__tests__/adapter-models.test.ts @@ -71,6 +71,7 @@ describe("adapter model listing", () => { expect(models.some((model) => model.id === "claude-mythos-5")).toBe(true); // Opus 5 is a current GA flagship and must be offered even when live discovery is unavailable. expect(models.some((model) => model.id === "claude-opus-5")).toBe(true); + expect(models).toContainEqual({ id: "claude-opus-5-5", label: "Claude Opus 5.5" }); expect(fetchSpy).not.toHaveBeenCalled(); }); @@ -93,6 +94,7 @@ describe("adapter model listing", () => { expect(first).toEqual(second); expect(first.some((model) => model.id === "claude-opus-4-8-20260529")).toBe(true); expect(first.some((model) => model.id === "claude-opus-4-8")).toBe(true); + expect(first.some((model) => model.id === "claude-opus-5-5")).toBe(true); }); it("refreshes cached claude models on demand", async () => { @@ -131,18 +133,21 @@ describe("adapter model listing", () => { expect(models).toEqual(claudeFallbackModels); }); - it("does not duplicate claude-fable-5-1 when discovery returns the identical ID", async () => { + it.each([ + ["claude-fable-5-1", "Claude Fable 5.1"], + ["claude-opus-5-5", "Claude Opus 5.5"], + ])("does not duplicate %s when discovery returns the identical ID", async (id, displayName) => { process.env.ANTHROPIC_API_KEY = "sk-ant-test"; vi.spyOn(globalThis, "fetch").mockResolvedValue({ ok: true, json: async () => ({ - data: [{ id: "claude-fable-5-1", display_name: "Claude Fable 5.1" }], + data: [{ id, display_name: displayName }], }), } as Response); const models = await listAdapterModels("claude_local"); - expect(models.filter((model) => model.id === "claude-fable-5-1")).toHaveLength(1); + expect(models.filter((model) => model.id === id)).toEqual([{ id, label: displayName }]); // Curated fallbacks discovery did not return are still merged in. expect(models.some((model) => model.id === "claude-fable-5")).toBe(true); expect(models.some((model) => model.id === "claude-opus-4-8")).toBe(true); @@ -154,13 +159,42 @@ describe("adapter model listing", () => { const models = await listAdapterModels("claude_local"); - // The Bedrock default (first entry) is unchanged. - expect(models[0]?.id).toBe("us.anthropic.claude-opus-4-8-v1"); - expect(models.some((model) => model.id === "us.anthropic.claude-fable-5-1")).toBe(true); + // Keep Opus 4.8 first, using its documented dateless Bedrock ID. + expect(models[0]?.id).toBe("us.anthropic.claude-opus-4-8"); + expect(models.map((model) => model.id)).toEqual(expect.arrayContaining([ + "us.anthropic.claude-opus-5-5", "us.anthropic.claude-opus-5", "us.anthropic.claude-sonnet-5", + "us.anthropic.claude-fable-5-1", "us.anthropic.claude-opus-4-7", "us.anthropic.claude-sonnet-4-6", + ])); + expect(models.map((model) => model.id)).not.toEqual(expect.arrayContaining(["us.anthropic.claude-opus-4-8-v1"])); expect(models.some((model) => model.id === "claude-fable-5-1")).toBe(false); expect(fetchSpy).not.toHaveBeenCalled(); }); + it.each([ + ["gemini_local", ["gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-flash-lite", "gemini-3-flash-preview"]], + ["grok_local", ["grok-build", "grok-4.7", "grok-4.6", "grok-4.5"]], + ["kimi_local", ["kimi-code/kimi-for-coding", "kimi-code/k3", "kimi-code/k3-256k"]], + ])("lists current %s models without a provider login", async (adapter, expectedIds) => { + const models = await listAdapterModels(adapter as string); + expect(models.map((model) => model.id)).toEqual(expect.arrayContaining(expectedIds as string[])); + expect(new Set(models.map((model) => model.id)).size).toBe(models.length); + if (adapter === "gemini_local") { + expect(models.some((model) => model.id.startsWith("gemini-2.0-"))).toBe(false); + } + if (adapter === "kimi_local") { + expect(models).toContainEqual({ id: "kimi-code/kimi-for-coding", label: "K2.8 Preview" }); + } + }); + + it("includes current Cursor fallbacks when runtime discovery is unavailable", async () => { + setCursorModelsRunnerForTests(() => ({ status: 1, stdout: "", stderr: "", hasError: true })); + const models = await listAdapterModels("cursor"); + expect(models.map((model) => model.id)).toEqual(expect.arrayContaining([ + "composer-2.5", "claude-opus-5-5", "claude-fable-5-1", "claude-sonnet-5", + "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna", "grok-4.7", "gemini-3.8-flash", "muse-spark-1.3", + ])); + }); + it("loads codex models dynamically and merges fallback options", async () => { process.env.OPENAI_API_KEY = "sk-test"; const fetchSpy = vi.spyOn(globalThis, "fetch").mockResolvedValue({ @@ -232,12 +266,13 @@ describe("adapter model listing", () => { expect(models).toEqual(cursorFallbackModels); }); - it("returns opencode fallback models including gpt-5.4", async () => { + it("returns current provider-qualified OpenCode models when discovery is unavailable", async () => { process.env.PAPERCLIP_OPENCODE_COMMAND = "__paperclip_missing_opencode_command__"; const models = await listAdapterModels("opencode_local"); expect(models).toEqual(opencodeFallbackModels); + expect(models.map((model) => model.id)).toEqual(expect.arrayContaining(["openai/gpt-6-astra", "openai/gpt-6-sol", "openai/gpt-6-luna", "openai/gpt-5.6-sol", "openai/gpt-5.6-terra", "openai/gpt-5.6-luna", "anthropic/claude-opus-5-5", "anthropic/claude-opus-5", "anthropic/claude-fable-5-1", "anthropic/claude-sonnet-5", "google/gemini-3.8-flash", "xai/grok-4.7"])); }); it("loads cursor models dynamically and caches them", async () => { diff --git a/ui/src/components/AgentConfigForm.render.test.tsx b/ui/src/components/AgentConfigForm.render.test.tsx index fb3715d9e5..e31aee17ba 100644 --- a/ui/src/components/AgentConfigForm.render.test.tsx +++ b/ui/src/components/AgentConfigForm.render.test.tsx @@ -784,6 +784,23 @@ describe("AgentConfigForm environment selector", () => { expect(result.onSave.mock.calls[0][0].adapterConfig.effort).toBeUndefined(); }); + it("saves Grok 4.7 reasoning effort using the runtime key", async () => { + const result = await renderForm([], { adapterType: "grok_local", adapterConfig: { model: "grok-4.7", reasoningEffort: "high" } }); + roots.push(result.root); + const effort = [...result.container.querySelectorAll("button")].find(button => button.textContent?.trim() === "High")!; + expect(effort).toBeTruthy(); + await act(async () => effort.click()); + await flushReact(); + const xhigh = [...document.querySelectorAll("button")].find(button => button.textContent?.trim() === "X-Highxhigh")!; + expect(xhigh).toBeTruthy(); + await act(async () => xhigh.click()); + await flushReact(); + const save = [...result.container.querySelectorAll("button")].find(button => button.textContent?.trim() === "Save")!; + await act(async () => save.click()); + expect(result.onSave).toHaveBeenCalledWith(expect.objectContaining({ adapterConfig: expect.objectContaining({ reasoningEffort: "xhigh" }) })); + expect(result.onSave.mock.calls[0][0].adapterConfig.effort).toBeUndefined(); + }); + it("hides the environment override when Local is the only configured environment", async () => { const result = await renderForm([ makeEnvironment({ id: "local-1", name: "Local", driver: "local" }), diff --git a/ui/src/components/AgentConfigForm.tsx b/ui/src/components/AgentConfigForm.tsx index fb4e0b4d12..8556e8d514 100644 --- a/ui/src/components/AgentConfigForm.tsx +++ b/ui/src/components/AgentConfigForm.tsx @@ -1,6 +1,7 @@ import { AiConnectionField } from "./ai-connections/AiConnectionField"; import { aiConnectionBindingSchema } from "@paperclipai/shared"; import { testAgentSetup } from "@/lib/test-agent-setup"; +import { setupEfforts } from "../lib/agent-setup-fields"; import { RuntimeTestCard } from "./RuntimeTestCard"; import { useState, useEffect, useRef, useMemo, useCallback, Children, isValidElement, type ReactNode } from "react"; import type { AdapterConfigSection } from "../adapters/types"; @@ -1268,6 +1269,7 @@ export function AgentConfigForm(props: AgentConfigFormProps) { ? "mode" : adapterType === "opencode_local" ? "variant" + : adapterType === "grok_local" ? "reasoningEffort" : adapterType === "pi_local" ? "thinking" : "effort"; const thinkingEffortOptions = adapterType === "codex_local" @@ -1283,7 +1285,12 @@ export function AgentConfigForm(props: AgentConfigFormProps) { ? kimiThinkingEffortOptions : adapterType === "pi_local" ? [{ id: "", label: "Auto" }, ...["off", "minimal", "low", "medium", "high", "xhigh"].map(id => ({ id, label: id }))] - : claudeThinkingEffortOptions; + : adapterType === "claude_local" || adapterType === "grok_local" + ? [{ id: "", label: "Auto" }, ...setupEfforts(adapterType, currentModelId).map((id) => ({ + id, + label: id === "xhigh" ? "X-High" : id[0].toUpperCase() + id.slice(1), + }))] + : claudeThinkingEffortOptions; const currentThinkingEffort = isCreate ? val!.thinkingEffort : adapterType === "codex_local" @@ -1713,10 +1720,10 @@ export function AgentConfigForm(props: AgentConfigFormProps) { models={models} value={currentModelId} onChange={(v) => { - const supportedEfforts = codexReasoningEffortOptions(v, "Auto"); - const clearUnsupportedEffort = adapterType === "codex_local" + const supportedEfforts = setupEfforts(adapterType, v); + const clearUnsupportedEffort = ["codex_local", "claude_local", "grok_local"].includes(adapterType) && Boolean(currentThinkingEffort) - && !supportedEfforts.some((option) => option.value === currentThinkingEffort); + && !supportedEfforts.includes(String(currentThinkingEffort)); if (isCreate) { set!({ model: v, diff --git a/ui/src/components/agent-config-primitives.tsx b/ui/src/components/agent-config-primitives.tsx index 118a53dd75..92ef43378c 100644 --- a/ui/src/components/agent-config-primitives.tsx +++ b/ui/src/components/agent-config-primitives.tsx @@ -34,7 +34,7 @@ export const help: Record = { dangerouslySkipPermissions: "Run unattended by auto-approving adapter permission prompts when supported.", dangerouslyBypassSandbox: "Run Codex without sandbox restrictions. Required for filesystem/network access.", search: "Enable Codex web search capability during runs.", - fastMode: "Enable Codex Fast mode. This burns credits/tokens much faster and is supported on GPT-6 Astra, GPT-5.6, GPT-5.5, GPT-5.4, and manual Codex model IDs.", + fastMode: "Enable Codex Fast mode. This burns credits/tokens much faster and is supported on GPT-6, GPT-5.6, GPT-5.5, GPT-5.4, and manual Codex model IDs.", workspaceStrategy: "How Paperclip should realize an execution workspace for this agent. Keep project_primary for normal cwd execution, or use git_worktree for issue-scoped isolated checkouts.", workspaceBaseRef: "Base git ref used when creating a worktree branch. Leave blank to use the resolved workspace ref or HEAD.", workspaceBranchTemplate: "Template for naming derived branches. Supports {{issue.identifier}}, {{issue.title}}, {{agent.name}}, {{project.id}}, {{workspace.repoRef}}, and {{slug}}.", diff --git a/ui/src/lib/agent-setup-fields.test.ts b/ui/src/lib/agent-setup-fields.test.ts new file mode 100644 index 0000000000..e0d14f7e7d --- /dev/null +++ b/ui/src/lib/agent-setup-fields.test.ts @@ -0,0 +1,21 @@ +// @vitest-environment node +import { describe, expect, it } from "vitest"; +import { setupEfforts } from "./agent-setup-fields"; + +describe("model-specific setup efforts", () => { + it("offers current Claude efforts without offering them on Haiku", () => { + expect(setupEfforts("claude_local", "claude-fable-5-1")).toEqual(["low", "medium", "high", "xhigh", "max"]); + expect(setupEfforts("claude_local", "claude-haiku-4-5")).toEqual([]); + }); + + it("offers xhigh on Grok 4.7 and 4.6, with the lower limit on 4.5", () => { + expect(setupEfforts("grok_local", "grok-4.7")).toEqual(["low", "medium", "high", "xhigh"]); + expect(setupEfforts("grok_local", "grok-4.6")).toContain("xhigh"); + expect(setupEfforts("grok_local", "grok-4.5")).toEqual(["low", "medium", "high"]); + }); + + it("caps Luna at max while exposing ultra on Sol", () => { + expect(setupEfforts("codex_local", "gpt-6-luna")).toEqual(["low", "medium", "high", "xhigh", "max"]); + expect(setupEfforts("codex_local", "gpt-6-sol")).toContain("ultra"); + }); +}); diff --git a/ui/src/lib/agent-setup-fields.ts b/ui/src/lib/agent-setup-fields.ts index e94adecd72..e0c6b70b7e 100644 --- a/ui/src/lib/agent-setup-fields.ts +++ b/ui/src/lib/agent-setup-fields.ts @@ -1,4 +1,6 @@ import { DEFAULT_CODEX_LOCAL_MODEL } from "@paperclipai/adapter-codex-local"; +import { claudeLocalReasoningEffortsForModel, DEFAULT_CLAUDE_LOCAL_MODEL } from "@paperclipai/adapter-claude-local"; +import { grokLocalReasoningEffortsForModel } from "@paperclipai/adapter-grok-local"; import { codexReasoningEffortOptions } from "./codex-reasoning-effort"; import { PROVIDER_ENV_KEYS } from "./provider-credential"; @@ -27,7 +29,7 @@ export function setupProviderKeys(adapter: string) { export function setupEfforts(adapter: string, model = ""): string[] { switch (adapter) { case "claude_local": - return ["low", "medium", "high"]; + return [...claudeLocalReasoningEffortsForModel(model || DEFAULT_CLAUDE_LOCAL_MODEL)]; case "codex_local": return codexReasoningEffortOptions(model || DEFAULT_CODEX_LOCAL_MODEL) .map((option) => option.value) @@ -35,7 +37,7 @@ export function setupEfforts(adapter: string, model = ""): string[] { case "pi_local": return ["off", "minimal", "low", "medium", "high", "xhigh"]; case "grok_local": - return ["low", "medium", "high"]; + return [...grokLocalReasoningEffortsForModel(model)]; default: return []; } diff --git a/ui/src/lib/codex-reasoning-effort.test.ts b/ui/src/lib/codex-reasoning-effort.test.ts index 293d6fe5d3..0d7424e52b 100644 --- a/ui/src/lib/codex-reasoning-effort.test.ts +++ b/ui/src/lib/codex-reasoning-effort.test.ts @@ -4,8 +4,8 @@ import { describe, expect, it } from "vitest"; import { codexReasoningEffortOptions } from "./codex-reasoning-effort"; describe("codexReasoningEffortOptions", () => { - it("exposes only the supported GPT-6 Astra reasoning efforts", () => { - expect(codexReasoningEffortOptions("gpt-6-astra")).toEqual([ + it.each(["gpt-6-astra", "gpt-6-sol", "gpt-5.6-sol", "gpt-5.6-terra"])("exposes only the supported %s reasoning efforts", (model) => { + expect(codexReasoningEffortOptions(model)).toEqual([ { value: "", label: "Default" }, { value: "low", label: "Low" }, { value: "medium", label: "Medium" }, @@ -16,8 +16,14 @@ describe("codexReasoningEffortOptions", () => { ]); }); + it.each(["gpt-6-luna", "gpt-5.6-luna"])("caps %s at Max", (model) => { + expect(codexReasoningEffortOptions(model).map((option) => option.value)).toEqual([ + "", "low", "medium", "high", "xhigh", "max", + ]); + }); + it("preserves the existing choices for other and manual models", () => { - expect(codexReasoningEffortOptions("gpt-5.6-sol").map((option) => option.value)).toEqual([ + expect(codexReasoningEffortOptions("custom-model").map((option) => option.value)).toEqual([ "", "minimal", "low",