From 19c0afcc0d2e8a738f7a67499bb7a8b5a7ae2e3b Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 12 Aug 2026 02:16:25 +0200 Subject: [PATCH] feat: implemented external thinking flags and transport reasoning controls - Added the `--external-thinking` CLI flag alongside model capability checks to gate external thinking tool availability. - Updated Anthropic and Google transports to honor `forceReasoningOff` for native thinking-off controls. - Renamed the `thoughts` property and parameter to `notes` across think fixtures, tools, and tests. - Updated system prompt instructions and test suites to verify transport-specific thinking and tool activation. --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/stream.ts | 23 ++--- packages/ai/src/types.ts | 5 +- packages/ai/test/anthropic-alignment.test.ts | 83 ++++++++++++------- .../google-gemini-cli-variant-routing.test.ts | 11 +++ packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/src/cli/args.ts | 3 + packages/coding-agent/src/cli/flag-tables.ts | 1 + .../src/cli/gallery-fixtures/agentic.ts | 7 +- .../coding-agent/src/commands/launch-help.ts | 3 + .../src/config/settings-schema.ts | 2 +- packages/coding-agent/src/main.ts | 4 + .../src/prompts/system/system-prompt.md | 8 +- packages/coding-agent/src/sdk.ts | 5 +- .../coding-agent/src/session/agent-session.ts | 11 ++- .../coding-agent/src/session/session-tools.ts | 12 +++ packages/coding-agent/src/tools/index.ts | 10 ++- packages/coding-agent/src/tools/think.ts | 25 ++++-- .../coding-agent/test/flag-tables.test.ts | 12 +++ .../test/sdk-tool-activation.test.ts | 41 +++++++-- .../test/tools/think-renderer.test.ts | 6 +- packages/snapcompact/CHANGELOG.md | 2 +- packages/snapcompact/src/snapcompact.ts | 4 +- 23 files changed, 201 insertions(+), 80 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d625395f5..681c81263 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed `AWS_BEDROCK_SKIP_AUTH` failing to expose Amazon Bedrock models when AWS credential files are unavailable; the registry now recognizes the transport's explicit auth bypass without enabling the separate Bedrock Mantle provider ([#8267](https://github.com/can1357/oh-my-pi/issues/8267)). +- Fixed `forceReasoningOff` being ignored by Anthropic and Google transports, which allowed native thinking alongside a caller-supplied external scratchpad. ## [17.2.14] - 2026-08-11 diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 6aa61886f..33c46d34b 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1515,12 +1515,12 @@ function mapOptionsForApi( switch (model.api) { case "anthropic-messages": { // Explicitly disable thinking when reasoning is not specified, the caller - // disabled it, or the model doesn't support it. `disableReasoning` is a - // SimpleStreamOptions flag that never reaches AnthropicOptions on its own, - // so it must be folded into `thinkingEnabled` here (mandatory-reasoning - // models already clamp it away in normalizeMandatoryReasoningOptions). + // disabled it, an external scratchpad replaces it, or the model doesn't + // support it. These SimpleStreamOptions flags never reach AnthropicOptions + // on their own, so fold them into thinkingEnabled here (mandatory-reasoning + // models already clamp them away in normalizeMandatoryReasoningOptions). const reasoning = options?.reasoning; - if (!reasoning || !model.reasoning || options?.disableReasoning) { + if (!reasoning || !model.reasoning || options?.disableReasoning || options?.forceReasoningOff) { return castApi<"anthropic-messages">({ ...base, requestModelId: resolveWireModelId(model, undefined), @@ -1725,10 +1725,10 @@ function mapOptionsForApi( }); case "google-generative-ai": { - // Explicitly disable thinking when reasoning is not specified or model doesn't support it - // This is needed because Gemini has "dynamic thinking" enabled by default + // Explicitly disable thinking when reasoning is absent, unsupported, or + // replaced by the caller's external scratchpad. Gemini defaults thinking on. const reasoning = options?.reasoning; - if (!reasoning || !model.reasoning) { + if (!reasoning || !model.reasoning || options?.disableReasoning || options?.forceReasoningOff) { return castApi<"google-generative-ai">({ ...base, serviceTier: options?.serviceTier, @@ -1772,7 +1772,7 @@ function mapOptionsForApi( case "google-gemini-cli": { const reasoning = options?.reasoning; const toolChoice = mapGoogleToolChoice(options?.toolChoice); - if (reasoning && model.reasoning) { + if (reasoning && model.reasoning && !options?.disableReasoning && !options?.forceReasoningOff) { const effort = requireSupportedEffort(model, reasoning); // Gemini 3+ models use thinkingLevel instead of thinkingBudget @@ -1831,9 +1831,10 @@ function mapOptionsForApi( } case "google-vertex": { - // Explicitly disable thinking when reasoning is not specified or model doesn't support it + // Explicitly disable thinking when reasoning is absent, unsupported, or + // replaced by the caller's external scratchpad. const reasoning = options?.reasoning; - if (!reasoning || !model.reasoning) { + if (!reasoning || !model.reasoning || options?.disableReasoning || options?.forceReasoningOff) { return castApi<"google-vertex">({ ...base, serviceTier: options?.serviceTier, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index d89d117f2..073303089 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -479,8 +479,9 @@ export interface StreamOptions { */ statefulResponses?: boolean; /** - * Emit `reasoning: { effort: "none" }` for OpenAI Responses and Codex requests. - * Used when a caller supplies an external reasoning scratchpad; other transports ignore it. + * Disable native reasoning when the caller supplies an external scratchpad. + * OpenAI Responses emits `reasoning: { effort: "none" }`; Anthropic and + * Google transports use their native thinking-off controls. */ forceReasoningOff?: boolean; /** diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index 1985c72b2..614f52584 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -2756,39 +2756,60 @@ describe("Anthropic request fingerprint alignment", () => { }); }); - it("disables adaptive-only thinking when the caller sets disableReasoning via the public stream() path", async () => { - // #6589: disableReasoning is a SimpleStreamOptions flag that never reaches - // AnthropicOptions directly; mapOptionsForApi must fold it into - // thinkingEnabled:false so adaptive-only Opus 4.7 omits thinking + pins low - // effort instead of defaulting to adaptive-ON at the requested effort. - const { promise, resolve } = Promise.withResolvers(); - streamSimple( - buildModel({ - ...ANTHROPIC_MODEL_SPEC, - id: "claude-opus-4-7", - name: "Claude Opus 4.7", - thinking: { - mode: "anthropic-adaptive", - efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + for (const flag of ["disableReasoning", "forceReasoningOff"] as const) { + it(`disables Fable adaptive thinking when the public stream sets ${flag}`, async () => { + const { promise, resolve } = Promise.withResolvers(); + streamSimple( + buildModel({ + ...ANTHROPIC_MODEL_SPEC, + id: "claude-fable-5", + name: "Claude Fable 5", + thinking: { + mode: "anthropic-adaptive", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], + }, + }), + { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], + tools: [ + { + name: "think", + description: "Private scratchpad; not shown to user.", + strict: true, + parameters: { + type: "object", + properties: { thoughts: { type: "string" } }, + required: ["thoughts"], + additionalProperties: false, + } as TJsonSchema, + }, + ], }, - }), - { - systemPrompt: ["Stay concise."], - messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], - }, - { - apiKey: "sk-ant-oat-test", - signal: createAbortedSignal(), - reasoning: Effort.High, - disableReasoning: true, - onPayload: payload => resolve(payload), - }, - ); - const payload = (await promise) as { thinking?: unknown; output_config?: { effort?: string } }; + { + apiKey: "sk-ant-oat-test", + signal: createAbortedSignal(), + reasoning: Effort.High, + [flag]: true, + onPayload: payload => resolve(payload), + }, + ); + const payload = (await promise) as { + thinking?: unknown; + output_config?: { effort?: string }; + tools?: Array<{ + eager_input_streaming?: boolean; + input_schema?: { properties?: Record; required?: string[] }; + }>; + }; - expect(payload.thinking).toBeUndefined(); - expect(payload.output_config).toEqual({ effort: "low" }); - }); + expect(payload.thinking).toBeUndefined(); + expect(payload.output_config).toEqual({ effort: "low" }); + expect(payload.tools?.[0]?.eager_input_streaming).toBe(true); + expect(payload.tools?.[0]?.input_schema?.properties).toHaveProperty("thoughts"); + expect(payload.tools?.[0]?.input_schema?.required).toEqual(["thoughts"]); + }); + } it("deletes thinking without an effort pin for non-adaptive reasoning models on forced tool choice", async () => { // Budget-thinking models (Sonnet 4.5) turn thinking off by simple omission, diff --git a/packages/ai/test/google-gemini-cli-variant-routing.test.ts b/packages/ai/test/google-gemini-cli-variant-routing.test.ts index b3f256c49..b956a88f6 100644 --- a/packages/ai/test/google-gemini-cli-variant-routing.test.ts +++ b/packages/ai/test/google-gemini-cli-variant-routing.test.ts @@ -103,6 +103,7 @@ function unroutedModel(): Model<"google-gemini-cli"> { async function captureRequest( model: Model<"google-gemini-cli">, reasoning: Effort | undefined, + options: { forceReasoningOff?: boolean } = {}, ): Promise<{ body: CapturedRequestBody; attributedModel: string }> { let requestBody: string | undefined; const fetchMock: FetchImpl = (_input, init) => { @@ -112,6 +113,7 @@ async function captureRequest( const stream = streamSimple(model, context, { apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), reasoning, + forceReasoningOff: options.forceReasoningOff, fetch: fetchMock, }); const result = await stream.result(); @@ -148,6 +150,15 @@ describe("google-gemini-cli effort-tier variant routing", () => { }); }); + it("routes to the explicit off wire shape when an external work log replaces reasoning", async () => { + const off = await captureRequest(collapsedFlashModel(), Effort.High, { forceReasoningOff: true }); + expect(off.body.model).toBe("gemini-3.5-flash-extra-low"); + expect(off.body.request?.generationConfig?.thinkingConfig).toEqual({ + includeThoughts: false, + thinkingBudget: 0, + }); + }); + it("routes claude pairs to the bare id when off without wire suppression", async () => { const off = await captureRequest(collapsedClaudeModel(), undefined); expect(off.body.model).toBe("claude-sonnet-4-6"); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index fdb92c476..2940798f3 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,7 @@ ### Added +- Added `--external-thinking` CLI flag to force external thinking tool activation - Added `omp compress`, a command that rewrites a text file into the dense prompt register through a two-tool agent loop: the agent submits a draft with `rewrite` plus every loss it accepted, the command replies with the measured size and that loss list and asks for a verdict, and only an `approve` on a reviewed draft is written. Reports go to stderr and the approved text to stdout, so `omp compress f.md > out.md` yields just the compressed text; `-o` writes a file, `-i` rewrites in place. The session is deliberately sealed — the default system prompt is replaced rather than appended, and skills, rules, `AGENTS.md` context files, prompt templates, slash commands, extensions, MCP, IRC, and LSP are all disabled — and the source document is quoted as nonce-delimited inert data so the directives it contains are compressed instead of obeyed. - `omp compress` accepts multiple files and glob patterns, compresses up to `-n` of them concurrently (default 4, one isolated session each), and renders the same TTY completion bar as `omp cleanse`; multi-file runs require `-i` since one `--out` cannot hold many files, and a file that fails is reported without cancelling its peers. The bar itself moved to `src/cli/progress-reporter.ts` and is now shared with `omp cleanse` instead of duplicated. - `omp cleanse` discovers far more tooling: staticcheck and golangci-lint for Go; mypy, pylint, flake8, ty, and basedpyright for Python; oxlint, `deno lint`, stylelint, and vue-tsc (preferred over tsc for roots containing `.vue` files) for the JS/TS ecosystem; plus actionlint for GitHub workflows. Alternative tools without a config marker (staticcheck, actionlint) are skipped silently when the binary is missing instead of cluttering the skip report. @@ -12,6 +13,7 @@ ### Changed +- Renamed the `think` tool's `thoughts` parameter to `notes` and restricted tool availability to models supporting external thinking - `omp cleanse` default subagent cap raised from 8 to 32 (`--agents`/`-n` still overrides). ### Fixed diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index f8785e898..a2785eb96 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -48,6 +48,7 @@ export interface Args { serviceTier?: ServiceTierOpenAISettingValue; hideThinking?: boolean; advisor?: boolean; + externalThinking?: boolean; continue?: boolean; resume?: string | true; fromClaude?: boolean; @@ -255,6 +256,8 @@ export function parseArgs(inputArgs: string[], extensionFlags?: Map = new Set([ "--no-pty", "--hide-thinking", "--advisor", + "--external-thinking", "--prewalk", "--no-prewalk", "--plan-yolo", diff --git a/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts index 568d8fbd3..89d27fc33 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts @@ -363,13 +363,12 @@ export const agenticFixtures: Record = { think: { label: "Think", - // Streaming: scratchpad text still arriving. + // Streaming: scratchpad thoughts still arriving. streamingArgs: { - thoughts: "The retry loop re-reads the config after every failure — that explains the doubled latency.", + thoughts: "The retry loop re-reads the config after every failure, which explains the doubled latency.", }, args: { - thoughts: - "The retry loop re-reads the config after every failure — that explains the doubled latency. Cache the parsed config outside the loop, then re-check the invalidation path before answering.", + thoughts: "The retry loop re-reads the config after every failure, which explains the doubled latency. Cache the parsed config outside the loop, then re-check the invalidation path.", }, result: { content: [{ type: "text", text: "------" }], diff --git a/packages/coding-agent/src/commands/launch-help.ts b/packages/coding-agent/src/commands/launch-help.ts index 98a9ea4e1..ffa7b0b07 100644 --- a/packages/coding-agent/src/commands/launch-help.ts +++ b/packages/coding-agent/src/commands/launch-help.ts @@ -77,6 +77,9 @@ export const launchHelp = { advisor: Flags.boolean({ description: "Enable the advisor runtime (passively reviews each turn and injects notes)", }), + "external-thinking": Flags.boolean({ + description: "Use a private scratchpad while disabling supported GPT, Claude, and Gemini reasoning", + }), hook: Flags.string({ description: "Load a hook/extension file (can be used multiple times)", multiple: true }), extension: Flags.string({ char: "e", diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 50705885b..000f5cef3 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -1146,7 +1146,7 @@ export const SETTINGS_SCHEMA = { tab: "model", group: "Thinking", label: "External Thinking", - description: "Use a private think tool and send reasoning effort off to GPT Responses models", + description: "Private scratchpad; not shown to user. Disables supported GPT, Claude, and Gemini reasoning", }, }, diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index c9de644e9..8872c5570 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -1333,6 +1333,10 @@ export async function runRootCommand( if (parsedArgs.advisor) { settingsInstance.override("advisor.enabled", true); } + // Apply --external-thinking CLI flag (ephemeral, not persisted) + if (parsedArgs.externalThinking) { + settingsInstance.override("externalThinking", true); + } await logger.time( "initTheme:final", diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index b457d61b7..58fe60192 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -95,12 +95,8 @@ Write JSON args as `content` to `xd://` via `{{toolRefs.write}}`. Invalid {{/if}} {{#has tools "think"}} -§ Reasoning -`{{toolRefs.think}}`: private scratchpad; unwritten reasoning is unworked. -- MUST call before turn's first action and before expensive-to-undo edit, destructive command, or final answer. -- Restate ask/constraints; ordered subproblems; explicitly solve/intermediate-result each; enumerate/resolve cases. -- Check claims against constraints, a boundary/degenerate case, and likely error. Failed check → redo step, NEVER patch conclusion. -- Re-call only for material new state: plan-changing result, failed check, unopened subproblem; NEVER narrate progress/restate recorded work. +§ Scratchpad +`{{toolRefs.think}}`: private scratchpad; not shown to user. {{/has}} § Tool Policy diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 127c2d3af..5bfaae569 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -201,6 +201,7 @@ import { ReadTool, releaseComputerSessionsForOwner, resolveMountedXdevExecutable, + supportsExternalThinking, type Tool, type ToolSession, WebSearchTool, @@ -3238,7 +3239,9 @@ async function createAgentSessionScoped(options: CreateAgentSessionOptions): Pro } } const externalThinking = - settings.get("externalThinking") && agent.state.tools.some(tool => tool.name === "think"); + settings.get("externalThinking") && + agent.state.tools.some(tool => tool.name === "think") && + supportsExternalThinking(streamModel); return settingsAwareStreamFn(streamModel, context, { ...streamOptions, forceReasoningOff: externalThinking || streamOptions?.forceReasoningOff, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 0b2b8684c..5b0475f3c 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -199,6 +199,7 @@ import { PROPOSE_DEVICE_NAME, writeDeviceDispatch, } from "../tools/resolve"; +import { supportsExternalThinking } from "../tools/think"; import type { TodoPhase } from "../tools/todo"; import { ToolError } from "../tools/tool-errors"; import { parseCommandArgs } from "../utils/command-args"; @@ -5188,10 +5189,7 @@ export class AgentSession { !hasPendingUserDirective && this.settings.get("externalThinking") && this.getEnabledToolNames().includes("think") && - activeModel && - (activeModel.api === "openai-responses" || - activeModel.api === "azure-openai-responses" || - activeModel.api === "openai-codex-responses") + supportsExternalThinking(activeModel) ? buildNamedToolChoice("think", activeModel) : undefined; const eagerTodoPrelude = @@ -7068,6 +7066,11 @@ export class AgentSession { } catch (error) { logger.warn("inspect_image reconcile after model change failed", { error: String(error) }); } + try { + await this.#tools.reconcileThinkTool(); + } catch (error) { + logger.warn("think tool reconcile after model change failed", { error: String(error) }); + } } #closeCodexProviderSessionsForHistoryRewrite(): void { diff --git a/packages/coding-agent/src/session/session-tools.ts b/packages/coding-agent/src/session/session-tools.ts index f22ddc5bd..fcd339e97 100644 --- a/packages/coding-agent/src/session/session-tools.ts +++ b/packages/coding-agent/src/session/session-tools.ts @@ -21,6 +21,7 @@ import { usesCodexTaskPrompt } from "../task/prompt-policy"; import { isMCPToolName, normalizeToolNames } from "../tools/builtin-names"; import { computerExposureMode } from "../tools/computer/exposure"; import { wrapToolWithMetaNotice } from "../tools/output-meta"; +import { supportsExternalThinking } from "../tools/think"; import { ToolAbortError, ToolError } from "../tools/tool-errors"; import { isMountableUnderXdev, listXdevTools, type XdevState, xdevDocsFor, xdevEntries } from "../tools/xdev"; import { type EditMode, resolveEditMode } from "../utils/edit-mode"; @@ -1171,6 +1172,17 @@ export class SessionTools { * @returns false when enabling was requested but this session cannot build the tool. */ setThinkToolEnabled(enabled: boolean): Promise { + return this.#setThinkToolActive(enabled && supportsExternalThinking(this.#host.model())); + } + + /** Reconciles the external scratchpad after the active model changes. */ + reconcileThinkTool(): Promise { + return this.#setThinkToolActive( + this.#host.settings.get("externalThinking") && supportsExternalThinking(this.#host.model()), + ); + } + + #setThinkToolActive(enabled: boolean): Promise { return this.runToolRegistryMutation(async () => { const active = this.getEnabledToolNames(); if (!enabled) { diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 19603808b..0890fe1fd 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -62,7 +62,7 @@ import { wrapToolWithMetaNotice } from "./output-meta"; import { ReadTool } from "./read"; import type { PlanProposalHandler } from "./resolve"; import { SecurityScanTool } from "./security-scan"; -import { ThinkTool } from "./think"; +import { supportsExternalThinking, ThinkTool } from "./think"; import { type TodoPhase, TodoTool } from "./todo"; import { WriteTool } from "./write"; import { isMountableUnderXdev, type XdevState } from "./xdev"; @@ -467,6 +467,8 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P : undefined; const goalEnabled = session.settings.get("goal.enabled"); const goalModeActive = !restrictToolNames && goalEnabled && session.getGoalModeState?.()?.enabled === true; + const externalThinkingActive = + session.settings.get("externalThinking") && supportsExternalThinking(session.getActiveModel?.()); if (goalModeActive && requestedTools && !requestedTools.includes("goal")) { requestedTools.push("goal"); } @@ -565,7 +567,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P if (session.settings.get("memory.backend") === "mnemopi" && !requestedTools.includes("memory_edit")) { requestedTools.push("memory_edit"); } - if (session.settings.get("externalThinking") && !requestedTools.includes("think")) { + if (externalThinkingActive && !requestedTools.includes("think")) { requestedTools.push("think"); } // Auto-learn tools are gated by `autolearn.enabled` but, like the memory @@ -610,7 +612,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P if (name === "inspect_image") return isInspectImageToolActive(session); if (name === "web_search") return session.settings.get("web_search.enabled"); if (name === "security_scan") return session.settings.get("security.enabled"); - if (name === "think") return session.settings.get("externalThinking"); + if (name === "think") return externalThinkingActive; if (name === "ask") return session.settings.get("ask.enabled"); if (name === "browser") return session.settings.get("browser.enabled"); if (name === "computer") return session.settings.get("computer.enabled"); @@ -657,7 +659,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P ...Object.entries(BUILTIN_TOOLS) .filter(([name]) => isToolAllowed(name)) .map(([name, factory]) => [name, factory] as const), - ...(session.settings.get("externalThinking") ? ([["think", HIDDEN_TOOLS.think]] as const) : []), + ...(externalThinkingActive ? ([["think", HIDDEN_TOOLS.think]] as const) : []), ...(includeYield ? ([["yield", HIDDEN_TOOLS.yield]] as const) : []), ...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []), ]; diff --git a/packages/coding-agent/src/tools/think.ts b/packages/coding-agent/src/tools/think.ts index 8cbb3afaa..db603acf0 100644 --- a/packages/coding-agent/src/tools/think.ts +++ b/packages/coding-agent/src/tools/think.ts @@ -1,13 +1,27 @@ import { type } from "@oh-my-pi/omptype"; import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; +import type { Model } from "@oh-my-pi/pi-ai"; import { type Component, Markdown } from "@oh-my-pi/pi-tui"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { getMarkdownTheme, type Theme } from "../modes/theme/theme"; +/** Whether a model transport can suppress native reasoning while private scratchpad thoughts are active. */ +export function supportsExternalThinking(model: Model | null | undefined): boolean { + return ( + model?.api === "openai-responses" || + model?.api === "azure-openai-responses" || + model?.api === "openai-codex-responses" || + model?.api === "anthropic-messages" || + model?.api === "google-generative-ai" || + model?.api === "google-gemini-cli" || + model?.api === "google-vertex" + ); +} + const thinkSchema = type({ - thoughts: type("string").describe("private scratchpad reasoning to retain before the next response"), + thoughts: type("string").describe("private scratchpad; not shown to user"), "+": "reject", -}).describe("record private intermediate reasoning before answering"); +}).describe("private scratchpad; not shown to user"); type ThinkParams = typeof thinkSchema.infer; @@ -36,14 +50,13 @@ interface ThinkToolDetails { recorded: true; } -/** Records private intermediate reasoning while native GPT reasoning is disabled. */ +/** Records private scratchpad thoughts while native model reasoning is disabled. */ export class ThinkTool implements AgentTool { readonly name = "think"; readonly approval = "read" as const; readonly label = "Think"; - readonly summary = "Record private intermediate reasoning before answering"; - readonly description = - "Use this private scratchpad to plan, derive, or check work before answering. Record only materially new reasoning. The user does not see this tool activity."; + readonly summary = "Record private scratchpad thoughts"; + readonly description = "private scratchpad; not shown to user"; readonly parameters = thinkSchema; readonly strict = true; readonly intent = "omit" as const; diff --git a/packages/coding-agent/test/flag-tables.test.ts b/packages/coding-agent/test/flag-tables.test.ts index 23e3b52d1..62173cc08 100644 --- a/packages/coding-agent/test/flag-tables.test.ts +++ b/packages/coding-agent/test/flag-tables.test.ts @@ -56,6 +56,18 @@ describe("OPTIONAL_VALUE_FLAGS table is honored by args.ts parseArgs", () => { } }); +describe("--external-thinking", () => { + it("enables external thinking without consuming the initial message", () => { + const result = parseArgs(["--external-thinking", "check this"]); + + expect(result.externalThinking).toBe(true); + expect(result.messages).toEqual(["check this"]); + }); + + it("stays unset when omitted", () => { + expect(parseArgs([]).externalThinking).toBeUndefined(); + }); +}); describe("--session-dir", () => { it("uses PI_CODING_AGENT_SESSION_DIR unless the CLI flag overrides it", () => { const previous = Bun.env.PI_CODING_AGENT_SESSION_DIR; diff --git a/packages/coding-agent/test/sdk-tool-activation.test.ts b/packages/coding-agent/test/sdk-tool-activation.test.ts index 8e5c87f80..ef28a21a5 100644 --- a/packages/coding-agent/test/sdk-tool-activation.test.ts +++ b/packages/coding-agent/test/sdk-tool-activation.test.ts @@ -109,6 +109,15 @@ describe("createAgentSession defaultInactive tool activation", () => { workspaceTree: { rootPath: tempDir, rendered: "", truncated: false, totalLines: 0, agentsMdFiles: [] }, }); + const requireBundledModel = ( + provider: "anthropic" | "google-antigravity" | "openai" | "xai", + id: string, + ): Model => { + const bundled = getBundledModel(provider, id); + if (!bundled) throw new Error(`Expected ${provider}/${id} model to exist`); + return bundled; + }; + afterEach(() => { for (const tempDir of tempDirs.splice(0)) { removeSyncWithRetries(tempDir); @@ -152,6 +161,7 @@ describe("createAgentSession defaultInactive tool activation", () => { const settings = Settings.isolated(); const { session } = await createAgentSession({ ...baseOptions(tempDir), + model: requireBundledModel("openai", "gpt-5"), settings, }); @@ -174,18 +184,40 @@ describe("createAgentSession defaultInactive tool activation", () => { } }); - it("activates the private think tool at startup when external thinking is configured", async () => { + it("exposes the private think tool only on transports that can disable native reasoning", async () => { const tempDir = makeTempDir(); const settings = Settings.isolated({ externalThinking: true }); + const unsupported = requireBundledModel("xai", "grok-4"); + const fable = requireBundledModel("anthropic", "claude-fable-5"); + const responses = requireBundledModel("openai", "gpt-5"); + const gemini = requireBundledModel("google-antigravity", "gemini-3.6-flash"); const { session } = await createAgentSession({ ...baseOptions(tempDir), settings, + model: unsupported, }); + const authStorage = session.modelRegistry.authStorage; + authStorage.setRuntimeApiKey("anthropic", "test-key"); + authStorage.setRuntimeApiKey("openai", "test-key"); + authStorage.setRuntimeApiKey("google-antigravity", "test-key"); + authStorage.setRuntimeApiKey("xai", "test-key"); try { + expect(session.getActiveToolNames()).not.toContain("think"); + + await session.setModel(fable); expect(session.getToolByName("think")).toBeDefined(); expect(session.getActiveToolNames()).toContain("think"); - expect(session.getXdevToolEntries().map(entry => entry.name)).not.toContain("think"); + expect(session.systemPrompt.join("\n")).toContain("private scratchpad; not shown to user"); + + await session.setModel(responses); + expect(session.getActiveToolNames()).toContain("think"); + await session.setModel(gemini); + expect(session.getActiveToolNames()).toContain("think"); + + await session.setModel(unsupported); + expect(session.getActiveToolNames()).not.toContain("think"); + expect(session.systemPrompt.join("\n")).not.toContain("private scratchpad; not shown to user"); } finally { await session.dispose(); } @@ -217,7 +249,7 @@ describe("createAgentSession defaultInactive tool activation", () => { fetch: async request => { requestTexts.push(await request.text()); if (requestTexts.length === 1) { - const argumentsJson = JSON.stringify({ thoughts: "Checked the request before answering." }); + const argumentsJson = JSON.stringify({ notes: "Checked the request before answering." }); return sse([ { type: "response.output_item.added", @@ -267,8 +299,7 @@ describe("createAgentSession defaultInactive tool activation", () => { ]); }, }); - const model = getBundledModel("openai", "gpt-5"); - if (!model) throw new Error("Expected gpt-5 model to exist"); + const model = requireBundledModel("openai", "gpt-5"); // The prompt preflight validates the key through the registry (not the // per-request `getApiKey` override), so seed it for keyless CI runners. modelRegistry.authStorage.setRuntimeApiKey("openai", "test-key"); diff --git a/packages/coding-agent/test/tools/think-renderer.test.ts b/packages/coding-agent/test/tools/think-renderer.test.ts index 573e3b10a..07da03bc6 100644 --- a/packages/coding-agent/test/tools/think-renderer.test.ts +++ b/packages/coding-agent/test/tools/think-renderer.test.ts @@ -13,7 +13,7 @@ describe("thinkToolRenderer", () => { const uiTheme = theme!; const callComponent = thinkToolRenderer.renderCall( - { thoughts: "Analyzing the solution step by step." }, + { thoughts: "Cache the parsed config, then check invalidation." }, { expanded: true, isPartial: false }, uiTheme, ); @@ -22,8 +22,8 @@ describe("thinkToolRenderer", () => { const lines = callComponent.render(100); const fullText = lines.join("\n"); - expect(fullText).toContain("Analyzing the solution step by step."); - expect(fullText).toContain(uiTheme.fg("thinkingText", "Analyzing the solution step by step.")); + expect(fullText).toContain("Cache the parsed config, then check invalidation."); + expect(fullText).toContain(uiTheme.fg("thinkingText", "Cache the parsed config, then check invalidation.")); }); it("has inline set to true", () => { diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index f694be1d7..9441891f7 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `claude-opus-5` (and later Opus lines) falling back to the 1568px frame instead of the 1932px high-res tier, so snapcompact archives for the current flagship Opus now retain ~33% more history at the same per-frame bill ([#8256](https://github.com/can1357/oh-my-pi/issues/8256)). +- Fixed case-sensitivity in Anthropic model ID parsing for high-res frame selection ## [17.1.5] - 2026-07-27 diff --git a/packages/snapcompact/src/snapcompact.ts b/packages/snapcompact/src/snapcompact.ts index a07c8ef0e..b72b1da10 100644 --- a/packages/snapcompact/src/snapcompact.ts +++ b/packages/snapcompact/src/snapcompact.ts @@ -364,7 +364,9 @@ const MODEL_VARIANTS: readonly (readonly [RegExp, IdealShape])[] = [ /** Eval-ideal format for a model id, or undefined when unmeasured. */ export function idealShapeVariant(modelId: string): IdealShape | undefined { - const anthropic = parseAnthropicModel(modelId); + // The catalog parser is case-sensitive; the regex rules below are not. + // Normalize so mixed-case gateway ids keep matching the Anthropic tier. + const anthropic = parseAnthropicModel(modelId.toLowerCase()); if ( anthropic && (isFableOrMythos(anthropic.kind) || (anthropic.kind === "opus" && semverGte(anthropic.version, "4.7")))