feat: implemented external thinking flags and transport reasoning controls

- Added the `--external-thinking` CLI flag alongside model capability checks to gate external thinking tool availability.
- Updated Anthropic and Google transports to honor `forceReasoningOff` for native thinking-off controls.
- Renamed the `thoughts` property and parameter to `notes` across think fixtures, tools, and tests.
- Updated system prompt instructions and test suites to verify transport-specific thinking and tool activation.
This commit is contained in:
can1357
2026-08-12 02:16:25 +02:00
parent 7e2663360f
commit 19c0afcc0d
23 changed files with 201 additions and 80 deletions
+1
View File
@@ -5,6 +5,7 @@
### Fixed ### Fixed
- Fixed `AWS_BEDROCK_SKIP_AUTH` failing to expose Amazon Bedrock models when AWS credential files are unavailable; the registry now recognizes the transport's explicit auth bypass without enabling the separate Bedrock Mantle provider ([#8267](https://github.com/can1357/oh-my-pi/issues/8267)). - Fixed `AWS_BEDROCK_SKIP_AUTH` failing to expose Amazon Bedrock models when AWS credential files are unavailable; the registry now recognizes the transport's explicit auth bypass without enabling the separate Bedrock Mantle provider ([#8267](https://github.com/can1357/oh-my-pi/issues/8267)).
- Fixed `forceReasoningOff` being ignored by Anthropic and Google transports, which allowed native thinking alongside a caller-supplied external scratchpad.
## [17.2.14] - 2026-08-11 ## [17.2.14] - 2026-08-11
+12 -11
View File
@@ -1515,12 +1515,12 @@ function mapOptionsForApi<TApi extends Api>(
switch (model.api) { switch (model.api) {
case "anthropic-messages": { case "anthropic-messages": {
// Explicitly disable thinking when reasoning is not specified, the caller // Explicitly disable thinking when reasoning is not specified, the caller
// disabled it, or the model doesn't support it. `disableReasoning` is a // disabled it, an external scratchpad replaces it, or the model doesn't
// SimpleStreamOptions flag that never reaches AnthropicOptions on its own, // support it. These SimpleStreamOptions flags never reach AnthropicOptions
// so it must be folded into `thinkingEnabled` here (mandatory-reasoning // on their own, so fold them into thinkingEnabled here (mandatory-reasoning
// models already clamp it away in normalizeMandatoryReasoningOptions). // models already clamp them away in normalizeMandatoryReasoningOptions).
const reasoning = options?.reasoning; const reasoning = options?.reasoning;
if (!reasoning || !model.reasoning || options?.disableReasoning) { if (!reasoning || !model.reasoning || options?.disableReasoning || options?.forceReasoningOff) {
return castApi<"anthropic-messages">({ return castApi<"anthropic-messages">({
...base, ...base,
requestModelId: resolveWireModelId(model, undefined), requestModelId: resolveWireModelId(model, undefined),
@@ -1725,10 +1725,10 @@ function mapOptionsForApi<TApi extends Api>(
}); });
case "google-generative-ai": { case "google-generative-ai": {
// Explicitly disable thinking when reasoning is not specified or model doesn't support it // Explicitly disable thinking when reasoning is absent, unsupported, or
// This is needed because Gemini has "dynamic thinking" enabled by default // replaced by the caller's external scratchpad. Gemini defaults thinking on.
const reasoning = options?.reasoning; const reasoning = options?.reasoning;
if (!reasoning || !model.reasoning) { if (!reasoning || !model.reasoning || options?.disableReasoning || options?.forceReasoningOff) {
return castApi<"google-generative-ai">({ return castApi<"google-generative-ai">({
...base, ...base,
serviceTier: options?.serviceTier, serviceTier: options?.serviceTier,
@@ -1772,7 +1772,7 @@ function mapOptionsForApi<TApi extends Api>(
case "google-gemini-cli": { case "google-gemini-cli": {
const reasoning = options?.reasoning; const reasoning = options?.reasoning;
const toolChoice = mapGoogleToolChoice(options?.toolChoice); const toolChoice = mapGoogleToolChoice(options?.toolChoice);
if (reasoning && model.reasoning) { if (reasoning && model.reasoning && !options?.disableReasoning && !options?.forceReasoningOff) {
const effort = requireSupportedEffort(model, reasoning); const effort = requireSupportedEffort(model, reasoning);
// Gemini 3+ models use thinkingLevel instead of thinkingBudget // Gemini 3+ models use thinkingLevel instead of thinkingBudget
@@ -1831,9 +1831,10 @@ function mapOptionsForApi<TApi extends Api>(
} }
case "google-vertex": { case "google-vertex": {
// Explicitly disable thinking when reasoning is not specified or model doesn't support it // Explicitly disable thinking when reasoning is absent, unsupported, or
// replaced by the caller's external scratchpad.
const reasoning = options?.reasoning; const reasoning = options?.reasoning;
if (!reasoning || !model.reasoning) { if (!reasoning || !model.reasoning || options?.disableReasoning || options?.forceReasoningOff) {
return castApi<"google-vertex">({ return castApi<"google-vertex">({
...base, ...base,
serviceTier: options?.serviceTier, serviceTier: options?.serviceTier,
+3 -2
View File
@@ -479,8 +479,9 @@ export interface StreamOptions {
*/ */
statefulResponses?: boolean; statefulResponses?: boolean;
/** /**
* Emit `reasoning: { effort: "none" }` for OpenAI Responses and Codex requests. * Disable native reasoning when the caller supplies an external scratchpad.
* Used when a caller supplies an external reasoning scratchpad; other transports ignore it. * OpenAI Responses emits `reasoning: { effort: "none" }`; Anthropic and
* Google transports use their native thinking-off controls.
*/ */
forceReasoningOff?: boolean; forceReasoningOff?: boolean;
/** /**
+52 -31
View File
@@ -2756,39 +2756,60 @@ describe("Anthropic request fingerprint alignment", () => {
}); });
}); });
it("disables adaptive-only thinking when the caller sets disableReasoning via the public stream() path", async () => { for (const flag of ["disableReasoning", "forceReasoningOff"] as const) {
// #6589: disableReasoning is a SimpleStreamOptions flag that never reaches it(`disables Fable adaptive thinking when the public stream sets ${flag}`, async () => {
// AnthropicOptions directly; mapOptionsForApi must fold it into const { promise, resolve } = Promise.withResolvers<unknown>();
// thinkingEnabled:false so adaptive-only Opus 4.7 omits thinking + pins low streamSimple(
// effort instead of defaulting to adaptive-ON at the requested effort. buildModel({
const { promise, resolve } = Promise.withResolvers<unknown>(); ...ANTHROPIC_MODEL_SPEC,
streamSimple( id: "claude-fable-5",
buildModel({ name: "Claude Fable 5",
...ANTHROPIC_MODEL_SPEC, thinking: {
id: "claude-opus-4-7", mode: "anthropic-adaptive",
name: "Claude Opus 4.7", efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max],
thinking: { },
mode: "anthropic-adaptive", }),
efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max], {
systemPrompt: ["Stay concise."],
messages: [{ role: "user", content: "Hi", timestamp: Date.now() }],
tools: [
{
name: "think",
description: "Private scratchpad; not shown to user.",
strict: true,
parameters: {
type: "object",
properties: { thoughts: { type: "string" } },
required: ["thoughts"],
additionalProperties: false,
} as TJsonSchema,
},
],
}, },
}), {
{ apiKey: "sk-ant-oat-test",
systemPrompt: ["Stay concise."], signal: createAbortedSignal(),
messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], reasoning: Effort.High,
}, [flag]: true,
{ onPayload: payload => resolve(payload),
apiKey: "sk-ant-oat-test", },
signal: createAbortedSignal(), );
reasoning: Effort.High, const payload = (await promise) as {
disableReasoning: true, thinking?: unknown;
onPayload: payload => resolve(payload), output_config?: { effort?: string };
}, tools?: Array<{
); eager_input_streaming?: boolean;
const payload = (await promise) as { thinking?: unknown; output_config?: { effort?: string } }; input_schema?: { properties?: Record<string, unknown>; required?: string[] };
}>;
};
expect(payload.thinking).toBeUndefined(); expect(payload.thinking).toBeUndefined();
expect(payload.output_config).toEqual({ effort: "low" }); expect(payload.output_config).toEqual({ effort: "low" });
}); expect(payload.tools?.[0]?.eager_input_streaming).toBe(true);
expect(payload.tools?.[0]?.input_schema?.properties).toHaveProperty("thoughts");
expect(payload.tools?.[0]?.input_schema?.required).toEqual(["thoughts"]);
});
}
it("deletes thinking without an effort pin for non-adaptive reasoning models on forced tool choice", async () => { it("deletes thinking without an effort pin for non-adaptive reasoning models on forced tool choice", async () => {
// Budget-thinking models (Sonnet 4.5) turn thinking off by simple omission, // Budget-thinking models (Sonnet 4.5) turn thinking off by simple omission,
@@ -103,6 +103,7 @@ function unroutedModel(): Model<"google-gemini-cli"> {
async function captureRequest( async function captureRequest(
model: Model<"google-gemini-cli">, model: Model<"google-gemini-cli">,
reasoning: Effort | undefined, reasoning: Effort | undefined,
options: { forceReasoningOff?: boolean } = {},
): Promise<{ body: CapturedRequestBody; attributedModel: string }> { ): Promise<{ body: CapturedRequestBody; attributedModel: string }> {
let requestBody: string | undefined; let requestBody: string | undefined;
const fetchMock: FetchImpl = (_input, init) => { const fetchMock: FetchImpl = (_input, init) => {
@@ -112,6 +113,7 @@ async function captureRequest(
const stream = streamSimple(model, context, { const stream = streamSimple(model, context, {
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
reasoning, reasoning,
forceReasoningOff: options.forceReasoningOff,
fetch: fetchMock, fetch: fetchMock,
}); });
const result = await stream.result(); const result = await stream.result();
@@ -148,6 +150,15 @@ describe("google-gemini-cli effort-tier variant routing", () => {
}); });
}); });
it("routes to the explicit off wire shape when an external work log replaces reasoning", async () => {
const off = await captureRequest(collapsedFlashModel(), Effort.High, { forceReasoningOff: true });
expect(off.body.model).toBe("gemini-3.5-flash-extra-low");
expect(off.body.request?.generationConfig?.thinkingConfig).toEqual({
includeThoughts: false,
thinkingBudget: 0,
});
});
it("routes claude pairs to the bare id when off without wire suppression", async () => { it("routes claude pairs to the bare id when off without wire suppression", async () => {
const off = await captureRequest(collapsedClaudeModel(), undefined); const off = await captureRequest(collapsedClaudeModel(), undefined);
expect(off.body.model).toBe("claude-sonnet-4-6"); expect(off.body.model).toBe("claude-sonnet-4-6");
+2
View File
@@ -4,6 +4,7 @@
### Added ### Added
- Added `--external-thinking` CLI flag to force external thinking tool activation
- Added `omp compress`, a command that rewrites a text file into the dense prompt register through a two-tool agent loop: the agent submits a draft with `rewrite` plus every loss it accepted, the command replies with the measured size and that loss list and asks for a verdict, and only an `approve` on a reviewed draft is written. Reports go to stderr and the approved text to stdout, so `omp compress f.md > out.md` yields just the compressed text; `-o` writes a file, `-i` rewrites in place. The session is deliberately sealed — the default system prompt is replaced rather than appended, and skills, rules, `AGENTS.md` context files, prompt templates, slash commands, extensions, MCP, IRC, and LSP are all disabled — and the source document is quoted as nonce-delimited inert data so the directives it contains are compressed instead of obeyed. - Added `omp compress`, a command that rewrites a text file into the dense prompt register through a two-tool agent loop: the agent submits a draft with `rewrite` plus every loss it accepted, the command replies with the measured size and that loss list and asks for a verdict, and only an `approve` on a reviewed draft is written. Reports go to stderr and the approved text to stdout, so `omp compress f.md > out.md` yields just the compressed text; `-o` writes a file, `-i` rewrites in place. The session is deliberately sealed — the default system prompt is replaced rather than appended, and skills, rules, `AGENTS.md` context files, prompt templates, slash commands, extensions, MCP, IRC, and LSP are all disabled — and the source document is quoted as nonce-delimited inert data so the directives it contains are compressed instead of obeyed.
- `omp compress` accepts multiple files and glob patterns, compresses up to `-n` of them concurrently (default 4, one isolated session each), and renders the same TTY completion bar as `omp cleanse`; multi-file runs require `-i` since one `--out` cannot hold many files, and a file that fails is reported without cancelling its peers. The bar itself moved to `src/cli/progress-reporter.ts` and is now shared with `omp cleanse` instead of duplicated. - `omp compress` accepts multiple files and glob patterns, compresses up to `-n` of them concurrently (default 4, one isolated session each), and renders the same TTY completion bar as `omp cleanse`; multi-file runs require `-i` since one `--out` cannot hold many files, and a file that fails is reported without cancelling its peers. The bar itself moved to `src/cli/progress-reporter.ts` and is now shared with `omp cleanse` instead of duplicated.
- `omp cleanse` discovers far more tooling: staticcheck and golangci-lint for Go; mypy, pylint, flake8, ty, and basedpyright for Python; oxlint, `deno lint`, stylelint, and vue-tsc (preferred over tsc for roots containing `.vue` files) for the JS/TS ecosystem; plus actionlint for GitHub workflows. Alternative tools without a config marker (staticcheck, actionlint) are skipped silently when the binary is missing instead of cluttering the skip report. - `omp cleanse` discovers far more tooling: staticcheck and golangci-lint for Go; mypy, pylint, flake8, ty, and basedpyright for Python; oxlint, `deno lint`, stylelint, and vue-tsc (preferred over tsc for roots containing `.vue` files) for the JS/TS ecosystem; plus actionlint for GitHub workflows. Alternative tools without a config marker (staticcheck, actionlint) are skipped silently when the binary is missing instead of cluttering the skip report.
@@ -12,6 +13,7 @@
### Changed ### Changed
- Renamed the `think` tool's `thoughts` parameter to `notes` and restricted tool availability to models supporting external thinking
- `omp cleanse` default subagent cap raised from 8 to 32 (`--agents`/`-n` still overrides). - `omp cleanse` default subagent cap raised from 8 to 32 (`--agents`/`-n` still overrides).
### Fixed ### Fixed
+3
View File
@@ -48,6 +48,7 @@ export interface Args {
serviceTier?: ServiceTierOpenAISettingValue; serviceTier?: ServiceTierOpenAISettingValue;
hideThinking?: boolean; hideThinking?: boolean;
advisor?: boolean; advisor?: boolean;
externalThinking?: boolean;
continue?: boolean; continue?: boolean;
resume?: string | true; resume?: string | true;
fromClaude?: boolean; fromClaude?: boolean;
@@ -255,6 +256,8 @@ export function parseArgs(inputArgs: string[], extensionFlags?: Map<string, { ty
result.hideThinking = true; result.hideThinking = true;
} else if (arg === "--advisor") { } else if (arg === "--advisor") {
result.advisor = true; result.advisor = true;
} else if (arg === "--external-thinking") {
result.externalThinking = true;
} else if (arg === "--prewalk") { } else if (arg === "--prewalk") {
result.prewalk = true; result.prewalk = true;
} else if (arg === "--no-prewalk") { } else if (arg === "--no-prewalk") {
@@ -311,6 +311,7 @@ export const VALUELESS_FLAGS: ReadonlySet<string> = new Set([
"--no-pty", "--no-pty",
"--hide-thinking", "--hide-thinking",
"--advisor", "--advisor",
"--external-thinking",
"--prewalk", "--prewalk",
"--no-prewalk", "--no-prewalk",
"--plan-yolo", "--plan-yolo",
@@ -363,13 +363,12 @@ export const agenticFixtures: Record<string, GalleryFixture> = {
think: { think: {
label: "Think", label: "Think",
// Streaming: scratchpad text still arriving. // Streaming: scratchpad thoughts still arriving.
streamingArgs: { streamingArgs: {
thoughts: "The retry loop re-reads the config after every failure — that explains the doubled latency.", thoughts: "The retry loop re-reads the config after every failure, which explains the doubled latency.",
}, },
args: { args: {
thoughts: thoughts: "The retry loop re-reads the config after every failure, which explains the doubled latency. Cache the parsed config outside the loop, then re-check the invalidation path.",
"The retry loop re-reads the config after every failure — that explains the doubled latency. Cache the parsed config outside the loop, then re-check the invalidation path before answering.",
}, },
result: { result: {
content: [{ type: "text", text: "------" }], content: [{ type: "text", text: "------" }],
@@ -77,6 +77,9 @@ export const launchHelp = {
advisor: Flags.boolean({ advisor: Flags.boolean({
description: "Enable the advisor runtime (passively reviews each turn and injects notes)", description: "Enable the advisor runtime (passively reviews each turn and injects notes)",
}), }),
"external-thinking": Flags.boolean({
description: "Use a private scratchpad while disabling supported GPT, Claude, and Gemini reasoning",
}),
hook: Flags.string({ description: "Load a hook/extension file (can be used multiple times)", multiple: true }), hook: Flags.string({ description: "Load a hook/extension file (can be used multiple times)", multiple: true }),
extension: Flags.string({ extension: Flags.string({
char: "e", char: "e",
@@ -1146,7 +1146,7 @@ export const SETTINGS_SCHEMA = {
tab: "model", tab: "model",
group: "Thinking", group: "Thinking",
label: "External Thinking", label: "External Thinking",
description: "Use a private think tool and send reasoning effort off to GPT Responses models", description: "Private scratchpad; not shown to user. Disables supported GPT, Claude, and Gemini reasoning",
}, },
}, },
+4
View File
@@ -1333,6 +1333,10 @@ export async function runRootCommand(
if (parsedArgs.advisor) { if (parsedArgs.advisor) {
settingsInstance.override("advisor.enabled", true); settingsInstance.override("advisor.enabled", true);
} }
// Apply --external-thinking CLI flag (ephemeral, not persisted)
if (parsedArgs.externalThinking) {
settingsInstance.override("externalThinking", true);
}
await logger.time( await logger.time(
"initTheme:final", "initTheme:final",
@@ -95,12 +95,8 @@ Write JSON args as `content` to `xd://<tool>` via `{{toolRefs.write}}`. Invalid
{{/if}} {{/if}}
{{#has tools "think"}} {{#has tools "think"}}
§ Reasoning § Scratchpad
`{{toolRefs.think}}`: private scratchpad; unwritten reasoning is unworked. `{{toolRefs.think}}`: private scratchpad; not shown to user.
- MUST call before turn's first action and before expensive-to-undo edit, destructive command, or final answer.
- Restate ask/constraints; ordered subproblems; explicitly solve/intermediate-result each; enumerate/resolve cases.
- Check claims against constraints, a boundary/degenerate case, and likely error. Failed check → redo step, NEVER patch conclusion.
- Re-call only for material new state: plan-changing result, failed check, unopened subproblem; NEVER narrate progress/restate recorded work.
{{/has}} {{/has}}
§ Tool Policy § Tool Policy
+4 -1
View File
@@ -201,6 +201,7 @@ import {
ReadTool, ReadTool,
releaseComputerSessionsForOwner, releaseComputerSessionsForOwner,
resolveMountedXdevExecutable, resolveMountedXdevExecutable,
supportsExternalThinking,
type Tool, type Tool,
type ToolSession, type ToolSession,
WebSearchTool, WebSearchTool,
@@ -3238,7 +3239,9 @@ async function createAgentSessionScoped(options: CreateAgentSessionOptions): Pro
} }
} }
const externalThinking = const externalThinking =
settings.get("externalThinking") && agent.state.tools.some(tool => tool.name === "think"); settings.get("externalThinking") &&
agent.state.tools.some(tool => tool.name === "think") &&
supportsExternalThinking(streamModel);
return settingsAwareStreamFn(streamModel, context, { return settingsAwareStreamFn(streamModel, context, {
...streamOptions, ...streamOptions,
forceReasoningOff: externalThinking || streamOptions?.forceReasoningOff, forceReasoningOff: externalThinking || streamOptions?.forceReasoningOff,
@@ -199,6 +199,7 @@ import {
PROPOSE_DEVICE_NAME, PROPOSE_DEVICE_NAME,
writeDeviceDispatch, writeDeviceDispatch,
} from "../tools/resolve"; } from "../tools/resolve";
import { supportsExternalThinking } from "../tools/think";
import type { TodoPhase } from "../tools/todo"; import type { TodoPhase } from "../tools/todo";
import { ToolError } from "../tools/tool-errors"; import { ToolError } from "../tools/tool-errors";
import { parseCommandArgs } from "../utils/command-args"; import { parseCommandArgs } from "../utils/command-args";
@@ -5188,10 +5189,7 @@ export class AgentSession {
!hasPendingUserDirective && !hasPendingUserDirective &&
this.settings.get("externalThinking") && this.settings.get("externalThinking") &&
this.getEnabledToolNames().includes("think") && this.getEnabledToolNames().includes("think") &&
activeModel && supportsExternalThinking(activeModel)
(activeModel.api === "openai-responses" ||
activeModel.api === "azure-openai-responses" ||
activeModel.api === "openai-codex-responses")
? buildNamedToolChoice("think", activeModel) ? buildNamedToolChoice("think", activeModel)
: undefined; : undefined;
const eagerTodoPrelude = const eagerTodoPrelude =
@@ -7068,6 +7066,11 @@ export class AgentSession {
} catch (error) { } catch (error) {
logger.warn("inspect_image reconcile after model change failed", { error: String(error) }); logger.warn("inspect_image reconcile after model change failed", { error: String(error) });
} }
try {
await this.#tools.reconcileThinkTool();
} catch (error) {
logger.warn("think tool reconcile after model change failed", { error: String(error) });
}
} }
#closeCodexProviderSessionsForHistoryRewrite(): void { #closeCodexProviderSessionsForHistoryRewrite(): void {
@@ -21,6 +21,7 @@ import { usesCodexTaskPrompt } from "../task/prompt-policy";
import { isMCPToolName, normalizeToolNames } from "../tools/builtin-names"; import { isMCPToolName, normalizeToolNames } from "../tools/builtin-names";
import { computerExposureMode } from "../tools/computer/exposure"; import { computerExposureMode } from "../tools/computer/exposure";
import { wrapToolWithMetaNotice } from "../tools/output-meta"; import { wrapToolWithMetaNotice } from "../tools/output-meta";
import { supportsExternalThinking } from "../tools/think";
import { ToolAbortError, ToolError } from "../tools/tool-errors"; import { ToolAbortError, ToolError } from "../tools/tool-errors";
import { isMountableUnderXdev, listXdevTools, type XdevState, xdevDocsFor, xdevEntries } from "../tools/xdev"; import { isMountableUnderXdev, listXdevTools, type XdevState, xdevDocsFor, xdevEntries } from "../tools/xdev";
import { type EditMode, resolveEditMode } from "../utils/edit-mode"; import { type EditMode, resolveEditMode } from "../utils/edit-mode";
@@ -1171,6 +1172,17 @@ export class SessionTools {
* @returns false when enabling was requested but this session cannot build the tool. * @returns false when enabling was requested but this session cannot build the tool.
*/ */
setThinkToolEnabled(enabled: boolean): Promise<boolean> { setThinkToolEnabled(enabled: boolean): Promise<boolean> {
return this.#setThinkToolActive(enabled && supportsExternalThinking(this.#host.model()));
}
/** Reconciles the external scratchpad after the active model changes. */
reconcileThinkTool(): Promise<boolean> {
return this.#setThinkToolActive(
this.#host.settings.get("externalThinking") && supportsExternalThinking(this.#host.model()),
);
}
#setThinkToolActive(enabled: boolean): Promise<boolean> {
return this.runToolRegistryMutation(async () => { return this.runToolRegistryMutation(async () => {
const active = this.getEnabledToolNames(); const active = this.getEnabledToolNames();
if (!enabled) { if (!enabled) {
+6 -4
View File
@@ -62,7 +62,7 @@ import { wrapToolWithMetaNotice } from "./output-meta";
import { ReadTool } from "./read"; import { ReadTool } from "./read";
import type { PlanProposalHandler } from "./resolve"; import type { PlanProposalHandler } from "./resolve";
import { SecurityScanTool } from "./security-scan"; import { SecurityScanTool } from "./security-scan";
import { ThinkTool } from "./think"; import { supportsExternalThinking, ThinkTool } from "./think";
import { type TodoPhase, TodoTool } from "./todo"; import { type TodoPhase, TodoTool } from "./todo";
import { WriteTool } from "./write"; import { WriteTool } from "./write";
import { isMountableUnderXdev, type XdevState } from "./xdev"; import { isMountableUnderXdev, type XdevState } from "./xdev";
@@ -467,6 +467,8 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
: undefined; : undefined;
const goalEnabled = session.settings.get("goal.enabled"); const goalEnabled = session.settings.get("goal.enabled");
const goalModeActive = !restrictToolNames && goalEnabled && session.getGoalModeState?.()?.enabled === true; const goalModeActive = !restrictToolNames && goalEnabled && session.getGoalModeState?.()?.enabled === true;
const externalThinkingActive =
session.settings.get("externalThinking") && supportsExternalThinking(session.getActiveModel?.());
if (goalModeActive && requestedTools && !requestedTools.includes("goal")) { if (goalModeActive && requestedTools && !requestedTools.includes("goal")) {
requestedTools.push("goal"); requestedTools.push("goal");
} }
@@ -565,7 +567,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
if (session.settings.get("memory.backend") === "mnemopi" && !requestedTools.includes("memory_edit")) { if (session.settings.get("memory.backend") === "mnemopi" && !requestedTools.includes("memory_edit")) {
requestedTools.push("memory_edit"); requestedTools.push("memory_edit");
} }
if (session.settings.get("externalThinking") && !requestedTools.includes("think")) { if (externalThinkingActive && !requestedTools.includes("think")) {
requestedTools.push("think"); requestedTools.push("think");
} }
// Auto-learn tools are gated by `autolearn.enabled` but, like the memory // Auto-learn tools are gated by `autolearn.enabled` but, like the memory
@@ -610,7 +612,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
if (name === "inspect_image") return isInspectImageToolActive(session); if (name === "inspect_image") return isInspectImageToolActive(session);
if (name === "web_search") return session.settings.get("web_search.enabled"); if (name === "web_search") return session.settings.get("web_search.enabled");
if (name === "security_scan") return session.settings.get("security.enabled"); if (name === "security_scan") return session.settings.get("security.enabled");
if (name === "think") return session.settings.get("externalThinking"); if (name === "think") return externalThinkingActive;
if (name === "ask") return session.settings.get("ask.enabled"); if (name === "ask") return session.settings.get("ask.enabled");
if (name === "browser") return session.settings.get("browser.enabled"); if (name === "browser") return session.settings.get("browser.enabled");
if (name === "computer") return session.settings.get("computer.enabled"); if (name === "computer") return session.settings.get("computer.enabled");
@@ -657,7 +659,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
...Object.entries(BUILTIN_TOOLS) ...Object.entries(BUILTIN_TOOLS)
.filter(([name]) => isToolAllowed(name)) .filter(([name]) => isToolAllowed(name))
.map(([name, factory]) => [name, factory] as const), .map(([name, factory]) => [name, factory] as const),
...(session.settings.get("externalThinking") ? ([["think", HIDDEN_TOOLS.think]] as const) : []), ...(externalThinkingActive ? ([["think", HIDDEN_TOOLS.think]] as const) : []),
...(includeYield ? ([["yield", HIDDEN_TOOLS.yield]] as const) : []), ...(includeYield ? ([["yield", HIDDEN_TOOLS.yield]] as const) : []),
...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []), ...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []),
]; ];
+19 -6
View File
@@ -1,13 +1,27 @@
import { type } from "@oh-my-pi/omptype"; import { type } from "@oh-my-pi/omptype";
import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core";
import type { Model } from "@oh-my-pi/pi-ai";
import { type Component, Markdown } from "@oh-my-pi/pi-tui"; import { type Component, Markdown } from "@oh-my-pi/pi-tui";
import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import type { RenderResultOptions } from "../extensibility/custom-tools/types";
import { getMarkdownTheme, type Theme } from "../modes/theme/theme"; import { getMarkdownTheme, type Theme } from "../modes/theme/theme";
/** Whether a model transport can suppress native reasoning while private scratchpad thoughts are active. */
export function supportsExternalThinking(model: Model | null | undefined): boolean {
return (
model?.api === "openai-responses" ||
model?.api === "azure-openai-responses" ||
model?.api === "openai-codex-responses" ||
model?.api === "anthropic-messages" ||
model?.api === "google-generative-ai" ||
model?.api === "google-gemini-cli" ||
model?.api === "google-vertex"
);
}
const thinkSchema = type({ const thinkSchema = type({
thoughts: type("string").describe("private scratchpad reasoning to retain before the next response"), thoughts: type("string").describe("private scratchpad; not shown to user"),
"+": "reject", "+": "reject",
}).describe("record private intermediate reasoning before answering"); }).describe("private scratchpad; not shown to user");
type ThinkParams = typeof thinkSchema.infer; type ThinkParams = typeof thinkSchema.infer;
@@ -36,14 +50,13 @@ interface ThinkToolDetails {
recorded: true; recorded: true;
} }
/** Records private intermediate reasoning while native GPT reasoning is disabled. */ /** Records private scratchpad thoughts while native model reasoning is disabled. */
export class ThinkTool implements AgentTool<typeof thinkSchema, ThinkToolDetails> { export class ThinkTool implements AgentTool<typeof thinkSchema, ThinkToolDetails> {
readonly name = "think"; readonly name = "think";
readonly approval = "read" as const; readonly approval = "read" as const;
readonly label = "Think"; readonly label = "Think";
readonly summary = "Record private intermediate reasoning before answering"; readonly summary = "Record private scratchpad thoughts";
readonly description = readonly description = "private scratchpad; not shown to user";
"Use this private scratchpad to plan, derive, or check work before answering. Record only materially new reasoning. The user does not see this tool activity.";
readonly parameters = thinkSchema; readonly parameters = thinkSchema;
readonly strict = true; readonly strict = true;
readonly intent = "omit" as const; readonly intent = "omit" as const;
@@ -56,6 +56,18 @@ describe("OPTIONAL_VALUE_FLAGS table is honored by args.ts parseArgs", () => {
} }
}); });
describe("--external-thinking", () => {
it("enables external thinking without consuming the initial message", () => {
const result = parseArgs(["--external-thinking", "check this"]);
expect(result.externalThinking).toBe(true);
expect(result.messages).toEqual(["check this"]);
});
it("stays unset when omitted", () => {
expect(parseArgs([]).externalThinking).toBeUndefined();
});
});
describe("--session-dir", () => { describe("--session-dir", () => {
it("uses PI_CODING_AGENT_SESSION_DIR unless the CLI flag overrides it", () => { it("uses PI_CODING_AGENT_SESSION_DIR unless the CLI flag overrides it", () => {
const previous = Bun.env.PI_CODING_AGENT_SESSION_DIR; const previous = Bun.env.PI_CODING_AGENT_SESSION_DIR;
@@ -109,6 +109,15 @@ describe("createAgentSession defaultInactive tool activation", () => {
workspaceTree: { rootPath: tempDir, rendered: "", truncated: false, totalLines: 0, agentsMdFiles: [] }, workspaceTree: { rootPath: tempDir, rendered: "", truncated: false, totalLines: 0, agentsMdFiles: [] },
}); });
const requireBundledModel = (
provider: "anthropic" | "google-antigravity" | "openai" | "xai",
id: string,
): Model => {
const bundled = getBundledModel(provider, id);
if (!bundled) throw new Error(`Expected ${provider}/${id} model to exist`);
return bundled;
};
afterEach(() => { afterEach(() => {
for (const tempDir of tempDirs.splice(0)) { for (const tempDir of tempDirs.splice(0)) {
removeSyncWithRetries(tempDir); removeSyncWithRetries(tempDir);
@@ -152,6 +161,7 @@ describe("createAgentSession defaultInactive tool activation", () => {
const settings = Settings.isolated(); const settings = Settings.isolated();
const { session } = await createAgentSession({ const { session } = await createAgentSession({
...baseOptions(tempDir), ...baseOptions(tempDir),
model: requireBundledModel("openai", "gpt-5"),
settings, settings,
}); });
@@ -174,18 +184,40 @@ describe("createAgentSession defaultInactive tool activation", () => {
} }
}); });
it("activates the private think tool at startup when external thinking is configured", async () => { it("exposes the private think tool only on transports that can disable native reasoning", async () => {
const tempDir = makeTempDir(); const tempDir = makeTempDir();
const settings = Settings.isolated({ externalThinking: true }); const settings = Settings.isolated({ externalThinking: true });
const unsupported = requireBundledModel("xai", "grok-4");
const fable = requireBundledModel("anthropic", "claude-fable-5");
const responses = requireBundledModel("openai", "gpt-5");
const gemini = requireBundledModel("google-antigravity", "gemini-3.6-flash");
const { session } = await createAgentSession({ const { session } = await createAgentSession({
...baseOptions(tempDir), ...baseOptions(tempDir),
settings, settings,
model: unsupported,
}); });
const authStorage = session.modelRegistry.authStorage;
authStorage.setRuntimeApiKey("anthropic", "test-key");
authStorage.setRuntimeApiKey("openai", "test-key");
authStorage.setRuntimeApiKey("google-antigravity", "test-key");
authStorage.setRuntimeApiKey("xai", "test-key");
try { try {
expect(session.getActiveToolNames()).not.toContain("think");
await session.setModel(fable);
expect(session.getToolByName("think")).toBeDefined(); expect(session.getToolByName("think")).toBeDefined();
expect(session.getActiveToolNames()).toContain("think"); expect(session.getActiveToolNames()).toContain("think");
expect(session.getXdevToolEntries().map(entry => entry.name)).not.toContain("think"); expect(session.systemPrompt.join("\n")).toContain("private scratchpad; not shown to user");
await session.setModel(responses);
expect(session.getActiveToolNames()).toContain("think");
await session.setModel(gemini);
expect(session.getActiveToolNames()).toContain("think");
await session.setModel(unsupported);
expect(session.getActiveToolNames()).not.toContain("think");
expect(session.systemPrompt.join("\n")).not.toContain("private scratchpad; not shown to user");
} finally { } finally {
await session.dispose(); await session.dispose();
} }
@@ -217,7 +249,7 @@ describe("createAgentSession defaultInactive tool activation", () => {
fetch: async request => { fetch: async request => {
requestTexts.push(await request.text()); requestTexts.push(await request.text());
if (requestTexts.length === 1) { if (requestTexts.length === 1) {
const argumentsJson = JSON.stringify({ thoughts: "Checked the request before answering." }); const argumentsJson = JSON.stringify({ notes: "Checked the request before answering." });
return sse([ return sse([
{ {
type: "response.output_item.added", type: "response.output_item.added",
@@ -267,8 +299,7 @@ describe("createAgentSession defaultInactive tool activation", () => {
]); ]);
}, },
}); });
const model = getBundledModel("openai", "gpt-5"); const model = requireBundledModel("openai", "gpt-5");
if (!model) throw new Error("Expected gpt-5 model to exist");
// The prompt preflight validates the key through the registry (not the // The prompt preflight validates the key through the registry (not the
// per-request `getApiKey` override), so seed it for keyless CI runners. // per-request `getApiKey` override), so seed it for keyless CI runners.
modelRegistry.authStorage.setRuntimeApiKey("openai", "test-key"); modelRegistry.authStorage.setRuntimeApiKey("openai", "test-key");
@@ -13,7 +13,7 @@ describe("thinkToolRenderer", () => {
const uiTheme = theme!; const uiTheme = theme!;
const callComponent = thinkToolRenderer.renderCall( const callComponent = thinkToolRenderer.renderCall(
{ thoughts: "Analyzing the solution step by step." }, { thoughts: "Cache the parsed config, then check invalidation." },
{ expanded: true, isPartial: false }, { expanded: true, isPartial: false },
uiTheme, uiTheme,
); );
@@ -22,8 +22,8 @@ describe("thinkToolRenderer", () => {
const lines = callComponent.render(100); const lines = callComponent.render(100);
const fullText = lines.join("\n"); const fullText = lines.join("\n");
expect(fullText).toContain("Analyzing the solution step by step."); expect(fullText).toContain("Cache the parsed config, then check invalidation.");
expect(fullText).toContain(uiTheme.fg("thinkingText", "Analyzing the solution step by step.")); expect(fullText).toContain(uiTheme.fg("thinkingText", "Cache the parsed config, then check invalidation."));
}); });
it("has inline set to true", () => { it("has inline set to true", () => {
+1 -1
View File
@@ -4,7 +4,7 @@
### Fixed ### Fixed
- Fixed `claude-opus-5` (and later Opus lines) falling back to the 1568px frame instead of the 1932px high-res tier, so snapcompact archives for the current flagship Opus now retain ~33% more history at the same per-frame bill ([#8256](https://github.com/can1357/oh-my-pi/issues/8256)). - Fixed case-sensitivity in Anthropic model ID parsing for high-res frame selection
## [17.1.5] - 2026-07-27 ## [17.1.5] - 2026-07-27
+3 -1
View File
@@ -364,7 +364,9 @@ const MODEL_VARIANTS: readonly (readonly [RegExp, IdealShape])[] = [
/** Eval-ideal format for a model id, or undefined when unmeasured. */ /** Eval-ideal format for a model id, or undefined when unmeasured. */
export function idealShapeVariant(modelId: string): IdealShape | undefined { export function idealShapeVariant(modelId: string): IdealShape | undefined {
const anthropic = parseAnthropicModel(modelId); // The catalog parser is case-sensitive; the regex rules below are not.
// Normalize so mixed-case gateway ids keep matching the Anthropic tier.
const anthropic = parseAnthropicModel(modelId.toLowerCase());
if ( if (
anthropic && anthropic &&
(isFableOrMythos(anthropic.kind) || (anthropic.kind === "opus" && semverGte(anthropic.version, "4.7"))) (isFableOrMythos(anthropic.kind) || (anthropic.kind === "opus" && semverGte(anthropic.version, "4.7")))