feat: introduced external thinking support and private scratchpad think tool

- Added support for external thinking and forced reasoning disablement across AI provider options and request transformers.
- Implemented the private scratchpad think tool along with its renderer, system prompt rules, and schema configuration.
- Updated agent session management and SDK tools to support dynamic runtime activation of the think tool via the externalThinking setting.
- Added comprehensive unit tests covering reasoning fallbacks, tool activation, and rendering behavior.
This commit is contained in:
can1357
2026-08-11 20:39:57 +02:00
parent d3b22a0db6
commit 10fd42289c
26 changed files with 560 additions and 29 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added `forceReasoningOff` and `disableReasoning` options to disable reasoning in OpenAI and Azure OpenAI models
## [17.2.13] - 2026-08-11
### Changed
@@ -65,6 +65,7 @@ export interface AzureOpenAIResponsesOptions extends StreamOptions {
azureDeploymentName?: string;
toolChoice?: ToolChoice;
serviceTier?: ServiceTier;
disableReasoning?: boolean;
}
type AzureOpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
@@ -1529,6 +1529,7 @@ export async function buildTransformedCodexRequestBody(
}
const codexOptions: CodexRequestOptions = {
reasoningEffort: options?.reasoning,
reasoningOff: options?.forceReasoningOff,
reasoningSummary: options?.reasoningSummary,
reasoningContext: options?.reasoningContext,
textVerbosity: options?.textVerbosity,
@@ -32,6 +32,8 @@ export interface ReasoningConfig {
export interface CodexRequestOptions {
/** User-facing effort; maps 1:1 onto the wire tier of the same name. */
reasoningEffort?: CodexCallerEffort | "none";
/** Suppress native reasoning by sending `reasoning.effort: "none"`. */
reasoningOff?: boolean;
reasoningSummary?: ReasoningConfig["summary"] | null;
/** Explicit `reasoning.context` override. Omitted by default; Responses Lite forces `all_turns` as required by that transport. */
reasoningContext?: CodexReasoningContext;
@@ -454,9 +456,12 @@ export async function transformRequestBody(
applyCodexResponsesLiteShape(body);
}
if (options.reasoningEffort !== undefined || responsesLite) {
const reasoningConfig =
options.reasoningEffort !== undefined ? getReasoningConfig(model, options.reasoningEffort, options) : {};
if (options.reasoningOff || options.reasoningEffort !== undefined || responsesLite) {
const reasoningConfig: Partial<ReasoningConfig> = options.reasoningOff
? { effort: "none" }
: options.reasoningEffort !== undefined
? getReasoningConfig(model, options.reasoningEffort, options)
: {};
body.reasoning = {
...body.reasoning,
...reasoningConfig,
@@ -478,7 +483,7 @@ export async function transformRequestBody(
// Catalog pro aliases (`gpt-5.6-*-pro`): applied after the effort branch so
// the mode is sent even when no effort is set (the branch above deletes
// `body.reasoning` in that case) — mode and effort are independent fields.
if (model.reasoningMode) {
if (model.reasoningMode && !options.reasoningOff) {
body.reasoning = { ...body.reasoning, mode: model.reasoningMode };
}
@@ -132,7 +132,13 @@ function collectMessageParts(error: unknown, captured: CapturedHttpErrorResponse
return parts.join("\n");
}
const REASONING_EFFORT_FIELD_PATTERN = /reasoning[_. ]effort|reasoning value/i;
/**
* Text that identifies a 400 as being about the reasoning-effort field.
* OpenAI-compatible gateways (cliproxy, …) never name the field — they reject
* the value alone with `level "none" not supported, valid levels: low, …` — so
* the allowed-level phrasing counts as a mention too.
*/
const REASONING_EFFORT_FIELD_PATTERN = /reasoning[_. ]effort|reasoning value|(?:valid|supported|allowed) levels?/i;
function mentionsReasoningEffort(error: unknown, captured: CapturedHttpErrorResponse | undefined): boolean {
const param = capturedStringField(captured, "param");
@@ -168,10 +174,13 @@ function isInvalidReasoningEffortError(
if (/(?:unsupported|not supported)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message)) {
return true;
}
return new RegExp(
`(?:invalid|unsupported|not supported)[^\\n]*["'\`]${escapeRegExp(currentEffort)}["'\`]`,
"i",
).test(message);
// Gateways put the rejected value first (`level "none" not supported`), the
// official API puts the verdict first (`Unsupported value: 'none'`).
const quoted = `["'\`]${escapeRegExp(currentEffort)}["'\`]`;
return (
new RegExp(`(?:invalid|unsupported|not supported)[^\\n]*${quoted}`, "i").test(message) ||
new RegExp(`${quoted}[^\\n]*(?:invalid|unsupported|not supported)`, "i").test(message)
);
}
function escapeRegExp(value: string): string {
@@ -186,9 +195,12 @@ function parseKnownReasoningValues(text: string): Set<string> {
values.add(quotedMatch[1]!.toLowerCase());
quotedMatch = quotedPattern.exec(text);
}
const allowedMatch = /(?:must be|one of|allowed values?|supported values?(?: are)?|expected)([^.\n]+)/i.exec(text);
const allowedMatch =
/(?:must be|one of|allowed values?|supported values?(?: are)?|expected|(?:valid|supported|allowed) levels?(?: are)?)[^.\n]+/i.exec(
text,
);
if (allowedMatch) {
const allowedText = allowedMatch[1]!;
const allowedText = allowedMatch[0]!;
const barePattern = /\b(none|minimal|low|medium|high|xhigh|max)\b/gi;
let bareMatch = barePattern.exec(allowedText);
while (bareMatch !== null) {
@@ -201,7 +213,8 @@ function parseKnownReasoningValues(text: string): Set<string> {
function parseAllowedReasoningValues(message: string, currentEffort: string): Set<string> | undefined {
const values = parseKnownReasoningValues(message);
const hasAllowedCue = /must be|one of|allowed values?|supported values?|expected/i.test(message);
const hasAllowedCue =
/must be|one of|allowed values?|supported values?|expected|(?:valid|supported|allowed) levels?/i.test(message);
values.delete(currentEffort.toLowerCase());
if (!hasAllowedCue && values.size === 0) return undefined;
return values;
+73 -12
View File
@@ -1,5 +1,6 @@
import { scheduler } from "node:timers/promises";
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
import { bareModelId, parseOpenAIModel, semverGte } from "@oh-my-pi/pi-catalog/identity";
import { $flag, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
import * as AIError from "../error";
import { getEnvApiKey } from "../stream";
@@ -80,6 +81,7 @@ import {
createInitialResponsesAssistantMessage,
createOpenAIStrictToolsState,
disableStrictToolsForScope,
getJuiceValue,
getOpenAIPromptCacheKey,
getOpenAIResponsesRoutingSessionId,
getOpenAIStrictToolsScope,
@@ -298,14 +300,23 @@ interface OpenAIResponsesChainedParams {
*/
function buildOpenAIResponsesChainedParams(
params: OpenAIResponsesSamplingParams,
trailingScaffoldingItems: number,
chain: OpenAIResponsesChainState,
): OpenAIResponsesChainedParams {
const historyParams =
trailingScaffoldingItems > 0 && Array.isArray(params.input)
? { ...params, input: params.input.slice(0, params.input.length - trailingScaffoldingItems) }
: params;
const deltaInput = chain.canAppend
? buildResponsesDeltaInput(chain.lastParams, chain.lastResponseItems, params)
? buildResponsesDeltaInput(chain.lastParams, chain.lastResponseItems, historyParams)
: null;
if (deltaInput && deltaInput.length > 0 && chain.lastResponseId) {
const scaffolding =
historyParams !== params && Array.isArray(params.input)
? params.input.slice(params.input.length - trailingScaffoldingItems)
: [];
return {
params: { ...params, previous_response_id: chain.lastResponseId, input: deltaInput },
params: { ...params, previous_response_id: chain.lastResponseId, input: [...deltaInput, ...scaffolding] },
previousResponseId: chain.lastResponseId,
};
}
@@ -462,8 +473,9 @@ const streamOpenAIResponsesOnce = (
false,
chainState?.canAppend ? chainState.lastParams?.input : undefined,
);
const params = builtParams.params;
const { params, trailingScaffoldingItems } = builtParams;
let activeParams = params;
let activeTrailingScaffoldingItems = trailingScaffoldingItems;
const resolvedBaseUrl = (baseUrl ?? "https://api.openai.com/v1").replace(/\/+$/, "");
const requestReasoningEffortFallbacks = new Map<string, OpenAIReasoningEffortFallback>();
const attemptedReasoningEffortFallbacks = new Set<string>();
@@ -490,7 +502,9 @@ const streamOpenAIResponsesOnce = (
}
applyReasoningEffortFallbackForRequest(params);
let chained: OpenAIResponsesChainedParams =
chainState && !chainState.disabled ? buildOpenAIResponsesChainedParams(params, chainState) : { params };
chainState && !chainState.disabled
? buildOpenAIResponsesChainedParams(params, trailingScaffoldingItems, chainState)
: { params };
sentPreviousResponseId = chained.previousResponseId;
const idleTimeoutMs =
options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(model.compat.streamIdleTimeoutMs);
@@ -586,7 +600,9 @@ const streamOpenAIResponsesOnce = (
const reasoningEffortFallback =
activeReasoningEffortFallbackKey && activeRequestParams && !requestSignal.aborted
? resolveOpenAIReasoningEffortFallback(error, capturedErrorResponse, activeRequestParams, {
explicitDisable: options?.disableReasoning === true && options.reasoning === undefined,
explicitDisable:
options?.forceReasoningOff === true ||
(options?.disableReasoning === true && options.reasoning === undefined),
})
: undefined;
if (reasoningEffortFallback !== undefined && activeReasoningEffortFallbackKey) {
@@ -632,7 +648,11 @@ const streamOpenAIResponsesOnce = (
if (chainState && !chainState.disabled) fallbackParams.store = true;
let fallbackChained: OpenAIResponsesChainedParams =
chainState && !chainState.disabled
? buildOpenAIResponsesChainedParams(fallbackParams, chainState)
? buildOpenAIResponsesChainedParams(
fallbackParams,
fallbackBuilt.trailingScaffoldingItems,
chainState,
)
: { params: fallbackParams };
sentPreviousResponseId = fallbackChained.previousResponseId;
fallbackChained = {
@@ -642,7 +662,7 @@ const streamOpenAIResponsesOnce = (
chained = fallbackChained;
activeRawRequestDump.body = chained.params;
activeParams = fallbackParams;
activeStrictToolsApplied = fallbackBuilt.strictToolsApplied;
activeTrailingScaffoldingItems = fallbackBuilt.trailingScaffoldingItems;
continue;
}
if (!chainState || !sentPreviousResponseId || requestSignal.aborted) {
@@ -688,6 +708,7 @@ const streamOpenAIResponsesOnce = (
chained = { params: retryParams };
activeRawRequestDump.body = retryParams;
activeParams = currentParams;
activeTrailingScaffoldingItems = currentBuilt.trailingScaffoldingItems;
activeStrictToolsApplied = currentBuilt.strictToolsApplied;
}
}
@@ -824,7 +845,17 @@ const streamOpenAIResponsesOnce = (
if (replayableResponseItems) {
if (providerSessionState) providerSessionState.nativeHistoryReplayWarmed = true;
if (chainState) {
chainState.lastParams = structuredCloneJSON(activeParams);
chainState.lastParams = structuredCloneJSON(
activeTrailingScaffoldingItems > 0 && Array.isArray(activeParams.input)
? {
...activeParams,
input: activeParams.input.slice(
0,
activeParams.input.length - activeTrailingScaffoldingItems,
),
}
: activeParams,
);
chainState.lastPromptCacheBreakpointPolicy = promptCacheBreakpointPolicy;
if (output.responseId) {
chainState.lastResponseId = output.responseId;
@@ -843,7 +874,14 @@ const streamOpenAIResponsesOnce = (
// baseline, but `lastParams` still records the successful wire controls
// without re-enabling `previous_response_id` chaining.
chainState.canAppend = false;
chainState.lastParams = structuredCloneJSON(activeParams);
chainState.lastParams = structuredCloneJSON(
activeTrailingScaffoldingItems > 0 && Array.isArray(activeParams.input)
? {
...activeParams,
input: activeParams.input.slice(0, activeParams.input.length - activeTrailingScaffoldingItems),
}
: activeParams,
);
chainState.lastPromptCacheBreakpointPolicy = promptCacheBreakpointPolicy;
chainState.lastResponseId = undefined;
chainState.lastResponseItems = undefined;
@@ -899,6 +937,17 @@ function isOfficialOpenAIResponsesEndpoint(model: Model<"openai-responses">): bo
}
}
/**
* GPT-5.6+ family check for Responses routes. The model id classifies the
* reasoning family regardless of the provider/host serving it — a cliproxy or
* other OpenAI-compatible gateway carrying `gpt-5.6-sol` gets the same
* scaffolding as the official endpoint.
*/
function isGpt56PlusResponsesModel(model: Model<"openai-responses">): boolean {
const parsed = parseOpenAIModel(bareModelId(model.requestModelId ?? model.id));
return parsed !== null && semverGte(parsed.version, "5.6");
}
function isResponsesPromptCacheableContentBlock(block: unknown): block is ResponseInputContent {
if (typeof block !== "object" || block === null || !("type" in block)) return false;
return block.type === "input_text" || block.type === "input_image" || block.type === "input_file";
@@ -1090,7 +1139,7 @@ export function buildParams(
strictToolsScope?: OpenAIStrictToolsScope,
disableStrictToolsOverride = false,
statefulCacheBaseline?: ResponseInput,
): { params: OpenAIResponsesSamplingParams; strictToolsApplied: boolean } {
): { params: OpenAIResponsesSamplingParams; trailingScaffoldingItems: number; strictToolsApplied: boolean } {
const policy = resolveOpenAICompatPolicy(model, {
endpoint: "responses",
reasoning: options?.reasoning,
@@ -1244,6 +1293,7 @@ export function buildParams(
: options?.reasoningSummary;
applyResponsesCompatPolicy(params, reasoningPolicy, {
reasoningSummary,
forceReasoningOff: options?.forceReasoningOff,
mapEffort: effort =>
model.compat.reasoningEffortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
model.thinking?.effortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
@@ -1253,7 +1303,7 @@ export function buildParams(
// mode survives every policy branch (disabled/omitted effort included) while
// keeping whatever effort/summary the policy produced — mode and effort are
// independent wire fields.
if (model.reasoningMode) {
if (model.reasoningMode && !options?.forceReasoningOff) {
params.reasoning = { ...params.reasoning, mode: model.reasoningMode };
}
@@ -1266,7 +1316,18 @@ export function buildParams(
applyOpenAIExtraBody(params, options?.extraBody);
applyOpenAIResponsesPromptCachePolicy(params, model, options, statefulCacheBaseline);
return { params, strictToolsApplied };
let trailingScaffoldingItems = 0;
if (options?.forceReasoningOff && isGpt56PlusResponsesModel(model)) {
const effort = options.reasoning ?? "medium";
const juice = getJuiceValue(effort);
messages.push({
role: "developer",
content: [{ type: "input_text", text: `# Juice: ${juice} !important` }],
});
trailingScaffoldingItems = 1;
}
return { params, trailingScaffoldingItems, strictToolsApplied };
}
/**
@@ -2428,6 +2428,20 @@ export function finalizeMessageText(item: ResponseOutputMessage, streamedText: s
if (!item.content?.length) return streamedText || "";
return item.content.map(part => (part.type === "output_text" ? (part.text ?? "") : (part.refusal ?? ""))).join("");
}
export const JUICE_EFFORT_MAP: Record<string, number> = {
none: 0,
minimal: 2,
low: 4,
medium: 8,
high: 48,
xhigh: 112,
max: 960,
};
export function getJuiceValue(effort?: string): number {
if (!effort) return 8;
return JUICE_EFFORT_MAP[effort] ?? 8;
}
export function accumulateToolCallArgumentsDelta(
block: ResponsesToolCallBlock,
@@ -3308,6 +3322,14 @@ type ReasoningOptions = {
export interface ApplyResponsesCompatPolicyOptions {
reasoningSummary?: "auto" | "detailed" | "concise" | null;
mapEffort?: (effort: string) => string;
/**
* Suppress native reasoning by sending `reasoning.effort: "none"` — the only
* disable level the Responses API defines (`"off"` is not a wire value and
* 400s everywhere). Gateways that reject `none` for a given model are
* handled by the reasoning-effort fallback retry, which clamps to the
* lowest level the error reports as allowed.
*/
forceReasoningOff?: boolean;
}
export function applyResponsesCompatPolicy<P extends ResponseCreateParamsStreaming>(
@@ -3316,6 +3338,10 @@ export function applyResponsesCompatPolicy<P extends ResponseCreateParamsStreami
options: ApplyResponsesCompatPolicyOptions | undefined,
): void {
const reasoning = policy.reasoning;
if (options?.forceReasoningOff) {
params.reasoning = { effort: "none" } as P["reasoning"];
return;
}
if (!reasoning.modelSupported) return;
if (reasoning.includeEncryptedReasoning) {
const include = params.include ?? [];
+4
View File
@@ -1692,6 +1692,7 @@ function mapOptionsForApi<TApi extends Api>(
openrouterVariant: options?.openrouterVariant,
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
disableReasoning: options?.disableReasoning,
forceReasoningOff: options?.forceReasoningOff,
textVerbosity: options?.textVerbosity,
promptCache: options?.promptCache,
statefulResponses: options?.statefulResponses,
@@ -1706,6 +1707,8 @@ function mapOptionsForApi<TApi extends Api>(
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
promptCache: options?.promptCache,
statefulResponses: options?.statefulResponses,
disableReasoning: options?.disableReasoning || options?.forceReasoningOff,
forceReasoningOff: options?.forceReasoningOff,
});
case "openai-codex-responses":
@@ -1718,6 +1721,7 @@ function mapOptionsForApi<TApi extends Api>(
codexCompaction: options?.codexCompaction,
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
textVerbosity: options?.textVerbosity,
forceReasoningOff: options?.forceReasoningOff,
});
case "google-generative-ai": {
+5
View File
@@ -478,6 +478,11 @@ export interface StreamOptions {
* `false` so `previous_response_id` cannot explain a result.
*/
statefulResponses?: boolean;
/**
* Emit `reasoning: { effort: "none" }` for OpenAI Responses and Codex requests.
* Used when a caller supplies an external reasoning scratchpad; other transports ignore it.
*/
forceReasoningOff?: boolean;
/**
* Provider-scoped mutable state store for this agent session.
* Providers can use this to persist transport/session state between turns.
@@ -166,6 +166,14 @@ describe("openai-codex optional response controls", () => {
expect("stream_options" in suppressed).toBe(false);
});
it("disables native reasoning with effort none when an external scratchpad replaces it", async () => {
const model = createCodexModel("gpt-5.5");
const body = await buildTransformedCodexRequestBody(model, createCodexTestContext(), {
forceReasoningOff: true,
});
expect(body.reasoning).toEqual({ effort: "none" });
});
it("forces reasoning.context to all_turns for Responses Lite", async () => {
const model = createCodexModel("gpt-5.5");
@@ -108,6 +108,18 @@ function pipeDelimitedReasoningEffortResponse(): Response {
);
}
/**
* cliproxy-style gateway rejection: the field is never named and the rejected
* value comes before the verdict (`level "none" not supported, valid levels: …`).
*/
function unsupportedLevelResponse(value: string): Response {
const message = `level "${value}" not supported, valid levels: low, medium, high, xhigh, max`;
return new Response(JSON.stringify({ error: { message, type: "invalid_request_error" } }), {
status: 400,
headers: { "content-type": "application/json" },
});
}
function summaryReasoningErrorResponse(): Response {
return new Response(
JSON.stringify({
@@ -343,6 +355,28 @@ describe("OpenAI reasoning effort fallback retry", () => {
expect(bodies.map(body => (body.reasoning as { effort?: string } | undefined)?.effort)).toEqual(["xhigh", "max"]);
});
it("clamps a rejected reasoning-off request to the lowest level the gateway allows", async () => {
const bodies: Record<string, unknown>[] = [];
const fetchMock: FetchImpl = Object.assign(
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const body = parseJsonBody(init);
bodies.push(body);
return bodies.length === 1 ? unsupportedLevelResponse("none") : createResponsesSseResponse();
},
{ preconnect: fetch.preconnect },
);
const result = await streamOpenAIResponses(createMaxLadderResponsesModel(), testContext, {
apiKey: "test-key",
fetch: fetchMock,
reasoning: "high",
forceReasoningOff: true,
}).result();
expect(result.stopReason).toBe("stop");
expect(bodies.map(body => (body.reasoning as { effort?: string } | undefined)?.effort)).toEqual(["none", "low"]);
});
it("does not retry unrelated reasoning parameter errors", async () => {
let attempts = 0;
const fetchMock: FetchImpl = Object.assign(
@@ -33,9 +33,12 @@ const ctx: Context = {
messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
};
async function drain(model: Model<"openai-responses">): Promise<Record<string, unknown>> {
async function drain(
model: Model<"openai-responses">,
options: { forceReasoningOff?: boolean } = {},
): Promise<Record<string, unknown>> {
const { fetchMock, captured } = mockSseFetch();
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock, temperature: 0 });
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock, temperature: 0, ...options });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
@@ -67,4 +70,10 @@ describe("openai-responses sampling-param gating (#5606)", () => {
const body = await drain(model);
expect(body.temperature).toBe(0);
});
it("disables native reasoning with effort none when an external scratchpad replaces it", async () => {
const model = getBundledModel("openai", "gpt-5") as Model<"openai-responses">;
const body = await drain(model, { forceReasoningOff: true });
expect(body.reasoning).toEqual({ effort: "none" });
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added `externalThinking` setting for private scratchpad reasoning via the new `think` tool
## [17.2.13] - 2026-08-11
### Added
@@ -1139,6 +1139,17 @@ export const SETTINGS_SCHEMA = {
},
},
externalThinking: {
type: "boolean",
default: false,
ui: {
tab: "model",
group: "Thinking",
label: "External Thinking",
description: "Use a private think tool and send reasoning effort off to GPT Responses models",
},
},
"model.loopGuard.enabled": {
type: "boolean",
default: true,
@@ -479,6 +479,11 @@ export class SelectorController {
this.ctx.showError(`Failed to apply vision mode: ${err}`);
});
break;
case "externalThinking":
void this.ctx.session.setThinkToolEnabled(value as boolean).catch(err => {
this.ctx.showError(`Failed to apply external thinking: ${err}`);
});
break;
case "autocompleteMaxVisible":
this.ctx.editor.setAutocompleteMaxVisible(typeof value === "number" ? value : Number(value));
@@ -101,6 +101,15 @@ Invalid args return the schema in the error — fix and retry
TOOL POLICY
==============
{{#has tools "think"}}
# Reasoning
`{{toolRefs.think}}` is your scratchpad and it is where your reasoning actually happens — whatever you do not write there, you have not worked out. Its content is private; the user never sees it.
- MUST call `{{toolRefs.think}}` before the first action of a turn, and again before any step that is expensive to undo: an edit, a destructive command, a final answer.
- Restate what is actually being asked and the constraints given. Split it into ordered sub-problems and solve each explicitly, writing the intermediate result instead of jumping to the conclusion. When a step splits into cases, enumerate them and resolve each.
- Then check the work: verify each claim against the constraints, test one boundary or degenerate case, and look specifically for the error you would most plausibly have made. If a check fails, redo that step — NEVER patch the conclusion.
- Call again only for materially new state: a tool result that changes the plan, a failed check, a sub-problem you had not opened. NEVER use it to narrate progress or restate what you already recorded.
{{/has}}
# General
Use tools whenever they improve correctness, completeness, or grounding.
- SHOULD resolve prerequisites before acting.
+7 -1
View File
@@ -3237,7 +3237,12 @@ async function createAgentSessionScoped(options: CreateAgentSessionOptions): Pro
});
}
}
return settingsAwareStreamFn(streamModel, context, streamOptions);
const externalThinking =
settings.get("externalThinking") && agent.state.tools.some(tool => tool.name === "think");
return settingsAwareStreamFn(streamModel, context, {
...streamOptions,
forceReasoningOff: externalThinking || streamOptions?.forceReasoningOff,
});
},
cursorExecHandlers,
getCursorTools: () => (toolSession.xdev ? listXdevTools(toolSession.xdev) : []),
@@ -3393,6 +3398,7 @@ async function createAgentSessionScoped(options: CreateAgentSessionOptions): Pro
createComputerTool: restrictToolNames
? undefined
: async () => (await BUILTIN_TOOLS.computer(toolSession)) ?? null,
createThinkTool: async () => (await HIDDEN_TOOLS.think(toolSession)) ?? null,
createInspectImageTool: restrictToolNames
? undefined
: async () => (await BUILTIN_TOOLS.inspect_image(toolSession)) ?? null,
@@ -164,6 +164,8 @@ export interface AgentSessionConfig {
createMemoryTools?: () => Promise<AgentTool[]>;
/** Creates the built-in `computer` tool for session-scoped runtime enablement (see {@link AgentSession.setComputerToolEnabled}). */
createComputerTool?: () => Promise<AgentTool | null>;
/** Creates the private `think` scratchpad tool for runtime setting changes. */
createThinkTool?: () => Promise<AgentTool | null>;
/** Creates the built-in `inspect_image` tool for session-scoped runtime enablement (see {@link AgentSession.setInspectImageMode}). */
createInspectImageTool?: () => Promise<AgentTool | null>;
/** Model registry for API key resolution and model discovery. */
@@ -1241,6 +1241,7 @@ export class AgentSession {
toolRegistry: config.toolRegistry,
createVibeTools: config.createVibeTools,
createComputerTool: config.createComputerTool,
createThinkTool: config.createThinkTool,
createInspectImageTool: config.createInspectImageTool,
builtInToolNames: config.builtInToolNames,
mcpManagerToolNames: config.mcpManagerToolNames,
@@ -4419,6 +4420,11 @@ export class AgentSession {
return this.#tools.setComputerToolEnabled(enabled);
}
/** Applies the external-thinking setting to the private scratchpad tool immediately. */
setThinkToolEnabled(enabled: boolean): Promise<boolean> {
return this.#tools.setThinkToolEnabled(enabled);
}
/**
* Session-scoped inspect_image mode (`/vision`). `auto` clears the override
* and returns to the persisted `inspect_image.mode` setting; `on`/`off`
@@ -5176,6 +5182,18 @@ export class AgentSession {
// Skip eager preludes when the user has already queued a directive
const hasPendingUserDirective = this.#toolChoiceQueue.inspect().includes("user-force");
const activeModel = this.agent.state.model;
const externalThinkingToolChoice =
!options?.synthetic &&
!hasPendingUserDirective &&
this.settings.get("externalThinking") &&
this.getEnabledToolNames().includes("think") &&
activeModel &&
(activeModel.api === "openai-responses" ||
activeModel.api === "azure-openai-responses" ||
activeModel.api === "openai-codex-responses")
? buildNamedToolChoice("think", activeModel)
: undefined;
const eagerTodoPrelude =
!options?.synthetic && !hasPendingUserDirective ? this.#todo.createEagerTodoPrelude(expandedText) : undefined;
const eagerTaskPrelude =
@@ -5193,6 +5211,12 @@ export class AgentSession {
: undefined;
const promptAttribution = options?.attribution ?? (options?.synthetic ? "agent" : "user");
if (externalThinkingToolChoice) {
this.#toolChoiceQueue.pushOnce(externalThinkingToolChoice, {
label: "external-thinking",
now: true,
});
}
const message = options?.synthetic
? { role: "developer" as const, content: userContent, attribution: promptAttribution, timestamp: Date.now() }
: { role: "user" as const, content: userContent, attribution: promptAttribution, timestamp: Date.now() };
@@ -5223,6 +5247,7 @@ export class AgentSession {
// Clean up residual eager-todo directive if the prompt never consumed it
// (e.g., compaction aborted, validation failed).
this.#toolChoiceQueue.removeByLabel("eager-todo");
this.#toolChoiceQueue.removeByLabel("external-thinking");
}
return true;
}
@@ -68,6 +68,8 @@ interface SessionToolsOptions {
toolRegistry?: Map<string, AgentTool>;
createVibeTools?: () => AgentTool[];
createComputerTool?: () => Promise<AgentTool | null>;
/** Creates the private `think` scratchpad tool for runtime setting changes. */
createThinkTool?: () => Promise<AgentTool | null>;
/** Creates the built-in `inspect_image` tool for session-scoped runtime enablement (see {@link SessionTools.setInspectImageMode}). */
createInspectImageTool?: () => Promise<AgentTool | null>;
builtInToolNames?: Iterable<string>;
@@ -184,6 +186,7 @@ export class SessionTools {
#toolRegistry: Map<string, AgentTool>;
#createVibeTools: (() => AgentTool[]) | undefined;
#createComputerTool: SessionToolsOptions["createComputerTool"];
#createThinkTool: SessionToolsOptions["createThinkTool"];
#createInspectImageTool: SessionToolsOptions["createInspectImageTool"];
#installedVibeToolNames = new Set<string>();
#builtInToolNames: Set<string>;
@@ -241,6 +244,7 @@ export class SessionTools {
this.#toolRegistry = options.toolRegistry ?? new Map();
this.#createVibeTools = options.createVibeTools;
this.#createComputerTool = options.createComputerTool;
this.#createThinkTool = options.createThinkTool;
this.#createInspectImageTool = options.createInspectImageTool;
this.#builtInToolNames = new Set(options.builtInToolNames ?? []);
this.#mcpManagerToolNames = new Set(options.mcpManagerToolNames ?? []);
@@ -1158,6 +1162,37 @@ export class SessionTools {
});
}
/**
* Session-scoped enable/disable for the private `think` scratchpad tool.
*
* Enabling constructs the tool once and refreshes the model's tool contract;
* disabling removes it from the active set while preserving its registry entry.
*
* @returns false when enabling was requested but this session cannot build the tool.
*/
setThinkToolEnabled(enabled: boolean): Promise<boolean> {
return this.runToolRegistryMutation(async () => {
const active = this.getEnabledToolNames();
if (!enabled) {
if (active.includes("think")) {
await this.#applyActiveToolsByName(active.filter(name => name !== "think"));
}
return true;
}
if (!this.#toolRegistry.has("think")) {
const tool = await this.#createThinkTool?.();
if (tool?.name !== "think") return false;
const wrapped = this.#wrapRuntimeTool(tool);
this.#toolRegistry.set(wrapped.name, wrapped);
this.#builtInToolNames.add(wrapped.name);
}
if (!active.includes("think")) {
await this.#applyActiveToolsByName([...active, "think"]);
}
return true;
});
}
/** Current effective inspect_image state for `/vision status`. */
inspectImageState(): { mode: InspectImageMode; active: boolean; model: string | undefined } {
const model = this.#host.model();
@@ -32,8 +32,7 @@ export const BUILTIN_TOOL_NAMES = [
export type BuiltinToolName = (typeof BUILTIN_TOOL_NAMES)[number];
/** Hidden built-ins: constructible and `--tools`-addressable, but never part of the default active set. */
export const HIDDEN_TOOL_NAMES = ["yield", "goal"] as const;
export const HIDDEN_TOOL_NAMES = ["yield", "goal", "think"] as const;
export type HiddenToolName = (typeof HIDDEN_TOOL_NAMES)[number];
+8
View File
@@ -62,6 +62,7 @@ import { wrapToolWithMetaNotice } from "./output-meta";
import { ReadTool } from "./read";
import type { PlanProposalHandler } from "./resolve";
import { SecurityScanTool } from "./security-scan";
import { ThinkTool } from "./think";
import { type TodoPhase, TodoTool } from "./todo";
import { WriteTool } from "./write";
import { isMountableUnderXdev, type XdevState } from "./xdev";
@@ -102,6 +103,7 @@ export * from "./report-tool-issue";
export * from "./resolve";
export * from "./review";
export * from "./security-scan";
export * from "./think";
export * from "./todo";
export * from "./tts";
export * from "./vibe";
@@ -444,6 +446,7 @@ export const BUILTIN_TOOLS: Record<BuiltinToolName, ToolFactory> = {
};
export const HIDDEN_TOOLS: Record<HiddenToolName, ToolFactory> = {
think: () => new ThinkTool(),
yield: s => new YieldTool(s),
goal: s => new GoalTool(s),
};
@@ -562,6 +565,9 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
if (session.settings.get("memory.backend") === "mnemopi" && !requestedTools.includes("memory_edit")) {
requestedTools.push("memory_edit");
}
if (session.settings.get("externalThinking") && !requestedTools.includes("think")) {
requestedTools.push("think");
}
// Auto-learn tools are gated by `autolearn.enabled` but, like the memory
// tools above, must also be force-included into an explicit requestedTools
// list so a restricted top-level session whose controller/guidance is
@@ -604,6 +610,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
if (name === "inspect_image") return isInspectImageToolActive(session);
if (name === "web_search") return session.settings.get("web_search.enabled");
if (name === "security_scan") return session.settings.get("security.enabled");
if (name === "think") return session.settings.get("externalThinking");
if (name === "ask") return session.settings.get("ask.enabled");
if (name === "browser") return session.settings.get("browser.enabled");
if (name === "computer") return session.settings.get("computer.enabled");
@@ -650,6 +657,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
...Object.entries(BUILTIN_TOOLS)
.filter(([name]) => isToolAllowed(name))
.map(([name, factory]) => [name, factory] as const),
...(session.settings.get("externalThinking") ? ([["think", HIDDEN_TOOLS.think]] as const) : []),
...(includeYield ? ([["yield", HIDDEN_TOOLS.yield]] as const) : []),
...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []),
];
@@ -27,6 +27,7 @@ import { inspectImageToolRenderer } from "./inspect-image-renderer";
import { recallToolRenderer, reflectToolRenderer, retainToolRenderer } from "./memory-render";
import { readToolRenderer } from "./read";
import { resolveRenderer } from "./resolve";
import { thinkToolRenderer } from "./think";
import { todoToolRenderer } from "./todo";
import { createVibeToolRenderer } from "./vibe";
import { writeToolRenderer } from "./write";
@@ -115,6 +116,7 @@ export const toolRenderers: Record<string, ToolRenderer> = {
get task(): ToolRenderer {
return taskToolRenderer as ToolRenderer;
},
think: thinkToolRenderer as ToolRenderer,
todo: todoToolRenderer as ToolRenderer,
github: githubToolRenderer as ToolRenderer,
goal: goalToolRenderer as ToolRenderer,
+62
View File
@@ -0,0 +1,62 @@
import { type } from "@oh-my-pi/omptype";
import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { type Component, Markdown } from "@oh-my-pi/pi-tui";
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
import { getMarkdownTheme, type Theme } from "../modes/theme/theme";
const thinkSchema = type({
thoughts: type("string").describe("private scratchpad reasoning to retain before the next response"),
"+": "reject",
}).describe("record private intermediate reasoning before answering");
type ThinkParams = typeof thinkSchema.infer;
export type ThinkRenderArgs = {
thoughts?: string;
};
export const thinkToolRenderer = {
inline: true,
renderCall(args: ThinkRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component {
const thoughts =
typeof args === "object" && args !== null && "thoughts" in args && typeof args.thoughts === "string"
? args.thoughts
: "";
return new Markdown(thoughts, 1, 0, getMarkdownTheme(), {
color: (text: string) => uiTheme.fg("thinkingText", text),
italic: true,
});
},
renderResult(): Component {
return undefined as unknown as Component;
},
};
interface ThinkToolDetails {
recorded: true;
}
/** Records private intermediate reasoning while native GPT reasoning is disabled. */
export class ThinkTool implements AgentTool<typeof thinkSchema, ThinkToolDetails> {
readonly name = "think";
readonly approval = "read" as const;
readonly label = "Think";
readonly summary = "Record private intermediate reasoning before answering";
readonly description =
"Use this private scratchpad to plan, derive, or check work before answering. Record only materially new reasoning. The user does not see this tool activity.";
readonly parameters = thinkSchema;
readonly strict = true;
readonly intent = "omit" as const;
async execute(_toolCallId: string, _params: ThinkParams): Promise<AgentToolResult<ThinkToolDetails>> {
return {
content: [
{
type: "text",
text: "------",
},
],
details: { recorded: true },
};
}
}
@@ -147,6 +147,153 @@ describe("createAgentSession defaultInactive tool activation", () => {
}
});
it("activates the private think tool when external thinking is enabled at runtime", async () => {
const tempDir = makeTempDir();
const settings = Settings.isolated();
const { session } = await createAgentSession({
...baseOptions(tempDir),
settings,
});
try {
expect(session.getToolByName("think")).toBeUndefined();
expect(session.getActiveToolNames()).not.toContain("think");
settings.set("externalThinking", true);
await session.setThinkToolEnabled(true);
expect(session.getToolByName("think")).toBeDefined();
expect(session.getActiveToolNames()).toContain("think");
expect(session.getXdevToolEntries().map(entry => entry.name)).not.toContain("think");
settings.set("externalThinking", false);
await session.setThinkToolEnabled(false);
expect(session.getActiveToolNames()).not.toContain("think");
} finally {
await session.dispose();
}
});
it("activates the private think tool at startup when external thinking is configured", async () => {
const tempDir = makeTempDir();
const settings = Settings.isolated({ externalThinking: true });
const { session } = await createAgentSession({
...baseOptions(tempDir),
settings,
});
try {
expect(session.getToolByName("think")).toBeDefined();
expect(session.getActiveToolNames()).toContain("think");
expect(session.getXdevToolEntries().map(entry => entry.name)).not.toContain("think");
} finally {
await session.dispose();
}
});
it("forces think and sends reasoning effort off for a Responses turn", async () => {
const tempDir = makeTempDir();
const settings = Settings.isolated({ externalThinking: true });
const requestTexts: string[] = [];
const sse = (events: unknown[]): Response =>
new Response(events.map(event => `data: ${JSON.stringify(event)}\n\n`).join(""), {
headers: { "content-type": "text/event-stream" },
});
const completed = (id: string) => ({
type: "response.completed",
response: {
id,
status: "completed",
usage: {
input_tokens: 1,
output_tokens: 1,
total_tokens: 2,
input_tokens_details: { cached_tokens: 0 },
},
},
});
const server = Bun.serve({
port: 0,
fetch: async request => {
requestTexts.push(await request.text());
if (requestTexts.length === 1) {
const argumentsJson = JSON.stringify({ thoughts: "Checked the request before answering." });
return sse([
{
type: "response.output_item.added",
output_index: 0,
item: {
type: "function_call",
id: "fc_think",
call_id: "call_think",
name: "think",
arguments: "",
},
},
{
type: "response.function_call_arguments.done",
output_index: 0,
item_id: "fc_think",
arguments: argumentsJson,
},
{
type: "response.output_item.done",
output_index: 0,
item: {
type: "function_call",
id: "fc_think",
call_id: "call_think",
name: "think",
arguments: argumentsJson,
},
},
completed("resp_think"),
]);
}
return sse([
{ type: "response.output_text.delta", output_index: 0, delta: "Done." },
{
type: "response.output_item.done",
output_index: 0,
item: {
type: "message",
id: "msg_done",
role: "assistant",
status: "completed",
content: [{ type: "output_text", text: "Done." }],
},
},
completed("resp_done"),
]);
},
});
const model = getBundledModel("openai", "gpt-5");
if (!model) throw new Error("Expected gpt-5 model to exist");
const { session } = await createAgentSession({
...baseOptions(tempDir),
settings,
model: { ...model, baseUrl: `${server.url}v1` },
getApiKey: () => "test-key",
});
expect(session.getActiveToolNames()).toContain("think");
try {
await session.prompt("Use the scratchpad before answering.");
const firstRequest = requestTexts.at(0);
if (!firstRequest) throw new Error("Expected the initial provider request.");
expect(requestTexts).toHaveLength(2);
expect(JSON.parse(firstRequest)).toEqual(
expect.objectContaining({
reasoning: { effort: "off" },
tool_choice: expect.objectContaining({ name: "think" }),
}),
);
} finally {
await session.dispose();
server.stop(true);
}
});
it("publishes tools from lazy session startup before the input lifecycle completes", async () => {
const tempDir = makeTempDir();
const startupGate = Promise.withResolvers<void>();
@@ -0,0 +1,45 @@
import { beforeAll, describe, expect, it } from "bun:test";
import { getThemeByName, initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import { thinkToolRenderer } from "../../src/tools/think";
beforeAll(async () => {
await initTheme();
});
describe("thinkToolRenderer", () => {
it("renders thoughts with thinkingText color and italic style", async () => {
const theme = await getThemeByName("dark");
expect(theme).toBeDefined();
const uiTheme = theme!;
const callComponent = thinkToolRenderer.renderCall(
{ thoughts: "Analyzing the solution step by step." },
{ expanded: true, isPartial: false },
uiTheme,
);
expect(callComponent).toBeDefined();
const lines = callComponent.render(100);
const fullText = lines.join("\n");
expect(fullText).toContain("Analyzing the solution step by step.");
expect(fullText).toContain(uiTheme.fg("thinkingText", "Analyzing the solution step by step."));
});
it("has inline set to true", () => {
expect(thinkToolRenderer.inline).toBe(true);
});
it("returns undefined for renderResult", () => {
expect(thinkToolRenderer.renderResult()).toBeUndefined();
});
it("handles empty or missing thoughts gracefully", async () => {
const theme = await getThemeByName("dark");
const uiTheme = theme!;
const emptyCall = thinkToolRenderer.renderCall({}, { expanded: true, isPartial: false }, uiTheme);
expect(emptyCall).toBeDefined();
expect(emptyCall.render(100)).toEqual([]);
});
});