feat: introduced external thinking support and private scratchpad think tool
- Added support for external thinking and forced reasoning disablement across AI provider options and request transformers. - Implemented the private scratchpad think tool along with its renderer, system prompt rules, and schema configuration. - Updated agent session management and SDK tools to support dynamic runtime activation of the think tool via the externalThinking setting. - Added comprehensive unit tests covering reasoning fallbacks, tool activation, and rendering behavior.
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `forceReasoningOff` and `disableReasoning` options to disable reasoning in OpenAI and Azure OpenAI models
|
||||
|
||||
## [17.2.13] - 2026-08-11
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -65,6 +65,7 @@ export interface AzureOpenAIResponsesOptions extends StreamOptions {
|
||||
azureDeploymentName?: string;
|
||||
toolChoice?: ToolChoice;
|
||||
serviceTier?: ServiceTier;
|
||||
disableReasoning?: boolean;
|
||||
}
|
||||
|
||||
type AzureOpenAIResponsesSamplingParams = ResponseCreateParamsStreaming & {
|
||||
|
||||
@@ -1529,6 +1529,7 @@ export async function buildTransformedCodexRequestBody(
|
||||
}
|
||||
const codexOptions: CodexRequestOptions = {
|
||||
reasoningEffort: options?.reasoning,
|
||||
reasoningOff: options?.forceReasoningOff,
|
||||
reasoningSummary: options?.reasoningSummary,
|
||||
reasoningContext: options?.reasoningContext,
|
||||
textVerbosity: options?.textVerbosity,
|
||||
|
||||
@@ -32,6 +32,8 @@ export interface ReasoningConfig {
|
||||
export interface CodexRequestOptions {
|
||||
/** User-facing effort; maps 1:1 onto the wire tier of the same name. */
|
||||
reasoningEffort?: CodexCallerEffort | "none";
|
||||
/** Suppress native reasoning by sending `reasoning.effort: "none"`. */
|
||||
reasoningOff?: boolean;
|
||||
reasoningSummary?: ReasoningConfig["summary"] | null;
|
||||
/** Explicit `reasoning.context` override. Omitted by default; Responses Lite forces `all_turns` as required by that transport. */
|
||||
reasoningContext?: CodexReasoningContext;
|
||||
@@ -454,9 +456,12 @@ export async function transformRequestBody(
|
||||
applyCodexResponsesLiteShape(body);
|
||||
}
|
||||
|
||||
if (options.reasoningEffort !== undefined || responsesLite) {
|
||||
const reasoningConfig =
|
||||
options.reasoningEffort !== undefined ? getReasoningConfig(model, options.reasoningEffort, options) : {};
|
||||
if (options.reasoningOff || options.reasoningEffort !== undefined || responsesLite) {
|
||||
const reasoningConfig: Partial<ReasoningConfig> = options.reasoningOff
|
||||
? { effort: "none" }
|
||||
: options.reasoningEffort !== undefined
|
||||
? getReasoningConfig(model, options.reasoningEffort, options)
|
||||
: {};
|
||||
body.reasoning = {
|
||||
...body.reasoning,
|
||||
...reasoningConfig,
|
||||
@@ -478,7 +483,7 @@ export async function transformRequestBody(
|
||||
// Catalog pro aliases (`gpt-5.6-*-pro`): applied after the effort branch so
|
||||
// the mode is sent even when no effort is set (the branch above deletes
|
||||
// `body.reasoning` in that case) — mode and effort are independent fields.
|
||||
if (model.reasoningMode) {
|
||||
if (model.reasoningMode && !options.reasoningOff) {
|
||||
body.reasoning = { ...body.reasoning, mode: model.reasoningMode };
|
||||
}
|
||||
|
||||
|
||||
@@ -132,7 +132,13 @@ function collectMessageParts(error: unknown, captured: CapturedHttpErrorResponse
|
||||
return parts.join("\n");
|
||||
}
|
||||
|
||||
const REASONING_EFFORT_FIELD_PATTERN = /reasoning[_. ]effort|reasoning value/i;
|
||||
/**
|
||||
* Text that identifies a 400 as being about the reasoning-effort field.
|
||||
* OpenAI-compatible gateways (cliproxy, …) never name the field — they reject
|
||||
* the value alone with `level "none" not supported, valid levels: low, …` — so
|
||||
* the allowed-level phrasing counts as a mention too.
|
||||
*/
|
||||
const REASONING_EFFORT_FIELD_PATTERN = /reasoning[_. ]effort|reasoning value|(?:valid|supported|allowed) levels?/i;
|
||||
|
||||
function mentionsReasoningEffort(error: unknown, captured: CapturedHttpErrorResponse | undefined): boolean {
|
||||
const param = capturedStringField(captured, "param");
|
||||
@@ -168,10 +174,13 @@ function isInvalidReasoningEffortError(
|
||||
if (/(?:unsupported|not supported)[^\n]*(?:reasoning[_. ]effort|reasoning value)/i.test(message)) {
|
||||
return true;
|
||||
}
|
||||
return new RegExp(
|
||||
`(?:invalid|unsupported|not supported)[^\\n]*["'\`]${escapeRegExp(currentEffort)}["'\`]`,
|
||||
"i",
|
||||
).test(message);
|
||||
// Gateways put the rejected value first (`level "none" not supported`), the
|
||||
// official API puts the verdict first (`Unsupported value: 'none'`).
|
||||
const quoted = `["'\`]${escapeRegExp(currentEffort)}["'\`]`;
|
||||
return (
|
||||
new RegExp(`(?:invalid|unsupported|not supported)[^\\n]*${quoted}`, "i").test(message) ||
|
||||
new RegExp(`${quoted}[^\\n]*(?:invalid|unsupported|not supported)`, "i").test(message)
|
||||
);
|
||||
}
|
||||
|
||||
function escapeRegExp(value: string): string {
|
||||
@@ -186,9 +195,12 @@ function parseKnownReasoningValues(text: string): Set<string> {
|
||||
values.add(quotedMatch[1]!.toLowerCase());
|
||||
quotedMatch = quotedPattern.exec(text);
|
||||
}
|
||||
const allowedMatch = /(?:must be|one of|allowed values?|supported values?(?: are)?|expected)([^.\n]+)/i.exec(text);
|
||||
const allowedMatch =
|
||||
/(?:must be|one of|allowed values?|supported values?(?: are)?|expected|(?:valid|supported|allowed) levels?(?: are)?)[^.\n]+/i.exec(
|
||||
text,
|
||||
);
|
||||
if (allowedMatch) {
|
||||
const allowedText = allowedMatch[1]!;
|
||||
const allowedText = allowedMatch[0]!;
|
||||
const barePattern = /\b(none|minimal|low|medium|high|xhigh|max)\b/gi;
|
||||
let bareMatch = barePattern.exec(allowedText);
|
||||
while (bareMatch !== null) {
|
||||
@@ -201,7 +213,8 @@ function parseKnownReasoningValues(text: string): Set<string> {
|
||||
|
||||
function parseAllowedReasoningValues(message: string, currentEffort: string): Set<string> | undefined {
|
||||
const values = parseKnownReasoningValues(message);
|
||||
const hasAllowedCue = /must be|one of|allowed values?|supported values?|expected/i.test(message);
|
||||
const hasAllowedCue =
|
||||
/must be|one of|allowed values?|supported values?|expected|(?:valid|supported|allowed) levels?/i.test(message);
|
||||
values.delete(currentEffort.toLowerCase());
|
||||
if (!hasAllowedCue && values.size === 0) return undefined;
|
||||
return values;
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { scheduler } from "node:timers/promises";
|
||||
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import { bareModelId, parseOpenAIModel, semverGte } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { $flag, logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
||||
import * as AIError from "../error";
|
||||
import { getEnvApiKey } from "../stream";
|
||||
@@ -80,6 +81,7 @@ import {
|
||||
createInitialResponsesAssistantMessage,
|
||||
createOpenAIStrictToolsState,
|
||||
disableStrictToolsForScope,
|
||||
getJuiceValue,
|
||||
getOpenAIPromptCacheKey,
|
||||
getOpenAIResponsesRoutingSessionId,
|
||||
getOpenAIStrictToolsScope,
|
||||
@@ -298,14 +300,23 @@ interface OpenAIResponsesChainedParams {
|
||||
*/
|
||||
function buildOpenAIResponsesChainedParams(
|
||||
params: OpenAIResponsesSamplingParams,
|
||||
trailingScaffoldingItems: number,
|
||||
chain: OpenAIResponsesChainState,
|
||||
): OpenAIResponsesChainedParams {
|
||||
const historyParams =
|
||||
trailingScaffoldingItems > 0 && Array.isArray(params.input)
|
||||
? { ...params, input: params.input.slice(0, params.input.length - trailingScaffoldingItems) }
|
||||
: params;
|
||||
const deltaInput = chain.canAppend
|
||||
? buildResponsesDeltaInput(chain.lastParams, chain.lastResponseItems, params)
|
||||
? buildResponsesDeltaInput(chain.lastParams, chain.lastResponseItems, historyParams)
|
||||
: null;
|
||||
if (deltaInput && deltaInput.length > 0 && chain.lastResponseId) {
|
||||
const scaffolding =
|
||||
historyParams !== params && Array.isArray(params.input)
|
||||
? params.input.slice(params.input.length - trailingScaffoldingItems)
|
||||
: [];
|
||||
return {
|
||||
params: { ...params, previous_response_id: chain.lastResponseId, input: deltaInput },
|
||||
params: { ...params, previous_response_id: chain.lastResponseId, input: [...deltaInput, ...scaffolding] },
|
||||
previousResponseId: chain.lastResponseId,
|
||||
};
|
||||
}
|
||||
@@ -462,8 +473,9 @@ const streamOpenAIResponsesOnce = (
|
||||
false,
|
||||
chainState?.canAppend ? chainState.lastParams?.input : undefined,
|
||||
);
|
||||
const params = builtParams.params;
|
||||
const { params, trailingScaffoldingItems } = builtParams;
|
||||
let activeParams = params;
|
||||
let activeTrailingScaffoldingItems = trailingScaffoldingItems;
|
||||
const resolvedBaseUrl = (baseUrl ?? "https://api.openai.com/v1").replace(/\/+$/, "");
|
||||
const requestReasoningEffortFallbacks = new Map<string, OpenAIReasoningEffortFallback>();
|
||||
const attemptedReasoningEffortFallbacks = new Set<string>();
|
||||
@@ -490,7 +502,9 @@ const streamOpenAIResponsesOnce = (
|
||||
}
|
||||
applyReasoningEffortFallbackForRequest(params);
|
||||
let chained: OpenAIResponsesChainedParams =
|
||||
chainState && !chainState.disabled ? buildOpenAIResponsesChainedParams(params, chainState) : { params };
|
||||
chainState && !chainState.disabled
|
||||
? buildOpenAIResponsesChainedParams(params, trailingScaffoldingItems, chainState)
|
||||
: { params };
|
||||
sentPreviousResponseId = chained.previousResponseId;
|
||||
const idleTimeoutMs =
|
||||
options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(model.compat.streamIdleTimeoutMs);
|
||||
@@ -586,7 +600,9 @@ const streamOpenAIResponsesOnce = (
|
||||
const reasoningEffortFallback =
|
||||
activeReasoningEffortFallbackKey && activeRequestParams && !requestSignal.aborted
|
||||
? resolveOpenAIReasoningEffortFallback(error, capturedErrorResponse, activeRequestParams, {
|
||||
explicitDisable: options?.disableReasoning === true && options.reasoning === undefined,
|
||||
explicitDisable:
|
||||
options?.forceReasoningOff === true ||
|
||||
(options?.disableReasoning === true && options.reasoning === undefined),
|
||||
})
|
||||
: undefined;
|
||||
if (reasoningEffortFallback !== undefined && activeReasoningEffortFallbackKey) {
|
||||
@@ -632,7 +648,11 @@ const streamOpenAIResponsesOnce = (
|
||||
if (chainState && !chainState.disabled) fallbackParams.store = true;
|
||||
let fallbackChained: OpenAIResponsesChainedParams =
|
||||
chainState && !chainState.disabled
|
||||
? buildOpenAIResponsesChainedParams(fallbackParams, chainState)
|
||||
? buildOpenAIResponsesChainedParams(
|
||||
fallbackParams,
|
||||
fallbackBuilt.trailingScaffoldingItems,
|
||||
chainState,
|
||||
)
|
||||
: { params: fallbackParams };
|
||||
sentPreviousResponseId = fallbackChained.previousResponseId;
|
||||
fallbackChained = {
|
||||
@@ -642,7 +662,7 @@ const streamOpenAIResponsesOnce = (
|
||||
chained = fallbackChained;
|
||||
activeRawRequestDump.body = chained.params;
|
||||
activeParams = fallbackParams;
|
||||
activeStrictToolsApplied = fallbackBuilt.strictToolsApplied;
|
||||
activeTrailingScaffoldingItems = fallbackBuilt.trailingScaffoldingItems;
|
||||
continue;
|
||||
}
|
||||
if (!chainState || !sentPreviousResponseId || requestSignal.aborted) {
|
||||
@@ -688,6 +708,7 @@ const streamOpenAIResponsesOnce = (
|
||||
chained = { params: retryParams };
|
||||
activeRawRequestDump.body = retryParams;
|
||||
activeParams = currentParams;
|
||||
activeTrailingScaffoldingItems = currentBuilt.trailingScaffoldingItems;
|
||||
activeStrictToolsApplied = currentBuilt.strictToolsApplied;
|
||||
}
|
||||
}
|
||||
@@ -824,7 +845,17 @@ const streamOpenAIResponsesOnce = (
|
||||
if (replayableResponseItems) {
|
||||
if (providerSessionState) providerSessionState.nativeHistoryReplayWarmed = true;
|
||||
if (chainState) {
|
||||
chainState.lastParams = structuredCloneJSON(activeParams);
|
||||
chainState.lastParams = structuredCloneJSON(
|
||||
activeTrailingScaffoldingItems > 0 && Array.isArray(activeParams.input)
|
||||
? {
|
||||
...activeParams,
|
||||
input: activeParams.input.slice(
|
||||
0,
|
||||
activeParams.input.length - activeTrailingScaffoldingItems,
|
||||
),
|
||||
}
|
||||
: activeParams,
|
||||
);
|
||||
chainState.lastPromptCacheBreakpointPolicy = promptCacheBreakpointPolicy;
|
||||
if (output.responseId) {
|
||||
chainState.lastResponseId = output.responseId;
|
||||
@@ -843,7 +874,14 @@ const streamOpenAIResponsesOnce = (
|
||||
// baseline, but `lastParams` still records the successful wire controls
|
||||
// without re-enabling `previous_response_id` chaining.
|
||||
chainState.canAppend = false;
|
||||
chainState.lastParams = structuredCloneJSON(activeParams);
|
||||
chainState.lastParams = structuredCloneJSON(
|
||||
activeTrailingScaffoldingItems > 0 && Array.isArray(activeParams.input)
|
||||
? {
|
||||
...activeParams,
|
||||
input: activeParams.input.slice(0, activeParams.input.length - activeTrailingScaffoldingItems),
|
||||
}
|
||||
: activeParams,
|
||||
);
|
||||
chainState.lastPromptCacheBreakpointPolicy = promptCacheBreakpointPolicy;
|
||||
chainState.lastResponseId = undefined;
|
||||
chainState.lastResponseItems = undefined;
|
||||
@@ -899,6 +937,17 @@ function isOfficialOpenAIResponsesEndpoint(model: Model<"openai-responses">): bo
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* GPT-5.6+ family check for Responses routes. The model id classifies the
|
||||
* reasoning family regardless of the provider/host serving it — a cliproxy or
|
||||
* other OpenAI-compatible gateway carrying `gpt-5.6-sol` gets the same
|
||||
* scaffolding as the official endpoint.
|
||||
*/
|
||||
function isGpt56PlusResponsesModel(model: Model<"openai-responses">): boolean {
|
||||
const parsed = parseOpenAIModel(bareModelId(model.requestModelId ?? model.id));
|
||||
return parsed !== null && semverGte(parsed.version, "5.6");
|
||||
}
|
||||
|
||||
function isResponsesPromptCacheableContentBlock(block: unknown): block is ResponseInputContent {
|
||||
if (typeof block !== "object" || block === null || !("type" in block)) return false;
|
||||
return block.type === "input_text" || block.type === "input_image" || block.type === "input_file";
|
||||
@@ -1090,7 +1139,7 @@ export function buildParams(
|
||||
strictToolsScope?: OpenAIStrictToolsScope,
|
||||
disableStrictToolsOverride = false,
|
||||
statefulCacheBaseline?: ResponseInput,
|
||||
): { params: OpenAIResponsesSamplingParams; strictToolsApplied: boolean } {
|
||||
): { params: OpenAIResponsesSamplingParams; trailingScaffoldingItems: number; strictToolsApplied: boolean } {
|
||||
const policy = resolveOpenAICompatPolicy(model, {
|
||||
endpoint: "responses",
|
||||
reasoning: options?.reasoning,
|
||||
@@ -1244,6 +1293,7 @@ export function buildParams(
|
||||
: options?.reasoningSummary;
|
||||
applyResponsesCompatPolicy(params, reasoningPolicy, {
|
||||
reasoningSummary,
|
||||
forceReasoningOff: options?.forceReasoningOff,
|
||||
mapEffort: effort =>
|
||||
model.compat.reasoningEffortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
|
||||
model.thinking?.effortMap?.[effort as NonNullable<OpenAIResponsesOptions["reasoning"]>] ??
|
||||
@@ -1253,7 +1303,7 @@ export function buildParams(
|
||||
// mode survives every policy branch (disabled/omitted effort included) while
|
||||
// keeping whatever effort/summary the policy produced — mode and effort are
|
||||
// independent wire fields.
|
||||
if (model.reasoningMode) {
|
||||
if (model.reasoningMode && !options?.forceReasoningOff) {
|
||||
params.reasoning = { ...params.reasoning, mode: model.reasoningMode };
|
||||
}
|
||||
|
||||
@@ -1266,7 +1316,18 @@ export function buildParams(
|
||||
applyOpenAIExtraBody(params, options?.extraBody);
|
||||
applyOpenAIResponsesPromptCachePolicy(params, model, options, statefulCacheBaseline);
|
||||
|
||||
return { params, strictToolsApplied };
|
||||
let trailingScaffoldingItems = 0;
|
||||
if (options?.forceReasoningOff && isGpt56PlusResponsesModel(model)) {
|
||||
const effort = options.reasoning ?? "medium";
|
||||
const juice = getJuiceValue(effort);
|
||||
messages.push({
|
||||
role: "developer",
|
||||
content: [{ type: "input_text", text: `# Juice: ${juice} !important` }],
|
||||
});
|
||||
trailingScaffoldingItems = 1;
|
||||
}
|
||||
|
||||
return { params, trailingScaffoldingItems, strictToolsApplied };
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -2428,6 +2428,20 @@ export function finalizeMessageText(item: ResponseOutputMessage, streamedText: s
|
||||
if (!item.content?.length) return streamedText || "";
|
||||
return item.content.map(part => (part.type === "output_text" ? (part.text ?? "") : (part.refusal ?? ""))).join("");
|
||||
}
|
||||
export const JUICE_EFFORT_MAP: Record<string, number> = {
|
||||
none: 0,
|
||||
minimal: 2,
|
||||
low: 4,
|
||||
medium: 8,
|
||||
high: 48,
|
||||
xhigh: 112,
|
||||
max: 960,
|
||||
};
|
||||
|
||||
export function getJuiceValue(effort?: string): number {
|
||||
if (!effort) return 8;
|
||||
return JUICE_EFFORT_MAP[effort] ?? 8;
|
||||
}
|
||||
|
||||
export function accumulateToolCallArgumentsDelta(
|
||||
block: ResponsesToolCallBlock,
|
||||
@@ -3308,6 +3322,14 @@ type ReasoningOptions = {
|
||||
export interface ApplyResponsesCompatPolicyOptions {
|
||||
reasoningSummary?: "auto" | "detailed" | "concise" | null;
|
||||
mapEffort?: (effort: string) => string;
|
||||
/**
|
||||
* Suppress native reasoning by sending `reasoning.effort: "none"` — the only
|
||||
* disable level the Responses API defines (`"off"` is not a wire value and
|
||||
* 400s everywhere). Gateways that reject `none` for a given model are
|
||||
* handled by the reasoning-effort fallback retry, which clamps to the
|
||||
* lowest level the error reports as allowed.
|
||||
*/
|
||||
forceReasoningOff?: boolean;
|
||||
}
|
||||
|
||||
export function applyResponsesCompatPolicy<P extends ResponseCreateParamsStreaming>(
|
||||
@@ -3316,6 +3338,10 @@ export function applyResponsesCompatPolicy<P extends ResponseCreateParamsStreami
|
||||
options: ApplyResponsesCompatPolicyOptions | undefined,
|
||||
): void {
|
||||
const reasoning = policy.reasoning;
|
||||
if (options?.forceReasoningOff) {
|
||||
params.reasoning = { effort: "none" } as P["reasoning"];
|
||||
return;
|
||||
}
|
||||
if (!reasoning.modelSupported) return;
|
||||
if (reasoning.includeEncryptedReasoning) {
|
||||
const include = params.include ?? [];
|
||||
|
||||
@@ -1692,6 +1692,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
openrouterVariant: options?.openrouterVariant,
|
||||
maxTokensExplicit: rawOptions?.maxTokens !== undefined,
|
||||
disableReasoning: options?.disableReasoning,
|
||||
forceReasoningOff: options?.forceReasoningOff,
|
||||
textVerbosity: options?.textVerbosity,
|
||||
promptCache: options?.promptCache,
|
||||
statefulResponses: options?.statefulResponses,
|
||||
@@ -1706,6 +1707,8 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
|
||||
promptCache: options?.promptCache,
|
||||
statefulResponses: options?.statefulResponses,
|
||||
disableReasoning: options?.disableReasoning || options?.forceReasoningOff,
|
||||
forceReasoningOff: options?.forceReasoningOff,
|
||||
});
|
||||
|
||||
case "openai-codex-responses":
|
||||
@@ -1718,6 +1721,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
codexCompaction: options?.codexCompaction,
|
||||
reasoningSummary: options?.hideThinkingSummary ? null : undefined,
|
||||
textVerbosity: options?.textVerbosity,
|
||||
forceReasoningOff: options?.forceReasoningOff,
|
||||
});
|
||||
|
||||
case "google-generative-ai": {
|
||||
|
||||
@@ -478,6 +478,11 @@ export interface StreamOptions {
|
||||
* `false` so `previous_response_id` cannot explain a result.
|
||||
*/
|
||||
statefulResponses?: boolean;
|
||||
/**
|
||||
* Emit `reasoning: { effort: "none" }` for OpenAI Responses and Codex requests.
|
||||
* Used when a caller supplies an external reasoning scratchpad; other transports ignore it.
|
||||
*/
|
||||
forceReasoningOff?: boolean;
|
||||
/**
|
||||
* Provider-scoped mutable state store for this agent session.
|
||||
* Providers can use this to persist transport/session state between turns.
|
||||
|
||||
@@ -166,6 +166,14 @@ describe("openai-codex optional response controls", () => {
|
||||
expect("stream_options" in suppressed).toBe(false);
|
||||
});
|
||||
|
||||
it("disables native reasoning with effort none when an external scratchpad replaces it", async () => {
|
||||
const model = createCodexModel("gpt-5.5");
|
||||
const body = await buildTransformedCodexRequestBody(model, createCodexTestContext(), {
|
||||
forceReasoningOff: true,
|
||||
});
|
||||
expect(body.reasoning).toEqual({ effort: "none" });
|
||||
});
|
||||
|
||||
it("forces reasoning.context to all_turns for Responses Lite", async () => {
|
||||
const model = createCodexModel("gpt-5.5");
|
||||
|
||||
|
||||
@@ -108,6 +108,18 @@ function pipeDelimitedReasoningEffortResponse(): Response {
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* cliproxy-style gateway rejection: the field is never named and the rejected
|
||||
* value comes before the verdict (`level "none" not supported, valid levels: …`).
|
||||
*/
|
||||
function unsupportedLevelResponse(value: string): Response {
|
||||
const message = `level "${value}" not supported, valid levels: low, medium, high, xhigh, max`;
|
||||
return new Response(JSON.stringify({ error: { message, type: "invalid_request_error" } }), {
|
||||
status: 400,
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
function summaryReasoningErrorResponse(): Response {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
@@ -343,6 +355,28 @@ describe("OpenAI reasoning effort fallback retry", () => {
|
||||
expect(bodies.map(body => (body.reasoning as { effort?: string } | undefined)?.effort)).toEqual(["xhigh", "max"]);
|
||||
});
|
||||
|
||||
it("clamps a rejected reasoning-off request to the lowest level the gateway allows", async () => {
|
||||
const bodies: Record<string, unknown>[] = [];
|
||||
const fetchMock: FetchImpl = Object.assign(
|
||||
async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
|
||||
const body = parseJsonBody(init);
|
||||
bodies.push(body);
|
||||
return bodies.length === 1 ? unsupportedLevelResponse("none") : createResponsesSseResponse();
|
||||
},
|
||||
{ preconnect: fetch.preconnect },
|
||||
);
|
||||
|
||||
const result = await streamOpenAIResponses(createMaxLadderResponsesModel(), testContext, {
|
||||
apiKey: "test-key",
|
||||
fetch: fetchMock,
|
||||
reasoning: "high",
|
||||
forceReasoningOff: true,
|
||||
}).result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(bodies.map(body => (body.reasoning as { effort?: string } | undefined)?.effort)).toEqual(["none", "low"]);
|
||||
});
|
||||
|
||||
it("does not retry unrelated reasoning parameter errors", async () => {
|
||||
let attempts = 0;
|
||||
const fetchMock: FetchImpl = Object.assign(
|
||||
|
||||
@@ -33,9 +33,12 @@ const ctx: Context = {
|
||||
messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
|
||||
};
|
||||
|
||||
async function drain(model: Model<"openai-responses">): Promise<Record<string, unknown>> {
|
||||
async function drain(
|
||||
model: Model<"openai-responses">,
|
||||
options: { forceReasoningOff?: boolean } = {},
|
||||
): Promise<Record<string, unknown>> {
|
||||
const { fetchMock, captured } = mockSseFetch();
|
||||
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock, temperature: 0 });
|
||||
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock, temperature: 0, ...options });
|
||||
for await (const event of stream) {
|
||||
if (event.type === "done" || event.type === "error") break;
|
||||
}
|
||||
@@ -67,4 +70,10 @@ describe("openai-responses sampling-param gating (#5606)", () => {
|
||||
const body = await drain(model);
|
||||
expect(body.temperature).toBe(0);
|
||||
});
|
||||
|
||||
it("disables native reasoning with effort none when an external scratchpad replaces it", async () => {
|
||||
const model = getBundledModel("openai", "gpt-5") as Model<"openai-responses">;
|
||||
const body = await drain(model, { forceReasoningOff: true });
|
||||
expect(body.reasoning).toEqual({ effort: "none" });
|
||||
});
|
||||
});
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `externalThinking` setting for private scratchpad reasoning via the new `think` tool
|
||||
|
||||
## [17.2.13] - 2026-08-11
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1139,6 +1139,17 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
externalThinking: {
|
||||
type: "boolean",
|
||||
default: false,
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Thinking",
|
||||
label: "External Thinking",
|
||||
description: "Use a private think tool and send reasoning effort off to GPT Responses models",
|
||||
},
|
||||
},
|
||||
|
||||
"model.loopGuard.enabled": {
|
||||
type: "boolean",
|
||||
default: true,
|
||||
|
||||
@@ -479,6 +479,11 @@ export class SelectorController {
|
||||
this.ctx.showError(`Failed to apply vision mode: ${err}`);
|
||||
});
|
||||
break;
|
||||
case "externalThinking":
|
||||
void this.ctx.session.setThinkToolEnabled(value as boolean).catch(err => {
|
||||
this.ctx.showError(`Failed to apply external thinking: ${err}`);
|
||||
});
|
||||
break;
|
||||
|
||||
case "autocompleteMaxVisible":
|
||||
this.ctx.editor.setAutocompleteMaxVisible(typeof value === "number" ? value : Number(value));
|
||||
|
||||
@@ -101,6 +101,15 @@ Invalid args return the schema in the error — fix and retry
|
||||
TOOL POLICY
|
||||
==============
|
||||
|
||||
{{#has tools "think"}}
|
||||
# Reasoning
|
||||
`{{toolRefs.think}}` is your scratchpad and it is where your reasoning actually happens — whatever you do not write there, you have not worked out. Its content is private; the user never sees it.
|
||||
- MUST call `{{toolRefs.think}}` before the first action of a turn, and again before any step that is expensive to undo: an edit, a destructive command, a final answer.
|
||||
- Restate what is actually being asked and the constraints given. Split it into ordered sub-problems and solve each explicitly, writing the intermediate result instead of jumping to the conclusion. When a step splits into cases, enumerate them and resolve each.
|
||||
- Then check the work: verify each claim against the constraints, test one boundary or degenerate case, and look specifically for the error you would most plausibly have made. If a check fails, redo that step — NEVER patch the conclusion.
|
||||
- Call again only for materially new state: a tool result that changes the plan, a failed check, a sub-problem you had not opened. NEVER use it to narrate progress or restate what you already recorded.
|
||||
{{/has}}
|
||||
|
||||
# General
|
||||
Use tools whenever they improve correctness, completeness, or grounding.
|
||||
- SHOULD resolve prerequisites before acting.
|
||||
|
||||
@@ -3237,7 +3237,12 @@ async function createAgentSessionScoped(options: CreateAgentSessionOptions): Pro
|
||||
});
|
||||
}
|
||||
}
|
||||
return settingsAwareStreamFn(streamModel, context, streamOptions);
|
||||
const externalThinking =
|
||||
settings.get("externalThinking") && agent.state.tools.some(tool => tool.name === "think");
|
||||
return settingsAwareStreamFn(streamModel, context, {
|
||||
...streamOptions,
|
||||
forceReasoningOff: externalThinking || streamOptions?.forceReasoningOff,
|
||||
});
|
||||
},
|
||||
cursorExecHandlers,
|
||||
getCursorTools: () => (toolSession.xdev ? listXdevTools(toolSession.xdev) : []),
|
||||
@@ -3393,6 +3398,7 @@ async function createAgentSessionScoped(options: CreateAgentSessionOptions): Pro
|
||||
createComputerTool: restrictToolNames
|
||||
? undefined
|
||||
: async () => (await BUILTIN_TOOLS.computer(toolSession)) ?? null,
|
||||
createThinkTool: async () => (await HIDDEN_TOOLS.think(toolSession)) ?? null,
|
||||
createInspectImageTool: restrictToolNames
|
||||
? undefined
|
||||
: async () => (await BUILTIN_TOOLS.inspect_image(toolSession)) ?? null,
|
||||
|
||||
@@ -164,6 +164,8 @@ export interface AgentSessionConfig {
|
||||
createMemoryTools?: () => Promise<AgentTool[]>;
|
||||
/** Creates the built-in `computer` tool for session-scoped runtime enablement (see {@link AgentSession.setComputerToolEnabled}). */
|
||||
createComputerTool?: () => Promise<AgentTool | null>;
|
||||
/** Creates the private `think` scratchpad tool for runtime setting changes. */
|
||||
createThinkTool?: () => Promise<AgentTool | null>;
|
||||
/** Creates the built-in `inspect_image` tool for session-scoped runtime enablement (see {@link AgentSession.setInspectImageMode}). */
|
||||
createInspectImageTool?: () => Promise<AgentTool | null>;
|
||||
/** Model registry for API key resolution and model discovery. */
|
||||
|
||||
@@ -1241,6 +1241,7 @@ export class AgentSession {
|
||||
toolRegistry: config.toolRegistry,
|
||||
createVibeTools: config.createVibeTools,
|
||||
createComputerTool: config.createComputerTool,
|
||||
createThinkTool: config.createThinkTool,
|
||||
createInspectImageTool: config.createInspectImageTool,
|
||||
builtInToolNames: config.builtInToolNames,
|
||||
mcpManagerToolNames: config.mcpManagerToolNames,
|
||||
@@ -4419,6 +4420,11 @@ export class AgentSession {
|
||||
return this.#tools.setComputerToolEnabled(enabled);
|
||||
}
|
||||
|
||||
/** Applies the external-thinking setting to the private scratchpad tool immediately. */
|
||||
setThinkToolEnabled(enabled: boolean): Promise<boolean> {
|
||||
return this.#tools.setThinkToolEnabled(enabled);
|
||||
}
|
||||
|
||||
/**
|
||||
* Session-scoped inspect_image mode (`/vision`). `auto` clears the override
|
||||
* and returns to the persisted `inspect_image.mode` setting; `on`/`off`
|
||||
@@ -5176,6 +5182,18 @@ export class AgentSession {
|
||||
|
||||
// Skip eager preludes when the user has already queued a directive
|
||||
const hasPendingUserDirective = this.#toolChoiceQueue.inspect().includes("user-force");
|
||||
const activeModel = this.agent.state.model;
|
||||
const externalThinkingToolChoice =
|
||||
!options?.synthetic &&
|
||||
!hasPendingUserDirective &&
|
||||
this.settings.get("externalThinking") &&
|
||||
this.getEnabledToolNames().includes("think") &&
|
||||
activeModel &&
|
||||
(activeModel.api === "openai-responses" ||
|
||||
activeModel.api === "azure-openai-responses" ||
|
||||
activeModel.api === "openai-codex-responses")
|
||||
? buildNamedToolChoice("think", activeModel)
|
||||
: undefined;
|
||||
const eagerTodoPrelude =
|
||||
!options?.synthetic && !hasPendingUserDirective ? this.#todo.createEagerTodoPrelude(expandedText) : undefined;
|
||||
const eagerTaskPrelude =
|
||||
@@ -5193,6 +5211,12 @@ export class AgentSession {
|
||||
: undefined;
|
||||
|
||||
const promptAttribution = options?.attribution ?? (options?.synthetic ? "agent" : "user");
|
||||
if (externalThinkingToolChoice) {
|
||||
this.#toolChoiceQueue.pushOnce(externalThinkingToolChoice, {
|
||||
label: "external-thinking",
|
||||
now: true,
|
||||
});
|
||||
}
|
||||
const message = options?.synthetic
|
||||
? { role: "developer" as const, content: userContent, attribution: promptAttribution, timestamp: Date.now() }
|
||||
: { role: "user" as const, content: userContent, attribution: promptAttribution, timestamp: Date.now() };
|
||||
@@ -5223,6 +5247,7 @@ export class AgentSession {
|
||||
// Clean up residual eager-todo directive if the prompt never consumed it
|
||||
// (e.g., compaction aborted, validation failed).
|
||||
this.#toolChoiceQueue.removeByLabel("eager-todo");
|
||||
this.#toolChoiceQueue.removeByLabel("external-thinking");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -68,6 +68,8 @@ interface SessionToolsOptions {
|
||||
toolRegistry?: Map<string, AgentTool>;
|
||||
createVibeTools?: () => AgentTool[];
|
||||
createComputerTool?: () => Promise<AgentTool | null>;
|
||||
/** Creates the private `think` scratchpad tool for runtime setting changes. */
|
||||
createThinkTool?: () => Promise<AgentTool | null>;
|
||||
/** Creates the built-in `inspect_image` tool for session-scoped runtime enablement (see {@link SessionTools.setInspectImageMode}). */
|
||||
createInspectImageTool?: () => Promise<AgentTool | null>;
|
||||
builtInToolNames?: Iterable<string>;
|
||||
@@ -184,6 +186,7 @@ export class SessionTools {
|
||||
#toolRegistry: Map<string, AgentTool>;
|
||||
#createVibeTools: (() => AgentTool[]) | undefined;
|
||||
#createComputerTool: SessionToolsOptions["createComputerTool"];
|
||||
#createThinkTool: SessionToolsOptions["createThinkTool"];
|
||||
#createInspectImageTool: SessionToolsOptions["createInspectImageTool"];
|
||||
#installedVibeToolNames = new Set<string>();
|
||||
#builtInToolNames: Set<string>;
|
||||
@@ -241,6 +244,7 @@ export class SessionTools {
|
||||
this.#toolRegistry = options.toolRegistry ?? new Map();
|
||||
this.#createVibeTools = options.createVibeTools;
|
||||
this.#createComputerTool = options.createComputerTool;
|
||||
this.#createThinkTool = options.createThinkTool;
|
||||
this.#createInspectImageTool = options.createInspectImageTool;
|
||||
this.#builtInToolNames = new Set(options.builtInToolNames ?? []);
|
||||
this.#mcpManagerToolNames = new Set(options.mcpManagerToolNames ?? []);
|
||||
@@ -1158,6 +1162,37 @@ export class SessionTools {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Session-scoped enable/disable for the private `think` scratchpad tool.
|
||||
*
|
||||
* Enabling constructs the tool once and refreshes the model's tool contract;
|
||||
* disabling removes it from the active set while preserving its registry entry.
|
||||
*
|
||||
* @returns false when enabling was requested but this session cannot build the tool.
|
||||
*/
|
||||
setThinkToolEnabled(enabled: boolean): Promise<boolean> {
|
||||
return this.runToolRegistryMutation(async () => {
|
||||
const active = this.getEnabledToolNames();
|
||||
if (!enabled) {
|
||||
if (active.includes("think")) {
|
||||
await this.#applyActiveToolsByName(active.filter(name => name !== "think"));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (!this.#toolRegistry.has("think")) {
|
||||
const tool = await this.#createThinkTool?.();
|
||||
if (tool?.name !== "think") return false;
|
||||
const wrapped = this.#wrapRuntimeTool(tool);
|
||||
this.#toolRegistry.set(wrapped.name, wrapped);
|
||||
this.#builtInToolNames.add(wrapped.name);
|
||||
}
|
||||
if (!active.includes("think")) {
|
||||
await this.#applyActiveToolsByName([...active, "think"]);
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
/** Current effective inspect_image state for `/vision status`. */
|
||||
inspectImageState(): { mode: InspectImageMode; active: boolean; model: string | undefined } {
|
||||
const model = this.#host.model();
|
||||
|
||||
@@ -32,8 +32,7 @@ export const BUILTIN_TOOL_NAMES = [
|
||||
|
||||
export type BuiltinToolName = (typeof BUILTIN_TOOL_NAMES)[number];
|
||||
|
||||
/** Hidden built-ins: constructible and `--tools`-addressable, but never part of the default active set. */
|
||||
export const HIDDEN_TOOL_NAMES = ["yield", "goal"] as const;
|
||||
export const HIDDEN_TOOL_NAMES = ["yield", "goal", "think"] as const;
|
||||
|
||||
export type HiddenToolName = (typeof HIDDEN_TOOL_NAMES)[number];
|
||||
|
||||
|
||||
@@ -62,6 +62,7 @@ import { wrapToolWithMetaNotice } from "./output-meta";
|
||||
import { ReadTool } from "./read";
|
||||
import type { PlanProposalHandler } from "./resolve";
|
||||
import { SecurityScanTool } from "./security-scan";
|
||||
import { ThinkTool } from "./think";
|
||||
import { type TodoPhase, TodoTool } from "./todo";
|
||||
import { WriteTool } from "./write";
|
||||
import { isMountableUnderXdev, type XdevState } from "./xdev";
|
||||
@@ -102,6 +103,7 @@ export * from "./report-tool-issue";
|
||||
export * from "./resolve";
|
||||
export * from "./review";
|
||||
export * from "./security-scan";
|
||||
export * from "./think";
|
||||
export * from "./todo";
|
||||
export * from "./tts";
|
||||
export * from "./vibe";
|
||||
@@ -444,6 +446,7 @@ export const BUILTIN_TOOLS: Record<BuiltinToolName, ToolFactory> = {
|
||||
};
|
||||
|
||||
export const HIDDEN_TOOLS: Record<HiddenToolName, ToolFactory> = {
|
||||
think: () => new ThinkTool(),
|
||||
yield: s => new YieldTool(s),
|
||||
goal: s => new GoalTool(s),
|
||||
};
|
||||
@@ -562,6 +565,9 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
if (session.settings.get("memory.backend") === "mnemopi" && !requestedTools.includes("memory_edit")) {
|
||||
requestedTools.push("memory_edit");
|
||||
}
|
||||
if (session.settings.get("externalThinking") && !requestedTools.includes("think")) {
|
||||
requestedTools.push("think");
|
||||
}
|
||||
// Auto-learn tools are gated by `autolearn.enabled` but, like the memory
|
||||
// tools above, must also be force-included into an explicit requestedTools
|
||||
// list so a restricted top-level session whose controller/guidance is
|
||||
@@ -604,6 +610,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
if (name === "inspect_image") return isInspectImageToolActive(session);
|
||||
if (name === "web_search") return session.settings.get("web_search.enabled");
|
||||
if (name === "security_scan") return session.settings.get("security.enabled");
|
||||
if (name === "think") return session.settings.get("externalThinking");
|
||||
if (name === "ask") return session.settings.get("ask.enabled");
|
||||
if (name === "browser") return session.settings.get("browser.enabled");
|
||||
if (name === "computer") return session.settings.get("computer.enabled");
|
||||
@@ -650,6 +657,7 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
...Object.entries(BUILTIN_TOOLS)
|
||||
.filter(([name]) => isToolAllowed(name))
|
||||
.map(([name, factory]) => [name, factory] as const),
|
||||
...(session.settings.get("externalThinking") ? ([["think", HIDDEN_TOOLS.think]] as const) : []),
|
||||
...(includeYield ? ([["yield", HIDDEN_TOOLS.yield]] as const) : []),
|
||||
...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []),
|
||||
];
|
||||
|
||||
@@ -27,6 +27,7 @@ import { inspectImageToolRenderer } from "./inspect-image-renderer";
|
||||
import { recallToolRenderer, reflectToolRenderer, retainToolRenderer } from "./memory-render";
|
||||
import { readToolRenderer } from "./read";
|
||||
import { resolveRenderer } from "./resolve";
|
||||
import { thinkToolRenderer } from "./think";
|
||||
import { todoToolRenderer } from "./todo";
|
||||
import { createVibeToolRenderer } from "./vibe";
|
||||
import { writeToolRenderer } from "./write";
|
||||
@@ -115,6 +116,7 @@ export const toolRenderers: Record<string, ToolRenderer> = {
|
||||
get task(): ToolRenderer {
|
||||
return taskToolRenderer as ToolRenderer;
|
||||
},
|
||||
think: thinkToolRenderer as ToolRenderer,
|
||||
todo: todoToolRenderer as ToolRenderer,
|
||||
github: githubToolRenderer as ToolRenderer,
|
||||
goal: goalToolRenderer as ToolRenderer,
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
import { type } from "@oh-my-pi/omptype";
|
||||
import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||
import { type Component, Markdown } from "@oh-my-pi/pi-tui";
|
||||
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
|
||||
import { getMarkdownTheme, type Theme } from "../modes/theme/theme";
|
||||
|
||||
const thinkSchema = type({
|
||||
thoughts: type("string").describe("private scratchpad reasoning to retain before the next response"),
|
||||
"+": "reject",
|
||||
}).describe("record private intermediate reasoning before answering");
|
||||
|
||||
type ThinkParams = typeof thinkSchema.infer;
|
||||
|
||||
export type ThinkRenderArgs = {
|
||||
thoughts?: string;
|
||||
};
|
||||
|
||||
export const thinkToolRenderer = {
|
||||
inline: true,
|
||||
renderCall(args: ThinkRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component {
|
||||
const thoughts =
|
||||
typeof args === "object" && args !== null && "thoughts" in args && typeof args.thoughts === "string"
|
||||
? args.thoughts
|
||||
: "";
|
||||
return new Markdown(thoughts, 1, 0, getMarkdownTheme(), {
|
||||
color: (text: string) => uiTheme.fg("thinkingText", text),
|
||||
italic: true,
|
||||
});
|
||||
},
|
||||
renderResult(): Component {
|
||||
return undefined as unknown as Component;
|
||||
},
|
||||
};
|
||||
|
||||
interface ThinkToolDetails {
|
||||
recorded: true;
|
||||
}
|
||||
|
||||
/** Records private intermediate reasoning while native GPT reasoning is disabled. */
|
||||
export class ThinkTool implements AgentTool<typeof thinkSchema, ThinkToolDetails> {
|
||||
readonly name = "think";
|
||||
readonly approval = "read" as const;
|
||||
readonly label = "Think";
|
||||
readonly summary = "Record private intermediate reasoning before answering";
|
||||
readonly description =
|
||||
"Use this private scratchpad to plan, derive, or check work before answering. Record only materially new reasoning. The user does not see this tool activity.";
|
||||
readonly parameters = thinkSchema;
|
||||
readonly strict = true;
|
||||
readonly intent = "omit" as const;
|
||||
|
||||
async execute(_toolCallId: string, _params: ThinkParams): Promise<AgentToolResult<ThinkToolDetails>> {
|
||||
return {
|
||||
content: [
|
||||
{
|
||||
type: "text",
|
||||
text: "------",
|
||||
},
|
||||
],
|
||||
details: { recorded: true },
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -147,6 +147,153 @@ describe("createAgentSession defaultInactive tool activation", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("activates the private think tool when external thinking is enabled at runtime", async () => {
|
||||
const tempDir = makeTempDir();
|
||||
const settings = Settings.isolated();
|
||||
const { session } = await createAgentSession({
|
||||
...baseOptions(tempDir),
|
||||
settings,
|
||||
});
|
||||
|
||||
try {
|
||||
expect(session.getToolByName("think")).toBeUndefined();
|
||||
expect(session.getActiveToolNames()).not.toContain("think");
|
||||
|
||||
settings.set("externalThinking", true);
|
||||
await session.setThinkToolEnabled(true);
|
||||
|
||||
expect(session.getToolByName("think")).toBeDefined();
|
||||
expect(session.getActiveToolNames()).toContain("think");
|
||||
expect(session.getXdevToolEntries().map(entry => entry.name)).not.toContain("think");
|
||||
|
||||
settings.set("externalThinking", false);
|
||||
await session.setThinkToolEnabled(false);
|
||||
expect(session.getActiveToolNames()).not.toContain("think");
|
||||
} finally {
|
||||
await session.dispose();
|
||||
}
|
||||
});
|
||||
|
||||
it("activates the private think tool at startup when external thinking is configured", async () => {
|
||||
const tempDir = makeTempDir();
|
||||
const settings = Settings.isolated({ externalThinking: true });
|
||||
const { session } = await createAgentSession({
|
||||
...baseOptions(tempDir),
|
||||
settings,
|
||||
});
|
||||
|
||||
try {
|
||||
expect(session.getToolByName("think")).toBeDefined();
|
||||
expect(session.getActiveToolNames()).toContain("think");
|
||||
expect(session.getXdevToolEntries().map(entry => entry.name)).not.toContain("think");
|
||||
} finally {
|
||||
await session.dispose();
|
||||
}
|
||||
});
|
||||
|
||||
it("forces think and sends reasoning effort off for a Responses turn", async () => {
|
||||
const tempDir = makeTempDir();
|
||||
const settings = Settings.isolated({ externalThinking: true });
|
||||
const requestTexts: string[] = [];
|
||||
const sse = (events: unknown[]): Response =>
|
||||
new Response(events.map(event => `data: ${JSON.stringify(event)}\n\n`).join(""), {
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
const completed = (id: string) => ({
|
||||
type: "response.completed",
|
||||
response: {
|
||||
id,
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 1,
|
||||
output_tokens: 1,
|
||||
total_tokens: 2,
|
||||
input_tokens_details: { cached_tokens: 0 },
|
||||
},
|
||||
},
|
||||
});
|
||||
const server = Bun.serve({
|
||||
port: 0,
|
||||
fetch: async request => {
|
||||
requestTexts.push(await request.text());
|
||||
if (requestTexts.length === 1) {
|
||||
const argumentsJson = JSON.stringify({ thoughts: "Checked the request before answering." });
|
||||
return sse([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "function_call",
|
||||
id: "fc_think",
|
||||
call_id: "call_think",
|
||||
name: "think",
|
||||
arguments: "",
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.function_call_arguments.done",
|
||||
output_index: 0,
|
||||
item_id: "fc_think",
|
||||
arguments: argumentsJson,
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "function_call",
|
||||
id: "fc_think",
|
||||
call_id: "call_think",
|
||||
name: "think",
|
||||
arguments: argumentsJson,
|
||||
},
|
||||
},
|
||||
completed("resp_think"),
|
||||
]);
|
||||
}
|
||||
return sse([
|
||||
{ type: "response.output_text.delta", output_index: 0, delta: "Done." },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_done",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "Done." }],
|
||||
},
|
||||
},
|
||||
completed("resp_done"),
|
||||
]);
|
||||
},
|
||||
});
|
||||
const model = getBundledModel("openai", "gpt-5");
|
||||
if (!model) throw new Error("Expected gpt-5 model to exist");
|
||||
const { session } = await createAgentSession({
|
||||
...baseOptions(tempDir),
|
||||
settings,
|
||||
model: { ...model, baseUrl: `${server.url}v1` },
|
||||
getApiKey: () => "test-key",
|
||||
});
|
||||
expect(session.getActiveToolNames()).toContain("think");
|
||||
|
||||
try {
|
||||
await session.prompt("Use the scratchpad before answering.");
|
||||
const firstRequest = requestTexts.at(0);
|
||||
if (!firstRequest) throw new Error("Expected the initial provider request.");
|
||||
expect(requestTexts).toHaveLength(2);
|
||||
expect(JSON.parse(firstRequest)).toEqual(
|
||||
expect.objectContaining({
|
||||
reasoning: { effort: "off" },
|
||||
tool_choice: expect.objectContaining({ name: "think" }),
|
||||
}),
|
||||
);
|
||||
} finally {
|
||||
await session.dispose();
|
||||
server.stop(true);
|
||||
}
|
||||
});
|
||||
|
||||
it("publishes tools from lazy session startup before the input lifecycle completes", async () => {
|
||||
const tempDir = makeTempDir();
|
||||
const startupGate = Promise.withResolvers<void>();
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
import { beforeAll, describe, expect, it } from "bun:test";
|
||||
import { getThemeByName, initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
|
||||
import { thinkToolRenderer } from "../../src/tools/think";
|
||||
|
||||
beforeAll(async () => {
|
||||
await initTheme();
|
||||
});
|
||||
|
||||
describe("thinkToolRenderer", () => {
|
||||
it("renders thoughts with thinkingText color and italic style", async () => {
|
||||
const theme = await getThemeByName("dark");
|
||||
expect(theme).toBeDefined();
|
||||
const uiTheme = theme!;
|
||||
|
||||
const callComponent = thinkToolRenderer.renderCall(
|
||||
{ thoughts: "Analyzing the solution step by step." },
|
||||
{ expanded: true, isPartial: false },
|
||||
uiTheme,
|
||||
);
|
||||
|
||||
expect(callComponent).toBeDefined();
|
||||
const lines = callComponent.render(100);
|
||||
const fullText = lines.join("\n");
|
||||
|
||||
expect(fullText).toContain("Analyzing the solution step by step.");
|
||||
expect(fullText).toContain(uiTheme.fg("thinkingText", "Analyzing the solution step by step."));
|
||||
});
|
||||
|
||||
it("has inline set to true", () => {
|
||||
expect(thinkToolRenderer.inline).toBe(true);
|
||||
});
|
||||
|
||||
it("returns undefined for renderResult", () => {
|
||||
expect(thinkToolRenderer.renderResult()).toBeUndefined();
|
||||
});
|
||||
|
||||
it("handles empty or missing thoughts gracefully", async () => {
|
||||
const theme = await getThemeByName("dark");
|
||||
const uiTheme = theme!;
|
||||
|
||||
const emptyCall = thinkToolRenderer.renderCall({}, { expanded: true, isPartial: false }, uiTheme);
|
||||
expect(emptyCall).toBeDefined();
|
||||
expect(emptyCall.render(100)).toEqual([]);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user