feat(ai): added disableReasoning option and fixed codex websocket reuse
- Added `disableReasoning` to `SimpleStreamOptions` and OpenAI completions, sending `reasoning: { enabled: false }` for OpenRouter requests to prevent reasoning models from consuming the full output budget on small calls like title generation.
- Fixed `canAppend` to accept `response.completed` as a terminal event, restoring websocket append reuse after codex sessions end.
- Replaced async blob-decoding and `addEventListener` with synchronous `onmessage`/`onopen`/`onerror`/`onclose` handlers and `binaryType = "nodebuffer"` for simpler, reliable message decoding.
- Simplified title generator to discard per-role thinking level and always pass `disableReasoning: true`.
This commit is contained in:
@@ -1,8 +1,10 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `disableReasoning` to stream and OpenAI completion options to force reasoning off for models that support it, sending `reasoning: { enabled: false }` for OpenRouter-compatible requests
|
||||
- Added `thinkingDisplay` option to Anthropic options to control whether adaptive and explicit reasoning is returned as `summarized` or `omitted`
|
||||
- Added Anthropic model compatibility flags `supportsEagerToolInputStreaming` and `supportsLongCacheRetention` for API-capability-specific request behavior
|
||||
|
||||
@@ -24,6 +26,7 @@
|
||||
- Fixed Anthropic stream handling to parse raw SSE envelopes directly, ignore unrelated events, and repair malformed JSON in SSE payloads
|
||||
- Fixed Anthropic streaming to emit an explicit error when the SSE stream ends without a `message_stop` event
|
||||
- Fixed OpenAI Codex websocket continuations to send true `previous_response_id` deltas for `store: false` transcripts, expose request stats, and default text verbosity to `low` unless explicitly overridden.
|
||||
- Fixed OpenAI Codex websocket append reuse after `response.completed` terminal events.
|
||||
|
||||
## [14.5.14] - 2026-05-01
|
||||
### Added
|
||||
|
||||
@@ -1202,7 +1202,7 @@ function handleResponseCompleted(
|
||||
state.lastResponseId = response.id;
|
||||
state.lastResponseItems = stripInputItemIds(structuredCloneJSON(runtime.nativeOutputItems));
|
||||
}
|
||||
state.canAppend = rawEvent.type === "response.done";
|
||||
state.canAppend = rawEvent.type === "response.done" || rawEvent.type === "response.completed";
|
||||
}
|
||||
|
||||
calculateCost(model, output.usage);
|
||||
@@ -1843,12 +1843,10 @@ class CodexWebSocketConnection {
|
||||
await this.#connectPromise;
|
||||
return;
|
||||
}
|
||||
const WebSocketWithHeaders = WebSocket as unknown as {
|
||||
new (url: string, options?: { headers?: Record<string, string> }): WebSocket;
|
||||
};
|
||||
const { promise, resolve, reject } = Promise.withResolvers<void>();
|
||||
this.#connectPromise = promise;
|
||||
const socket = new WebSocketWithHeaders(this.#url, { headers: this.#headers });
|
||||
const socket = new WebSocket(this.#url, { headers: this.#headers });
|
||||
socket.binaryType = "nodebuffer";
|
||||
this.#socket = socket;
|
||||
let settled = false;
|
||||
let timeout: NodeJS.Timeout | undefined;
|
||||
@@ -1878,15 +1876,15 @@ class CodexWebSocketConnection {
|
||||
}
|
||||
}, CODEX_WEBSOCKET_CONNECT_TIMEOUT_MS);
|
||||
|
||||
socket.addEventListener("open", event => {
|
||||
socket.onopen = event => {
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
clearPending();
|
||||
this.#captureHandshakeHeaders(socket, event);
|
||||
resolve();
|
||||
}
|
||||
});
|
||||
socket.addEventListener("error", event => {
|
||||
};
|
||||
socket.onerror = event => {
|
||||
const eventRecord = event as unknown as Record<string, unknown>;
|
||||
const detail =
|
||||
(typeof eventRecord.message === "string" && eventRecord.message) ||
|
||||
@@ -1900,8 +1898,8 @@ class CodexWebSocketConnection {
|
||||
return;
|
||||
}
|
||||
this.#push(error);
|
||||
});
|
||||
socket.addEventListener("close", event => {
|
||||
};
|
||||
socket.onclose = event => {
|
||||
this.#socket = null;
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
@@ -1911,28 +1909,26 @@ class CodexWebSocketConnection {
|
||||
}
|
||||
this.#push(createCodexWebSocketTransportError(`websocket closed (${event.code})`));
|
||||
this.#push(null);
|
||||
});
|
||||
socket.addEventListener("message", event => {
|
||||
void (async () => {
|
||||
try {
|
||||
const text = await decodeCodexWebSocketData(event.data);
|
||||
if (!text) return;
|
||||
const parsed = JSON.parse(text) as Record<string, unknown>;
|
||||
if (parsed.type === "error" && typeof parsed.error === "object" && parsed.error) {
|
||||
const inner = parsed.error as Record<string, unknown>;
|
||||
if (typeof parsed.code !== "string" && typeof inner.code === "string") {
|
||||
parsed.code = inner.code;
|
||||
}
|
||||
if (typeof parsed.message !== "string" && typeof inner.message === "string") {
|
||||
parsed.message = inner.message;
|
||||
}
|
||||
};
|
||||
socket.onmessage = event => {
|
||||
try {
|
||||
const text = typeof event.data === "string" ? event.data : Buffer.from(event.data).toString("utf-8");
|
||||
if (!text) return;
|
||||
const parsed = JSON.parse(text) as Record<string, unknown>;
|
||||
if (parsed.type === "error" && typeof parsed.error === "object" && parsed.error) {
|
||||
const inner = parsed.error as Record<string, unknown>;
|
||||
if (typeof parsed.code !== "string" && typeof inner.code === "string") {
|
||||
parsed.code = inner.code;
|
||||
}
|
||||
if (typeof parsed.message !== "string" && typeof inner.message === "string") {
|
||||
parsed.message = inner.message;
|
||||
}
|
||||
this.#push(parsed);
|
||||
} catch (error) {
|
||||
this.#push(createCodexWebSocketTransportError(String(error)));
|
||||
}
|
||||
})();
|
||||
});
|
||||
this.#push(parsed);
|
||||
} catch (error) {
|
||||
this.#push(createCodexWebSocketTransportError(String(error)));
|
||||
}
|
||||
};
|
||||
|
||||
logger.time("codexWs:awaitTcpHandshake");
|
||||
try {
|
||||
@@ -2280,23 +2276,6 @@ function resolveCodexResponsesUrl(baseUrl: string | undefined): string {
|
||||
return `${normalized}/codex/responses`;
|
||||
}
|
||||
|
||||
async function decodeCodexWebSocketData(data: unknown): Promise<string | null> {
|
||||
if (typeof data === "string") return data;
|
||||
if (data instanceof ArrayBuffer) {
|
||||
return new TextDecoder().decode(new Uint8Array(data));
|
||||
}
|
||||
if (ArrayBuffer.isView(data)) {
|
||||
const view = data;
|
||||
return new TextDecoder().decode(new Uint8Array(view.buffer, view.byteOffset, view.byteLength));
|
||||
}
|
||||
if (data && typeof data === "object" && "arrayBuffer" in data) {
|
||||
const blobLike = data as { arrayBuffer: () => Promise<ArrayBuffer> };
|
||||
const arrayBuffer = await blobLike.arrayBuffer();
|
||||
return new TextDecoder().decode(new Uint8Array(arrayBuffer));
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function getAccountId(accessToken: string): string {
|
||||
const accountId = getCodexAccountId(accessToken);
|
||||
if (!accountId) {
|
||||
|
||||
@@ -9,6 +9,7 @@ import type {
|
||||
ChatCompletionMessageParam,
|
||||
ChatCompletionToolMessageParam,
|
||||
} from "openai/resources/chat/completions";
|
||||
import type { Effort } from "../model-thinking";
|
||||
import { calculateCost } from "../models";
|
||||
import { getEnvApiKey } from "../stream";
|
||||
import {
|
||||
@@ -17,6 +18,7 @@ import {
|
||||
type Message,
|
||||
type MessageAttribution,
|
||||
type Model,
|
||||
type OpenAICompat,
|
||||
type ProviderSessionState,
|
||||
type ServiceTier,
|
||||
type StopReason,
|
||||
@@ -125,13 +127,21 @@ function hasToolHistory(messages: Message[]): boolean {
|
||||
export interface OpenAICompletionsOptions extends StreamOptions {
|
||||
toolChoice?: ToolChoice;
|
||||
reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
/** Force-disable reasoning for OpenRouter-format requests (sends `reasoning: { enabled: false }`). */
|
||||
disableReasoning?: boolean;
|
||||
serviceTier?: ServiceTier;
|
||||
}
|
||||
|
||||
type OpenAICompletionsSamplingParams = OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming & {
|
||||
type OpenAICompletionsParams = OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming & {
|
||||
top_k?: number;
|
||||
min_p?: number;
|
||||
repetition_penalty?: number;
|
||||
thinking?: { type: "enabled" | "disabled" };
|
||||
enable_thinking?: boolean;
|
||||
chat_template_kwargs?: { enable_thinking: boolean };
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
provider?: OpenAICompat["openRouterRouting"];
|
||||
providerOptions?: { gateway?: { only?: string[]; order?: string[] } };
|
||||
};
|
||||
|
||||
type AppliedToolStrictMode = "mixed" | "all_strict" | "none";
|
||||
@@ -824,7 +834,7 @@ function buildParams(
|
||||
options: OpenAICompletionsOptions | undefined,
|
||||
resolvedBaseUrl?: string,
|
||||
toolStrictModeOverride?: ToolStrictModeOverride,
|
||||
): { params: OpenAICompletionsSamplingParams; toolStrictMode: AppliedToolStrictMode } {
|
||||
): { params: OpenAICompletionsParams; toolStrictMode: AppliedToolStrictMode } {
|
||||
const compat = getCompat(model, resolvedBaseUrl);
|
||||
const messages = convertMessages(model, context, compat);
|
||||
maybeAddOpenRouterAnthropicCacheControl(model, messages);
|
||||
@@ -837,7 +847,7 @@ function buildParams(
|
||||
const effectiveMaxTokens = options?.maxTokens ?? (isKimi ? model.maxTokens : undefined);
|
||||
|
||||
const requestModelId = model.provider === "fireworks" ? toFireworksWireModelId(model.id) : model.id;
|
||||
const params: OpenAICompletionsSamplingParams = {
|
||||
const params: OpenAICompletionsParams = {
|
||||
model: requestModelId,
|
||||
messages,
|
||||
stream: true,
|
||||
@@ -845,7 +855,7 @@ function buildParams(
|
||||
let toolStrictMode: AppliedToolStrictMode = "none";
|
||||
|
||||
if (compat.supportsUsageInStreaming !== false) {
|
||||
(params as { stream_options?: { include_usage: boolean } }).stream_options = { include_usage: true };
|
||||
params.stream_options = { include_usage: true };
|
||||
}
|
||||
|
||||
if (compat.supportsStore) {
|
||||
@@ -854,7 +864,7 @@ function buildParams(
|
||||
|
||||
if (effectiveMaxTokens) {
|
||||
if (compat.maxTokensField === "max_tokens") {
|
||||
(params as any).max_tokens = effectiveMaxTokens;
|
||||
params.max_tokens = effectiveMaxTokens;
|
||||
} else {
|
||||
params.max_completion_tokens = effectiveMaxTokens;
|
||||
}
|
||||
@@ -897,27 +907,40 @@ function buildParams(
|
||||
|
||||
if (supportsReasoningParams && compat.thinkingFormat === "zai" && model.reasoning) {
|
||||
// Z.ai uses binary thinking: { type: "enabled" | "disabled" }
|
||||
// Must explicitly disable since z.ai defaults to thinking enabled
|
||||
Reflect.set(params, "thinking", { type: options?.reasoning ? "enabled" : "disabled" });
|
||||
// Must explicitly disable since z.ai defaults to thinking enabled.
|
||||
const enabled = options?.reasoning && !options?.disableReasoning;
|
||||
params.thinking = { type: enabled ? "enabled" : "disabled" };
|
||||
} else if (supportsReasoningParams && compat.thinkingFormat === "qwen" && model.reasoning) {
|
||||
// Qwen uses top-level enable_thinking: boolean
|
||||
Reflect.set(params, "enable_thinking", !!options?.reasoning);
|
||||
params.enable_thinking = !!options?.reasoning && !options?.disableReasoning;
|
||||
} else if (supportsReasoningParams && compat.thinkingFormat === "qwen-chat-template" && model.reasoning) {
|
||||
Reflect.set(params, "chat_template_kwargs", { enable_thinking: !!options?.reasoning });
|
||||
params.chat_template_kwargs = {
|
||||
enable_thinking: !!options?.reasoning && !options?.disableReasoning,
|
||||
};
|
||||
} else if (supportsReasoningParams && compat.thinkingFormat === "openrouter" && model.reasoning) {
|
||||
// OpenRouter normalizes reasoning across providers via a nested reasoning object.
|
||||
// Without an explicit signal, OpenRouter defaults reasoning models to thinking, which
|
||||
// silently consumes the entire output budget on small `max_tokens` requests (e.g.
|
||||
// title generation). Honor `disableReasoning` to opt out cleanly.
|
||||
const openRouterParams = params as typeof params & {
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
};
|
||||
if (options?.disableReasoning) {
|
||||
openRouterParams.reasoning = { enabled: false };
|
||||
} else if (options?.reasoning) {
|
||||
openRouterParams.reasoning = {
|
||||
effort: mapReasoningEffort(options.reasoning, compat.reasoningEffortMap),
|
||||
};
|
||||
}
|
||||
} else if (
|
||||
supportsReasoningParams &&
|
||||
compat.thinkingFormat === "openrouter" &&
|
||||
options?.reasoning &&
|
||||
model.reasoning
|
||||
!options?.disableReasoning &&
|
||||
model.reasoning &&
|
||||
compat.supportsReasoningEffort
|
||||
) {
|
||||
// OpenRouter normalizes reasoning across providers via a nested reasoning object.
|
||||
const openRouterParams = params as typeof params & { reasoning?: { effort?: string } };
|
||||
openRouterParams.reasoning = {
|
||||
effort: mapReasoningEffort(options.reasoning, compat.reasoningEffortMap),
|
||||
};
|
||||
} else if (supportsReasoningParams && options?.reasoning && model.reasoning && compat.supportsReasoningEffort) {
|
||||
// OpenAI-style reasoning_effort
|
||||
Reflect.set(params, "reasoning_effort", mapReasoningEffort(options.reasoning, compat.reasoningEffortMap));
|
||||
params.reasoning_effort = mapReasoningEffort(options.reasoning, compat.reasoningEffortMap) as Effort;
|
||||
}
|
||||
|
||||
if (compat.disableReasoningOnForcedToolChoice && isForcedToolChoice(params.tool_choice)) {
|
||||
@@ -925,13 +948,13 @@ function buildParams(
|
||||
// Kimi 400 with `tool_choice 'specified' is incompatible with thinking
|
||||
// enabled`. Drop reasoning for this turn instead of dropping tool_choice;
|
||||
// the agent still gets the forced tool call, just without thinking.
|
||||
delete (params as { reasoning_effort?: unknown }).reasoning_effort;
|
||||
delete (params as { reasoning?: unknown }).reasoning;
|
||||
delete params.reasoning_effort;
|
||||
delete params.reasoning;
|
||||
}
|
||||
|
||||
// OpenRouter provider routing preferences
|
||||
if (model.baseUrl.includes("openrouter.ai") && compat.openRouterRouting) {
|
||||
Reflect.set(params, "provider", compat.openRouterRouting);
|
||||
params.provider = compat.openRouterRouting;
|
||||
}
|
||||
|
||||
// Vercel AI Gateway provider routing preferences
|
||||
@@ -941,7 +964,7 @@ function buildParams(
|
||||
const gatewayOptions: Record<string, string[]> = {};
|
||||
if (routing.only) gatewayOptions.only = routing.only;
|
||||
if (routing.order) gatewayOptions.order = routing.order;
|
||||
Reflect.set(params, "providerOptions", { gateway: gatewayOptions });
|
||||
params.providerOptions = { gateway: gatewayOptions };
|
||||
}
|
||||
}
|
||||
|
||||
@@ -949,13 +972,6 @@ function buildParams(
|
||||
Object.assign(params, compat.extraBody);
|
||||
}
|
||||
|
||||
return buildParamsResult(params, toolStrictMode);
|
||||
}
|
||||
|
||||
function buildParamsResult(
|
||||
params: OpenAICompletionsSamplingParams,
|
||||
toolStrictMode: AppliedToolStrictMode,
|
||||
): { params: OpenAICompletionsSamplingParams; toolStrictMode: AppliedToolStrictMode } {
|
||||
return { params, toolStrictMode };
|
||||
}
|
||||
|
||||
|
||||
@@ -553,6 +553,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
return castApi<"openai-completions">({
|
||||
...base,
|
||||
reasoning: resolveOpenAiReasoningEffort(model, options),
|
||||
disableReasoning: options?.disableReasoning,
|
||||
toolChoice: mapOpenAiToolChoice(options?.toolChoice),
|
||||
serviceTier: options?.serviceTier,
|
||||
});
|
||||
|
||||
@@ -246,6 +246,15 @@ export interface StreamOptions {
|
||||
// Unified options with reasoning passed to streamSimple() and completeSimple()
|
||||
export interface SimpleStreamOptions extends StreamOptions {
|
||||
reasoning?: Effort;
|
||||
/**
|
||||
* Force-disable reasoning for the request even when the model supports it.
|
||||
* Takes precedence over `reasoning`. Useful for fast utility calls
|
||||
* (e.g. title generation) where the model would otherwise burn the entire
|
||||
* output budget on internal thinking. Currently honored by OpenRouter
|
||||
* (sends `reasoning: { enabled: false }`); other providers already behave
|
||||
* this way when `reasoning` is undefined.
|
||||
*/
|
||||
disableReasoning?: boolean;
|
||||
/** Custom token budgets for thinking levels (token-based providers only) */
|
||||
thinkingBudgets?: ThinkingBudgets;
|
||||
/** Cursor exec handlers for local tool execution */
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -2,7 +2,7 @@
|
||||
* Generate session titles using a smol, fast model.
|
||||
*/
|
||||
import * as path from "node:path";
|
||||
import type { ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
|
||||
import { type Api, completeSimple, type Model } from "@oh-my-pi/pi-ai";
|
||||
import { logger, prompt } from "@oh-my-pi/pi-utils";
|
||||
import type { ModelRegistry } from "../config/model-registry";
|
||||
@@ -17,22 +17,14 @@ const TERMINAL_TITLE_CONTROL_CHARS = /[\u0000-\u001f\u007f-\u009f]/g;
|
||||
|
||||
const MAX_INPUT_CHARS = 2000;
|
||||
|
||||
function getTitleModel(
|
||||
registry: ModelRegistry,
|
||||
settings: Settings,
|
||||
currentModel?: Model<Api>,
|
||||
): { model: Model<Api>; thinkingLevel?: ThinkingLevel } | undefined {
|
||||
function getTitleModel(registry: ModelRegistry, settings: Settings, currentModel?: Model<Api>): Model<Api> | undefined {
|
||||
const availableModels = registry.getAvailable();
|
||||
if (availableModels.length === 0) return undefined;
|
||||
|
||||
const titleModel = resolveRoleSelection(["commit", "smol"], settings, availableModels, registry);
|
||||
if (titleModel) {
|
||||
return { model: titleModel.model, thinkingLevel: titleModel.thinkingLevel };
|
||||
}
|
||||
const titleModel = resolveRoleSelection(["commit", "smol"], settings, availableModels, registry)?.model;
|
||||
if (titleModel) return titleModel;
|
||||
|
||||
if (currentModel) {
|
||||
return { model: currentModel };
|
||||
}
|
||||
if (currentModel) return currentModel;
|
||||
|
||||
return undefined;
|
||||
}
|
||||
@@ -42,7 +34,7 @@ function getTitleModel(
|
||||
*
|
||||
* @param firstMessage The first user message
|
||||
* @param registry Model registry
|
||||
* @param settings Settings used to resolve the smol role, including per-role thinking
|
||||
* @param settings Settings used to resolve the smol role
|
||||
* @param sessionId Optional session id for sticky API key selection
|
||||
*/
|
||||
export async function generateSessionTitle(
|
||||
@@ -52,8 +44,8 @@ export async function generateSessionTitle(
|
||||
sessionId?: string,
|
||||
currentModel?: Model<Api>,
|
||||
): Promise<string | null> {
|
||||
const candidate = getTitleModel(registry, settings, currentModel);
|
||||
if (!candidate) {
|
||||
const model = getTitleModel(registry, settings, currentModel);
|
||||
if (!model) {
|
||||
logger.debug("title-generator: no title model found");
|
||||
return null;
|
||||
}
|
||||
@@ -65,11 +57,11 @@ export async function generateSessionTitle(
|
||||
${truncatedMessage}
|
||||
</user-message>`;
|
||||
|
||||
const apiKey = await registry.getApiKey(candidate.model, sessionId);
|
||||
const apiKey = await registry.getApiKey(model, sessionId);
|
||||
if (!apiKey) {
|
||||
logger.debug("title-generator: no API key for smol model", {
|
||||
provider: candidate.model.provider,
|
||||
id: candidate.model.id,
|
||||
provider: model.provider,
|
||||
id: model.id,
|
||||
});
|
||||
return null;
|
||||
}
|
||||
@@ -78,7 +70,7 @@ ${truncatedMessage}
|
||||
// don't burn the entire output budget on internal thinking and return an empty
|
||||
// string. With reasoning disabled, 30 tokens of output is plenty.
|
||||
const request = {
|
||||
model: `${candidate.model.provider}/${candidate.model.id}`,
|
||||
model: `${model.provider}/${model.id}`,
|
||||
systemPrompt: TITLE_SYSTEM_PROMPT,
|
||||
userMessage,
|
||||
maxTokens: 30,
|
||||
@@ -87,7 +79,7 @@ ${truncatedMessage}
|
||||
|
||||
try {
|
||||
const response = await completeSimple(
|
||||
candidate.model,
|
||||
model,
|
||||
{
|
||||
systemPrompt: request.systemPrompt,
|
||||
messages: [{ role: "user", content: request.userMessage, timestamp: Date.now() }],
|
||||
|
||||
@@ -50,7 +50,7 @@ describe("role thinking helper propagation", () => {
|
||||
expect(completeSimpleMock.mock.calls[0]?.[2]).toMatchObject({ reasoning: Effort.Minimal });
|
||||
});
|
||||
|
||||
it("passes smol-role thinking to title generation", async () => {
|
||||
it("disables reasoning for title generation even when smol role has thinking", async () => {
|
||||
const model = getModelOrThrow("claude-sonnet-4-5");
|
||||
const settings = createSettings({
|
||||
default: `${model.provider}/${model.id}:high`,
|
||||
@@ -67,6 +67,6 @@ describe("role thinking helper propagation", () => {
|
||||
|
||||
const title = await generateSessionTitle("Investigate resolver", registry as never, settings);
|
||||
expect(title).toBe("Investigate resolver");
|
||||
expect(completeSimpleMock.mock.calls[0]?.[2]).toMatchObject({ reasoning: Effort.Low });
|
||||
expect(completeSimpleMock.mock.calls[0]?.[2]).toMatchObject({ disableReasoning: true });
|
||||
});
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user