diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index ff1e54094..4a89ed9d7 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -8,6 +8,7 @@ - Fixed Cursor reads with inline OMP range selectors reporting the returned slice length as the source file's `totalLines`, which made sequential reads of an unchanged file appear inconsistent ([#7590](https://github.com/can1357/oh-my-pi/issues/7590)). - Made model-scoped usage health ignore Codex accounts that cannot use the requested plan-gated model while retaining conservative unknown-state handling and independent usage-window resets. - Fixed OpenAI Codex usage telemetry blocking explicitly allowed ChatGPT Team credentials when a weekly `used_percent` rounded to 100, which could route multi-account sessions to an actually exhausted sibling instead ([#7617](https://github.com/can1357/oh-my-pi/issues/7617)). +- Fixed OpenAI Codex GPT-5.x requests sending optional `reasoning.summary`, `reasoning.context`, and `text.verbosity` controls by default, reducing Codex `server_error` disconnects from unsupported request shapes. ([#4949](https://github.com/can1357/oh-my-pi/issues/4949)) ## [17.2.7] - 2026-08-03 diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 8c27fe700..9e320b70c 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -128,7 +128,7 @@ import { redactSensitiveInObject, transformMessages } from "./transform-messages export interface OpenAICodexResponsesOptions extends StreamOptions { reasoning?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; reasoningSummary?: "auto" | "concise" | "detailed" | null; - /** `reasoning.context` replay scope; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ + /** Explicit `reasoning.context` replay scope. Omitted by default so Codex applies its native request policy. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; codexMode?: boolean; @@ -1530,7 +1530,7 @@ export async function buildTransformedCodexRequestBody( } const codexOptions: CodexRequestOptions = { reasoningEffort: options?.reasoning, - reasoningSummary: options?.reasoningSummary === undefined ? "auto" : options.reasoningSummary, + reasoningSummary: options?.reasoningSummary, reasoningContext: options?.reasoningContext, textVerbosity: options?.textVerbosity, include: options?.include, diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index b0b157e9c..13a2cc7f9 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -33,7 +33,7 @@ export interface CodexRequestOptions { /** User-facing effort; maps 1:1 onto the wire tier of the same name. */ reasoningEffort?: CodexCallerEffort | "none"; reasoningSummary?: ReasoningConfig["summary"] | null; - /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. Gated to gpt-5.4+ Codex models (older ids reject it, so it is suppressed and `context` omitted). Note that under Responses Lite (`responsesLite`), the server strictly requires `reasoning.context` to be `all_turns`, which overrides this option and forces `all_turns`. */ + /** Explicit `reasoning.context` override. Omitted by default; Responses Lite forces `all_turns` as required by that transport. */ reasoningContext?: CodexReasoningContext; textVerbosity?: "low" | "medium" | "high"; include?: string[]; @@ -145,13 +145,12 @@ function getReasoningConfig( const config: ReasoningConfig = { effort: effort === "none" ? "none" : mapCodexWireEffort(model, effort), }; - // `reasoning.summary` is accepted only from gpt-5.4 onward; earlier Codex ids - // (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with - // "Unsupported parameter: 'reasoning.summary' is not supported with this model". - // Mirrors the all_turns gate: an explicit summary is suppressed on unsupported - // ids, letting the server skip the human-readable summary stream. - if (options.reasoningSummary !== null && supportsCodexReasoningSummary(model.id)) { - config.summary = options.reasoningSummary ?? "detailed"; + if ( + options.reasoningSummary !== undefined && + options.reasoningSummary !== null && + supportsCodexReasoningSummary(model.id) + ) { + config.summary = options.reasoningSummary; } return config; } @@ -444,21 +443,14 @@ export async function transformRequestBody( ...body.reasoning, ...reasoningConfig, }; - // Default reasoning replay to `all_turns`, mirroring codex-rs; an - // explicit `reasoningContext` overrides the default. The `all_turns` - // value is only accepted from gpt-5.4 onward — earlier Codex ids - // (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with - // "Unsupported value: 'all_turns' is not supported with this model". - // For those, drop `context` so the server applies its `current_turn` - // default. The version gate is authoritative: even an explicit - // `all_turns` override is suppressed on unsupported models, while - // `current_turn`/`auto` (universally supported) always pass through. - // Note: Responses Lite forces `all_turns` to satisfy the transport's server invariant. - const context = responsesLite ? "all_turns" : (options.reasoningContext ?? "all_turns"); - if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) { - delete body.reasoning.context; - } else { - body.reasoning.context = context; + // Responses Lite requires `all_turns`; the full transport leaves context to the server unless explicitly set. + const context = responsesLite ? "all_turns" : options.reasoningContext; + if (context !== undefined) { + if (context === "all_turns" && !supportsAllTurnsReasoningContext(model.id)) { + delete body.reasoning.context; + } else { + body.reasoning.context = context; + } } } else { delete body.reasoning; @@ -481,10 +473,12 @@ export async function transformRequestBody( delete body.stream_options; } - body.text = { - ...body.text, - verbosity: options.textVerbosity || "medium", - }; + if (options.textVerbosity !== undefined) { + body.text = { + ...body.text, + verbosity: options.textVerbosity, + }; + } const include = Array.isArray(options.include) ? [...options.include] : []; include.push("reasoning.encrypted_content"); diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index d8c3e1a28..f89dfd0e3 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1705,7 +1705,7 @@ function mapOptionsForApi( serviceTier: options?.serviceTier, preferWebsockets: options?.preferWebsockets, codexCompaction: options?.codexCompaction, - reasoningSummary: options?.hideThinkingSummary ? null : "detailed", + reasoningSummary: options?.hideThinkingSummary ? null : undefined, textVerbosity: options?.textVerbosity, }); diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index 3f4ff7a74..1db6daf4b 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -128,71 +128,58 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe }) as FetchImpl; } -describe("openai-codex reasoning.context", () => { - it("defaults to all_turns on gpt-5.4+ models and forwards explicit overrides", async () => { - const model = createCodexModel("gpt-5.4"); +describe("openai-codex optional response controls", () => { + it("omits optional controls on full requests and forwards explicit controls", async () => { + const model = createCodexModel("gpt-5.5"); const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning?.context).toBe("all_turns"); + expect(defaulted.reasoning).toEqual({ effort: "medium" }); + expect("summary" in (defaulted.reasoning ?? {})).toBe(false); + expect("context" in (defaulted.reasoning ?? {})).toBe(false); + expect("text" in defaulted).toBe(false); + expect("stream_options" in defaulted).toBe(false); const explicit = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", - reasoningContext: "current_turn", + reasoningSummary: "concise", + reasoningContext: "all_turns", + textVerbosity: "low", }); - expect(explicit.reasoning?.context).toBe("current_turn"); + expect(explicit.reasoning).toEqual({ + effort: "medium", + summary: "concise", + context: "all_turns", + }); + expect(explicit.text).toEqual({ verbosity: "low" }); + expect(explicit.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); }); - it("keeps the all_turns default for the lite transport on supported models", async () => { + it("omits reasoning.summary when explicitly suppressed", async () => { const model = createCodexModel("gpt-5.5"); - - const lite = await transformRequestBody({ model: model.id }, model, { + const suppressed = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", - responsesLite: true, + reasoningSummary: null, }); - expect(lite.reasoning?.context).toBe("all_turns"); - - const overridden = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: "medium", - responsesLite: true, - reasoningContext: "auto", - }); - expect(overridden.reasoning?.context).toBe("all_turns"); + expect(suppressed.reasoning).toEqual({ effort: "medium" }); + expect("summary" in (suppressed.reasoning ?? {})).toBe(false); + expect("stream_options" in suppressed).toBe(false); }); - it("enforces reasoning.context to be all_turns for the lite transport even when effort is unset or none", async () => { + it("forces reasoning.context to all_turns for Responses Lite", async () => { const model = createCodexModel("gpt-5.5"); - // Case 1: reasoningEffort is undefined (missing effort) const missingEffort = await transformRequestBody({ model: model.id }, model, { responsesLite: true, }); - expect(missingEffort.reasoning?.context).toBe("all_turns"); - expect(missingEffort.reasoning?.effort).toBeUndefined(); + expect(missingEffort.reasoning).toEqual({ context: "all_turns" }); - // Case 2: reasoningEffort is explicitly "none" (effort set to off) const noneEffort = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "none", responsesLite: true, - }); - expect(noneEffort.reasoning?.context).toBe("all_turns"); - expect(noneEffort.reasoning?.effort).toBe("none"); - - // Case 3: Conflicting explicit reasoningContext with missing effort under Lite - const conflictingUnsetEffort = await transformRequestBody({ model: model.id }, model, { - responsesLite: true, reasoningContext: "current_turn", }); - expect(conflictingUnsetEffort.reasoning?.context).toBe("all_turns"); + expect(noneEffort.reasoning).toEqual({ effort: "none", context: "all_turns" }); - // Case 4: Conflicting explicit reasoningContext with "none" effort under Lite - const conflictingNoneEffort = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: "none", - responsesLite: true, - reasoningContext: "current_turn", - }); - expect(conflictingNoneEffort.reasoning?.context).toBe("all_turns"); - - // Case 5: responsesLite is false and reasoningEffort is undefined (regular request with no effort) const plainRequest = await transformRequestBody({ model: model.id }, model, { responsesLite: false, }); @@ -202,73 +189,35 @@ describe("openai-codex reasoning.context", () => { // gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `all_turns` // ("Unsupported value: 'all_turns' is not supported with this model"). it.each(["gpt-5.1-codex", "gpt-5.3-codex", "gpt-5.3-codex-spark"])( - "omits the all_turns default for pre-5.4 model %s", + "omits unsupported all_turns context for pre-5.4 model %s", async modelId => { const model = createCodexModel(modelId); + const forced = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: "medium", + reasoningContext: "all_turns", + }); + expect(forced.reasoning).toEqual({ effort: "medium" }); - const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning).toBeDefined(); - expect(defaulted.reasoning?.context).toBeUndefined(); - expect("context" in (defaulted.reasoning ?? {})).toBe(false); - - // A supported override (current_turn/auto) is still honored. const overridden = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", reasoningContext: "current_turn", }); - expect(overridden.reasoning?.context).toBe("current_turn"); + expect(overridden.reasoning).toEqual({ effort: "medium", context: "current_turn" }); }, ); - it("suppresses an explicit all_turns override on a pre-5.4 model", async () => { - const model = createCodexModel("gpt-5.3-codex-spark"); - - const forced = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: "medium", - reasoningContext: "all_turns", - }); - expect(forced.reasoning).toBeDefined(); - expect(forced.reasoning?.context).toBeUndefined(); - }); -}); - -describe("openai-codex reasoning.summary", () => { - it("sends summary on gpt-5.4+ models and honors explicit levels", async () => { - const model = createCodexModel("gpt-5.4"); - - const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning?.summary).toBe("detailed"); - - const explicit = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: "medium", - reasoningSummary: "concise", - }); - expect(explicit.reasoning?.summary).toBe("concise"); - - const suppressed = await transformRequestBody({ model: model.id }, model, { - reasoningEffort: "medium", - reasoningSummary: null, - }); - expect("summary" in (suppressed.reasoning ?? {})).toBe(false); - }); - // gpt-5.1-codex / gpt-5.3-codex / gpt-5.3-codex-spark reject `reasoning.summary` // ("Unsupported parameter: 'reasoning.summary' is not supported with this model"). it.each(["gpt-5.1-codex", "gpt-5.3-codex", "gpt-5.3-codex-spark"])( "omits reasoning.summary for pre-5.4 model %s", async modelId => { const model = createCodexModel(modelId); - - const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning).toBeDefined(); - expect("summary" in (defaulted.reasoning ?? {})).toBe(false); - - // Even an explicit summary level is suppressed on unsupported ids. const forced = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium", reasoningSummary: "detailed", }); - expect("summary" in (forced.reasoning ?? {})).toBe(false); + expect(forced.reasoning).toEqual({ effort: "medium" }); + expect("stream_options" in forced).toBe(false); }, ); }); @@ -816,7 +765,11 @@ describe("openai-codex concurrent reasoning summaries", () => { it("sends stream_options only when a summary is requested and supported", async () => { const terra = createCodexModel("gpt-5.6-terra"); - const withSummary = await transformRequestBody({ model: terra.id }, terra, { reasoningEffort: "medium" }); + const withSummary = await transformRequestBody( + { model: terra.id }, + terra, + { reasoningEffort: "medium", reasoningSummary: "detailed" }, + ); expect(withSummary.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); expect(withSummary.reasoning?.summary).toBe("detailed"); @@ -830,7 +783,11 @@ describe("openai-codex concurrent reasoning summaries", () => { expect(noReasoning.stream_options).toBeUndefined(); const legacy = createCodexModel("gpt-5.1-codex"); - const unsupported = await transformRequestBody({ model: legacy.id }, legacy, { reasoningEffort: "medium" }); + const unsupported = await transformRequestBody( + { model: legacy.id }, + legacy, + { reasoningEffort: "medium", reasoningSummary: "detailed" }, + ); expect(unsupported.stream_options).toBeUndefined(); }); @@ -889,6 +846,7 @@ describe("openai-codex concurrent reasoning summaries", () => { apiKey: createCodexTestToken(), fetch: fetchMock, reasoning: "medium", + reasoningSummary: "detailed", }); const thinkingDeltas: string[] = []; for await (const event of stream) { @@ -1060,6 +1018,7 @@ describe("openai-codex concurrent reasoning summaries", () => { apiKey: createCodexTestToken(), fetch: fetchMock, reasoning: "medium", + reasoningSummary: "detailed", }); const thinkingDeltas: string[] = []; for await (const event of stream) { @@ -1215,6 +1174,7 @@ describe("openai-codex concurrent reasoning summaries", () => { apiKey: createCodexTestToken(), fetch: fetchMock, reasoning: "medium", + reasoningSummary: "detailed", }); const deltasByBlock = new Map(); for await (const event of stream) { diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 274cc02f3..a650a3e43 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -501,6 +501,32 @@ describe("openai-codex streaming", () => { expect(capturedText).toEqual({ verbosity: "low" }); }); + it("omits optional response controls from default SimpleStreamOptions", async () => { + const tempDir = TempDir.createSync("@pi-codex-stream-"); + setAgentDir(tempDir.path()); + const token = createCodexTestToken(); + const context = createCodexTestContext(); + const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false }; + let capturedBody: Record | undefined; + const fetchMock: FetchImpl = async (_input, init) => { + capturedBody = JSON.parse(decodeCodexRequestBody(init?.body)) as Record; + return new Response(createCompletedCodexSse("Hello"), { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); + }; + + const result = await streamSimple(model, context, { + apiKey: token, + fetch: fetchMock, + reasoning: "medium", + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(capturedBody?.reasoning).toEqual({ effort: "medium" }); + expect(capturedBody?.text).toBeUndefined(); + }); + async function runCodexSseEvents(events: unknown[]) { const token = createCodexTestToken(); const context = createCodexTestContext(); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 45ea58480..8fe24bd59 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -14,6 +14,7 @@ ### Fixed - Retried concurrent-request caps with a short backoff without deleting valid Copilot credentials or rotating through sibling accounts. +- Fixed the default `textVerbosity` setting being forwarded to OpenAI Codex requests unless the user explicitly configures it, preserving Codex's native response-control defaults. ([#4949](https://github.com/can1357/oh-my-pi/issues/4949)) - Reduced streaming CPU usage by coalescing the cumulative `message_update` deltas of a turn at the event-controller dispatch boundary: at most one streaming-state rebuild runs per ~33ms window instead of one per token, cutting the per-token handler work that dominated the CPU profile of streaming sessions (especially at high token rates) while preserving per-delta speech output. Subscriber dispatch is serialized so a rapid stream tail (`message_update` → `message_end` → `agent_end`) cannot overtake the coalesced flush. ([#7443](https://github.com/can1357/oh-my-pi/issues/7443)) - Fixed translated MCP importers (Claude Code, Cursor, Gemini CLI, Windsurf, VS Code) silently dropping a server's `enabled: false` flag, so a server disabled at the source config stayed mounted; the flag is now propagated and honored like Codex, OpenCode, and native `mcp.json`. These importers now also load project entries before same-named user entries (matching native/Codex) so a project `enabled: false` suppresses a same-named user server ([#7652](https://github.com/can1357/oh-my-pi/issues/7652)). - Removed the per-call `model` override from the eval `agent()` helper (all runtimes), completing the earlier task-tool removal (`9f8aa87dbf`). Subagents always use their selected agent's frontmatter model and settings; a legacy `model` argument is silently ignored, so an explicit `model: "default"` can no longer route children onto the parent session model ([#6438](https://github.com/can1357/oh-my-pi/issues/6438)). diff --git a/packages/coding-agent/src/session/settings-stream-fn.ts b/packages/coding-agent/src/session/settings-stream-fn.ts index be58243ed..41dcdc386 100644 --- a/packages/coding-agent/src/session/settings-stream-fn.ts +++ b/packages/coding-agent/src/session/settings-stream-fn.ts @@ -34,9 +34,13 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn = openrouterRoutingPreset && openrouterRoutingPreset !== "default" ? openrouterRoutingPreset : undefined; const antigravityEndpointMode = settings.get("providers.antigravityEndpoint"); const textVerbosity = - model.api === "openai-codex-responses" || model.api === "openai-responses" - ? settings.get("textVerbosity") - : undefined; + model.api === "openai-codex-responses" + ? settings.isConfigured("textVerbosity") + ? settings.get("textVerbosity") + : undefined + : model.api === "openai-responses" + ? settings.get("textVerbosity") + : undefined; const streamFirstEventTimeoutMs = timeoutSecondsToMs(settings.get("providers.streamFirstEventTimeoutSeconds")); const streamIdleTimeoutMs = timeoutSecondsToMs(settings.get("providers.streamIdleTimeoutSeconds")); // Server-side fallback (opt-in): when the user enables it AND the diff --git a/packages/coding-agent/test/settings-stream-fn.test.ts b/packages/coding-agent/test/settings-stream-fn.test.ts index 424889447..44be30735 100644 --- a/packages/coding-agent/test/settings-stream-fn.test.ts +++ b/packages/coding-agent/test/settings-stream-fn.test.ts @@ -81,7 +81,17 @@ describe("createSettingsAwareStreamFn", () => { expect(calls[0]?.options?.hideThinkingSummary).toBe(true); }); - it("applies Responses-family text verbosity from settings while preserving caller overrides", () => { + it("applies Codex text verbosity only when settings or caller options configure it", () => { + const unconfiguredSettings = Settings.isolated({}); + const { fn: unconfiguredBase, calls: unconfiguredCalls } = captureBase(); + const unconfiguredWrapped = createSettingsAwareStreamFn(unconfiguredSettings, unconfiguredBase); + + unconfiguredWrapped(stubCodexModel, stubContext, undefined); + unconfiguredWrapped(stubCodexModel, stubContext, { textVerbosity: "medium" }); + + expect(unconfiguredCalls[0]?.options?.textVerbosity).toBeUndefined(); + expect(unconfiguredCalls[1]?.options?.textVerbosity).toBe("medium"); + const settings = Settings.isolated({ textVerbosity: "low" }); const { fn: base, calls } = captureBase(); const wrapped = createSettingsAwareStreamFn(settings, base);