diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index 32d657a9b..d8817b85d 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -109,6 +109,22 @@ export function resolveCodexResponsesLite( return model.useResponsesLite === true; } +/** + * Whether to request `stream_options.reasoning_summary_delivery = + * "sequential_cutoff"` (codex-rs `concurrent_reasoning_summaries`), enabled by + * `PI_CODEX_CONCURRENT_SUMMARIES=1`. + * + * Off by default because the mode cancels summary sections still in flight when + * the reasoning item closes: measured over 12 interleaved turns it halved + * visible thinking (0.83 vs 1.67 summary parts, 37 vs 69 chars per turn) and + * produced no summary at all on 3 of 12 turns. codex-rs ships it disabled too + * (`Stage::UnderDevelopment`, `default_enabled: false`). + */ +function concurrentSummariesEnabled(): boolean { + const env = $env.PI_CODEX_CONCURRENT_SUMMARIES?.trim().toLowerCase(); + return env === "1" || env === "true"; +} + /** * Clamp a user-facing effort to the model's ladder, then remap to the wire * tier. User efforts map 1:1 onto wire tiers; the effort map only covers @@ -145,12 +161,14 @@ function getReasoningConfig( const config: ReasoningConfig = { effort: effort === "none" ? "none" : mapCodexWireEffort(model, effort), }; - if ( - options.reasoningSummary !== undefined && - options.reasoningSummary !== null && - supportsCodexReasoningSummary(model.id) - ) { - config.summary = options.reasoningSummary; + // The backend only emits reasoning summaries when `reasoning.summary` is + // present: omitting it yields zero `response.reasoning_summary_text.*` + // events (measured against gpt-5.5, gpt-5.6-sol and gpt-5.6-terra). So + // `undefined` means "default on" — matching `applyResponsesCompatPolicy` + // on the plain Responses path — and only an explicit `null` (the caller + // hiding thinking) opts out. + if (options.reasoningSummary !== null && supportsCodexReasoningSummary(model.id)) { + config.summary = options.reasoningSummary ?? "auto"; } return config; } @@ -464,12 +482,13 @@ export async function transformRequestBody( body.reasoning = { ...body.reasoning, mode: model.reasoningMode }; } - // Concurrent reasoning summaries (codex-rs `concurrent_reasoning_summaries` - // feature): `sequential_cutoff` lets the server stream output without - // blocking on summary generation. Only meaningful when a summary is - // requested; codex-rs additionally gates on its OpenAI provider check, - // which is inherent here. - if (body.reasoning?.summary !== undefined) { + // Concurrent reasoning summaries (codex-rs `concurrent_reasoning_summaries`): + // `sequential_cutoff` lets the server stream output without blocking on + // summary generation, delivering each completed section as an atomic + // `response.reasoning_summary_text.done`. Opt-in only — see + // {@link concurrentSummariesEnabled} for why. Requires a requested summary; + // codex-rs additionally gates on its OpenAI provider check, inherent here. + if (body.reasoning?.summary !== undefined && concurrentSummariesEnabled()) { body.stream_options = { reasoning_summary_delivery: "sequential_cutoff" }; } else { delete body.stream_options; diff --git a/packages/ai/test/openai-codex-responses-lite.test.ts b/packages/ai/test/openai-codex-responses-lite.test.ts index d176ec1c2..3cfa63fb2 100644 --- a/packages/ai/test/openai-codex-responses-lite.test.ts +++ b/packages/ai/test/openai-codex-responses-lite.test.ts @@ -129,12 +129,13 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe } describe("openai-codex optional response controls", () => { - it("omits optional controls on full requests and forwards explicit controls", async () => { + it("defaults reasoning.summary on and forwards explicit controls", async () => { const model = createCodexModel("gpt-5.5"); + // The backend emits no reasoning summaries at all unless `summary` is + // sent, so an unset `reasoningSummary` must still request one. const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" }); - expect(defaulted.reasoning).toEqual({ effort: "medium" }); - expect("summary" in (defaulted.reasoning ?? {})).toBe(false); + expect(defaulted.reasoning).toEqual({ effort: "medium", summary: "auto" }); expect("context" in (defaulted.reasoning ?? {})).toBe(false); expect("text" in defaulted).toBe(false); expect("stream_options" in defaulted).toBe(false); @@ -151,7 +152,7 @@ describe("openai-codex optional response controls", () => { context: "all_turns", }); expect(explicit.text).toEqual({ verbosity: "low" }); - expect(explicit.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); + expect("stream_options" in explicit).toBe(false); }); it("omits reasoning.summary when explicitly suppressed", async () => { @@ -178,7 +179,7 @@ describe("openai-codex optional response controls", () => { responsesLite: true, reasoningContext: "current_turn", }); - expect(noneEffort.reasoning).toEqual({ effort: "none", context: "all_turns" }); + expect(noneEffort.reasoning).toEqual({ effort: "none", summary: "auto", context: "all_turns" }); const plainRequest = await transformRequestBody({ model: model.id }, model, { responsesLite: false, @@ -759,19 +760,37 @@ describe("openai-codex websocket append with client metadata", () => { }); describe("openai-codex concurrent reasoning summaries", () => { + // Sequential-cutoff delivery is opt-in (it cancels in-flight summary + // sections), so the response-side contract is exercised with it enabled. + let previousConcurrent: string | undefined; + beforeEach(() => { + previousConcurrent = Bun.env.PI_CODEX_CONCURRENT_SUMMARIES; + Bun.env.PI_CODEX_CONCURRENT_SUMMARIES = "1"; + }); + afterEach(() => { + if (previousConcurrent === undefined) delete Bun.env.PI_CODEX_CONCURRENT_SUMMARIES; + else Bun.env.PI_CODEX_CONCURRENT_SUMMARIES = previousConcurrent; + }); + it("counts atomic summary dones as websocket watchdog progress", () => { expect(isOpenAIResponsesProgressEvent({ type: "response.reasoning_summary_text.done" })).toBe(true); }); - it("sends stream_options only when a summary is requested and supported", async () => { + it("sends stream_options only when opted in, with a supported summary requested", async () => { const terra = createCodexModel("gpt-5.6-terra"); - const withSummary = await transformRequestBody({ model: terra.id }, terra, { - reasoningEffort: "medium", - reasoningSummary: "detailed", - }); + const summaryRequest = { reasoningEffort: "medium", reasoningSummary: "detailed" } as const; + + const withSummary = await transformRequestBody({ model: terra.id }, terra, summaryRequest); expect(withSummary.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" }); expect(withSummary.reasoning?.summary).toBe("detailed"); + // Opted out: the summary is still requested, only the delivery mode drops. + delete Bun.env.PI_CODEX_CONCURRENT_SUMMARIES; + const optedOut = await transformRequestBody({ model: terra.id }, terra, summaryRequest); + expect(optedOut.stream_options).toBeUndefined(); + expect(optedOut.reasoning?.summary).toBe("detailed"); + Bun.env.PI_CODEX_CONCURRENT_SUMMARIES = "1"; + const suppressed = await transformRequestBody({ model: terra.id }, terra, { reasoningEffort: "medium", reasoningSummary: null,