feat(ai): enabled automatic reasoning summaries in codex requests

- Default reasoning.summary to auto in openai-codex requests to ensure summaries are emitted by the backend.
- Gate stream_options reasoning_summary_delivery behind the PI_CODEX_CONCURRENT_SUMMARIES environment variable opt-in.
- Update tests in openai-codex-responses-lite to verify default summary behavior and opt-in concurrent delivery controls.
This commit is contained in:
can1357
2026-08-11 15:29:26 +02:00
parent 8fed93632a
commit f22cb606bf
2 changed files with 60 additions and 22 deletions
@@ -109,6 +109,22 @@ export function resolveCodexResponsesLite(
return model.useResponsesLite === true;
}
/**
* Whether to request `stream_options.reasoning_summary_delivery =
* "sequential_cutoff"` (codex-rs `concurrent_reasoning_summaries`), enabled by
* `PI_CODEX_CONCURRENT_SUMMARIES=1`.
*
* Off by default because the mode cancels summary sections still in flight when
* the reasoning item closes: measured over 12 interleaved turns it halved
* visible thinking (0.83 vs 1.67 summary parts, 37 vs 69 chars per turn) and
* produced no summary at all on 3 of 12 turns. codex-rs ships it disabled too
* (`Stage::UnderDevelopment`, `default_enabled: false`).
*/
function concurrentSummariesEnabled(): boolean {
const env = $env.PI_CODEX_CONCURRENT_SUMMARIES?.trim().toLowerCase();
return env === "1" || env === "true";
}
/**
* Clamp a user-facing effort to the model's ladder, then remap to the wire
* tier. User efforts map 1:1 onto wire tiers; the effort map only covers
@@ -145,12 +161,14 @@ function getReasoningConfig(
const config: ReasoningConfig = {
effort: effort === "none" ? "none" : mapCodexWireEffort(model, effort),
};
if (
options.reasoningSummary !== undefined &&
options.reasoningSummary !== null &&
supportsCodexReasoningSummary(model.id)
) {
config.summary = options.reasoningSummary;
// The backend only emits reasoning summaries when `reasoning.summary` is
// present: omitting it yields zero `response.reasoning_summary_text.*`
// events (measured against gpt-5.5, gpt-5.6-sol and gpt-5.6-terra). So
// `undefined` means "default on" — matching `applyResponsesCompatPolicy`
// on the plain Responses path — and only an explicit `null` (the caller
// hiding thinking) opts out.
if (options.reasoningSummary !== null && supportsCodexReasoningSummary(model.id)) {
config.summary = options.reasoningSummary ?? "auto";
}
return config;
}
@@ -464,12 +482,13 @@ export async function transformRequestBody(
body.reasoning = { ...body.reasoning, mode: model.reasoningMode };
}
// Concurrent reasoning summaries (codex-rs `concurrent_reasoning_summaries`
// feature): `sequential_cutoff` lets the server stream output without
// blocking on summary generation. Only meaningful when a summary is
// requested; codex-rs additionally gates on its OpenAI provider check,
// which is inherent here.
if (body.reasoning?.summary !== undefined) {
// Concurrent reasoning summaries (codex-rs `concurrent_reasoning_summaries`):
// `sequential_cutoff` lets the server stream output without blocking on
// summary generation, delivering each completed section as an atomic
// `response.reasoning_summary_text.done`. Opt-in only — see
// {@link concurrentSummariesEnabled} for why. Requires a requested summary;
// codex-rs additionally gates on its OpenAI provider check, inherent here.
if (body.reasoning?.summary !== undefined && concurrentSummariesEnabled()) {
body.stream_options = { reasoning_summary_delivery: "sequential_cutoff" };
} else {
delete body.stream_options;
@@ -129,12 +129,13 @@ function createCodexFetchMock(sse: string, onRequest: (captured: CapturedCodexRe
}
describe("openai-codex optional response controls", () => {
it("omits optional controls on full requests and forwards explicit controls", async () => {
it("defaults reasoning.summary on and forwards explicit controls", async () => {
const model = createCodexModel("gpt-5.5");
// The backend emits no reasoning summaries at all unless `summary` is
// sent, so an unset `reasoningSummary` must still request one.
const defaulted = await transformRequestBody({ model: model.id }, model, { reasoningEffort: "medium" });
expect(defaulted.reasoning).toEqual({ effort: "medium" });
expect("summary" in (defaulted.reasoning ?? {})).toBe(false);
expect(defaulted.reasoning).toEqual({ effort: "medium", summary: "auto" });
expect("context" in (defaulted.reasoning ?? {})).toBe(false);
expect("text" in defaulted).toBe(false);
expect("stream_options" in defaulted).toBe(false);
@@ -151,7 +152,7 @@ describe("openai-codex optional response controls", () => {
context: "all_turns",
});
expect(explicit.text).toEqual({ verbosity: "low" });
expect(explicit.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" });
expect("stream_options" in explicit).toBe(false);
});
it("omits reasoning.summary when explicitly suppressed", async () => {
@@ -178,7 +179,7 @@ describe("openai-codex optional response controls", () => {
responsesLite: true,
reasoningContext: "current_turn",
});
expect(noneEffort.reasoning).toEqual({ effort: "none", context: "all_turns" });
expect(noneEffort.reasoning).toEqual({ effort: "none", summary: "auto", context: "all_turns" });
const plainRequest = await transformRequestBody({ model: model.id }, model, {
responsesLite: false,
@@ -759,19 +760,37 @@ describe("openai-codex websocket append with client metadata", () => {
});
describe("openai-codex concurrent reasoning summaries", () => {
// Sequential-cutoff delivery is opt-in (it cancels in-flight summary
// sections), so the response-side contract is exercised with it enabled.
let previousConcurrent: string | undefined;
beforeEach(() => {
previousConcurrent = Bun.env.PI_CODEX_CONCURRENT_SUMMARIES;
Bun.env.PI_CODEX_CONCURRENT_SUMMARIES = "1";
});
afterEach(() => {
if (previousConcurrent === undefined) delete Bun.env.PI_CODEX_CONCURRENT_SUMMARIES;
else Bun.env.PI_CODEX_CONCURRENT_SUMMARIES = previousConcurrent;
});
it("counts atomic summary dones as websocket watchdog progress", () => {
expect(isOpenAIResponsesProgressEvent({ type: "response.reasoning_summary_text.done" })).toBe(true);
});
it("sends stream_options only when a summary is requested and supported", async () => {
it("sends stream_options only when opted in, with a supported summary requested", async () => {
const terra = createCodexModel("gpt-5.6-terra");
const withSummary = await transformRequestBody({ model: terra.id }, terra, {
reasoningEffort: "medium",
reasoningSummary: "detailed",
});
const summaryRequest = { reasoningEffort: "medium", reasoningSummary: "detailed" } as const;
const withSummary = await transformRequestBody({ model: terra.id }, terra, summaryRequest);
expect(withSummary.stream_options).toEqual({ reasoning_summary_delivery: "sequential_cutoff" });
expect(withSummary.reasoning?.summary).toBe("detailed");
// Opted out: the summary is still requested, only the delivery mode drops.
delete Bun.env.PI_CODEX_CONCURRENT_SUMMARIES;
const optedOut = await transformRequestBody({ model: terra.id }, terra, summaryRequest);
expect(optedOut.stream_options).toBeUndefined();
expect(optedOut.reasoning?.summary).toBe("detailed");
Bun.env.PI_CODEX_CONCURRENT_SUMMARIES = "1";
const suppressed = await transformRequestBody({ model: terra.id }, terra, {
reasoningEffort: "medium",
reasoningSummary: null,