fix(ai): stripped unsupported sampling keys from openai-codex-responses payloads

The Codex Responses backend (ChatGPT-subscription `/backend-api/codex/responses`)
returns 400 `{"detail":"Unsupported parameter: temperature"}` for every sampling
control, but `buildTransformedCodexRequestBody` forwarded `temperature`, `top_p`,
`top_k`, `min_p`, `presence_penalty`, and `repetition_penalty` straight through
whenever the caller's `StreamOptions` carried them — so any non-default sampling
setting failed every turn.

- Provider drops the full sampling set (matching codex-rs, which sends none).
- `RequestBody` no longer types fields it never populates for codex.
- Auth-gateway defensive strip in `buildStreamOptions` and the pi-native handler
  widened from `{temperature, topP}` to the same set plus
  `stopSequences`/`frequencyPenalty`.
- Regression test pins the captured body to omit every key even when the caller
  sets all of them.

Fixes #3117
This commit is contained in:
roboomp
2026-06-20 09:48:58 +00:00
parent 62756efb8f
commit f6c4d6fb62
5 changed files with 82 additions and 37 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed `openai-codex-responses` forwarding sampling controls (`temperature`, `top_p`, `top_k`, `min_p`, `presence_penalty`, `repetition_penalty`) into the Codex request body — the ChatGPT-subscription Codex backend rejects each of them with a 400 `{"detail":"Unsupported parameter: temperature"}`, so any caller setting non-default `StreamOptions` saw every turn fail. The provider now drops the full sampling set (matching codex-rs), and the auth-gateway's defensive strip on both `buildStreamOptions` and the pi-native path was widened from `{temperature, topP}` to the same set plus `stopSequences`/`frequencyPenalty`. ([#3117](https://github.com/can1357/oh-my-pi/issues/3117))
## [16.1.4] - 2026-06-19
### Added
+18 -12
View File
@@ -120,20 +120,20 @@ function deriveSessionId(modelId: string, context: Context): string {
function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: AbortSignal): SimpleStreamOptions {
const opts: SimpleStreamOptions = { signal };
const { options } = parsed;
// Codex backend rejects `temperature` / `top_p` (per-model defaults only),
// so we drop them silently for that one provider. Every other unsupported
// option is just ignored by `streamSimple` if the underlying provider
// doesn't honour it.
// Codex backend rejects every sampling control with
// `Unsupported parameter: …` (#3117). Strip the full set for that one
// provider; everything else is harmless to forward — `streamSimple` ignores
// what the underlying provider doesn't honour.
const isCodex = api === "openai-codex-responses";
if (options.maxOutputTokens !== undefined) opts.maxTokens = options.maxOutputTokens;
if (options.temperature !== undefined && !isCodex) opts.temperature = options.temperature;
if (options.topP !== undefined && !isCodex) opts.topP = options.topP;
if (options.topK !== undefined) opts.topK = options.topK;
if (options.minP !== undefined) opts.minP = options.minP;
if (options.stopSequences !== undefined) opts.stopSequences = options.stopSequences;
if (options.presencePenalty !== undefined) opts.presencePenalty = options.presencePenalty;
if (options.frequencyPenalty !== undefined) opts.frequencyPenalty = options.frequencyPenalty;
if (options.repetitionPenalty !== undefined) opts.repetitionPenalty = options.repetitionPenalty;
if (options.topK !== undefined && !isCodex) opts.topK = options.topK;
if (options.minP !== undefined && !isCodex) opts.minP = options.minP;
if (options.stopSequences !== undefined && !isCodex) opts.stopSequences = options.stopSequences;
if (options.presencePenalty !== undefined && !isCodex) opts.presencePenalty = options.presencePenalty;
if (options.frequencyPenalty !== undefined && !isCodex) opts.frequencyPenalty = options.frequencyPenalty;
if (options.repetitionPenalty !== undefined && !isCodex) opts.repetitionPenalty = options.repetitionPenalty;
if (options.metadata !== undefined) opts.metadata = options.metadata;
if (options.headers !== undefined) opts.headers = { ...(opts.headers ?? {}), ...options.headers };
if (options.toolChoice !== undefined) {
@@ -662,8 +662,8 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe
// Build the SimpleStreamOptions actually handed to `streamSimple`. We
// trust the client's options (already allow-listed by `parseRequest`) and
// only inject server-controlled fields. The codex temperature/topP strip
// matches `buildStreamOptions` — Codex rejects them with a 400.
// only inject server-controlled fields. The codex sampling strip mirrors
// `buildStreamOptions` — Codex rejects every one with a 400 (#3117).
const streamOpts: SimpleStreamOptions = { ...parsed.options, apiKey, signal: controller.signal };
streamOpts.apiKey = buildGatewayApiKeyResolver(
bootOpts.storage,
@@ -677,6 +677,12 @@ async function handlePiNative(bootOpts: AuthGatewayBootOptions, req: Request, pe
if (model.api === "openai-codex-responses") {
delete streamOpts.temperature;
delete streamOpts.topP;
delete streamOpts.topK;
delete streamOpts.minP;
delete streamOpts.stopSequences;
delete streamOpts.presencePenalty;
delete streamOpts.frequencyPenalty;
delete streamOpts.repetitionPenalty;
}
// Merge gateway-captured passthrough headers under the client's own
// headers — the client's values win when they collide.
@@ -729,25 +729,12 @@ async function buildTransformedCodexRequestBody(
// `maxTokens` is intentionally not forwarded: transformRequestBody strips
// `max_output_tokens`/`max_completion_tokens` (the Codex backend rejects
// caller-supplied output caps).
if (options?.temperature !== undefined) {
params.temperature = options.temperature;
}
if (options?.topP !== undefined) {
params.top_p = options.topP;
}
if (options?.topK !== undefined) {
params.top_k = options.topK;
}
if (options?.minP !== undefined) {
params.min_p = options.minP;
}
if (options?.presencePenalty !== undefined) {
params.presence_penalty = options.presencePenalty;
}
if (options?.repetitionPenalty !== undefined) {
params.repetition_penalty = options.repetitionPenalty;
}
// caller-supplied output caps). Sampling controls (`temperature`, `top_p`,
// `top_k`, `min_p`, `presence_penalty`, `repetition_penalty`,
// `frequency_penalty`, `stop`) are likewise refused with
// `{"detail":"Unsupported parameter: temperature"}` etc., so we drop
// everything from `StreamOptions` rather than forwarding any of them.
// (#3117 — codex-rs sends none of these either.)
applyOpenAIServiceTier(params, options?.serviceTier, model.provider);
if (context.tools && context.tools.length > 0) {
params.tools = convertOpenAICodexResponsesTools(context.tools, model);
@@ -41,12 +41,10 @@ export interface RequestBody {
input?: InputItem[];
tools?: unknown;
tool_choice?: unknown;
temperature?: number;
top_p?: number;
top_k?: number;
min_p?: number;
presence_penalty?: number;
repetition_penalty?: number;
// Sampling controls (temperature/top_p/top_k/min_p/presence_penalty/
// repetition_penalty/frequency_penalty/stop) are intentionally absent: the
// Codex backend rejects every one with a 400 `Unsupported parameter`, so
// the transformer never sets them (#3117).
reasoning?: Partial<ReasoningConfig>;
text?: {
verbosity?: "low" | "medium" | "high";
@@ -1586,6 +1586,56 @@ describe("openai-codex streaming", () => {
expect(capturedBody?.prompt_cache_key).toBe(promptCacheKey);
});
it("omits unsupported sampling keys (temperature/top_p/top_k/min_p/penalties) from the Codex Responses body", async () => {
// Regression for #3117 — Codex backend returns
// `{"detail":"Unsupported parameter: temperature"}` 400 for any of
// these keys, so the provider MUST drop them even when the caller's
// `StreamOptions` carries non-default values.
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
const token = createCodexTestToken();
const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false };
let capturedBody: Record<string, unknown> | undefined;
const fetchMock = vi.fn(async (input: string | URL, init?: RequestInit) => {
const url = typeof input === "string" ? input : input.toString();
if (url === "https://chatgpt.com/backend-api/codex/responses") {
capturedBody =
typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : undefined;
return new Response(createCompletedCodexSse("Hello"), {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
return new Response("not found", { status: 404 });
});
await streamOpenAICodexResponses(model, createCodexTestContext(), {
fetch: fetchMock as FetchImpl,
apiKey: token,
temperature: 0.2,
topP: 0.9,
topK: 40,
minP: 0.05,
presencePenalty: 0.1,
frequencyPenalty: 0.1,
repetitionPenalty: 1.1,
stopSequences: ["STOP"],
}).result();
expect(capturedBody).toBeDefined();
expect(capturedBody?.temperature).toBeUndefined();
expect(capturedBody?.top_p).toBeUndefined();
expect(capturedBody?.top_k).toBeUndefined();
expect(capturedBody?.min_p).toBeUndefined();
expect(capturedBody?.presence_penalty).toBeUndefined();
expect(capturedBody?.frequency_penalty).toBeUndefined();
expect(capturedBody?.repetition_penalty).toBeUndefined();
expect(capturedBody?.stop).toBeUndefined();
expect(capturedBody?.stop_sequences).toBeUndefined();
});
it("rejects gpt-5.3-codex minimal reasoning effort instead of clamping", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());