fix(ai): omitted unsupported reasoning sampling params

- Added resolved sampling-parameter compatibility for OpenAI o-series and GPT-5+ models.
- Stopped Responses and Chat Completions payloads from forwarding temperature and related controls to restricted models.
- Added catalog and wire-payload regressions for GitHub Copilot GPT-5.6 Luna.

Fixes #5606
This commit is contained in:
roboomp
2026-07-15 19:30:53 +00:00
parent d5cd24f39a
commit f6de373506
12 changed files with 187 additions and 27 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed OpenAI Responses and Chat Completions requests forwarding unsupported sampling parameters such as `temperature` to o-series and GPT-5+ models, preventing 400 errors for mnemopi memory calls through GitHub Copilot GPT-5.6 Luna. ([#5606](https://github.com/can1357/oh-my-pi/issues/5606))
## [17.0.0] - 2026-07-15
### Changed
+24 -20
View File
@@ -1460,31 +1460,35 @@ function buildParams(
params.store = false;
}
if (options?.temperature !== undefined) {
params.temperature = options.temperature;
}
if (options?.topP !== undefined) {
params.top_p = options.topP;
}
if (options?.topK !== undefined) {
params.top_k = options.topK;
}
if (options?.minP !== undefined) {
params.min_p = options.minP;
}
if (options?.presencePenalty !== undefined) {
params.presence_penalty = options.presencePenalty;
}
if (options?.repetitionPenalty !== undefined) {
params.repetition_penalty = options.repetitionPenalty;
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
// sampling params with a 400 on every serving host (#5606).
if (initialCompat.supportsSamplingParams) {
if (options?.temperature !== undefined) {
params.temperature = options.temperature;
}
if (options?.topP !== undefined) {
params.top_p = options.topP;
}
if (options?.topK !== undefined) {
params.top_k = options.topK;
}
if (options?.minP !== undefined) {
params.min_p = options.minP;
}
if (options?.presencePenalty !== undefined) {
params.presence_penalty = options.presencePenalty;
}
if (options?.repetitionPenalty !== undefined) {
params.repetition_penalty = options.repetitionPenalty;
}
if (options?.frequencyPenalty !== undefined) {
params.frequency_penalty = options.frequencyPenalty;
}
}
if (options?.stopSequences?.length) {
const seqs = options.stopSequences;
params.stop = seqs.length === 1 ? seqs[0] : seqs.slice(0, 4);
}
if (options?.frequencyPenalty !== undefined) {
params.frequency_penalty = options.frequencyPenalty;
}
applyOpenAIServiceTier(params, options?.serviceTier, model);
if (context.tools?.length) {
+13 -7
View File
@@ -2717,7 +2717,9 @@ type CommonSamplingOptions = Pick<
export function applyCommonResponsesSamplingParams<P extends CommonResponsesParams>(
params: P,
options: CommonSamplingOptions | undefined,
model: Pick<Model, "provider" | "api" | "id" | "omitMaxOutputTokens" | "maxTokens">,
model: Pick<Model, "provider" | "api" | "id" | "omitMaxOutputTokens" | "maxTokens"> & {
compat: Pick<ResolvedOpenAISharedCompat, "supportsSamplingParams">;
},
): void {
if (options?.maxTokens && !model.omitMaxOutputTokens) {
params.max_output_tokens = Math.min(
@@ -2726,12 +2728,16 @@ export function applyCommonResponsesSamplingParams<P extends CommonResponsesPara
OPENAI_MAX_OUTPUT_TOKENS,
);
}
if (options?.temperature !== undefined) params.temperature = options.temperature;
if (options?.topP !== undefined) params.top_p = options.topP;
if (options?.topK !== undefined) params.top_k = options.topK;
if (options?.minP !== undefined) params.min_p = options.minP;
if (options?.presencePenalty !== undefined) params.presence_penalty = options.presencePenalty;
if (options?.repetitionPenalty !== undefined) params.repetition_penalty = options.repetitionPenalty;
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
// sampling params with a 400 on every serving host (#5606).
if (model.compat.supportsSamplingParams) {
if (options?.temperature !== undefined) params.temperature = options.temperature;
if (options?.topP !== undefined) params.top_p = options.topP;
if (options?.topK !== undefined) params.top_k = options.topK;
if (options?.minP !== undefined) params.min_p = options.minP;
if (options?.presencePenalty !== undefined) params.presence_penalty = options.presencePenalty;
if (options?.repetitionPenalty !== undefined) params.repetition_penalty = options.repetitionPenalty;
}
applyOpenAIServiceTier(params, options?.serviceTier, model);
}
@@ -27,6 +27,7 @@ const compat: ResolvedOpenAICompat = {
supportsMultipleSystemMessages: true,
supportsReasoningEffort: true,
supportsReasoningParams: true,
supportsSamplingParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
@@ -196,6 +196,7 @@ describe("openai-completions compatibility", () => {
supportsStrictMode: true,
toolStrictMode: "none",
supportsReasoningParams: true,
supportsSamplingParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
@@ -50,6 +50,7 @@ const compat: ResolvedOpenAICompat = {
supportsStrictMode: true,
toolStrictMode: "none",
supportsReasoningParams: true,
supportsSamplingParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
@@ -0,0 +1,70 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
function mockSseFetch(): { fetchMock: FetchImpl; captured: Record<string, unknown> } {
const captured: Record<string, unknown> = {};
const fetchMock: FetchImpl = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => {
const body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record<string, unknown>) : {};
Object.assign(captured, body);
const event = {
type: "response.completed",
response: {
status: "completed",
usage: {
input_tokens: 1,
output_tokens: 1,
total_tokens: 2,
input_tokens_details: { cached_tokens: 0 },
},
},
};
return new Response(`data: ${JSON.stringify(event)}\n\n`, {
status: 200,
headers: { "content-type": "text/event-stream" },
});
});
return { fetchMock, captured };
}
const ctx: Context = {
systemPrompt: ["hi"],
messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
};
async function drain(model: Model<"openai-responses">): Promise<Record<string, unknown>> {
const { fetchMock, captured } = mockSseFetch();
const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock, temperature: 0 });
for await (const event of stream) {
if (event.type === "done" || event.type === "error") break;
}
return captured;
}
afterEach(() => {
vi.restoreAllMocks();
});
describe("openai-responses sampling-param gating (#5606)", () => {
it("omits temperature for OpenAI reasoning models that reject it", async () => {
const model = getBundledModel("openai", "gpt-5") as Model<"openai-responses">;
expect(model.compat.supportsSamplingParams).toBe(false);
const body = await drain(model);
expect(body).not.toHaveProperty("temperature");
});
it("omits temperature for GitHub Copilot gpt-5.6 (the reported model)", async () => {
const model = getBundledModel("github-copilot", "gpt-5.6-luna") as Model<"openai-responses">;
expect(model.compat.supportsSamplingParams).toBe(false);
const body = await drain(model);
expect(body).not.toHaveProperty("temperature");
});
it("still forwards temperature for non-restricted OpenAI models", async () => {
const model = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-responses">;
expect(model.compat.supportsSamplingParams).toBe(true);
const body = await drain(model);
expect(body.temperature).toBe(0);
});
});
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Added
- Added resolved OpenAI sampling-parameter compatibility metadata for o-series and GPT-5+ models.
## [16.5.2] - 2026-07-14
### Fixed
+7
View File
@@ -18,6 +18,7 @@ import {
isKimiK26ModelId,
isKimiModelId,
isMimoModelIdOrName,
isOpenAISamplingRestrictedModelId,
isQwenModelId,
modelFamilyToken,
} from "../identity/family";
@@ -409,6 +410,9 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
supportsReasoningEffort: !isGrok && !isXiaomiMimo && (!(isZai || isZhipu) || supportsZaiReasoningEffort),
// GitHub Copilot's chat-completions endpoint rejects reasoning params wholesale.
supportsReasoningParams: provider !== "github-copilot",
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
// temperature/top_p/… with a 400 on every serving host (#5606).
supportsSamplingParams: !isOpenAISamplingRestrictedModelId(spec.id),
reasoningEffortMap: isMimoReasoningEffortModel ? MIMO_REASONING_EFFORT_MAP : {},
supportsUsageInStreaming: !isCerebras,
// pi-ai's thinking-loop guard is gemini-only; default the flag from the
@@ -604,6 +608,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
spec.provider !== "xai-oauth" && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
reasoningEffortMap: {},
supportsReasoningParams: true,
// OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
// temperature/top_p/… with a 400 on every serving host (#5606).
supportsSamplingParams: !isOpenAISamplingRestrictedModelId(id),
thinkingFormat,
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
omitReasoningEffort: false,
+26
View File
@@ -163,6 +163,32 @@ export const supportsAllTurnsReasoningContext = isOpenAIWireGen54Plus;
*/
export const supportsCodexReasoningSummary = isOpenAIWireGen54Plus;
/** OpenAI proprietary reasoning families keyed off the parsed gpt version (gpt-5+). */
const isOpenAIWireGen5Plus = memo((modelId: string): boolean => {
const parsed = parseOpenAIModel(bareModelId(modelId));
if (!parsed) return false;
return semverGte(parsed.version, "5");
});
/** o-series reasoning ids (`o1`, `o1-pro`, `o3`, `o3-mini`, `o4-mini`, `openai/o3`, …). */
const O_SERIES_REASONING_RE = /(^|\/)o[134](?:[-.]|$)/i;
/**
* OpenAI proprietary models whose serving path rejects explicit sampling
* parameters (`temperature`, `top_p`, `top_k`, …) with
* `400 Unsupported parameter: 'temperature' is not supported with this model`.
* Covers the o-series and the entire gpt-5+ generation — base, `mini`, `nano`,
* `codex*`, the `luna`/`sol`/`terra` SKUs, and the `-chat-latest` variants,
* since even the non-reasoning gpt-5 chat models reject sampling params (see
* litellm#13781). Holds regardless of which OpenAI-serving host proxies the
* model (official, Azure, GitHub Copilot). Version floor (not an allowlist) so
* 6.x inherits automatically. Issue #5606.
*/
export const isOpenAISamplingRestrictedModelId = memo((modelId: string): boolean => {
const bare = bareModelId(modelId);
return isOpenAIWireGen5Plus(modelId) || O_SERIES_REASONING_RE.test(bare);
});
/**
* Reasoning-capable GLM coding SKUs: glm-4.5 and up on the base / `-air` /
* `-turbo` lines. Excludes the vision (`…v`) shape, the non-reasoning
+11
View File
@@ -327,6 +327,15 @@ export interface OpenAICompat {
toolStrictMode?: "all_strict" | "none";
/** Whether request shaping may send reasoning params at all. Default: auto-detected (disabled for GitHub Copilot chat-completions). */
supportsReasoningParams?: boolean;
/**
* Whether the endpoint accepts explicit sampling parameters (`temperature`,
* `top_p`, `top_k`, `min_p`, penalties). OpenAI proprietary reasoning models
* (o-series, gpt-5+) reject them with `400 Unsupported parameter:
* 'temperature' is not supported with this model` on every serving host
* (official, Azure, GitHub Copilot). When unset, auto-detected from the
* model id. Default: true. Issue #5606.
*/
supportsSamplingParams?: boolean;
/** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
alwaysSendMaxTokens?: boolean;
/** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
@@ -464,6 +473,7 @@ export interface ResolvedOpenAISharedCompat {
supportsReasoningEffort: boolean;
reasoningEffortMap: Partial<Record<Effort, string>>;
supportsReasoningParams: boolean;
supportsSamplingParams: boolean;
thinkingFormat: OpenAIReasoningFormat;
reasoningDisableMode: OpenAIReasoningDisableMode;
omitReasoningEffort: boolean;
@@ -516,6 +526,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
| "supportsReasoningEffort"
| "reasoningEffortMap"
| "supportsReasoningParams"
| "supportsSamplingParams"
| "thinkingFormat"
| "reasoningDisableMode"
| "omitReasoningEffort"
@@ -584,6 +584,31 @@ describe("model thinking derivation", () => {
expect(fable.compat.supportsSamplingParams).toBe(false);
});
it("bakes sampling-param rejection into OpenAI reasoning compat (#5606)", () => {
// GitHub Copilot Responses gpt-5.6 — the reported failing model.
const luna = createModel({
id: "gpt-5.6-luna",
api: "openai-responses",
provider: "github-copilot",
baseUrl: "https://api.githubcopilot.com",
});
const gpt5 = createModel({ id: "gpt-5", api: "openai-responses", provider: "openai" });
const gpt5Mini = createModel({ id: "gpt-5-mini", api: "openai-completions", provider: "openai" });
const gpt5Chat = createModel({ id: "gpt-5-chat-latest", api: "openai-responses", provider: "openai" });
const oThree = createModel({ id: "o3-mini", api: "openai-responses", provider: "openai" });
// Non-restricted OpenAI + non-OpenAI models keep sampling support.
const gpt4o = createModel({ id: "gpt-4o", api: "openai-responses", provider: "openai", reasoning: false });
const kimi = createModel({ id: "kimi-k2.6", api: "openai-completions", provider: "moonshot" });
expect(luna.compat.supportsSamplingParams).toBe(false);
expect(gpt5.compat.supportsSamplingParams).toBe(false);
expect(gpt5Mini.compat.supportsSamplingParams).toBe(false);
expect(gpt5Chat.compat.supportsSamplingParams).toBe(false);
expect(oThree.compat.supportsSamplingParams).toBe(false);
expect(gpt4o.compat.supportsSamplingParams).toBe(true);
expect(kimi.compat.supportsSamplingParams).toBe(true);
});
it("encodes effort-dial-less reasoners as thinking: undefined", () => {
const model = createModel({
id: "grok-build",