diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md
index d3095bf01..2779aa270 100644
--- a/packages/ai/CHANGELOG.md
+++ b/packages/ai/CHANGELOG.md
@@ -2,6 +2,10 @@
## [Unreleased]
+### Fixed
+
+- Fixed OpenAI Responses and Chat Completions requests forwarding unsupported sampling parameters such as `temperature` to o-series and GPT-5+ models, preventing 400 errors for mnemopi memory calls through GitHub Copilot GPT-5.6 Luna. ([#5606](https://github.com/can1357/oh-my-pi/issues/5606))
+
## [17.0.0] - 2026-07-15
### Changed
diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts
index 83cc11666..22203967e 100644
--- a/packages/ai/src/providers/openai-completions.ts
+++ b/packages/ai/src/providers/openai-completions.ts
@@ -1460,31 +1460,35 @@ function buildParams(
params.store = false;
}
- if (options?.temperature !== undefined) {
- params.temperature = options.temperature;
- }
- if (options?.topP !== undefined) {
- params.top_p = options.topP;
- }
- if (options?.topK !== undefined) {
- params.top_k = options.topK;
- }
- if (options?.minP !== undefined) {
- params.min_p = options.minP;
- }
- if (options?.presencePenalty !== undefined) {
- params.presence_penalty = options.presencePenalty;
- }
- if (options?.repetitionPenalty !== undefined) {
- params.repetition_penalty = options.repetitionPenalty;
+ // OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
+ // sampling params with a 400 on every serving host (#5606).
+ if (initialCompat.supportsSamplingParams) {
+ if (options?.temperature !== undefined) {
+ params.temperature = options.temperature;
+ }
+ if (options?.topP !== undefined) {
+ params.top_p = options.topP;
+ }
+ if (options?.topK !== undefined) {
+ params.top_k = options.topK;
+ }
+ if (options?.minP !== undefined) {
+ params.min_p = options.minP;
+ }
+ if (options?.presencePenalty !== undefined) {
+ params.presence_penalty = options.presencePenalty;
+ }
+ if (options?.repetitionPenalty !== undefined) {
+ params.repetition_penalty = options.repetitionPenalty;
+ }
+ if (options?.frequencyPenalty !== undefined) {
+ params.frequency_penalty = options.frequencyPenalty;
+ }
}
if (options?.stopSequences?.length) {
const seqs = options.stopSequences;
params.stop = seqs.length === 1 ? seqs[0] : seqs.slice(0, 4);
}
- if (options?.frequencyPenalty !== undefined) {
- params.frequency_penalty = options.frequencyPenalty;
- }
applyOpenAIServiceTier(params, options?.serviceTier, model);
if (context.tools?.length) {
diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts
index 64189c5e6..0943b9dee 100644
--- a/packages/ai/src/providers/openai-shared.ts
+++ b/packages/ai/src/providers/openai-shared.ts
@@ -2717,7 +2717,9 @@ type CommonSamplingOptions = Pick<
export function applyCommonResponsesSamplingParams
(
params: P,
options: CommonSamplingOptions | undefined,
- model: Pick,
+ model: Pick & {
+ compat: Pick;
+ },
): void {
if (options?.maxTokens && !model.omitMaxOutputTokens) {
params.max_output_tokens = Math.min(
@@ -2726,12 +2728,16 @@ export function applyCommonResponsesSamplingParams {
supportsStrictMode: true,
toolStrictMode: "none",
supportsReasoningParams: true,
+ supportsSamplingParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts
index fe2731ecf..e598f6068 100644
--- a/packages/ai/test/openai-completions-tool-result-images.test.ts
+++ b/packages/ai/test/openai-completions-tool-result-images.test.ts
@@ -50,6 +50,7 @@ const compat: ResolvedOpenAICompat = {
supportsStrictMode: true,
toolStrictMode: "none",
supportsReasoningParams: true,
+ supportsSamplingParams: true,
alwaysSendMaxTokens: false,
isOpenRouterHost: false,
isVercelGatewayHost: false,
diff --git a/packages/ai/test/openai-responses-sampling-params.test.ts b/packages/ai/test/openai-responses-sampling-params.test.ts
new file mode 100644
index 000000000..cedab6f30
--- /dev/null
+++ b/packages/ai/test/openai-responses-sampling-params.test.ts
@@ -0,0 +1,70 @@
+import { afterEach, describe, expect, it, vi } from "bun:test";
+import { streamSimple } from "@oh-my-pi/pi-ai/stream";
+import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
+import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
+
+function mockSseFetch(): { fetchMock: FetchImpl; captured: Record } {
+ const captured: Record = {};
+ const fetchMock: FetchImpl = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => {
+ const body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record) : {};
+ Object.assign(captured, body);
+ const event = {
+ type: "response.completed",
+ response: {
+ status: "completed",
+ usage: {
+ input_tokens: 1,
+ output_tokens: 1,
+ total_tokens: 2,
+ input_tokens_details: { cached_tokens: 0 },
+ },
+ },
+ };
+ return new Response(`data: ${JSON.stringify(event)}\n\n`, {
+ status: 200,
+ headers: { "content-type": "text/event-stream" },
+ });
+ });
+ return { fetchMock, captured };
+}
+
+const ctx: Context = {
+ systemPrompt: ["hi"],
+ messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
+};
+
+async function drain(model: Model<"openai-responses">): Promise> {
+ const { fetchMock, captured } = mockSseFetch();
+ const stream = streamSimple(model, ctx, { apiKey: "k", fetch: fetchMock, temperature: 0 });
+ for await (const event of stream) {
+ if (event.type === "done" || event.type === "error") break;
+ }
+ return captured;
+}
+
+afterEach(() => {
+ vi.restoreAllMocks();
+});
+
+describe("openai-responses sampling-param gating (#5606)", () => {
+ it("omits temperature for OpenAI reasoning models that reject it", async () => {
+ const model = getBundledModel("openai", "gpt-5") as Model<"openai-responses">;
+ expect(model.compat.supportsSamplingParams).toBe(false);
+ const body = await drain(model);
+ expect(body).not.toHaveProperty("temperature");
+ });
+
+ it("omits temperature for GitHub Copilot gpt-5.6 (the reported model)", async () => {
+ const model = getBundledModel("github-copilot", "gpt-5.6-luna") as Model<"openai-responses">;
+ expect(model.compat.supportsSamplingParams).toBe(false);
+ const body = await drain(model);
+ expect(body).not.toHaveProperty("temperature");
+ });
+
+ it("still forwards temperature for non-restricted OpenAI models", async () => {
+ const model = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-responses">;
+ expect(model.compat.supportsSamplingParams).toBe(true);
+ const body = await drain(model);
+ expect(body.temperature).toBe(0);
+ });
+});
diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md
index 7124fdc93..c890466ca 100644
--- a/packages/catalog/CHANGELOG.md
+++ b/packages/catalog/CHANGELOG.md
@@ -2,6 +2,10 @@
## [Unreleased]
+### Added
+
+- Added resolved OpenAI sampling-parameter compatibility metadata for o-series and GPT-5+ models.
+
## [16.5.2] - 2026-07-14
### Fixed
diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts
index 2425a4744..dc70d322a 100644
--- a/packages/catalog/src/compat/openai.ts
+++ b/packages/catalog/src/compat/openai.ts
@@ -18,6 +18,7 @@ import {
isKimiK26ModelId,
isKimiModelId,
isMimoModelIdOrName,
+ isOpenAISamplingRestrictedModelId,
isQwenModelId,
modelFamilyToken,
} from "../identity/family";
@@ -409,6 +410,9 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
supportsReasoningEffort: !isGrok && !isXiaomiMimo && (!(isZai || isZhipu) || supportsZaiReasoningEffort),
// GitHub Copilot's chat-completions endpoint rejects reasoning params wholesale.
supportsReasoningParams: provider !== "github-copilot",
+ // OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
+ // temperature/top_p/… with a 400 on every serving host (#5606).
+ supportsSamplingParams: !isOpenAISamplingRestrictedModelId(spec.id),
reasoningEffortMap: isMimoReasoningEffortModel ? MIMO_REASONING_EFFORT_MAP : {},
supportsUsageInStreaming: !isCerebras,
// pi-ai's thinking-loop guard is gemini-only; default the flag from the
@@ -604,6 +608,9 @@ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): Resol
spec.provider !== "xai-oauth" && !modelMatchesHost({ provider: spec.provider, baseUrl }, "githubCopilot"),
reasoningEffortMap: {},
supportsReasoningParams: true,
+ // OpenAI proprietary reasoning models (o-series, gpt-5+) reject explicit
+ // temperature/top_p/… with a 400 on every serving host (#5606).
+ supportsSamplingParams: !isOpenAISamplingRestrictedModelId(id),
thinkingFormat,
reasoningDisableMode: resolveReasoningDisableMode(thinkingFormat),
omitReasoningEffort: false,
diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts
index aa206cd9f..79653c862 100644
--- a/packages/catalog/src/identity/family.ts
+++ b/packages/catalog/src/identity/family.ts
@@ -163,6 +163,32 @@ export const supportsAllTurnsReasoningContext = isOpenAIWireGen54Plus;
*/
export const supportsCodexReasoningSummary = isOpenAIWireGen54Plus;
+/** OpenAI proprietary reasoning families keyed off the parsed gpt version (gpt-5+). */
+const isOpenAIWireGen5Plus = memo((modelId: string): boolean => {
+ const parsed = parseOpenAIModel(bareModelId(modelId));
+ if (!parsed) return false;
+ return semverGte(parsed.version, "5");
+});
+
+/** o-series reasoning ids (`o1`, `o1-pro`, `o3`, `o3-mini`, `o4-mini`, `openai/o3`, …). */
+const O_SERIES_REASONING_RE = /(^|\/)o[134](?:[-.]|$)/i;
+
+/**
+ * OpenAI proprietary models whose serving path rejects explicit sampling
+ * parameters (`temperature`, `top_p`, `top_k`, …) with
+ * `400 Unsupported parameter: 'temperature' is not supported with this model`.
+ * Covers the o-series and the entire gpt-5+ generation — base, `mini`, `nano`,
+ * `codex*`, the `luna`/`sol`/`terra` SKUs, and the `-chat-latest` variants,
+ * since even the non-reasoning gpt-5 chat models reject sampling params (see
+ * litellm#13781). Holds regardless of which OpenAI-serving host proxies the
+ * model (official, Azure, GitHub Copilot). Version floor (not an allowlist) so
+ * 6.x inherits automatically. Issue #5606.
+ */
+export const isOpenAISamplingRestrictedModelId = memo((modelId: string): boolean => {
+ const bare = bareModelId(modelId);
+ return isOpenAIWireGen5Plus(modelId) || O_SERIES_REASONING_RE.test(bare);
+});
+
/**
* Reasoning-capable GLM coding SKUs: glm-4.5 and up on the base / `-air` /
* `-turbo` lines. Excludes the vision (`…v`) shape, the non-reasoning
diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts
index ec5d93526..ec522e550 100644
--- a/packages/catalog/src/types.ts
+++ b/packages/catalog/src/types.ts
@@ -327,6 +327,15 @@ export interface OpenAICompat {
toolStrictMode?: "all_strict" | "none";
/** Whether request shaping may send reasoning params at all. Default: auto-detected (disabled for GitHub Copilot chat-completions). */
supportsReasoningParams?: boolean;
+ /**
+ * Whether the endpoint accepts explicit sampling parameters (`temperature`,
+ * `top_p`, `top_k`, `min_p`, penalties). OpenAI proprietary reasoning models
+ * (o-series, gpt-5+) reject them with `400 Unsupported parameter:
+ * 'temperature' is not supported with this model` on every serving host
+ * (official, Azure, GitHub Copilot). When unset, auto-detected from the
+ * model id. Default: true. Issue #5606.
+ */
+ supportsSamplingParams?: boolean;
/** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */
alwaysSendMaxTokens?: boolean;
/** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */
@@ -464,6 +473,7 @@ export interface ResolvedOpenAISharedCompat {
supportsReasoningEffort: boolean;
reasoningEffortMap: Partial>;
supportsReasoningParams: boolean;
+ supportsSamplingParams: boolean;
thinkingFormat: OpenAIReasoningFormat;
reasoningDisableMode: OpenAIReasoningDisableMode;
omitReasoningEffort: boolean;
@@ -516,6 +526,7 @@ export type ResolvedOpenAICompat = ResolvedOpenAISharedCompat &
| "supportsReasoningEffort"
| "reasoningEffortMap"
| "supportsReasoningParams"
+ | "supportsSamplingParams"
| "thinkingFormat"
| "reasoningDisableMode"
| "omitReasoningEffort"
diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts
index 5c12451f6..650d87bc8 100644
--- a/packages/catalog/test/model-thinking.test.ts
+++ b/packages/catalog/test/model-thinking.test.ts
@@ -584,6 +584,31 @@ describe("model thinking derivation", () => {
expect(fable.compat.supportsSamplingParams).toBe(false);
});
+ it("bakes sampling-param rejection into OpenAI reasoning compat (#5606)", () => {
+ // GitHub Copilot Responses gpt-5.6 — the reported failing model.
+ const luna = createModel({
+ id: "gpt-5.6-luna",
+ api: "openai-responses",
+ provider: "github-copilot",
+ baseUrl: "https://api.githubcopilot.com",
+ });
+ const gpt5 = createModel({ id: "gpt-5", api: "openai-responses", provider: "openai" });
+ const gpt5Mini = createModel({ id: "gpt-5-mini", api: "openai-completions", provider: "openai" });
+ const gpt5Chat = createModel({ id: "gpt-5-chat-latest", api: "openai-responses", provider: "openai" });
+ const oThree = createModel({ id: "o3-mini", api: "openai-responses", provider: "openai" });
+ // Non-restricted OpenAI + non-OpenAI models keep sampling support.
+ const gpt4o = createModel({ id: "gpt-4o", api: "openai-responses", provider: "openai", reasoning: false });
+ const kimi = createModel({ id: "kimi-k2.6", api: "openai-completions", provider: "moonshot" });
+
+ expect(luna.compat.supportsSamplingParams).toBe(false);
+ expect(gpt5.compat.supportsSamplingParams).toBe(false);
+ expect(gpt5Mini.compat.supportsSamplingParams).toBe(false);
+ expect(gpt5Chat.compat.supportsSamplingParams).toBe(false);
+ expect(oThree.compat.supportsSamplingParams).toBe(false);
+ expect(gpt4o.compat.supportsSamplingParams).toBe(true);
+ expect(kimi.compat.supportsSamplingParams).toBe(true);
+ });
+
it("encodes effort-dial-less reasoners as thinking: undefined", () => {
const model = createModel({
id: "grok-build",