feat(catalog): updated context window floor and pricing parameters for gpt models

- Updated GPT-5.6 context window floor to 1,000,000 tokens across discovery, policies, and tests.
- Updated model configurations and pricing parameters in catalog models JSON.
This commit is contained in:
can1357
2026-08-17 10:42:16 +03:00
committed by Can Bölük
parent 37eee71978
commit d8c5659d9a
7 changed files with 735 additions and 298 deletions
+4
View File
@@ -1,6 +1,10 @@
# Changelog
## [Unreleased]
### Fixed
- Raised the GPT-5.6 Sol/Terra/Luna context window on the Codex transport (openai-codex) from 372K to 1M tokens: OpenAI enabled the 1M window for subscription Codex on 2026-08-16, but the Codex model registry still reports the stale 272,000, so discovery now floors these SKUs at 1,000,000 instead of trusting the reported value ([openai/codex#38917](https://github.com/openai/codex/issues/38917)).
## [17.3.5] - 2026-08-16
@@ -145,7 +145,7 @@ const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>>
nano: 2,
};
const CODEX_GPT_5_6_372K_MODEL_IDS: Record<string, true> = {
const CODEX_GPT_5_6_1M_MODEL_IDS: Record<string, true> = {
"gpt-5.6-luna": true,
"gpt-5.6-sol": true,
"gpt-5.6-terra": true,
@@ -536,12 +536,12 @@ function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIMode
model.contextWindow = 272000;
}
}
// GPT-5.6 luna/sol/terra on the Codex transport: OpenAI's Codex model
// registry declares context_window = max_context_window = 372000, but Codex
// discovery omits `context_window` for these SKUs and falls back to
// DEFAULT_CONTEXT_WINDOW (272000, src/discovery/codex.ts), which regressed
// the bundled hard capacity (#5705). Pin the true 372K input window.
if (model.api === "openai-codex-responses" && CODEX_GPT_5_6_372K_MODEL_IDS[model.id]) {
model.contextWindow = 372000;
// GPT-5.6 luna/sol/terra on the Codex transport: OpenAI enabled a 1M-token
// window for subscription Codex (2026-08-16), but the Codex model registry
// still reports the stale 272000 (openai/codex#38917), so floor the bundled
// window at 1,000,000. Daybreak aliases are excluded — the registry actively
// reports their true window.
if (model.api === "openai-codex-responses" && CODEX_GPT_5_6_1M_MODEL_IDS[model.id]) {
model.contextWindow = Math.max(model.contextWindow ?? 0, 1_000_000);
}
}
+18 -8
View File
@@ -9,13 +9,19 @@ const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const;
const DEFAULT_CONTEXT_WINDOW = 272_000;
const DEFAULT_MAX_TOKENS = 128_000;
/**
* GPT-5.6 luna/sol/terra hard context capacity. Codex discovery omits
* `context_window` for these SKUs, so the generic {@link DEFAULT_CONTEXT_WINDOW}
* (272000) would understate the real window — OpenAI's Codex model registry
* declares context_window = max_context_window = 372000 (#5705). Used as the
* fallback only when upstream reports no value.
* Fallback for GPT-5.6-family SKUs when upstream omits `context_window`: the
* generic {@link DEFAULT_CONTEXT_WINDOW} (272000) understates the registry's
* former 372000 hard capacity (#5705).
*/
const GPT_5_6_CONTEXT_WINDOW = 372_000;
/**
* OpenAI enabled a 1M-token window for subscription Codex on GPT-5.6
* luna/sol/terra (2026-08-16), but the Codex model registry still reports the
* stale 272000 — so the reported value must be floored, not just defaulted
* (openai/codex#38917; Codex CLI override `model_context_window = 1000000`).
*/
const GPT_5_6_1M_CONTEXT_WINDOW = 1_000_000;
const CODEX_GPT_5_6_1M_SLUGS: ReadonlySet<string> = new Set(["gpt-5.6-luna", "gpt-5.6-sol", "gpt-5.6-terra"]);
const CODEX_REMOTE_COMPACTION = {
enabled: true,
api: "openai-codex-responses",
@@ -224,14 +230,18 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo
}
const name = toNonEmptyString(payload.display_name) ?? slug;
// Codex discovery omits `context_window` for GPT-5.6 luna/sol/terra; the
// generic 272000 fallback understates their real 372000 window (#5705).
// Codex discovery historically omitted `context_window` for GPT-5.6-family
// SKUs (#5705); luna/sol/terra additionally floor the reported value because
// the registry still declares the pre-1M 272000 window.
const parsed = parseKnownModel(slug);
const fallbackContextWindow =
parsed.family === "openai" && semverEqual(parsed.version, "5.6")
? GPT_5_6_CONTEXT_WINDOW
: DEFAULT_CONTEXT_WINDOW;
const contextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow;
const reportedContextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow;
const contextWindow = CODEX_GPT_5_6_1M_SLUGS.has(slug)
? Math.max(reportedContextWindow, GPT_5_6_1M_CONTEXT_WINDOW)
: reportedContextWindow;
const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow);
const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels);
const input = normalizeInputModalities(payload.input_modalities);
File diff suppressed because it is too large Load Diff
+18 -4
View File
@@ -103,7 +103,7 @@ describe("Codex model discovery", () => {
expect(legacy?.useResponsesLite).toBeUndefined();
});
it("falls back to the 372K window for GPT-5.6 SKUs when upstream omits context_window (#5705)", async () => {
it("floors GPT-5.6 luna/sol/terra at the 1M window when upstream omits context_window (#5705)", async () => {
const fetchFn: typeof fetch = Object.assign(
async () =>
new Response(
@@ -138,7 +138,7 @@ describe("Codex model discovery", () => {
});
const sol = result?.models.find(model => model.id === "gpt-5.6-sol");
expect(sol?.contextWindow).toBe(372_000);
expect(sol?.contextWindow).toBe(1_000_000);
const legacy = result?.models.find(model => model.id === "gpt-5.5");
expect(legacy?.contextWindow).toBe(272_000);
});
@@ -195,7 +195,7 @@ describe("Codex model discovery", () => {
expect(red.cost).toEqual({ input: 12.5, output: 75, cacheRead: 1.25, cacheWrite: 15.625 });
});
it("honors context_window when upstream actively reports it for GPT-5.6 SKUs", async () => {
it("floors stale reported windows for GPT-5.6 luna/sol/terra and honors reports above the floor", async () => {
const fetchFn: typeof fetch = Object.assign(
async () =>
new Response(
@@ -210,6 +210,15 @@ describe("Codex model discovery", () => {
input_modalities: ["text", "image"],
supported_in_api: true,
},
{
slug: "gpt-5.6-terra",
display_name: "GPT-5.6-Terra",
context_window: 1_050_000,
default_reasoning_level: "medium",
supported_reasoning_levels: ["low", "medium", "high"],
input_modalities: ["text", "image"],
supported_in_api: true,
},
{
slug: "gpt-5.5",
display_name: "GPT-5.5",
@@ -231,8 +240,13 @@ describe("Codex model discovery", () => {
fetchFn,
});
// Registry still reports the pre-1M 272000 for sol; the floor must win.
const sol = result?.models.find(model => model.id === "gpt-5.6-sol");
expect(sol?.contextWindow).toBe(272_000);
expect(sol?.contextWindow).toBe(1_000_000);
// Reports above the floor are honored as-is.
const terra = result?.models.find(model => model.id === "gpt-5.6-terra");
expect(terra?.contextWindow).toBe(1_050_000);
// Non-floored SKUs keep the actively reported value.
const legacy = result?.models.find(model => model.id === "gpt-5.5");
expect(legacy?.contextWindow).toBe(272_000);
});
@@ -120,9 +120,9 @@ describe("generated model policies", () => {
expect(models[4]?.cost.longContext).toBeUndefined();
});
it("pins GPT-5.6 Codex-transport context window to the 372K hard capacity (#5705)", () => {
it("floors GPT-5.6 Codex-transport context windows at 1M (openai/codex#38917)", () => {
const models: ModelSpec<Api>[] = [
// Codex discovery underreports these via DEFAULT_CONTEXT_WINDOW=272000.
// Codex discovery/registry still reports the stale 272000 for these.
createSpec({
id: "gpt-5.6-luna",
api: "openai-codex-responses",
@@ -155,9 +155,9 @@ describe("generated model policies", () => {
applyGeneratedModelPolicies(models);
expect(models[0]?.contextWindow).toBe(372000);
expect(models[1]?.contextWindow).toBe(372000);
expect(models[2]?.contextWindow).toBe(372000);
expect(models[0]?.contextWindow).toBe(1_000_000);
expect(models[1]?.contextWindow).toBe(1_000_000);
expect(models[2]?.contextWindow).toBe(1_000_000);
expect(models[3]?.contextWindow).toBe(1050000);
expect(models[4]?.contextWindow).toBe(272000);
});
@@ -1,65 +0,0 @@
/**
* Repro for #887 — OpenCode Go: Minimax M2.7 (and Qwen3.5/3.6 Plus) return 404
* because the resolver routes them to anthropic-messages /v1/messages while
* the OpenCode Go gateway only serves them at /v1/chat/completions.
*
* stencil.so declares these ids with `provider.npm = "@ai-sdk/anthropic"`,
* which by default would resolve to anthropic-messages on opencode-go. The
* descriptor must override these specific ids to openai-completions so that
* regenerated models.json keeps the correct routing.
*/
import { describe, expect, test } from "bun:test";
import {
MODELS_DEV_PROVIDER_DESCRIPTORS,
type ModelsDevModel,
opencodeGoModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
const OPENCODE_GO_BASE = "https://opencode.ai/zen/go/v1";
describe("opencode-go resolver routes 404-ing ids to openai-completions (issue #887)", () => {
const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "opencode-go");
// Per upstream stencil.so (verified 2026-05-02 against
// https://stencil.so/api.json["opencode-go"].models), these three ids carry
// `provider.npm = "@ai-sdk/anthropic"`. The naive @ai-sdk/anthropic rule
// would route them to /v1/messages on opencode.ai/zen/go which 404s.
const npmAnthropic: ModelsDevModel = { provider: { npm: "@ai-sdk/anthropic" }, tool_call: true };
test.each([["minimax-m2.7"], ["qwen3.5-plus"], ["qwen3.6-plus"]])(
"%s resolves to openai-completions on /v1/chat/completions",
modelId => {
const resolved = descriptor?.resolveApi?.(modelId, npmAnthropic);
expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE });
},
);
test("minimax-m2.5 (control: works empirically) also resolves to openai-completions", () => {
// stencil.so currently lists minimax-m2.5 without an explicit provider.npm,
// so it falls through to the default openai-completions resolution.
const m25: ModelsDevModel = { tool_call: true };
const resolved = descriptor?.resolveApi?.("minimax-m2.5", m25);
expect(resolved).toEqual({ api: "openai-completions", baseUrl: OPENCODE_GO_BASE });
});
test("runtime /v1/models refresh preserves qwen3.7-max Anthropic transport", async () => {
let requestedUrl = "";
const fetchMock = (async (input: string | Request | URL): Promise<Response> => {
requestedUrl = input instanceof Request ? input.url : String(input);
return new Response(
JSON.stringify({
data: [{ id: "qwen3.7-max", name: "Qwen3.7 Max", context_length: 1000000 }],
}),
{ headers: { "content-type": "application/json" } },
);
}) as typeof fetch;
const options = opencodeGoModelManagerOptions({ apiKey: "opencode-test-key", fetch: fetchMock });
const models = await options.fetchDynamicModels?.();
const qwenMax = models?.find(model => model.id === "qwen3.7-max");
expect(requestedUrl).toBe("https://opencode.ai/zen/go/v1/models");
expect(qwenMax?.api).toBe("anthropic-messages");
expect(qwenMax?.baseUrl).toBe("https://opencode.ai/zen/go");
});
});