From 930bb33f4a74acee3f81344fcf15d2b6e9ee9635 Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 22 Jul 2026 04:03:25 +0000 Subject: [PATCH] fix(catalog): floored gpt-5.6 codex context window at 372k Codex discovery under-reports the gpt-5.6 sol/terra/luna window: some accounts omit `context_window`, others actively return 272000. The #5707 `?? fallback` only fired on absence, so an actively-reported 272000 passed through and `preferDiscoveryLimit` overwrote the bundled 372K pin at runtime, dropping compaction from 279000 to 204000 tokens. Treat GPT_5_6_CONTEXT_WINDOW as a floor for these SKUs via Math.max so neither omission nor active under-report regresses the real capacity; other models keep honoring the reported value. Corrected the stale "omits" comments in codex.ts and generated-policies.ts. Fixes #6259 --- packages/catalog/CHANGELOG.md | 4 ++ .../catalog/scripts/generated-policies.ts | 6 +-- packages/catalog/src/discovery/codex.ts | 28 ++++++------ packages/catalog/test/codex-discovery.test.ts | 43 +++++++++++++++++++ 4 files changed, 66 insertions(+), 15 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 50018ad2e..9b41ce4e7 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed GPT-5.6 Codex SKUs (`gpt-5.6-{sol,terra,luna}`) losing ~75K of usable context when the Codex discovery endpoint actively reports `context_window: 272000`: discovery now floors these SKUs at the 372K hard capacity instead of only substituting it when the field is absent, so the runtime dynamic value no longer overwrites the bundled pin ([#6259](https://github.com/can1357/oh-my-pi/issues/6259)). + ## [17.0.6] - 2026-07-20 ### Added diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index cf8bce014..7ebf256ec 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -364,9 +364,9 @@ function applyOpenAICatalogPolicy(model: ModelSpec, parsedModel: OpenAIMode } // GPT-5.6 luna/sol/terra on the Codex transport: OpenAI's Codex model // registry declares context_window = max_context_window = 372000, but Codex - // discovery omits `context_window` for these SKUs and falls back to - // DEFAULT_CONTEXT_WINDOW (272000, src/discovery/codex.ts), which regressed - // the bundled hard capacity (#5705). Pin the true 372K input window. + // discovery under-reports it — omitting the field for some accounts and + // actively returning 272000 for others (#5705, #6259). Pin the true 372K + // input window on the bundled catalog; discovery enforces the same floor. if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.6")) { model.contextWindow = 372000; } diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index 5dcd6ea8c..88f98de1f 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -8,11 +8,11 @@ const DEFAULT_MODEL_LIST_PATHS = ["/codex/models", "/models"] as const; const DEFAULT_CONTEXT_WINDOW = 272_000; const DEFAULT_MAX_TOKENS = 128_000; /** - * GPT-5.6 luna/sol/terra hard context capacity. Codex discovery omits - * `context_window` for these SKUs, so the generic {@link DEFAULT_CONTEXT_WINDOW} - * (272000) would understate the real window — OpenAI's Codex model registry - * declares context_window = max_context_window = 372000 (#5705). Used as the - * fallback only when upstream reports no value. + * GPT-5.6 luna/sol/terra hard context capacity. OpenAI's Codex model registry + * declares context_window = max_context_window = 372000 (#5705), but Codex + * discovery under-reports it — omitting the field for some accounts and + * actively returning 272000 for others (#6259). Applied as a floor for these + * SKUs so the reported/absent value never regresses the real window. */ const GPT_5_6_CONTEXT_WINDOW = 372_000; const CODEX_REMOTE_COMPACTION = { @@ -223,14 +223,18 @@ function normalizeCodexModelEntry(entry: unknown, baseUrl: string): NormalizedCo } const name = toNonEmptyString(payload.display_name) ?? slug; - // Codex discovery omits `context_window` for GPT-5.6 luna/sol/terra; the - // generic 272000 fallback understates their real 372000 window (#5705). + // GPT-5.6 luna/sol/terra have a 372000 hard window, but Codex discovery + // under-reports it: for some accounts the field is omitted, for others it is + // actively returned as 272000 (#6259). Treat GPT_5_6_CONTEXT_WINDOW as a + // floor for these SKUs so neither the omission nor the active under-report + // regresses the real capacity; other models honor the reported value with + // the generic 272000 fallback. const parsed = parseKnownModel(slug); - const fallbackContextWindow = - parsed.family === "openai" && semverEqual(parsed.version, "5.6") - ? GPT_5_6_CONTEXT_WINDOW - : DEFAULT_CONTEXT_WINDOW; - const contextWindow = toPositiveInt(payload.context_window) ?? fallbackContextWindow; + const isGpt56 = parsed.family === "openai" && semverEqual(parsed.version, "5.6"); + const reportedContextWindow = toPositiveInt(payload.context_window); + const contextWindow = isGpt56 + ? Math.max(GPT_5_6_CONTEXT_WINDOW, reportedContextWindow ?? 0) + : (reportedContextWindow ?? DEFAULT_CONTEXT_WINDOW); const maxTokens = Math.min(DEFAULT_MAX_TOKENS, contextWindow); const reasoning = supportsReasoning(payload.default_reasoning_level, payload.supported_reasoning_levels); const input = normalizeInputModalities(payload.input_modalities); diff --git a/packages/catalog/test/codex-discovery.test.ts b/packages/catalog/test/codex-discovery.test.ts index bd1bb3346..a16f9b3b6 100644 --- a/packages/catalog/test/codex-discovery.test.ts +++ b/packages/catalog/test/codex-discovery.test.ts @@ -141,6 +141,49 @@ describe("Codex model discovery", () => { expect(legacy?.contextWindow).toBe(272_000); }); + it("floors GPT-5.6 SKUs to 372K when upstream actively reports 272000 (#6259)", async () => { + const fetchFn: typeof fetch = Object.assign( + async () => + new Response( + JSON.stringify({ + models: [ + { + slug: "gpt-5.6-sol", + display_name: "GPT-5.6-Sol", + context_window: 272_000, + default_reasoning_level: "medium", + supported_reasoning_levels: ["low", "medium", "high"], + input_modalities: ["text", "image"], + supported_in_api: true, + }, + { + slug: "gpt-5.5", + display_name: "GPT-5.5", + context_window: 272_000, + default_reasoning_level: "high", + supported_reasoning_levels: ["low", "high"], + input_modalities: ["text"], + supported_in_api: true, + }, + ], + }), + ), + { preconnect() {} }, + ); + const result = await fetchCodexModels({ + accessToken: "test-token", + baseUrl: "https://codex.example/backend-api", + clientVersion: "0.144.1", + fetchFn, + }); + + const sol = result?.models.find(model => model.id === "gpt-5.6-sol"); + expect(sol?.contextWindow).toBe(372_000); + // Non-5.6 SKUs still honor the reported value verbatim. + const legacy = result?.models.find(model => model.id === "gpt-5.5"); + expect(legacy?.contextWindow).toBe(272_000); + }); + it("uses the discovered account catalog as authoritative", async () => { const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-codex-authoritative-")); const staticOnlyModel: ModelSpec<"openai-codex-responses"> = {