From e7e280a6fbe01e276f857f5af98d029393b6d249 Mon Sep 17 00:00:00 2001 From: pickpocket Date: Wed, 12 Aug 2026 00:28:54 -0400 Subject: [PATCH] fix(catalog): complete GPT-5.6 off and pricing support (cherry picked from commit fc034d61ff69e8ad1870f674c72ff3f761741862) --- packages/ai/CHANGELOG.md | 2 +- packages/ai/test/models-cost.test.ts | 38 ++++++++ .../ai/test/openai-daybreak-effort.test.ts | 26 ++++-- packages/catalog/CHANGELOG.md | 1 + .../catalog/scripts/generated-policies.ts | 38 ++++++-- packages/catalog/src/model-manager.ts | 15 +++- packages/catalog/src/models.json | 90 +++++++++++++++++-- packages/catalog/src/models.ts | 27 ++++-- .../src/provider-models/openai-compat.ts | 31 ++++++- packages/catalog/src/types.ts | 29 ++++-- packages/catalog/test/build.test.ts | 46 ++++++++++ .../catalog/test/generated-policies.test.ts | 29 ++++++ packages/catalog/test/openai-daybreak.test.ts | 40 ++++++++- .../coding-agent/src/config/model-patch.ts | 2 + .../coding-agent/test/model-registry.test.ts | 17 ++-- 15 files changed, 381 insertions(+), 50 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 33c9c31da..f1e346513 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -51,7 +51,7 @@ - Fixed the AWS credential resolver ignoring `role_arn` profiles: shared-config role chaining (`source_profile` recursion, `web_identity_token_file`, `credential_source`) now resolves via STS `AssumeRole`/`AssumeRoleWithWebIdentity`, honoring `role_session_name`/`duration_seconds`/`external_id`, so Bedrock is detected on EKS/IRSA and multi-account setups instead of reporting "No models available" ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)). - Fixed Bedrock availability being under-detected on Nitro/EKS hosts: the EC2 metadata probe now recognizes Nitro DMI markers (`board_asset_tag` instance ids, `Amazon EC2` vendor fields) in addition to the Xen `ec2` UUID prefix ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)). - Fixed DeepSeek Responses targets (opencode-go) rejecting a thinking-mode continuation with `400 The reasoning_text in the thinking mode must be passed back to the API` after a prewalk hand-off plus mid-run compaction: the Responses input builder re-encoded replayed assistant turns without a reasoning item, so the request enabled reasoning but shipped no `reasoning_text`. The encoder now synthesizes a `reasoning_text` reasoning item for every replayed assistant turn when the target requires reasoning replay in thinking mode (`requiresReasoningContentForAllAssistantTurns` / `requiresReasoningContentForToolCalls`), mirroring the chat-completions `reasoning_content` safety net ([#8248](https://github.com/can1357/oh-my-pi/issues/8248)). -- Fixed OpenAI Daybreak `off` thinking requests serializing as `low`; models with explicit wire-level off support now send `reasoning.effort: "none"`. +- Fixed OpenAI GPT-5.6 and Daybreak `off` thinking requests serializing as `low`; every first-party alias with explicit wire-level off support now sends `reasoning.effort: "none"`. ## [17.2.12] - 2026-08-08 diff --git a/packages/ai/test/models-cost.test.ts b/packages/ai/test/models-cost.test.ts index a0c78e24c..a6c6bd532 100644 --- a/packages/ai/test/models-cost.test.ts +++ b/packages/ai/test/models-cost.test.ts @@ -196,4 +196,42 @@ describe("calculateCost", () => { expect(usage.cost.total).toBeCloseTo(0.01005, 8); }); + + it("keeps Daybreak Blue at short-context rates through 272K prompt tokens", () => { + const model = getBundledModel("openai", "daybreak-blue-latest"); + const usage: Usage = { + input: 270_000, + output: 1_000, + cacheRead: 1_000, + cacheWrite: 1_000, + totalTokens: 273_000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; + + calculateCost(model, usage); + + expect(usage.cost.input).toBeCloseTo(1.35, 12); + expect(usage.cost.output).toBeCloseTo(0.03, 12); + expect(usage.cost.cacheRead).toBeCloseTo(0.0005, 12); + expect(usage.cost.cacheWrite).toBeCloseTo(0.00625, 12); + }); + + it("prices the full Daybreak Blue request at long-context rates above 272K prompt tokens", () => { + const model = getBundledModel("openai", "daybreak-blue-latest"); + const usage: Usage = { + input: 270_001, + output: 1_000, + cacheRead: 1_000, + cacheWrite: 1_000, + totalTokens: 273_001, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; + + calculateCost(model, usage); + + expect(usage.cost.input).toBeCloseTo(2.70001, 12); + expect(usage.cost.output).toBeCloseTo(0.045, 12); + expect(usage.cost.cacheRead).toBeCloseTo(0.001, 12); + expect(usage.cost.cacheWrite).toBeCloseTo(0.0125, 12); + }); }); diff --git a/packages/ai/test/openai-daybreak-effort.test.ts b/packages/ai/test/openai-daybreak-effort.test.ts index dfb6cec83..98645022e 100644 --- a/packages/ai/test/openai-daybreak-effort.test.ts +++ b/packages/ai/test/openai-daybreak-effort.test.ts @@ -4,24 +4,36 @@ import type { Context } from "@oh-my-pi/pi-ai/types"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; -const DAYBREAK_MODEL_IDS = ["daybreak-blue-latest", "daybreak-red-latest", "gpt-5.6-cyber", "gpt-5.6-sol"]; -const DAYBREAK_EFFORTS = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]; +const GPT_56_MODEL_IDS = [ + "daybreak-blue-latest", + "daybreak-red-latest", + "gpt-5.6", + "gpt-5.6-cyber", + "gpt-5.6-luna", + "gpt-5.6-luna-pro", + "gpt-5.6-sol", + "gpt-5.6-sol-pro", + "gpt-5.6-terra", + "gpt-5.6-terra-pro", +]; +const GPT_56_EFFORTS = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]; const CONTEXT: Context = { messages: [{ role: "user", content: "hello", timestamp: 0 }], }; -describe("OpenAI Daybreak Responses reasoning payload", () => { - for (const id of DAYBREAK_MODEL_IDS) { +describe("OpenAI GPT-5.6 Responses reasoning payload", () => { + for (const id of GPT_56_MODEL_IDS) { test(`${id} serializes off and every supported thinking level`, () => { const model = getBundledModel<"openai-responses">("openai", id); if (!model) throw new Error(`openai/${id} must be in bundled models.json`); + const mode = model.reasoningMode ? { mode: model.reasoningMode } : {}; const disabled = buildParams(model, CONTEXT, { disableReasoning: true }, undefined); - expect(disabled.params.reasoning).toEqual({ effort: "none" }); + expect(disabled.params.reasoning).toEqual({ effort: "none", ...mode }); - for (const effort of DAYBREAK_EFFORTS) { + for (const effort of GPT_56_EFFORTS) { const enabled = buildParams(model, CONTEXT, { reasoning: effort }, undefined); - expect(enabled.params.reasoning).toEqual({ effort, summary: "auto" }); + expect(enabled.params.reasoning).toEqual({ effort, summary: "auto", ...mode }); } }); } diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index f959a4d68..72222ad69 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -13,6 +13,7 @@ ### Fixed - Fixed Codex-discovered `gpt-daybreak-*` aliases being treated as unknown models, restoring the GPT-5.6 `low`/`medium`/`high`/`xhigh`/`max` effort ladder and its 372K fallback only when the Codex registry omits `context_window`. +- Fixed first-party OpenAI GPT-5.6 aliases to preserve wire-level `off` through generated pro aliases and to price requests above 272K input at each SKU's documented long-context rates. ## [17.2.15] - 2026-08-12 diff --git a/packages/catalog/scripts/generated-policies.ts b/packages/catalog/scripts/generated-policies.ts index 286fa7101..2b97e0dfb 100644 --- a/packages/catalog/scripts/generated-policies.ts +++ b/packages/catalog/scripts/generated-policies.ts @@ -21,9 +21,10 @@ import { resolveModelThinking } from "../src/model-thinking"; import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../src/provider-models/ollama"; import { ALIBABA_TOKEN_PLAN_STATIC_MODELS, + OPENAI_GPT_56_LONG_CONTEXT_COSTS, resolveWaferServerlessThinkingFormat, } from "../src/provider-models/openai-compat"; -import type { Api, Model, ModelSpec } from "../src/types"; +import type { Api, LongContextTokenCost, Model, ModelSpec } from "../src/types"; import { isVariantCollapsedSpec } from "../src/variant-collapse"; import { buildCanonicalModelIndex, buildCanonicalReferenceData } from "./equivalence"; @@ -149,13 +150,33 @@ const CODEX_GPT_5_6_372K_MODEL_IDS: Record = { "gpt-5.6-terra": true, }; -const OPENAI_DAYBREAK_REASONING_MODEL_IDS: Record = { +const OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID: Readonly> = { + "daybreak-blue-latest": OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol, + "gpt-5.6": OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol, + "gpt-5.6-luna": OPENAI_GPT_56_LONG_CONTEXT_COSTS.luna, + "gpt-5.6-sol": OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol, + "gpt-5.6-terra": OPENAI_GPT_56_LONG_CONTEXT_COSTS.terra, +}; + +const OPENAI_NONE_EFFORT_MODEL_IDS: Record = { "daybreak-blue-latest": true, "daybreak-red-latest": true, + "gpt-5.6": true, "gpt-5.6-cyber": true, + "gpt-5.6-luna": true, "gpt-5.6-sol": true, + "gpt-5.6-terra": true, }; +function modelOrRequestIdValue( + model: Pick, "id" | "requestModelId">, + values: Readonly>, +): T | undefined { + const direct = values[bareModelId(model.id)]; + if (direct !== undefined) return direct; + return model.requestModelId === undefined ? undefined : values[bareModelId(model.requestModelId)]; +} + const COPILOT_GENERATED_LIMITS: Record = { "claude-opus-4.6": { contextWindow: 168000, maxTokens: 32000 }, "gpt-5.2": { contextWindow: 272000, maxTokens: 128000 }, @@ -477,13 +498,16 @@ function inferGeneratedApplyPatchToolType( } function applyOpenAICatalogPolicy(model: ModelSpec, parsedModel: OpenAIModel): void { - if ( - model.provider === "openai" && - model.api === "openai-responses" && - OPENAI_DAYBREAK_REASONING_MODEL_IDS[model.id] - ) { + const isFirstPartyResponses = model.provider === "openai" && model.api === "openai-responses"; + if (isFirstPartyResponses && modelOrRequestIdValue(model, OPENAI_NONE_EFFORT_MODEL_IDS)) { model.compat = { ...(model.compat ?? {}), reasoningDisableMode: "none-effort" }; } + const longContextCost = isFirstPartyResponses + ? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID) + : undefined; + if (longContextCost) { + model.cost = { ...model.cost, longContext: longContextCost }; + } // Codex models: 400K figure includes output budget; input window is 272K. if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") { diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index 185477c68..48bcd3a8b 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -1,7 +1,7 @@ import { buildModel } from "./build"; import { readModelCache, writeModelCache } from "./model-cache"; import { type GeneratedProvider, getBundledModels } from "./models"; -import type { Api, Model, ModelSpec, Provider } from "./types"; +import type { Api, Model, ModelCost, ModelSpec, Provider, TokenCost } from "./types"; import { isRecord } from "./utils"; import { collapseBuiltModelVariants } from "./variant-collapse"; @@ -510,6 +510,7 @@ function mergeDynamicModel(existingModel: Model, dynamic const reasoning = dynamicReasoningAuthoritative ? dynamicModel.reasoning : existingModel.reasoning || dynamicModel.reasoning; + const longContextCost = dynamicModel.cost.longContext ?? existingModel.cost.longContext; // Re-build from spec stage: sparse compat comes from `compatConfig` (the // verbatim override vocabulary), never the resolved `compat` record. return buildModel({ @@ -523,6 +524,7 @@ function mergeDynamicModel(existingModel: Model, dynamic output: preferDiscoveryCost(dynamicModel.cost.output, existingModel.cost.output), cacheRead: preferDiscoveryCost(dynamicModel.cost.cacheRead, existingModel.cost.cacheRead), cacheWrite: preferDiscoveryCost(dynamicModel.cost.cacheWrite, existingModel.cost.cacheWrite), + ...(longContextCost ? { longContext: longContextCost } : {}), }, contextWindow: preferDiscoveryLimit(dynamicModel.contextWindow, existingModel.contextWindow), maxTokens: preferDiscoveryLimit(dynamicModel.maxTokens, existingModel.maxTokens), @@ -640,7 +642,7 @@ function isModelInputArray(value: unknown): value is ("text" | "image")[] { return true; } -function isModelCost(value: unknown): value is Model["cost"] { +function isTokenCost(value: unknown): value is TokenCost { if (!isRecord(value)) { return false; } @@ -670,3 +672,12 @@ function isModelCost(value: unknown): value is Model["cost"] { } return true; } + +function isModelCost(value: unknown): value is ModelCost { + if (!isTokenCost(value)) return false; + const longContext = (value as TokenCost & { longContext?: unknown }).longContext; + if (longContext === undefined) return true; + if (!isTokenCost(longContext) || !isRecord(longContext)) return false; + const threshold = longContext.inputThreshold; + return typeof threshold === "number" && threshold > 0 && threshold < Infinity; +} diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 794053b38..3fa594af8 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -72354,7 +72354,14 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 6.25 + "cacheWrite": 6.25, + "longContext": { + "inputThreshold": 272000, + "input": 10, + "output": 45, + "cacheRead": 1, + "cacheWrite": 12.5 + } }, "contextWindow": 1050000, "maxTokens": 128000, @@ -72368,6 +72375,9 @@ "xhigh", "max" ] + }, + "compat": { + "reasoningDisableMode": "none-effort" } }, "gpt-5.6-luna": { @@ -72385,7 +72395,14 @@ "input": 0.2, "output": 1.2, "cacheRead": 0.02, - "cacheWrite": 0.25 + "cacheWrite": 0.25, + "longContext": { + "inputThreshold": 272000, + "input": 0.4, + "output": 1.8, + "cacheRead": 0.04, + "cacheWrite": 0.5 + } }, "contextWindow": 1050000, "maxTokens": 128000, @@ -72399,6 +72416,9 @@ "xhigh", "max" ] + }, + "compat": { + "reasoningDisableMode": "none-effort" } }, "gpt-5.6-luna-pro": { @@ -72416,7 +72436,14 @@ "input": 0.2, "output": 1.2, "cacheRead": 0.02, - "cacheWrite": 0.25 + "cacheWrite": 0.25, + "longContext": { + "inputThreshold": 272000, + "input": 0.4, + "output": 1.8, + "cacheRead": 0.04, + "cacheWrite": 0.5 + } }, "contextWindow": 1050000, "maxTokens": 128000, @@ -72432,6 +72459,9 @@ "xhigh", "max" ] + }, + "compat": { + "reasoningDisableMode": "none-effort" } }, "gpt-5.6-sol": { @@ -72449,7 +72479,14 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 6.25 + "cacheWrite": 6.25, + "longContext": { + "inputThreshold": 272000, + "input": 10, + "output": 45, + "cacheRead": 1, + "cacheWrite": 12.5 + } }, "contextWindow": 1050000, "maxTokens": 128000, @@ -72483,7 +72520,14 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 6.25 + "cacheWrite": 6.25, + "longContext": { + "inputThreshold": 272000, + "input": 10, + "output": 45, + "cacheRead": 1, + "cacheWrite": 12.5 + } }, "contextWindow": 1050000, "maxTokens": 128000, @@ -72499,6 +72543,9 @@ "xhigh", "max" ] + }, + "compat": { + "reasoningDisableMode": "none-effort" } }, "gpt-5.6-terra": { @@ -72516,7 +72563,14 @@ "input": 2, "output": 12, "cacheRead": 0.2, - "cacheWrite": 2.5 + "cacheWrite": 2.5, + "longContext": { + "inputThreshold": 272000, + "input": 4, + "output": 18, + "cacheRead": 0.4, + "cacheWrite": 5 + } }, "contextWindow": 1050000, "maxTokens": 128000, @@ -72530,6 +72584,9 @@ "xhigh", "max" ] + }, + "compat": { + "reasoningDisableMode": "none-effort" } }, "gpt-5.6-terra-pro": { @@ -72547,7 +72604,14 @@ "input": 2, "output": 12, "cacheRead": 0.2, - "cacheWrite": 2.5 + "cacheWrite": 2.5, + "longContext": { + "inputThreshold": 272000, + "input": 4, + "output": 18, + "cacheRead": 0.4, + "cacheWrite": 5 + } }, "contextWindow": 1050000, "maxTokens": 128000, @@ -72563,6 +72627,9 @@ "xhigh", "max" ] + }, + "compat": { + "reasoningDisableMode": "none-effort" } }, "gpt-realtime-2.1": { @@ -72857,7 +72924,14 @@ "input": 5, "output": 30, "cacheRead": 0.5, - "cacheWrite": 6.25 + "cacheWrite": 6.25, + "longContext": { + "inputThreshold": 272000, + "input": 10, + "output": 45, + "cacheRead": 1, + "cacheWrite": 12.5 + } }, "contextWindow": 1050000, "maxTokens": 128000, diff --git a/packages/catalog/src/models.ts b/packages/catalog/src/models.ts index 0c43aa1d6..dd0f498f7 100644 --- a/packages/catalog/src/models.ts +++ b/packages/catalog/src/models.ts @@ -1,6 +1,6 @@ import { buildModel } from "./build"; import MODELS from "./models.json" with { type: "json" }; -import type { Api, KnownProvider, Model, ModelSpec, Usage } from "./types"; +import type { Api, KnownProvider, Model, ModelSpec, TokenCost, Usage } from "./types"; /** * Static bundled model registry loaded from `models.json`. @@ -42,13 +42,22 @@ export function getBundledModels(provider: GeneratedProvider): Model[] { const models = getProviderModels(provider); return models ? (Array.from(models.values()) as Model[]) : []; } +function resolveTokenCost(cost: Model["cost"], usage: Usage): TokenCost { + const longContext = cost.longContext; + if (!longContext) return cost; + const orchestration = usage.orchestration; + const promptInputTokens = + usage.input + usage.cacheRead + usage.cacheWrite + (orchestration?.input ?? 0) + (orchestration?.cacheRead ?? 0); + return promptInputTokens > longContext.inputThreshold ? longContext : cost; +} export function calculateCost(model: Model, usage: Usage): Usage["cost"] { + const rates = resolveTokenCost(model.cost, usage); const orchestration = usage.orchestration; - usage.cost.input = (model.cost.input / 1000000) * (usage.input + (orchestration?.input ?? 0)); - usage.cost.output = (model.cost.output / 1000000) * (usage.output + (orchestration?.output ?? 0)); - usage.cost.cacheRead = (model.cost.cacheRead / 1000000) * (usage.cacheRead + (orchestration?.cacheRead ?? 0)); - usage.cost.cacheWrite = cacheWriteCost(model, usage); + usage.cost.input = (rates.input / 1000000) * (usage.input + (orchestration?.input ?? 0)); + usage.cost.output = (rates.output / 1000000) * (usage.output + (orchestration?.output ?? 0)); + usage.cost.cacheRead = (rates.cacheRead / 1000000) * (usage.cacheRead + (orchestration?.cacheRead ?? 0)); + usage.cost.cacheWrite = cacheWriteCost(rates, usage); usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite; return usage.cost; } @@ -56,7 +65,7 @@ export function calculateCost(model: Model, usage: Usage /** * Price cache-write tokens, honoring the TTL breakdown when the provider reports one. * - * `model.cost.cacheWrite` is the 5-minute write rate (Anthropic bills 5m writes at + * `rates.cacheWrite` is the 5-minute write rate (Anthropic bills 5m writes at * 1.25x base input). When `usage.cttl` is present the write mixes 5m and 1h * breakpoints — omp defaults to 1h retention on first-party Anthropic, and 1h writes * bill at 2x base input — so each component is priced at its own rate instead of the @@ -70,14 +79,14 @@ export function calculateCost(model: Model, usage: Usage * so any unattributed remainder is priced at the flat rate instead of being dropped: * a partial or stale breakdown must never make write tokens free. */ -function cacheWriteCost(model: Model, usage: Usage): number { - const rate5m = model.cost.cacheWrite / 1000000; +function cacheWriteCost(rates: TokenCost, usage: Usage): number { + const rate5m = rates.cacheWrite / 1000000; const cttl = usage.cttl; if (!cttl) return rate5m * usage.cacheWrite; const fiveMinute = cttl.ephemeral5m ?? 0; const oneHour = cttl.ephemeral1h ?? 0; const residual = Math.max(0, usage.cacheWrite - fiveMinute - oneHour); - return rate5m * (fiveMinute + residual) + ((model.cost.input * 2) / 1000000) * oneHour; + return rate5m * (fiveMinute + residual) + ((rates.input * 2) / 1000000) * oneHour; } /** diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index f25f8f2b7..76ee02bd0 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -867,7 +867,36 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode // --------------------------------------------------------------------------- const OPENAI_API_BASE_URL = "https://api.openai.com/v1"; -const OPENAI_GPT_56_SOL_STANDARD_COST = { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 6.25 } as const; +export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = { + luna: { + inputThreshold: 272_000, + input: 0.4, + output: 1.8, + cacheRead: 0.04, + cacheWrite: 0.5, + }, + sol: { + inputThreshold: 272_000, + input: 10, + output: 45, + cacheRead: 1, + cacheWrite: 12.5, + }, + terra: { + inputThreshold: 272_000, + input: 4, + output: 18, + cacheRead: 0.4, + cacheWrite: 5, + }, +} as const; +const OPENAI_GPT_56_SOL_STANDARD_COST = { + input: 5, + output: 30, + cacheRead: 0.5, + cacheWrite: 6.25, + longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol, +} as const; const OPENAI_GPT_56_CYBER_STANDARD_COST = { input: 12.5, output: 75, diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 151a67723..8dfa0dcdd 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -820,6 +820,28 @@ export interface RemoteCompactionConfig { model?: string; } +/** Per-million-token rates for one model pricing tier. */ +export interface TokenCost { + input: number; + output: number; + cacheRead: number; + cacheWrite: number; +} + +/** + * Rates applied to the full request when its prompt exceeds `inputThreshold`. + * Prompt input is the sum of uncached, cached-read, cache-write, and + * provider-orchestration input tokens. + */ +export interface LongContextTokenCost extends TokenCost { + inputThreshold: number; +} + +/** Base token rates plus an optional long-context tier. */ +export interface ModelCost extends TokenCost { + longContext?: LongContextTokenCost; +} + // Model interface for the unified model system export interface Model { id: string; @@ -865,12 +887,7 @@ export interface Model { gitlabDuoWorkflowRootNamespaceId?: string; /** Cursor `max_mode` request flag returned by `GetUsableModels` for premium models that require max mode. */ cursorMaxMode?: boolean; - cost: { - input: number; // $/million tokens - output: number; // $/million tokens - cacheRead: number; // $/million tokens - cacheWrite: number; // $/million tokens - }; + cost: ModelCost; /** Premium Copilot requests charged per user-initiated request (defaults to 1). */ premiumMultiplier?: number; contextWindow: number | null; diff --git a/packages/catalog/test/build.test.ts b/packages/catalog/test/build.test.ts index 28882ae7a..cf9a672d2 100644 --- a/packages/catalog/test/build.test.ts +++ b/packages/catalog/test/build.test.ts @@ -763,6 +763,52 @@ describe("model cache spec round trip", () => { } }); + it("preserves static long-context pricing through dynamic refresh and cache restore", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-tiered-cost-")); + const dbPath = path.join(tempDir, "models.db"); + const staticModel = completionsSpec({ + id: "tiered-model", + provider: "tiered-cost-test", + cost: { + input: 1, + output: 2, + cacheRead: 0.1, + cacheWrite: 1.25, + longContext: { + inputThreshold: 272_000, + input: 2, + output: 3, + cacheRead: 0.2, + cacheWrite: 2.5, + }, + }, + }); + const dynamicModel = completionsSpec({ + ...staticModel, + cost: { input: 3, output: 4, cacheRead: 0.3, cacheWrite: 3.75 }, + }); + const options = { + providerId: "tiered-cost-test", + staticModels: [staticModel], + cacheDbPath: dbPath, + }; + try { + const online = await resolveProviderModels<"openai-completions">( + { ...options, fetchDynamicModels: async () => [dynamicModel] }, + "online", + ); + expect(online.models[0]?.cost).toEqual({ + ...dynamicModel.cost, + longContext: staticModel.cost.longContext, + }); + + const offline = await resolveProviderModels<"openai-completions">(options, "offline"); + expect(offline.models[0]?.cost.longContext).toEqual(staticModel.cost.longContext); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); + it("invalidates schema-v10 rows that predate computer-use capability provenance", async () => { const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-legacy-computer-cache-")); const dbPath = path.join(tempDir, "models.db"); diff --git a/packages/catalog/test/generated-policies.test.ts b/packages/catalog/test/generated-policies.test.ts index 1ac4c78c0..3c88063ba 100644 --- a/packages/catalog/test/generated-policies.test.ts +++ b/packages/catalog/test/generated-policies.test.ts @@ -91,6 +91,35 @@ describe("generated model policies", () => { expect(models[3]?.priority).toBe(1); }); + it("applies GPT-5.6 off and long-context pricing through request-model aliases", () => { + const models: ModelSpec[] = [ + createSpec({ id: "gpt-5.6", api: "openai-responses", provider: "openai" }), + createSpec({ id: "gpt-5.6-luna", api: "openai-responses", provider: "openai" }), + { + ...createSpec({ id: "gpt-5.6-sol-pro", api: "openai-responses", provider: "openai" }), + requestModelId: "gpt-5.6-sol", + }, + { + ...createSpec({ id: "gpt-5.6-terra-pro", api: "openai-responses", provider: "openai" }), + requestModelId: "gpt-5.6-terra", + }, + createSpec({ id: "gpt-5.6", api: "openai-responses", provider: "openrouter" }), + ]; + + applyGeneratedModelPolicies(models); + + for (const model of models.slice(0, 4)) { + expect(model.compat).toMatchObject({ reasoningDisableMode: "none-effort" }); + expect(model.cost.longContext?.inputThreshold).toBe(272_000); + } + expect(models[0]?.cost.longContext).toMatchObject({ input: 10, output: 45 }); + expect(models[1]?.cost.longContext).toMatchObject({ input: 0.4, output: 1.8 }); + expect(models[2]?.cost.longContext).toMatchObject({ input: 10, output: 45 }); + expect(models[3]?.cost.longContext).toMatchObject({ input: 4, output: 18 }); + expect(models[4]?.compat).toBeUndefined(); + expect(models[4]?.cost.longContext).toBeUndefined(); + }); + it("pins GPT-5.6 Codex-transport context window to the 372K hard capacity (#5705)", () => { const models: ModelSpec[] = [ // Codex discovery underreports these via DEFAULT_CONTEXT_WINDOW=272000. diff --git a/packages/catalog/test/openai-daybreak.test.ts b/packages/catalog/test/openai-daybreak.test.ts index 15e4da662..d2d1e2296 100644 --- a/packages/catalog/test/openai-daybreak.test.ts +++ b/packages/catalog/test/openai-daybreak.test.ts @@ -2,19 +2,32 @@ import { describe, expect, test } from "bun:test"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; import type { Api, ModelSpec } from "@oh-my-pi/pi-catalog/types"; import { applyGeneratedModelPolicies } from "../scripts/generated-policies"; const DAYBREAK_EFFORTS = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max]; -describe("OpenAI Daybreak models", () => { +describe("OpenAI Daybreak and GPT-5.6 models", () => { test("curates the documented aliases and Cyber snapshot with standard API pricing", () => { const byId = Object.fromEntries(OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS.map(model => [model.id, model])); expect(Object.keys(byId)).toEqual(["daybreak-blue-latest", "daybreak-red-latest", "gpt-5.6-cyber"]); expect(byId["daybreak-blue-latest"]).toMatchObject({ name: "Daybreak Blue", - cost: { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 6.25 }, + cost: { + input: 5, + output: 30, + cacheRead: 0.5, + cacheWrite: 6.25, + longContext: { + inputThreshold: 272_000, + input: 10, + output: 45, + cacheRead: 1, + cacheWrite: 12.5, + }, + }, contextWindow: 1_050_000, maxTokens: 128_000, }); @@ -27,6 +40,29 @@ describe("OpenAI Daybreak models", () => { } }); + test("bakes off support and long-context pricing onto every first-party GPT-5.6 alias", () => { + const longContextCosts = { + "daybreak-blue-latest": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 }, + "gpt-5.6": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 }, + "gpt-5.6-luna": { input: 0.4, output: 1.8, cacheRead: 0.04, cacheWrite: 0.5 }, + "gpt-5.6-luna-pro": { input: 0.4, output: 1.8, cacheRead: 0.04, cacheWrite: 0.5 }, + "gpt-5.6-sol": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 }, + "gpt-5.6-sol-pro": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 }, + "gpt-5.6-terra": { input: 4, output: 18, cacheRead: 0.4, cacheWrite: 5 }, + "gpt-5.6-terra-pro": { input: 4, output: 18, cacheRead: 0.4, cacheWrite: 5 }, + } as const; + for (const [id, longContext] of Object.entries(longContextCosts)) { + const model = getBundledModel<"openai-responses">("openai", id); + expect(model.compat.reasoningDisableMode).toBe("none-effort"); + expect(model.cost.longContext).toEqual({ inputThreshold: 272_000, ...longContext }); + } + for (const id of ["daybreak-red-latest", "gpt-5.6-cyber"]) { + const model = getBundledModel<"openai-responses">("openai", id); + expect(model.compat.reasoningDisableMode).toBe("none-effort"); + expect(model.cost.longContext).toBeUndefined(); + } + }); + test("exposes off and every GPT-5.6 wire effort on all Daybreak IDs", () => { const generated: ModelSpec[] = OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS.map(model => ({ ...model, diff --git a/packages/coding-agent/src/config/model-patch.ts b/packages/coding-agent/src/config/model-patch.ts index 38e9920d0..d0d25a711 100644 --- a/packages/coding-agent/src/config/model-patch.ts +++ b/packages/coding-agent/src/config/model-patch.ts @@ -223,11 +223,13 @@ export function applyModelPatch(base: Model, patch: ModelPatch, transport: } if (patch.premiumMultiplier !== undefined) result.premiumMultiplier = patch.premiumMultiplier; if (patch.cost) { + const longContext = patch.cost.longContext ?? base.cost.longContext; result.cost = { input: patch.cost.input ?? base.cost.input, output: patch.cost.output ?? base.cost.output, cacheRead: patch.cost.cacheRead ?? base.cost.cacheRead, cacheWrite: patch.cost.cacheWrite ?? base.cost.cacheWrite, + ...(longContext ? { longContext } : {}), }; } let compat: ModelSpec["compat"]; diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 591c3cb53..088cf3d19 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -1312,7 +1312,7 @@ describe("ModelRegistry", () => { }, }); costPartial = readonlyRegistry({ - providers: { openrouter: { modelOverrides: { "anthropic/claude-sonnet-4": { cost: { input: 99 } } } } }, + providers: { openai: { modelOverrides: { "gpt-5.6": { cost: { input: 99 } } } } }, }); addHeaders = readonlyRegistry({ providers: { @@ -1438,12 +1438,15 @@ describe("ModelRegistry", () => { expect(invalid.find("myprovider", "my-model")).toBeUndefined(); }); - test("model override can change cost fields partially", () => { - const sonnet = getModelsForProvider(costPartial, "openrouter").find(m => m.id === "anthropic/claude-sonnet-4"); - // Input cost should be overridden - expect(sonnet?.cost.input).toBe(99); - // Other cost fields should be preserved from built-in - expect(sonnet?.cost.output).toBeGreaterThan(0); + test("model override can change cost fields partially without dropping long-context pricing", () => { + const gpt56 = getModelsForProvider(costPartial, "openai").find(m => m.id === "gpt-5.6"); + expect(gpt56?.cost.input).toBe(99); + expect(gpt56?.cost.output).toBeGreaterThan(0); + expect(gpt56?.cost.longContext).toMatchObject({ + inputThreshold: 272_000, + input: 10, + output: 45, + }); }); test("model override can add headers", () => {