fix(catalog): complete GPT-5.6 off and pricing support
(cherry picked from commit fc034d61ff69e8ad1870f674c72ff3f761741862)
This commit is contained in:
@@ -51,7 +51,7 @@
|
||||
- Fixed the AWS credential resolver ignoring `role_arn` profiles: shared-config role chaining (`source_profile` recursion, `web_identity_token_file`, `credential_source`) now resolves via STS `AssumeRole`/`AssumeRoleWithWebIdentity`, honoring `role_session_name`/`duration_seconds`/`external_id`, so Bedrock is detected on EKS/IRSA and multi-account setups instead of reporting "No models available" ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
|
||||
- Fixed Bedrock availability being under-detected on Nitro/EKS hosts: the EC2 metadata probe now recognizes Nitro DMI markers (`board_asset_tag` instance ids, `Amazon EC2` vendor fields) in addition to the Xen `ec2` UUID prefix ([#8209](https://github.com/can1357/oh-my-pi/issues/8209)).
|
||||
- Fixed DeepSeek Responses targets (opencode-go) rejecting a thinking-mode continuation with `400 The reasoning_text in the thinking mode must be passed back to the API` after a prewalk hand-off plus mid-run compaction: the Responses input builder re-encoded replayed assistant turns without a reasoning item, so the request enabled reasoning but shipped no `reasoning_text`. The encoder now synthesizes a `reasoning_text` reasoning item for every replayed assistant turn when the target requires reasoning replay in thinking mode (`requiresReasoningContentForAllAssistantTurns` / `requiresReasoningContentForToolCalls`), mirroring the chat-completions `reasoning_content` safety net ([#8248](https://github.com/can1357/oh-my-pi/issues/8248)).
|
||||
- Fixed OpenAI Daybreak `off` thinking requests serializing as `low`; models with explicit wire-level off support now send `reasoning.effort: "none"`.
|
||||
- Fixed OpenAI GPT-5.6 and Daybreak `off` thinking requests serializing as `low`; every first-party alias with explicit wire-level off support now sends `reasoning.effort: "none"`.
|
||||
|
||||
## [17.2.12] - 2026-08-08
|
||||
|
||||
|
||||
@@ -196,4 +196,42 @@ describe("calculateCost", () => {
|
||||
|
||||
expect(usage.cost.total).toBeCloseTo(0.01005, 8);
|
||||
});
|
||||
|
||||
it("keeps Daybreak Blue at short-context rates through 272K prompt tokens", () => {
|
||||
const model = getBundledModel("openai", "daybreak-blue-latest");
|
||||
const usage: Usage = {
|
||||
input: 270_000,
|
||||
output: 1_000,
|
||||
cacheRead: 1_000,
|
||||
cacheWrite: 1_000,
|
||||
totalTokens: 273_000,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
};
|
||||
|
||||
calculateCost(model, usage);
|
||||
|
||||
expect(usage.cost.input).toBeCloseTo(1.35, 12);
|
||||
expect(usage.cost.output).toBeCloseTo(0.03, 12);
|
||||
expect(usage.cost.cacheRead).toBeCloseTo(0.0005, 12);
|
||||
expect(usage.cost.cacheWrite).toBeCloseTo(0.00625, 12);
|
||||
});
|
||||
|
||||
it("prices the full Daybreak Blue request at long-context rates above 272K prompt tokens", () => {
|
||||
const model = getBundledModel("openai", "daybreak-blue-latest");
|
||||
const usage: Usage = {
|
||||
input: 270_001,
|
||||
output: 1_000,
|
||||
cacheRead: 1_000,
|
||||
cacheWrite: 1_000,
|
||||
totalTokens: 273_001,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
};
|
||||
|
||||
calculateCost(model, usage);
|
||||
|
||||
expect(usage.cost.input).toBeCloseTo(2.70001, 12);
|
||||
expect(usage.cost.output).toBeCloseTo(0.045, 12);
|
||||
expect(usage.cost.cacheRead).toBeCloseTo(0.001, 12);
|
||||
expect(usage.cost.cacheWrite).toBeCloseTo(0.0125, 12);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -4,24 +4,36 @@ import type { Context } from "@oh-my-pi/pi-ai/types";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
const DAYBREAK_MODEL_IDS = ["daybreak-blue-latest", "daybreak-red-latest", "gpt-5.6-cyber", "gpt-5.6-sol"];
|
||||
const DAYBREAK_EFFORTS = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max];
|
||||
const GPT_56_MODEL_IDS = [
|
||||
"daybreak-blue-latest",
|
||||
"daybreak-red-latest",
|
||||
"gpt-5.6",
|
||||
"gpt-5.6-cyber",
|
||||
"gpt-5.6-luna",
|
||||
"gpt-5.6-luna-pro",
|
||||
"gpt-5.6-sol",
|
||||
"gpt-5.6-sol-pro",
|
||||
"gpt-5.6-terra",
|
||||
"gpt-5.6-terra-pro",
|
||||
];
|
||||
const GPT_56_EFFORTS = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max];
|
||||
const CONTEXT: Context = {
|
||||
messages: [{ role: "user", content: "hello", timestamp: 0 }],
|
||||
};
|
||||
|
||||
describe("OpenAI Daybreak Responses reasoning payload", () => {
|
||||
for (const id of DAYBREAK_MODEL_IDS) {
|
||||
describe("OpenAI GPT-5.6 Responses reasoning payload", () => {
|
||||
for (const id of GPT_56_MODEL_IDS) {
|
||||
test(`${id} serializes off and every supported thinking level`, () => {
|
||||
const model = getBundledModel<"openai-responses">("openai", id);
|
||||
if (!model) throw new Error(`openai/${id} must be in bundled models.json`);
|
||||
|
||||
const mode = model.reasoningMode ? { mode: model.reasoningMode } : {};
|
||||
const disabled = buildParams(model, CONTEXT, { disableReasoning: true }, undefined);
|
||||
expect(disabled.params.reasoning).toEqual({ effort: "none" });
|
||||
expect(disabled.params.reasoning).toEqual({ effort: "none", ...mode });
|
||||
|
||||
for (const effort of DAYBREAK_EFFORTS) {
|
||||
for (const effort of GPT_56_EFFORTS) {
|
||||
const enabled = buildParams(model, CONTEXT, { reasoning: effort }, undefined);
|
||||
expect(enabled.params.reasoning).toEqual({ effort, summary: "auto" });
|
||||
expect(enabled.params.reasoning).toEqual({ effort, summary: "auto", ...mode });
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
### Fixed
|
||||
|
||||
- Fixed Codex-discovered `gpt-daybreak-*` aliases being treated as unknown models, restoring the GPT-5.6 `low`/`medium`/`high`/`xhigh`/`max` effort ladder and its 372K fallback only when the Codex registry omits `context_window`.
|
||||
- Fixed first-party OpenAI GPT-5.6 aliases to preserve wire-level `off` through generated pro aliases and to price requests above 272K input at each SKU's documented long-context rates.
|
||||
|
||||
## [17.2.15] - 2026-08-12
|
||||
|
||||
|
||||
@@ -21,9 +21,10 @@ import { resolveModelThinking } from "../src/model-thinking";
|
||||
import { isOllamaCloudOutputCapped, OLLAMA_CLOUD_MAX_OUTPUT_TOKENS } from "../src/provider-models/ollama";
|
||||
import {
|
||||
ALIBABA_TOKEN_PLAN_STATIC_MODELS,
|
||||
OPENAI_GPT_56_LONG_CONTEXT_COSTS,
|
||||
resolveWaferServerlessThinkingFormat,
|
||||
} from "../src/provider-models/openai-compat";
|
||||
import type { Api, Model, ModelSpec } from "../src/types";
|
||||
import type { Api, LongContextTokenCost, Model, ModelSpec } from "../src/types";
|
||||
import { isVariantCollapsedSpec } from "../src/variant-collapse";
|
||||
import { buildCanonicalModelIndex, buildCanonicalReferenceData } from "./equivalence";
|
||||
|
||||
@@ -149,13 +150,33 @@ const CODEX_GPT_5_6_372K_MODEL_IDS: Record<string, true> = {
|
||||
"gpt-5.6-terra": true,
|
||||
};
|
||||
|
||||
const OPENAI_DAYBREAK_REASONING_MODEL_IDS: Record<string, true> = {
|
||||
const OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID: Readonly<Record<string, LongContextTokenCost>> = {
|
||||
"daybreak-blue-latest": OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
||||
"gpt-5.6": OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
||||
"gpt-5.6-luna": OPENAI_GPT_56_LONG_CONTEXT_COSTS.luna,
|
||||
"gpt-5.6-sol": OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
||||
"gpt-5.6-terra": OPENAI_GPT_56_LONG_CONTEXT_COSTS.terra,
|
||||
};
|
||||
|
||||
const OPENAI_NONE_EFFORT_MODEL_IDS: Record<string, true> = {
|
||||
"daybreak-blue-latest": true,
|
||||
"daybreak-red-latest": true,
|
||||
"gpt-5.6": true,
|
||||
"gpt-5.6-cyber": true,
|
||||
"gpt-5.6-luna": true,
|
||||
"gpt-5.6-sol": true,
|
||||
"gpt-5.6-terra": true,
|
||||
};
|
||||
|
||||
function modelOrRequestIdValue<T>(
|
||||
model: Pick<ModelSpec<Api>, "id" | "requestModelId">,
|
||||
values: Readonly<Record<string, T>>,
|
||||
): T | undefined {
|
||||
const direct = values[bareModelId(model.id)];
|
||||
if (direct !== undefined) return direct;
|
||||
return model.requestModelId === undefined ? undefined : values[bareModelId(model.requestModelId)];
|
||||
}
|
||||
|
||||
const COPILOT_GENERATED_LIMITS: Record<string, { contextWindow: number; maxTokens: number }> = {
|
||||
"claude-opus-4.6": { contextWindow: 168000, maxTokens: 32000 },
|
||||
"gpt-5.2": { contextWindow: 272000, maxTokens: 128000 },
|
||||
@@ -477,13 +498,16 @@ function inferGeneratedApplyPatchToolType(
|
||||
}
|
||||
|
||||
function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIModel): void {
|
||||
if (
|
||||
model.provider === "openai" &&
|
||||
model.api === "openai-responses" &&
|
||||
OPENAI_DAYBREAK_REASONING_MODEL_IDS[model.id]
|
||||
) {
|
||||
const isFirstPartyResponses = model.provider === "openai" && model.api === "openai-responses";
|
||||
if (isFirstPartyResponses && modelOrRequestIdValue(model, OPENAI_NONE_EFFORT_MODEL_IDS)) {
|
||||
model.compat = { ...(model.compat ?? {}), reasoningDisableMode: "none-effort" };
|
||||
}
|
||||
const longContextCost = isFirstPartyResponses
|
||||
? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID)
|
||||
: undefined;
|
||||
if (longContextCost) {
|
||||
model.cost = { ...model.cost, longContext: longContextCost };
|
||||
}
|
||||
|
||||
// Codex models: 400K figure includes output budget; input window is 272K.
|
||||
if (parsedModel.variant.startsWith("codex") && parsedModel.variant !== "codex-spark") {
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import { buildModel } from "./build";
|
||||
import { readModelCache, writeModelCache } from "./model-cache";
|
||||
import { type GeneratedProvider, getBundledModels } from "./models";
|
||||
import type { Api, Model, ModelSpec, Provider } from "./types";
|
||||
import type { Api, Model, ModelCost, ModelSpec, Provider, TokenCost } from "./types";
|
||||
import { isRecord } from "./utils";
|
||||
import { collapseBuiltModelVariants } from "./variant-collapse";
|
||||
|
||||
@@ -510,6 +510,7 @@ function mergeDynamicModel<TApi extends Api>(existingModel: Model<TApi>, dynamic
|
||||
const reasoning = dynamicReasoningAuthoritative
|
||||
? dynamicModel.reasoning
|
||||
: existingModel.reasoning || dynamicModel.reasoning;
|
||||
const longContextCost = dynamicModel.cost.longContext ?? existingModel.cost.longContext;
|
||||
// Re-build from spec stage: sparse compat comes from `compatConfig` (the
|
||||
// verbatim override vocabulary), never the resolved `compat` record.
|
||||
return buildModel({
|
||||
@@ -523,6 +524,7 @@ function mergeDynamicModel<TApi extends Api>(existingModel: Model<TApi>, dynamic
|
||||
output: preferDiscoveryCost(dynamicModel.cost.output, existingModel.cost.output),
|
||||
cacheRead: preferDiscoveryCost(dynamicModel.cost.cacheRead, existingModel.cost.cacheRead),
|
||||
cacheWrite: preferDiscoveryCost(dynamicModel.cost.cacheWrite, existingModel.cost.cacheWrite),
|
||||
...(longContextCost ? { longContext: longContextCost } : {}),
|
||||
},
|
||||
contextWindow: preferDiscoveryLimit(dynamicModel.contextWindow, existingModel.contextWindow),
|
||||
maxTokens: preferDiscoveryLimit(dynamicModel.maxTokens, existingModel.maxTokens),
|
||||
@@ -640,7 +642,7 @@ function isModelInputArray(value: unknown): value is ("text" | "image")[] {
|
||||
return true;
|
||||
}
|
||||
|
||||
function isModelCost(value: unknown): value is Model<Api>["cost"] {
|
||||
function isTokenCost(value: unknown): value is TokenCost {
|
||||
if (!isRecord(value)) {
|
||||
return false;
|
||||
}
|
||||
@@ -670,3 +672,12 @@ function isModelCost(value: unknown): value is Model<Api>["cost"] {
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
function isModelCost(value: unknown): value is ModelCost {
|
||||
if (!isTokenCost(value)) return false;
|
||||
const longContext = (value as TokenCost & { longContext?: unknown }).longContext;
|
||||
if (longContext === undefined) return true;
|
||||
if (!isTokenCost(longContext) || !isRecord(longContext)) return false;
|
||||
const threshold = longContext.inputThreshold;
|
||||
return typeof threshold === "number" && threshold > 0 && threshold < Infinity;
|
||||
}
|
||||
|
||||
@@ -72354,7 +72354,14 @@
|
||||
"input": 5,
|
||||
"output": 30,
|
||||
"cacheRead": 0.5,
|
||||
"cacheWrite": 6.25
|
||||
"cacheWrite": 6.25,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 10,
|
||||
"output": 45,
|
||||
"cacheRead": 1,
|
||||
"cacheWrite": 12.5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
@@ -72368,6 +72375,9 @@
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"compat": {
|
||||
"reasoningDisableMode": "none-effort"
|
||||
}
|
||||
},
|
||||
"gpt-5.6-luna": {
|
||||
@@ -72385,7 +72395,14 @@
|
||||
"input": 0.2,
|
||||
"output": 1.2,
|
||||
"cacheRead": 0.02,
|
||||
"cacheWrite": 0.25
|
||||
"cacheWrite": 0.25,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 0.4,
|
||||
"output": 1.8,
|
||||
"cacheRead": 0.04,
|
||||
"cacheWrite": 0.5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
@@ -72399,6 +72416,9 @@
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"compat": {
|
||||
"reasoningDisableMode": "none-effort"
|
||||
}
|
||||
},
|
||||
"gpt-5.6-luna-pro": {
|
||||
@@ -72416,7 +72436,14 @@
|
||||
"input": 0.2,
|
||||
"output": 1.2,
|
||||
"cacheRead": 0.02,
|
||||
"cacheWrite": 0.25
|
||||
"cacheWrite": 0.25,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 0.4,
|
||||
"output": 1.8,
|
||||
"cacheRead": 0.04,
|
||||
"cacheWrite": 0.5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
@@ -72432,6 +72459,9 @@
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"compat": {
|
||||
"reasoningDisableMode": "none-effort"
|
||||
}
|
||||
},
|
||||
"gpt-5.6-sol": {
|
||||
@@ -72449,7 +72479,14 @@
|
||||
"input": 5,
|
||||
"output": 30,
|
||||
"cacheRead": 0.5,
|
||||
"cacheWrite": 6.25
|
||||
"cacheWrite": 6.25,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 10,
|
||||
"output": 45,
|
||||
"cacheRead": 1,
|
||||
"cacheWrite": 12.5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
@@ -72483,7 +72520,14 @@
|
||||
"input": 5,
|
||||
"output": 30,
|
||||
"cacheRead": 0.5,
|
||||
"cacheWrite": 6.25
|
||||
"cacheWrite": 6.25,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 10,
|
||||
"output": 45,
|
||||
"cacheRead": 1,
|
||||
"cacheWrite": 12.5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
@@ -72499,6 +72543,9 @@
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"compat": {
|
||||
"reasoningDisableMode": "none-effort"
|
||||
}
|
||||
},
|
||||
"gpt-5.6-terra": {
|
||||
@@ -72516,7 +72563,14 @@
|
||||
"input": 2,
|
||||
"output": 12,
|
||||
"cacheRead": 0.2,
|
||||
"cacheWrite": 2.5
|
||||
"cacheWrite": 2.5,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 4,
|
||||
"output": 18,
|
||||
"cacheRead": 0.4,
|
||||
"cacheWrite": 5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
@@ -72530,6 +72584,9 @@
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"compat": {
|
||||
"reasoningDisableMode": "none-effort"
|
||||
}
|
||||
},
|
||||
"gpt-5.6-terra-pro": {
|
||||
@@ -72547,7 +72604,14 @@
|
||||
"input": 2,
|
||||
"output": 12,
|
||||
"cacheRead": 0.2,
|
||||
"cacheWrite": 2.5
|
||||
"cacheWrite": 2.5,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 4,
|
||||
"output": 18,
|
||||
"cacheRead": 0.4,
|
||||
"cacheWrite": 5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
@@ -72563,6 +72627,9 @@
|
||||
"xhigh",
|
||||
"max"
|
||||
]
|
||||
},
|
||||
"compat": {
|
||||
"reasoningDisableMode": "none-effort"
|
||||
}
|
||||
},
|
||||
"gpt-realtime-2.1": {
|
||||
@@ -72857,7 +72924,14 @@
|
||||
"input": 5,
|
||||
"output": 30,
|
||||
"cacheRead": 0.5,
|
||||
"cacheWrite": 6.25
|
||||
"cacheWrite": 6.25,
|
||||
"longContext": {
|
||||
"inputThreshold": 272000,
|
||||
"input": 10,
|
||||
"output": 45,
|
||||
"cacheRead": 1,
|
||||
"cacheWrite": 12.5
|
||||
}
|
||||
},
|
||||
"contextWindow": 1050000,
|
||||
"maxTokens": 128000,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { buildModel } from "./build";
|
||||
import MODELS from "./models.json" with { type: "json" };
|
||||
import type { Api, KnownProvider, Model, ModelSpec, Usage } from "./types";
|
||||
import type { Api, KnownProvider, Model, ModelSpec, TokenCost, Usage } from "./types";
|
||||
|
||||
/**
|
||||
* Static bundled model registry loaded from `models.json`.
|
||||
@@ -42,13 +42,22 @@ export function getBundledModels(provider: GeneratedProvider): Model<Api>[] {
|
||||
const models = getProviderModels(provider);
|
||||
return models ? (Array.from(models.values()) as Model<Api>[]) : [];
|
||||
}
|
||||
function resolveTokenCost(cost: Model["cost"], usage: Usage): TokenCost {
|
||||
const longContext = cost.longContext;
|
||||
if (!longContext) return cost;
|
||||
const orchestration = usage.orchestration;
|
||||
const promptInputTokens =
|
||||
usage.input + usage.cacheRead + usage.cacheWrite + (orchestration?.input ?? 0) + (orchestration?.cacheRead ?? 0);
|
||||
return promptInputTokens > longContext.inputThreshold ? longContext : cost;
|
||||
}
|
||||
|
||||
export function calculateCost<TApi extends Api>(model: Model<TApi>, usage: Usage): Usage["cost"] {
|
||||
const rates = resolveTokenCost(model.cost, usage);
|
||||
const orchestration = usage.orchestration;
|
||||
usage.cost.input = (model.cost.input / 1000000) * (usage.input + (orchestration?.input ?? 0));
|
||||
usage.cost.output = (model.cost.output / 1000000) * (usage.output + (orchestration?.output ?? 0));
|
||||
usage.cost.cacheRead = (model.cost.cacheRead / 1000000) * (usage.cacheRead + (orchestration?.cacheRead ?? 0));
|
||||
usage.cost.cacheWrite = cacheWriteCost(model, usage);
|
||||
usage.cost.input = (rates.input / 1000000) * (usage.input + (orchestration?.input ?? 0));
|
||||
usage.cost.output = (rates.output / 1000000) * (usage.output + (orchestration?.output ?? 0));
|
||||
usage.cost.cacheRead = (rates.cacheRead / 1000000) * (usage.cacheRead + (orchestration?.cacheRead ?? 0));
|
||||
usage.cost.cacheWrite = cacheWriteCost(rates, usage);
|
||||
usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite;
|
||||
return usage.cost;
|
||||
}
|
||||
@@ -56,7 +65,7 @@ export function calculateCost<TApi extends Api>(model: Model<TApi>, usage: Usage
|
||||
/**
|
||||
* Price cache-write tokens, honoring the TTL breakdown when the provider reports one.
|
||||
*
|
||||
* `model.cost.cacheWrite` is the 5-minute write rate (Anthropic bills 5m writes at
|
||||
* `rates.cacheWrite` is the 5-minute write rate (Anthropic bills 5m writes at
|
||||
* 1.25x base input). When `usage.cttl` is present the write mixes 5m and 1h
|
||||
* breakpoints — omp defaults to 1h retention on first-party Anthropic, and 1h writes
|
||||
* bill at 2x base input — so each component is priced at its own rate instead of the
|
||||
@@ -70,14 +79,14 @@ export function calculateCost<TApi extends Api>(model: Model<TApi>, usage: Usage
|
||||
* so any unattributed remainder is priced at the flat rate instead of being dropped:
|
||||
* a partial or stale breakdown must never make write tokens free.
|
||||
*/
|
||||
function cacheWriteCost<TApi extends Api>(model: Model<TApi>, usage: Usage): number {
|
||||
const rate5m = model.cost.cacheWrite / 1000000;
|
||||
function cacheWriteCost(rates: TokenCost, usage: Usage): number {
|
||||
const rate5m = rates.cacheWrite / 1000000;
|
||||
const cttl = usage.cttl;
|
||||
if (!cttl) return rate5m * usage.cacheWrite;
|
||||
const fiveMinute = cttl.ephemeral5m ?? 0;
|
||||
const oneHour = cttl.ephemeral1h ?? 0;
|
||||
const residual = Math.max(0, usage.cacheWrite - fiveMinute - oneHour);
|
||||
return rate5m * (fiveMinute + residual) + ((model.cost.input * 2) / 1000000) * oneHour;
|
||||
return rate5m * (fiveMinute + residual) + ((rates.input * 2) / 1000000) * oneHour;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -867,7 +867,36 @@ export function umansModelManagerOptions(config?: UmansModelManagerConfig): Mode
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
const OPENAI_API_BASE_URL = "https://api.openai.com/v1";
|
||||
const OPENAI_GPT_56_SOL_STANDARD_COST = { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 6.25 } as const;
|
||||
export const OPENAI_GPT_56_LONG_CONTEXT_COSTS = {
|
||||
luna: {
|
||||
inputThreshold: 272_000,
|
||||
input: 0.4,
|
||||
output: 1.8,
|
||||
cacheRead: 0.04,
|
||||
cacheWrite: 0.5,
|
||||
},
|
||||
sol: {
|
||||
inputThreshold: 272_000,
|
||||
input: 10,
|
||||
output: 45,
|
||||
cacheRead: 1,
|
||||
cacheWrite: 12.5,
|
||||
},
|
||||
terra: {
|
||||
inputThreshold: 272_000,
|
||||
input: 4,
|
||||
output: 18,
|
||||
cacheRead: 0.4,
|
||||
cacheWrite: 5,
|
||||
},
|
||||
} as const;
|
||||
const OPENAI_GPT_56_SOL_STANDARD_COST = {
|
||||
input: 5,
|
||||
output: 30,
|
||||
cacheRead: 0.5,
|
||||
cacheWrite: 6.25,
|
||||
longContext: OPENAI_GPT_56_LONG_CONTEXT_COSTS.sol,
|
||||
} as const;
|
||||
const OPENAI_GPT_56_CYBER_STANDARD_COST = {
|
||||
input: 12.5,
|
||||
output: 75,
|
||||
|
||||
@@ -820,6 +820,28 @@ export interface RemoteCompactionConfig<TApi extends Api = Api> {
|
||||
model?: string;
|
||||
}
|
||||
|
||||
/** Per-million-token rates for one model pricing tier. */
|
||||
export interface TokenCost {
|
||||
input: number;
|
||||
output: number;
|
||||
cacheRead: number;
|
||||
cacheWrite: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rates applied to the full request when its prompt exceeds `inputThreshold`.
|
||||
* Prompt input is the sum of uncached, cached-read, cache-write, and
|
||||
* provider-orchestration input tokens.
|
||||
*/
|
||||
export interface LongContextTokenCost extends TokenCost {
|
||||
inputThreshold: number;
|
||||
}
|
||||
|
||||
/** Base token rates plus an optional long-context tier. */
|
||||
export interface ModelCost extends TokenCost {
|
||||
longContext?: LongContextTokenCost;
|
||||
}
|
||||
|
||||
// Model interface for the unified model system
|
||||
export interface Model<TApi extends Api = Api> {
|
||||
id: string;
|
||||
@@ -865,12 +887,7 @@ export interface Model<TApi extends Api = Api> {
|
||||
gitlabDuoWorkflowRootNamespaceId?: string;
|
||||
/** Cursor `max_mode` request flag returned by `GetUsableModels` for premium models that require max mode. */
|
||||
cursorMaxMode?: boolean;
|
||||
cost: {
|
||||
input: number; // $/million tokens
|
||||
output: number; // $/million tokens
|
||||
cacheRead: number; // $/million tokens
|
||||
cacheWrite: number; // $/million tokens
|
||||
};
|
||||
cost: ModelCost;
|
||||
/** Premium Copilot requests charged per user-initiated request (defaults to 1). */
|
||||
premiumMultiplier?: number;
|
||||
contextWindow: number | null;
|
||||
|
||||
@@ -763,6 +763,52 @@ describe("model cache spec round trip", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("preserves static long-context pricing through dynamic refresh and cache restore", async () => {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-tiered-cost-"));
|
||||
const dbPath = path.join(tempDir, "models.db");
|
||||
const staticModel = completionsSpec({
|
||||
id: "tiered-model",
|
||||
provider: "tiered-cost-test",
|
||||
cost: {
|
||||
input: 1,
|
||||
output: 2,
|
||||
cacheRead: 0.1,
|
||||
cacheWrite: 1.25,
|
||||
longContext: {
|
||||
inputThreshold: 272_000,
|
||||
input: 2,
|
||||
output: 3,
|
||||
cacheRead: 0.2,
|
||||
cacheWrite: 2.5,
|
||||
},
|
||||
},
|
||||
});
|
||||
const dynamicModel = completionsSpec({
|
||||
...staticModel,
|
||||
cost: { input: 3, output: 4, cacheRead: 0.3, cacheWrite: 3.75 },
|
||||
});
|
||||
const options = {
|
||||
providerId: "tiered-cost-test",
|
||||
staticModels: [staticModel],
|
||||
cacheDbPath: dbPath,
|
||||
};
|
||||
try {
|
||||
const online = await resolveProviderModels<"openai-completions">(
|
||||
{ ...options, fetchDynamicModels: async () => [dynamicModel] },
|
||||
"online",
|
||||
);
|
||||
expect(online.models[0]?.cost).toEqual({
|
||||
...dynamicModel.cost,
|
||||
longContext: staticModel.cost.longContext,
|
||||
});
|
||||
|
||||
const offline = await resolveProviderModels<"openai-completions">(options, "offline");
|
||||
expect(offline.models[0]?.cost.longContext).toEqual(staticModel.cost.longContext);
|
||||
} finally {
|
||||
await fs.rm(tempDir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it("invalidates schema-v10 rows that predate computer-use capability provenance", async () => {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-legacy-computer-cache-"));
|
||||
const dbPath = path.join(tempDir, "models.db");
|
||||
|
||||
@@ -91,6 +91,35 @@ describe("generated model policies", () => {
|
||||
expect(models[3]?.priority).toBe(1);
|
||||
});
|
||||
|
||||
it("applies GPT-5.6 off and long-context pricing through request-model aliases", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({ id: "gpt-5.6", api: "openai-responses", provider: "openai" }),
|
||||
createSpec({ id: "gpt-5.6-luna", api: "openai-responses", provider: "openai" }),
|
||||
{
|
||||
...createSpec({ id: "gpt-5.6-sol-pro", api: "openai-responses", provider: "openai" }),
|
||||
requestModelId: "gpt-5.6-sol",
|
||||
},
|
||||
{
|
||||
...createSpec({ id: "gpt-5.6-terra-pro", api: "openai-responses", provider: "openai" }),
|
||||
requestModelId: "gpt-5.6-terra",
|
||||
},
|
||||
createSpec({ id: "gpt-5.6", api: "openai-responses", provider: "openrouter" }),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
for (const model of models.slice(0, 4)) {
|
||||
expect(model.compat).toMatchObject({ reasoningDisableMode: "none-effort" });
|
||||
expect(model.cost.longContext?.inputThreshold).toBe(272_000);
|
||||
}
|
||||
expect(models[0]?.cost.longContext).toMatchObject({ input: 10, output: 45 });
|
||||
expect(models[1]?.cost.longContext).toMatchObject({ input: 0.4, output: 1.8 });
|
||||
expect(models[2]?.cost.longContext).toMatchObject({ input: 10, output: 45 });
|
||||
expect(models[3]?.cost.longContext).toMatchObject({ input: 4, output: 18 });
|
||||
expect(models[4]?.compat).toBeUndefined();
|
||||
expect(models[4]?.cost.longContext).toBeUndefined();
|
||||
});
|
||||
|
||||
it("pins GPT-5.6 Codex-transport context window to the 372K hard capacity (#5705)", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
// Codex discovery underreports these via DEFAULT_CONTEXT_WINDOW=272000.
|
||||
|
||||
@@ -2,19 +2,32 @@ import { describe, expect, test } from "bun:test";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { Api, ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
import { applyGeneratedModelPolicies } from "../scripts/generated-policies";
|
||||
|
||||
const DAYBREAK_EFFORTS = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max];
|
||||
|
||||
describe("OpenAI Daybreak models", () => {
|
||||
describe("OpenAI Daybreak and GPT-5.6 models", () => {
|
||||
test("curates the documented aliases and Cyber snapshot with standard API pricing", () => {
|
||||
const byId = Object.fromEntries(OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS.map(model => [model.id, model]));
|
||||
expect(Object.keys(byId)).toEqual(["daybreak-blue-latest", "daybreak-red-latest", "gpt-5.6-cyber"]);
|
||||
expect(byId["daybreak-blue-latest"]).toMatchObject({
|
||||
name: "Daybreak Blue",
|
||||
cost: { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 6.25 },
|
||||
cost: {
|
||||
input: 5,
|
||||
output: 30,
|
||||
cacheRead: 0.5,
|
||||
cacheWrite: 6.25,
|
||||
longContext: {
|
||||
inputThreshold: 272_000,
|
||||
input: 10,
|
||||
output: 45,
|
||||
cacheRead: 1,
|
||||
cacheWrite: 12.5,
|
||||
},
|
||||
},
|
||||
contextWindow: 1_050_000,
|
||||
maxTokens: 128_000,
|
||||
});
|
||||
@@ -27,6 +40,29 @@ describe("OpenAI Daybreak models", () => {
|
||||
}
|
||||
});
|
||||
|
||||
test("bakes off support and long-context pricing onto every first-party GPT-5.6 alias", () => {
|
||||
const longContextCosts = {
|
||||
"daybreak-blue-latest": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 },
|
||||
"gpt-5.6": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 },
|
||||
"gpt-5.6-luna": { input: 0.4, output: 1.8, cacheRead: 0.04, cacheWrite: 0.5 },
|
||||
"gpt-5.6-luna-pro": { input: 0.4, output: 1.8, cacheRead: 0.04, cacheWrite: 0.5 },
|
||||
"gpt-5.6-sol": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 },
|
||||
"gpt-5.6-sol-pro": { input: 10, output: 45, cacheRead: 1, cacheWrite: 12.5 },
|
||||
"gpt-5.6-terra": { input: 4, output: 18, cacheRead: 0.4, cacheWrite: 5 },
|
||||
"gpt-5.6-terra-pro": { input: 4, output: 18, cacheRead: 0.4, cacheWrite: 5 },
|
||||
} as const;
|
||||
for (const [id, longContext] of Object.entries(longContextCosts)) {
|
||||
const model = getBundledModel<"openai-responses">("openai", id);
|
||||
expect(model.compat.reasoningDisableMode).toBe("none-effort");
|
||||
expect(model.cost.longContext).toEqual({ inputThreshold: 272_000, ...longContext });
|
||||
}
|
||||
for (const id of ["daybreak-red-latest", "gpt-5.6-cyber"]) {
|
||||
const model = getBundledModel<"openai-responses">("openai", id);
|
||||
expect(model.compat.reasoningDisableMode).toBe("none-effort");
|
||||
expect(model.cost.longContext).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
test("exposes off and every GPT-5.6 wire effort on all Daybreak IDs", () => {
|
||||
const generated: ModelSpec<Api>[] = OPENAI_DAYBREAK_CURATED_FALLBACK_MODELS.map(model => ({
|
||||
...model,
|
||||
|
||||
@@ -223,11 +223,13 @@ export function applyModelPatch(base: Model<Api>, patch: ModelPatch, transport:
|
||||
}
|
||||
if (patch.premiumMultiplier !== undefined) result.premiumMultiplier = patch.premiumMultiplier;
|
||||
if (patch.cost) {
|
||||
const longContext = patch.cost.longContext ?? base.cost.longContext;
|
||||
result.cost = {
|
||||
input: patch.cost.input ?? base.cost.input,
|
||||
output: patch.cost.output ?? base.cost.output,
|
||||
cacheRead: patch.cost.cacheRead ?? base.cost.cacheRead,
|
||||
cacheWrite: patch.cost.cacheWrite ?? base.cost.cacheWrite,
|
||||
...(longContext ? { longContext } : {}),
|
||||
};
|
||||
}
|
||||
let compat: ModelSpec<Api>["compat"];
|
||||
|
||||
@@ -1312,7 +1312,7 @@ describe("ModelRegistry", () => {
|
||||
},
|
||||
});
|
||||
costPartial = readonlyRegistry({
|
||||
providers: { openrouter: { modelOverrides: { "anthropic/claude-sonnet-4": { cost: { input: 99 } } } } },
|
||||
providers: { openai: { modelOverrides: { "gpt-5.6": { cost: { input: 99 } } } } },
|
||||
});
|
||||
addHeaders = readonlyRegistry({
|
||||
providers: {
|
||||
@@ -1438,12 +1438,15 @@ describe("ModelRegistry", () => {
|
||||
expect(invalid.find("myprovider", "my-model")).toBeUndefined();
|
||||
});
|
||||
|
||||
test("model override can change cost fields partially", () => {
|
||||
const sonnet = getModelsForProvider(costPartial, "openrouter").find(m => m.id === "anthropic/claude-sonnet-4");
|
||||
// Input cost should be overridden
|
||||
expect(sonnet?.cost.input).toBe(99);
|
||||
// Other cost fields should be preserved from built-in
|
||||
expect(sonnet?.cost.output).toBeGreaterThan(0);
|
||||
test("model override can change cost fields partially without dropping long-context pricing", () => {
|
||||
const gpt56 = getModelsForProvider(costPartial, "openai").find(m => m.id === "gpt-5.6");
|
||||
expect(gpt56?.cost.input).toBe(99);
|
||||
expect(gpt56?.cost.output).toBeGreaterThan(0);
|
||||
expect(gpt56?.cost.longContext).toMatchObject({
|
||||
inputThreshold: 272_000,
|
||||
input: 10,
|
||||
output: 45,
|
||||
});
|
||||
});
|
||||
|
||||
test("model override can add headers", () => {
|
||||
|
||||
Reference in New Issue
Block a user