feat: implemented long-context pricing and configuration support
- Added long-context pricing tiers and billing policies for subscription Codex models in the catalog. - Introduced the `extendedContext` configuration setting to control premium long-context windows. - Implemented runtime policy refresh and model re-binding when context settings change. - Added comprehensive unit tests for pricing tiers, context capping, and policy toggling behavior.
This commit is contained in:
@@ -5,6 +5,7 @@
|
||||
### Added
|
||||
|
||||
- Models now materialize an optional `tokenizer` family in the catalog (`claude-v3`/`v47`/`v5`, Qwen 3.5+, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5+). The field follows `requestModelId`, applies to bundled, discovered, and custom models, and can be explicitly overridden in model configuration.
|
||||
- Subscription Codex GPT-5.6 Sol/Terra/Luna now carry the same `cost.longContext` tier as their first-party API siblings (2x input / 1.5x output above 272K input tokens, [openai/codex#32486](https://github.com/openai/codex/issues/32486)), so cost attribution reflects the higher rating above the threshold and downstream consumers can locate the standard-pricing boundary.
|
||||
|
||||
### Fixed
|
||||
|
||||
|
||||
@@ -508,12 +508,17 @@ function inferGeneratedApplyPatchToolType(
|
||||
|
||||
function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIModel): void {
|
||||
const isFirstPartyResponses = model.provider === "openai" && model.api === "openai-responses";
|
||||
// Subscription Codex rates usage at the same >272K long-context tier as the
|
||||
// API (openai/codex#32486), so first-party Codex SKUs carry the tier too —
|
||||
// it drives both cost attribution and the extended-context window clamp.
|
||||
const isFirstPartyCodex = model.provider === "openai-codex" && model.api === "openai-codex-responses";
|
||||
if (isFirstPartyResponses && modelOrRequestIdValue(model, OPENAI_NONE_EFFORT_MODEL_IDS)) {
|
||||
model.compat = { ...(model.compat ?? {}), reasoningDisableMode: "none-effort" };
|
||||
}
|
||||
const longContextCost = isFirstPartyResponses
|
||||
? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID)
|
||||
: undefined;
|
||||
const longContextCost =
|
||||
isFirstPartyResponses || isFirstPartyCodex
|
||||
? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID)
|
||||
: undefined;
|
||||
if (longContextCost) {
|
||||
model.cost = { ...model.cost, longContext: longContextCost };
|
||||
}
|
||||
|
||||
+2639
-894
File diff suppressed because it is too large
Load Diff
@@ -162,6 +162,21 @@ describe("generated model policies", () => {
|
||||
expect(models[4]?.contextWindow).toBe(272000);
|
||||
});
|
||||
|
||||
it("applies GPT-5.6 long-context pricing to Codex-transport SKUs (openai/codex#32486)", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({ id: "gpt-5.6-sol", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
createSpec({ id: "gpt-5.6-luna", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
// Third-party carriers of the same id must not inherit the tier.
|
||||
createSpec({ id: "gpt-5.6-sol", api: "openai-completions", provider: "openrouter" }),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.cost.longContext).toMatchObject({ inputThreshold: 272_000, input: 10, output: 45 });
|
||||
expect(models[1]?.cost.longContext).toMatchObject({ inputThreshold: 272_000, input: 0.4, output: 1.8 });
|
||||
expect(models[2]?.cost.longContext).toBeUndefined();
|
||||
});
|
||||
|
||||
it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { calculateCost, getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import type { Usage } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
function usage(fields: Pick<Usage, "input" | "output" | "cacheRead" | "cacheWrite">): Usage {
|
||||
return {
|
||||
...fields,
|
||||
totalTokens: fields.input + fields.output + fields.cacheRead + fields.cacheWrite,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
};
|
||||
}
|
||||
|
||||
describe("long-context pricing tier", () => {
|
||||
// GPT-5.6 Sol: $5/$30 standard, $10/$45 above 272K input tokens; the tier
|
||||
// rate applies to the ENTIRE request once total prompt input (input +
|
||||
// cacheRead + cacheWrite) crosses the threshold, matching OpenAI's billing.
|
||||
const sol = getBundledModel("openai", "gpt-5.6-sol");
|
||||
|
||||
it("bills at standard rates at or below the 272K input threshold", () => {
|
||||
const atThreshold = usage({ input: 72_000, output: 10_000, cacheRead: 200_000, cacheWrite: 0 });
|
||||
calculateCost(sol, atThreshold);
|
||||
expect(atThreshold.cost.input).toBeCloseTo((5 / 1e6) * 72_000, 10);
|
||||
expect(atThreshold.cost.output).toBeCloseTo((30 / 1e6) * 10_000, 10);
|
||||
expect(atThreshold.cost.cacheRead).toBeCloseTo((0.5 / 1e6) * 200_000, 10);
|
||||
});
|
||||
|
||||
it("bills the whole request at tier rates once prompt input crosses the threshold", () => {
|
||||
const overThreshold = usage({ input: 72_001, output: 10_000, cacheRead: 200_000, cacheWrite: 0 });
|
||||
calculateCost(sol, overThreshold);
|
||||
expect(overThreshold.cost.input).toBeCloseTo((10 / 1e6) * 72_001, 10);
|
||||
expect(overThreshold.cost.output).toBeCloseTo((45 / 1e6) * 10_000, 10);
|
||||
expect(overThreshold.cost.cacheRead).toBeCloseTo((1 / 1e6) * 200_000, 10);
|
||||
expect(overThreshold.cost.total).toBeCloseTo(
|
||||
overThreshold.cost.input + overThreshold.cost.output + overThreshold.cost.cacheRead,
|
||||
10,
|
||||
);
|
||||
});
|
||||
|
||||
it("prices subscription Codex SKUs with the same tier for cost attribution", () => {
|
||||
const codexSol = getBundledModel("openai-codex", "gpt-5.6-sol");
|
||||
const overThreshold = usage({ input: 300_000, output: 1_000, cacheRead: 0, cacheWrite: 0 });
|
||||
calculateCost(codexSol, overThreshold);
|
||||
expect(overThreshold.cost.input).toBeCloseTo((10 / 1e6) * 300_000, 10);
|
||||
expect(overThreshold.cost.output).toBeCloseTo((45 / 1e6) * 1_000, 10);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user