feat: implemented long-context pricing and configuration support
- Added long-context pricing tiers and billing policies for subscription Codex models in the catalog. - Introduced the `extendedContext` configuration setting to control premium long-context windows. - Implemented runtime policy refresh and model re-binding when context settings change. - Added comprehensive unit tests for pricing tiers, context capping, and policy toggling behavior.
This commit is contained in:
@@ -5,6 +5,7 @@
|
||||
### Added
|
||||
|
||||
- Models now materialize an optional `tokenizer` family in the catalog (`claude-v3`/`v47`/`v5`, Qwen 3.5+, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5+). The field follows `requestModelId`, applies to bundled, discovered, and custom models, and can be explicitly overridden in model configuration.
|
||||
- Subscription Codex GPT-5.6 Sol/Terra/Luna now carry the same `cost.longContext` tier as their first-party API siblings (2x input / 1.5x output above 272K input tokens, [openai/codex#32486](https://github.com/openai/codex/issues/32486)), so cost attribution reflects the higher rating above the threshold and downstream consumers can locate the standard-pricing boundary.
|
||||
|
||||
### Fixed
|
||||
|
||||
|
||||
@@ -508,12 +508,17 @@ function inferGeneratedApplyPatchToolType(
|
||||
|
||||
function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIModel): void {
|
||||
const isFirstPartyResponses = model.provider === "openai" && model.api === "openai-responses";
|
||||
// Subscription Codex rates usage at the same >272K long-context tier as the
|
||||
// API (openai/codex#32486), so first-party Codex SKUs carry the tier too —
|
||||
// it drives both cost attribution and the extended-context window clamp.
|
||||
const isFirstPartyCodex = model.provider === "openai-codex" && model.api === "openai-codex-responses";
|
||||
if (isFirstPartyResponses && modelOrRequestIdValue(model, OPENAI_NONE_EFFORT_MODEL_IDS)) {
|
||||
model.compat = { ...(model.compat ?? {}), reasoningDisableMode: "none-effort" };
|
||||
}
|
||||
const longContextCost = isFirstPartyResponses
|
||||
? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID)
|
||||
: undefined;
|
||||
const longContextCost =
|
||||
isFirstPartyResponses || isFirstPartyCodex
|
||||
? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID)
|
||||
: undefined;
|
||||
if (longContextCost) {
|
||||
model.cost = { ...model.cost, longContext: longContextCost };
|
||||
}
|
||||
|
||||
+2639
-894
File diff suppressed because it is too large
Load Diff
@@ -162,6 +162,21 @@ describe("generated model policies", () => {
|
||||
expect(models[4]?.contextWindow).toBe(272000);
|
||||
});
|
||||
|
||||
it("applies GPT-5.6 long-context pricing to Codex-transport SKUs (openai/codex#32486)", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({ id: "gpt-5.6-sol", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
createSpec({ id: "gpt-5.6-luna", api: "openai-codex-responses", provider: "openai-codex" }),
|
||||
// Third-party carriers of the same id must not inherit the tier.
|
||||
createSpec({ id: "gpt-5.6-sol", api: "openai-completions", provider: "openrouter" }),
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
|
||||
expect(models[0]?.cost.longContext).toMatchObject({ inputThreshold: 272_000, input: 10, output: 45 });
|
||||
expect(models[1]?.cost.longContext).toMatchObject({ inputThreshold: 272_000, input: 0.4, output: 1.8 });
|
||||
expect(models[2]?.cost.longContext).toBeUndefined();
|
||||
});
|
||||
|
||||
it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
createSpec({
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { calculateCost, getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import type { Usage } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
function usage(fields: Pick<Usage, "input" | "output" | "cacheRead" | "cacheWrite">): Usage {
|
||||
return {
|
||||
...fields,
|
||||
totalTokens: fields.input + fields.output + fields.cacheRead + fields.cacheWrite,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
};
|
||||
}
|
||||
|
||||
describe("long-context pricing tier", () => {
|
||||
// GPT-5.6 Sol: $5/$30 standard, $10/$45 above 272K input tokens; the tier
|
||||
// rate applies to the ENTIRE request once total prompt input (input +
|
||||
// cacheRead + cacheWrite) crosses the threshold, matching OpenAI's billing.
|
||||
const sol = getBundledModel("openai", "gpt-5.6-sol");
|
||||
|
||||
it("bills at standard rates at or below the 272K input threshold", () => {
|
||||
const atThreshold = usage({ input: 72_000, output: 10_000, cacheRead: 200_000, cacheWrite: 0 });
|
||||
calculateCost(sol, atThreshold);
|
||||
expect(atThreshold.cost.input).toBeCloseTo((5 / 1e6) * 72_000, 10);
|
||||
expect(atThreshold.cost.output).toBeCloseTo((30 / 1e6) * 10_000, 10);
|
||||
expect(atThreshold.cost.cacheRead).toBeCloseTo((0.5 / 1e6) * 200_000, 10);
|
||||
});
|
||||
|
||||
it("bills the whole request at tier rates once prompt input crosses the threshold", () => {
|
||||
const overThreshold = usage({ input: 72_001, output: 10_000, cacheRead: 200_000, cacheWrite: 0 });
|
||||
calculateCost(sol, overThreshold);
|
||||
expect(overThreshold.cost.input).toBeCloseTo((10 / 1e6) * 72_001, 10);
|
||||
expect(overThreshold.cost.output).toBeCloseTo((45 / 1e6) * 10_000, 10);
|
||||
expect(overThreshold.cost.cacheRead).toBeCloseTo((1 / 1e6) * 200_000, 10);
|
||||
expect(overThreshold.cost.total).toBeCloseTo(
|
||||
overThreshold.cost.input + overThreshold.cost.output + overThreshold.cost.cacheRead,
|
||||
10,
|
||||
);
|
||||
});
|
||||
|
||||
it("prices subscription Codex SKUs with the same tier for cost attribution", () => {
|
||||
const codexSol = getBundledModel("openai-codex", "gpt-5.6-sol");
|
||||
const overThreshold = usage({ input: 300_000, output: 1_000, cacheRead: 0, cacheWrite: 0 });
|
||||
calculateCost(codexSol, overThreshold);
|
||||
expect(overThreshold.cost.input).toBeCloseTo((10 / 1e6) * 300_000, 10);
|
||||
expect(overThreshold.cost.output).toBeCloseTo((45 / 1e6) * 1_000, 10);
|
||||
});
|
||||
});
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`.
|
||||
- Added `tokenizer` to custom model and `modelOverrides` configuration. It overrides the catalog-resolved local tokenizer family for a model when a proxy serves a known model id with a different tokenizer.
|
||||
- Added `extendedContext` setting (`/settings` → Context → General, default on). When off, models with a premium long-context price tier (OpenAI GPT-5.6 Sol/Terra/Luna bill 2x input / 1.5x output above 272K input tokens, on both the API and subscription Codex) are capped at the standard-pricing threshold — they appear as 272K again and compaction fires before a request crosses into premium billing. Toggling mid-session re-clamps or restores the active model's window immediately. Anthropic Claude 4.6+ serves its full 1M window at standard pricing, so no Anthropic model is affected.
|
||||
|
||||
### Changed
|
||||
|
||||
|
||||
@@ -140,6 +140,18 @@ function getDisabledProviderIdsFromSettings(): Set<string> {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether premium long-context windows are enabled. Defaults to true when the
|
||||
* settings singleton is not initialized (SDK embedding, early boot).
|
||||
*/
|
||||
function isExtendedContextEnabledFromSettings(): boolean {
|
||||
try {
|
||||
return settings.get("extendedContext");
|
||||
} catch {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/** Authentication material returned to legacy extensions for one model request. */
|
||||
export type ResolvedRequestAuth =
|
||||
| {
|
||||
@@ -176,6 +188,7 @@ export class ModelRegistry {
|
||||
#cacheDbPath?: string;
|
||||
#suppressedSelectors: Map<string, number> = new Map();
|
||||
#backgroundRefresh?: Promise<void>;
|
||||
#policyReapply?: Promise<void>;
|
||||
#lastDiscoveryWarnings: Map<string, string> = new Map();
|
||||
// Runtime extension model overlays — persist across refresh() cycles so that
|
||||
// models registered by extensions survive the model selector's offline reload.
|
||||
@@ -269,6 +282,27 @@ export class ModelRegistry {
|
||||
await this.#refreshRuntimeDiscoveries(strategy);
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild the catalog after a policy-affecting setting change (e.g.
|
||||
* `extendedContext`). Forces the static reload past the models.yml mtime
|
||||
* gate, then restores runtime-discovered models from the SQLite cache —
|
||||
* offline, a settings flip must never hit the network. Concurrent calls
|
||||
* coalesce onto one rebuild.
|
||||
*/
|
||||
reapplyModelPolicies(): Promise<void> {
|
||||
this.#policyReapply ??= this.#runPolicyReapply();
|
||||
return this.#policyReapply;
|
||||
}
|
||||
|
||||
async #runPolicyReapply(): Promise<void> {
|
||||
try {
|
||||
this.#lastStaticLoadMtime = null;
|
||||
await this.refresh("offline");
|
||||
} finally {
|
||||
this.#policyReapply = undefined;
|
||||
}
|
||||
}
|
||||
|
||||
refreshInBackground(strategy: ModelRefreshStrategy = "online-if-uncached"): void {
|
||||
if (this.#backgroundRefresh) {
|
||||
return;
|
||||
@@ -1558,7 +1592,19 @@ export class ModelRegistry {
|
||||
});
|
||||
}
|
||||
#applyHardcodedModelPolicies(models: Model<Api>[]): Model<Api>[] {
|
||||
const extendedContext = isExtendedContextEnabledFromSettings();
|
||||
return models.map(model => {
|
||||
// Extended context off: cap models with a premium long-context price
|
||||
// tier (e.g. GPT-5.6 bills 2x input above 272K) at the standard-pricing
|
||||
// threshold so compaction fires before a request crosses into the tier.
|
||||
// Explicit per-model `contextWindow` overrides reapply later in
|
||||
// composition and win over this cap.
|
||||
if (!extendedContext) {
|
||||
const threshold = model.cost.longContext?.inputThreshold;
|
||||
if (threshold !== undefined && model.contextWindow !== null && model.contextWindow > threshold) {
|
||||
model = applyModelOverride(model, { contextWindow: threshold });
|
||||
}
|
||||
}
|
||||
if (model.provider === "ollama-cloud" && model.omitMaxOutputTokens !== true) {
|
||||
model = applyModelOverride(model, { omitMaxOutputTokens: true });
|
||||
}
|
||||
|
||||
@@ -2149,6 +2149,21 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
// Premium long-context tiers (OpenAI GPT-5.6 bills 2x input / 1.5x output
|
||||
// above 272K input tokens). Off caps affected models at the threshold so
|
||||
// compaction kicks in before any request crosses into premium billing.
|
||||
extendedContext: {
|
||||
type: "boolean",
|
||||
default: true,
|
||||
ui: {
|
||||
tab: "context",
|
||||
group: "General",
|
||||
label: "Extended Context",
|
||||
description:
|
||||
"Use premium long-context windows on models that bill extra past a threshold (e.g. GPT-5.6 1M charges 2x input above 272K); off caps them at the standard-pricing window",
|
||||
},
|
||||
},
|
||||
|
||||
// Compaction
|
||||
"compaction.enabled": {
|
||||
type: "boolean",
|
||||
|
||||
@@ -2466,6 +2466,7 @@ const SETTING_HOOKS: Partial<Record<SettingPath, SettingHook<any>>> = {
|
||||
"hindsight.bankId": () => hindsightScopeSignal.fire(),
|
||||
"hindsight.bankIdPrefix": () => hindsightScopeSignal.fire(),
|
||||
"hindsight.scoping": () => hindsightScopeSignal.fire(),
|
||||
extendedContext: () => extendedContextSignal.fire(),
|
||||
"worktree.base": value => {
|
||||
const dir = typeof value === "string" && value.trim() ? value : undefined;
|
||||
// Always call so an unset/empty value clears a previously-applied override.
|
||||
@@ -2494,6 +2495,17 @@ const modelRolesSignal = new SettingSignal("modelRoles");
|
||||
/** Subscribe to model role changes. Returns an unsubscribe function. */
|
||||
export const onModelRolesChanged: (cb: () => void) => () => void = modelRolesSignal.on.bind(modelRolesSignal);
|
||||
|
||||
/** Fires when `extendedContext` changes at runtime. */
|
||||
const extendedContextSignal = new SettingSignal("extendedContext");
|
||||
|
||||
/**
|
||||
* Subscribe to extended-context setting changes. Sessions re-derive their
|
||||
* model's effective context window (the registry clamps premium long-context
|
||||
* models to the standard-pricing threshold while the setting is off).
|
||||
* Returns an unsubscribe function.
|
||||
*/
|
||||
export const onExtendedContextChanged = (cb: () => void) => extendedContextSignal.on(cb);
|
||||
|
||||
/** Fires when `statusLine.sessionAccent` changes at runtime. */
|
||||
const statusLineSessionAccentSignal = new SettingSignal("statusLine.sessionAccent");
|
||||
|
||||
|
||||
@@ -105,7 +105,7 @@ import type { ResolvedModelRoleValue } from "../config/model-resolver";
|
||||
import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates";
|
||||
import { buildServiceTierByFamily } from "../config/service-tier";
|
||||
import type { Settings, SkillsSettings } from "../config/settings";
|
||||
import { onAppendOnlyModeChanged, onModelRolesChanged } from "../config/settings";
|
||||
import { onAppendOnlyModeChanged, onExtendedContextChanged, onModelRolesChanged } from "../config/settings";
|
||||
import { RawSseDebugBuffer } from "../debug/raw-sse-buffer";
|
||||
import { getFileSnapshotStore } from "../edit/file-snapshot-store";
|
||||
import type { PythonResult } from "../eval/py/executor";
|
||||
@@ -489,6 +489,7 @@ export class AgentSession {
|
||||
#exitRecorded = false;
|
||||
#unsubscribeAppendOnly?: () => void;
|
||||
#unsubscribeModelRoles?: () => void;
|
||||
#unsubscribeExtendedContext?: () => void;
|
||||
/** Last (enable, providerId) tuple resolved by `#syncAppendOnlyContext` — used to skip no-op invalidations. */
|
||||
#lastAppendOnlyResolution?: { enable: boolean; providerId: string | undefined };
|
||||
#eventListeners: AgentSessionEventListener[] = [];
|
||||
@@ -1638,6 +1639,11 @@ export class AgentSession {
|
||||
// Re-evaluate append-only context mode when the setting changes at runtime.
|
||||
this.#unsubscribeAppendOnly = onAppendOnlyModeChanged(_value => this.#syncAppendOnlyContext(this.model));
|
||||
this.#unsubscribeModelRoles = onModelRolesChanged(() => this.#advisors.onModelRolesChanged());
|
||||
// Re-derive the active model's effective context window when the
|
||||
// extended-context setting flips at runtime: the registry re-clamps (or
|
||||
// restores) premium long-context windows, and the live model object must
|
||||
// follow so compaction thresholds and context display react immediately.
|
||||
this.#unsubscribeExtendedContext = onExtendedContextChanged(() => void this.#reapplyExtendedContextPolicy());
|
||||
}
|
||||
/** Model registry for API key resolution and model discovery */
|
||||
get modelRegistry(): ModelRegistry {
|
||||
@@ -4094,6 +4100,10 @@ export class AgentSession {
|
||||
this.#unsubscribeModelRoles();
|
||||
this.#unsubscribeModelRoles = undefined;
|
||||
}
|
||||
if (this.#unsubscribeExtendedContext) {
|
||||
this.#unsubscribeExtendedContext();
|
||||
this.#unsubscribeExtendedContext = undefined;
|
||||
}
|
||||
this.#eventListeners = [];
|
||||
this.#runStateListeners.clear();
|
||||
this.#sessionChangeCallbacks.clear();
|
||||
@@ -7319,6 +7329,26 @@ export class AgentSession {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild the model catalog after an `extendedContext` toggle and rebind the
|
||||
* active model when its effective context window changed. Same-model rebinds
|
||||
* skip provider-session resets (`modelsAreEqual` sees no change), so this
|
||||
* only refreshes metadata consumers (compaction thresholds, context display).
|
||||
*/
|
||||
async #reapplyExtendedContextPolicy(): Promise<void> {
|
||||
try {
|
||||
await this.#modelRegistry.reapplyModelPolicies();
|
||||
const currentModel = this.model;
|
||||
if (!currentModel || this.#isDisposed) return;
|
||||
const updated = this.#modelRegistry.find(currentModel.provider, currentModel.id);
|
||||
if (updated && updated.contextWindow !== currentModel.contextWindow) {
|
||||
await this.#setModelWithProviderSessionReset(updated);
|
||||
}
|
||||
} catch (error) {
|
||||
logger.warn("extended-context policy reapply failed", { error: String(error) });
|
||||
}
|
||||
}
|
||||
|
||||
async #setModelWithProviderSessionReset(model: Model): Promise<void> {
|
||||
const currentModel = this.model;
|
||||
const isChanging = !currentModel || !modelsAreEqual(currentModel, model);
|
||||
|
||||
@@ -7,7 +7,7 @@ import { Effort, type FetchImpl, type Model, type OpenAICompat, type ThinkingCon
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
|
||||
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { resetSettingsForTest, Settings, settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
|
||||
|
||||
@@ -1663,6 +1663,33 @@ describe("ModelRegistry", () => {
|
||||
expect(disabledProbeUrls).toEqual([]);
|
||||
});
|
||||
});
|
||||
describe("extended context", () => {
|
||||
test("off caps premium long-context models at the standard-pricing threshold", async () => {
|
||||
await Settings.init({ inMemory: true, overrides: { extendedContext: false } });
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath);
|
||||
|
||||
// GPT-5.6 bills 2x input above 272K on both the API and Codex.
|
||||
expect(registry.find("openai", "gpt-5.6-sol")?.contextWindow).toBe(272_000);
|
||||
expect(registry.find("openai-codex", "gpt-5.6-sol")?.contextWindow).toBe(272_000);
|
||||
// Standard-priced 1M models (no long-context tier) keep their window.
|
||||
expect(registry.find("anthropic", "claude-opus-4-8")?.contextWindow).toBe(1_000_000);
|
||||
});
|
||||
|
||||
test("reapplyModelPolicies re-clamps and restores premium windows on toggle", async () => {
|
||||
await Settings.init({ inMemory: true });
|
||||
const registry = new ModelRegistry(authStorage, modelsJsonPath);
|
||||
expect(registry.find("openai", "gpt-5.6-terra")?.contextWindow).toBe(1_050_000);
|
||||
|
||||
settings.set("extendedContext", false);
|
||||
await registry.reapplyModelPolicies();
|
||||
expect(registry.find("openai", "gpt-5.6-terra")?.contextWindow).toBe(272_000);
|
||||
|
||||
settings.set("extendedContext", true);
|
||||
await registry.reapplyModelPolicies();
|
||||
expect(registry.find("openai", "gpt-5.6-terra")?.contextWindow).toBe(1_050_000);
|
||||
expect(registry.find("openai-codex", "gpt-5.6-terra")?.contextWindow).toBe(1_000_000);
|
||||
});
|
||||
});
|
||||
describe("bundled Anthropic catalog availability", () => {
|
||||
let anthropicAuth: AuthStorage;
|
||||
let registry: ModelRegistry;
|
||||
|
||||
Reference in New Issue
Block a user