feat: implemented long-context pricing and configuration support

- Added long-context pricing tiers and billing policies for subscription Codex models in the catalog.
- Introduced the `extendedContext` configuration setting to control premium long-context windows.
- Implemented runtime policy refresh and model re-binding when context settings change.
- Added comprehensive unit tests for pricing tiers, context capping, and policy toggling behavior.
This commit is contained in:
can1357
2026-08-20 02:47:32 +02:00
parent 577c0d795f
commit 6d2bae2c41
11 changed files with 2842 additions and 899 deletions
+1
View File
@@ -5,6 +5,7 @@
### Added
- Models now materialize an optional `tokenizer` family in the catalog (`claude-v3`/`v47`/`v5`, Qwen 3.5+, DeepSeek V3/V4/R1, Kimi K2/K3, and GLM-5+). The field follows `requestModelId`, applies to bundled, discovered, and custom models, and can be explicitly overridden in model configuration.
- Subscription Codex GPT-5.6 Sol/Terra/Luna now carry the same `cost.longContext` tier as their first-party API siblings (2x input / 1.5x output above 272K input tokens, [openai/codex#32486](https://github.com/openai/codex/issues/32486)), so cost attribution reflects the higher rating above the threshold and downstream consumers can locate the standard-pricing boundary.
### Fixed
@@ -508,12 +508,17 @@ function inferGeneratedApplyPatchToolType(
function applyOpenAICatalogPolicy(model: ModelSpec<Api>, parsedModel: OpenAIModel): void {
const isFirstPartyResponses = model.provider === "openai" && model.api === "openai-responses";
// Subscription Codex rates usage at the same >272K long-context tier as the
// API (openai/codex#32486), so first-party Codex SKUs carry the tier too —
// it drives both cost attribution and the extended-context window clamp.
const isFirstPartyCodex = model.provider === "openai-codex" && model.api === "openai-codex-responses";
if (isFirstPartyResponses && modelOrRequestIdValue(model, OPENAI_NONE_EFFORT_MODEL_IDS)) {
model.compat = { ...(model.compat ?? {}), reasoningDisableMode: "none-effort" };
}
const longContextCost = isFirstPartyResponses
? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID)
: undefined;
const longContextCost =
isFirstPartyResponses || isFirstPartyCodex
? modelOrRequestIdValue(model, OPENAI_GPT_5_6_LONG_CONTEXT_COST_BY_MODEL_ID)
: undefined;
if (longContextCost) {
model.cost = { ...model.cost, longContext: longContextCost };
}
File diff suppressed because it is too large Load Diff
@@ -162,6 +162,21 @@ describe("generated model policies", () => {
expect(models[4]?.contextWindow).toBe(272000);
});
it("applies GPT-5.6 long-context pricing to Codex-transport SKUs (openai/codex#32486)", () => {
const models: ModelSpec<Api>[] = [
createSpec({ id: "gpt-5.6-sol", api: "openai-codex-responses", provider: "openai-codex" }),
createSpec({ id: "gpt-5.6-luna", api: "openai-codex-responses", provider: "openai-codex" }),
// Third-party carriers of the same id must not inherit the tier.
createSpec({ id: "gpt-5.6-sol", api: "openai-completions", provider: "openrouter" }),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.cost.longContext).toMatchObject({ inputThreshold: 272_000, input: 10, output: 45 });
expect(models[1]?.cost.longContext).toMatchObject({ inputThreshold: 272_000, input: 0.4, output: 1.8 });
expect(models[2]?.cost.longContext).toBeUndefined();
});
it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => {
const models: ModelSpec<Api>[] = [
createSpec({
@@ -0,0 +1,46 @@
import { describe, expect, it } from "bun:test";
import { calculateCost, getBundledModel } from "@oh-my-pi/pi-catalog/models";
import type { Usage } from "@oh-my-pi/pi-catalog/types";
function usage(fields: Pick<Usage, "input" | "output" | "cacheRead" | "cacheWrite">): Usage {
return {
...fields,
totalTokens: fields.input + fields.output + fields.cacheRead + fields.cacheWrite,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
describe("long-context pricing tier", () => {
// GPT-5.6 Sol: $5/$30 standard, $10/$45 above 272K input tokens; the tier
// rate applies to the ENTIRE request once total prompt input (input +
// cacheRead + cacheWrite) crosses the threshold, matching OpenAI's billing.
const sol = getBundledModel("openai", "gpt-5.6-sol");
it("bills at standard rates at or below the 272K input threshold", () => {
const atThreshold = usage({ input: 72_000, output: 10_000, cacheRead: 200_000, cacheWrite: 0 });
calculateCost(sol, atThreshold);
expect(atThreshold.cost.input).toBeCloseTo((5 / 1e6) * 72_000, 10);
expect(atThreshold.cost.output).toBeCloseTo((30 / 1e6) * 10_000, 10);
expect(atThreshold.cost.cacheRead).toBeCloseTo((0.5 / 1e6) * 200_000, 10);
});
it("bills the whole request at tier rates once prompt input crosses the threshold", () => {
const overThreshold = usage({ input: 72_001, output: 10_000, cacheRead: 200_000, cacheWrite: 0 });
calculateCost(sol, overThreshold);
expect(overThreshold.cost.input).toBeCloseTo((10 / 1e6) * 72_001, 10);
expect(overThreshold.cost.output).toBeCloseTo((45 / 1e6) * 10_000, 10);
expect(overThreshold.cost.cacheRead).toBeCloseTo((1 / 1e6) * 200_000, 10);
expect(overThreshold.cost.total).toBeCloseTo(
overThreshold.cost.input + overThreshold.cost.output + overThreshold.cost.cacheRead,
10,
);
});
it("prices subscription Codex SKUs with the same tier for cost attribution", () => {
const codexSol = getBundledModel("openai-codex", "gpt-5.6-sol");
const overThreshold = usage({ input: 300_000, output: 1_000, cacheRead: 0, cacheWrite: 0 });
calculateCost(codexSol, overThreshold);
expect(overThreshold.cost.input).toBeCloseTo((10 / 1e6) * 300_000, 10);
expect(overThreshold.cost.output).toBeCloseTo((45 / 1e6) * 1_000, 10);
});
});
+1
View File
@@ -6,6 +6,7 @@
- Added `qwenTemplateReasoningEffort` to the `models.yml` `compat` schema, so the auto-enabled Qwen 3.8+ template effort dialect (`chat_template_kwargs.reasoning_effort`) can be switched off per provider/model for strict local servers that reject unknown `chat_template_kwargs`.
- Added `tokenizer` to custom model and `modelOverrides` configuration. It overrides the catalog-resolved local tokenizer family for a model when a proxy serves a known model id with a different tokenizer.
- Added `extendedContext` setting (`/settings` → Context → General, default on). When off, models with a premium long-context price tier (OpenAI GPT-5.6 Sol/Terra/Luna bill 2x input / 1.5x output above 272K input tokens, on both the API and subscription Codex) are capped at the standard-pricing threshold — they appear as 272K again and compaction fires before a request crosses into premium billing. Toggling mid-session re-clamps or restores the active model's window immediately. Anthropic Claude 4.6+ serves its full 1M window at standard pricing, so no Anthropic model is affected.
### Changed
@@ -140,6 +140,18 @@ function getDisabledProviderIdsFromSettings(): Set<string> {
}
}
/**
* Whether premium long-context windows are enabled. Defaults to true when the
* settings singleton is not initialized (SDK embedding, early boot).
*/
function isExtendedContextEnabledFromSettings(): boolean {
try {
return settings.get("extendedContext");
} catch {
return true;
}
}
/** Authentication material returned to legacy extensions for one model request. */
export type ResolvedRequestAuth =
| {
@@ -176,6 +188,7 @@ export class ModelRegistry {
#cacheDbPath?: string;
#suppressedSelectors: Map<string, number> = new Map();
#backgroundRefresh?: Promise<void>;
#policyReapply?: Promise<void>;
#lastDiscoveryWarnings: Map<string, string> = new Map();
// Runtime extension model overlays — persist across refresh() cycles so that
// models registered by extensions survive the model selector's offline reload.
@@ -269,6 +282,27 @@ export class ModelRegistry {
await this.#refreshRuntimeDiscoveries(strategy);
}
/**
* Rebuild the catalog after a policy-affecting setting change (e.g.
* `extendedContext`). Forces the static reload past the models.yml mtime
* gate, then restores runtime-discovered models from the SQLite cache —
* offline, a settings flip must never hit the network. Concurrent calls
* coalesce onto one rebuild.
*/
reapplyModelPolicies(): Promise<void> {
this.#policyReapply ??= this.#runPolicyReapply();
return this.#policyReapply;
}
async #runPolicyReapply(): Promise<void> {
try {
this.#lastStaticLoadMtime = null;
await this.refresh("offline");
} finally {
this.#policyReapply = undefined;
}
}
refreshInBackground(strategy: ModelRefreshStrategy = "online-if-uncached"): void {
if (this.#backgroundRefresh) {
return;
@@ -1558,7 +1592,19 @@ export class ModelRegistry {
});
}
#applyHardcodedModelPolicies(models: Model<Api>[]): Model<Api>[] {
const extendedContext = isExtendedContextEnabledFromSettings();
return models.map(model => {
// Extended context off: cap models with a premium long-context price
// tier (e.g. GPT-5.6 bills 2x input above 272K) at the standard-pricing
// threshold so compaction fires before a request crosses into the tier.
// Explicit per-model `contextWindow` overrides reapply later in
// composition and win over this cap.
if (!extendedContext) {
const threshold = model.cost.longContext?.inputThreshold;
if (threshold !== undefined && model.contextWindow !== null && model.contextWindow > threshold) {
model = applyModelOverride(model, { contextWindow: threshold });
}
}
if (model.provider === "ollama-cloud" && model.omitMaxOutputTokens !== true) {
model = applyModelOverride(model, { omitMaxOutputTokens: true });
}
@@ -2149,6 +2149,21 @@ export const SETTINGS_SCHEMA = {
},
},
// Premium long-context tiers (OpenAI GPT-5.6 bills 2x input / 1.5x output
// above 272K input tokens). Off caps affected models at the threshold so
// compaction kicks in before any request crosses into premium billing.
extendedContext: {
type: "boolean",
default: true,
ui: {
tab: "context",
group: "General",
label: "Extended Context",
description:
"Use premium long-context windows on models that bill extra past a threshold (e.g. GPT-5.6 1M charges 2x input above 272K); off caps them at the standard-pricing window",
},
},
// Compaction
"compaction.enabled": {
type: "boolean",
@@ -2466,6 +2466,7 @@ const SETTING_HOOKS: Partial<Record<SettingPath, SettingHook<any>>> = {
"hindsight.bankId": () => hindsightScopeSignal.fire(),
"hindsight.bankIdPrefix": () => hindsightScopeSignal.fire(),
"hindsight.scoping": () => hindsightScopeSignal.fire(),
extendedContext: () => extendedContextSignal.fire(),
"worktree.base": value => {
const dir = typeof value === "string" && value.trim() ? value : undefined;
// Always call so an unset/empty value clears a previously-applied override.
@@ -2494,6 +2495,17 @@ const modelRolesSignal = new SettingSignal("modelRoles");
/** Subscribe to model role changes. Returns an unsubscribe function. */
export const onModelRolesChanged: (cb: () => void) => () => void = modelRolesSignal.on.bind(modelRolesSignal);
/** Fires when `extendedContext` changes at runtime. */
const extendedContextSignal = new SettingSignal("extendedContext");
/**
* Subscribe to extended-context setting changes. Sessions re-derive their
* model's effective context window (the registry clamps premium long-context
* models to the standard-pricing threshold while the setting is off).
* Returns an unsubscribe function.
*/
export const onExtendedContextChanged = (cb: () => void) => extendedContextSignal.on(cb);
/** Fires when `statusLine.sessionAccent` changes at runtime. */
const statusLineSessionAccentSignal = new SettingSignal("statusLine.sessionAccent");
@@ -105,7 +105,7 @@ import type { ResolvedModelRoleValue } from "../config/model-resolver";
import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates";
import { buildServiceTierByFamily } from "../config/service-tier";
import type { Settings, SkillsSettings } from "../config/settings";
import { onAppendOnlyModeChanged, onModelRolesChanged } from "../config/settings";
import { onAppendOnlyModeChanged, onExtendedContextChanged, onModelRolesChanged } from "../config/settings";
import { RawSseDebugBuffer } from "../debug/raw-sse-buffer";
import { getFileSnapshotStore } from "../edit/file-snapshot-store";
import type { PythonResult } from "../eval/py/executor";
@@ -489,6 +489,7 @@ export class AgentSession {
#exitRecorded = false;
#unsubscribeAppendOnly?: () => void;
#unsubscribeModelRoles?: () => void;
#unsubscribeExtendedContext?: () => void;
/** Last (enable, providerId) tuple resolved by `#syncAppendOnlyContext` — used to skip no-op invalidations. */
#lastAppendOnlyResolution?: { enable: boolean; providerId: string | undefined };
#eventListeners: AgentSessionEventListener[] = [];
@@ -1638,6 +1639,11 @@ export class AgentSession {
// Re-evaluate append-only context mode when the setting changes at runtime.
this.#unsubscribeAppendOnly = onAppendOnlyModeChanged(_value => this.#syncAppendOnlyContext(this.model));
this.#unsubscribeModelRoles = onModelRolesChanged(() => this.#advisors.onModelRolesChanged());
// Re-derive the active model's effective context window when the
// extended-context setting flips at runtime: the registry re-clamps (or
// restores) premium long-context windows, and the live model object must
// follow so compaction thresholds and context display react immediately.
this.#unsubscribeExtendedContext = onExtendedContextChanged(() => void this.#reapplyExtendedContextPolicy());
}
/** Model registry for API key resolution and model discovery */
get modelRegistry(): ModelRegistry {
@@ -4094,6 +4100,10 @@ export class AgentSession {
this.#unsubscribeModelRoles();
this.#unsubscribeModelRoles = undefined;
}
if (this.#unsubscribeExtendedContext) {
this.#unsubscribeExtendedContext();
this.#unsubscribeExtendedContext = undefined;
}
this.#eventListeners = [];
this.#runStateListeners.clear();
this.#sessionChangeCallbacks.clear();
@@ -7319,6 +7329,26 @@ export class AgentSession {
return true;
}
/**
* Rebuild the model catalog after an `extendedContext` toggle and rebind the
* active model when its effective context window changed. Same-model rebinds
* skip provider-session resets (`modelsAreEqual` sees no change), so this
* only refreshes metadata consumers (compaction thresholds, context display).
*/
async #reapplyExtendedContextPolicy(): Promise<void> {
try {
await this.#modelRegistry.reapplyModelPolicies();
const currentModel = this.model;
if (!currentModel || this.#isDisposed) return;
const updated = this.#modelRegistry.find(currentModel.provider, currentModel.id);
if (updated && updated.contextWindow !== currentModel.contextWindow) {
await this.#setModelWithProviderSessionReset(updated);
}
} catch (error) {
logger.warn("extended-context policy reapply failed", { error: String(error) });
}
}
async #setModelWithProviderSessionReset(model: Model): Promise<void> {
const currentModel = this.model;
const isChanging = !currentModel || !modelsAreEqual(currentModel, model);
@@ -7,7 +7,7 @@ import { Effort, type FetchImpl, type Model, type OpenAICompat, type ThinkingCon
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { resetSettingsForTest, Settings, settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
@@ -1663,6 +1663,33 @@ describe("ModelRegistry", () => {
expect(disabledProbeUrls).toEqual([]);
});
});
describe("extended context", () => {
test("off caps premium long-context models at the standard-pricing threshold", async () => {
await Settings.init({ inMemory: true, overrides: { extendedContext: false } });
const registry = new ModelRegistry(authStorage, modelsJsonPath);
// GPT-5.6 bills 2x input above 272K on both the API and Codex.
expect(registry.find("openai", "gpt-5.6-sol")?.contextWindow).toBe(272_000);
expect(registry.find("openai-codex", "gpt-5.6-sol")?.contextWindow).toBe(272_000);
// Standard-priced 1M models (no long-context tier) keep their window.
expect(registry.find("anthropic", "claude-opus-4-8")?.contextWindow).toBe(1_000_000);
});
test("reapplyModelPolicies re-clamps and restores premium windows on toggle", async () => {
await Settings.init({ inMemory: true });
const registry = new ModelRegistry(authStorage, modelsJsonPath);
expect(registry.find("openai", "gpt-5.6-terra")?.contextWindow).toBe(1_050_000);
settings.set("extendedContext", false);
await registry.reapplyModelPolicies();
expect(registry.find("openai", "gpt-5.6-terra")?.contextWindow).toBe(272_000);
settings.set("extendedContext", true);
await registry.reapplyModelPolicies();
expect(registry.find("openai", "gpt-5.6-terra")?.contextWindow).toBe(1_050_000);
expect(registry.find("openai-codex", "gpt-5.6-terra")?.contextWindow).toBe(1_000_000);
});
});
describe("bundled Anthropic catalog availability", () => {
let anthropicAuth: AuthStorage;
let registry: ModelRegistry;