fix: handled unknown model limits as null to avoid artificial token caps

- Replaced unknown model contextWindow/maxTokens sentinels with nullable values across types and catalog data.
- Mapped request token calculations to treat null maxTokens as unlimited output caps.
- Updated remote compaction and context checks to ignore unknown limits by using Infinity/0 fallbacks.
- Adjusted CLI/model registry flows to skip cap enforcement for null limits and render unknown values as '-'.
This commit is contained in:
can1357
2026-06-13 15:35:40 +02:00
parent 2a7e10cc44
commit f0c6a54f51
43 changed files with 3014 additions and 2798 deletions
+3
View File
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
### Fixed
- Fixed remote compaction input trimming to use unlimited context when `model.contextWindow` is unset
## [15.12.1] - 2026-06-12
### Breaking Changes
+1 -1
View File
@@ -456,7 +456,7 @@ export async function requestOpenAiRemoteCompaction(
const endpoint = resolveOpenAiCompactEndpoint(model);
const request: OpenAiRemoteCompactionRequest = {
model: model.id,
input: trimOpenAiCompactInput(compactInput, model.contextWindow, instructions),
input: trimOpenAiCompactInput(compactInput, model.contextWindow ?? Number.POSITIVE_INFINITY, instructions),
instructions,
};
const headers: Record<string, string> = {
+1
View File
@@ -11,6 +11,7 @@
### Fixed
- Fixed: provider request builders treat unknown `model.maxTokens` (`null`) as "no model cap" instead of coercing to `0` via `Math.min`; Anthropic falls back to the 64k Claude-Code cap for its required `max_tokens`.
- Fixed transient stream failures on OpenAI-compatible providers by retrying HTTP 408/429/5xx responses and transient network errors with Retry-After/quota-hint aware backoff
- Fixed SSE stream handling for OpenAI-compatible responses by parsing wire-level JSON frames directly and honoring `[DONE]` termination
- Fixed stream error handling for OpenAI-compatible providers by preserving structured HTTP status/headers and response body details from failed requests for retry and strict-tool fallback logic
+3 -2
View File
@@ -2817,7 +2817,8 @@ function buildParams(
// Claude Code requests at most 64k output tokens; clamp only OAuth requests,
// where the wire fingerprint must match. API-key callers keep the full model
// ceiling (e.g. 128k on Opus 4.8).
const maxOutputTokens = isOAuthToken ? Math.min(CLAUDE_CODE_MAX_OUTPUT_TOKENS, model.maxTokens) : model.maxTokens;
const modelMaxTokens = model.maxTokens ?? CLAUDE_CODE_MAX_OUTPUT_TOKENS;
const maxOutputTokens = isOAuthToken ? Math.min(CLAUDE_CODE_MAX_OUTPUT_TOKENS, modelMaxTokens) : modelMaxTokens;
// Build params in the canonical field order: model → messages → system → tools →
// metadata → max_tokens → thinking → context_management → output_config → stream.
@@ -2827,7 +2828,7 @@ function buildParams(
...(systemBlocks && { system: systemBlocks }),
...(tools !== undefined && { tools }),
...(metadata && { metadata }),
max_tokens: Math.min(maxOutputTokens, options?.maxTokens || model.maxTokens),
max_tokens: Math.min(maxOutputTokens, options?.maxTokens || modelMaxTokens),
...(thinking && { thinking }),
...(contextManagement && { context_management: contextManagement }),
...(outputConfig && { output_config: outputConfig }),
+3 -3
View File
@@ -275,7 +275,7 @@ export function streamGitLabDuo(
minP: options.minP,
presencePenalty: options.presencePenalty,
repetitionPenalty: options.repetitionPenalty,
maxTokens: options.maxTokens ?? model.maxTokens,
maxTokens: options.maxTokens ?? model.maxTokens ?? undefined,
signal: options.signal,
cacheRetention: options.cacheRetention,
headers,
@@ -313,7 +313,7 @@ export function streamGitLabDuo(
minP: options.minP,
presencePenalty: options.presencePenalty,
repetitionPenalty: options.repetitionPenalty,
maxTokens: options.maxTokens ?? model.maxTokens,
maxTokens: options.maxTokens ?? model.maxTokens ?? undefined,
signal: options.signal,
cacheRetention: options.cacheRetention,
headers,
@@ -346,7 +346,7 @@ export function streamGitLabDuo(
minP: options.minP,
presencePenalty: options.presencePenalty,
repetitionPenalty: options.repetitionPenalty,
maxTokens: options.maxTokens ?? model.maxTokens,
maxTokens: options.maxTokens ?? model.maxTokens ?? undefined,
signal: options.signal,
cacheRetention: options.cacheRetention,
headers,
@@ -85,7 +85,7 @@ export function streamOpenAIAnthropicShim(
minP: options?.minP,
presencePenalty: options?.presencePenalty,
repetitionPenalty: options?.repetitionPenalty,
maxTokens: options?.maxTokens ?? model.maxTokens,
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
signal: options?.signal,
headers: mergedHeaders,
sessionId: options?.sessionId,
@@ -119,7 +119,7 @@ export function streamOpenAIAnthropicShim(
minP: options?.minP,
presencePenalty: options?.presencePenalty,
repetitionPenalty: options?.repetitionPenalty,
maxTokens: options?.maxTokens ?? model.maxTokens,
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
signal: options?.signal,
headers: mergedHeaders,
sessionId: options?.sessionId,
@@ -1239,7 +1239,8 @@ function buildParams(
// Kimi-family models calculate TPM rate limits from max_tokens (not actual
// output) and the official guidance requires sending it on every call —
// `compat.alwaysSendMaxTokens` carries that detection.
const requestedMaxTokens = options?.maxTokens ?? (compat.alwaysSendMaxTokens ? model.maxTokens : undefined);
const requestedMaxTokens =
options?.maxTokens ?? (compat.alwaysSendMaxTokens ? (model.maxTokens ?? undefined) : undefined);
// OpenRouter fans out to upstreams whose output caps differ from the catalog
// value (which tracks the highest-cap provider). A max_tokens above the routed
// upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras
@@ -1250,7 +1251,7 @@ function buildParams(
const effectiveMaxTokens =
requestedMaxTokens === undefined || omitMaxTokensForRouting
? undefined
: Math.min(requestedMaxTokens, model.maxTokens, OPENAI_MAX_OUTPUT_TOKENS);
: Math.min(requestedMaxTokens, model.maxTokens ?? Number.POSITIVE_INFINITY, OPENAI_MAX_OUTPUT_TOKENS);
const requestModelId = resolveOpenAICompletionsModelId(model, options);
const params: OpenAICompletionsParams = {
@@ -1025,7 +1025,11 @@ export function applyCommonResponsesSamplingParams<P extends CommonResponsesPara
model: Pick<Model, "provider" | "omitMaxOutputTokens" | "maxTokens">,
): void {
if (options?.maxTokens && !model.omitMaxOutputTokens) {
params.max_output_tokens = Math.min(options.maxTokens, model.maxTokens, OPENAI_MAX_OUTPUT_TOKENS);
params.max_output_tokens = Math.min(
options.maxTokens,
model.maxTokens ?? Number.POSITIVE_INFINITY,
OPENAI_MAX_OUTPUT_TOKENS,
);
}
if (options?.temperature !== undefined) params.temperature = options.temperature;
if (options?.topP !== undefined) params.top_p = options.topP;
+15 -5
View File
@@ -557,6 +557,7 @@ export async function completeSimple<TApi extends Api>(
}
const MIN_OUTPUT_TOKENS = 1024;
const OUTPUT_CAP_WHEN_UNKNOWN = 64_000;
export const OUTPUT_FALLBACK_BUFFER = 4000;
const ANTHROPIC_USE_INTERLEAVED_THINKING = Bun.env.PI_NO_INTERLEAVED_THINKING !== "1";
@@ -709,7 +710,7 @@ function mapOptionsForApi<TApi extends Api>(
minP: options?.minP,
presencePenalty: options?.presencePenalty,
repetitionPenalty: options?.repetitionPenalty,
maxTokens: options?.maxTokens ?? model.maxTokens,
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
signal: options?.signal,
apiKey: apiKey ?? (typeof options?.apiKey === "string" ? options.apiKey : undefined),
cacheRetention: options?.cacheRetention,
@@ -785,7 +786,10 @@ function mapOptionsForApi<TApi extends Api>(
}
// Caller's maxTokens is the desired output; add thinking budget on top, capped at model limit
const maxTokens = Math.min((base.maxTokens || 0) + thinkingBudget, model.maxTokens);
const maxTokens = Math.min(
(base.maxTokens || 0) + thinkingBudget,
model.maxTokens ?? Number.POSITIVE_INFINITY,
);
// If not enough room for thinking + output, reduce thinking budget
if (maxTokens <= thinkingBudget) {
@@ -830,10 +834,13 @@ function mapOptionsForApi<TApi extends Api>(
}
const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
if (!budgetInfo) return bedrockBase as OptionsForApi<TApi>;
let maxTokens = bedrockBase.maxTokens ?? model.maxTokens;
let maxTokens = bedrockBase.maxTokens ?? model.maxTokens ?? OUTPUT_CAP_WHEN_UNKNOWN;
let thinkingBudgets = bedrockBase.thinkingBudgets;
if (maxTokens <= budgetInfo.budget) {
const desiredMaxTokens = Math.min(model.maxTokens, budgetInfo.budget + MIN_OUTPUT_TOKENS);
const desiredMaxTokens = Math.min(
model.maxTokens ?? Number.POSITIVE_INFINITY,
budgetInfo.budget + MIN_OUTPUT_TOKENS,
);
if (desiredMaxTokens > maxTokens) {
maxTokens = desiredMaxTokens;
}
@@ -943,7 +950,10 @@ function mapOptionsForApi<TApi extends Api>(
let thinkingBudget = options.thinkingBudgets?.[effort] ?? GOOGLE_THINKING[effort];
// Caller's maxTokens is the desired output; add thinking budget on top, capped at model limit
const maxTokens = Math.min((base.maxTokens || 0) + thinkingBudget, model.maxTokens);
const maxTokens = Math.min(
(base.maxTokens || 0) + thinkingBudget,
model.maxTokens ?? Number.POSITIVE_INFINITY,
);
// If not enough room for thinking + output, reduce thinking budget
if (maxTokens <= thinkingBudget) {
+9 -4
View File
@@ -16,10 +16,15 @@ import type { ChildProcess } from "node:child_process";
import { execSync, spawn } from "node:child_process";
import { complete } from "@oh-my-pi/pi-ai/stream";
import type { AssistantMessage, Context, Model, Usage } from "@oh-my-pi/pi-ai/types";
import { isContextOverflow } from "@oh-my-pi/pi-ai/utils/overflow";
import { isContextOverflow as originalIsContextOverflow } from "@oh-my-pi/pi-ai/utils/overflow";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { $which } from "@oh-my-pi/pi-utils";
function isContextOverflow(message: AssistantMessage, contextWindow: number | null): boolean {
return originalIsContextOverflow(message, contextWindow ?? 0);
}
import { e2eApiKey, resolveApiKey } from "./oauth";
// Resolve OAuth tokens at module level (async, runs before tests)
@@ -55,7 +60,7 @@ interface OverflowResult {
}
async function testContextOverflow(model: Model, apiKey: string): Promise<OverflowResult> {
const overflowContent = generateOverflowContent(model.contextWindow);
const overflowContent = generateOverflowContent(model.contextWindow ?? 0);
const context: Context = {
systemPrompt: ["You are a helpful assistant."],
@@ -75,7 +80,7 @@ async function testContextOverflow(model: Model, apiKey: string): Promise<Overfl
return {
provider: model.provider,
model: model.id,
contextWindow: model.contextWindow,
contextWindow: model.contextWindow ?? 0,
stopReason: response.stopReason,
errorMessage: response.errorMessage,
usage: response.usage,
@@ -459,7 +464,7 @@ describe("Context overflow error handling", () => {
// Either way, isContextOverflow should detect it (via usage check or we skip if rate limited)
if (result.stopReason === "stop") {
expect(result.hasUsageData).toBe(true);
expect(result.usage.input).toBeGreaterThan(model.contextWindow);
expect(result.usage.input).toBeGreaterThan(model.contextWindow ?? 0);
expect(isContextOverflow(result.response, model.contextWindow)).toBe(true);
} else {
// Rate limited or other error - just log and pass
@@ -0,0 +1,139 @@
import { describe, expect, it } from "bun:test";
import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { Context, FetchImpl, ModelSpec } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
const ctx: Context = {
systemPrompt: ["hi"],
messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
};
function completionsSse(): Response {
const events: unknown[] = [
{
id: "c",
object: "chat.completion.chunk",
choices: [{ index: 0, delta: { content: "ok" } }],
},
{
id: "c",
object: "chat.completion.chunk",
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
},
"[DONE]",
];
const payload = `${events.map(e => `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`;
return new Response(payload, { status: 200, headers: { "content-type": "text/event-stream" } });
}
function responsesSse(): Response {
const event = {
type: "response.completed",
response: {
status: "completed",
usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2, input_tokens_details: { cached_tokens: 0 } },
},
};
return new Response(`data: ${JSON.stringify(event)}\n\n`, {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
describe("null maxTokens fallback wire tests", () => {
it("verifies anthropic messages wire format max_tokens fallback is finite when maxTokens is null", async () => {
const spec: ModelSpec<"anthropic-messages"> = {
id: "claude-custom",
name: "Claude Custom",
api: "anthropic-messages",
provider: "anthropic",
baseUrl: "https://api.anthropic.com",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 100000,
maxTokens: null, // unknown limit
};
const model = buildModel(spec);
let capturedPayload: Record<string, unknown> | null = null;
const mockFetch: FetchImpl = async (_url, init) => {
capturedPayload = JSON.parse(init?.body as string) as Record<string, unknown>;
return new Response("{}", { status: 400 }); // we just want to capture
};
// streamAnthropic triggers fetch. We ignore errors.
try {
await streamAnthropic(model, ctx, { apiKey: "test-key", fetch: mockFetch }).result();
} catch {
// expected 400
}
expect(capturedPayload).not.toBeNull();
expect(capturedPayload!.max_tokens).toBe(64000); // fallback CLAUDE_CODE_MAX_OUTPUT_TOKENS
});
it("verifies openai completions clamps to OPENAI_MAX_OUTPUT_TOKENS when maxTokens is null", async () => {
const spec: ModelSpec<"openai-completions"> = {
id: "gpt-custom",
name: "GPT Custom",
api: "openai-completions",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 100000,
maxTokens: null, // unknown limit
};
const model = buildModel(spec);
let capturedPayload: Record<string, unknown> | null = null;
const mockFetch: FetchImpl = async (_url, init) => {
capturedPayload = JSON.parse(init?.body as string) as Record<string, unknown>;
return completionsSse();
};
await streamOpenAICompletions(model, ctx, {
apiKey: "test-key",
maxTokens: 100000, // requested max tokens
fetch: mockFetch,
}).result();
expect(capturedPayload).not.toBeNull();
expect(capturedPayload!.max_completion_tokens).toBe(64000); // clamps to OPENAI_MAX_OUTPUT_TOKENS
});
it("verifies openai responses clamps to OPENAI_MAX_OUTPUT_TOKENS when maxTokens is null", async () => {
const spec: ModelSpec<"openai-responses"> = {
id: "gpt-responses-custom",
name: "GPT Responses Custom",
api: "openai-responses",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 100000,
maxTokens: null, // unknown limit
};
const model = buildModel(spec);
let capturedPayload: Record<string, unknown> | null = null;
const mockFetch: FetchImpl = async (_url, init) => {
capturedPayload = JSON.parse(init?.body as string) as Record<string, unknown>;
return responsesSse();
};
await streamOpenAIResponses(model, ctx, {
apiKey: "test-key",
maxTokens: 100000, // requested max tokens
fetch: mockFetch,
}).result();
expect(capturedPayload).not.toBeNull();
expect(capturedPayload!.max_output_tokens).toBe(64000); // clamps to OPENAI_MAX_OUTPUT_TOKENS
});
});
+1
View File
@@ -7,6 +7,7 @@
- Changed
### Changed
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
- Changed the `github-copilot` model context window to `524288` tokens
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
+7 -10
View File
@@ -34,8 +34,6 @@ import {
MODELS_DEV_PROVIDER_DESCRIPTORS,
mapModelsDevToModels,
stripFireworksDeepSeekThinkingToggle,
UNK_CONTEXT_WINDOW,
UNK_MAX_TOKENS,
} from "../src/provider-models/openai-compat";
import type { ModelSpec } from "../src/types";
import { cleanModelName } from "../src/utils";
@@ -158,21 +156,20 @@ function createGlobalModelsDevReferenceMap(modelsDevModels: readonly ModelSpec[]
references.set(model.id, model);
continue;
}
if (model.contextWindow > existing.contextWindow) {
if ((model.contextWindow ?? 0) > (existing.contextWindow ?? 0)) {
references.set(model.id, model);
continue;
}
if (model.contextWindow === existing.contextWindow && model.maxTokens > existing.maxTokens) {
if (
(model.contextWindow ?? 0) === (existing.contextWindow ?? 0) &&
(model.maxTokens ?? 0) > (existing.maxTokens ?? 0)
) {
references.set(model.id, model);
}
}
return references;
}
function inheritModelsDevLimit(value: number, referenceValue: number, unspecifiedValue: number): number {
return value === unspecifiedValue ? referenceValue : value;
}
function applyGlobalModelsDevFallback(
models: readonly ModelSpec[],
modelsDevModels: readonly ModelSpec[],
@@ -194,8 +191,8 @@ function applyGlobalModelsDevFallback(
input: reference.input,
// Fill unknown endpoint limits from same-id models.dev references, but keep
// provider-specific values when discovery returned them explicitly.
contextWindow: inheritModelsDevLimit(model.contextWindow, reference.contextWindow, UNK_CONTEXT_WINDOW),
maxTokens: inheritModelsDevLimit(model.maxTokens, reference.maxTokens, UNK_MAX_TOKENS),
contextWindow: model.contextWindow ?? reference.contextWindow,
maxTokens: model.maxTokens ?? reference.maxTokens,
};
});
}
+5 -4
View File
@@ -1,7 +1,6 @@
import { z } from "zod/v4";
import { getBundledModels } from "../models";
import { toModelSpec } from "../provider-models/bundled-references";
import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "../provider-models/discovery-constants";
import type { FetchImpl, Model, ModelSpec } from "../types";
const GOOGLE_GENERATIVE_AI_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
@@ -148,7 +147,9 @@ function normalizeBaseUrl(baseUrl?: string): string {
return value.replace(/\/+$/, "");
}
function normalizePositiveInt(value: number | undefined, fallback: number): number {
function normalizePositiveInt(value: number | undefined, fallback: number): number;
function normalizePositiveInt(value: number | undefined, fallback: number | null): number | null;
function normalizePositiveInt(value: number | undefined, fallback: number | null): number | null {
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) {
return fallback;
}
@@ -178,8 +179,8 @@ function normalizeModel(
}
const reference = bundledById.get(id);
const contextWindow = normalizePositiveInt(item.inputTokenLimit, reference?.contextWindow ?? UNK_CONTEXT_WINDOW);
const maxTokens = normalizePositiveInt(item.outputTokenLimit, reference?.maxTokens ?? UNK_MAX_TOKENS);
const contextWindow = normalizePositiveInt(item.inputTokenLimit, reference?.contextWindow ?? null);
const maxTokens = normalizePositiveInt(item.outputTokenLimit, reference?.maxTokens ?? null);
const name = normalizeModelName(item.displayName, reference?.name ?? id);
if (reference) {
@@ -1,5 +1,4 @@
import { z } from "zod/v4";
import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "../provider-models/discovery-constants";
import type { Api, FetchImpl, ModelSpec, Provider } from "../types";
const MODELS_PATH = "/models";
@@ -165,8 +164,8 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: UNK_CONTEXT_WINDOW,
maxTokens: UNK_MAX_TOKENS,
contextWindow: null,
maxTokens: null,
};
// `mapModel` returning null skips the entry (documented contract); only a
+2 -2
View File
@@ -61,10 +61,10 @@ const FAMILY_EXTRACTION_PATTERNS = [
function shouldReplaceReference(existing: Model<Api> | undefined, candidate: Model<Api>): boolean {
if (!existing) return true;
if (candidate.contextWindow !== existing.contextWindow) {
return candidate.contextWindow > existing.contextWindow;
return (candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0);
}
if (candidate.maxTokens !== existing.maxTokens) {
return candidate.maxTokens > existing.maxTokens;
return (candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0);
}
return existing.provider !== "openai" && candidate.provider === "openai";
}
+2 -2
View File
@@ -34,10 +34,10 @@ export function isZeroCostXaiOAuthReference(candidate: Model<Api>): boolean {
function shouldReplaceReference(existing: Model<Api> | undefined, candidate: Model<Api>): boolean {
if (!existing) return true;
if (candidate.contextWindow !== existing.contextWindow) {
return candidate.contextWindow > existing.contextWindow;
return (candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0);
}
if (candidate.maxTokens !== existing.maxTokens) {
return candidate.maxTokens > existing.maxTokens;
return (candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0);
}
const existingHasCachePricing = existing.cost.cacheRead > 0 || existing.cost.cacheWrite > 0;
const candidateHasCachePricing = candidate.cost.cacheRead > 0 || candidate.cost.cacheWrite > 0;
+7 -5
View File
@@ -363,11 +363,13 @@ function preferDiscoveryName(discoveryName: string, fallbackName: string, modelI
return normalizedDiscoveryName;
}
function preferDiscoveryLimit(discoveryLimit: number, fallbackLimit: number): number {
if (!Number.isFinite(discoveryLimit) || discoveryLimit <= 0) {
function preferDiscoveryLimit(discoveryLimit: number, fallbackLimit: number): number;
function preferDiscoveryLimit(discoveryLimit: number | null, fallbackLimit: number | null): number | null;
function preferDiscoveryLimit(discoveryLimit: number | null, fallbackLimit: number | null): number | null {
if (discoveryLimit === null || !Number.isFinite(discoveryLimit) || discoveryLimit <= 0) {
return fallbackLimit;
}
if (discoveryLimit === 4096 && fallbackLimit > discoveryLimit) {
if (discoveryLimit === 4096 && fallbackLimit !== null && fallbackLimit > discoveryLimit) {
return fallbackLimit;
}
return discoveryLimit;
@@ -428,11 +430,11 @@ function isModelLike(value: unknown): value is ModelSpec<Api> {
}
// Finite positive: NaN > 0 is false, +Infinity < Infinity is false.
const cw = v.contextWindow;
if (typeof cw !== "number" || !(cw > 0 && cw < Infinity)) {
if (cw !== null && (typeof cw !== "number" || !(cw > 0 && cw < Infinity))) {
return false;
}
const mt = v.maxTokens;
if (typeof mt !== "number" || !(mt > 0 && mt < Infinity)) {
if (mt !== null && (typeof mt !== "number" || !(mt > 0 && mt < Infinity))) {
return false;
}
return true;
File diff suppressed because it is too large Load Diff
@@ -37,11 +37,11 @@ export function createReferenceResolver<TApi extends Api>(
if (!existing) {
globalRefs.set(candidate.id, candidate);
} else if (candidate.contextWindow !== existing.contextWindow) {
if (candidate.contextWindow > existing.contextWindow) {
if ((candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0)) {
globalRefs.set(candidate.id, candidate);
}
} else if (candidate.maxTokens !== existing.maxTokens) {
if (candidate.maxTokens > existing.maxTokens) {
if ((candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0)) {
globalRefs.set(candidate.id, candidate);
}
} else if (existing.provider !== "openai" && candidate.provider === "openai") {
@@ -1,11 +0,0 @@
/**
* Fallback context-window / max-output-token values for models discovered
* without limit metadata.
*
* Kept in a dependency-free leaf module (rather than `openai-compat.ts`) so the
* model-discovery helpers in `utils/discovery/*` can import them without pulling
* the package root barrel (`@oh-my-pi/pi-ai`) into the model-manager init graph,
* which would otherwise form an import cycle through the provider registry.
*/
export const UNK_CONTEXT_WINDOW = 222_222;
export const UNK_MAX_TOKENS = 8_888;
@@ -11,7 +11,6 @@ import type { Api, FetchImpl, Model, ModelSpec, Provider, ThinkingConfig } from
import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
import { COPILOT_API_HEADERS, getGitHubCopilotBaseUrl, parseGitHubCopilotApiKey } from "../wire/github-copilot";
import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references";
import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "./discovery-constants";
const MODELS_DEV_URL = "https://models.dev/api.json";
const ANTHROPIC_BASE_URL = "https://api.anthropic.com/v1";
@@ -103,8 +102,8 @@ function mapAnthropicModelsDev(payload: unknown, baseUrl: string): ModelSpec<"an
cacheRead: toNumber(model.cost?.cache_read) ?? 0,
cacheWrite: toNumber(model.cost?.cache_write) ?? 0,
},
contextWindow: toPositiveNumber(model.limit?.context, UNK_CONTEXT_WINDOW),
maxTokens: toPositiveNumber(model.limit?.output, UNK_MAX_TOKENS),
contextWindow: toPositiveNumber(model.limit?.context, null),
maxTokens: toPositiveNumber(model.limit?.output, null),
});
}
@@ -1095,7 +1094,10 @@ export function isFireworksKimiK2ModelId(modelId: string): boolean {
* Clamp the Kimi K2 family's `maxTokens` to {@link FIREWORKS_KIMI_MAX_TOKENS}
* on Fireworks-backed providers, leaving every other model untouched.
*/
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): number {
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): number;
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number | null): number | null;
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number | null): number | null {
if (candidate === null) return null;
return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate;
}
@@ -1176,10 +1178,10 @@ function mapFireworksControlPlaneModel(
): ModelSpec<"openai-completions"> {
const name = toModelName(record.displayName, reference?.name ?? publicModelId);
const supportsImage = toBoolean(record.supportsImageInput) === true;
const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? UNK_CONTEXT_WINDOW);
const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? null);
// The control plane reports no max-output budget; default the Kimi family to
// its published cap, everyone else to the discovery fallback, then clamp.
const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : UNK_MAX_TOKENS;
const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : null;
const maxTokens = clampFireworksKimiMaxTokens(publicModelId, reference?.maxTokens ?? fallbackMaxTokens);
const base: ModelSpec<"openai-completions"> = reference ?? {
id: publicModelId,
@@ -1290,11 +1292,14 @@ function createModelsDevReferenceMap<TApi extends Api>(
references.set(candidate.id, candidate);
continue;
}
if (candidate.contextWindow > existing.contextWindow) {
if ((candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0)) {
references.set(candidate.id, candidate);
continue;
}
if (candidate.contextWindow === existing.contextWindow && candidate.maxTokens > existing.maxTokens) {
if (
candidate.contextWindow === existing.contextWindow &&
(candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0)
) {
references.set(candidate.id, candidate);
}
}
@@ -1413,7 +1418,7 @@ function mapWaferModel(
wafer?.context_length,
toPositiveNumber((entry as { max_model_len?: unknown }).max_model_len, defaults.contextWindow),
);
const maxTokens = Math.min(contextWindow, WAFER_MAX_TOKENS_CAP);
const maxTokens = contextWindow !== null ? Math.min(contextWindow, WAFER_MAX_TOKENS_CAP) : null;
const pricing = wafer?.pricing ?? {};
// Wafer's `/v1/models` exposes pricing through `*_cents_per_million` fields,
// but the values are an internal wholesale unit, not literal cents — across
@@ -2610,16 +2615,16 @@ function copilotTierCost(
*/
function createCopilotLongContextVariant(
base: ModelSpec<Api>,
fullContextWindow: number,
maxTokens: number,
fullContextWindow: number | null,
maxTokens: number | null,
longContext: CopilotTokenPriceTier | undefined,
): ModelSpec<Api> | undefined {
const longContextMax = longContext?.contextMax;
if (longContextMax === undefined || longContextMax <= 0) {
if (longContextMax === undefined || longContextMax <= 0 || fullContextWindow === null || maxTokens === null) {
return undefined;
}
const variantWindow = Math.min(fullContextWindow, longContextMax + maxTokens);
if (variantWindow <= base.contextWindow) {
if (base.contextWindow === null || variantWindow <= base.contextWindow) {
return undefined;
}
const longCost = copilotTierCost(longContext);
@@ -2708,7 +2713,10 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
const tokenPrices = extractCopilotTokenPrices(entry);
const defaultContextMax = tokenPrices.defaultTier?.contextMax;
const defaultTierWindow =
defaultContextMax !== undefined && defaultContextMax > 0
defaultContextMax !== undefined &&
defaultContextMax > 0 &&
contextWindow !== null &&
maxTokens !== null
? Math.min(contextWindow, defaultContextMax + maxTokens)
: contextWindow;
const base: ModelSpec<Api> = reference
@@ -2853,8 +2861,6 @@ export function anthropicModelManagerOptions(
// Models.dev provider descriptors for generate-models.ts
// ---------------------------------------------------------------------------
export { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "./discovery-constants";
/** Describes how to map models.dev API data for a single provider. */
export interface ModelsDevProviderDescriptor {
/** Key in the models.dev API response JSON (e.g., "anthropic", "amazon-bedrock") */
@@ -2934,8 +2940,8 @@ export function mapModelsDevToModels(
cacheRead: toNumber(m.cost?.cache_read) ?? 0,
cacheWrite: toNumber(m.cost?.cache_write) ?? 0,
},
contextWindow: toPositiveNumber(m.limit?.context, desc.defaultContextWindow ?? UNK_CONTEXT_WINDOW),
maxTokens: toPositiveNumber(m.limit?.output, desc.defaultMaxTokens ?? UNK_MAX_TOKENS),
contextWindow: toPositiveNumber(m.limit?.context, desc.defaultContextWindow ?? null),
maxTokens: toPositiveNumber(m.limit?.output, desc.defaultMaxTokens ?? null),
...(desc.compat && { compat: desc.compat }),
...(desc.headers && { headers: { ...desc.headers } }),
};
+2 -2
View File
@@ -438,8 +438,8 @@ export interface Model<TApi extends Api = Api> {
};
/** Premium Copilot requests charged per user-initiated request (defaults to 1). */
premiumMultiplier?: number;
contextWindow: number;
maxTokens: number;
contextWindow: number | null;
maxTokens: number | null;
/**
* When `true`, providers MUST omit `max_output_tokens` (Responses) /
* `max_tokens` / `max_completion_tokens` (Completions) from the outbound
+9 -1
View File
@@ -13,11 +13,19 @@ export function toNumber(value: unknown): number | undefined {
return undefined;
}
export function toPositiveNumber(value: unknown, fallback: number): number {
export function toPositiveNumber(value: unknown, fallback: number): number;
export function toPositiveNumber(value: unknown, fallback: number | null): number | null;
export function toPositiveNumber(value: unknown, fallback: number | null): number | null {
const parsed = toNumber(value);
return parsed !== undefined && parsed > 0 ? parsed : fallback;
}
/** Positive finite number, or `null` when the value is missing/non-positive. */
export function toPositiveNumberOrNull(value: unknown): number | null {
const parsed = toNumber(value);
return parsed !== undefined && parsed > 0 ? parsed : null;
}
export function toBoolean(value: unknown): boolean | undefined {
return typeof value === "boolean" ? value : undefined;
}
+7 -2
View File
@@ -444,8 +444,8 @@ export function collapseEffortVariants<TSpec extends VariantSpecLike>(
name: family.name,
reasoning,
input,
contextWindow: Math.max(...memberSpecs.map(spec => spec.contextWindow)),
maxTokens: Math.max(...memberSpecs.map(spec => spec.maxTokens)),
contextWindow: maxOrNull(memberSpecs.map(spec => spec.contextWindow)),
maxTokens: maxOrNull(memberSpecs.map(spec => spec.maxTokens)),
};
// The default wire id is the highest-priority live member; omit when it
// equals the logical id (bare/thinking pairs) — `resolveWireModelId`
@@ -620,3 +620,8 @@ export function getVariantAliasSources(provider: Provider, modelId: string): rea
if (!table) return [];
return getAliasIndex(table).reverse.get(modelId) ?? [];
}
function maxOrNull(values: ReadonlyArray<number | null>): number | null {
const known = values.filter((v): v is number => v != null);
return known.length ? Math.max(...known) : null;
}
@@ -0,0 +1,32 @@
import { describe, expect, it } from "bun:test";
import { fetchOpenAICompatibleModels } from "../src/discovery/openai-compatible";
describe("discovery null limits", () => {
it("emits null for contextWindow and maxTokens when limits are unknown", async () => {
const mockFetch = async () => {
return new Response(
JSON.stringify({
data: [
{
id: "some-model",
name: "Some Model",
},
],
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
};
const models = await fetchOpenAICompatibleModels({
provider: "custom",
api: "openai-completions",
baseUrl: "https://api.example.com/v1",
fetch: mockFetch,
});
expect(models).toBeDefined();
expect(models!.length).toBe(1);
expect(models![0].contextWindow).toBeNull();
expect(models![0].maxTokens).toBeNull();
});
});
@@ -67,7 +67,7 @@ describe("xai-oauth bundled catalog (regression)", () => {
// curated catalog owns maxTokens — set to mirror each model's contextWindow
// (the openai-responses wire still clamps the actual request to
// OPENAI_MAX_OUTPUT_TOKENS). Pin maxTokens === contextWindow on both the
// static-seed and bundled paths so the 8888 UNK_MAX_TOKENS placeholder can
// static-seed and bundled paths so a null placeholder can
// never silently leak back into the bundle.
it("sets maxTokens equal to contextWindow for every xai-oauth model", () => {
for (const model of seed) {
+2 -1
View File
@@ -1,7 +1,6 @@
# Changelog
## [Unreleased]
### Breaking Changes
- Removed the top-level `--list-models` flag path and migrated model listing to the new `omp models` command
@@ -14,6 +13,7 @@
### Changed
- Model registry merge and `omp models` / model picker handle unknown context/output limits (`null`) — unknown limits render as `-` instead of a fake `222K`/`8.9K`.
- Changed `omp models` to use cached provider data by default and require `omp models refresh` for a forced online re-fetch
- Updated model-resolution errors to point to `omp models` when a provider or model is not found
- Upgraded workspace catalog packages to their latest versions as of 3 days ago, and refactored the ACP agent implementation to be compatible with `@agentclientprotocol/sdk` version `0.25.0`.
@@ -21,6 +21,7 @@
### Fixed
- Fixed model auth gateway probing to avoid skipping candidates with unknown `maxTokens` limits (`null`)
- Fixed model listings so providers registered via extensions are now included from `-e` and configured `extensions` sources
- Fixed `/mcp reauth`, `/mcp test`, and `/mcp unauth` to find and operate on MCP servers reported by `/mcp list` even when they are only runtime-discovered and not stored in writable config, including namespaced plugin servers like `cloudflare:cloudflare-api`
- Fixed MCP server name validation so colon-namespaced server IDs are accepted when persisting reauth overrides so namespaced OAuth MCP servers can be stored in user config as `server:subserver` entries
@@ -409,7 +409,7 @@ function pickProbeCandidates(provider: string): Model<Api>[] {
if (!model.input.includes("text")) return false;
const totalCost = (model.cost?.input ?? 0) + (model.cost?.output ?? 0);
if (!Number.isFinite(totalCost) || totalCost < 0) return false;
if (model.maxTokens <= 0) return false;
if (model.maxTokens !== null && model.maxTokens <= 0) return false;
return true;
});
candidates.sort((a, b) => a.cost.input + a.cost.output - (b.cost.input + b.cost.output) || a.id.localeCompare(b.id));
+1 -1
View File
@@ -180,7 +180,7 @@ async function runBenchRequest(
apiKey: options.apiKey,
sessionId: options.sessionId,
maxTokens:
Number.isFinite(model.maxTokens) && model.maxTokens > 0
model.maxTokens !== null && Number.isFinite(model.maxTokens) && model.maxTokens > 0
? Math.min(options.maxTokens, model.maxTokens)
: options.maxTokens,
reasoning: options.reasoning,
@@ -279,7 +279,7 @@ function isBenchFirstTokenEvent(event: AssistantMessageEvent): boolean {
}
function resolveBenchMaxTokens(model: Model<Api>): number {
return Number.isFinite(model.maxTokens) && model.maxTokens > 0
return model.maxTokens !== null && Number.isFinite(model.maxTokens) && model.maxTokens > 0
? Math.min(BENCH_MAX_TOKENS, model.maxTokens)
: BENCH_MAX_TOKENS;
}
+12 -8
View File
@@ -68,8 +68,8 @@ interface ModelJson {
id: string;
selector: string;
name: string;
contextWindow: number;
maxTokens: number;
contextWindow: number | null;
maxTokens: number | null;
reasoning: boolean;
/** Supported thinking efforts when the model thinks, otherwise null. */
thinking: readonly Effort[] | null;
@@ -81,8 +81,8 @@ interface CanonicalJson {
id: string;
selected: string;
variants: number;
contextWindow: number;
maxTokens: number;
contextWindow: number | null;
maxTokens: number | null;
}
interface ModelsJson {
@@ -97,6 +97,10 @@ function writeLine(line = ""): void {
process.stdout.write(`${line}\n`);
}
function formatLimit(n: number | null): string {
return n === null ? "-" : formatNumber(n);
}
function byProviderThenId(left: Model<Api>, right: Model<Api>): number {
const providerCmp = left.provider.localeCompare(right.provider);
if (providerCmp !== 0) return providerCmp;
@@ -229,8 +233,8 @@ function renderProviderModels(
writeLine(`${chalk.bold.cyan(provider)} ${chalk.dim(`(${models.length})`)}`);
const rows = models.map(model => [
model.id,
formatNumber(model.contextWindow),
formatNumber(model.maxTokens),
formatLimit(model.contextWindow),
formatLimit(model.maxTokens),
model.thinking ? getSupportedEfforts(model).join(",") : model.reasoning ? "yes" : "-",
model.input.includes("image") ? "yes" : "no",
]);
@@ -293,8 +297,8 @@ function renderCanonicalModels(modelRegistry: ModelRegistry, pattern: string | u
record.id,
`${model.provider}/${model.id}`,
String(record.variants.length),
formatNumber(model.contextWindow),
formatNumber(model.maxTokens),
formatLimit(model.contextWindow),
formatLimit(model.maxTokens),
]);
for (const line of boxTable(
[
@@ -17,8 +17,6 @@ import {
googleGeminiCliModelManagerOptions,
openaiCodexModelManagerOptions,
PROVIDER_DESCRIPTORS,
UNK_CONTEXT_WINDOW,
UNK_MAX_TOKENS,
} from "@oh-my-pi/pi-catalog/provider-models";
import {
collapseBuiltModelVariants,
@@ -530,9 +528,8 @@ function finalizeCustomModel(model: CustomModelOverlay, options: CustomModelBuil
thinking: resolvedModel.thinking ?? reference?.thinking,
input: input as ("text" | "image")[],
cost,
contextWindow:
resolvedModel.contextWindow ?? reference?.contextWindow ?? (options.useDefaults ? 128000 : undefined),
maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : undefined),
contextWindow: resolvedModel.contextWindow ?? reference?.contextWindow ?? (options.useDefaults ? 128000 : null),
maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : null),
headers: resolvedModel.headers,
omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens,
compat: mergeCompat(reference?.compatConfig, resolvedModel.compat),
@@ -866,11 +863,8 @@ export class ModelRegistry {
if (!existing) return replacementModel;
return {
...replacementModel,
contextWindow:
replacementModel.contextWindow === UNK_CONTEXT_WINDOW
? existing.contextWindow
: replacementModel.contextWindow,
maxTokens: replacementModel.maxTokens === UNK_MAX_TOKENS ? existing.maxTokens : replacementModel.maxTokens,
contextWindow: replacementModel.contextWindow ?? existing.contextWindow,
maxTokens: replacementModel.maxTokens ?? existing.maxTokens,
};
});
}
+3 -1
View File
@@ -1071,7 +1071,9 @@ function truncateByApproxTokens(text: string, tokenLimit: number): string {
function computeModelTokenBudget(model: Model, config: MemoryRuntimeConfig): number {
const maxTokens =
Number.isFinite(model.contextWindow) && model.contextWindow > 0 ? model.contextWindow : config.fallbackTokenLimit;
model.contextWindow !== null && Number.isFinite(model.contextWindow) && model.contextWindow > 0
? model.contextWindow
: config.fallbackTokenLimit;
return Math.max(2048, Math.floor(maxTokens));
}
@@ -711,7 +711,7 @@ export class ModelSelectorComponent extends Container {
if (!this.#isModelOverContextLimit(model)) {
return "";
}
return ` ${theme.status.disabled} context>${formatNumber(model.contextWindow).toLowerCase()}`;
return ` ${theme.status.disabled} context>${formatNumber(model.contextWindow ?? 0).toLowerCase()}`;
}
#getVisibleItems(): ReadonlyArray<ModelItem | CanonicalModelItem> {
@@ -1016,7 +1016,7 @@ export class ModelSelectorComponent extends Container {
const limitWarning = this.#isItemDisabled(selected)
? theme.fg(
"dim",
` — current context ${formatNumber(this.#currentContextTokens).toLowerCase()} > ${formatNumber(selected.model.contextWindow).toLowerCase()} limit`,
` — current context ${formatNumber(this.#currentContextTokens).toLowerCase()} > ${formatNumber(selected.model.contextWindow ?? 0).toLowerCase()} limit`,
)
: "";
this.#listContainer.addChild(
@@ -7328,7 +7328,7 @@ export class AgentSession {
const candidate = this.#resolveContextPromotionConfiguredTarget(currentModel, availableModels);
if (!candidate) return undefined;
if (modelsAreEqual(candidate, currentModel)) return undefined;
if (candidate.contextWindow <= contextWindow) return undefined;
if (candidate.contextWindow == null || candidate.contextWindow <= contextWindow) return undefined;
const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId);
if (!apiKey) return undefined;
return candidate;
@@ -7639,7 +7639,7 @@ export class AgentSession {
addCandidate(this.#resolveRoleModelFull(role, availableModels, currentModel).model);
}
const sortedByContext = [...availableModels].sort((a, b) => b.contextWindow - a.contextWindow);
const sortedByContext = [...availableModels].sort((a, b) => (b.contextWindow ?? 0) - (a.contextWindow ?? 0));
for (const model of sortedByContext) {
if (!seen.has(this.#getModelKey(model))) {
addCandidate(model);
@@ -1860,8 +1860,8 @@ describe("ModelRegistry", () => {
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 222_222, // UNK_CONTEXT_WINDOW
maxTokens: 8_888, // UNK_MAX_TOKENS
contextWindow: null, // unknown limit
maxTokens: null, // unknown limit
}),
],
true,
@@ -1872,9 +1872,9 @@ describe("ModelRegistry", () => {
expect(model).toBeDefined();
// The bundled gpt-4o has a correct contextWindow, not the UNK sentinel
expect(model!.contextWindow).not.toBe(222_222);
expect(model!.contextWindow).not.toBeNull();
expect(model!.contextWindow).toBeGreaterThan(100_000);
expect(model!.maxTokens).not.toBe(8_888);
expect(model!.maxTokens).not.toBeNull();
expect(model!.maxTokens).toBeGreaterThan(1000);
});
+3
View File
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
### Fixed
- Fixed context usage percentage calculations to return null when context window is missing or non-positive, preventing invalid or Infinity/NaN usage display
## [15.12.2] - 2026-06-12
+1 -1
View File
@@ -356,7 +356,7 @@ export function makeProbeProgress(tick: number): SubagentProgressPayload {
requests: 4 + Math.floor(tick / 3),
tokens: 18_400 + tick * 450,
contextTokens: 22_300 + tick * 510,
contextWindow: fixtureModel.contextWindow,
contextWindow: fixtureModel.contextWindow ?? undefined,
cost: 0.041 + tick * 0.0012,
durationMs: 95_000 + tick * 2_000,
resolvedModel: fixtureModel.id,
+4 -1
View File
@@ -106,7 +106,10 @@ function buildState(): SessionState {
contextUsage: {
tokens,
contextWindow: fixtureModel.contextWindow,
percent: (tokens / fixtureModel.contextWindow) * 100,
percent:
fixtureModel.contextWindow !== null && fixtureModel.contextWindow > 0
? (tokens / fixtureModel.contextWindow) * 100
: null,
},
participants,
};
@@ -19,7 +19,9 @@ export function HeaderBar({ snapshot, subCount, railOpen, onToggleRail, onLeave
if (usage) {
pct =
usage.percent ??
(usage.tokens != null && usage.contextWindow > 0 ? (usage.tokens / usage.contextWindow) * 100 : null);
(usage.tokens != null && usage.contextWindow !== null && usage.contextWindow > 0
? (usage.tokens / usage.contextWindow) * 100
: null);
}
return (
+3
View File
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
### Changed
- Changed `WireModel.contextWindow` and `ContextUsage.contextWindow` to `number | null` to allow representing unavailable context-window values
## [15.12.0] - 2026-06-12
### Added
+2 -2
View File
@@ -205,12 +205,12 @@ export interface WireModel {
id: string;
name: string;
provider: string;
contextWindow: number;
contextWindow: number | null;
}
export interface ContextUsage {
tokens: number | null;
contextWindow: number;
contextWindow: number | null;
percent: number | null;
}