fix: handled unknown model limits as null to avoid artificial token caps
- Replaced unknown model contextWindow/maxTokens sentinels with nullable values across types and catalog data. - Mapped request token calculations to treat null maxTokens as unlimited output caps. - Updated remote compaction and context checks to ignore unknown limits by using Infinity/0 fallbacks. - Adjusted CLI/model registry flows to skip cap enforcement for null limits and render unknown values as '-'.
This commit is contained in:
@@ -1,6 +1,9 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Fixed
|
||||
|
||||
- Fixed remote compaction input trimming to use unlimited context when `model.contextWindow` is unset
|
||||
|
||||
## [15.12.1] - 2026-06-12
|
||||
### Breaking Changes
|
||||
|
||||
@@ -456,7 +456,7 @@ export async function requestOpenAiRemoteCompaction(
|
||||
const endpoint = resolveOpenAiCompactEndpoint(model);
|
||||
const request: OpenAiRemoteCompactionRequest = {
|
||||
model: model.id,
|
||||
input: trimOpenAiCompactInput(compactInput, model.contextWindow, instructions),
|
||||
input: trimOpenAiCompactInput(compactInput, model.contextWindow ?? Number.POSITIVE_INFINITY, instructions),
|
||||
instructions,
|
||||
};
|
||||
const headers: Record<string, string> = {
|
||||
|
||||
@@ -11,6 +11,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed: provider request builders treat unknown `model.maxTokens` (`null`) as "no model cap" instead of coercing to `0` via `Math.min`; Anthropic falls back to the 64k Claude-Code cap for its required `max_tokens`.
|
||||
- Fixed transient stream failures on OpenAI-compatible providers by retrying HTTP 408/429/5xx responses and transient network errors with Retry-After/quota-hint aware backoff
|
||||
- Fixed SSE stream handling for OpenAI-compatible responses by parsing wire-level JSON frames directly and honoring `[DONE]` termination
|
||||
- Fixed stream error handling for OpenAI-compatible providers by preserving structured HTTP status/headers and response body details from failed requests for retry and strict-tool fallback logic
|
||||
|
||||
@@ -2817,7 +2817,8 @@ function buildParams(
|
||||
// Claude Code requests at most 64k output tokens; clamp only OAuth requests,
|
||||
// where the wire fingerprint must match. API-key callers keep the full model
|
||||
// ceiling (e.g. 128k on Opus 4.8).
|
||||
const maxOutputTokens = isOAuthToken ? Math.min(CLAUDE_CODE_MAX_OUTPUT_TOKENS, model.maxTokens) : model.maxTokens;
|
||||
const modelMaxTokens = model.maxTokens ?? CLAUDE_CODE_MAX_OUTPUT_TOKENS;
|
||||
const maxOutputTokens = isOAuthToken ? Math.min(CLAUDE_CODE_MAX_OUTPUT_TOKENS, modelMaxTokens) : modelMaxTokens;
|
||||
|
||||
// Build params in the canonical field order: model → messages → system → tools →
|
||||
// metadata → max_tokens → thinking → context_management → output_config → stream.
|
||||
@@ -2827,7 +2828,7 @@ function buildParams(
|
||||
...(systemBlocks && { system: systemBlocks }),
|
||||
...(tools !== undefined && { tools }),
|
||||
...(metadata && { metadata }),
|
||||
max_tokens: Math.min(maxOutputTokens, options?.maxTokens || model.maxTokens),
|
||||
max_tokens: Math.min(maxOutputTokens, options?.maxTokens || modelMaxTokens),
|
||||
...(thinking && { thinking }),
|
||||
...(contextManagement && { context_management: contextManagement }),
|
||||
...(outputConfig && { output_config: outputConfig }),
|
||||
|
||||
@@ -275,7 +275,7 @@ export function streamGitLabDuo(
|
||||
minP: options.minP,
|
||||
presencePenalty: options.presencePenalty,
|
||||
repetitionPenalty: options.repetitionPenalty,
|
||||
maxTokens: options.maxTokens ?? model.maxTokens,
|
||||
maxTokens: options.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options.signal,
|
||||
cacheRetention: options.cacheRetention,
|
||||
headers,
|
||||
@@ -313,7 +313,7 @@ export function streamGitLabDuo(
|
||||
minP: options.minP,
|
||||
presencePenalty: options.presencePenalty,
|
||||
repetitionPenalty: options.repetitionPenalty,
|
||||
maxTokens: options.maxTokens ?? model.maxTokens,
|
||||
maxTokens: options.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options.signal,
|
||||
cacheRetention: options.cacheRetention,
|
||||
headers,
|
||||
@@ -346,7 +346,7 @@ export function streamGitLabDuo(
|
||||
minP: options.minP,
|
||||
presencePenalty: options.presencePenalty,
|
||||
repetitionPenalty: options.repetitionPenalty,
|
||||
maxTokens: options.maxTokens ?? model.maxTokens,
|
||||
maxTokens: options.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options.signal,
|
||||
cacheRetention: options.cacheRetention,
|
||||
headers,
|
||||
|
||||
@@ -85,7 +85,7 @@ export function streamOpenAIAnthropicShim(
|
||||
minP: options?.minP,
|
||||
presencePenalty: options?.presencePenalty,
|
||||
repetitionPenalty: options?.repetitionPenalty,
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens,
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options?.signal,
|
||||
headers: mergedHeaders,
|
||||
sessionId: options?.sessionId,
|
||||
@@ -119,7 +119,7 @@ export function streamOpenAIAnthropicShim(
|
||||
minP: options?.minP,
|
||||
presencePenalty: options?.presencePenalty,
|
||||
repetitionPenalty: options?.repetitionPenalty,
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens,
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options?.signal,
|
||||
headers: mergedHeaders,
|
||||
sessionId: options?.sessionId,
|
||||
|
||||
@@ -1239,7 +1239,8 @@ function buildParams(
|
||||
// Kimi-family models calculate TPM rate limits from max_tokens (not actual
|
||||
// output) and the official guidance requires sending it on every call —
|
||||
// `compat.alwaysSendMaxTokens` carries that detection.
|
||||
const requestedMaxTokens = options?.maxTokens ?? (compat.alwaysSendMaxTokens ? model.maxTokens : undefined);
|
||||
const requestedMaxTokens =
|
||||
options?.maxTokens ?? (compat.alwaysSendMaxTokens ? (model.maxTokens ?? undefined) : undefined);
|
||||
// OpenRouter fans out to upstreams whose output caps differ from the catalog
|
||||
// value (which tracks the highest-cap provider). A max_tokens above the routed
|
||||
// upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras
|
||||
@@ -1250,7 +1251,7 @@ function buildParams(
|
||||
const effectiveMaxTokens =
|
||||
requestedMaxTokens === undefined || omitMaxTokensForRouting
|
||||
? undefined
|
||||
: Math.min(requestedMaxTokens, model.maxTokens, OPENAI_MAX_OUTPUT_TOKENS);
|
||||
: Math.min(requestedMaxTokens, model.maxTokens ?? Number.POSITIVE_INFINITY, OPENAI_MAX_OUTPUT_TOKENS);
|
||||
|
||||
const requestModelId = resolveOpenAICompletionsModelId(model, options);
|
||||
const params: OpenAICompletionsParams = {
|
||||
|
||||
@@ -1025,7 +1025,11 @@ export function applyCommonResponsesSamplingParams<P extends CommonResponsesPara
|
||||
model: Pick<Model, "provider" | "omitMaxOutputTokens" | "maxTokens">,
|
||||
): void {
|
||||
if (options?.maxTokens && !model.omitMaxOutputTokens) {
|
||||
params.max_output_tokens = Math.min(options.maxTokens, model.maxTokens, OPENAI_MAX_OUTPUT_TOKENS);
|
||||
params.max_output_tokens = Math.min(
|
||||
options.maxTokens,
|
||||
model.maxTokens ?? Number.POSITIVE_INFINITY,
|
||||
OPENAI_MAX_OUTPUT_TOKENS,
|
||||
);
|
||||
}
|
||||
if (options?.temperature !== undefined) params.temperature = options.temperature;
|
||||
if (options?.topP !== undefined) params.top_p = options.topP;
|
||||
|
||||
@@ -557,6 +557,7 @@ export async function completeSimple<TApi extends Api>(
|
||||
}
|
||||
|
||||
const MIN_OUTPUT_TOKENS = 1024;
|
||||
const OUTPUT_CAP_WHEN_UNKNOWN = 64_000;
|
||||
export const OUTPUT_FALLBACK_BUFFER = 4000;
|
||||
const ANTHROPIC_USE_INTERLEAVED_THINKING = Bun.env.PI_NO_INTERLEAVED_THINKING !== "1";
|
||||
|
||||
@@ -709,7 +710,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
minP: options?.minP,
|
||||
presencePenalty: options?.presencePenalty,
|
||||
repetitionPenalty: options?.repetitionPenalty,
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens,
|
||||
maxTokens: options?.maxTokens ?? model.maxTokens ?? undefined,
|
||||
signal: options?.signal,
|
||||
apiKey: apiKey ?? (typeof options?.apiKey === "string" ? options.apiKey : undefined),
|
||||
cacheRetention: options?.cacheRetention,
|
||||
@@ -785,7 +786,10 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
}
|
||||
|
||||
// Caller's maxTokens is the desired output; add thinking budget on top, capped at model limit
|
||||
const maxTokens = Math.min((base.maxTokens || 0) + thinkingBudget, model.maxTokens);
|
||||
const maxTokens = Math.min(
|
||||
(base.maxTokens || 0) + thinkingBudget,
|
||||
model.maxTokens ?? Number.POSITIVE_INFINITY,
|
||||
);
|
||||
|
||||
// If not enough room for thinking + output, reduce thinking budget
|
||||
if (maxTokens <= thinkingBudget) {
|
||||
@@ -830,10 +834,13 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
}
|
||||
const budgetInfo = resolveBedrockThinkingBudget(model as Model<"bedrock-converse-stream">, options);
|
||||
if (!budgetInfo) return bedrockBase as OptionsForApi<TApi>;
|
||||
let maxTokens = bedrockBase.maxTokens ?? model.maxTokens;
|
||||
let maxTokens = bedrockBase.maxTokens ?? model.maxTokens ?? OUTPUT_CAP_WHEN_UNKNOWN;
|
||||
let thinkingBudgets = bedrockBase.thinkingBudgets;
|
||||
if (maxTokens <= budgetInfo.budget) {
|
||||
const desiredMaxTokens = Math.min(model.maxTokens, budgetInfo.budget + MIN_OUTPUT_TOKENS);
|
||||
const desiredMaxTokens = Math.min(
|
||||
model.maxTokens ?? Number.POSITIVE_INFINITY,
|
||||
budgetInfo.budget + MIN_OUTPUT_TOKENS,
|
||||
);
|
||||
if (desiredMaxTokens > maxTokens) {
|
||||
maxTokens = desiredMaxTokens;
|
||||
}
|
||||
@@ -943,7 +950,10 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
let thinkingBudget = options.thinkingBudgets?.[effort] ?? GOOGLE_THINKING[effort];
|
||||
|
||||
// Caller's maxTokens is the desired output; add thinking budget on top, capped at model limit
|
||||
const maxTokens = Math.min((base.maxTokens || 0) + thinkingBudget, model.maxTokens);
|
||||
const maxTokens = Math.min(
|
||||
(base.maxTokens || 0) + thinkingBudget,
|
||||
model.maxTokens ?? Number.POSITIVE_INFINITY,
|
||||
);
|
||||
|
||||
// If not enough room for thinking + output, reduce thinking budget
|
||||
if (maxTokens <= thinkingBudget) {
|
||||
|
||||
@@ -16,10 +16,15 @@ import type { ChildProcess } from "node:child_process";
|
||||
import { execSync, spawn } from "node:child_process";
|
||||
import { complete } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { AssistantMessage, Context, Model, Usage } from "@oh-my-pi/pi-ai/types";
|
||||
import { isContextOverflow } from "@oh-my-pi/pi-ai/utils/overflow";
|
||||
import { isContextOverflow as originalIsContextOverflow } from "@oh-my-pi/pi-ai/utils/overflow";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { $which } from "@oh-my-pi/pi-utils";
|
||||
|
||||
function isContextOverflow(message: AssistantMessage, contextWindow: number | null): boolean {
|
||||
return originalIsContextOverflow(message, contextWindow ?? 0);
|
||||
}
|
||||
|
||||
import { e2eApiKey, resolveApiKey } from "./oauth";
|
||||
|
||||
// Resolve OAuth tokens at module level (async, runs before tests)
|
||||
@@ -55,7 +60,7 @@ interface OverflowResult {
|
||||
}
|
||||
|
||||
async function testContextOverflow(model: Model, apiKey: string): Promise<OverflowResult> {
|
||||
const overflowContent = generateOverflowContent(model.contextWindow);
|
||||
const overflowContent = generateOverflowContent(model.contextWindow ?? 0);
|
||||
|
||||
const context: Context = {
|
||||
systemPrompt: ["You are a helpful assistant."],
|
||||
@@ -75,7 +80,7 @@ async function testContextOverflow(model: Model, apiKey: string): Promise<Overfl
|
||||
return {
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
contextWindow: model.contextWindow,
|
||||
contextWindow: model.contextWindow ?? 0,
|
||||
stopReason: response.stopReason,
|
||||
errorMessage: response.errorMessage,
|
||||
usage: response.usage,
|
||||
@@ -459,7 +464,7 @@ describe("Context overflow error handling", () => {
|
||||
// Either way, isContextOverflow should detect it (via usage check or we skip if rate limited)
|
||||
if (result.stopReason === "stop") {
|
||||
expect(result.hasUsageData).toBe(true);
|
||||
expect(result.usage.input).toBeGreaterThan(model.contextWindow);
|
||||
expect(result.usage.input).toBeGreaterThan(model.contextWindow ?? 0);
|
||||
expect(isContextOverflow(result.response, model.contextWindow)).toBe(true);
|
||||
} else {
|
||||
// Rate limited or other error - just log and pass
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import type { Context, FetchImpl, ModelSpec } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
const ctx: Context = {
|
||||
systemPrompt: ["hi"],
|
||||
messages: [{ role: "user", content: "ping", timestamp: Date.now() }],
|
||||
};
|
||||
|
||||
function completionsSse(): Response {
|
||||
const events: unknown[] = [
|
||||
{
|
||||
id: "c",
|
||||
object: "chat.completion.chunk",
|
||||
choices: [{ index: 0, delta: { content: "ok" } }],
|
||||
},
|
||||
{
|
||||
id: "c",
|
||||
object: "chat.completion.chunk",
|
||||
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
||||
},
|
||||
"[DONE]",
|
||||
];
|
||||
const payload = `${events.map(e => `data: ${typeof e === "string" ? e : JSON.stringify(e)}`).join("\n\n")}\n\n`;
|
||||
return new Response(payload, { status: 200, headers: { "content-type": "text/event-stream" } });
|
||||
}
|
||||
|
||||
function responsesSse(): Response {
|
||||
const event = {
|
||||
type: "response.completed",
|
||||
response: {
|
||||
status: "completed",
|
||||
usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2, input_tokens_details: { cached_tokens: 0 } },
|
||||
},
|
||||
};
|
||||
return new Response(`data: ${JSON.stringify(event)}\n\n`, {
|
||||
status: 200,
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
}
|
||||
|
||||
describe("null maxTokens fallback wire tests", () => {
|
||||
it("verifies anthropic messages wire format max_tokens fallback is finite when maxTokens is null", async () => {
|
||||
const spec: ModelSpec<"anthropic-messages"> = {
|
||||
id: "claude-custom",
|
||||
name: "Claude Custom",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
baseUrl: "https://api.anthropic.com",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 100000,
|
||||
maxTokens: null, // unknown limit
|
||||
};
|
||||
const model = buildModel(spec);
|
||||
|
||||
let capturedPayload: Record<string, unknown> | null = null;
|
||||
const mockFetch: FetchImpl = async (_url, init) => {
|
||||
capturedPayload = JSON.parse(init?.body as string) as Record<string, unknown>;
|
||||
return new Response("{}", { status: 400 }); // we just want to capture
|
||||
};
|
||||
|
||||
// streamAnthropic triggers fetch. We ignore errors.
|
||||
try {
|
||||
await streamAnthropic(model, ctx, { apiKey: "test-key", fetch: mockFetch }).result();
|
||||
} catch {
|
||||
// expected 400
|
||||
}
|
||||
|
||||
expect(capturedPayload).not.toBeNull();
|
||||
expect(capturedPayload!.max_tokens).toBe(64000); // fallback CLAUDE_CODE_MAX_OUTPUT_TOKENS
|
||||
});
|
||||
|
||||
it("verifies openai completions clamps to OPENAI_MAX_OUTPUT_TOKENS when maxTokens is null", async () => {
|
||||
const spec: ModelSpec<"openai-completions"> = {
|
||||
id: "gpt-custom",
|
||||
name: "GPT Custom",
|
||||
api: "openai-completions",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 100000,
|
||||
maxTokens: null, // unknown limit
|
||||
};
|
||||
const model = buildModel(spec);
|
||||
|
||||
let capturedPayload: Record<string, unknown> | null = null;
|
||||
const mockFetch: FetchImpl = async (_url, init) => {
|
||||
capturedPayload = JSON.parse(init?.body as string) as Record<string, unknown>;
|
||||
return completionsSse();
|
||||
};
|
||||
|
||||
await streamOpenAICompletions(model, ctx, {
|
||||
apiKey: "test-key",
|
||||
maxTokens: 100000, // requested max tokens
|
||||
fetch: mockFetch,
|
||||
}).result();
|
||||
|
||||
expect(capturedPayload).not.toBeNull();
|
||||
expect(capturedPayload!.max_completion_tokens).toBe(64000); // clamps to OPENAI_MAX_OUTPUT_TOKENS
|
||||
});
|
||||
|
||||
it("verifies openai responses clamps to OPENAI_MAX_OUTPUT_TOKENS when maxTokens is null", async () => {
|
||||
const spec: ModelSpec<"openai-responses"> = {
|
||||
id: "gpt-responses-custom",
|
||||
name: "GPT Responses Custom",
|
||||
api: "openai-responses",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 100000,
|
||||
maxTokens: null, // unknown limit
|
||||
};
|
||||
const model = buildModel(spec);
|
||||
|
||||
let capturedPayload: Record<string, unknown> | null = null;
|
||||
const mockFetch: FetchImpl = async (_url, init) => {
|
||||
capturedPayload = JSON.parse(init?.body as string) as Record<string, unknown>;
|
||||
return responsesSse();
|
||||
};
|
||||
|
||||
await streamOpenAIResponses(model, ctx, {
|
||||
apiKey: "test-key",
|
||||
maxTokens: 100000, // requested max tokens
|
||||
fetch: mockFetch,
|
||||
}).result();
|
||||
|
||||
expect(capturedPayload).not.toBeNull();
|
||||
expect(capturedPayload!.max_output_tokens).toBe(64000); // clamps to OPENAI_MAX_OUTPUT_TOKENS
|
||||
});
|
||||
});
|
||||
@@ -7,6 +7,7 @@
|
||||
- Changed
|
||||
|
||||
### Changed
|
||||
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
|
||||
|
||||
- Changed the `github-copilot` model context window to `524288` tokens
|
||||
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
|
||||
|
||||
@@ -34,8 +34,6 @@ import {
|
||||
MODELS_DEV_PROVIDER_DESCRIPTORS,
|
||||
mapModelsDevToModels,
|
||||
stripFireworksDeepSeekThinkingToggle,
|
||||
UNK_CONTEXT_WINDOW,
|
||||
UNK_MAX_TOKENS,
|
||||
} from "../src/provider-models/openai-compat";
|
||||
import type { ModelSpec } from "../src/types";
|
||||
import { cleanModelName } from "../src/utils";
|
||||
@@ -158,21 +156,20 @@ function createGlobalModelsDevReferenceMap(modelsDevModels: readonly ModelSpec[]
|
||||
references.set(model.id, model);
|
||||
continue;
|
||||
}
|
||||
if (model.contextWindow > existing.contextWindow) {
|
||||
if ((model.contextWindow ?? 0) > (existing.contextWindow ?? 0)) {
|
||||
references.set(model.id, model);
|
||||
continue;
|
||||
}
|
||||
if (model.contextWindow === existing.contextWindow && model.maxTokens > existing.maxTokens) {
|
||||
if (
|
||||
(model.contextWindow ?? 0) === (existing.contextWindow ?? 0) &&
|
||||
(model.maxTokens ?? 0) > (existing.maxTokens ?? 0)
|
||||
) {
|
||||
references.set(model.id, model);
|
||||
}
|
||||
}
|
||||
return references;
|
||||
}
|
||||
|
||||
function inheritModelsDevLimit(value: number, referenceValue: number, unspecifiedValue: number): number {
|
||||
return value === unspecifiedValue ? referenceValue : value;
|
||||
}
|
||||
|
||||
function applyGlobalModelsDevFallback(
|
||||
models: readonly ModelSpec[],
|
||||
modelsDevModels: readonly ModelSpec[],
|
||||
@@ -194,8 +191,8 @@ function applyGlobalModelsDevFallback(
|
||||
input: reference.input,
|
||||
// Fill unknown endpoint limits from same-id models.dev references, but keep
|
||||
// provider-specific values when discovery returned them explicitly.
|
||||
contextWindow: inheritModelsDevLimit(model.contextWindow, reference.contextWindow, UNK_CONTEXT_WINDOW),
|
||||
maxTokens: inheritModelsDevLimit(model.maxTokens, reference.maxTokens, UNK_MAX_TOKENS),
|
||||
contextWindow: model.contextWindow ?? reference.contextWindow,
|
||||
maxTokens: model.maxTokens ?? reference.maxTokens,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
import { z } from "zod/v4";
|
||||
import { getBundledModels } from "../models";
|
||||
import { toModelSpec } from "../provider-models/bundled-references";
|
||||
import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "../provider-models/discovery-constants";
|
||||
import type { FetchImpl, Model, ModelSpec } from "../types";
|
||||
|
||||
const GOOGLE_GENERATIVE_AI_BASE_URL = "https://generativelanguage.googleapis.com/v1beta";
|
||||
@@ -148,7 +147,9 @@ function normalizeBaseUrl(baseUrl?: string): string {
|
||||
return value.replace(/\/+$/, "");
|
||||
}
|
||||
|
||||
function normalizePositiveInt(value: number | undefined, fallback: number): number {
|
||||
function normalizePositiveInt(value: number | undefined, fallback: number): number;
|
||||
function normalizePositiveInt(value: number | undefined, fallback: number | null): number | null;
|
||||
function normalizePositiveInt(value: number | undefined, fallback: number | null): number | null {
|
||||
if (typeof value !== "number" || !Number.isFinite(value) || value <= 0) {
|
||||
return fallback;
|
||||
}
|
||||
@@ -178,8 +179,8 @@ function normalizeModel(
|
||||
}
|
||||
|
||||
const reference = bundledById.get(id);
|
||||
const contextWindow = normalizePositiveInt(item.inputTokenLimit, reference?.contextWindow ?? UNK_CONTEXT_WINDOW);
|
||||
const maxTokens = normalizePositiveInt(item.outputTokenLimit, reference?.maxTokens ?? UNK_MAX_TOKENS);
|
||||
const contextWindow = normalizePositiveInt(item.inputTokenLimit, reference?.contextWindow ?? null);
|
||||
const maxTokens = normalizePositiveInt(item.outputTokenLimit, reference?.maxTokens ?? null);
|
||||
const name = normalizeModelName(item.displayName, reference?.name ?? id);
|
||||
|
||||
if (reference) {
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
import { z } from "zod/v4";
|
||||
import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "../provider-models/discovery-constants";
|
||||
import type { Api, FetchImpl, ModelSpec, Provider } from "../types";
|
||||
|
||||
const MODELS_PATH = "/models";
|
||||
@@ -165,8 +164,8 @@ export async function fetchOpenAICompatibleModels<TApi extends Api>(
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: UNK_CONTEXT_WINDOW,
|
||||
maxTokens: UNK_MAX_TOKENS,
|
||||
contextWindow: null,
|
||||
maxTokens: null,
|
||||
};
|
||||
|
||||
// `mapModel` returning null skips the entry (documented contract); only a
|
||||
|
||||
@@ -61,10 +61,10 @@ const FAMILY_EXTRACTION_PATTERNS = [
|
||||
function shouldReplaceReference(existing: Model<Api> | undefined, candidate: Model<Api>): boolean {
|
||||
if (!existing) return true;
|
||||
if (candidate.contextWindow !== existing.contextWindow) {
|
||||
return candidate.contextWindow > existing.contextWindow;
|
||||
return (candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0);
|
||||
}
|
||||
if (candidate.maxTokens !== existing.maxTokens) {
|
||||
return candidate.maxTokens > existing.maxTokens;
|
||||
return (candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0);
|
||||
}
|
||||
return existing.provider !== "openai" && candidate.provider === "openai";
|
||||
}
|
||||
|
||||
@@ -34,10 +34,10 @@ export function isZeroCostXaiOAuthReference(candidate: Model<Api>): boolean {
|
||||
function shouldReplaceReference(existing: Model<Api> | undefined, candidate: Model<Api>): boolean {
|
||||
if (!existing) return true;
|
||||
if (candidate.contextWindow !== existing.contextWindow) {
|
||||
return candidate.contextWindow > existing.contextWindow;
|
||||
return (candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0);
|
||||
}
|
||||
if (candidate.maxTokens !== existing.maxTokens) {
|
||||
return candidate.maxTokens > existing.maxTokens;
|
||||
return (candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0);
|
||||
}
|
||||
const existingHasCachePricing = existing.cost.cacheRead > 0 || existing.cost.cacheWrite > 0;
|
||||
const candidateHasCachePricing = candidate.cost.cacheRead > 0 || candidate.cost.cacheWrite > 0;
|
||||
|
||||
@@ -363,11 +363,13 @@ function preferDiscoveryName(discoveryName: string, fallbackName: string, modelI
|
||||
return normalizedDiscoveryName;
|
||||
}
|
||||
|
||||
function preferDiscoveryLimit(discoveryLimit: number, fallbackLimit: number): number {
|
||||
if (!Number.isFinite(discoveryLimit) || discoveryLimit <= 0) {
|
||||
function preferDiscoveryLimit(discoveryLimit: number, fallbackLimit: number): number;
|
||||
function preferDiscoveryLimit(discoveryLimit: number | null, fallbackLimit: number | null): number | null;
|
||||
function preferDiscoveryLimit(discoveryLimit: number | null, fallbackLimit: number | null): number | null {
|
||||
if (discoveryLimit === null || !Number.isFinite(discoveryLimit) || discoveryLimit <= 0) {
|
||||
return fallbackLimit;
|
||||
}
|
||||
if (discoveryLimit === 4096 && fallbackLimit > discoveryLimit) {
|
||||
if (discoveryLimit === 4096 && fallbackLimit !== null && fallbackLimit > discoveryLimit) {
|
||||
return fallbackLimit;
|
||||
}
|
||||
return discoveryLimit;
|
||||
@@ -428,11 +430,11 @@ function isModelLike(value: unknown): value is ModelSpec<Api> {
|
||||
}
|
||||
// Finite positive: NaN > 0 is false, +Infinity < Infinity is false.
|
||||
const cw = v.contextWindow;
|
||||
if (typeof cw !== "number" || !(cw > 0 && cw < Infinity)) {
|
||||
if (cw !== null && (typeof cw !== "number" || !(cw > 0 && cw < Infinity))) {
|
||||
return false;
|
||||
}
|
||||
const mt = v.maxTokens;
|
||||
if (typeof mt !== "number" || !(mt > 0 && mt < Infinity)) {
|
||||
if (mt !== null && (typeof mt !== "number" || !(mt > 0 && mt < Infinity))) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
|
||||
+2679
-2679
File diff suppressed because it is too large
Load Diff
@@ -37,11 +37,11 @@ export function createReferenceResolver<TApi extends Api>(
|
||||
if (!existing) {
|
||||
globalRefs.set(candidate.id, candidate);
|
||||
} else if (candidate.contextWindow !== existing.contextWindow) {
|
||||
if (candidate.contextWindow > existing.contextWindow) {
|
||||
if ((candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0)) {
|
||||
globalRefs.set(candidate.id, candidate);
|
||||
}
|
||||
} else if (candidate.maxTokens !== existing.maxTokens) {
|
||||
if (candidate.maxTokens > existing.maxTokens) {
|
||||
if ((candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0)) {
|
||||
globalRefs.set(candidate.id, candidate);
|
||||
}
|
||||
} else if (existing.provider !== "openai" && candidate.provider === "openai") {
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
/**
|
||||
* Fallback context-window / max-output-token values for models discovered
|
||||
* without limit metadata.
|
||||
*
|
||||
* Kept in a dependency-free leaf module (rather than `openai-compat.ts`) so the
|
||||
* model-discovery helpers in `utils/discovery/*` can import them without pulling
|
||||
* the package root barrel (`@oh-my-pi/pi-ai`) into the model-manager init graph,
|
||||
* which would otherwise form an import cycle through the provider registry.
|
||||
*/
|
||||
export const UNK_CONTEXT_WINDOW = 222_222;
|
||||
export const UNK_MAX_TOKENS = 8_888;
|
||||
@@ -11,7 +11,6 @@ import type { Api, FetchImpl, Model, ModelSpec, Provider, ThinkingConfig } from
|
||||
import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils";
|
||||
import { COPILOT_API_HEADERS, getGitHubCopilotBaseUrl, parseGitHubCopilotApiKey } from "../wire/github-copilot";
|
||||
import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references";
|
||||
import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "./discovery-constants";
|
||||
|
||||
const MODELS_DEV_URL = "https://models.dev/api.json";
|
||||
const ANTHROPIC_BASE_URL = "https://api.anthropic.com/v1";
|
||||
@@ -103,8 +102,8 @@ function mapAnthropicModelsDev(payload: unknown, baseUrl: string): ModelSpec<"an
|
||||
cacheRead: toNumber(model.cost?.cache_read) ?? 0,
|
||||
cacheWrite: toNumber(model.cost?.cache_write) ?? 0,
|
||||
},
|
||||
contextWindow: toPositiveNumber(model.limit?.context, UNK_CONTEXT_WINDOW),
|
||||
maxTokens: toPositiveNumber(model.limit?.output, UNK_MAX_TOKENS),
|
||||
contextWindow: toPositiveNumber(model.limit?.context, null),
|
||||
maxTokens: toPositiveNumber(model.limit?.output, null),
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1095,7 +1094,10 @@ export function isFireworksKimiK2ModelId(modelId: string): boolean {
|
||||
* Clamp the Kimi K2 family's `maxTokens` to {@link FIREWORKS_KIMI_MAX_TOKENS}
|
||||
* on Fireworks-backed providers, leaving every other model untouched.
|
||||
*/
|
||||
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): number {
|
||||
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): number;
|
||||
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number | null): number | null;
|
||||
export function clampFireworksKimiMaxTokens(modelId: string, candidate: number | null): number | null {
|
||||
if (candidate === null) return null;
|
||||
return isFireworksKimiK2ModelId(modelId) ? Math.min(candidate, FIREWORKS_KIMI_MAX_TOKENS) : candidate;
|
||||
}
|
||||
|
||||
@@ -1176,10 +1178,10 @@ function mapFireworksControlPlaneModel(
|
||||
): ModelSpec<"openai-completions"> {
|
||||
const name = toModelName(record.displayName, reference?.name ?? publicModelId);
|
||||
const supportsImage = toBoolean(record.supportsImageInput) === true;
|
||||
const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? UNK_CONTEXT_WINDOW);
|
||||
const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? null);
|
||||
// The control plane reports no max-output budget; default the Kimi family to
|
||||
// its published cap, everyone else to the discovery fallback, then clamp.
|
||||
const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : UNK_MAX_TOKENS;
|
||||
const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : null;
|
||||
const maxTokens = clampFireworksKimiMaxTokens(publicModelId, reference?.maxTokens ?? fallbackMaxTokens);
|
||||
const base: ModelSpec<"openai-completions"> = reference ?? {
|
||||
id: publicModelId,
|
||||
@@ -1290,11 +1292,14 @@ function createModelsDevReferenceMap<TApi extends Api>(
|
||||
references.set(candidate.id, candidate);
|
||||
continue;
|
||||
}
|
||||
if (candidate.contextWindow > existing.contextWindow) {
|
||||
if ((candidate.contextWindow ?? 0) > (existing.contextWindow ?? 0)) {
|
||||
references.set(candidate.id, candidate);
|
||||
continue;
|
||||
}
|
||||
if (candidate.contextWindow === existing.contextWindow && candidate.maxTokens > existing.maxTokens) {
|
||||
if (
|
||||
candidate.contextWindow === existing.contextWindow &&
|
||||
(candidate.maxTokens ?? 0) > (existing.maxTokens ?? 0)
|
||||
) {
|
||||
references.set(candidate.id, candidate);
|
||||
}
|
||||
}
|
||||
@@ -1413,7 +1418,7 @@ function mapWaferModel(
|
||||
wafer?.context_length,
|
||||
toPositiveNumber((entry as { max_model_len?: unknown }).max_model_len, defaults.contextWindow),
|
||||
);
|
||||
const maxTokens = Math.min(contextWindow, WAFER_MAX_TOKENS_CAP);
|
||||
const maxTokens = contextWindow !== null ? Math.min(contextWindow, WAFER_MAX_TOKENS_CAP) : null;
|
||||
const pricing = wafer?.pricing ?? {};
|
||||
// Wafer's `/v1/models` exposes pricing through `*_cents_per_million` fields,
|
||||
// but the values are an internal wholesale unit, not literal cents — across
|
||||
@@ -2610,16 +2615,16 @@ function copilotTierCost(
|
||||
*/
|
||||
function createCopilotLongContextVariant(
|
||||
base: ModelSpec<Api>,
|
||||
fullContextWindow: number,
|
||||
maxTokens: number,
|
||||
fullContextWindow: number | null,
|
||||
maxTokens: number | null,
|
||||
longContext: CopilotTokenPriceTier | undefined,
|
||||
): ModelSpec<Api> | undefined {
|
||||
const longContextMax = longContext?.contextMax;
|
||||
if (longContextMax === undefined || longContextMax <= 0) {
|
||||
if (longContextMax === undefined || longContextMax <= 0 || fullContextWindow === null || maxTokens === null) {
|
||||
return undefined;
|
||||
}
|
||||
const variantWindow = Math.min(fullContextWindow, longContextMax + maxTokens);
|
||||
if (variantWindow <= base.contextWindow) {
|
||||
if (base.contextWindow === null || variantWindow <= base.contextWindow) {
|
||||
return undefined;
|
||||
}
|
||||
const longCost = copilotTierCost(longContext);
|
||||
@@ -2708,7 +2713,10 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
|
||||
const tokenPrices = extractCopilotTokenPrices(entry);
|
||||
const defaultContextMax = tokenPrices.defaultTier?.contextMax;
|
||||
const defaultTierWindow =
|
||||
defaultContextMax !== undefined && defaultContextMax > 0
|
||||
defaultContextMax !== undefined &&
|
||||
defaultContextMax > 0 &&
|
||||
contextWindow !== null &&
|
||||
maxTokens !== null
|
||||
? Math.min(contextWindow, defaultContextMax + maxTokens)
|
||||
: contextWindow;
|
||||
const base: ModelSpec<Api> = reference
|
||||
@@ -2853,8 +2861,6 @@ export function anthropicModelManagerOptions(
|
||||
// Models.dev provider descriptors for generate-models.ts
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
export { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "./discovery-constants";
|
||||
|
||||
/** Describes how to map models.dev API data for a single provider. */
|
||||
export interface ModelsDevProviderDescriptor {
|
||||
/** Key in the models.dev API response JSON (e.g., "anthropic", "amazon-bedrock") */
|
||||
@@ -2934,8 +2940,8 @@ export function mapModelsDevToModels(
|
||||
cacheRead: toNumber(m.cost?.cache_read) ?? 0,
|
||||
cacheWrite: toNumber(m.cost?.cache_write) ?? 0,
|
||||
},
|
||||
contextWindow: toPositiveNumber(m.limit?.context, desc.defaultContextWindow ?? UNK_CONTEXT_WINDOW),
|
||||
maxTokens: toPositiveNumber(m.limit?.output, desc.defaultMaxTokens ?? UNK_MAX_TOKENS),
|
||||
contextWindow: toPositiveNumber(m.limit?.context, desc.defaultContextWindow ?? null),
|
||||
maxTokens: toPositiveNumber(m.limit?.output, desc.defaultMaxTokens ?? null),
|
||||
...(desc.compat && { compat: desc.compat }),
|
||||
...(desc.headers && { headers: { ...desc.headers } }),
|
||||
};
|
||||
|
||||
@@ -438,8 +438,8 @@ export interface Model<TApi extends Api = Api> {
|
||||
};
|
||||
/** Premium Copilot requests charged per user-initiated request (defaults to 1). */
|
||||
premiumMultiplier?: number;
|
||||
contextWindow: number;
|
||||
maxTokens: number;
|
||||
contextWindow: number | null;
|
||||
maxTokens: number | null;
|
||||
/**
|
||||
* When `true`, providers MUST omit `max_output_tokens` (Responses) /
|
||||
* `max_tokens` / `max_completion_tokens` (Completions) from the outbound
|
||||
|
||||
@@ -13,11 +13,19 @@ export function toNumber(value: unknown): number | undefined {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function toPositiveNumber(value: unknown, fallback: number): number {
|
||||
export function toPositiveNumber(value: unknown, fallback: number): number;
|
||||
export function toPositiveNumber(value: unknown, fallback: number | null): number | null;
|
||||
export function toPositiveNumber(value: unknown, fallback: number | null): number | null {
|
||||
const parsed = toNumber(value);
|
||||
return parsed !== undefined && parsed > 0 ? parsed : fallback;
|
||||
}
|
||||
|
||||
/** Positive finite number, or `null` when the value is missing/non-positive. */
|
||||
export function toPositiveNumberOrNull(value: unknown): number | null {
|
||||
const parsed = toNumber(value);
|
||||
return parsed !== undefined && parsed > 0 ? parsed : null;
|
||||
}
|
||||
|
||||
export function toBoolean(value: unknown): boolean | undefined {
|
||||
return typeof value === "boolean" ? value : undefined;
|
||||
}
|
||||
|
||||
@@ -444,8 +444,8 @@ export function collapseEffortVariants<TSpec extends VariantSpecLike>(
|
||||
name: family.name,
|
||||
reasoning,
|
||||
input,
|
||||
contextWindow: Math.max(...memberSpecs.map(spec => spec.contextWindow)),
|
||||
maxTokens: Math.max(...memberSpecs.map(spec => spec.maxTokens)),
|
||||
contextWindow: maxOrNull(memberSpecs.map(spec => spec.contextWindow)),
|
||||
maxTokens: maxOrNull(memberSpecs.map(spec => spec.maxTokens)),
|
||||
};
|
||||
// The default wire id is the highest-priority live member; omit when it
|
||||
// equals the logical id (bare/thinking pairs) — `resolveWireModelId`
|
||||
@@ -620,3 +620,8 @@ export function getVariantAliasSources(provider: Provider, modelId: string): rea
|
||||
if (!table) return [];
|
||||
return getAliasIndex(table).reverse.get(modelId) ?? [];
|
||||
}
|
||||
|
||||
function maxOrNull(values: ReadonlyArray<number | null>): number | null {
|
||||
const known = values.filter((v): v is number => v != null);
|
||||
return known.length ? Math.max(...known) : null;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { fetchOpenAICompatibleModels } from "../src/discovery/openai-compatible";
|
||||
|
||||
describe("discovery null limits", () => {
|
||||
it("emits null for contextWindow and maxTokens when limits are unknown", async () => {
|
||||
const mockFetch = async () => {
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
data: [
|
||||
{
|
||||
id: "some-model",
|
||||
name: "Some Model",
|
||||
},
|
||||
],
|
||||
}),
|
||||
{ status: 200, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
};
|
||||
|
||||
const models = await fetchOpenAICompatibleModels({
|
||||
provider: "custom",
|
||||
api: "openai-completions",
|
||||
baseUrl: "https://api.example.com/v1",
|
||||
fetch: mockFetch,
|
||||
});
|
||||
|
||||
expect(models).toBeDefined();
|
||||
expect(models!.length).toBe(1);
|
||||
expect(models![0].contextWindow).toBeNull();
|
||||
expect(models![0].maxTokens).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -67,7 +67,7 @@ describe("xai-oauth bundled catalog (regression)", () => {
|
||||
// curated catalog owns maxTokens — set to mirror each model's contextWindow
|
||||
// (the openai-responses wire still clamps the actual request to
|
||||
// OPENAI_MAX_OUTPUT_TOKENS). Pin maxTokens === contextWindow on both the
|
||||
// static-seed and bundled paths so the 8888 UNK_MAX_TOKENS placeholder can
|
||||
// static-seed and bundled paths so a null placeholder can
|
||||
// never silently leak back into the bundle.
|
||||
it("sets maxTokens equal to contextWindow for every xai-oauth model", () => {
|
||||
for (const model of seed) {
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Removed the top-level `--list-models` flag path and migrated model listing to the new `omp models` command
|
||||
@@ -14,6 +13,7 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Model registry merge and `omp models` / model picker handle unknown context/output limits (`null`) — unknown limits render as `-` instead of a fake `222K`/`8.9K`.
|
||||
- Changed `omp models` to use cached provider data by default and require `omp models refresh` for a forced online re-fetch
|
||||
- Updated model-resolution errors to point to `omp models` when a provider or model is not found
|
||||
- Upgraded workspace catalog packages to their latest versions as of 3 days ago, and refactored the ACP agent implementation to be compatible with `@agentclientprotocol/sdk` version `0.25.0`.
|
||||
@@ -21,6 +21,7 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed model auth gateway probing to avoid skipping candidates with unknown `maxTokens` limits (`null`)
|
||||
- Fixed model listings so providers registered via extensions are now included from `-e` and configured `extensions` sources
|
||||
- Fixed `/mcp reauth`, `/mcp test`, and `/mcp unauth` to find and operate on MCP servers reported by `/mcp list` even when they are only runtime-discovered and not stored in writable config, including namespaced plugin servers like `cloudflare:cloudflare-api`
|
||||
- Fixed MCP server name validation so colon-namespaced server IDs are accepted when persisting reauth overrides so namespaced OAuth MCP servers can be stored in user config as `server:subserver` entries
|
||||
|
||||
@@ -409,7 +409,7 @@ function pickProbeCandidates(provider: string): Model<Api>[] {
|
||||
if (!model.input.includes("text")) return false;
|
||||
const totalCost = (model.cost?.input ?? 0) + (model.cost?.output ?? 0);
|
||||
if (!Number.isFinite(totalCost) || totalCost < 0) return false;
|
||||
if (model.maxTokens <= 0) return false;
|
||||
if (model.maxTokens !== null && model.maxTokens <= 0) return false;
|
||||
return true;
|
||||
});
|
||||
candidates.sort((a, b) => a.cost.input + a.cost.output - (b.cost.input + b.cost.output) || a.id.localeCompare(b.id));
|
||||
|
||||
@@ -180,7 +180,7 @@ async function runBenchRequest(
|
||||
apiKey: options.apiKey,
|
||||
sessionId: options.sessionId,
|
||||
maxTokens:
|
||||
Number.isFinite(model.maxTokens) && model.maxTokens > 0
|
||||
model.maxTokens !== null && Number.isFinite(model.maxTokens) && model.maxTokens > 0
|
||||
? Math.min(options.maxTokens, model.maxTokens)
|
||||
: options.maxTokens,
|
||||
reasoning: options.reasoning,
|
||||
|
||||
@@ -279,7 +279,7 @@ function isBenchFirstTokenEvent(event: AssistantMessageEvent): boolean {
|
||||
}
|
||||
|
||||
function resolveBenchMaxTokens(model: Model<Api>): number {
|
||||
return Number.isFinite(model.maxTokens) && model.maxTokens > 0
|
||||
return model.maxTokens !== null && Number.isFinite(model.maxTokens) && model.maxTokens > 0
|
||||
? Math.min(BENCH_MAX_TOKENS, model.maxTokens)
|
||||
: BENCH_MAX_TOKENS;
|
||||
}
|
||||
|
||||
@@ -68,8 +68,8 @@ interface ModelJson {
|
||||
id: string;
|
||||
selector: string;
|
||||
name: string;
|
||||
contextWindow: number;
|
||||
maxTokens: number;
|
||||
contextWindow: number | null;
|
||||
maxTokens: number | null;
|
||||
reasoning: boolean;
|
||||
/** Supported thinking efforts when the model thinks, otherwise null. */
|
||||
thinking: readonly Effort[] | null;
|
||||
@@ -81,8 +81,8 @@ interface CanonicalJson {
|
||||
id: string;
|
||||
selected: string;
|
||||
variants: number;
|
||||
contextWindow: number;
|
||||
maxTokens: number;
|
||||
contextWindow: number | null;
|
||||
maxTokens: number | null;
|
||||
}
|
||||
|
||||
interface ModelsJson {
|
||||
@@ -97,6 +97,10 @@ function writeLine(line = ""): void {
|
||||
process.stdout.write(`${line}\n`);
|
||||
}
|
||||
|
||||
function formatLimit(n: number | null): string {
|
||||
return n === null ? "-" : formatNumber(n);
|
||||
}
|
||||
|
||||
function byProviderThenId(left: Model<Api>, right: Model<Api>): number {
|
||||
const providerCmp = left.provider.localeCompare(right.provider);
|
||||
if (providerCmp !== 0) return providerCmp;
|
||||
@@ -229,8 +233,8 @@ function renderProviderModels(
|
||||
writeLine(`${chalk.bold.cyan(provider)} ${chalk.dim(`(${models.length})`)}`);
|
||||
const rows = models.map(model => [
|
||||
model.id,
|
||||
formatNumber(model.contextWindow),
|
||||
formatNumber(model.maxTokens),
|
||||
formatLimit(model.contextWindow),
|
||||
formatLimit(model.maxTokens),
|
||||
model.thinking ? getSupportedEfforts(model).join(",") : model.reasoning ? "yes" : "-",
|
||||
model.input.includes("image") ? "yes" : "no",
|
||||
]);
|
||||
@@ -293,8 +297,8 @@ function renderCanonicalModels(modelRegistry: ModelRegistry, pattern: string | u
|
||||
record.id,
|
||||
`${model.provider}/${model.id}`,
|
||||
String(record.variants.length),
|
||||
formatNumber(model.contextWindow),
|
||||
formatNumber(model.maxTokens),
|
||||
formatLimit(model.contextWindow),
|
||||
formatLimit(model.maxTokens),
|
||||
]);
|
||||
for (const line of boxTable(
|
||||
[
|
||||
|
||||
@@ -17,8 +17,6 @@ import {
|
||||
googleGeminiCliModelManagerOptions,
|
||||
openaiCodexModelManagerOptions,
|
||||
PROVIDER_DESCRIPTORS,
|
||||
UNK_CONTEXT_WINDOW,
|
||||
UNK_MAX_TOKENS,
|
||||
} from "@oh-my-pi/pi-catalog/provider-models";
|
||||
import {
|
||||
collapseBuiltModelVariants,
|
||||
@@ -530,9 +528,8 @@ function finalizeCustomModel(model: CustomModelOverlay, options: CustomModelBuil
|
||||
thinking: resolvedModel.thinking ?? reference?.thinking,
|
||||
input: input as ("text" | "image")[],
|
||||
cost,
|
||||
contextWindow:
|
||||
resolvedModel.contextWindow ?? reference?.contextWindow ?? (options.useDefaults ? 128000 : undefined),
|
||||
maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : undefined),
|
||||
contextWindow: resolvedModel.contextWindow ?? reference?.contextWindow ?? (options.useDefaults ? 128000 : null),
|
||||
maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : null),
|
||||
headers: resolvedModel.headers,
|
||||
omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens,
|
||||
compat: mergeCompat(reference?.compatConfig, resolvedModel.compat),
|
||||
@@ -866,11 +863,8 @@ export class ModelRegistry {
|
||||
if (!existing) return replacementModel;
|
||||
return {
|
||||
...replacementModel,
|
||||
contextWindow:
|
||||
replacementModel.contextWindow === UNK_CONTEXT_WINDOW
|
||||
? existing.contextWindow
|
||||
: replacementModel.contextWindow,
|
||||
maxTokens: replacementModel.maxTokens === UNK_MAX_TOKENS ? existing.maxTokens : replacementModel.maxTokens,
|
||||
contextWindow: replacementModel.contextWindow ?? existing.contextWindow,
|
||||
maxTokens: replacementModel.maxTokens ?? existing.maxTokens,
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1071,7 +1071,9 @@ function truncateByApproxTokens(text: string, tokenLimit: number): string {
|
||||
|
||||
function computeModelTokenBudget(model: Model, config: MemoryRuntimeConfig): number {
|
||||
const maxTokens =
|
||||
Number.isFinite(model.contextWindow) && model.contextWindow > 0 ? model.contextWindow : config.fallbackTokenLimit;
|
||||
model.contextWindow !== null && Number.isFinite(model.contextWindow) && model.contextWindow > 0
|
||||
? model.contextWindow
|
||||
: config.fallbackTokenLimit;
|
||||
return Math.max(2048, Math.floor(maxTokens));
|
||||
}
|
||||
|
||||
|
||||
@@ -711,7 +711,7 @@ export class ModelSelectorComponent extends Container {
|
||||
if (!this.#isModelOverContextLimit(model)) {
|
||||
return "";
|
||||
}
|
||||
return ` ${theme.status.disabled} context>${formatNumber(model.contextWindow).toLowerCase()}`;
|
||||
return ` ${theme.status.disabled} context>${formatNumber(model.contextWindow ?? 0).toLowerCase()}`;
|
||||
}
|
||||
|
||||
#getVisibleItems(): ReadonlyArray<ModelItem | CanonicalModelItem> {
|
||||
@@ -1016,7 +1016,7 @@ export class ModelSelectorComponent extends Container {
|
||||
const limitWarning = this.#isItemDisabled(selected)
|
||||
? theme.fg(
|
||||
"dim",
|
||||
` — current context ${formatNumber(this.#currentContextTokens).toLowerCase()} > ${formatNumber(selected.model.contextWindow).toLowerCase()} limit`,
|
||||
` — current context ${formatNumber(this.#currentContextTokens).toLowerCase()} > ${formatNumber(selected.model.contextWindow ?? 0).toLowerCase()} limit`,
|
||||
)
|
||||
: "";
|
||||
this.#listContainer.addChild(
|
||||
|
||||
@@ -7328,7 +7328,7 @@ export class AgentSession {
|
||||
const candidate = this.#resolveContextPromotionConfiguredTarget(currentModel, availableModels);
|
||||
if (!candidate) return undefined;
|
||||
if (modelsAreEqual(candidate, currentModel)) return undefined;
|
||||
if (candidate.contextWindow <= contextWindow) return undefined;
|
||||
if (candidate.contextWindow == null || candidate.contextWindow <= contextWindow) return undefined;
|
||||
const apiKey = await this.#modelRegistry.getApiKey(candidate, this.sessionId);
|
||||
if (!apiKey) return undefined;
|
||||
return candidate;
|
||||
@@ -7639,7 +7639,7 @@ export class AgentSession {
|
||||
addCandidate(this.#resolveRoleModelFull(role, availableModels, currentModel).model);
|
||||
}
|
||||
|
||||
const sortedByContext = [...availableModels].sort((a, b) => b.contextWindow - a.contextWindow);
|
||||
const sortedByContext = [...availableModels].sort((a, b) => (b.contextWindow ?? 0) - (a.contextWindow ?? 0));
|
||||
for (const model of sortedByContext) {
|
||||
if (!seen.has(this.#getModelKey(model))) {
|
||||
addCandidate(model);
|
||||
|
||||
@@ -1860,8 +1860,8 @@ describe("ModelRegistry", () => {
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 222_222, // UNK_CONTEXT_WINDOW
|
||||
maxTokens: 8_888, // UNK_MAX_TOKENS
|
||||
contextWindow: null, // unknown limit
|
||||
maxTokens: null, // unknown limit
|
||||
}),
|
||||
],
|
||||
true,
|
||||
@@ -1872,9 +1872,9 @@ describe("ModelRegistry", () => {
|
||||
|
||||
expect(model).toBeDefined();
|
||||
// The bundled gpt-4o has a correct contextWindow, not the UNK sentinel
|
||||
expect(model!.contextWindow).not.toBe(222_222);
|
||||
expect(model!.contextWindow).not.toBeNull();
|
||||
expect(model!.contextWindow).toBeGreaterThan(100_000);
|
||||
expect(model!.maxTokens).not.toBe(8_888);
|
||||
expect(model!.maxTokens).not.toBeNull();
|
||||
expect(model!.maxTokens).toBeGreaterThan(1000);
|
||||
});
|
||||
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Fixed
|
||||
|
||||
- Fixed context usage percentage calculations to return null when context window is missing or non-positive, preventing invalid or Infinity/NaN usage display
|
||||
|
||||
## [15.12.2] - 2026-06-12
|
||||
|
||||
|
||||
@@ -356,7 +356,7 @@ export function makeProbeProgress(tick: number): SubagentProgressPayload {
|
||||
requests: 4 + Math.floor(tick / 3),
|
||||
tokens: 18_400 + tick * 450,
|
||||
contextTokens: 22_300 + tick * 510,
|
||||
contextWindow: fixtureModel.contextWindow,
|
||||
contextWindow: fixtureModel.contextWindow ?? undefined,
|
||||
cost: 0.041 + tick * 0.0012,
|
||||
durationMs: 95_000 + tick * 2_000,
|
||||
resolvedModel: fixtureModel.id,
|
||||
|
||||
@@ -106,7 +106,10 @@ function buildState(): SessionState {
|
||||
contextUsage: {
|
||||
tokens,
|
||||
contextWindow: fixtureModel.contextWindow,
|
||||
percent: (tokens / fixtureModel.contextWindow) * 100,
|
||||
percent:
|
||||
fixtureModel.contextWindow !== null && fixtureModel.contextWindow > 0
|
||||
? (tokens / fixtureModel.contextWindow) * 100
|
||||
: null,
|
||||
},
|
||||
participants,
|
||||
};
|
||||
|
||||
@@ -19,7 +19,9 @@ export function HeaderBar({ snapshot, subCount, railOpen, onToggleRail, onLeave
|
||||
if (usage) {
|
||||
pct =
|
||||
usage.percent ??
|
||||
(usage.tokens != null && usage.contextWindow > 0 ? (usage.tokens / usage.contextWindow) * 100 : null);
|
||||
(usage.tokens != null && usage.contextWindow !== null && usage.contextWindow > 0
|
||||
? (usage.tokens / usage.contextWindow) * 100
|
||||
: null);
|
||||
}
|
||||
|
||||
return (
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Changed
|
||||
|
||||
- Changed `WireModel.contextWindow` and `ContextUsage.contextWindow` to `number | null` to allow representing unavailable context-window values
|
||||
|
||||
## [15.12.0] - 2026-06-12
|
||||
### Added
|
||||
|
||||
@@ -205,12 +205,12 @@ export interface WireModel {
|
||||
id: string;
|
||||
name: string;
|
||||
provider: string;
|
||||
contextWindow: number;
|
||||
contextWindow: number | null;
|
||||
}
|
||||
|
||||
export interface ContextUsage {
|
||||
tokens: number | null;
|
||||
contextWindow: number;
|
||||
contextWindow: number | null;
|
||||
percent: number | null;
|
||||
}
|
||||
|
||||
|
||||
Reference in New Issue
Block a user