fix(catalog): backfilled missing model limit fields using canonical fallback

- Applied canonical limit fallback in model generation before provider grouping.
- Backfilled null contextWindow and maxTokens with canonical and suffix alias lookups.
- Preserved existing limit values and skipped zero-cost xai-oauth fallbacks.
- Added canonical-limit-fallback test coverage for donor matching and no-donor cases.
This commit is contained in:
can1357
2026-06-13 15:38:17 +02:00
parent f0c6a54f51
commit 2baabead25
5 changed files with 1016 additions and 836 deletions
+4 -1
View File
@@ -1,19 +1,22 @@
# Changelog
## [Unreleased]
### Added
- Added bundled Fireworks models `deepseek-v4-flash`, `kimi-k2.7-code`, `minimax-m2.5`, `minimax-m3`, `nemotron-3-ultra-nvfp4`, `qwen3.6-plus`, and `qwen3.7-plus`
- Changed
### Changed
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
- Changed the `github-copilot` model context window to `524288` tokens
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
### Fixed
- Filled missing `contextWindow` and `maxTokens` in generated `models.json` for proxy/reseller variants by inheriting limits from canonical-family and segment-reference models
- Ignored zero-cost `x-ai` subscription entries as reference sources when backfilling limits so inflated values are not propagated
- Fixed the model cache opening with `PRAGMA journal_mode=WAL` before `PRAGMA busy_timeout`, so concurrent omp startups could crash inside `getDb()` on `SQLITE_BUSY` during WAL recovery instead of waiting through the transient lock. The busy handler is now installed before the first lock-taking statement ([#2421](https://github.com/can1357/oh-my-pi/issues/2421)).
## [15.11.8] - 2026-06-12
@@ -40,6 +40,7 @@ import { cleanModelName } from "../src/utils";
import { collapseEffortVariantsAcrossProviders } from "../src/variant-collapse";
import { JWT_CLAIM_PATH } from "../src/wire/codex";
import {
applyCanonicalLimitFallback,
applyGeneratedModelPolicies,
CLOUDFLARE_FALLBACK_MODEL,
linkOpenAIPromotionTargets,
@@ -475,6 +476,9 @@ async function generateModels() {
// entries are already collapsed (rebake skips them); this pass folds
// previous-snapshot raw members into their logical families.
allModels = collapseEffortVariantsAcrossProviders(allModels);
// Fill remaining null endpoint limits from each model's canonical-family
// reference. Runs last so canonical ids and explicit policy limits are final.
applyCanonicalLimitFallback(allModels);
// Group by provider and sort each provider's models
const providers: Record<string, Record<string, ModelSpec>> = {};
+65 -1
View File
@@ -13,8 +13,11 @@ import {
parseKnownModel,
semverEqual,
} from "../src/identity/classify";
import { buildCanonicalModelIndex, buildCanonicalReferenceData } from "../src/identity/equivalence";
import { getLongestModelLikeIdSegment } from "../src/identity/id";
import { buildModelReferenceIndex, resolveModelReference } from "../src/identity/reference";
import { resolveModelThinking } from "../src/model-thinking";
import type { Api, ModelSpec } from "../src/types";
import type { Api, Model, ModelSpec } from "../src/types";
import { isVariantCollapsedSpec } from "../src/variant-collapse";
const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
@@ -113,6 +116,67 @@ export function linkOpenAIPromotionTargets(models: ModelSpec<Api>[]): void {
}
}
/**
* Fill `null` `contextWindow` / `maxTokens` from a model's family reference.
* Proxies and resellers serve first-party models under mangled ids and report
* no limits, so discovery emits `null` rather than a magic number. Two lookups
* cover the two ways an id drifts from its family head:
*
* 1. Compact / re-spelled versions (`venice/openai-gpt-54-mini`,
* `aimlapi/moonshot/kimi-k2-5`) — the canonical-equivalence index maps these
* to their head (`gpt-5.4-mini`, `kimi-k2.5`).
* 2. Org-namespace variance (`aimlapi/alibaba/qwen3-32b` vs `groq/qwen/qwen3-32b`)
* — these never share an exact id, so the bare model-segment (`qwen3-32b`)
* is resolved through the proxy-reference suffix-alias map instead.
*
* Both lookups draw metadata from the proxy-reference index, which prefers the
* largest limits with complete cache pricing and first-party providers, and
* excludes zero-cost xai-oauth subscription entries (inflated `maxTokens`) as
* sources. The canonical head is tried first (more precise); the segment alias
* backfills any field it leaves null.
*
* Only `null` fields are filled; provider-specific limits that discovery
* returned explicitly are never overwritten.
*/
export function applyCanonicalLimitFallback(models: ModelSpec<Api>[]): void {
if (!models.some(model => model.contextWindow === null || model.maxTokens === null)) {
return;
}
// The identity indices read only id/provider/name/limit/cost fields, all of
// which ModelSpec carries — no built-only field (compat/thinking) is read —
// so reusing the runtime Model<Api> builders over raw specs is sound.
const catalog = models as unknown as readonly Model<Api>[];
const referenceData = buildCanonicalReferenceData(catalog);
const canonicalIndex = buildCanonicalModelIndex(catalog, referenceData);
const referenceIndex = buildModelReferenceIndex(catalog);
for (const model of models) {
if (model.contextWindow !== null && model.maxTokens !== null) {
continue;
}
const canonicalId = canonicalIndex.bySelector.get(`${model.provider}/${model.id}`.toLowerCase());
const segment = getLongestModelLikeIdSegment(model.id);
const references = [
canonicalId ? resolveModelReference(canonicalId, referenceIndex) : undefined,
segment ? referenceIndex.suffixAlias.get(segment) : undefined,
];
for (const reference of references) {
if (!reference || (reference.provider === model.provider && reference.id === model.id)) {
continue;
}
if (model.contextWindow === null && reference.contextWindow !== null) {
model.contextWindow = reference.contextWindow;
}
if (model.maxTokens === null && reference.maxTokens !== null) {
model.maxTokens = reference.maxTokens;
}
if (model.contextWindow !== null && model.maxTokens !== null) {
break;
}
}
}
}
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
if (copilotLimits) {
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,109 @@
import { describe, expect, it } from "bun:test";
import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
import { applyCanonicalLimitFallback } from "../scripts/generated-policies";
function spec(overrides: {
id: string;
provider: Provider;
contextWindow: number | null;
maxTokens: number | null;
cost?: ModelSpec<"openai-completions">["cost"];
}): ModelSpec<"openai-completions"> {
return {
id: overrides.id,
name: overrides.id,
api: "openai-completions",
provider: overrides.provider,
baseUrl: "https://example.com",
reasoning: false,
input: ["text"],
cost: overrides.cost ?? { input: 1, output: 2, cacheRead: 0, cacheWrite: 0 },
contextWindow: overrides.contextWindow,
maxTokens: overrides.maxTokens,
};
}
function find(models: ModelSpec<Api>[], provider: Provider, id: string): ModelSpec<Api> {
const model = models.find(m => m.provider === provider && m.id === id);
if (!model) throw new Error(`missing ${provider}/${id}`);
return model;
}
describe("applyCanonicalLimitFallback", () => {
it("fills a proxy's null maxTokens from the canonical first-party reference (gpt-5.4 family)", () => {
const models: ModelSpec<Api>[] = [
spec({ id: "gpt-5.4-mini", provider: "openai", contextWindow: 400000, maxTokens: 128000 }),
// Venice's mangled id with a maxTokens hole; contextWindow is provider-supplied.
spec({ id: "openai-gpt-54-mini", provider: "venice", contextWindow: 400000, maxTokens: null }),
];
applyCanonicalLimitFallback(models);
const venice = find(models, "venice", "openai-gpt-54-mini");
expect(venice.maxTokens).toBe(128000);
// Provider-supplied contextWindow is left intact, not overwritten by the reference.
expect(venice.contextWindow).toBe(400000);
});
it("fills both null limits when the proxy reports neither", () => {
const models: ModelSpec<Api>[] = [
spec({ id: "gpt-5.4-pro", provider: "openai", contextWindow: 1050000, maxTokens: 128000 }),
spec({ id: "openai-gpt-54-pro", provider: "venice", contextWindow: null, maxTokens: null }),
];
applyCanonicalLimitFallback(models);
const venice = find(models, "venice", "openai-gpt-54-pro");
expect(venice.contextWindow).toBe(1050000);
expect(venice.maxTokens).toBe(128000);
});
it("never sources limits from zero-cost xai-oauth subscription entries", () => {
const models: ModelSpec<Api>[] = [
// Inflated subscription entry: zero cost, oversized limits. Must be ignored.
spec({
id: "grok-4.3",
provider: "xai-oauth",
contextWindow: 2000000,
maxTokens: 1000000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
}),
// Public paid entry: the legitimate reference.
spec({ id: "grok-4.3", provider: "xai", contextWindow: 1000000, maxTokens: 30000 }),
spec({ id: "x-ai/grok-4-3", provider: "aimlapi", contextWindow: null, maxTokens: null }),
];
applyCanonicalLimitFallback(models);
const proxy = find(models, "aimlapi", "x-ai/grok-4-3");
expect(proxy.maxTokens).toBe(30000);
expect(proxy.contextWindow).toBe(1000000);
});
it("fills across org-namespace variance via the bare model segment", () => {
const models: ModelSpec<Api>[] = [
// Donor and hole share no exact id (alibaba/ vs qwen/ namespace) and no
// compact-version relationship — only the `qwen3-32b` segment unifies them.
spec({ id: "qwen/qwen3-32b", provider: "groq", contextWindow: 131072, maxTokens: 40960 }),
spec({ id: "alibaba/qwen3-32b", provider: "aimlapi", contextWindow: null, maxTokens: null }),
];
applyCanonicalLimitFallback(models);
const proxy = find(models, "aimlapi", "alibaba/qwen3-32b");
expect(proxy.contextWindow).toBe(131072);
expect(proxy.maxTokens).toBe(40960);
});
it("leaves holes null when no canonical-family reference exists", () => {
const models: ModelSpec<Api>[] = [
spec({ id: "some-bespoke-model-xyz", provider: "custom", contextWindow: null, maxTokens: null }),
];
applyCanonicalLimitFallback(models);
const model = find(models, "custom", "some-bespoke-model-xyz");
expect(model.contextWindow).toBeNull();
expect(model.maxTokens).toBeNull();
});
});