fix(catalog): backfilled missing model limit fields using canonical fallback
- Applied canonical limit fallback in model generation before provider grouping. - Backfilled null contextWindow and maxTokens with canonical and suffix alias lookups. - Preserved existing limit values and skipped zero-cost xai-oauth fallbacks. - Added canonical-limit-fallback test coverage for donor matching and no-donor cases.
This commit is contained in:
@@ -1,19 +1,22 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added bundled Fireworks models `deepseek-v4-flash`, `kimi-k2.7-code`, `minimax-m2.5`, `minimax-m3`, `nemotron-3-ultra-nvfp4`, `qwen3.6-plus`, and `qwen3.7-plus`
|
||||
- Changed
|
||||
|
||||
### Changed
|
||||
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
|
||||
|
||||
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
|
||||
- Changed the `github-copilot` model context window to `524288` tokens
|
||||
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Filled missing `contextWindow` and `maxTokens` in generated `models.json` for proxy/reseller variants by inheriting limits from canonical-family and segment-reference models
|
||||
- Ignored zero-cost `x-ai` subscription entries as reference sources when backfilling limits so inflated values are not propagated
|
||||
- Fixed the model cache opening with `PRAGMA journal_mode=WAL` before `PRAGMA busy_timeout`, so concurrent omp startups could crash inside `getDb()` on `SQLITE_BUSY` during WAL recovery instead of waiting through the transient lock. The busy handler is now installed before the first lock-taking statement ([#2421](https://github.com/can1357/oh-my-pi/issues/2421)).
|
||||
|
||||
## [15.11.8] - 2026-06-12
|
||||
|
||||
@@ -40,6 +40,7 @@ import { cleanModelName } from "../src/utils";
|
||||
import { collapseEffortVariantsAcrossProviders } from "../src/variant-collapse";
|
||||
import { JWT_CLAIM_PATH } from "../src/wire/codex";
|
||||
import {
|
||||
applyCanonicalLimitFallback,
|
||||
applyGeneratedModelPolicies,
|
||||
CLOUDFLARE_FALLBACK_MODEL,
|
||||
linkOpenAIPromotionTargets,
|
||||
@@ -475,6 +476,9 @@ async function generateModels() {
|
||||
// entries are already collapsed (rebake skips them); this pass folds
|
||||
// previous-snapshot raw members into their logical families.
|
||||
allModels = collapseEffortVariantsAcrossProviders(allModels);
|
||||
// Fill remaining null endpoint limits from each model's canonical-family
|
||||
// reference. Runs last so canonical ids and explicit policy limits are final.
|
||||
applyCanonicalLimitFallback(allModels);
|
||||
|
||||
// Group by provider and sort each provider's models
|
||||
const providers: Record<string, Record<string, ModelSpec>> = {};
|
||||
|
||||
@@ -13,8 +13,11 @@ import {
|
||||
parseKnownModel,
|
||||
semverEqual,
|
||||
} from "../src/identity/classify";
|
||||
import { buildCanonicalModelIndex, buildCanonicalReferenceData } from "../src/identity/equivalence";
|
||||
import { getLongestModelLikeIdSegment } from "../src/identity/id";
|
||||
import { buildModelReferenceIndex, resolveModelReference } from "../src/identity/reference";
|
||||
import { resolveModelThinking } from "../src/model-thinking";
|
||||
import type { Api, ModelSpec } from "../src/types";
|
||||
import type { Api, Model, ModelSpec } from "../src/types";
|
||||
import { isVariantCollapsedSpec } from "../src/variant-collapse";
|
||||
|
||||
const CLOUDFLARE_AI_GATEWAY_BASE_URL = "https://gateway.ai.cloudflare.com/v1/<account>/<gateway>/anthropic";
|
||||
@@ -113,6 +116,67 @@ export function linkOpenAIPromotionTargets(models: ModelSpec<Api>[]): void {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fill `null` `contextWindow` / `maxTokens` from a model's family reference.
|
||||
* Proxies and resellers serve first-party models under mangled ids and report
|
||||
* no limits, so discovery emits `null` rather than a magic number. Two lookups
|
||||
* cover the two ways an id drifts from its family head:
|
||||
*
|
||||
* 1. Compact / re-spelled versions (`venice/openai-gpt-54-mini`,
|
||||
* `aimlapi/moonshot/kimi-k2-5`) — the canonical-equivalence index maps these
|
||||
* to their head (`gpt-5.4-mini`, `kimi-k2.5`).
|
||||
* 2. Org-namespace variance (`aimlapi/alibaba/qwen3-32b` vs `groq/qwen/qwen3-32b`)
|
||||
* — these never share an exact id, so the bare model-segment (`qwen3-32b`)
|
||||
* is resolved through the proxy-reference suffix-alias map instead.
|
||||
*
|
||||
* Both lookups draw metadata from the proxy-reference index, which prefers the
|
||||
* largest limits with complete cache pricing and first-party providers, and
|
||||
* excludes zero-cost xai-oauth subscription entries (inflated `maxTokens`) as
|
||||
* sources. The canonical head is tried first (more precise); the segment alias
|
||||
* backfills any field it leaves null.
|
||||
*
|
||||
* Only `null` fields are filled; provider-specific limits that discovery
|
||||
* returned explicitly are never overwritten.
|
||||
*/
|
||||
export function applyCanonicalLimitFallback(models: ModelSpec<Api>[]): void {
|
||||
if (!models.some(model => model.contextWindow === null || model.maxTokens === null)) {
|
||||
return;
|
||||
}
|
||||
// The identity indices read only id/provider/name/limit/cost fields, all of
|
||||
// which ModelSpec carries — no built-only field (compat/thinking) is read —
|
||||
// so reusing the runtime Model<Api> builders over raw specs is sound.
|
||||
const catalog = models as unknown as readonly Model<Api>[];
|
||||
const referenceData = buildCanonicalReferenceData(catalog);
|
||||
const canonicalIndex = buildCanonicalModelIndex(catalog, referenceData);
|
||||
const referenceIndex = buildModelReferenceIndex(catalog);
|
||||
|
||||
for (const model of models) {
|
||||
if (model.contextWindow !== null && model.maxTokens !== null) {
|
||||
continue;
|
||||
}
|
||||
const canonicalId = canonicalIndex.bySelector.get(`${model.provider}/${model.id}`.toLowerCase());
|
||||
const segment = getLongestModelLikeIdSegment(model.id);
|
||||
const references = [
|
||||
canonicalId ? resolveModelReference(canonicalId, referenceIndex) : undefined,
|
||||
segment ? referenceIndex.suffixAlias.get(segment) : undefined,
|
||||
];
|
||||
for (const reference of references) {
|
||||
if (!reference || (reference.provider === model.provider && reference.id === model.id)) {
|
||||
continue;
|
||||
}
|
||||
if (model.contextWindow === null && reference.contextWindow !== null) {
|
||||
model.contextWindow = reference.contextWindow;
|
||||
}
|
||||
if (model.maxTokens === null && reference.maxTokens !== null) {
|
||||
model.maxTokens = reference.maxTokens;
|
||||
}
|
||||
if (model.contextWindow !== null && model.maxTokens !== null) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
|
||||
const copilotLimits = model.provider === "github-copilot" ? COPILOT_GENERATED_LIMITS[model.id] : undefined;
|
||||
if (copilotLimits) {
|
||||
|
||||
+834
-834
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,109 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types";
|
||||
import { applyCanonicalLimitFallback } from "../scripts/generated-policies";
|
||||
|
||||
function spec(overrides: {
|
||||
id: string;
|
||||
provider: Provider;
|
||||
contextWindow: number | null;
|
||||
maxTokens: number | null;
|
||||
cost?: ModelSpec<"openai-completions">["cost"];
|
||||
}): ModelSpec<"openai-completions"> {
|
||||
return {
|
||||
id: overrides.id,
|
||||
name: overrides.id,
|
||||
api: "openai-completions",
|
||||
provider: overrides.provider,
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: overrides.cost ?? { input: 1, output: 2, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: overrides.contextWindow,
|
||||
maxTokens: overrides.maxTokens,
|
||||
};
|
||||
}
|
||||
|
||||
function find(models: ModelSpec<Api>[], provider: Provider, id: string): ModelSpec<Api> {
|
||||
const model = models.find(m => m.provider === provider && m.id === id);
|
||||
if (!model) throw new Error(`missing ${provider}/${id}`);
|
||||
return model;
|
||||
}
|
||||
|
||||
describe("applyCanonicalLimitFallback", () => {
|
||||
it("fills a proxy's null maxTokens from the canonical first-party reference (gpt-5.4 family)", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
spec({ id: "gpt-5.4-mini", provider: "openai", contextWindow: 400000, maxTokens: 128000 }),
|
||||
// Venice's mangled id with a maxTokens hole; contextWindow is provider-supplied.
|
||||
spec({ id: "openai-gpt-54-mini", provider: "venice", contextWindow: 400000, maxTokens: null }),
|
||||
];
|
||||
|
||||
applyCanonicalLimitFallback(models);
|
||||
|
||||
const venice = find(models, "venice", "openai-gpt-54-mini");
|
||||
expect(venice.maxTokens).toBe(128000);
|
||||
// Provider-supplied contextWindow is left intact, not overwritten by the reference.
|
||||
expect(venice.contextWindow).toBe(400000);
|
||||
});
|
||||
|
||||
it("fills both null limits when the proxy reports neither", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
spec({ id: "gpt-5.4-pro", provider: "openai", contextWindow: 1050000, maxTokens: 128000 }),
|
||||
spec({ id: "openai-gpt-54-pro", provider: "venice", contextWindow: null, maxTokens: null }),
|
||||
];
|
||||
|
||||
applyCanonicalLimitFallback(models);
|
||||
|
||||
const venice = find(models, "venice", "openai-gpt-54-pro");
|
||||
expect(venice.contextWindow).toBe(1050000);
|
||||
expect(venice.maxTokens).toBe(128000);
|
||||
});
|
||||
|
||||
it("never sources limits from zero-cost xai-oauth subscription entries", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
// Inflated subscription entry: zero cost, oversized limits. Must be ignored.
|
||||
spec({
|
||||
id: "grok-4.3",
|
||||
provider: "xai-oauth",
|
||||
contextWindow: 2000000,
|
||||
maxTokens: 1000000,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
}),
|
||||
// Public paid entry: the legitimate reference.
|
||||
spec({ id: "grok-4.3", provider: "xai", contextWindow: 1000000, maxTokens: 30000 }),
|
||||
spec({ id: "x-ai/grok-4-3", provider: "aimlapi", contextWindow: null, maxTokens: null }),
|
||||
];
|
||||
|
||||
applyCanonicalLimitFallback(models);
|
||||
|
||||
const proxy = find(models, "aimlapi", "x-ai/grok-4-3");
|
||||
expect(proxy.maxTokens).toBe(30000);
|
||||
expect(proxy.contextWindow).toBe(1000000);
|
||||
});
|
||||
it("fills across org-namespace variance via the bare model segment", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
// Donor and hole share no exact id (alibaba/ vs qwen/ namespace) and no
|
||||
// compact-version relationship — only the `qwen3-32b` segment unifies them.
|
||||
spec({ id: "qwen/qwen3-32b", provider: "groq", contextWindow: 131072, maxTokens: 40960 }),
|
||||
spec({ id: "alibaba/qwen3-32b", provider: "aimlapi", contextWindow: null, maxTokens: null }),
|
||||
];
|
||||
|
||||
applyCanonicalLimitFallback(models);
|
||||
|
||||
const proxy = find(models, "aimlapi", "alibaba/qwen3-32b");
|
||||
expect(proxy.contextWindow).toBe(131072);
|
||||
expect(proxy.maxTokens).toBe(40960);
|
||||
});
|
||||
|
||||
|
||||
it("leaves holes null when no canonical-family reference exists", () => {
|
||||
const models: ModelSpec<Api>[] = [
|
||||
spec({ id: "some-bespoke-model-xyz", provider: "custom", contextWindow: null, maxTokens: null }),
|
||||
];
|
||||
|
||||
applyCanonicalLimitFallback(models);
|
||||
|
||||
const model = find(models, "custom", "some-bespoke-model-xyz");
|
||||
expect(model.contextWindow).toBeNull();
|
||||
expect(model.maxTokens).toBeNull();
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user