diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 390d0adee..eb5e7b41d 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -14,6 +14,7 @@ - Fixed SuperGrok (`xai-oauth`) Grok 4.6 hiding the thinking-level picker: the Responses effort-capable allowlist now includes `grok-4.6`, so `/model` can select the documented `low`/`medium`/`high`/`xhigh` ladder (`max` is rejected by api.x.ai). - Marked CoreWeave runtime discovery as authoritative so stale bundled model ids that the endpoint no longer serves stop appearing as selectable models. - ChatGPT Codex discovery that advertises only worker `-wm` SKUs now also registers the plain model route, so a configured `openai-codex/` keeps resolving instead of fuzzy-falling-back to the `-wm` SKU some accounts reject. +- Fixed GMI Cloud (`gmi-cloud`) models resolved via `/v1/models` discovery surfacing with `null` context windows, zero pricing, and no reasoning/thinking metadata for every model except the bundled `deepseek-ai/DeepSeek-V4-Flash` seed. GMI's endpoint returns only bare `{id}` rows, so the mapper now recovers intrinsic capability metadata (context window, output limit, reasoning, thinking ladder) for resold open-weight models from the cross-provider canonical reference index — matching the SiliconFlow behavior — while never borrowing another provider's pricing ([#8890](https://github.com/can1357/oh-my-pi/issues/8890)). ## [17.3.6] - 2026-08-17 diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 7fc21a0ab..328c337da 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -1074,10 +1074,61 @@ export interface GmiCloudModelManagerConfig { fetch?: FetchImpl; } +/** + * Map a discovered GMI Cloud model to a full spec. + * + * GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults carry + * no limits, reasoning, or thinking metadata. When a gmi-cloud bundled + * reference exists (the seeded default) it supplies GMI's published tariff and + * limits directly. Every other id is an open-weight model GMI resells under its + * canonical id (`deepseek-ai/…`, `moonshotai/…`, `zai-org/…`, `Qwen/…`), so its + * intrinsic capabilities — context window, output limit, reasoning, thinking + * ladder — are recovered from any bundled upstream entry via the canonical + * reference index. Pricing is deliberately never borrowed across providers: + * GMI's per-model tariff is unknown for these ids, so cost stays zeroed rather + * than inheriting another provider's rate. + */ +function mapGmiCloudModel( + entry: OpenAICompatibleModelRecord, + defaults: ModelSpec<"openai-completions">, + reference: ModelSpec<"openai-completions"> | undefined, +): ModelSpec<"openai-completions"> { + if (reference) { + return mapWithBundledReference(entry, defaults, reference); + } + const canonical = resolveModelReference(defaults.id, getBundledModelReferenceIndex()) as + | ModelSpec<"openai-completions"> + | undefined; + if (!canonical) { + return { ...defaults, name: toModelName(entry.name, defaults.name) }; + } + const contextWindow = canonical.contextWindow ?? defaults.contextWindow; + const maxTokens = + canonical.maxTokens != null && contextWindow != null + ? Math.min(canonical.maxTokens, contextWindow) + : (canonical.maxTokens ?? defaults.maxTokens); + return { + ...defaults, + name: toModelName(entry.name, canonical.name ?? defaults.name), + reasoning: canonical.reasoning, + input: canonical.input, + ...(canonical.thinking && { thinking: canonical.thinking }), + contextWindow, + maxTokens, + }; +} + export function gmiCloudModelManagerOptions( config?: GmiCloudModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { - return createSimpleOpenAICompletionsOptions("gmi-cloud", GMI_CLOUD_BASE_URL, config); + return createOpenAICompatibleModelManagerOptions({ + api: "openai-completions", + providerId: "gmi-cloud", + defaultBaseUrl: GMI_CLOUD_BASE_URL, + config, + requireApiKey: true, + mapModel: mapGmiCloudModel, + }); } // --------------------------------------------------------------------------- diff --git a/packages/catalog/test/gmi-cloud-provider.test.ts b/packages/catalog/test/gmi-cloud-provider.test.ts index 2c133349f..79c471586 100644 --- a/packages/catalog/test/gmi-cloud-provider.test.ts +++ b/packages/catalog/test/gmi-cloud-provider.test.ts @@ -1,6 +1,11 @@ import { describe, expect, test } from "bun:test"; +import { getBundledModelReferenceIndex } from "@oh-my-pi/pi-catalog/identity/bundled"; +import { resolveModelReference } from "@oh-my-pi/pi-catalog/identity/reference"; import { CATALOG_PROVIDERS } from "@oh-my-pi/pi-catalog/provider-models/descriptors"; -import { GMI_CLOUD_STATIC_MODELS } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import { + GMI_CLOUD_STATIC_MODELS, + gmiCloudModelManagerOptions, +} from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; describe("GMI Cloud provider", () => { test("static seed covers the descriptor's default model", () => { @@ -15,4 +20,52 @@ describe("GMI Cloud provider", () => { }); expect(GMI_CLOUD_STATIC_MODELS.map(model => model.id)).toContain("deepseek-ai/DeepSeek-V4-Flash"); }); + + // GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults + // carry no limits/reasoning/thinking. The mapper must recover intrinsic + // capability metadata for resold open-weight models from the cross-provider + // canonical index while never inheriting another provider's pricing. + test("dynamic discovery recovers canonical params without borrowing pricing", async () => { + // Self-select a bundled reasoning model with a thinking ladder that GMI + // does not seed — the very shape that regressed to null params — instead + // of hardcoding a churning model id. + const index = getBundledModelReferenceIndex(); + const resold = [...index.exact.values()].find(model => { + if (model.provider === "gmi-cloud" || !model.id.includes("/")) return false; + if (GMI_CLOUD_STATIC_MODELS.some(seed => seed.id === model.id)) return false; + const ref = resolveModelReference(model.id, index); + return ref?.reasoning === true && ref.thinking?.mode === "effort" && (ref.contextWindow ?? 0) > 0; + }); + if (!resold) { + throw new Error("no bundled resold reasoning model available to exercise canonical recovery"); + } + + const discoveredIds = [resold.id, "gmi-only/nonexistent-model"]; + const fetch = (async () => + new Response( + JSON.stringify({ + object: "list", + data: discoveredIds.map(id => ({ id, object: "model", created: 0, owned_by: "public" })), + }), + { status: 200, headers: { "content-type": "application/json" } }, + )) as unknown as typeof globalThis.fetch; + + const options = gmiCloudModelManagerOptions({ apiKey: "test-key", fetch }); + const models = (await options.fetchDynamicModels?.()) ?? []; + const byId = new Map(models.map(model => [model.id, model])); + + // Resold model recovers intrinsic capabilities from the canonical index... + const recovered = byId.get(resold.id); + const canonical = resolveModelReference(resold.id, index); + expect(recovered?.contextWindow).toBe(canonical?.contextWindow ?? null); + expect(recovered?.reasoning).toBe(true); + expect(recovered?.thinking?.mode).toBe("effort"); + // ...but pricing is never borrowed across providers. + expect(recovered?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }); + + // An id absent from every bundle stays bare rather than fabricating params. + const unknown = byId.get("gmi-only/nonexistent-model"); + expect(unknown?.contextWindow).toBeNull(); + expect(unknown?.reasoning).toBe(false); + }); });