fix(catalog): recover gmi-cloud model params from canonical index

GMI Cloud's /v1/models returns only bare {id} rows, so dynamic discovery
resolved every model except the single bundled DeepSeek-V4-Flash seed with
a null context window, zero pricing, and no reasoning/thinking config.

Give the gmi-cloud mapper the same cross-provider canonical fallback that
SiliconFlow uses for the identical open-weight models: recover context
window, output limit, reasoning, and thinking ladder from the bundled
reference index while never borrowing another provider's pricing.

Fixes #8890
This commit is contained in:
roboomp
2026-08-19 09:57:19 +00:00
parent d94bdfa1bb
commit 8fc4555914
3 changed files with 107 additions and 2 deletions
+1
View File
@@ -14,6 +14,7 @@
- Fixed SuperGrok (`xai-oauth`) Grok 4.6 hiding the thinking-level picker: the Responses effort-capable allowlist now includes `grok-4.6`, so `/model` can select the documented `low`/`medium`/`high`/`xhigh` ladder (`max` is rejected by api.x.ai).
- Marked CoreWeave runtime discovery as authoritative so stale bundled model ids that the endpoint no longer serves stop appearing as selectable models.
- ChatGPT Codex discovery that advertises only worker `-wm` SKUs now also registers the plain model route, so a configured `openai-codex/<model>` keeps resolving instead of fuzzy-falling-back to the `-wm` SKU some accounts reject.
- Fixed GMI Cloud (`gmi-cloud`) models resolved via `/v1/models` discovery surfacing with `null` context windows, zero pricing, and no reasoning/thinking metadata for every model except the bundled `deepseek-ai/DeepSeek-V4-Flash` seed. GMI's endpoint returns only bare `{id}` rows, so the mapper now recovers intrinsic capability metadata (context window, output limit, reasoning, thinking ladder) for resold open-weight models from the cross-provider canonical reference index — matching the SiliconFlow behavior — while never borrowing another provider's pricing ([#8890](https://github.com/can1357/oh-my-pi/issues/8890)).
## [17.3.6] - 2026-08-17
@@ -1074,10 +1074,61 @@ export interface GmiCloudModelManagerConfig {
fetch?: FetchImpl;
}
/**
* Map a discovered GMI Cloud model to a full spec.
*
* GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults carry
* no limits, reasoning, or thinking metadata. When a gmi-cloud bundled
* reference exists (the seeded default) it supplies GMI's published tariff and
* limits directly. Every other id is an open-weight model GMI resells under its
* canonical id (`deepseek-ai/…`, `moonshotai/…`, `zai-org/…`, `Qwen/…`), so its
* intrinsic capabilities — context window, output limit, reasoning, thinking
* ladder — are recovered from any bundled upstream entry via the canonical
* reference index. Pricing is deliberately never borrowed across providers:
* GMI's per-model tariff is unknown for these ids, so cost stays zeroed rather
* than inheriting another provider's rate.
*/
function mapGmiCloudModel(
entry: OpenAICompatibleModelRecord,
defaults: ModelSpec<"openai-completions">,
reference: ModelSpec<"openai-completions"> | undefined,
): ModelSpec<"openai-completions"> {
if (reference) {
return mapWithBundledReference(entry, defaults, reference);
}
const canonical = resolveModelReference(defaults.id, getBundledModelReferenceIndex()) as
| ModelSpec<"openai-completions">
| undefined;
if (!canonical) {
return { ...defaults, name: toModelName(entry.name, defaults.name) };
}
const contextWindow = canonical.contextWindow ?? defaults.contextWindow;
const maxTokens =
canonical.maxTokens != null && contextWindow != null
? Math.min(canonical.maxTokens, contextWindow)
: (canonical.maxTokens ?? defaults.maxTokens);
return {
...defaults,
name: toModelName(entry.name, canonical.name ?? defaults.name),
reasoning: canonical.reasoning,
input: canonical.input,
...(canonical.thinking && { thinking: canonical.thinking }),
contextWindow,
maxTokens,
};
}
export function gmiCloudModelManagerOptions(
config?: GmiCloudModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
return createSimpleOpenAICompletionsOptions("gmi-cloud", GMI_CLOUD_BASE_URL, config);
return createOpenAICompatibleModelManagerOptions({
api: "openai-completions",
providerId: "gmi-cloud",
defaultBaseUrl: GMI_CLOUD_BASE_URL,
config,
requireApiKey: true,
mapModel: mapGmiCloudModel,
});
}
// ---------------------------------------------------------------------------
@@ -1,6 +1,11 @@
import { describe, expect, test } from "bun:test";
import { getBundledModelReferenceIndex } from "@oh-my-pi/pi-catalog/identity/bundled";
import { resolveModelReference } from "@oh-my-pi/pi-catalog/identity/reference";
import { CATALOG_PROVIDERS } from "@oh-my-pi/pi-catalog/provider-models/descriptors";
import { GMI_CLOUD_STATIC_MODELS } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import {
GMI_CLOUD_STATIC_MODELS,
gmiCloudModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
describe("GMI Cloud provider", () => {
test("static seed covers the descriptor's default model", () => {
@@ -15,4 +20,52 @@ describe("GMI Cloud provider", () => {
});
expect(GMI_CLOUD_STATIC_MODELS.map(model => model.id)).toContain("deepseek-ai/DeepSeek-V4-Flash");
});
// GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults
// carry no limits/reasoning/thinking. The mapper must recover intrinsic
// capability metadata for resold open-weight models from the cross-provider
// canonical index while never inheriting another provider's pricing.
test("dynamic discovery recovers canonical params without borrowing pricing", async () => {
// Self-select a bundled reasoning model with a thinking ladder that GMI
// does not seed — the very shape that regressed to null params — instead
// of hardcoding a churning model id.
const index = getBundledModelReferenceIndex();
const resold = [...index.exact.values()].find(model => {
if (model.provider === "gmi-cloud" || !model.id.includes("/")) return false;
if (GMI_CLOUD_STATIC_MODELS.some(seed => seed.id === model.id)) return false;
const ref = resolveModelReference(model.id, index);
return ref?.reasoning === true && ref.thinking?.mode === "effort" && (ref.contextWindow ?? 0) > 0;
});
if (!resold) {
throw new Error("no bundled resold reasoning model available to exercise canonical recovery");
}
const discoveredIds = [resold.id, "gmi-only/nonexistent-model"];
const fetch = (async () =>
new Response(
JSON.stringify({
object: "list",
data: discoveredIds.map(id => ({ id, object: "model", created: 0, owned_by: "public" })),
}),
{ status: 200, headers: { "content-type": "application/json" } },
)) as unknown as typeof globalThis.fetch;
const options = gmiCloudModelManagerOptions({ apiKey: "test-key", fetch });
const models = (await options.fetchDynamicModels?.()) ?? [];
const byId = new Map(models.map(model => [model.id, model]));
// Resold model recovers intrinsic capabilities from the canonical index...
const recovered = byId.get(resold.id);
const canonical = resolveModelReference(resold.id, index);
expect(recovered?.contextWindow).toBe(canonical?.contextWindow ?? null);
expect(recovered?.reasoning).toBe(true);
expect(recovered?.thinking?.mode).toBe("effort");
// ...but pricing is never borrowed across providers.
expect(recovered?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
// An id absent from every bundle stays bare rather than fabricating params.
const unknown = byId.get("gmi-only/nonexistent-model");
expect(unknown?.contextWindow).toBeNull();
expect(unknown?.reasoning).toBe(false);
});
});