Merge PR #8991: fix(catalog): recover gmi-cloud model params from canonical index (@roboomp)

This commit is contained in:
can1357
2026-08-19 12:13:16 +02:00
3 changed files with 107 additions and 2 deletions
+1
View File
@@ -19,6 +19,7 @@
- Fixed `opencode-go/muse-spark-1.2` (and `muse-spark-1.2-contributor`) failing every tool-call turn with `OpenAI completions stream closed before a finish_reason was received`. The Go gateway serves these ids only at `/zen/go/v1/responses`, but the `/zen/go/v1/models` discovery omits the `provider.npm` hint, so the resolver fell through to `openai-completions`; both ids are now pinned to `openai-responses` like `deepseek-v4-flash` ([#8957](https://github.com/can1357/oh-my-pi/issues/8957)).
- Fixed GitHub Copilot `grok-4.6` / `grok-4.6-1m` failing with HTTP 400 `unsupported_api_for_model` by routing them through the OpenAI Responses API (`/responses`) instead of `/chat/completions`, matching `grok-4.5`. Stale cached completion routes are invalidated on refresh ([#8807](https://github.com/can1357/oh-my-pi/issues/8807)).
- Fixed Cursor Grok 4.5/4.6 discovery classifying the versioned ids as non-reasoning: `GetUsableModels` ships no `thinkingDetails` and the bundled references read `reasoning: false`, so the picker hid the effort ladder. Discovery now marks `cursor-grok-<version>` ids as reasoning models (the non-reasoning `grok-code-*` ids stay out) ([#8803](https://github.com/can1357/oh-my-pi/issues/8803)).
- Fixed GMI Cloud (`gmi-cloud`) models resolved via `/v1/models` discovery surfacing with `null` context windows, zero pricing, and no reasoning/thinking metadata for every model except the bundled `deepseek-ai/DeepSeek-V4-Flash` seed. GMI's endpoint returns only bare `{id}` rows, so the mapper now recovers intrinsic capability metadata (context window, output limit, reasoning, thinking ladder) for resold open-weight models from the cross-provider canonical reference index — matching the SiliconFlow behavior — while never borrowing another provider's pricing ([#8890](https://github.com/can1357/oh-my-pi/issues/8890)).
## [17.3.6] - 2026-08-17
@@ -1074,10 +1074,61 @@ export interface GmiCloudModelManagerConfig {
fetch?: FetchImpl;
}
/**
* Map a discovered GMI Cloud model to a full spec.
*
* GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults carry
* no limits, reasoning, or thinking metadata. When a gmi-cloud bundled
* reference exists (the seeded default) it supplies GMI's published tariff and
* limits directly. Every other id is an open-weight model GMI resells under its
* canonical id (`deepseek-ai/…`, `moonshotai/…`, `zai-org/…`, `Qwen/…`), so its
* intrinsic capabilities — context window, output limit, reasoning, thinking
* ladder — are recovered from any bundled upstream entry via the canonical
* reference index. Pricing is deliberately never borrowed across providers:
* GMI's per-model tariff is unknown for these ids, so cost stays zeroed rather
* than inheriting another provider's rate.
*/
function mapGmiCloudModel(
entry: OpenAICompatibleModelRecord,
defaults: ModelSpec<"openai-completions">,
reference: ModelSpec<"openai-completions"> | undefined,
): ModelSpec<"openai-completions"> {
if (reference) {
return mapWithBundledReference(entry, defaults, reference);
}
const canonical = resolveModelReference(defaults.id, getBundledModelReferenceIndex()) as
| ModelSpec<"openai-completions">
| undefined;
if (!canonical) {
return { ...defaults, name: toModelName(entry.name, defaults.name) };
}
const contextWindow = canonical.contextWindow ?? defaults.contextWindow;
const maxTokens =
canonical.maxTokens != null && contextWindow != null
? Math.min(canonical.maxTokens, contextWindow)
: (canonical.maxTokens ?? defaults.maxTokens);
return {
...defaults,
name: toModelName(entry.name, canonical.name ?? defaults.name),
reasoning: canonical.reasoning,
input: canonical.input,
...(canonical.thinking && { thinking: canonical.thinking }),
contextWindow,
maxTokens,
};
}
export function gmiCloudModelManagerOptions(
config?: GmiCloudModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
return createSimpleOpenAICompletionsOptions("gmi-cloud", GMI_CLOUD_BASE_URL, config);
return createOpenAICompatibleModelManagerOptions({
api: "openai-completions",
providerId: "gmi-cloud",
defaultBaseUrl: GMI_CLOUD_BASE_URL,
config,
requireApiKey: true,
mapModel: mapGmiCloudModel,
});
}
// ---------------------------------------------------------------------------
@@ -1,6 +1,11 @@
import { describe, expect, test } from "bun:test";
import { getBundledModelReferenceIndex } from "@oh-my-pi/pi-catalog/identity/bundled";
import { resolveModelReference } from "@oh-my-pi/pi-catalog/identity/reference";
import { CATALOG_PROVIDERS } from "@oh-my-pi/pi-catalog/provider-models/descriptors";
import { GMI_CLOUD_STATIC_MODELS } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import {
GMI_CLOUD_STATIC_MODELS,
gmiCloudModelManagerOptions,
} from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
describe("GMI Cloud provider", () => {
test("static seed covers the descriptor's default model", () => {
@@ -15,4 +20,52 @@ describe("GMI Cloud provider", () => {
});
expect(GMI_CLOUD_STATIC_MODELS.map(model => model.id)).toContain("deepseek-ai/DeepSeek-V4-Flash");
});
// GMI's `/v1/models` returns only bare `{id}` rows, so discovery defaults
// carry no limits/reasoning/thinking. The mapper must recover intrinsic
// capability metadata for resold open-weight models from the cross-provider
// canonical index while never inheriting another provider's pricing.
test("dynamic discovery recovers canonical params without borrowing pricing", async () => {
// Self-select a bundled reasoning model with a thinking ladder that GMI
// does not seed — the very shape that regressed to null params — instead
// of hardcoding a churning model id.
const index = getBundledModelReferenceIndex();
const resold = [...index.exact.values()].find(model => {
if (model.provider === "gmi-cloud" || !model.id.includes("/")) return false;
if (GMI_CLOUD_STATIC_MODELS.some(seed => seed.id === model.id)) return false;
const ref = resolveModelReference(model.id, index);
return ref?.reasoning === true && ref.thinking?.mode === "effort" && (ref.contextWindow ?? 0) > 0;
});
if (!resold) {
throw new Error("no bundled resold reasoning model available to exercise canonical recovery");
}
const discoveredIds = [resold.id, "gmi-only/nonexistent-model"];
const fetch = (async () =>
new Response(
JSON.stringify({
object: "list",
data: discoveredIds.map(id => ({ id, object: "model", created: 0, owned_by: "public" })),
}),
{ status: 200, headers: { "content-type": "application/json" } },
)) as unknown as typeof globalThis.fetch;
const options = gmiCloudModelManagerOptions({ apiKey: "test-key", fetch });
const models = (await options.fetchDynamicModels?.()) ?? [];
const byId = new Map(models.map(model => [model.id, model]));
// Resold model recovers intrinsic capabilities from the canonical index...
const recovered = byId.get(resold.id);
const canonical = resolveModelReference(resold.id, index);
expect(recovered?.contextWindow).toBe(canonical?.contextWindow ?? null);
expect(recovered?.reasoning).toBe(true);
expect(recovered?.thinking?.mode).toBe("effort");
// ...but pricing is never borrowed across providers.
expect(recovered?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
// An id absent from every bundle stays bare rather than fabricating params.
const unknown = byId.get("gmi-only/nonexistent-model");
expect(unknown?.contextWindow).toBeNull();
expect(unknown?.reasoning).toBe(false);
});
});