From 7099457f8424de31675f581b6a296156aa4c24bc Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 19 Aug 2026 16:40:50 +0200 Subject: [PATCH] fix(catalog): resolved model routing and fallback failures in catalog - Fixed `muse-spark-1.2` and `muse-spark-1.2-contributor` failing on tool-call turns by mapping them to `openai-responses` in `OPENCODE_GO_API_ID_OVERRIDES`. - Added automatic fallback routing to borrow `openai-responses` routes from sibling gateways or billing-variant base IDs for gateway-first models. - Configured `dropCachedModelIdsOnStaticMismatch` to ensure cached models holding stale static metadata are invalidated on discovery. --- packages/catalog/CHANGELOG.md | 5 + .../src/provider-models/openai-compat.ts | 170 ++++++++++++------ .../catalog/test/opencode-provider.test.ts | 50 ++++++ 3 files changed, 175 insertions(+), 50 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 35e5de7f7..09200593f 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,11 @@ ## [Unreleased] +### Fixed + +- Fixed `opencode-go/muse-spark-1.2` and `muse-spark-1.2-contributor` still failing every tool-call turn with `OpenAI completions stream closed before a finish_reason was received` on 17.3.8. The earlier pin only covered the models.dev resolver, but models.dev omits these ids under `opencode-go` entirely, so live `/zen/go/v1/models` discovery had no bundled reference and defaulted them to chat completions. The per-id API pins now also apply inside the discovery mapper, and pinned ids invalidate cached routes written before the pin ([#8957](https://github.com/can1357/oh-my-pi/issues/8957)). +- Future gateway-first OpenCode models (ids the gateway serves before models.dev lists them, like muse-spark-1.2 was) no longer default to chat completions blindly: discovery now borrows the `openai-responses` route from the sibling gateway's catalog or the billing-variant base id (`-free`/`-contributor`). Only the responses signal is borrowed — anthropic transports genuinely diverge across the gateways (e.g. `minimax-m2.5`) and are never inferred. + ## [17.3.8] - 2026-08-19 ### Added diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 641b3d37b..f593e500a 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -2491,6 +2491,73 @@ function openCodeBaseUrlForApi(api: Api, basePath: string): string { return api === "anthropic-messages" ? basePath : `${basePath}/v1`; } +// Per-id API pins correcting upstream metadata mismatches on the OpenCode +// gateways. Applied in two places: the models.dev resolver rules +// (OPENCODE_ZEN_API_RESOLUTION / OPENCODE_GO_API_RESOLUTION) and the live +// /v1/models discovery mapper in openCodeModelManagerOptions — the gateways +// list ids that models.dev omits entirely (muse-spark-1.2[-contributor] on +// opencode-go, #8957), so discovery cannot rely on bundled references alone. +// +// OpenCode Zen: models.dev declares minimax-m3-free (and forward-compat +// minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but the Zen gateway +// only serves them at https://opencode.ai/zen/v1/chat/completions (verified +// against the live /v1/models response — minimax-m3-free is listed there, and +// the gateway has no /v1/messages route for it). Without this override the +// resolver POSTs anthropic-shaped requests to /v1/messages and the UI surfaces +// raw /<|minimax|>/ markup (#1617). +const OPENCODE_ZEN_API_ID_OVERRIDES: Readonly> = { + "minimax-m3": "openai-completions", + "minimax-m3-free": "openai-completions", +}; +// OpenCode Go: models.dev declares minimax-m2.7 / qwen3.5-plus / qwen3.6-plus +// (and now also minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but +// the OpenCode Go gateway only serves them at +// `https://opencode.ai/zen/go/v1/chat/completions` (verified against +// https://opencode.ai/zen/go/v1/models and the upstream endpoint table at +// https://opencode.ai/docs/go/#endpoints — minimax-m2.5 works the same way +// and lacks an `npm` field on models.dev so it already falls through to the +// openai-completions default). Without this override the resolver would POST +// anthropic-style requests to /v1/messages and the gateway would return its +// `Page Not Found` HTML (issue #887 for the qwen/m2.7 entries; minimax-m3 +// and minimax-m3-free added under #1617 for the same root cause). +// +// deepseek-v4-flash is the inverse case: it falls through to +// openai-completions by default, but the Go gateway's +// /zen/go/v1/chat/completions route does not work for this model while +// /zen/go/v1/responses does (user-verified against the live gateway, +// 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions). +// +// muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case, but +// worse: models.dev does not list them under opencode-go at all, so they have +// no bundled reference and only exist via live gateway discovery. Without the +// discovery-side pin they default to openai-completions even though the +// gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per +// https://opencode.ai/docs/go/#endpoints). The completions parser then closes +// the stream with no finish_reason on every tool-call turn (#8957). +const OPENCODE_GO_API_ID_OVERRIDES: Readonly> = { + "deepseek-v4-flash": "openai-responses", + "muse-spark-1.2": "openai-responses", + "muse-spark-1.2-contributor": "openai-responses", + "minimax-m2.7": "openai-completions", + "minimax-m3": "openai-completions", + "minimax-m3-free": "openai-completions", + "qwen3.5-plus": "openai-completions", + "qwen3.6-plus": "openai-completions", +}; + +// Billing-variant suffixes the OpenCode gateways append to a base model id +// without changing its transport (`deepseek-v4-flash-free`, +// `muse-spark-1.2-contributor`). +const OPENCODE_VARIANT_SUFFIXES = ["-contributor", "-free"] as const; + +/** Strips a billing-variant suffix; null when `id` is not a variant. */ +function openCodeBaseModelId(id: string): string | null { + for (const suffix of OPENCODE_VARIANT_SUFFIXES) { + if (id.endsWith(suffix) && id.length > suffix.length) return id.slice(0, -suffix.length); + } + return null; +} + function openCodeModelManagerOptions( providerId: "opencode-go" | "opencode-zen", config?: OpenCodeModelManagerConfig, @@ -2501,10 +2568,38 @@ function openCodeModelManagerOptions( const basePath = normalizeOpenCodeBasePath(config?.baseUrl, defaultBasePath); const discoveryBaseUrl = openCodeBaseUrlForApi("openai-completions", basePath); const references = createBundledReferenceMap(providerId); + // Both gateways share one operator with identical endpoint semantics, so + // the sibling's bundled catalog is a routing hint for ids models.dev has + // not picked up under this gateway yet. + const siblingReferences = createBundledReferenceMap( + providerId === "opencode-go" ? "opencode-zen" : "opencode-go", + ); + const apiOverrides = providerId === "opencode-go" ? OPENCODE_GO_API_ID_OVERRIDES : OPENCODE_ZEN_API_ID_OVERRIDES; + // Routes a discovered id with no same-provider metadata. models.dev lags + // the gateway (muse-spark-1.2[-contributor] shipped gateway-first, #8957), + // so borrow the openai-responses route from the sibling gateway or the + // billing-variant base id. Responses ONLY: openai-completions is already + // the default, and anthropic-messages transports genuinely diverge across + // gateways (e.g. minimax-m2.5), so borrowing them would import upstream + // metadata noise as hard routing errors. + const fallbackApi = (id: string, base: string | null): Api | undefined => { + const hints = [ + siblingReferences.get(id)?.api, + base ? references.get(base)?.api : undefined, + base ? siblingReferences.get(base)?.api : undefined, + ]; + return hints.includes("openai-responses") ? "openai-responses" : undefined; + }; return { providerId, cacheProviderId: resolveModelCacheProviderId(providerId, { apiKey, baseUrl: discoveryBaseUrl }), dynamicModelsAuthoritative: true, + // The per-id API pins are cache identity: without this, rows cached + // before a pin was added keep the wrong endpoint until TTL expiry + // (#8957 — 17.3.7 caches held muse-spark-1.2[-contributor] on chat + // completions after the pin shipped). Sibling-catalog drift is bounded + // by the 2h cache TTL instead. + dropCachedModelIdsOnStaticMismatch: Object.keys(apiOverrides), ...(apiKey && { fetchDynamicModels: () => fetchOpenAICompatibleModels({ @@ -2515,17 +2610,26 @@ function openCodeModelManagerOptions( mapModel: (entry, defaults) => { const reference = references.get(defaults.id); const name = toModelName(entry.name, reference?.name ?? defaults.name); + const base = openCodeBaseModelId(defaults.id); + // Pins win over bundled references (stale bundled routes + // must not stick), and a base-id pin covers its billing + // variants; the responses fallback covers gateway-first ids. + const api = + apiOverrides[defaults.id] ?? + (base ? apiOverrides[base] : undefined) ?? + reference?.api ?? + fallbackApi(defaults.id, base) ?? + defaults.api; + const baseUrl = openCodeBaseUrlForApi(api, basePath); if (!reference) { - return { - ...defaults, - name, - }; + return { ...defaults, name, api, baseUrl }; } return { ...reference, id: defaults.id, name, - baseUrl: openCodeBaseUrlForApi(reference.api, basePath), + api, + baseUrl, contextWindow: toPositiveNumber(entry.context_length, reference.contextWindow), maxTokens: toPositiveNumber(entry.max_completion_tokens, reference.maxTokens), }; @@ -5748,51 +5852,17 @@ function createOpenCodeApiResolution( }; } -// OpenCode Zen: models.dev declares minimax-m3-free (and forward-compat -// minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but the Zen gateway -// only serves them at https://opencode.ai/zen/v1/chat/completions (verified -// against the live /v1/models response — minimax-m3-free is listed there, and -// the gateway has no /v1/messages route for it). Without this override the -// resolver POSTs anthropic-shaped requests to /v1/messages and the UI surfaces -// raw /<|minimax|>/ markup (#1617). -const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen", { - "minimax-m3": "openai-completions", - "minimax-m3-free": "openai-completions", -}); -// OpenCode Go: models.dev declares minimax-m2.7 / qwen3.5-plus / qwen3.6-plus -// (and now also minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but -// the OpenCode Go gateway only serves them at -// `https://opencode.ai/zen/go/v1/chat/completions` (verified against -// https://opencode.ai/zen/go/v1/models and the upstream endpoint table at -// https://opencode.ai/docs/go/#endpoints — minimax-m2.5 works the same way -// and lacks an `npm` field on models.dev so it already falls through to the -// openai-completions default). Without this override the resolver would POST -// anthropic-style requests to /v1/messages and the gateway would return its -// `Page Not Found` HTML (issue #887 for the qwen/m2.7 entries; minimax-m3 -// and minimax-m3-free added under #1617 for the same root cause). -// -// deepseek-v4-flash is the inverse case: it falls through to -// openai-completions by default, but the Go gateway's -// /zen/go/v1/chat/completions route does not work for this model while -// /zen/go/v1/responses does (user-verified against the live gateway, -// 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions). -// -// muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case: the -// Go gateway's /zen/go/v1/models discovery drops the `provider.npm` hint, so -// without an override they fall through to openai-completions even though the -// gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per -// https://opencode.ai/docs/go/#endpoints). The completions parser then closes -// the stream with no finish_reason on every tool-call turn (#8957). -const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen/go", { - "deepseek-v4-flash": "openai-responses", - "muse-spark-1.2": "openai-responses", - "muse-spark-1.2-contributor": "openai-responses", - "minimax-m2.7": "openai-completions", - "minimax-m3": "openai-completions", - "minimax-m3-free": "openai-completions", - "qwen3.5-plus": "openai-completions", - "qwen3.6-plus": "openai-completions", -}); +// Resolver rules for the models.dev descriptor path; the per-id pins live in +// OPENCODE_ZEN_API_ID_OVERRIDES / OPENCODE_GO_API_ID_OVERRIDES (section 8), +// which also drive the live-discovery mapper. +const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution( + "https://opencode.ai/zen", + OPENCODE_ZEN_API_ID_OVERRIDES, +); +const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution( + "https://opencode.ai/zen/go", + OPENCODE_GO_API_ID_OVERRIDES, +); const COPILOT_BASE_URL = "https://api.githubcopilot.com"; diff --git a/packages/catalog/test/opencode-provider.test.ts b/packages/catalog/test/opencode-provider.test.ts index b8d157c17..f54422b5b 100644 --- a/packages/catalog/test/opencode-provider.test.ts +++ b/packages/catalog/test/opencode-provider.test.ts @@ -69,6 +69,56 @@ describe("OpenCode provider discovery", () => { } }); + test("pins gateway-only muse-spark ids to responses in live discovery (#8957)", async () => { + // models.dev omits muse-spark-1.2[-contributor] under opencode-go, so + // there is no bundled reference row. Without the discovery-side pin the + // mapper defaults them to openai-completions and every tool-call turn + // fails with "stream closed before a finish_reason was received". + const options = opencodeGoModelManagerOptions({ + apiKey: "test-key", + fetch: async () => modelListResponse(["muse-spark-1.2", "muse-spark-1.2-contributor", "kimi-k3"]), + }); + const models = await options.fetchDynamicModels?.(); + expect(models).not.toBeNull(); + const byId = new Map((models ?? []).map(model => [model.id, model])); + for (const id of ["muse-spark-1.2", "muse-spark-1.2-contributor"]) { + expect(byId.get(id)).toMatchObject({ + api: "openai-responses", + baseUrl: "https://opencode.ai/zen/go/v1", + }); + } + // Contrast: an unpinned id with a bundled reference keeps its route. + expect(byId.get("kimi-k3")).toMatchObject({ api: "openai-completions" }); + // Upgrade path: pinned ids invalidate caches written before the pin, + // otherwise 17.3.7-era rows keep the completions route until TTL. + expect(options.dropCachedModelIdsOnStaticMismatch).toContain("muse-spark-1.2-contributor"); + }); + + test("routes gateway-first ids via sibling catalog and variant-base hints", async () => { + // The Go gateway ships models before models.dev lists them under + // opencode-go (muse-spark-1.2[-contributor] did exactly this, #8957). + // With no same-provider metadata, the mapper borrows the + // openai-responses route from the sibling Zen catalog or the + // billing-variant base id — responses only, never anthropic-messages + // (cross-gateway transports genuinely diverge there). + const options = opencodeGoModelManagerOptions({ + apiKey: "test-key", + fetch: async () => + modelListResponse([ + "gpt-5.5", // zen bundles it as openai-responses; absent from the go bundle + "deepseek-v4-flash-free", // base id is pinned to responses on go + "minimax-m2.5-free", // anthropic hints only -> must keep the completions default + "brand-new-model", // no hint anywhere -> completions default + ]), + }); + const models = await options.fetchDynamicModels?.(); + const apiById = new Map((models ?? []).map(model => [model.id, model.api])); + expect(apiById.get("gpt-5.5")).toBe("openai-responses"); + expect(apiById.get("deepseek-v4-flash-free")).toBe("openai-responses"); + expect(apiById.get("minimax-m2.5-free")).toBe("openai-completions"); + expect(apiById.get("brand-new-model")).toBe("openai-completions"); + }); + test("replaces stale bundled Zen models with each credential's live endpoint list", async () => { const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-opencode-zen-")); try {