fix(catalog): resolved model routing and fallback failures in catalog

- Fixed `muse-spark-1.2` and `muse-spark-1.2-contributor` failing on tool-call turns by mapping them to `openai-responses` in `OPENCODE_GO_API_ID_OVERRIDES`.
- Added automatic fallback routing to borrow `openai-responses` routes from sibling gateways or billing-variant base IDs for gateway-first models.
- Configured `dropCachedModelIdsOnStaticMismatch` to ensure cached models holding stale static metadata are invalidated on discovery.
This commit is contained in:
can1357
2026-08-19 16:40:50 +02:00
parent 416b30a8d5
commit 7099457f84
3 changed files with 175 additions and 50 deletions
+5
View File
@@ -2,6 +2,11 @@
## [Unreleased]
### Fixed
- Fixed `opencode-go/muse-spark-1.2` and `muse-spark-1.2-contributor` still failing every tool-call turn with `OpenAI completions stream closed before a finish_reason was received` on 17.3.8. The earlier pin only covered the models.dev resolver, but models.dev omits these ids under `opencode-go` entirely, so live `/zen/go/v1/models` discovery had no bundled reference and defaulted them to chat completions. The per-id API pins now also apply inside the discovery mapper, and pinned ids invalidate cached routes written before the pin ([#8957](https://github.com/can1357/oh-my-pi/issues/8957)).
- Future gateway-first OpenCode models (ids the gateway serves before models.dev lists them, like muse-spark-1.2 was) no longer default to chat completions blindly: discovery now borrows the `openai-responses` route from the sibling gateway's catalog or the billing-variant base id (`-free`/`-contributor`). Only the responses signal is borrowed — anthropic transports genuinely diverge across the gateways (e.g. `minimax-m2.5`) and are never inferred.
## [17.3.8] - 2026-08-19
### Added
@@ -2491,6 +2491,73 @@ function openCodeBaseUrlForApi(api: Api, basePath: string): string {
return api === "anthropic-messages" ? basePath : `${basePath}/v1`;
}
// Per-id API pins correcting upstream metadata mismatches on the OpenCode
// gateways. Applied in two places: the models.dev resolver rules
// (OPENCODE_ZEN_API_RESOLUTION / OPENCODE_GO_API_RESOLUTION) and the live
// /v1/models discovery mapper in openCodeModelManagerOptions — the gateways
// list ids that models.dev omits entirely (muse-spark-1.2[-contributor] on
// opencode-go, #8957), so discovery cannot rely on bundled references alone.
//
// OpenCode Zen: models.dev declares minimax-m3-free (and forward-compat
// minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but the Zen gateway
// only serves them at https://opencode.ai/zen/v1/chat/completions (verified
// against the live /v1/models response — minimax-m3-free is listed there, and
// the gateway has no /v1/messages route for it). Without this override the
// resolver POSTs anthropic-shaped requests to /v1/messages and the UI surfaces
// raw <invoke>/<|minimax|>/<tool_call> markup (#1617).
const OPENCODE_ZEN_API_ID_OVERRIDES: Readonly<Record<string, Api>> = {
"minimax-m3": "openai-completions",
"minimax-m3-free": "openai-completions",
};
// OpenCode Go: models.dev declares minimax-m2.7 / qwen3.5-plus / qwen3.6-plus
// (and now also minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but
// the OpenCode Go gateway only serves them at
// `https://opencode.ai/zen/go/v1/chat/completions` (verified against
// https://opencode.ai/zen/go/v1/models and the upstream endpoint table at
// https://opencode.ai/docs/go/#endpoints — minimax-m2.5 works the same way
// and lacks an `npm` field on models.dev so it already falls through to the
// openai-completions default). Without this override the resolver would POST
// anthropic-style requests to /v1/messages and the gateway would return its
// `Page Not Found` HTML (issue #887 for the qwen/m2.7 entries; minimax-m3
// and minimax-m3-free added under #1617 for the same root cause).
//
// deepseek-v4-flash is the inverse case: it falls through to
// openai-completions by default, but the Go gateway's
// /zen/go/v1/chat/completions route does not work for this model while
// /zen/go/v1/responses does (user-verified against the live gateway,
// 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions).
//
// muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case, but
// worse: models.dev does not list them under opencode-go at all, so they have
// no bundled reference and only exist via live gateway discovery. Without the
// discovery-side pin they default to openai-completions even though the
// gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per
// https://opencode.ai/docs/go/#endpoints). The completions parser then closes
// the stream with no finish_reason on every tool-call turn (#8957).
const OPENCODE_GO_API_ID_OVERRIDES: Readonly<Record<string, Api>> = {
"deepseek-v4-flash": "openai-responses",
"muse-spark-1.2": "openai-responses",
"muse-spark-1.2-contributor": "openai-responses",
"minimax-m2.7": "openai-completions",
"minimax-m3": "openai-completions",
"minimax-m3-free": "openai-completions",
"qwen3.5-plus": "openai-completions",
"qwen3.6-plus": "openai-completions",
};
// Billing-variant suffixes the OpenCode gateways append to a base model id
// without changing its transport (`deepseek-v4-flash-free`,
// `muse-spark-1.2-contributor`).
const OPENCODE_VARIANT_SUFFIXES = ["-contributor", "-free"] as const;
/** Strips a billing-variant suffix; null when `id` is not a variant. */
function openCodeBaseModelId(id: string): string | null {
for (const suffix of OPENCODE_VARIANT_SUFFIXES) {
if (id.endsWith(suffix) && id.length > suffix.length) return id.slice(0, -suffix.length);
}
return null;
}
function openCodeModelManagerOptions(
providerId: "opencode-go" | "opencode-zen",
config?: OpenCodeModelManagerConfig,
@@ -2501,10 +2568,38 @@ function openCodeModelManagerOptions(
const basePath = normalizeOpenCodeBasePath(config?.baseUrl, defaultBasePath);
const discoveryBaseUrl = openCodeBaseUrlForApi("openai-completions", basePath);
const references = createBundledReferenceMap<Api>(providerId);
// Both gateways share one operator with identical endpoint semantics, so
// the sibling's bundled catalog is a routing hint for ids models.dev has
// not picked up under this gateway yet.
const siblingReferences = createBundledReferenceMap<Api>(
providerId === "opencode-go" ? "opencode-zen" : "opencode-go",
);
const apiOverrides = providerId === "opencode-go" ? OPENCODE_GO_API_ID_OVERRIDES : OPENCODE_ZEN_API_ID_OVERRIDES;
// Routes a discovered id with no same-provider metadata. models.dev lags
// the gateway (muse-spark-1.2[-contributor] shipped gateway-first, #8957),
// so borrow the openai-responses route from the sibling gateway or the
// billing-variant base id. Responses ONLY: openai-completions is already
// the default, and anthropic-messages transports genuinely diverge across
// gateways (e.g. minimax-m2.5), so borrowing them would import upstream
// metadata noise as hard routing errors.
const fallbackApi = (id: string, base: string | null): Api | undefined => {
const hints = [
siblingReferences.get(id)?.api,
base ? references.get(base)?.api : undefined,
base ? siblingReferences.get(base)?.api : undefined,
];
return hints.includes("openai-responses") ? "openai-responses" : undefined;
};
return {
providerId,
cacheProviderId: resolveModelCacheProviderId(providerId, { apiKey, baseUrl: discoveryBaseUrl }),
dynamicModelsAuthoritative: true,
// The per-id API pins are cache identity: without this, rows cached
// before a pin was added keep the wrong endpoint until TTL expiry
// (#8957 — 17.3.7 caches held muse-spark-1.2[-contributor] on chat
// completions after the pin shipped). Sibling-catalog drift is bounded
// by the 2h cache TTL instead.
dropCachedModelIdsOnStaticMismatch: Object.keys(apiOverrides),
...(apiKey && {
fetchDynamicModels: () =>
fetchOpenAICompatibleModels<Api>({
@@ -2515,17 +2610,26 @@ function openCodeModelManagerOptions(
mapModel: (entry, defaults) => {
const reference = references.get(defaults.id);
const name = toModelName(entry.name, reference?.name ?? defaults.name);
const base = openCodeBaseModelId(defaults.id);
// Pins win over bundled references (stale bundled routes
// must not stick), and a base-id pin covers its billing
// variants; the responses fallback covers gateway-first ids.
const api =
apiOverrides[defaults.id] ??
(base ? apiOverrides[base] : undefined) ??
reference?.api ??
fallbackApi(defaults.id, base) ??
defaults.api;
const baseUrl = openCodeBaseUrlForApi(api, basePath);
if (!reference) {
return {
...defaults,
name,
};
return { ...defaults, name, api, baseUrl };
}
return {
...reference,
id: defaults.id,
name,
baseUrl: openCodeBaseUrlForApi(reference.api, basePath),
api,
baseUrl,
contextWindow: toPositiveNumber(entry.context_length, reference.contextWindow),
maxTokens: toPositiveNumber(entry.max_completion_tokens, reference.maxTokens),
};
@@ -5748,51 +5852,17 @@ function createOpenCodeApiResolution(
};
}
// OpenCode Zen: models.dev declares minimax-m3-free (and forward-compat
// minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but the Zen gateway
// only serves them at https://opencode.ai/zen/v1/chat/completions (verified
// against the live /v1/models response — minimax-m3-free is listed there, and
// the gateway has no /v1/messages route for it). Without this override the
// resolver POSTs anthropic-shaped requests to /v1/messages and the UI surfaces
// raw <invoke>/<|minimax|>/<tool_call> markup (#1617).
const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen", {
"minimax-m3": "openai-completions",
"minimax-m3-free": "openai-completions",
});
// OpenCode Go: models.dev declares minimax-m2.7 / qwen3.5-plus / qwen3.6-plus
// (and now also minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but
// the OpenCode Go gateway only serves them at
// `https://opencode.ai/zen/go/v1/chat/completions` (verified against
// https://opencode.ai/zen/go/v1/models and the upstream endpoint table at
// https://opencode.ai/docs/go/#endpoints — minimax-m2.5 works the same way
// and lacks an `npm` field on models.dev so it already falls through to the
// openai-completions default). Without this override the resolver would POST
// anthropic-style requests to /v1/messages and the gateway would return its
// `Page Not Found` HTML (issue #887 for the qwen/m2.7 entries; minimax-m3
// and minimax-m3-free added under #1617 for the same root cause).
//
// deepseek-v4-flash is the inverse case: it falls through to
// openai-completions by default, but the Go gateway's
// /zen/go/v1/chat/completions route does not work for this model while
// /zen/go/v1/responses does (user-verified against the live gateway,
// 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions).
//
// muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case: the
// Go gateway's /zen/go/v1/models discovery drops the `provider.npm` hint, so
// without an override they fall through to openai-completions even though the
// gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per
// https://opencode.ai/docs/go/#endpoints). The completions parser then closes
// the stream with no finish_reason on every tool-call turn (#8957).
const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen/go", {
"deepseek-v4-flash": "openai-responses",
"muse-spark-1.2": "openai-responses",
"muse-spark-1.2-contributor": "openai-responses",
"minimax-m2.7": "openai-completions",
"minimax-m3": "openai-completions",
"minimax-m3-free": "openai-completions",
"qwen3.5-plus": "openai-completions",
"qwen3.6-plus": "openai-completions",
});
// Resolver rules for the models.dev descriptor path; the per-id pins live in
// OPENCODE_ZEN_API_ID_OVERRIDES / OPENCODE_GO_API_ID_OVERRIDES (section 8),
// which also drive the live-discovery mapper.
const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution(
"https://opencode.ai/zen",
OPENCODE_ZEN_API_ID_OVERRIDES,
);
const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution(
"https://opencode.ai/zen/go",
OPENCODE_GO_API_ID_OVERRIDES,
);
const COPILOT_BASE_URL = "https://api.githubcopilot.com";
@@ -69,6 +69,56 @@ describe("OpenCode provider discovery", () => {
}
});
test("pins gateway-only muse-spark ids to responses in live discovery (#8957)", async () => {
// models.dev omits muse-spark-1.2[-contributor] under opencode-go, so
// there is no bundled reference row. Without the discovery-side pin the
// mapper defaults them to openai-completions and every tool-call turn
// fails with "stream closed before a finish_reason was received".
const options = opencodeGoModelManagerOptions({
apiKey: "test-key",
fetch: async () => modelListResponse(["muse-spark-1.2", "muse-spark-1.2-contributor", "kimi-k3"]),
});
const models = await options.fetchDynamicModels?.();
expect(models).not.toBeNull();
const byId = new Map((models ?? []).map(model => [model.id, model]));
for (const id of ["muse-spark-1.2", "muse-spark-1.2-contributor"]) {
expect(byId.get(id)).toMatchObject({
api: "openai-responses",
baseUrl: "https://opencode.ai/zen/go/v1",
});
}
// Contrast: an unpinned id with a bundled reference keeps its route.
expect(byId.get("kimi-k3")).toMatchObject({ api: "openai-completions" });
// Upgrade path: pinned ids invalidate caches written before the pin,
// otherwise 17.3.7-era rows keep the completions route until TTL.
expect(options.dropCachedModelIdsOnStaticMismatch).toContain("muse-spark-1.2-contributor");
});
test("routes gateway-first ids via sibling catalog and variant-base hints", async () => {
// The Go gateway ships models before models.dev lists them under
// opencode-go (muse-spark-1.2[-contributor] did exactly this, #8957).
// With no same-provider metadata, the mapper borrows the
// openai-responses route from the sibling Zen catalog or the
// billing-variant base id — responses only, never anthropic-messages
// (cross-gateway transports genuinely diverge there).
const options = opencodeGoModelManagerOptions({
apiKey: "test-key",
fetch: async () =>
modelListResponse([
"gpt-5.5", // zen bundles it as openai-responses; absent from the go bundle
"deepseek-v4-flash-free", // base id is pinned to responses on go
"minimax-m2.5-free", // anthropic hints only -> must keep the completions default
"brand-new-model", // no hint anywhere -> completions default
]),
});
const models = await options.fetchDynamicModels?.();
const apiById = new Map((models ?? []).map(model => [model.id, model.api]));
expect(apiById.get("gpt-5.5")).toBe("openai-responses");
expect(apiById.get("deepseek-v4-flash-free")).toBe("openai-responses");
expect(apiById.get("minimax-m2.5-free")).toBe("openai-completions");
expect(apiById.get("brand-new-model")).toBe("openai-completions");
});
test("replaces stale bundled Zen models with each credential's live endpoint list", async () => {
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-opencode-zen-"));
try {