fix(catalog): resolved model routing and fallback failures in catalog
- Fixed `muse-spark-1.2` and `muse-spark-1.2-contributor` failing on tool-call turns by mapping them to `openai-responses` in `OPENCODE_GO_API_ID_OVERRIDES`. - Added automatic fallback routing to borrow `openai-responses` routes from sibling gateways or billing-variant base IDs for gateway-first models. - Configured `dropCachedModelIdsOnStaticMismatch` to ensure cached models holding stale static metadata are invalidated on discovery.
This commit is contained in:
@@ -2,6 +2,11 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed `opencode-go/muse-spark-1.2` and `muse-spark-1.2-contributor` still failing every tool-call turn with `OpenAI completions stream closed before a finish_reason was received` on 17.3.8. The earlier pin only covered the models.dev resolver, but models.dev omits these ids under `opencode-go` entirely, so live `/zen/go/v1/models` discovery had no bundled reference and defaulted them to chat completions. The per-id API pins now also apply inside the discovery mapper, and pinned ids invalidate cached routes written before the pin ([#8957](https://github.com/can1357/oh-my-pi/issues/8957)).
|
||||
- Future gateway-first OpenCode models (ids the gateway serves before models.dev lists them, like muse-spark-1.2 was) no longer default to chat completions blindly: discovery now borrows the `openai-responses` route from the sibling gateway's catalog or the billing-variant base id (`-free`/`-contributor`). Only the responses signal is borrowed — anthropic transports genuinely diverge across the gateways (e.g. `minimax-m2.5`) and are never inferred.
|
||||
|
||||
## [17.3.8] - 2026-08-19
|
||||
|
||||
### Added
|
||||
|
||||
@@ -2491,6 +2491,73 @@ function openCodeBaseUrlForApi(api: Api, basePath: string): string {
|
||||
return api === "anthropic-messages" ? basePath : `${basePath}/v1`;
|
||||
}
|
||||
|
||||
// Per-id API pins correcting upstream metadata mismatches on the OpenCode
|
||||
// gateways. Applied in two places: the models.dev resolver rules
|
||||
// (OPENCODE_ZEN_API_RESOLUTION / OPENCODE_GO_API_RESOLUTION) and the live
|
||||
// /v1/models discovery mapper in openCodeModelManagerOptions — the gateways
|
||||
// list ids that models.dev omits entirely (muse-spark-1.2[-contributor] on
|
||||
// opencode-go, #8957), so discovery cannot rely on bundled references alone.
|
||||
//
|
||||
// OpenCode Zen: models.dev declares minimax-m3-free (and forward-compat
|
||||
// minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but the Zen gateway
|
||||
// only serves them at https://opencode.ai/zen/v1/chat/completions (verified
|
||||
// against the live /v1/models response — minimax-m3-free is listed there, and
|
||||
// the gateway has no /v1/messages route for it). Without this override the
|
||||
// resolver POSTs anthropic-shaped requests to /v1/messages and the UI surfaces
|
||||
// raw <invoke>/<|minimax|>/<tool_call> markup (#1617).
|
||||
const OPENCODE_ZEN_API_ID_OVERRIDES: Readonly<Record<string, Api>> = {
|
||||
"minimax-m3": "openai-completions",
|
||||
"minimax-m3-free": "openai-completions",
|
||||
};
|
||||
// OpenCode Go: models.dev declares minimax-m2.7 / qwen3.5-plus / qwen3.6-plus
|
||||
// (and now also minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but
|
||||
// the OpenCode Go gateway only serves them at
|
||||
// `https://opencode.ai/zen/go/v1/chat/completions` (verified against
|
||||
// https://opencode.ai/zen/go/v1/models and the upstream endpoint table at
|
||||
// https://opencode.ai/docs/go/#endpoints — minimax-m2.5 works the same way
|
||||
// and lacks an `npm` field on models.dev so it already falls through to the
|
||||
// openai-completions default). Without this override the resolver would POST
|
||||
// anthropic-style requests to /v1/messages and the gateway would return its
|
||||
// `Page Not Found` HTML (issue #887 for the qwen/m2.7 entries; minimax-m3
|
||||
// and minimax-m3-free added under #1617 for the same root cause).
|
||||
//
|
||||
// deepseek-v4-flash is the inverse case: it falls through to
|
||||
// openai-completions by default, but the Go gateway's
|
||||
// /zen/go/v1/chat/completions route does not work for this model while
|
||||
// /zen/go/v1/responses does (user-verified against the live gateway,
|
||||
// 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions).
|
||||
//
|
||||
// muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case, but
|
||||
// worse: models.dev does not list them under opencode-go at all, so they have
|
||||
// no bundled reference and only exist via live gateway discovery. Without the
|
||||
// discovery-side pin they default to openai-completions even though the
|
||||
// gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per
|
||||
// https://opencode.ai/docs/go/#endpoints). The completions parser then closes
|
||||
// the stream with no finish_reason on every tool-call turn (#8957).
|
||||
const OPENCODE_GO_API_ID_OVERRIDES: Readonly<Record<string, Api>> = {
|
||||
"deepseek-v4-flash": "openai-responses",
|
||||
"muse-spark-1.2": "openai-responses",
|
||||
"muse-spark-1.2-contributor": "openai-responses",
|
||||
"minimax-m2.7": "openai-completions",
|
||||
"minimax-m3": "openai-completions",
|
||||
"minimax-m3-free": "openai-completions",
|
||||
"qwen3.5-plus": "openai-completions",
|
||||
"qwen3.6-plus": "openai-completions",
|
||||
};
|
||||
|
||||
// Billing-variant suffixes the OpenCode gateways append to a base model id
|
||||
// without changing its transport (`deepseek-v4-flash-free`,
|
||||
// `muse-spark-1.2-contributor`).
|
||||
const OPENCODE_VARIANT_SUFFIXES = ["-contributor", "-free"] as const;
|
||||
|
||||
/** Strips a billing-variant suffix; null when `id` is not a variant. */
|
||||
function openCodeBaseModelId(id: string): string | null {
|
||||
for (const suffix of OPENCODE_VARIANT_SUFFIXES) {
|
||||
if (id.endsWith(suffix) && id.length > suffix.length) return id.slice(0, -suffix.length);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function openCodeModelManagerOptions(
|
||||
providerId: "opencode-go" | "opencode-zen",
|
||||
config?: OpenCodeModelManagerConfig,
|
||||
@@ -2501,10 +2568,38 @@ function openCodeModelManagerOptions(
|
||||
const basePath = normalizeOpenCodeBasePath(config?.baseUrl, defaultBasePath);
|
||||
const discoveryBaseUrl = openCodeBaseUrlForApi("openai-completions", basePath);
|
||||
const references = createBundledReferenceMap<Api>(providerId);
|
||||
// Both gateways share one operator with identical endpoint semantics, so
|
||||
// the sibling's bundled catalog is a routing hint for ids models.dev has
|
||||
// not picked up under this gateway yet.
|
||||
const siblingReferences = createBundledReferenceMap<Api>(
|
||||
providerId === "opencode-go" ? "opencode-zen" : "opencode-go",
|
||||
);
|
||||
const apiOverrides = providerId === "opencode-go" ? OPENCODE_GO_API_ID_OVERRIDES : OPENCODE_ZEN_API_ID_OVERRIDES;
|
||||
// Routes a discovered id with no same-provider metadata. models.dev lags
|
||||
// the gateway (muse-spark-1.2[-contributor] shipped gateway-first, #8957),
|
||||
// so borrow the openai-responses route from the sibling gateway or the
|
||||
// billing-variant base id. Responses ONLY: openai-completions is already
|
||||
// the default, and anthropic-messages transports genuinely diverge across
|
||||
// gateways (e.g. minimax-m2.5), so borrowing them would import upstream
|
||||
// metadata noise as hard routing errors.
|
||||
const fallbackApi = (id: string, base: string | null): Api | undefined => {
|
||||
const hints = [
|
||||
siblingReferences.get(id)?.api,
|
||||
base ? references.get(base)?.api : undefined,
|
||||
base ? siblingReferences.get(base)?.api : undefined,
|
||||
];
|
||||
return hints.includes("openai-responses") ? "openai-responses" : undefined;
|
||||
};
|
||||
return {
|
||||
providerId,
|
||||
cacheProviderId: resolveModelCacheProviderId(providerId, { apiKey, baseUrl: discoveryBaseUrl }),
|
||||
dynamicModelsAuthoritative: true,
|
||||
// The per-id API pins are cache identity: without this, rows cached
|
||||
// before a pin was added keep the wrong endpoint until TTL expiry
|
||||
// (#8957 — 17.3.7 caches held muse-spark-1.2[-contributor] on chat
|
||||
// completions after the pin shipped). Sibling-catalog drift is bounded
|
||||
// by the 2h cache TTL instead.
|
||||
dropCachedModelIdsOnStaticMismatch: Object.keys(apiOverrides),
|
||||
...(apiKey && {
|
||||
fetchDynamicModels: () =>
|
||||
fetchOpenAICompatibleModels<Api>({
|
||||
@@ -2515,17 +2610,26 @@ function openCodeModelManagerOptions(
|
||||
mapModel: (entry, defaults) => {
|
||||
const reference = references.get(defaults.id);
|
||||
const name = toModelName(entry.name, reference?.name ?? defaults.name);
|
||||
const base = openCodeBaseModelId(defaults.id);
|
||||
// Pins win over bundled references (stale bundled routes
|
||||
// must not stick), and a base-id pin covers its billing
|
||||
// variants; the responses fallback covers gateway-first ids.
|
||||
const api =
|
||||
apiOverrides[defaults.id] ??
|
||||
(base ? apiOverrides[base] : undefined) ??
|
||||
reference?.api ??
|
||||
fallbackApi(defaults.id, base) ??
|
||||
defaults.api;
|
||||
const baseUrl = openCodeBaseUrlForApi(api, basePath);
|
||||
if (!reference) {
|
||||
return {
|
||||
...defaults,
|
||||
name,
|
||||
};
|
||||
return { ...defaults, name, api, baseUrl };
|
||||
}
|
||||
return {
|
||||
...reference,
|
||||
id: defaults.id,
|
||||
name,
|
||||
baseUrl: openCodeBaseUrlForApi(reference.api, basePath),
|
||||
api,
|
||||
baseUrl,
|
||||
contextWindow: toPositiveNumber(entry.context_length, reference.contextWindow),
|
||||
maxTokens: toPositiveNumber(entry.max_completion_tokens, reference.maxTokens),
|
||||
};
|
||||
@@ -5748,51 +5852,17 @@ function createOpenCodeApiResolution(
|
||||
};
|
||||
}
|
||||
|
||||
// OpenCode Zen: models.dev declares minimax-m3-free (and forward-compat
|
||||
// minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but the Zen gateway
|
||||
// only serves them at https://opencode.ai/zen/v1/chat/completions (verified
|
||||
// against the live /v1/models response — minimax-m3-free is listed there, and
|
||||
// the gateway has no /v1/messages route for it). Without this override the
|
||||
// resolver POSTs anthropic-shaped requests to /v1/messages and the UI surfaces
|
||||
// raw <invoke>/<|minimax|>/<tool_call> markup (#1617).
|
||||
const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen", {
|
||||
"minimax-m3": "openai-completions",
|
||||
"minimax-m3-free": "openai-completions",
|
||||
});
|
||||
// OpenCode Go: models.dev declares minimax-m2.7 / qwen3.5-plus / qwen3.6-plus
|
||||
// (and now also minimax-m3) with `provider.npm = "@ai-sdk/anthropic"`, but
|
||||
// the OpenCode Go gateway only serves them at
|
||||
// `https://opencode.ai/zen/go/v1/chat/completions` (verified against
|
||||
// https://opencode.ai/zen/go/v1/models and the upstream endpoint table at
|
||||
// https://opencode.ai/docs/go/#endpoints — minimax-m2.5 works the same way
|
||||
// and lacks an `npm` field on models.dev so it already falls through to the
|
||||
// openai-completions default). Without this override the resolver would POST
|
||||
// anthropic-style requests to /v1/messages and the gateway would return its
|
||||
// `Page Not Found` HTML (issue #887 for the qwen/m2.7 entries; minimax-m3
|
||||
// and minimax-m3-free added under #1617 for the same root cause).
|
||||
//
|
||||
// deepseek-v4-flash is the inverse case: it falls through to
|
||||
// openai-completions by default, but the Go gateway's
|
||||
// /zen/go/v1/chat/completions route does not work for this model while
|
||||
// /zen/go/v1/responses does (user-verified against the live gateway,
|
||||
// 2026-08-08; Flash only — deepseek-v4-pro serves fine on chat completions).
|
||||
//
|
||||
// muse-spark-1.2 / muse-spark-1.2-contributor are the same inverse case: the
|
||||
// Go gateway's /zen/go/v1/models discovery drops the `provider.npm` hint, so
|
||||
// without an override they fall through to openai-completions even though the
|
||||
// gateway only serves them at /zen/go/v1/responses (@ai-sdk/openai per
|
||||
// https://opencode.ai/docs/go/#endpoints). The completions parser then closes
|
||||
// the stream with no finish_reason on every tool-call turn (#8957).
|
||||
const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution("https://opencode.ai/zen/go", {
|
||||
"deepseek-v4-flash": "openai-responses",
|
||||
"muse-spark-1.2": "openai-responses",
|
||||
"muse-spark-1.2-contributor": "openai-responses",
|
||||
"minimax-m2.7": "openai-completions",
|
||||
"minimax-m3": "openai-completions",
|
||||
"minimax-m3-free": "openai-completions",
|
||||
"qwen3.5-plus": "openai-completions",
|
||||
"qwen3.6-plus": "openai-completions",
|
||||
});
|
||||
// Resolver rules for the models.dev descriptor path; the per-id pins live in
|
||||
// OPENCODE_ZEN_API_ID_OVERRIDES / OPENCODE_GO_API_ID_OVERRIDES (section 8),
|
||||
// which also drive the live-discovery mapper.
|
||||
const OPENCODE_ZEN_API_RESOLUTION = createOpenCodeApiResolution(
|
||||
"https://opencode.ai/zen",
|
||||
OPENCODE_ZEN_API_ID_OVERRIDES,
|
||||
);
|
||||
const OPENCODE_GO_API_RESOLUTION = createOpenCodeApiResolution(
|
||||
"https://opencode.ai/zen/go",
|
||||
OPENCODE_GO_API_ID_OVERRIDES,
|
||||
);
|
||||
|
||||
const COPILOT_BASE_URL = "https://api.githubcopilot.com";
|
||||
|
||||
|
||||
@@ -69,6 +69,56 @@ describe("OpenCode provider discovery", () => {
|
||||
}
|
||||
});
|
||||
|
||||
test("pins gateway-only muse-spark ids to responses in live discovery (#8957)", async () => {
|
||||
// models.dev omits muse-spark-1.2[-contributor] under opencode-go, so
|
||||
// there is no bundled reference row. Without the discovery-side pin the
|
||||
// mapper defaults them to openai-completions and every tool-call turn
|
||||
// fails with "stream closed before a finish_reason was received".
|
||||
const options = opencodeGoModelManagerOptions({
|
||||
apiKey: "test-key",
|
||||
fetch: async () => modelListResponse(["muse-spark-1.2", "muse-spark-1.2-contributor", "kimi-k3"]),
|
||||
});
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
expect(models).not.toBeNull();
|
||||
const byId = new Map((models ?? []).map(model => [model.id, model]));
|
||||
for (const id of ["muse-spark-1.2", "muse-spark-1.2-contributor"]) {
|
||||
expect(byId.get(id)).toMatchObject({
|
||||
api: "openai-responses",
|
||||
baseUrl: "https://opencode.ai/zen/go/v1",
|
||||
});
|
||||
}
|
||||
// Contrast: an unpinned id with a bundled reference keeps its route.
|
||||
expect(byId.get("kimi-k3")).toMatchObject({ api: "openai-completions" });
|
||||
// Upgrade path: pinned ids invalidate caches written before the pin,
|
||||
// otherwise 17.3.7-era rows keep the completions route until TTL.
|
||||
expect(options.dropCachedModelIdsOnStaticMismatch).toContain("muse-spark-1.2-contributor");
|
||||
});
|
||||
|
||||
test("routes gateway-first ids via sibling catalog and variant-base hints", async () => {
|
||||
// The Go gateway ships models before models.dev lists them under
|
||||
// opencode-go (muse-spark-1.2[-contributor] did exactly this, #8957).
|
||||
// With no same-provider metadata, the mapper borrows the
|
||||
// openai-responses route from the sibling Zen catalog or the
|
||||
// billing-variant base id — responses only, never anthropic-messages
|
||||
// (cross-gateway transports genuinely diverge there).
|
||||
const options = opencodeGoModelManagerOptions({
|
||||
apiKey: "test-key",
|
||||
fetch: async () =>
|
||||
modelListResponse([
|
||||
"gpt-5.5", // zen bundles it as openai-responses; absent from the go bundle
|
||||
"deepseek-v4-flash-free", // base id is pinned to responses on go
|
||||
"minimax-m2.5-free", // anthropic hints only -> must keep the completions default
|
||||
"brand-new-model", // no hint anywhere -> completions default
|
||||
]),
|
||||
});
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
const apiById = new Map((models ?? []).map(model => [model.id, model.api]));
|
||||
expect(apiById.get("gpt-5.5")).toBe("openai-responses");
|
||||
expect(apiById.get("deepseek-v4-flash-free")).toBe("openai-responses");
|
||||
expect(apiById.get("minimax-m2.5-free")).toBe("openai-completions");
|
||||
expect(apiById.get("brand-new-model")).toBe("openai-completions");
|
||||
});
|
||||
|
||||
test("replaces stale bundled Zen models with each credential's live endpoint list", async () => {
|
||||
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-opencode-zen-"));
|
||||
try {
|
||||
|
||||
Reference in New Issue
Block a user