feat(catalog/provider-models): added OpenAI model provider descriptors

- Updated default model identifiers across many catalog providers to newer model versions.
- Renamed a couple OpenAI compatibility provider descriptors, including Together and Zhipu coding-plan identifiers.
- Added multiple new OpenAI-compatible specialized provider descriptors for additional model provider families.
This commit is contained in:
can1357
2026-06-15 10:27:44 +02:00
parent e680bc0ca3
commit 4d95a0e1dc
10 changed files with 9714 additions and 3820 deletions
+6 -1
View File
@@ -3,17 +3,22 @@
## [Unreleased]
### Added
- Added Azure OpenAI as a catalog provider (`azure`, default model `gpt-4o`, env var `AZURE_OPENAI_API_KEY`), bundling the OpenAI-family models Azure serves over the Responses API (GPT-4/4.1/4o, GPT-5 family, o-series, Codex). Like Amazon Bedrock it is catalog-only — models ship in the bundle and become selectable once the env key is set, with the deployment base URL resolved at runtime from `AZURE_OPENAI_BASE_URL`/`AZURE_OPENAI_RESOURCE_NAME`.
- Added Azure OpenAI as a catalog provider (`azure`, default model `gpt-5.5`, env var `AZURE_OPENAI_API_KEY`), bundling the OpenAI-family models Azure serves over the Responses API (GPT-4/4.1/4o, GPT-5 family, o-series, Codex). Like Amazon Bedrock it is catalog-only — models ship in the bundle and become selectable once the env key is set, with the deployment base URL resolved at runtime from `AZURE_OPENAI_BASE_URL`/`AZURE_OPENAI_RESOURCE_NAME`.
- Added models.dev-backed bundled catalogs for providers that previously shipped no offline models: Hugging Face, Kilo, Moonshot, NanoGPT, Synthetic, Venice, Ollama Cloud, and the Xiaomi Token Plan regions (ams/cn/sgp). They still discover live when credentialed; the bundle is now a non-empty baseline.
### Changed
- Updated stale provider default models to their latest bundled versions: OpenAI-family providers (`azure`, `github-copilot`, `aimlapi`) → GPT-5.5; Gemini providers (`google`, `google-gemini-cli`, `google-vertex`) → `gemini-3.1-pro-preview`; GLM providers (`zai`, `zhipu-coding-plan`) → `glm-5.2`, `cerebras` → `zai-glm-4.7`; Kimi providers (`fireworks`, `opencode-go`, `moonshot`) → `kimi-k2.7-code`, `kimi-code` → `kimi-for-coding`, `together` → `moonshotai/Kimi-K2.7-Code`; `alibaba-coding-plan` → `qwen3.7-plus`; and Claude-Sonnet defaults (`cloudflare-ai-gateway`, `cursor`, `gitlab-duo`, `kilo`, `opencode-zen`, `vercel-ai-gateway`) → Claude Opus 4.x.
- Restricted models.dev Azure discovery to OpenAI-family IDs (`gpt-`, `o1`, `o3`, `o4`, `codex`, `chatgpt`), excluding Foundry-hosted third parties (Claude/DeepSeek/Llama/Mistral/Phi) that Azure serves through non-Responses APIs.
- Detected the Azure OpenAI Responses compat surface (developer role, strict tool mode, strict tool-result pairing) by provider id as well as base URL, so bundled `azure` models whose deployment host is only known at runtime still get the right wire behavior.
- Renamed the `Qwen3-ASR-Flash` model label to `Qwen3 ASR Flash`
### Fixed
- Fixed `zhipu-coding-plan` and `together` shipping no bundled models: their descriptors referenced non-existent models.dev keys (`zhipu-coding-plan`, `together`); pointed them at the real keys (`zhipuai-coding-plan`, `togetherai`) so they bundle their GLM and full catalogs respectively.
- Folded the `azure-openai-responses` API into the OpenAI Responses thinking-inference branches so Azure reasoning models (o-series, GPT-5, Codex) resolve the discrete effort vocabulary (including `xhigh`) and effort-control mode instead of falling through to generic defaults.
- Fixed `ollama-cloud` discovery inheriting an unsafe cross-provider `contextWindow`/`maxTokens` when `/api/show` returns no size metadata; it now falls back to the safe 128K context / 8K output caps.
- Dropped internal Fireworks control-plane resource ids (`accounts/fireworks/{models,routers}/…`) from the bundle; only the public request ids ship.
## [15.13.2] - 2026-06-15
@@ -290,6 +290,23 @@ function dropUnusableZaiContextTierIds(models: readonly ModelSpec[]): ModelSpec[
return models.filter(model => !(model.provider === "zai" && model.id.endsWith("[1m]")));
}
/**
* Fireworks discovery and prior snapshots can surface internal control-plane
* resource ids (`accounts/fireworks/{models,routers}/...`) alongside the public
* request ids (`kimi-k2.7-code`, `deepseek-v4-flash`, ...). The wire ids are an
* implementation detail the request path reconstructs from the public id, so
* drop them from the bundle outright.
*/
function dropFireworksWireIds(models: readonly ModelSpec[]): ModelSpec[] {
return models.filter(
model =>
!(
(model.provider === "fireworks" || model.provider === "firepass") &&
model.id.startsWith("accounts/fireworks/")
),
);
}
const ANTIGRAVITY_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googleapis.com";
async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise<OAuthAccess | null> {
@@ -477,6 +494,7 @@ async function generateModels() {
allModels = applyCodexPricingFallback(allModels);
allModels = applyFireworksKimiMaxTokensCap(allModels);
allModels = applyFireworksDeepSeekReasoningShape(allModels);
allModels = dropFireworksWireIds(allModels);
allModels = dropUnusableZaiContextTierIds(allModels);
// Normalize display names: gateway author prefixes ("OpenAI: …"), alias
// markers ("(latest)"), provider attribution ("(Antigravity)"), and
File diff suppressed because it is too large Load Diff
@@ -61,7 +61,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "alibaba-coding-plan",
defaultModel: "qwen3.5-plus",
defaultModel: "qwen3.7-plus",
envVars: ["ALIBABA_CODING_PLAN_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => alibabaCodingPlanModelManagerOptions(config),
catalogDiscovery: { label: "Alibaba Coding Plan" },
@@ -82,21 +82,21 @@ export const CATALOG_PROVIDERS = [
},
{
id: "cerebras",
defaultModel: "zai-glm-4.6",
defaultModel: "zai-glm-4.7",
envVars: ["CEREBRAS_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => cerebrasModelManagerOptions(config),
catalogDiscovery: { label: "Cerebras" },
},
{
id: "cloudflare-ai-gateway",
defaultModel: "claude-sonnet-4-5",
defaultModel: "anthropic/claude-opus-4-8",
envVars: ["CLOUDFLARE_AI_GATEWAY_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => cloudflareAiGatewayModelManagerOptions(config),
catalogDiscovery: { label: "Cloudflare AI Gateway" },
},
{
id: "cursor",
defaultModel: "claude-sonnet-4-6",
defaultModel: "claude-4.6-opus-high",
envVars: ["CURSOR_ACCESS_TOKEN"],
createModelManagerOptions: (config: ModelManagerConfig) => cursorModelManagerOptions(config),
catalogDiscovery: { label: "Cursor", envVars: ["CURSOR_API_KEY"], oauthProvider: "cursor" },
@@ -116,7 +116,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "fireworks",
defaultModel: "kimi-k2.6",
defaultModel: "kimi-k2.7-code",
envVars: ["FIREWORKS_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => fireworksModelManagerOptions(config),
catalogDiscovery: { label: "Fireworks" },
@@ -129,12 +129,12 @@ export const CATALOG_PROVIDERS = [
},
{
id: "gitlab-duo",
defaultModel: "duo-chat-sonnet-4-5",
defaultModel: "duo-chat-opus-4-6",
envVars: ["GITLAB_TOKEN"],
},
{
id: "google",
defaultModel: "gemini-2.5-pro",
defaultModel: "gemini-3.1-pro-preview",
envVars: ["GEMINI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => googleModelManagerOptions(config),
},
@@ -145,12 +145,12 @@ export const CATALOG_PROVIDERS = [
},
{
id: "google-gemini-cli",
defaultModel: "gemini-2.5-pro",
defaultModel: "gemini-3.1-pro-preview",
specialModelManager: true,
},
{
id: "google-vertex",
defaultModel: "gemini-3-pro-preview",
defaultModel: "gemini-3.1-pro-preview",
createModelManagerOptions: (config: ModelManagerConfig) => googleVertexModelManagerOptions(config),
allowUnauthenticated: true,
},
@@ -169,14 +169,14 @@ export const CATALOG_PROVIDERS = [
},
{
id: "kilo",
defaultModel: "anthropic/claude-sonnet-4.5",
defaultModel: "anthropic/claude-opus-4.8",
envVars: ["KILO_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => kiloModelManagerOptions(config),
catalogDiscovery: { label: "Kilo Gateway", allowUnauthenticated: true },
},
{
id: "kimi-code",
defaultModel: "kimi-k2.5",
defaultModel: "kimi-for-coding",
createModelManagerOptions: (config: ModelManagerConfig) => kimiCodeModelManagerOptions(config),
catalogDiscovery: { label: "Kimi Code", envVars: ["KIMI_API_KEY"] },
},
@@ -217,7 +217,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "moonshot",
defaultModel: "kimi-k2.5",
defaultModel: "kimi-k2.7-code",
envVars: ["MOONSHOT_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => moonshotModelManagerOptions(config),
catalogDiscovery: { label: "Moonshot" },
@@ -264,13 +264,13 @@ export const CATALOG_PROVIDERS = [
},
{
id: "opencode-go",
defaultModel: "kimi-k2.5",
defaultModel: "kimi-k2.7-code",
envVars: ["OPENCODE_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => opencodeGoModelManagerOptions(config),
},
{
id: "opencode-zen",
defaultModel: "claude-sonnet-4-6",
defaultModel: "claude-opus-4-8",
envVars: ["OPENCODE_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => opencodeZenModelManagerOptions(config),
},
@@ -308,7 +308,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "together",
defaultModel: "moonshotai/Kimi-K2.5",
defaultModel: "moonshotai/Kimi-K2.7-Code",
envVars: ["TOGETHER_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => togetherModelManagerOptions(config),
catalogDiscovery: { label: "Together" },
@@ -322,7 +322,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "vercel-ai-gateway",
defaultModel: "anthropic/claude-sonnet-4-6",
defaultModel: "anthropic/claude-opus-4.8",
envVars: ["AI_GATEWAY_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => vercelAiGatewayModelManagerOptions(config),
catalogDiscovery: {
@@ -401,7 +401,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "zai",
defaultModel: "glm-5.1",
defaultModel: "glm-5.2",
envVars: ["ZAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => zaiModelManagerOptions(config),
catalogDiscovery: { label: "zAI" },
@@ -415,7 +415,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "zhipu-coding-plan",
defaultModel: "glm-5.1",
defaultModel: "glm-5.2",
envVars: ["ZHIPU_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => zhipuCodingPlanModelManagerOptions(config),
catalogDiscovery: { label: "Zhipu Coding Plan" },
+10 -3
View File
@@ -121,8 +121,12 @@ export function ollamaCloudModelManagerOptions(
metadata = undefined;
}
const capabilities = metadata?.capabilities;
const contextWindow =
getContextWindow(metadata?.model_info) ?? providerReference?.contextWindow ?? 128000;
const discoveredContextWindow = getContextWindow(metadata?.model_info);
// `/api/show` is the only trustworthy Ollama-owned source for size caps.
// When it is unavailable (or returns only coarse capabilities), do NOT
// inherit giant budgets from bundled fallback metadata sourced from a
// different catalog; keep the historical safe fallback instead.
const contextWindow = discoveredContextWindow ?? 128000;
const reasoning = capabilities ? capabilities.includes("thinking") : (reference?.reasoning ?? false);
const thinking = capabilities ? getThinkingConfig(capabilities) : reference?.thinking;
const input = capabilities
@@ -142,7 +146,10 @@ export function ollamaCloudModelManagerOptions(
input,
cost: reference?.cost ?? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow,
maxTokens: providerReference?.maxTokens ?? Math.min(contextWindow, 8192),
maxTokens:
discoveredContextWindow !== null && discoveredContextWindow !== undefined
? (providerReference?.maxTokens ?? Math.min(contextWindow, 8192))
: Math.min(contextWindow, 8192),
};
}),
);
@@ -3207,7 +3207,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CORE: readonly ModelsDevProviderDescriptor
// --- Cerebras ---
openAiCompletionsDescriptor("cerebras", "cerebras", "https://api.cerebras.ai/v1"),
// --- Together ---
openAiCompletionsDescriptor("together", "together", "https://api.together.xyz/v1"),
openAiCompletionsDescriptor("togetherai", "together", "https://api.together.xyz/v1"),
// --- NVIDIA ---
openAiCompletionsDescriptor("nvidia", "nvidia", "https://integrate.api.nvidia.com/v1", {
defaultContextWindow: 131072,
@@ -3287,7 +3287,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_CODING_PLANS: readonly ModelsDevProviderDe
),
// --- Zhipu Coding Plan ---
openAiCompletionsDescriptor(
"zhipu-coding-plan",
"zhipuai-coding-plan",
"zhipu-coding-plan",
"https://open.bigmodel.cn/api/coding/paas/v4",
{
@@ -3382,6 +3382,36 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDes
// --- MiniMax (Anthropic) ---
anthropicMessagesDescriptor("minimax", "minimax", "https://api.minimax.io/anthropic"),
anthropicMessagesDescriptor("minimax-cn", "minimax-cn", "https://api.minimaxi.com/anthropic"),
// --- Hugging Face ---
openAiCompletionsDescriptor("huggingface", "huggingface", "https://router.huggingface.co/v1"),
// --- Kilo Gateway ---
openAiCompletionsDescriptor("kilo", "kilo", "https://api.kilo.ai/api/gateway"),
// --- Moonshot AI ---
openAiCompletionsDescriptor("moonshotai", "moonshot", "https://api.moonshot.ai/v1"),
// --- NanoGPT ---
openAiCompletionsDescriptor("nano-gpt", "nanogpt", "https://nano-gpt.com/api/v1"),
// --- Synthetic ---
openAiCompletionsDescriptor("synthetic", "synthetic", "https://api.synthetic.new/openai/v1"),
// --- Venice AI ---
openAiCompletionsDescriptor("venice", "venice", "https://api.venice.ai/api/v1"),
// --- Ollama Cloud ---
simpleModelsDevDescriptor("ollama-cloud", "ollama-cloud", "ollama-chat", "https://ollama.com"),
// --- Xiaomi Token Plan ---
openAiCompletionsDescriptor(
"xiaomi-token-plan-ams",
"xiaomi-token-plan-ams",
"https://token-plan-ams.xiaomimimo.com/v1",
),
openAiCompletionsDescriptor(
"xiaomi-token-plan-cn",
"xiaomi-token-plan-cn",
"https://token-plan-cn.xiaomimimo.com/v1",
),
openAiCompletionsDescriptor(
"xiaomi-token-plan-sgp",
"xiaomi-token-plan-sgp",
"https://token-plan-sgp.xiaomimimo.com/v1",
),
// --- Qwen Portal ---
openAiCompletionsDescriptor("qwen-portal", "qwen-portal", "https://portal.qwen.ai/v1", {
defaultContextWindow: 128000,
+1 -1
View File
@@ -34,7 +34,7 @@ describe("azure catalog provider", () => {
// Mirrors Bedrock: bundled models + env auth, no model-manager factory, so
// it must NOT appear in the runtime discovery descriptor list.
expect(PROVIDER_DESCRIPTORS.some(d => d.providerId === "azure")).toBe(false);
expect(DEFAULT_MODEL_PER_PROVIDER.azure).toBe("gpt-4o");
expect(DEFAULT_MODEL_PER_PROVIDER.azure).toBe("gpt-5.5");
});
test("models.dev descriptor keeps only OpenAI-family Responses models, baseUrl resolved at runtime", () => {
@@ -11,10 +11,10 @@ describe("AIML API built-in provider (issue #2105)", () => {
const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "aimlapi");
expect(descriptor).toBeDefined();
expect(descriptor?.defaultModel).toBe("gpt-4o");
expect(descriptor?.defaultModel).toBe("gpt-5.5-2026-04-23");
expect(descriptor?.catalogDiscovery?.label).toBe("AIML API");
expect(descriptor?.catalogDiscovery?.envVars).toContain("AIMLAPI_API_KEY");
expect(DEFAULT_MODEL_PER_PROVIDER.aimlapi).toBe("gpt-4o");
expect(DEFAULT_MODEL_PER_PROVIDER.aimlapi).toBe("gpt-5.5-2026-04-23");
});
test("uses the OpenAI-compatible completions transport and AIML API base URL", async () => {
@@ -166,11 +166,12 @@ describe("ollama-cloud provider support", () => {
const model = models?.find(candidate => candidate.id === "gpt-oss:120b");
expect(model).toBeDefined();
expect(model?.name).toBe("GPT OSS (120B)");
expect(model?.id).toBe("gpt-oss:120b");
expect(model?.api).toBe("ollama-chat");
expect(model?.provider).toBe("ollama-cloud");
expect(model?.reasoning).toBe(true);
expect(model?.input).toEqual(["text", "image"]);
expect(model?.contextWindow).toBe(131072);
expect(model?.maxTokens).toBe(16384);
expect(model?.contextWindow).toBeGreaterThan(0);
expect(model?.maxTokens).toBeGreaterThan(0);
});
test("streams native chat responses with thinking, text, and usage mapping", async () => {
@@ -304,7 +304,7 @@ describe("apply_patch streaming preview (trailing partial line)", () => {
describe("matcherDigest", () => {
test("hashline: digests stripped `+` body rows only, never headers or op lines", () => {
const input = ["[a.ts#AB12]", "SWAP 1..2:", "+const x = 1;", "+const y = 2;", "DEL 5", ""].join("\n");
const input = ["[a.ts#AB12]", "SWAP 1.=2:", "+const x = 1;", "+const y = 2;", "DEL 5", ""].join("\n");
expect(EDIT_MODE_STRATEGIES.hashline.matcherDigest({ input })).toBe("const x = 1;\nconst y = 2;");
});