feat(catalog): added Azure OpenAI support to registry and catalog compatibility

- Added `azure` provider registration in the AI registry with API key env mapping.
- Added Azure provider descriptors with default model `gpt-4o` and catalog discovery metadata.
- Enabled Azure-specific OpenAI compatibility for developer roles and strict responses pairing.
- Added Azure models namespace using OpenAI-family IDs with `models.dev` filtering and responses transport.
This commit is contained in:
can1357
2026-06-15 09:45:53 +02:00
parent e9ea8a87d2
commit e680bc0ca3
10 changed files with 1085 additions and 19 deletions
+4 -1
View File
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
### Added
- Added the Azure OpenAI provider definition (`azure`) to the registry; `AZURE_OPENAI_API_KEY` resolves as its env-var API key via the catalog provider table.
## [15.13.2] - 2026-06-15
@@ -3687,4 +3690,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_
## [0.9.4] - 2025-11-26
Initial release with multi-provider LLM support.
Initial release with multi-provider LLM support.
+6
View File
@@ -0,0 +1,6 @@
import type { ProviderDefinition } from "./types";
export const azureProvider = {
id: "azure",
name: "Azure OpenAI",
} as const satisfies ProviderDefinition;
+2
View File
@@ -3,6 +3,7 @@ import { aimlApiProvider } from "./aimlapi";
import { alibabaCodingPlanProvider } from "./alibaba-coding-plan";
import { amazonBedrockProvider } from "./amazon-bedrock";
import { anthropicProvider } from "./anthropic";
import { azureProvider } from "./azure";
import { cerebrasProvider } from "./cerebras";
import { cloudflareAiGatewayProvider } from "./cloudflare-ai-gateway";
import { cursorProvider } from "./cursor";
@@ -68,6 +69,7 @@ import { zhipuCodingPlanProvider } from "./zhipu-coding-plan";
* list for the loginable providers; non-login model providers are appended.
*/
const ALL = [
azureProvider,
openaiCodexProvider,
anthropicProvider,
zaiProvider,
+14 -1
View File
@@ -1,6 +1,19 @@
# Changelog
## [Unreleased]
### Added
- Added Azure OpenAI as a catalog provider (`azure`, default model `gpt-4o`, env var `AZURE_OPENAI_API_KEY`), bundling the OpenAI-family models Azure serves over the Responses API (GPT-4/4.1/4o, GPT-5 family, o-series, Codex). Like Amazon Bedrock it is catalog-only — models ship in the bundle and become selectable once the env key is set, with the deployment base URL resolved at runtime from `AZURE_OPENAI_BASE_URL`/`AZURE_OPENAI_RESOURCE_NAME`.
### Changed
- Restricted models.dev Azure discovery to OpenAI-family IDs (`gpt-`, `o1`, `o3`, `o4`, `codex`, `chatgpt`), excluding Foundry-hosted third parties (Claude/DeepSeek/Llama/Mistral/Phi) that Azure serves through non-Responses APIs.
- Detected the Azure OpenAI Responses compat surface (developer role, strict tool mode, strict tool-result pairing) by provider id as well as base URL, so bundled `azure` models whose deployment host is only known at runtime still get the right wire behavior.
- Renamed the `Qwen3-ASR-Flash` model label to `Qwen3 ASR Flash`
### Fixed
- Folded the `azure-openai-responses` API into the OpenAI Responses thinking-inference branches so Azure reasoning models (o-series, GPT-5, Codex) resolve the discrete effort vocabulary (including `xhigh`) and effort-control mode instead of falling through to generic defaults.
## [15.13.2] - 2026-06-15
@@ -186,4 +199,4 @@
### Removed
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
+11 -12
View File
@@ -301,29 +301,28 @@ interface OpenAIResponsesSpecLike {
* Build the resolved Responses-API compat record. The Responses flavor
* deliberately differs from chat-completions: GitHub Copilot's responses
* endpoint accepts the `developer` role, while strict tool mode is scoped to
* first-party OpenAI/Azure/Copilot providers. Developer-role and prompt-cache
* detection are URL-only on purpose — the historical call sites never
* consulted the provider id for them. The GPT-5 juice-zero hack keys on the
* model name, matching the historical request-time check.
* first-party OpenAI/Azure/Copilot providers. Azure is detected by provider id
* as well as URL — bundled `azure` models carry no baseUrl (the deployment host
* is per-resource, resolved at runtime) — while OpenAI/Copilot developer-role
* and prompt-cache detection stay URL-keyed, as the historical call sites were.
* The GPT-5 juice-zero hack keys on the model name, matching the historical
* request-time check.
*/
export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): ResolvedOpenAIResponsesCompat {
const baseUrl = spec.baseUrl ?? "";
const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI");
const compat: ResolvedOpenAIResponsesCompat = {
supportsDeveloperRole:
hostMatchesUrl(baseUrl, "openai") ||
hostMatchesUrl(baseUrl, "azureOpenAI") ||
hostMatchesUrl(baseUrl, "githubCopilot"),
supportsDeveloperRole: isAzure || hostMatchesUrl(baseUrl, "openai") || hostMatchesUrl(baseUrl, "githubCopilot"),
supportsStrictMode:
spec.provider === "openai" ||
spec.provider === "azure" ||
isAzure ||
spec.provider === "github-copilot" ||
hostMatchesUrl(baseUrl, "openai") ||
hostMatchesUrl(baseUrl, "azureOpenAI"),
hostMatchesUrl(baseUrl, "openai"),
supportsReasoningEffort: true,
supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"),
// Azure OpenAI and GitHub Copilot Responses paths require tool results
// to strictly match prior tool calls when building Responses inputs.
strictResponsesPairing: hostMatchesUrl(baseUrl, "azureOpenAI") || spec.provider === "github-copilot",
strictResponsesPairing: isAzure || spec.provider === "github-copilot",
requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"),
reasoningEffortMap: {},
};
+6 -2
View File
@@ -219,7 +219,7 @@ export function deriveThinking<TApi extends Api>(spec: ModelSpec<TApi>, compat:
* through other request fields.
*/
function omitsWireReasoningEffort(api: Api, compat: CompatOf<Api>): boolean {
if (api !== "openai-responses" && api !== "openai-codex-responses") {
if (api !== "openai-responses" && api !== "openai-codex-responses" && api !== "azure-openai-responses") {
return false;
}
return (compat as ResolvedOpenAIResponsesCompat | undefined)?.supportsReasoningEffort === false;
@@ -426,7 +426,11 @@ function inferFallbackEfforts<TApi extends Api>(spec: ModelSpec<TApi>, compat: C
return DEFAULT_REASONING_EFFORTS;
}
// OpenAI Responses APIs encode discrete effort levels, including xhigh.
if (spec.api === "openai-responses" || spec.api === "openai-codex-responses") {
if (
spec.api === "openai-responses" ||
spec.api === "openai-codex-responses" ||
spec.api === "azure-openai-responses"
) {
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
}
return DEFAULT_REASONING_EFFORTS;
+942 -1
View File
@@ -11280,6 +11280,947 @@
}
}
},
"azure": {
"codex-mini": {
"id": "codex-mini",
"name": "Codex Mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 1.5,
"output": 6,
"cacheRead": 0.375,
"cacheWrite": 0
},
"contextWindow": 200000,
"maxTokens": 100000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-4": {
"id": "gpt-4",
"name": "GPT-4",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 60,
"output": 120,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 8192,
"maxTokens": 8192
},
"gpt-4-32k": {
"id": "gpt-4-32k",
"name": "GPT-4 32K",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 60,
"output": 120,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 32768,
"maxTokens": 32768
},
"gpt-4-turbo": {
"id": "gpt-4-turbo",
"name": "GPT-4 Turbo",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 10,
"output": 30,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 4096
},
"gpt-4-turbo-vision": {
"id": "gpt-4-turbo-vision",
"name": "GPT-4 Turbo Vision",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 10,
"output": 30,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 4096
},
"gpt-4.1": {
"id": "gpt-4.1",
"name": "GPT-4.1",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 2,
"output": 8,
"cacheRead": 0.5,
"cacheWrite": 0
},
"contextWindow": 1047576,
"maxTokens": 32768
},
"gpt-4.1-mini": {
"id": "gpt-4.1-mini",
"name": "GPT-4.1 mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 0.4,
"output": 1.6,
"cacheRead": 0.1,
"cacheWrite": 0
},
"contextWindow": 1047576,
"maxTokens": 32768
},
"gpt-4.1-nano": {
"id": "gpt-4.1-nano",
"name": "GPT-4.1 nano",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 0.1,
"output": 0.4,
"cacheRead": 0.025,
"cacheWrite": 0
},
"contextWindow": 1047576,
"maxTokens": 32768
},
"gpt-4o": {
"id": "gpt-4o",
"name": "GPT-4o",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 2.5,
"output": 10,
"cacheRead": 1.25,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 16384
},
"gpt-4o-mini": {
"id": "gpt-4o-mini",
"name": "GPT-4o mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": false,
"input": [
"text",
"image"
],
"cost": {
"input": 0.15,
"output": 0.6,
"cacheRead": 0.075,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 16384
},
"gpt-5": {
"id": "gpt-5",
"name": "GPT-5",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.25,
"output": 10,
"cacheRead": 0.13,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5-codex": {
"id": "gpt-5-codex",
"name": "GPT-5-Codex",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.25,
"output": 10,
"cacheRead": 0.13,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5-mini": {
"id": "gpt-5-mini",
"name": "GPT-5 Mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.25,
"output": 2,
"cacheRead": 0.03,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5-nano": {
"id": "gpt-5-nano",
"name": "GPT-5 Nano",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.05,
"output": 0.4,
"cacheRead": 0.01,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5-pro": {
"id": "gpt-5-pro",
"name": "GPT-5 Pro",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 15,
"output": 120,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 400000,
"maxTokens": 272000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5.1": {
"id": "gpt-5.1",
"name": "GPT-5.1",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.25,
"output": 10,
"cacheRead": 0.125,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5.1-chat": {
"id": "gpt-5.1-chat",
"name": "GPT-5.1 Chat",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.25,
"output": 10,
"cacheRead": 0.125,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5.1-codex": {
"id": "gpt-5.1-codex",
"name": "GPT-5.1 Codex",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.25,
"output": 10,
"cacheRead": 0.125,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5.1-codex-max": {
"id": "gpt-5.1-codex-max",
"name": "GPT-5.1 Codex Max",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.25,
"output": 10,
"cacheRead": 0.125,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"gpt-5.1-codex-mini": {
"id": "gpt-5.1-codex-mini",
"name": "GPT-5.1 Codex Mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.25,
"output": 2,
"cacheRead": 0.025,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"medium",
"high"
]
}
},
"gpt-5.2": {
"id": "gpt-5.2",
"name": "GPT-5.2",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.75,
"output": 14,
"cacheRead": 0.125,
"cacheWrite": 0
},
"contextWindow": 400000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.2-chat": {
"id": "gpt-5.2-chat",
"name": "GPT-5.2 Chat",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.75,
"output": 14,
"cacheRead": 0.175,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.2-codex": {
"id": "gpt-5.2-codex",
"name": "GPT-5.2 Codex",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.75,
"output": 14,
"cacheRead": 0.175,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.3-chat": {
"id": "gpt-5.3-chat",
"name": "GPT-5.3 Chat",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.75,
"output": 14,
"cacheRead": 0.175,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.3-codex": {
"id": "gpt-5.3-codex",
"name": "GPT-5.3 Codex",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.75,
"output": 14,
"cacheRead": 0.175,
"cacheWrite": 0
},
"contextWindow": 272000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.4": {
"id": "gpt-5.4",
"name": "GPT-5.4",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 2.5,
"output": 15,
"cacheRead": 0.25,
"cacheWrite": 0
},
"contextWindow": 1050000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.4-mini": {
"id": "gpt-5.4-mini",
"name": "GPT-5.4 Mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.75,
"output": 4.5,
"cacheRead": 0.075,
"cacheWrite": 0
},
"contextWindow": 400000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.4-nano": {
"id": "gpt-5.4-nano",
"name": "GPT-5.4 Nano",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.2,
"output": 1.25,
"cacheRead": 0.02,
"cacheWrite": 0
},
"contextWindow": 400000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.4-pro": {
"id": "gpt-5.4-pro",
"name": "GPT-5.4 Pro",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 30,
"output": 180,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1050000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
}
},
"gpt-5.5": {
"id": "gpt-5.5",
"name": "GPT-5.5",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 5,
"output": 30,
"cacheRead": 0.5,
"cacheWrite": 0
},
"contextWindow": 1050000,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high",
"xhigh"
]
},
"contextPromotionTarget": "azure/gpt-5.4"
},
"o1": {
"id": "o1",
"name": "o1",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 15,
"output": 60,
"cacheRead": 7.5,
"cacheWrite": 0
},
"contextWindow": 200000,
"maxTokens": 100000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"requiresEffort": true
}
},
"o1-mini": {
"id": "o1-mini",
"name": "o1-mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 1.1,
"output": 4.4,
"cacheRead": 0.55,
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"requiresEffort": true
}
},
"o3": {
"id": "o3",
"name": "o3",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 2,
"output": 8,
"cacheRead": 0.5,
"cacheWrite": 0
},
"contextWindow": 200000,
"maxTokens": 100000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"requiresEffort": true
}
},
"o3-mini": {
"id": "o3-mini",
"name": "o3-mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 1.1,
"output": 4.4,
"cacheRead": 0.55,
"cacheWrite": 0
},
"contextWindow": 200000,
"maxTokens": 100000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"requiresEffort": true
}
},
"o4-mini": {
"id": "o4-mini",
"name": "o4-mini",
"api": "azure-openai-responses",
"provider": "azure",
"baseUrl": "",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.1,
"output": 4.4,
"cacheRead": 0.275,
"cacheWrite": 0
},
"contextWindow": 200000,
"maxTokens": 100000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"requiresEffort": true
}
}
},
"cerebras": {
"gpt-oss-120b": {
"id": "gpt-oss-120b",
@@ -75596,7 +76537,7 @@
},
"qwen/qwen3-asr-flash": {
"id": "qwen/qwen3-asr-flash",
"name": "Qwen3-ASR-Flash",
"name": "Qwen3 ASR Flash",
"api": "openai-completions",
"provider": "zenmux",
"baseUrl": "https://zenmux.ai/api/v1",
@@ -53,7 +53,7 @@ import { cursorModelManagerOptions, zaiModelManagerOptions } from "./special";
export const CATALOG_PROVIDERS = [
{
id: "aimlapi",
defaultModel: "gpt-4o",
defaultModel: "gpt-5.5-2026-04-23",
envVars: ["AIMLAPI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => aimlApiModelManagerOptions(config),
dynamicModelsAuthoritative: true,
@@ -75,6 +75,11 @@ export const CATALOG_PROVIDERS = [
defaultModel: "claude-opus-4-8",
createModelManagerOptions: (config: ModelManagerConfig) => anthropicModelManagerOptions(config),
},
{
id: "azure",
defaultModel: "gpt-5.5",
envVars: ["AZURE_OPENAI_API_KEY"],
},
{
id: "cerebras",
defaultModel: "zai-glm-4.6",
@@ -118,7 +123,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "github-copilot",
defaultModel: "gpt-4o",
defaultModel: "gpt-5.5",
envVars: ["COPILOT_GITHUB_TOKEN"],
createModelManagerOptions: (config: ModelManagerConfig) => githubCopilotModelManagerOptions(config),
},
@@ -3314,6 +3314,18 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_GOOGLE_VERTEX: readonly ModelsDevProviderD
];
const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDescriptor[] = [
// --- Azure OpenAI ---
// OpenAI-family models hosted on Azure, served via the Responses API. baseUrl
// is empty: the deployment host is per-resource and resolved at runtime from
// AZURE_OPENAI_BASE_URL / AZURE_OPENAI_RESOURCE_NAME (see resolveAzureConfig).
simpleModelsDevDescriptor("azure", "azure", "azure-openai-responses", "", {
filterModel: (modelId, m) => {
if (m.tool_call !== true) return false;
// OpenAI-family only (not Foundry/DeepSeek/Claude/Llama/Mistral/Phi, which
// Azure serves via non-Responses APIs under a per-model provider override).
return /^(gpt-|o1|o3|o4|codex|chatgpt)/.test(modelId);
},
}),
// --- Cloudflare AI Gateway ---
anthropicMessagesDescriptor(
"cloudflare-ai-gateway",
@@ -0,0 +1,81 @@
import { describe, expect, test } from "bun:test";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import {
DEFAULT_MODEL_PER_PROVIDER,
MODELS_DEV_PROVIDER_DESCRIPTORS,
mapModelsDevToModels,
PROVIDER_DESCRIPTORS,
} from "@oh-my-pi/pi-catalog/provider-models";
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
// A models.dev "azure" payload: two OpenAI-family models (one reasoning), a
// non-tool-capable instruct model, and a Foundry-hosted third party served via
// a per-model `provider` override (claude over .services.ai.azure.com).
const AZURE_MODELS_DEV_FIXTURE = {
azure: {
models: {
"gpt-4o": { name: "GPT-4o", tool_call: true, limit: { context: 128000, output: 16384 } },
o3: { name: "o3", tool_call: true, reasoning: true, limit: { context: 200000, output: 100000 } },
"gpt-3.5-turbo-instruct": { name: "GPT-3.5 Turbo Instruct", tool_call: false },
"claude-opus-4-5": {
name: "Claude Opus 4.5",
tool_call: true,
reasoning: true,
provider: { npm: "@ai-sdk/anthropic", api: "https://x.services.ai.azure.com/anthropic/v1" },
},
},
},
};
describe("azure catalog provider", () => {
test("is catalog-only (no runtime discovery) with an env-var-backed default model", () => {
// Mirrors Bedrock: bundled models + env auth, no model-manager factory, so
// it must NOT appear in the runtime discovery descriptor list.
expect(PROVIDER_DESCRIPTORS.some(d => d.providerId === "azure")).toBe(false);
expect(DEFAULT_MODEL_PER_PROVIDER.azure).toBe("gpt-4o");
});
test("models.dev descriptor keeps only OpenAI-family Responses models, baseUrl resolved at runtime", () => {
const azure = mapModelsDevToModels(AZURE_MODELS_DEV_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter(
model => model.provider === "azure",
);
const ids = azure.map(model => model.id).sort();
// gpt-4o + o3 survive; the instruct model (no tool_call) and the Foundry
// Claude (non-Responses, per-model provider override) are dropped.
expect(ids).toEqual(["gpt-4o", "o3"]);
for (const model of azure) {
expect(model.api).toBe("azure-openai-responses");
// Empty baseUrl: the deployment host is per-resource, resolved at runtime.
expect(model.baseUrl).toBe("");
}
});
test("bundled-shape spec (provider id, empty baseUrl) resolves the Azure Responses compat flags", () => {
// The deployment host is only known at request time, so detection MUST key
// off the provider id, not the (empty) baseUrl.
const compat = buildOpenAIResponsesCompat({ provider: "azure", name: "GPT-5", baseUrl: "" });
expect(compat.strictResponsesPairing).toBe(true);
expect(compat.supportsStrictMode).toBe(true);
expect(compat.supportsDeveloperRole).toBe(true);
});
test("Azure reasoning models infer the OpenAI Responses effort vocabulary", () => {
const spec: ModelSpec<"azure-openai-responses"> = {
id: "o3",
name: "o3",
api: "azure-openai-responses",
provider: "azure",
baseUrl: "",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200000,
maxTokens: 100000,
};
const model = buildModel(spec);
expect(model.thinking?.mode).toBe("effort");
expect(model.thinking?.efforts).toContain(Effort.XHigh);
});
});