diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 88f374b0b..abc815cc5 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] +### Added + +- Added the Azure OpenAI provider definition (`azure`) to the registry; `AZURE_OPENAI_API_KEY` resolves as its env-var API key via the catalog provider table. ## [15.13.2] - 2026-06-15 @@ -3687,4 +3690,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_ ## [0.9.4] - 2025-11-26 -Initial release with multi-provider LLM support. +Initial release with multi-provider LLM support. \ No newline at end of file diff --git a/packages/ai/src/registry/azure.ts b/packages/ai/src/registry/azure.ts new file mode 100644 index 000000000..759a47bf3 --- /dev/null +++ b/packages/ai/src/registry/azure.ts @@ -0,0 +1,6 @@ +import type { ProviderDefinition } from "./types"; + +export const azureProvider = { + id: "azure", + name: "Azure OpenAI", +} as const satisfies ProviderDefinition; diff --git a/packages/ai/src/registry/registry.ts b/packages/ai/src/registry/registry.ts index e49787b77..288071422 100644 --- a/packages/ai/src/registry/registry.ts +++ b/packages/ai/src/registry/registry.ts @@ -3,6 +3,7 @@ import { aimlApiProvider } from "./aimlapi"; import { alibabaCodingPlanProvider } from "./alibaba-coding-plan"; import { amazonBedrockProvider } from "./amazon-bedrock"; import { anthropicProvider } from "./anthropic"; +import { azureProvider } from "./azure"; import { cerebrasProvider } from "./cerebras"; import { cloudflareAiGatewayProvider } from "./cloudflare-ai-gateway"; import { cursorProvider } from "./cursor"; @@ -68,6 +69,7 @@ import { zhipuCodingPlanProvider } from "./zhipu-coding-plan"; * list for the loginable providers; non-login model providers are appended. */ const ALL = [ + azureProvider, openaiCodexProvider, anthropicProvider, zaiProvider, diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index fb65cf315..c5f56745d 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -1,6 +1,19 @@ # Changelog ## [Unreleased] +### Added + +- Added Azure OpenAI as a catalog provider (`azure`, default model `gpt-4o`, env var `AZURE_OPENAI_API_KEY`), bundling the OpenAI-family models Azure serves over the Responses API (GPT-4/4.1/4o, GPT-5 family, o-series, Codex). Like Amazon Bedrock it is catalog-only — models ship in the bundle and become selectable once the env key is set, with the deployment base URL resolved at runtime from `AZURE_OPENAI_BASE_URL`/`AZURE_OPENAI_RESOURCE_NAME`. + +### Changed + +- Restricted models.dev Azure discovery to OpenAI-family IDs (`gpt-`, `o1`, `o3`, `o4`, `codex`, `chatgpt`), excluding Foundry-hosted third parties (Claude/DeepSeek/Llama/Mistral/Phi) that Azure serves through non-Responses APIs. +- Detected the Azure OpenAI Responses compat surface (developer role, strict tool mode, strict tool-result pairing) by provider id as well as base URL, so bundled `azure` models whose deployment host is only known at runtime still get the right wire behavior. +- Renamed the `Qwen3-ASR-Flash` model label to `Qwen3 ASR Flash` + +### Fixed + +- Folded the `azure-openai-responses` API into the OpenAI Responses thinking-inference branches so Azure reasoning models (o-series, GPT-5, Codex) resolve the discrete effort vocabulary (including `xhigh`) and effort-control mode instead of falling through to generic defaults. ## [15.13.2] - 2026-06-15 @@ -186,4 +199,4 @@ ### Removed -- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads. +- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads. \ No newline at end of file diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 8574d01ae..a7265f448 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -301,29 +301,28 @@ interface OpenAIResponsesSpecLike { * Build the resolved Responses-API compat record. The Responses flavor * deliberately differs from chat-completions: GitHub Copilot's responses * endpoint accepts the `developer` role, while strict tool mode is scoped to - * first-party OpenAI/Azure/Copilot providers. Developer-role and prompt-cache - * detection are URL-only on purpose — the historical call sites never - * consulted the provider id for them. The GPT-5 juice-zero hack keys on the - * model name, matching the historical request-time check. + * first-party OpenAI/Azure/Copilot providers. Azure is detected by provider id + * as well as URL — bundled `azure` models carry no baseUrl (the deployment host + * is per-resource, resolved at runtime) — while OpenAI/Copilot developer-role + * and prompt-cache detection stay URL-keyed, as the historical call sites were. + * The GPT-5 juice-zero hack keys on the model name, matching the historical + * request-time check. */ export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): ResolvedOpenAIResponsesCompat { const baseUrl = spec.baseUrl ?? ""; + const isAzure = modelMatchesHost({ provider: spec.provider, baseUrl }, "azureOpenAI"); const compat: ResolvedOpenAIResponsesCompat = { - supportsDeveloperRole: - hostMatchesUrl(baseUrl, "openai") || - hostMatchesUrl(baseUrl, "azureOpenAI") || - hostMatchesUrl(baseUrl, "githubCopilot"), + supportsDeveloperRole: isAzure || hostMatchesUrl(baseUrl, "openai") || hostMatchesUrl(baseUrl, "githubCopilot"), supportsStrictMode: spec.provider === "openai" || - spec.provider === "azure" || + isAzure || spec.provider === "github-copilot" || - hostMatchesUrl(baseUrl, "openai") || - hostMatchesUrl(baseUrl, "azureOpenAI"), + hostMatchesUrl(baseUrl, "openai"), supportsReasoningEffort: true, supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"), // Azure OpenAI and GitHub Copilot Responses paths require tool results // to strictly match prior tool calls when building Responses inputs. - strictResponsesPairing: hostMatchesUrl(baseUrl, "azureOpenAI") || spec.provider === "github-copilot", + strictResponsesPairing: isAzure || spec.provider === "github-copilot", requiresJuiceZeroHack: spec.name.toLowerCase().startsWith("gpt-5"), reasoningEffortMap: {}, }; diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index e8aa17925..bcaaad94a 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -219,7 +219,7 @@ export function deriveThinking(spec: ModelSpec, compat: * through other request fields. */ function omitsWireReasoningEffort(api: Api, compat: CompatOf): boolean { - if (api !== "openai-responses" && api !== "openai-codex-responses") { + if (api !== "openai-responses" && api !== "openai-codex-responses" && api !== "azure-openai-responses") { return false; } return (compat as ResolvedOpenAIResponsesCompat | undefined)?.supportsReasoningEffort === false; @@ -426,7 +426,11 @@ function inferFallbackEfforts(spec: ModelSpec, compat: C return DEFAULT_REASONING_EFFORTS; } // OpenAI Responses APIs encode discrete effort levels, including xhigh. - if (spec.api === "openai-responses" || spec.api === "openai-codex-responses") { + if ( + spec.api === "openai-responses" || + spec.api === "openai-codex-responses" || + spec.api === "azure-openai-responses" + ) { return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; } return DEFAULT_REASONING_EFFORTS; diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 809d52d50..b5923f4f8 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -11280,6 +11280,947 @@ } } }, + "azure": { + "codex-mini": { + "id": "codex-mini", + "name": "Codex Mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.5, + "output": 6, + "cacheRead": 0.375, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 100000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-4": { + "id": "gpt-4", + "name": "GPT-4", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 60, + "output": 120, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 8192, + "maxTokens": 8192 + }, + "gpt-4-32k": { + "id": "gpt-4-32k", + "name": "GPT-4 32K", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 60, + "output": 120, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 32768, + "maxTokens": 32768 + }, + "gpt-4-turbo": { + "id": "gpt-4-turbo", + "name": "GPT-4 Turbo", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 30, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 4096 + }, + "gpt-4-turbo-vision": { + "id": "gpt-4-turbo-vision", + "name": "GPT-4 Turbo Vision", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 10, + "output": 30, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 4096 + }, + "gpt-4.1": { + "id": "gpt-4.1", + "name": "GPT-4.1", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 8, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1047576, + "maxTokens": 32768 + }, + "gpt-4.1-mini": { + "id": "gpt-4.1-mini", + "name": "GPT-4.1 mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.4, + "output": 1.6, + "cacheRead": 0.1, + "cacheWrite": 0 + }, + "contextWindow": 1047576, + "maxTokens": 32768 + }, + "gpt-4.1-nano": { + "id": "gpt-4.1-nano", + "name": "GPT-4.1 nano", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.1, + "output": 0.4, + "cacheRead": 0.025, + "cacheWrite": 0 + }, + "contextWindow": 1047576, + "maxTokens": 32768 + }, + "gpt-4o": { + "id": "gpt-4o", + "name": "GPT-4o", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 10, + "cacheRead": 1.25, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384 + }, + "gpt-4o-mini": { + "id": "gpt-4o-mini", + "name": "GPT-4o mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.15, + "output": 0.6, + "cacheRead": 0.075, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384 + }, + "gpt-5": { + "id": "gpt-5", + "name": "GPT-5", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 10, + "cacheRead": 0.13, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5-codex": { + "id": "gpt-5-codex", + "name": "GPT-5-Codex", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 10, + "cacheRead": 0.13, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5-mini": { + "id": "gpt-5-mini", + "name": "GPT-5 Mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 2, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5-nano": { + "id": "gpt-5-nano", + "name": "GPT-5 Nano", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.05, + "output": 0.4, + "cacheRead": 0.01, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5-pro": { + "id": "gpt-5-pro", + "name": "GPT-5 Pro", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 15, + "output": 120, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 400000, + "maxTokens": 272000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5.1": { + "id": "gpt-5.1", + "name": "GPT-5.1", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 10, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5.1-chat": { + "id": "gpt-5.1-chat", + "name": "GPT-5.1 Chat", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 10, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5.1-codex": { + "id": "gpt-5.1-codex", + "name": "GPT-5.1 Codex", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 10, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5.1-codex-max": { + "id": "gpt-5.1-codex-max", + "name": "GPT-5.1 Codex Max", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 10, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "gpt-5.1-codex-mini": { + "id": "gpt-5.1-codex-mini", + "name": "GPT-5.1 Codex Mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 2, + "cacheRead": 0.025, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "medium", + "high" + ] + } + }, + "gpt-5.2": { + "id": "gpt-5.2", + "name": "GPT-5.2", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.75, + "output": 14, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 400000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.2-chat": { + "id": "gpt-5.2-chat", + "name": "GPT-5.2 Chat", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.75, + "output": 14, + "cacheRead": 0.175, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.2-codex": { + "id": "gpt-5.2-codex", + "name": "GPT-5.2 Codex", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.75, + "output": 14, + "cacheRead": 0.175, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.3-chat": { + "id": "gpt-5.3-chat", + "name": "GPT-5.3 Chat", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.75, + "output": 14, + "cacheRead": 0.175, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.3-codex": { + "id": "gpt-5.3-codex", + "name": "GPT-5.3 Codex", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.75, + "output": 14, + "cacheRead": 0.175, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.4": { + "id": "gpt-5.4", + "name": "GPT-5.4", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.4-mini": { + "id": "gpt-5.4-mini", + "name": "GPT-5.4 Mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.75, + "output": 4.5, + "cacheRead": 0.075, + "cacheWrite": 0 + }, + "contextWindow": 400000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.4-nano": { + "id": "gpt-5.4-nano", + "name": "GPT-5.4 Nano", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.2, + "output": 1.25, + "cacheRead": 0.02, + "cacheWrite": 0 + }, + "contextWindow": 400000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.4-pro": { + "id": "gpt-5.4-pro", + "name": "GPT-5.4 Pro", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 30, + "output": 180, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.5": { + "id": "gpt-5.5", + "name": "GPT-5.5", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + }, + "contextPromotionTarget": "azure/gpt-5.4" + }, + "o1": { + "id": "o1", + "name": "o1", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 15, + "output": 60, + "cacheRead": 7.5, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 100000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + }, + "o1-mini": { + "id": "o1-mini", + "name": "o1-mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.1, + "output": 4.4, + "cacheRead": 0.55, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + }, + "o3": { + "id": "o3", + "name": "o3", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 8, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 100000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + }, + "o3-mini": { + "id": "o3-mini", + "name": "o3-mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.1, + "output": 4.4, + "cacheRead": 0.55, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 100000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + }, + "o4-mini": { + "id": "o4-mini", + "name": "o4-mini", + "api": "azure-openai-responses", + "provider": "azure", + "baseUrl": "", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.1, + "output": 4.4, + "cacheRead": 0.275, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 100000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "requiresEffort": true + } + } + }, "cerebras": { "gpt-oss-120b": { "id": "gpt-oss-120b", @@ -75596,7 +76537,7 @@ }, "qwen/qwen3-asr-flash": { "id": "qwen/qwen3-asr-flash", - "name": "Qwen3-ASR-Flash", + "name": "Qwen3 ASR Flash", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index 2810795d4..e1362e86d 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -53,7 +53,7 @@ import { cursorModelManagerOptions, zaiModelManagerOptions } from "./special"; export const CATALOG_PROVIDERS = [ { id: "aimlapi", - defaultModel: "gpt-4o", + defaultModel: "gpt-5.5-2026-04-23", envVars: ["AIMLAPI_API_KEY"], createModelManagerOptions: (config: ModelManagerConfig) => aimlApiModelManagerOptions(config), dynamicModelsAuthoritative: true, @@ -75,6 +75,11 @@ export const CATALOG_PROVIDERS = [ defaultModel: "claude-opus-4-8", createModelManagerOptions: (config: ModelManagerConfig) => anthropicModelManagerOptions(config), }, + { + id: "azure", + defaultModel: "gpt-5.5", + envVars: ["AZURE_OPENAI_API_KEY"], + }, { id: "cerebras", defaultModel: "zai-glm-4.6", @@ -118,7 +123,7 @@ export const CATALOG_PROVIDERS = [ }, { id: "github-copilot", - defaultModel: "gpt-4o", + defaultModel: "gpt-5.5", envVars: ["COPILOT_GITHUB_TOKEN"], createModelManagerOptions: (config: ModelManagerConfig) => githubCopilotModelManagerOptions(config), }, diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 7ae1dd7a9..78380aafd 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3314,6 +3314,18 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_GOOGLE_VERTEX: readonly ModelsDevProviderD ]; const MODELS_DEV_PROVIDER_DESCRIPTORS_SPECIALIZED: readonly ModelsDevProviderDescriptor[] = [ + // --- Azure OpenAI --- + // OpenAI-family models hosted on Azure, served via the Responses API. baseUrl + // is empty: the deployment host is per-resource and resolved at runtime from + // AZURE_OPENAI_BASE_URL / AZURE_OPENAI_RESOURCE_NAME (see resolveAzureConfig). + simpleModelsDevDescriptor("azure", "azure", "azure-openai-responses", "", { + filterModel: (modelId, m) => { + if (m.tool_call !== true) return false; + // OpenAI-family only (not Foundry/DeepSeek/Claude/Llama/Mistral/Phi, which + // Azure serves via non-Responses APIs under a per-model provider override). + return /^(gpt-|o1|o3|o4|codex|chatgpt)/.test(modelId); + }, + }), // --- Cloudflare AI Gateway --- anthropicMessagesDescriptor( "cloudflare-ai-gateway", diff --git a/packages/catalog/test/azure-provider.test.ts b/packages/catalog/test/azure-provider.test.ts new file mode 100644 index 000000000..33b850d74 --- /dev/null +++ b/packages/catalog/test/azure-provider.test.ts @@ -0,0 +1,81 @@ +import { describe, expect, test } from "bun:test"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { + DEFAULT_MODEL_PER_PROVIDER, + MODELS_DEV_PROVIDER_DESCRIPTORS, + mapModelsDevToModels, + PROVIDER_DESCRIPTORS, +} from "@oh-my-pi/pi-catalog/provider-models"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; + +// A models.dev "azure" payload: two OpenAI-family models (one reasoning), a +// non-tool-capable instruct model, and a Foundry-hosted third party served via +// a per-model `provider` override (claude over .services.ai.azure.com). +const AZURE_MODELS_DEV_FIXTURE = { + azure: { + models: { + "gpt-4o": { name: "GPT-4o", tool_call: true, limit: { context: 128000, output: 16384 } }, + o3: { name: "o3", tool_call: true, reasoning: true, limit: { context: 200000, output: 100000 } }, + "gpt-3.5-turbo-instruct": { name: "GPT-3.5 Turbo Instruct", tool_call: false }, + "claude-opus-4-5": { + name: "Claude Opus 4.5", + tool_call: true, + reasoning: true, + provider: { npm: "@ai-sdk/anthropic", api: "https://x.services.ai.azure.com/anthropic/v1" }, + }, + }, + }, +}; + +describe("azure catalog provider", () => { + test("is catalog-only (no runtime discovery) with an env-var-backed default model", () => { + // Mirrors Bedrock: bundled models + env auth, no model-manager factory, so + // it must NOT appear in the runtime discovery descriptor list. + expect(PROVIDER_DESCRIPTORS.some(d => d.providerId === "azure")).toBe(false); + expect(DEFAULT_MODEL_PER_PROVIDER.azure).toBe("gpt-4o"); + }); + + test("models.dev descriptor keeps only OpenAI-family Responses models, baseUrl resolved at runtime", () => { + const azure = mapModelsDevToModels(AZURE_MODELS_DEV_FIXTURE, MODELS_DEV_PROVIDER_DESCRIPTORS).filter( + model => model.provider === "azure", + ); + const ids = azure.map(model => model.id).sort(); + // gpt-4o + o3 survive; the instruct model (no tool_call) and the Foundry + // Claude (non-Responses, per-model provider override) are dropped. + expect(ids).toEqual(["gpt-4o", "o3"]); + for (const model of azure) { + expect(model.api).toBe("azure-openai-responses"); + // Empty baseUrl: the deployment host is per-resource, resolved at runtime. + expect(model.baseUrl).toBe(""); + } + }); + + test("bundled-shape spec (provider id, empty baseUrl) resolves the Azure Responses compat flags", () => { + // The deployment host is only known at request time, so detection MUST key + // off the provider id, not the (empty) baseUrl. + const compat = buildOpenAIResponsesCompat({ provider: "azure", name: "GPT-5", baseUrl: "" }); + expect(compat.strictResponsesPairing).toBe(true); + expect(compat.supportsStrictMode).toBe(true); + expect(compat.supportsDeveloperRole).toBe(true); + }); + + test("Azure reasoning models infer the OpenAI Responses effort vocabulary", () => { + const spec: ModelSpec<"azure-openai-responses"> = { + id: "o3", + name: "o3", + api: "azure-openai-responses", + provider: "azure", + baseUrl: "", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200000, + maxTokens: 100000, + }; + const model = buildModel(spec); + expect(model.thinking?.mode).toBe("effort"); + expect(model.thinking?.efforts).toContain(Effort.XHigh); + }); +});