From e14a63da6dd445b276233db597fb3acffc251eb2 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 10 Jun 2026 04:30:19 +0200 Subject: [PATCH] feat(cross-cutting): centralized model compatibility handling with catalog resolvers - Added catalog-level host/model predicates and compat resolvers. - Extended compatibility types and model schema with timeout and replay flags. - Replaced provider-specific heuristics with shared resolver-based checks. - Updated host/identity and resolver tests to validate the new behavior. --- bun.lock | 1 + packages/ai/src/providers/amazon-bedrock.ts | 17 +- packages/ai/src/providers/anthropic.ts | 129 ++---------- .../src/providers/azure-openai-responses.ts | 8 +- .../ai/src/providers/openai-completions.ts | 56 +---- packages/ai/src/providers/openai-responses.ts | 61 ++---- packages/ai/src/providers/vision-guard.ts | 8 +- packages/ai/src/stream.ts | 11 +- .../ai/src/utils/stream-markup-healing.ts | 4 +- ...anthropic-unsigned-thinking-replay.test.ts | 2 +- .../openai-completions-progress-chunk.test.ts | 18 +- .../openai-responses-developer-role.test.ts | 62 +++--- packages/catalog/CHANGELOG.md | 7 +- packages/catalog/src/compat/anthropic.ts | 109 ++++++++++ packages/catalog/src/compat/openai.ts | 199 ++++++++++++------ packages/catalog/src/hosts.ts | 110 ++++++++++ packages/catalog/src/identity/family.ts | 59 ++++++ packages/catalog/src/identity/index.ts | 1 + packages/catalog/src/model-thinking.ts | 3 +- .../src/provider-models/openai-compat.ts | 12 +- packages/catalog/src/types.ts | 23 ++ packages/catalog/test/hosts.test.ts | 75 +++++++ packages/catalog/test/identity-family.test.ts | 44 ++++ packages/coding-agent/CHANGELOG.md | 6 +- .../src/config/append-only-context-mode.ts | 13 +- .../coding-agent/src/config/model-registry.ts | 3 +- .../coding-agent/src/config/model-resolver.ts | 5 +- .../src/config/models-config-schema.ts | 5 + packages/coding-agent/test/usage-cli.test.ts | 68 +++--- packages/mnemopi/CHANGELOG.md | 4 + packages/mnemopi/package.json | 1 + packages/mnemopi/src/config.ts | 5 +- packages/mnemopi/src/core/embeddings.ts | 7 +- 33 files changed, 740 insertions(+), 396 deletions(-) create mode 100644 packages/catalog/src/compat/anthropic.ts create mode 100644 packages/catalog/src/hosts.ts create mode 100644 packages/catalog/src/identity/family.ts create mode 100644 packages/catalog/test/hosts.test.ts create mode 100644 packages/catalog/test/identity-family.test.ts diff --git a/bun.lock b/bun.lock index 61738b81b..d5d5a5919 100644 --- a/bun.lock +++ b/bun.lock @@ -123,6 +123,7 @@ }, "dependencies": { "@oh-my-pi/pi-ai": "catalog:", + "@oh-my-pi/pi-catalog": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "fastembed": "catalog:", "lru-cache": "catalog:", diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 31e8af89d..6bd5a2b1b 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -8,6 +8,7 @@ */ import type { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity"; import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils"; @@ -852,22 +853,6 @@ function buildAdditionalModelRequestFields( return result; } -/** - * Adaptive thinking `display` is supported starting with Claude Opus 4.7 and - * Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet - * 4.6+) reject the field. Bedrock model ids are prefixed with region/inference- - * profile slugs (e.g. `eu.anthropic.claude-opus-4-7-...`); the regex matches - * the Claude model fragment regardless of prefix. - */ -function supportsAdaptiveThinkingDisplay(modelId: string): boolean { - if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true; - const match = /claude-opus-(\d+)-(\d+)/.exec(modelId); - if (!match) return false; - const major = Number(match[1]); - const minor = Number(match[2]); - return major > 4 || (major === 4 && minor >= 7); -} - /** * Bedrock's wire format expects the image as `{ source: { bytes: }, format }`. * The caller already passes base64-encoded data, so no decode/re-encode round-trip is needed. diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index ce2c4df34..8b05244c6 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2,12 +2,9 @@ import * as nodeCrypto from "node:crypto"; import * as fs from "node:fs"; import { scheduler } from "node:timers/promises"; import * as tls from "node:tls"; -import { - hasOpus47ApiRestrictions, - isAnthropicFableOrMythosModel, - mapEffortToAnthropicAdaptiveEffort, - supportsMidConversationSystemMessages, -} from "@oh-my-pi/pi-catalog/model-thinking"; +import { isOfficialAnthropicApiUrl, resolveAnthropicCompat } from "@oh-my-pi/pi-catalog/compat/anthropic"; +import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity"; +import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { isAnthropicOAuthToken } from "@oh-my-pi/pi-catalog/utils"; import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot"; @@ -181,16 +178,6 @@ function isClaudeCodeClientUserAgent(userAgent: string | undefined): userAgent i return userAgent.toLowerCase().startsWith("claude-cli"); } -export function isAnthropicApiBaseUrl(baseUrl?: string): boolean { - if (!baseUrl) return true; - try { - const url = new URL(baseUrl); - return url.protocol.toLowerCase() === "https:" && url.hostname.toLowerCase() === "api.anthropic.com"; - } catch { - return false; - } -} - const sharedHeaders = { "Accept-Encoding": "gzip, deflate, br, zstd", Connection: "keep-alive", @@ -263,7 +250,7 @@ export function buildAnthropicHeaders(options: AnthropicHeaderOptions): Record 4 || (major === 4 && minor >= 7); -} - const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages"; type AnthropicProviderSessionState = ProviderSessionState & { @@ -450,7 +421,9 @@ function getCacheControl( return { retention }; } const ttl = - retention === "long" && isAnthropicApiBaseUrl(baseUrl) && getAnthropicCompat(model).supportsLongCacheRetention + retention === "long" && + isOfficialAnthropicApiUrl(baseUrl) && + resolveAnthropicCompat(model).supportsLongCacheRetention ? "1h" : undefined; return { @@ -1146,7 +1119,7 @@ function parseAnthropicCustomHeaders(rawHeaders: string | undefined): Record | undefined { - if (!isFoundryEnabled() && isAnthropicApiBaseUrl(baseUrl)) return undefined; + if (!isFoundryEnabled() && isOfficialAnthropicApiUrl(baseUrl)) return undefined; return parseAnthropicCustomHeaders($env.ANTHROPIC_CUSTOM_HEADERS); } @@ -1404,24 +1377,6 @@ async function* observeDecodedAnthropicSdkEvents( } } -function getAnthropicCompat( - model: Model<"anthropic-messages">, -): Required["compat"]>> { - return { - disableStrictTools: model.compat?.disableStrictTools ?? false, - disableAdaptiveThinking: model.compat?.disableAdaptiveThinking ?? false, - supportsEagerToolInputStreaming: model.compat?.supportsEagerToolInputStreaming ?? true, - supportsLongCacheRetention: model.compat?.supportsLongCacheRetention ?? true, - supportsMidConversationSystem: - model.compat?.supportsMidConversationSystem ?? - // First-party Claude API only. Bedrock/Vertex/Foundry and other - // Anthropic-compatible proxies reject the role; gate auto-detection on - // the canonical api.anthropic.com host plus a supported model id. - (isAnthropicApiBaseUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id)), - supportsForcedToolChoice: model.compat?.supportsForcedToolChoice ?? !isAnthropicFableOrMythosModel(model.id), - }; -} - const PROVIDER_MAX_RETRIES = 3; const PROVIDER_BASE_DELAY_MS = 2000; @@ -1626,7 +1581,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( const sendsAdaptiveEffortPin = options?.thinkingEnabled === false && model.thinking?.mode === "anthropic-adaptive" && - !getAnthropicCompat(model).disableAdaptiveThinking; + !resolveAnthropicCompat(model).disableAdaptiveThinking; if ( model.reasoning && (options?.thinkingEnabled || sendsAdaptiveEffortPin) && @@ -1635,7 +1590,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( extraBetas.push(effortBeta); } if ( - getAnthropicCompat(model).supportsMidConversationSystem && + resolveAnthropicCompat(model).supportsMidConversationSystem && !extraBetas.includes(midConversationSystemBeta) ) { // convertAnthropicMessages may upgrade developer turns to the @@ -2332,7 +2287,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A isOAuth, claudeCodeSessionId, } = args; - const compat = getAnthropicCompat(model); + const compat = resolveAnthropicCompat(model); const needsInterleavedBeta = interleavedThinking && !supportsAdaptiveThinkingDisplay(model.id); const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming; const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey); @@ -2443,7 +2398,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A const authorizationHeader = getHeaderCaseInsensitive(defaultHeaders, "Authorization"); const shouldSuppressClientApiKey = !oauthToken && - !isAnthropicApiBaseUrl(baseUrl) && + !isOfficialAnthropicApiUrl(baseUrl) && typeof authorizationHeader === "string" && /^Bearer\s+/i.test(authorizationHeader); @@ -2777,7 +2732,7 @@ function buildParams( context.tools, isOAuthToken, disableStrictTools || model.provider === "github-copilot", - getAnthropicCompat(model).supportsEagerToolInputStreaming, + resolveAnthropicCompat(model).supportsEagerToolInputStreaming, ); } else if (isOAuthToken) { tools = []; @@ -2800,7 +2755,7 @@ function buildParams( if (options?.thinkingEnabled) { const mode = model.thinking?.mode; const effort = resolveAnthropicAdaptiveEffort(model, options); - const compat = getAnthropicCompat(model); + const compat = resolveAnthropicCompat(model); if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" }; // Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking @@ -2823,7 +2778,7 @@ function buildParams( if (mode === "anthropic-budget-effort" && effort) outputConfigEffort = effort; } } else if (options?.thinkingEnabled === false) { - const compat = getAnthropicCompat(model); + const compat = resolveAnthropicCompat(model); if (model.thinking?.mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { // Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject // `thinking.type: "disabled"` — adaptive thinking cannot be switched off. @@ -2912,7 +2867,7 @@ function buildParams( // request succeeds; the tool stays available and the caller's prompt steers // the model toward it. const choiceType = params.tool_choice?.type; - if ((choiceType === "any" || choiceType === "tool") && !getAnthropicCompat(model).supportsForcedToolChoice) { + if ((choiceType === "any" || choiceType === "tool") && !resolveAnthropicCompat(model).supportsForcedToolChoice) { params.tool_choice = { type: "auto" }; } } @@ -2926,52 +2881,6 @@ function buildParams( return params; } -/** - * Z.AI's Anthropic-compatible proxy at `api.z.ai/api/anthropic` deserializes - * tool_result blocks into a Python class that accesses `.id`, even though - * Anthropic's standard tool_result schema only carries `tool_use_id`. Detect - * that endpoint so we can emit the non-standard alias for it without - * polluting requests to api.anthropic.com or other compatible proxies. - * See: https://github.com/can1357/oh-my-pi/issues/814 - */ -function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { - if (model.provider === "zai") return true; - const baseUrl = model.baseUrl; - if (!baseUrl) return false; - try { - return new URL(baseUrl).hostname.toLowerCase() === "api.z.ai"; - } catch { - return false; - } -} - -/** - * Returns true when unsigned `thinking` blocks from prior assistant turns should - * be replayed as Anthropic-native thinking instead of demoted to text. - * - * Official Anthropic (matched via `isAnthropicApiBaseUrl`, which intentionally - * treats a missing baseUrl as official since `resolveAnthropicBaseUrl` routes - * it to `https://api.anthropic.com`) enforces signature-based thinking-chain - * integrity, so unsigned blocks must remain text there. Anthropic-compatible - * reasoning endpoints commonly emit unsigned thinking blocks while still - * expecting them back as `type: "thinking"` on continuation; demoting them - * loses the model's reasoning chain and can destabilize the next tool-call - * arguments (#2005). Known non-signing hosts are also preserved for - * compatibility. - */ -function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">, baseUrl: string | undefined): boolean { - if (model.provider === "zai" || model.provider === "deepseek") return true; - if (baseUrl) { - try { - const hostname = new URL(baseUrl).hostname.toLowerCase(); - if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true; - } catch { - // Fall through to the protocol-level reasoning rule below. - } - } - return model.reasoning && !isAnthropicApiBaseUrl(baseUrl); -} - function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam { const block: ContentBlockParam = { type: "tool_result", @@ -2979,7 +2888,7 @@ function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResul content: convertContentBlocks(msg.content, model.input.includes("image")), is_error: msg.isError, }; - if (isZaiAnthropicEndpoint(model)) { + if (resolveAnthropicCompat(model).requiresToolResultId) { // Z.AI workaround (issue #814): include `id` aliased to `tool_use_id`. (block as unknown as Record).id = msg.toolCallId; } @@ -3092,7 +3001,7 @@ export function convertAnthropicMessages( } if (block.thinking.trim().length === 0) continue; if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) { - if (shouldReplayUnsignedThinking(model, baseUrl)) { + if (resolveAnthropicCompat(model, baseUrl).replayUnsignedThinking) { blocks.push({ type: "thinking", thinking: block.thinking.toWellFormed(), @@ -3170,7 +3079,7 @@ export function convertAnthropicMessages( // never consecutive. Requiring the next param to be `assistant` (or absent) // covers both the "followed by assistant / last" and "no consecutive system" // constraints. Anything that does not qualify stays a `user` message. - if (developerParamIndices.length > 0 && getAnthropicCompat(model).supportsMidConversationSystem) { + if (developerParamIndices.length > 0 && resolveAnthropicCompat(model).supportsMidConversationSystem) { for (const idx of developerParamIndices) { const followsUser = idx > 0 && params[idx - 1]?.role === "user"; const next = params[idx + 1]; diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 36bf5c58e..15506057c 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -1,3 +1,4 @@ +import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import { AzureOpenAI, APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai"; import type { @@ -31,7 +32,7 @@ import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schem import { createSdkStreamRequestOptions } from "../utils/sdk-stream-timeout"; import { notifyRawSseEvent } from "../utils/sse-debug"; import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice"; -import { getOpenAIResponsesCacheSessionId, supportsDeveloperRole } from "./openai-responses"; +import { getOpenAIResponsesCacheSessionId } from "./openai-responses"; import { appendResponsesToolResultMessages, applyCommonResponsesSamplingParams, @@ -337,7 +338,10 @@ function convertMessages( const systemPrompts = normalizeSystemPrompts(context.systemPrompt); if (systemPrompts.length > 0) { - const role = model.reasoning && supportsDeveloperRole(resolvedBaseUrl ?? model) ? "developer" : "system"; + const role = + model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole + ? "developer" + : "system"; for (const systemPrompt of systemPrompts) { messages.push({ role, content: systemPrompt }); } diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index bf30aed6a..3acdc904b 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -1,6 +1,8 @@ import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; import type { Effort } from "@oh-my-pi/pi-catalog/effort"; import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; +import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts"; +import { isDeepseekModelIdOrName, isKimiModelId } from "@oh-my-pi/pi-catalog/identity"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot"; @@ -390,44 +392,6 @@ function getTrailingPartialDeepseekToken(text: string): string { const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE = "OpenAI completions stream timed out while waiting for the first event"; -const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; -const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i; - -// DeepSeek V4 reasoning models on the official api.deepseek.com emit no SSE -// bytes while the model finishes its private chain-of-thought, which routinely -// takes longer than the generic 100s first-event floor under load (issue -// #2177). Mirror the GLM coding-plan widening: a 5-minute idle floor lifts the -// first-event watchdog (it floors at idle) without changing the runtime -// streaming behavior, so reasoning warm-ups stop aborting and retrying. -const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; - -function isDirectDeepseekReasoningModel(model: Model<"openai-completions">): boolean { - if (!model.reasoning) return false; - if (model.provider === "deepseek") return true; - return model.baseUrl.toLowerCase().includes("api.deepseek.com"); -} - -/** Returns the widened OpenAI stream watchdog floor for slow reasoning models hosted on OpenAI-compatible endpoints. */ -export function getOpenAICompletionsStreamIdleTimeoutFallbackMs( - model: Model<"openai-completions">, -): number | undefined { - if (GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) { - if (model.provider === "zhipu-coding-plan" || model.provider === "zai") - return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; - - const baseUrl = model.baseUrl.toLowerCase(); - if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) { - return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS; - } - } - - if (isDirectDeepseekReasoningModel(model)) { - return DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS; - } - - return undefined; -} - async function* observeDecodedOpenAICompletionChunks( chunks: AsyncIterable, observer: (event: RawSseEvent) => void, @@ -468,7 +432,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( try { const apiKey = options?.apiKey || getEnvApiKey(model.provider) || ""; - const idleTimeoutFallbackMs = getOpenAICompletionsStreamIdleTimeoutFallbackMs(model); + const idleTimeoutFallbackMs = resolveOpenAICompat(model).streamIdleTimeoutMs; const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs); const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs); @@ -576,7 +540,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( // though tool calls are also surfaced structurally. Strip the leaked markers // so users don't see raw `<|...|>` tokens. const stripDeepseekChatTemplateTokens = - /deepseek/i.test(model.id) && (model.provider === "nvidia" || model.provider === "deepseek"); + isDeepseekModelIdOrName(model.id) && (model.provider === "nvidia" || model.provider === "deepseek"); type ToolCallStreamBlock = ToolCall & { partialArgs?: string | Record; streamIndex?: number; @@ -1252,8 +1216,8 @@ function buildParams( compat.allowsSyntheticReasoningContentForToolCalls = false; compat.reasoningContentField = "reasoning_content"; } - const isKimiModelId = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id); - const isOpenRouter = model.baseUrl.includes("openrouter.ai"); + const isKimiFamilyModel = isKimiModelId(model.id); + const isOpenRouter = modelMatchesHost(model, "openrouter"); const messages = convertMessages(model, context, compat); maybeAddAnthropicCacheControl(compat, messages); const supportsReasoningParams = model.provider !== "github-copilot"; @@ -1266,14 +1230,14 @@ function buildParams( // before the final answer. Always send max_tokens — match the same // Kimi-family regex used by the compat detector. // Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts. - const requestedMaxTokens = options?.maxTokens ?? (isKimiModelId ? model.maxTokens : undefined); + const requestedMaxTokens = options?.maxTokens ?? (isKimiFamilyModel ? model.maxTokens : undefined); // OpenRouter fans out to upstreams whose output caps differ from the catalog // value (which tracks the highest-cap provider). A max_tokens above the routed // upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras // GLM-4.7, ~40k) for a higher-cap one, defeating `provider.order`/`only`. Omit // it for OpenRouter so each upstream self-caps and routing is honored. Kimi is // exempt — it derives TPM rate limits from max_tokens (see above). - const omitMaxTokensForRouting = isOpenRouter && !isKimiModelId; + const omitMaxTokensForRouting = isOpenRouter && !isKimiFamilyModel; const effectiveMaxTokens = requestedMaxTokens === undefined || omitMaxTokensForRouting ? undefined @@ -1442,12 +1406,12 @@ function buildParams( } // OpenRouter provider routing preferences - if (model.baseUrl.includes("openrouter.ai") && compat.openRouterRouting) { + if (modelMatchesHost(model, "openrouter") && compat.openRouterRouting) { params.provider = compat.openRouterRouting; } // Vercel AI Gateway provider routing preferences - if (model.baseUrl.includes("ai-gateway.vercel.sh") && model.compat?.vercelGatewayRouting) { + if (modelMatchesHost(model, "vercelAIGateway") && model.compat?.vercelGatewayRouting) { const routing = model.compat.vercelGatewayRouting; if (routing.only || routing.order) { const gatewayOptions: Record = {}; diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 45ac36a0b..ef7e0653c 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -1,3 +1,5 @@ +import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts"; import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot"; import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import OpenAI, { APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai"; @@ -10,7 +12,6 @@ import type { import { getEnvApiKey } from "../stream"; import type { AssistantMessage, - CacheRetention, Context, FetchImpl, MessageAttribution, @@ -69,20 +70,6 @@ import { } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; -/** - * Get prompt cache retention based on cacheRetention and base URL. - * Only applies to direct OpenAI API calls (api.openai.com). - */ -function getPromptCacheRetention(baseUrl: string, cacheRetention: CacheRetention): "24h" | undefined { - if (cacheRetention !== "long") { - return undefined; - } - if (baseUrl.includes("api.openai.com")) { - return "24h"; - } - return undefined; -} - export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined { if (!sessionId || sessionId.length === 0) return undefined; const wellFormed = sessionId.toWellFormed(); @@ -442,13 +429,14 @@ function buildParams( ): OpenAIResponsesSamplingParams { const strictResponsesPairing = options?.strictResponsesPairing ?? - (isAzureOpenAIBaseUrl(model.baseUrl ?? "") || model.provider === "github-copilot"); + (hostMatchesUrl(model.baseUrl ?? "", "azureOpenAI") || model.provider === "github-copilot"); const messages = convertConversationMessages(model, context, strictResponsesPairing, providerSessionState, options); const systemPrompts = normalizeSystemPrompts(context.systemPrompt); let systemInstructions: string | undefined; if (systemPrompts.length > 0) { - const needsDeveloperRole = model.reasoning && supportsDeveloperRole(resolvedBaseUrl ?? model); + const needsDeveloperRole = + model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole; if (needsDeveloperRole) { // Reasoning models on known OpenAI-compatible endpoints require the // `developer` role. Send all system prompts inline in `input`. @@ -472,7 +460,10 @@ function buildParams( stream: true, prompt_cache_key: promptCacheKey, prompt_cache_retention: promptCacheKey - ? getPromptCacheRetention(resolvedBaseUrl ?? model.baseUrl, cacheRetention) + ? cacheRetention === "long" && + resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsLongPromptCacheRetention + ? "24h" + : undefined : undefined, store: false, stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined, @@ -485,7 +476,11 @@ function buildParams( // `StreamOptions.frequencyPenalty` is intentionally dropped for this provider. if (context.tools) { - params.tools = convertTools(context.tools, supportsStrictMode(model), model); + params.tools = convertTools( + context.tools, + resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsStrictMode, + model, + ); if (options?.toolChoice) { params.tool_choice = mapOpenAIResponsesToolChoiceForTools(options.toolChoice, context.tools, model); } @@ -528,34 +523,6 @@ function mapReasoningEffort( return reasoningEffortMap?.[effort] ?? effort; } -function isAzureOpenAIBaseUrl(baseUrl: string): boolean { - return baseUrl.includes(".openai.azure.com") || baseUrl.includes("azure.com/openai"); -} - -function supportsStrictMode(model: Model<"openai-responses">): boolean { - if (model.provider === "openai" || model.provider === "azure" || model.provider === "github-copilot") return true; - - const baseUrl = model.baseUrl.toLowerCase(); - return ( - baseUrl.includes("api.openai.com") || - baseUrl.includes(".openai.azure.com") || - baseUrl.includes("models.inference.ai.azure.com") - ); -} - -export function supportsDeveloperRole(modelOrBaseUrl: Pick | string): boolean { - const baseUrl = - typeof modelOrBaseUrl === "string" ? modelOrBaseUrl.toLowerCase() : (modelOrBaseUrl.baseUrl ?? "").toLowerCase(); - return ( - baseUrl.includes("api.openai.com") || - baseUrl.includes(".openai.azure.com") || - baseUrl.includes("azure.com/openai") || - baseUrl.includes("models.inference.ai.azure.com") || - baseUrl.includes("githubcopilot.com") || - baseUrl.includes("copilot-api.") - ); -} - function convertConversationMessages( model: Model<"openai-responses">, context: Context, diff --git a/packages/ai/src/providers/vision-guard.ts b/packages/ai/src/providers/vision-guard.ts index c376b3ee7..a16a39b31 100644 --- a/packages/ai/src/providers/vision-guard.ts +++ b/packages/ai/src/providers/vision-guard.ts @@ -1,3 +1,6 @@ +import { isDashscopeCompatibleModeUrl } from "@oh-my-pi/pi-catalog/hosts"; +import { isQwenModelId } from "@oh-my-pi/pi-catalog/identity"; + import type { ImageContent, Model, TextContent } from "../types"; export const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]"; @@ -42,11 +45,10 @@ export function joinTextWithImagePlaceholder(text: string, omittedImages: boolea * provider (issue #1859) can't drive the request into an unrecoverable 400. */ export function isDashscopeCompatibleModeTextOnlyQwen(model: Model<"openai-completions">): boolean { - const baseUrl = model.baseUrl.toLowerCase(); - if (!baseUrl.includes("dashscope") || !baseUrl.includes("aliyuncs.com") || !baseUrl.includes("/compatible-mode")) { + if (!isDashscopeCompatibleModeUrl(model.baseUrl)) { return false; } const id = model.id.toLowerCase(); - if (!id.includes("qwen")) return false; + if (!isQwenModelId(model.id)) return false; return /\bqwen(?:[\d.]+)?-max\b/.test(id) || /\bqwen(?:[\d.]+)?-coder\b/.test(id); } diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 15bc3eacf..b0e87e037 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1,4 +1,5 @@ import type { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { isVertexExpressOpenAIUrl, isVertexRawPredictUrl } from "@oh-my-pi/pi-catalog/hosts"; import { mapEffortToAnthropicAdaptiveEffort, mapEffortToGoogleThinkingLevel, @@ -65,8 +66,8 @@ import { withRequestDebugFetch } from "./utils/request-debug"; function isGoogleVertexAuthenticatedModel(model: Model): boolean { return ( model.provider === "google-vertex" && - ((model.api === "openai-completions" && model.baseUrl.includes("/endpoints/openapi")) || - (model.api === "anthropic-messages" && model.baseUrl.includes(":streamRawPredict"))) + ((model.api === "openai-completions" && isVertexExpressOpenAIUrl(model.baseUrl)) || + (model.api === "anthropic-messages" && isVertexRawPredictUrl(model.baseUrl))) ); } @@ -78,7 +79,7 @@ function createVertexAuthenticatedFetch(options: StreamOptions | undefined): Fet headers.set("Authorization", `Bearer ${token}`); const rewritten = resolveVertexRequest(input); const url = rewritten instanceof Request ? rewritten.url : rewritten.toString(); - if (isVertexAnthropicRawPredict(url)) { + if (isVertexRawPredictUrl(url)) { const bodyText = await readVertexRequestBody(rewritten, init); const transformed = transformVertexAnthropicBody(bodyText); return baseFetch(url, { @@ -93,10 +94,6 @@ function createVertexAuthenticatedFetch(options: StreamOptions | undefined): Fet return Object.assign(vertexFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {}); } -function isVertexAnthropicRawPredict(url: string): boolean { - return url.includes(":streamRawPredict") || url.includes(":rawPredict"); -} - async function readVertexRequestBody(input: string | URL | Request, init: RequestInit | undefined): Promise { if (input instanceof Request) return input.clone().text(); const body = init?.body; diff --git a/packages/ai/src/utils/stream-markup-healing.ts b/packages/ai/src/utils/stream-markup-healing.ts index 3598c9868..bbc688ab0 100644 --- a/packages/ai/src/utils/stream-markup-healing.ts +++ b/packages/ai/src/utils/stream-markup-healing.ts @@ -13,6 +13,8 @@ * deltas for thinking blocks, and holds partial tags across chunk boundaries. */ +import { isDeepseekModelIdOrName } from "@oh-my-pi/pi-catalog/identity"; + import { parseJsonWithRepair } from "./json-parse"; const KIMI_SECTION_BEGIN = "<|tool_calls_section_begin|>"; @@ -622,7 +624,7 @@ export function modelMayLeakKimiToolCalls(provider: string, modelId: string): bo /** Cheap model/provider gate for DeepSeek DSML envelope leaks. */ export function modelMayLeakDsmlToolCalls(provider: string, modelId: string): boolean { - if (!/deepseek/i.test(modelId)) return false; + if (!isDeepseekModelIdOrName(modelId)) return false; return ( provider === "ollama" || provider === "ollama-cloud" || diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts index d18e77412..4284b2e9f 100644 --- a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -161,7 +161,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { }); it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => { - // `isAnthropicApiBaseUrl(undefined) === true` because the actual HTTP + // `isOfficialAnthropicApiUrl(undefined) === true` because the actual HTTP // dispatch falls back to https://api.anthropic.com. Same-id custom // overrides that only tweak model metadata (no baseUrl override) must // not regress to native-thinking replay against the first-party API. diff --git a/packages/ai/test/openai-completions-progress-chunk.test.ts b/packages/ai/test/openai-completions-progress-chunk.test.ts index 42520d6c5..831fc9ab5 100644 --- a/packages/ai/test/openai-completions-progress-chunk.test.ts +++ b/packages/ai/test/openai-completions-progress-chunk.test.ts @@ -1,10 +1,10 @@ import { describe, expect, it } from "bun:test"; import { - getOpenAICompletionsStreamIdleTimeoutFallbackMs, isOpenAICompletionsProgressChunk, streamOpenAICompletions, } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; const openAICompletionsModel = { @@ -78,7 +78,7 @@ function createKeepaliveOnlyCompletionsResponse(modelId: string, signal: AbortSi }); } -describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { +describe("resolveOpenAICompat stream idle timeout", () => { it("widens GLM 5.1 coding-plan stream watchdogs", () => { const model = { ...openAICompletionsModel, @@ -88,7 +88,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4", } satisfies Model<"openai-completions">; - expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000); + expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000); }); it("also widens custom Z.AI OpenAI-compatible GLM 5.1 endpoints", () => { @@ -100,7 +100,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { baseUrl: "https://api.z.ai/api/coding/paas/v4", } satisfies Model<"openai-completions">; - expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000); + expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000); }); it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => { @@ -113,7 +113,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { reasoning: true, } satisfies Model<"openai-completions">; - expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000); + expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000); }); it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => { @@ -126,7 +126,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { reasoning: true, } satisfies Model<"openai-completions">; - expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000); + expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000); }); it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => { @@ -139,7 +139,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { reasoning: false, } satisfies Model<"openai-completions">; - expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined(); + expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined(); }); it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => { @@ -152,11 +152,11 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => { reasoning: true, } satisfies Model<"openai-completions">; - expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined(); + expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined(); }); it("keeps ordinary OpenAI-compatible models on the global timeout", () => { - expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined(); + expect(resolveOpenAICompat(openAICompletionsModel).streamIdleTimeoutMs).toBeUndefined(); }); }); diff --git a/packages/ai/test/openai-responses-developer-role.test.ts b/packages/ai/test/openai-responses-developer-role.test.ts index 6789f2e3b..5129a0048 100644 --- a/packages/ai/test/openai-responses-developer-role.test.ts +++ b/packages/ai/test/openai-responses-developer-role.test.ts @@ -1,78 +1,78 @@ import { describe, expect, it } from "bun:test"; -import { supportsDeveloperRole } from "@oh-my-pi/pi-ai/providers/openai-responses"; -import type { Model } from "@oh-my-pi/pi-ai/types"; -describe("supportsDeveloperRole", () => { +import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; + +describe("resolveOpenAIResponsesCompat supportsDeveloperRole", () => { it("returns true for openai provider with official API base URL", () => { - const model = { provider: "openai", baseUrl: "https://api.openai.com/v1" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "openai", baseUrl: "https://api.openai.com/v1" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns false for openai provider with custom proxy base URL", () => { - const model = { provider: "openai", baseUrl: "https://my-proxy.example.com/v1" } as Model; - expect(supportsDeveloperRole(model)).toBe(false); + const model = { provider: "openai", baseUrl: "https://my-proxy.example.com/v1" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("returns true for github-copilot provider", () => { - const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns false for github-copilot provider with custom proxy base URL", () => { - const model = { provider: "github-copilot", baseUrl: "https://proxy.example.com/v1" } as Model; - expect(supportsDeveloperRole(model)).toBe(false); + const model = { provider: "github-copilot", baseUrl: "https://proxy.example.com/v1" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("returns true for Azure OpenAI base URL", () => { - const model = { provider: "azure-openai", baseUrl: "https://my-resource.openai.azure.com/openai" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "azure-openai", baseUrl: "https://my-resource.openai.azure.com/openai" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for Azure AI Inference base URL", () => { const model = { provider: "azure-openai", baseUrl: "https://models.inference.ai.azure.com/v1/chat/completions", - } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for api.openai.com base URL", () => { - const model = { provider: "custom", baseUrl: "https://api.openai.com/v1/chat/completions" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "custom", baseUrl: "https://api.openai.com/v1/chat/completions" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns false for generic third-party provider", () => { - const model = { provider: "custom", baseUrl: "https://api.example.com/v1" } as Model; - expect(supportsDeveloperRole(model)).toBe(false); + const model = { provider: "custom", baseUrl: "https://api.example.com/v1" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("returns false for local/localhost endpoints", () => { - const model = { provider: "custom", baseUrl: "http://localhost:8080/v1" } as Model; - expect(supportsDeveloperRole(model)).toBe(false); + const model = { provider: "custom", baseUrl: "http://localhost:8080/v1" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("is case-insensitive for base URL matching", () => { - const model = { provider: "custom", baseUrl: "https://API.OPENAI.COM/v1" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "custom", baseUrl: "https://API.OPENAI.COM/v1" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for azure.com/openai base URL", () => { - const model = { provider: "custom", baseUrl: "https://azure.com/openai/deployments/my-model" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "custom", baseUrl: "https://azure.com/openai/deployments/my-model" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for github-copilot provider with api.githubcopilot.com", () => { - const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for github-copilot provider with api.enterprise.githubcopilot.com", () => { - const model = { provider: "github-copilot", baseUrl: "https://api.enterprise.githubcopilot.com" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "github-copilot", baseUrl: "https://api.enterprise.githubcopilot.com" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for github-copilot provider with copilot-api enterprise domain", () => { - const model = { provider: "github-copilot", baseUrl: "https://copilot-api.mycompany.com" } as Model; - expect(supportsDeveloperRole(model)).toBe(true); + const model = { provider: "github-copilot", baseUrl: "https://copilot-api.mycompany.com" }; + expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); }); diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 642ffbda7..823a04d88 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -1,9 +1,12 @@ # Changelog ## [Unreleased] - ### Added +- Added `hostMatchesUrl`, `modelMatchesHost`, and endpoint-shape helpers in the new `hosts` module for consistent provider/baseUrl matching +- Added `streamIdleTimeoutMs` to `OpenAICompat` and now auto-populated it for GLM coding-plan and direct DeepSeek reasoning models +- Added `supportsLongPromptCacheRetention` and the OpenAI Responses helpers `detectOpenAIResponsesCompat`/`resolveOpenAIResponsesCompat` +- Added anthropic-messages compatibility resolution with new `AnthropicCompat` fields `requiresToolResultId` and `replayUnsignedThinking` - New `@oh-my-pi/pi-catalog` package: the model catalog extracted from `@oh-my-pi/pi-ai`. Owns the bundled `models.json` and its generation pipeline (`scripts/generate-models.ts`), the core model data types (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces), thinking metadata enrichment and generated policies (`model-thinking.ts`), the SQLite model cache and model manager, per-provider discovery factories (`provider-models/`), the discovery protocol clients (`discovery/`), and the new `CATALOG_PROVIDERS` table — the single source of truth for provider ids, default models, and discovery wiring (`KnownProvider`, `PROVIDER_DESCRIPTORS`, and `DEFAULT_MODEL_PER_PROVIDER` are derived from it). - New `identity/` module centralizing model-identity concerns that were previously duplicated across packages: family classification and version parsing (`identity/classify.ts`, extracted from pi-ai's `model-thinking` internals), canonical model equivalence with injected reference data (`identity/equivalence.ts`, from coding-agent's `model-equivalence`), proxy/reseller reference lookup (`identity/reference.ts`, from coding-agent's `model-registry`), bracket-affix and id-segment helpers (`identity/id.ts`), a single trailing-marker vocabulary with canonical vs reference flavors (`identity/markers.ts` — `search` stays reference-only so Perplexity's `sonar-pro-search` remains canonical-distinct), and provider priority ordering (`identity/priority.ts`). - Memoized bundled-reference accessors (`getBundledCanonicalReferenceData` / `getBundledModelReferenceIndex` in `identity/bundled.ts`): one lazy walk of the bundled catalog feeds both canonical equivalence and proxy-reference lookup, so consumers no longer hand-roll the glue. @@ -11,6 +14,8 @@ ### Changed +- Changed OpenAI compatibility detection to use shared host classifiers (`modelMatchesHost`/`hostMatchesUrl`) with normalized matching instead of raw URL substring checks +- Changed `hostMatchesUrl`/`modelMatchesHost` usage in compatibility detection to reduce mismatches across case variants and provider alias hosts - Provider catalog entries now carry the runtime API-key env fallback as an ordered `envVars` list; `catalogDiscovery.envVars` became an optional generation-time override (only `cursor` and `vercel-ai-gateway` differ) and `PROVIDER_DESCRIPTORS` materializes the resolved list for `generate-models.ts`. - `Model`'s api parameter now defaults to `Api` instead of `any` (`Model`), so bare `Model` no longer behaves as `Model` at call sites. diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts new file mode 100644 index 000000000..2e06c66ea --- /dev/null +++ b/packages/catalog/src/compat/anthropic.ts @@ -0,0 +1,109 @@ +/** + * Anthropic-messages compatibility detection and resolution — the + * anthropic-side analogue of `./openai`. Detect-time defaults come from + * provider ids, strict URL checks, and model-id classification; explicit + * `model.compat` overrides always win. + */ +import { isAnthropicFableOrMythosModel, supportsMidConversationSystemMessages } from "../model-thinking"; +import type { AnthropicCompat, Model } from "../types"; + +/** + * Official first-party Anthropic API check (https + exact host). A missing + * baseUrl is official on purpose: request dispatch falls back to + * `https://api.anthropic.com`. Strict URL parsing (not substring) because the + * callers gate auth flows and body mutations on it. + */ +export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean { + if (!baseUrl) return true; + try { + const url = new URL(baseUrl); + return url.protocol.toLowerCase() === "https:" && url.hostname.toLowerCase() === "api.anthropic.com"; + } catch { + return false; + } +} + +/** Z.AI's Anthropic-compatible proxy (`api.z.ai/api/anthropic`), strict-host matched. */ +function isZaiAnthropicUrl(baseUrl: string | undefined): boolean { + if (!baseUrl) return false; + try { + return new URL(baseUrl).hostname.toLowerCase() === "api.z.ai"; + } catch { + return false; + } +} + +/** DeepSeek-operated host, strict-host matched (`api.deepseek.com` or any `*.deepseek.com`). */ +function isDeepseekHostUrl(baseUrl: string | undefined): boolean { + if (!baseUrl) return false; + try { + const hostname = new URL(baseUrl).hostname.toLowerCase(); + return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com"); + } catch { + return false; + } +} + +export type ResolvedAnthropicCompat = Required; + +/** + * Detect anthropic-messages compatibility defaults from provider/baseUrl/model id. + * @param resolvedBaseUrl - Effective request base URL when it differs from + * `model.baseUrl` (e.g. an options-level override). + */ +export function detectAnthropicCompat( + model: Model<"anthropic-messages">, + resolvedBaseUrl?: string, +): ResolvedAnthropicCompat { + const baseUrl = resolvedBaseUrl ?? model.baseUrl; + const isZai = model.provider === "zai" || isZaiAnthropicUrl(baseUrl); + return { + disableStrictTools: false, + disableAdaptiveThinking: false, + supportsEagerToolInputStreaming: true, + supportsLongCacheRetention: true, + // First-party Claude API only. Bedrock/Vertex/Foundry and other + // Anthropic-compatible gateways reject mid-conversation system roles, so + // detection requires the canonical api.anthropic.com host plus a + // supported model id. + supportsMidConversationSystem: + isOfficialAnthropicApiUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id), + supportsForcedToolChoice: !isAnthropicFableOrMythosModel(model.id), + // Z.AI workaround (issue #814): its proxy deserializes tool_result blocks + // into a class that reads `.id`. + requiresToolResultId: isZai, + // Official Anthropic enforces signature-based thinking-chain integrity, so + // unsigned thinking blocks must stay text there. Anthropic-compatible + // reasoning endpoints commonly emit unsigned thinking blocks while still + // expecting them back as `type: "thinking"` on continuation; demoting them + // loses the reasoning chain and can destabilize the next tool-call + // arguments (#2005). Known non-signing hosts (Z.AI, DeepSeek) are also + // preserved for compatibility. + replayUnsignedThinking: + isZai || + model.provider === "deepseek" || + isDeepseekHostUrl(baseUrl) || + (model.reasoning && !isOfficialAnthropicApiUrl(baseUrl)), + }; +} + +/** Layer explicit `model.compat` overrides onto the detected anthropic defaults. */ +export function resolveAnthropicCompat( + model: Model<"anthropic-messages">, + resolvedBaseUrl?: string, +): ResolvedAnthropicCompat { + const detected = detectAnthropicCompat(model, resolvedBaseUrl); + const compat = model.compat; + if (!compat) return detected; + return { + disableStrictTools: compat.disableStrictTools ?? detected.disableStrictTools, + disableAdaptiveThinking: compat.disableAdaptiveThinking ?? detected.disableAdaptiveThinking, + supportsEagerToolInputStreaming: + compat.supportsEagerToolInputStreaming ?? detected.supportsEagerToolInputStreaming, + supportsLongCacheRetention: compat.supportsLongCacheRetention ?? detected.supportsLongCacheRetention, + supportsMidConversationSystem: compat.supportsMidConversationSystem ?? detected.supportsMidConversationSystem, + supportsForcedToolChoice: compat.supportsForcedToolChoice ?? detected.supportsForcedToolChoice, + requiresToolResultId: compat.requiresToolResultId ?? detected.requiresToolResultId, + replayUnsignedThinking: compat.replayUnsignedThinking ?? detected.replayUnsignedThinking, + }; +} diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index df32da8be..8215d9074 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -1,3 +1,13 @@ +import { hostMatchesUrl, modelMatchesHost } from "../hosts"; +import { + isAnthropicNamespacedModelId, + isClaudeModelId, + isDeepseekModelIdOrName, + isKimiK26ModelId, + isKimiModelId, + isMimoModelIdOrName, + isQwenModelId, +} from "../identity/family"; import type { Model, OpenAICompat } from "../types"; type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh"; @@ -10,6 +20,8 @@ export type ResolvedOpenAICompat = Required< | "vercelGatewayRouting" | "extraBody" | "toolStrictMode" + | "streamIdleTimeoutMs" + | "supportsLongPromptCacheRetention" | "cacheControlFormat" | "thinkingKeep" > @@ -19,9 +31,16 @@ export type ResolvedOpenAICompat = Required< extraBody?: OpenAICompat["extraBody"]; cacheControlFormat?: OpenAICompat["cacheControlFormat"]; thinkingKeep?: OpenAICompat["thinkingKeep"]; + streamIdleTimeoutMs?: number; toolStrictMode: ResolvedToolStrictMode; }; +/** GLM coding-plan SKUs idle for minutes mid-reasoning; see `streamIdleTimeoutMs`. */ +const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i; +const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; +/** Direct DeepSeek reasoning models stall between thinking and answer phases. */ +const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; + function detectStrictModeSupport(provider: string, baseUrl: string): boolean { if ( provider === "openai" || @@ -33,17 +52,13 @@ function detectStrictModeSupport(provider: string, baseUrl: string): boolean { ) { return true; } - - const normalizedBaseUrl = baseUrl.toLowerCase(); return ( - normalizedBaseUrl.includes("api.openai.com") || - normalizedBaseUrl.includes(".openai.azure.com") || - normalizedBaseUrl.includes("models.inference.ai.azure.com") || - normalizedBaseUrl.includes("api.cerebras.ai") || - normalizedBaseUrl.includes("api.together.xyz") || - normalizedBaseUrl.includes("openrouter.ai") || - normalizedBaseUrl.includes("api.deepseek.com") || - normalizedBaseUrl.includes("deepseek.com") + hostMatchesUrl(baseUrl, "openai") || + hostMatchesUrl(baseUrl, "azureOpenAI") || + hostMatchesUrl(baseUrl, "cerebras") || + hostMatchesUrl(baseUrl, "together") || + hostMatchesUrl(baseUrl, "openrouter") || + hostMatchesUrl(baseUrl, "deepseekFamily") ); } @@ -87,23 +102,19 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB const provider = model.provider; // Use resolvedBaseUrl if provided (e.g., after GitHub Copilot proxy-ep resolution) const baseUrl = resolvedBaseUrl ?? model.baseUrl; + const hostModel = { provider, baseUrl }; - const isCerebras = provider === "cerebras" || baseUrl.includes("cerebras.ai"); - const isZai = provider === "zai" || baseUrl.includes("api.z.ai"); - const isZhipu = provider === "zhipu-coding-plan" || baseUrl.includes("open.bigmodel.cn"); - const isKilo = provider === "kilo" || baseUrl.includes("api.kilo.ai"); - const isKimiModel = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id); - const isMoonshotNativeHost = - provider === "moonshot" || provider === "kimi-code" || /api\.moonshot\.ai|api\.kimi\.com/i.test(baseUrl); - const isMoonshotKimi = isKimiModel && isMoonshotNativeHost; - const usesMoonshotKimiPreservedThinking = isMoonshotKimi && /(^|\/)kimi-k2\.6(?:[-:]|$)/i.test(model.id); + const isCerebras = modelMatchesHost(hostModel, "cerebras"); + const isZai = modelMatchesHost(hostModel, "zai"); + const isZhipu = modelMatchesHost(hostModel, "zhipu"); + const isKilo = modelMatchesHost(hostModel, "kilo"); + const isKimiModel = isKimiModelId(model.id); + const isMoonshotKimi = isKimiModel && modelMatchesHost(hostModel, "moonshotNative"); + const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(model.id); const isAnthropicModel = - provider === "anthropic" || - baseUrl.includes("api.anthropic.com") || - /(^|\/)claude[-.]/i.test(model.id) || - /(^|\/)anthropic\//i.test(model.id); - const isAlibaba = provider === "alibaba-coding-plan" || baseUrl.includes("dashscope"); - const isQwen = model.id.toLowerCase().includes("qwen"); + modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(model.id) || isAnthropicNamespacedModelId(model.id); + const isAlibaba = modelMatchesHost(hostModel, "alibabaDashscope"); + const isQwen = isQwenModelId(model.id); // DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in // thinking mode unless prior assistant tool-call turns include `reasoning_content`. The // upstream model is reachable through many OpenAI-compat hosts (api.deepseek.com, Deepinfra, @@ -112,66 +123,52 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB // applies when thinking mode is actually engaged. const lowerId = model.id.toLowerCase(); const lowerName = (model.name ?? "").toLowerCase(); - const isXiaomiHost = - provider === "xiaomi" || provider.startsWith("xiaomi-token-plan-") || baseUrl.includes("xiaomimimo.com"); - const isMimoModel = lowerId.includes("mimo") || lowerName.includes("mimo"); - const isXiaomiMimo = isXiaomiHost && isMimoModel; + const isXiaomiHost = modelMatchesHost(hostModel, "xiaomi"); + const isXiaomiMimo = isXiaomiHost && (isMimoModelIdOrName(model.id) || isMimoModelIdOrName(model.name ?? "")); // OpenCode Zen's `big-pickle` is a DeepSeek reasoning alias; the upstream // 400s come from DeepSeek and require exact reasoning_content replay. const isOpenCodeDeepseekAlias = provider === "opencode-zen" && (lowerId === "big-pickle" || lowerName === "big pickle"); const isDeepseekFamily = - provider === "deepseek" || - baseUrl.includes("deepseek.com") || - lowerId.includes("deepseek") || - lowerName.includes("deepseek") || + modelMatchesHost(hostModel, "deepseekFamily") || + isDeepseekModelIdOrName(model.id) || + isDeepseekModelIdOrName(model.name ?? "") || isOpenCodeDeepseekAlias; - const isDirectDeepseekApi = provider === "deepseek" || baseUrl.includes("api.deepseek.com"); + const isDirectDeepseekApi = modelMatchesHost(hostModel, "deepseekDirect"); const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(model.reasoning); + const isGrok = modelMatchesHost(hostModel, "xai"); + const isMistral = modelMatchesHost(hostModel, "mistral"); + const isOpenCodeHost = modelMatchesHost(hostModel, "opencode"); const isNonStandard = isCerebras || - provider === "xai" || - baseUrl.includes("api.x.ai") || - provider === "mistral" || - baseUrl.includes("mistral.ai") || - baseUrl.includes("chutes.ai") || - baseUrl.includes("deepseek.com") || - baseUrl.includes("fireworks.ai") || + isGrok || + isMistral || + hostMatchesUrl(baseUrl, "chutes") || + hostMatchesUrl(baseUrl, "deepseekFamily") || + hostMatchesUrl(baseUrl, "fireworks") || isAlibaba || isZai || isZhipu || isKilo || isQwen || isXiaomiHost || - provider === "opencode-zen" || - provider === "opencode-go" || - baseUrl.includes("opencode.ai"); + isOpenCodeHost; const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen"; const useMaxTokens = - provider === "mistral" || - baseUrl.includes("mistral.ai") || - baseUrl.includes("chutes.ai") || - baseUrl.includes("fireworks.ai") || - isDirectDeepseekApi; - const isGrok = provider === "xai" || baseUrl.includes("api.x.ai"); - const isMistral = provider === "mistral" || baseUrl.includes("mistral.ai"); + isMistral || hostMatchesUrl(baseUrl, "chutes") || hostMatchesUrl(baseUrl, "fireworks") || isDirectDeepseekApi; // Hosts whose chat-completions endpoints are known to accept multiple // leading `system`/`developer` messages (preferred for KV-cache reuse). // Anything outside this allowlist defaults to coalescing because // strict chat templates (Qwen 3.5+ via vLLM, MiniMax, etc.) reject // follow-up system messages with a 400. - const isOpenAIHost = provider === "openai" || baseUrl.includes("api.openai.com"); - const isAzureHost = - provider === "azure" || - baseUrl.includes(".openai.azure.com") || - baseUrl.includes("models.inference.ai.azure.com") || - baseUrl.includes("azure.com/openai"); - const isOpenRouter = provider === "openrouter" || baseUrl.includes("openrouter.ai"); - const isTogether = provider === "together" || baseUrl.includes("api.together.xyz"); - const isFireworks = baseUrl.includes("fireworks.ai"); - const isGroqHost = provider === "groq" || baseUrl.includes("api.groq.com"); + const isOpenAIHost = modelMatchesHost(hostModel, "openai"); + const isAzureHost = modelMatchesHost(hostModel, "azureOpenAI"); + const isOpenRouter = modelMatchesHost(hostModel, "openrouter"); + const isTogether = modelMatchesHost(hostModel, "together"); + const isFireworks = hostMatchesUrl(baseUrl, "fireworks"); + const isGroqHost = modelMatchesHost(hostModel, "groq"); const isCopilotHost = provider === "github-copilot"; const isZenmuxHost = provider === "zenmux"; // Endpoints that MUST receive a single system block. MiniMax's OpenAI @@ -179,12 +176,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB // Dashscope and Qwen Portal serve Qwen models whose chat template // raises "System message must be at the beginning" if any system // message appears past index 0. - const isMiniMaxHost = - provider === "minimax-code" || - provider === "minimax-code-cn" || - baseUrl.includes("api.minimax.io") || - baseUrl.includes("api.minimaxi.com"); - const isQwenPortal = provider === "qwen-portal" || baseUrl.includes("portal.qwen.ai"); + const isMiniMaxHost = modelMatchesHost(hostModel, "minimax"); + const isQwenPortal = modelMatchesHost(hostModel, "qwenPortal"); const supportsMultipleSystemMessagesDefault = !isMiniMaxHost && !isAlibaba && @@ -234,6 +227,16 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB } satisfies Partial>) : {}; + // Stream-watchdog floor: GLM coding-plan SKUs and direct DeepSeek reasoning + // models idle for minutes mid-reasoning; widen the idle timeout so warm-ups + // stop aborting and retrying. + const streamIdleTimeoutMs = + GLM_CODING_PLAN_MODEL_PATTERN.test(model.id) && (isZai || isZhipu) + ? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS + : model.reasoning && isDirectDeepseekApi + ? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS + : undefined; + return { supportsStore: !isNonStandard, // `developer` is an OpenAI-Responses-era extension to the chat-completions schema. Almost @@ -263,7 +266,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB thinkingFormat: isZai || isZhipu || isMoonshotKimi || isXiaomiMimo ? "zai" - : provider === "openrouter" || baseUrl.includes("openrouter.ai") + : isOpenRouter ? "openrouter" : isAlibaba || isQwen ? "qwen" @@ -286,7 +289,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB (isKimiModel && !isOpenCodeProvider) || (isDeepseekFamily && Boolean(model.reasoning)) || isXiaomiMimo || - ((provider === "openrouter" || baseUrl.includes("openrouter.ai")) && Boolean(model.reasoning)), + (isOpenRouter && Boolean(model.reasoning)), // DeepSeek V4 and Xiaomi MiMo reject synthetic reasoning_content placeholders (".") on tool-call turns. // Kimi and OpenRouter accept them when actual reasoning is unavailable. allowsSyntheticReasoningContentForToolCalls: (!isDeepseekFamily || !model.reasoning) && !isXiaomiMimo, @@ -297,6 +300,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB supportsStrictMode: detectStrictModeSupport(provider, baseUrl), extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined, toolStrictMode: isCerebras ? "all_strict" : "mixed", + streamIdleTimeoutMs, }; } @@ -350,5 +354,62 @@ export function resolveOpenAICompat( supportsStrictMode: model.compat.supportsStrictMode ?? detected.supportsStrictMode, extraBody: model.compat.extraBody ?? detected.extraBody, toolStrictMode: model.compat.toolStrictMode ?? detected.toolStrictMode, + streamIdleTimeoutMs: model.compat.streamIdleTimeoutMs ?? detected.streamIdleTimeoutMs, + }; +} + +/** Resolved Responses-API compatibility view (see `detectOpenAIResponsesCompat`). */ +export interface ResolvedOpenAIResponsesCompat { + supportsDeveloperRole: boolean; + supportsStrictMode: boolean; + supportsLongPromptCacheRetention: boolean; +} + +/** + * Detect Responses-API compatibility from provider/baseUrl. The Responses + * flavor deliberately differs from chat-completions: GitHub Copilot's + * responses endpoint accepts the `developer` role, while strict tool mode is + * scoped to first-party OpenAI/Azure/Copilot providers. Developer-role and + * prompt-cache detection are URL-only on purpose — the historical call sites + * never consulted the provider id for them. + */ +export function detectOpenAIResponsesCompat( + model: { provider: string; baseUrl: string }, + resolvedBaseUrl?: string, +): ResolvedOpenAIResponsesCompat { + const baseUrl = resolvedBaseUrl ?? model.baseUrl ?? ""; + return { + supportsDeveloperRole: + hostMatchesUrl(baseUrl, "openai") || + hostMatchesUrl(baseUrl, "azureOpenAI") || + hostMatchesUrl(baseUrl, "githubCopilot"), + supportsStrictMode: + model.provider === "openai" || + model.provider === "azure" || + model.provider === "github-copilot" || + hostMatchesUrl(baseUrl, "openai") || + hostMatchesUrl(baseUrl, "azureOpenAI"), + supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"), + }; +} + +/** + * Resolve Responses-API compatibility by layering explicit `model.compat` + * overrides onto the detected defaults — the Responses-side analogue of + * `resolveOpenAICompat`. Models bundled with `supportsDeveloperRole: false` + * (codex-mini-style SKUs) take effect here. + */ +export function resolveOpenAIResponsesCompat( + model: { provider: string; baseUrl: string; compat?: OpenAICompat }, + resolvedBaseUrl?: string, +): ResolvedOpenAIResponsesCompat { + const detected = detectOpenAIResponsesCompat(model, resolvedBaseUrl); + const compat = model.compat; + if (!compat) return detected; + return { + supportsDeveloperRole: compat.supportsDeveloperRole ?? detected.supportsDeveloperRole, + supportsStrictMode: compat.supportsStrictMode ?? detected.supportsStrictMode, + supportsLongPromptCacheRetention: + compat.supportsLongPromptCacheRetention ?? detected.supportsLongPromptCacheRetention, }; } diff --git a/packages/catalog/src/hosts.ts b/packages/catalog/src/hosts.ts new file mode 100644 index 000000000..af79151bd --- /dev/null +++ b/packages/catalog/src/hosts.ts @@ -0,0 +1,110 @@ +/** + * Known model-endpoint host classification — the single vocabulary for the + * `provider === id || baseUrl.includes(marker)` idiom that gates wire-level + * behavior (compat detection, routing, header shaping, watchdog floors). + * + * Markers are case-insensitive substrings matched against the base URL, NOT + * parsed hostnames: proxies regularly embed the upstream host in a path + * segment, and the historical call sites all used substring semantics. + * Callers needing strict hostname matching (e.g. guards before request-body + * mutation) should keep their own `new URL().hostname` checks. + */ + +interface HostClassSpec { + /** Provider ids that imply this host class regardless of baseUrl. */ + readonly providers?: readonly string[]; + /** Provider-id prefixes that imply this host class (e.g. `xiaomi-token-plan-`). */ + readonly providerPrefixes?: readonly string[]; + /** Case-insensitive substrings matched against the base URL. */ + readonly urlMarkers: readonly string[]; +} + +export const KNOWN_HOSTS = { + openai: { providers: ["openai"], urlMarkers: ["api.openai.com"] }, + azureOpenAI: { + providers: ["azure"], + urlMarkers: [".openai.azure.com", "azure.com/openai", "models.inference.ai.azure.com"], + }, + openrouter: { providers: ["openrouter"], urlMarkers: ["openrouter.ai"] }, + vercelAIGateway: { providers: ["vercel-ai-gateway"], urlMarkers: ["ai-gateway.vercel.sh"] }, + githubCopilot: { providers: ["github-copilot"], urlMarkers: ["githubcopilot.com", "copilot-api."] }, + anthropic: { providers: ["anthropic"], urlMarkers: ["api.anthropic.com"] }, + /** DeepSeek's first-party API only — gates direct-API quirks (max_tokens field, thinking extraBody). */ + deepseekDirect: { providers: ["deepseek"], urlMarkers: ["api.deepseek.com"] }, + /** Any DeepSeek-operated host (first-party API, web-chat fronts). Wider than `deepseekDirect` on purpose. */ + deepseekFamily: { providers: ["deepseek"], urlMarkers: ["deepseek.com"] }, + cerebras: { providers: ["cerebras"], urlMarkers: ["cerebras.ai"] }, + zai: { providers: ["zai"], urlMarkers: ["api.z.ai"] }, + zhipu: { providers: ["zhipu-coding-plan"], urlMarkers: ["open.bigmodel.cn"] }, + kilo: { providers: ["kilo"], urlMarkers: ["api.kilo.ai"] }, + alibabaDashscope: { providers: ["alibaba-coding-plan"], urlMarkers: ["dashscope"] }, + xiaomi: { providers: ["xiaomi"], providerPrefixes: ["xiaomi-token-plan-"], urlMarkers: ["xiaomimimo.com"] }, + xai: { providers: ["xai"], urlMarkers: ["api.x.ai"] }, + mistral: { providers: ["mistral"], urlMarkers: ["mistral.ai"] }, + together: { providers: ["together"], urlMarkers: ["api.together.xyz"] }, + /** URL-only on purpose: the `fireworks`/`firepass` providers route per-model and not every model is Fireworks-shaped. */ + fireworks: { urlMarkers: ["fireworks.ai"] }, + groq: { providers: ["groq"], urlMarkers: ["api.groq.com"] }, + minimax: { + providers: ["minimax", "minimax-code", "minimax-code-cn"], + urlMarkers: ["api.minimax.io", "api.minimaxi.com"], + }, + qwenPortal: { providers: ["qwen-portal"], urlMarkers: ["portal.qwen.ai"] }, + moonshotNative: { providers: ["moonshot", "kimi-code"], urlMarkers: ["api.moonshot.ai", "api.kimi.com"] }, + opencode: { providers: ["opencode-go", "opencode-zen"], urlMarkers: ["opencode.ai"] }, + chutes: { urlMarkers: ["chutes.ai"] }, +} as const satisfies Record; + +export type KnownHost = keyof typeof KNOWN_HOSTS; + +/** URL-only host check (for call sites that have no provider id, e.g. raw env config). */ +export function hostMatchesUrl(baseUrl: string | undefined, host: KnownHost): boolean { + if (!baseUrl) return false; + const spec: HostClassSpec = KNOWN_HOSTS[host]; + const normalized = baseUrl.toLowerCase(); + for (const marker of spec.urlMarkers) { + if (normalized.includes(marker)) return true; + } + return false; +} + +/** Provider-or-URL host check — the canonical `provider === id || baseUrl.includes(marker)` idiom. */ +export function modelMatchesHost(model: { provider: string; baseUrl: string }, host: KnownHost): boolean { + const spec: HostClassSpec = KNOWN_HOSTS[host]; + if (spec.providers) { + for (const provider of spec.providers) { + if (model.provider === provider) return true; + } + } + if (spec.providerPrefixes) { + for (const prefix of spec.providerPrefixes) { + if (model.provider.startsWith(prefix)) return true; + } + } + return hostMatchesUrl(model.baseUrl, host); +} + +// --- Endpoint-shape predicates (URL path/verb shapes, not vendor hosts) --- + +/** Vertex AI express-mode OpenAI-compatible endpoint (`…/endpoints/openapi`). */ +export function isVertexExpressOpenAIUrl(baseUrl: string): boolean { + return baseUrl.includes("/endpoints/openapi"); +} + +/** Vertex AI Anthropic raw-predict endpoints (`:streamRawPredict` / `:rawPredict`). */ +export function isVertexRawPredictUrl(baseUrl: string): boolean { + return baseUrl.includes(":streamRawPredict") || baseUrl.includes(":rawPredict"); +} + +/** Azure OpenAI deployment-scoped path (`…/deployments//…`). */ +export function isAzureDeploymentsUrl(baseUrl: string): boolean { + return baseUrl.includes("/deployments/"); +} + +/** Alibaba DashScope consumer `compatible-mode` endpoint (rejects multimodal arrays for some text-only SKUs). */ +export function isDashscopeCompatibleModeUrl(baseUrl: string): boolean { + const normalized = baseUrl.toLowerCase(); + return ( + normalized.includes("dashscope") && normalized.includes("aliyuncs.com") && normalized.includes("/compatible-mode") + ); +} diff --git a/packages/catalog/src/identity/family.ts b/packages/catalog/src/identity/family.ts new file mode 100644 index 000000000..c812ffe4d --- /dev/null +++ b/packages/catalog/src/identity/family.ts @@ -0,0 +1,59 @@ +/** + * Model-family id predicates: the shared vocabulary for "is this id a member + * of family X" checks that gate wire-level behavior across hosts (a Kimi or + * DeepSeek model keeps its quirks no matter which OpenAI-compatible proxy + * serves it). Looser per-feature heuristics (e.g. stream-markup healing) + * deliberately keep their own patterns — only provably-shared matchers live + * here. + */ + +/** Kimi family ids in any namespace form (`moonshotai/kimi-*`, `kimi-k2.6`, `vendor/kimi.x`). */ +export function isKimiModelId(modelId: string): boolean { + return modelId.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(modelId); +} + +/** Kimi K2.6 specifically (preserved-thinking transport on Moonshot-native hosts). */ +export function isKimiK26ModelId(modelId: string): boolean { + return /(^|\/)kimi-k2\.6(?:[-:]|$)/i.test(modelId); +} + +/** Claude ids in any namespace form (`claude-*`, `vendor/claude.x`). */ +export function isClaudeModelId(modelId: string): boolean { + return /(^|\/)claude[-.]/i.test(modelId); +} + +/** `anthropic/`-namespaced ids (aggregator catalogs like OpenRouter). */ +export function isAnthropicNamespacedModelId(modelId: string): boolean { + return /(^|\/)anthropic\//i.test(modelId); +} + +/** Qwen family ids (substring match — Qwen SKUs have no stable prefix shape). */ +export function isQwenModelId(modelId: string): boolean { + return modelId.toLowerCase().includes("qwen"); +} + +/** DeepSeek family by id or display name (proxies often rename the id but keep the name). */ +export function isDeepseekModelIdOrName(value: string): boolean { + return value.toLowerCase().includes("deepseek"); +} + +/** Xiaomi MiMo family by id or display name. */ +export function isMimoModelIdOrName(value: string): boolean { + return value.toLowerCase().includes("mimo"); +} + +/** + * Adaptive thinking `display` is supported starting with Claude Opus 4.7 and + * Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet + * 4.6+) reject the field. + */ +export function supportsAdaptiveThinkingDisplay(modelId: string): boolean { + if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true; + // Bound the minor to non-date digits: bare dated ids like + // `claude-opus-4-20250514` (Opus 4.0) must not parse as minor=20250514. + const match = /claude-opus-(\d+)-(\d{1,2})(?!\d)/.exec(modelId); + if (!match) return false; + const major = Number(match[1]); + const minor = Number(match[2]); + return major > 4 || (major === 4 && minor >= 7); +} diff --git a/packages/catalog/src/identity/index.ts b/packages/catalog/src/identity/index.ts index cf16518b8..69c28db81 100644 --- a/packages/catalog/src/identity/index.ts +++ b/packages/catalog/src/identity/index.ts @@ -1,6 +1,7 @@ export * from "./bundled"; export * from "./classify"; export * from "./equivalence"; +export * from "./family"; export * from "./id"; export * from "./markers"; export * from "./priority"; diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index bd0e0bcce..6e99501bd 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -1,5 +1,6 @@ import { resolveOpenAICompat } from "./compat/openai"; import { Effort, THINKING_EFFORTS } from "./effort"; +import { modelMatchesHost } from "./hosts"; import { type AnthropicModel, bareModelId, @@ -353,7 +354,7 @@ function isOpenRouterAnthropicAdaptiveReasoningModel( model: ApiModel, ): boolean { if (model.api !== "openai-completions") return false; - if (model.provider !== "openrouter" && !model.baseUrl.includes("openrouter.ai")) return false; + if (!modelMatchesHost(model, "openrouter")) return false; return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6")); } diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 7bb55a532..7fd765abf 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -2349,11 +2349,15 @@ export interface GithubCopilotModelManagerConfig { fetch?: FetchImpl; } +const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus)-4([.-]|$)/; +const isCopilotResponsesModelId = (modelId: string): boolean => + modelId.startsWith("gpt-5") || modelId.startsWith("oswe"); + function inferCopilotApi(modelId: string): Api { - if (/^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId)) { + if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) { return "anthropic-messages"; } - if (modelId.startsWith("gpt-5") || modelId.startsWith("oswe")) { + if (isCopilotResponsesModelId(modelId)) { return "openai-responses"; } return "openai-completions"; @@ -2776,11 +2780,11 @@ const COPILOT_DEFAULT_RESOLUTION = { const COPILOT_API_RESOLUTION_RULES: readonly ApiResolutionRule[] = [ { - matches: modelId => /^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId), + matches: modelId => COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId), resolved: { api: "anthropic-messages", baseUrl: COPILOT_BASE_URL }, }, { - matches: modelId => modelId.startsWith("gpt-5") || modelId.startsWith("oswe"), + matches: isCopilotResponsesModelId, resolved: { api: "openai-responses", baseUrl: COPILOT_BASE_URL }, }, ]; diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index eedfeb2e9..0b974e0cf 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -182,6 +182,13 @@ export interface OpenAICompat { cacheControlFormat?: "anthropic" | undefined; /** Whether the provider supports the `strict` field in tool definitions. Default: auto-detected per provider/baseUrl (conservative for unknown providers). */ supportsStrictMode?: boolean; + /** + * Stream-watchdog idle-timeout floor in ms for slow reasoning hosts. + * Default: auto-detected (GLM coding-plan hosts, direct DeepSeek reasoning). + */ + streamIdleTimeoutMs?: number; + /** Whether the host honors `prompt_cache_retention: "24h"` on the Responses API. Default: auto-detected (api.openai.com). */ + supportsLongPromptCacheRetention?: boolean; /** Whether tool schemas must be sent either all strict or all non-strict. Undefined keeps the existing per-tool mixed behavior. */ toolStrictMode?: "all_strict" | "none"; } @@ -225,6 +232,22 @@ export interface AnthropicCompat { * When unset, auto-detected from the model id. Default: true. */ supportsForcedToolChoice?: boolean; + /** + * Include a non-standard `id` field (aliasing `tool_use_id`) on + * `tool_result` blocks. Z.AI's Anthropic-compatible proxy deserializes + * tool results into a class that reads `.id` (issue #814). Default: + * auto-detected (Z.AI hosts). + */ + requiresToolResultId?: boolean; + /** + * Replay unsigned `thinking` blocks from prior assistant turns as native + * thinking instead of demoting them to text. Official Anthropic enforces + * signature-based thinking-chain integrity, so unsigned blocks must stay + * text there; compatible reasoning endpoints (Z.AI, DeepSeek, …) emit + * unsigned blocks and expect them back as `type: "thinking"` (#2005). + * Default: auto-detected from provider/baseUrl and `model.reasoning`. + */ + replayUnsignedThinking?: boolean; } /** diff --git a/packages/catalog/test/hosts.test.ts b/packages/catalog/test/hosts.test.ts new file mode 100644 index 000000000..68b61b146 --- /dev/null +++ b/packages/catalog/test/hosts.test.ts @@ -0,0 +1,75 @@ +import { describe, expect, test } from "bun:test"; +import { + hostMatchesUrl, + isDashscopeCompatibleModeUrl, + isVertexExpressOpenAIUrl, + isVertexRawPredictUrl, + modelMatchesHost, +} from "@oh-my-pi/pi-catalog/hosts"; + +describe("hostMatchesUrl", () => { + test("matches OpenRouter URLs and rejects other or missing URLs", () => { + expect(hostMatchesUrl("https://openrouter.ai/api/v1", "openrouter")).toBe(true); + expect(hostMatchesUrl("https://api.openai.com/v1", "openrouter")).toBe(false); + expect(hostMatchesUrl(undefined, "openrouter")).toBe(false); + }); + + test("matches Z.AI URLs case-insensitively", () => { + expect(hostMatchesUrl("https://API.Z.AI/api/paas/v4", "zai")).toBe(true); + }); + + test("keeps DeepSeek direct host narrower than DeepSeek family", () => { + expect(hostMatchesUrl("https://api.deepseek.com/v1", "deepseekDirect")).toBe(true); + expect(hostMatchesUrl("https://api.deepseek.com/v1", "deepseekFamily")).toBe(true); + expect(hostMatchesUrl("https://chat.deepseek.com/api", "deepseekFamily")).toBe(true); + expect(hostMatchesUrl("https://chat.deepseek.com/api", "deepseekDirect")).toBe(false); + }); +}); + +describe("modelMatchesHost", () => { + test("matches by provider id, provider prefix, and URL-only Fireworks markers", () => { + expect(modelMatchesHost({ provider: "openrouter", baseUrl: "https://example.com/v1" }, "openrouter")).toBe(true); + expect(modelMatchesHost({ provider: "xiaomi-token-plan-eu", baseUrl: "https://example.com/v1" }, "xiaomi")).toBe( + true, + ); + expect(modelMatchesHost({ provider: "fireworks", baseUrl: "https://example.com/v1" }, "fireworks")).toBe(false); + expect( + modelMatchesHost({ provider: "custom", baseUrl: "https://api.fireworks.ai/inference/v1" }, "fireworks"), + ).toBe(true); + }); +}); + +describe("endpoint shape predicates", () => { + test("recognizes Vertex express OpenAI-compatible URLs", () => { + expect( + isVertexExpressOpenAIUrl( + "https://us-central1-aiplatform.googleapis.com/v1/projects/p/locations/us/endpoints/openapi", + ), + ).toBe(true); + expect( + isVertexExpressOpenAIUrl( + "https://us-central1-aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/google/models/gemini", + ), + ).toBe(false); + }); + + test("recognizes Vertex rawPredict and streamRawPredict URLs", () => { + expect( + isVertexRawPredictUrl( + "https://aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/anthropic/models/claude:rawPredict", + ), + ).toBe(true); + expect( + isVertexRawPredictUrl( + "https://aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/anthropic/models/claude:streamRawPredict", + ), + ).toBe(true); + }); + + test("requires all DashScope compatible-mode URL markers", () => { + expect(isDashscopeCompatibleModeUrl("https://dashscope.aliyuncs.com/compatible-mode/v1")).toBe(true); + expect(isDashscopeCompatibleModeUrl("https://example.aliyuncs.com/compatible-mode/v1")).toBe(false); + expect(isDashscopeCompatibleModeUrl("https://dashscope.example.com/compatible-mode/v1")).toBe(false); + expect(isDashscopeCompatibleModeUrl("https://dashscope.aliyuncs.com/api/v1")).toBe(false); + }); +}); diff --git a/packages/catalog/test/identity-family.test.ts b/packages/catalog/test/identity-family.test.ts new file mode 100644 index 000000000..9bfc7af6c --- /dev/null +++ b/packages/catalog/test/identity-family.test.ts @@ -0,0 +1,44 @@ +import { describe, expect, test } from "bun:test"; +import { + isClaudeModelId, + isKimiK26ModelId, + isKimiModelId, + supportsAdaptiveThinkingDisplay, +} from "@oh-my-pi/pi-catalog/identity"; + +describe("isKimiModelId", () => { + test("matches Kimi namespace and delimiter forms", () => { + expect(isKimiModelId("moonshotai/kimi-k2")).toBe(true); + expect(isKimiModelId("kimi-k2.6")).toBe(true); + expect(isKimiModelId("vendor/kimi.x")).toBe(true); + expect(isKimiModelId("akimbo-model")).toBe(false); + }); +}); + +describe("isKimiK26ModelId", () => { + test("matches Kimi K2.6 without accepting adjacent versions", () => { + expect(isKimiK26ModelId("kimi-k2.6")).toBe(true); + expect(isKimiK26ModelId("kimi-k2.6-thinking")).toBe(true); + expect(isKimiK26ModelId("kimi-k2.61")).toBe(false); + expect(isKimiK26ModelId("kimi-k2.5")).toBe(false); + }); +}); + +describe("isClaudeModelId", () => { + test("matches Claude namespace and delimiter forms", () => { + expect(isClaudeModelId("claude-sonnet-4-6")).toBe(true); + expect(isClaudeModelId("anthropic/claude.3")).toBe(true); + expect(isClaudeModelId("my-claudius")).toBe(false); + }); +}); + +describe("supportsAdaptiveThinkingDisplay", () => { + test("allows Claude Fable 5 and Opus 4.7 or newer only", () => { + expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true); + expect(supportsAdaptiveThinkingDisplay("claude-opus-4-7")).toBe(true); + expect(supportsAdaptiveThinkingDisplay("claude-opus-5-0")).toBe(true); + expect(supportsAdaptiveThinkingDisplay("claude-opus-4-6")).toBe(false); + expect(supportsAdaptiveThinkingDisplay("claude-opus-4-20250514")).toBe(false); + expect(supportsAdaptiveThinkingDisplay("claude-sonnet-4-6")).toBe(false); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8d4763b12..d2e8ca477 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,9 +1,9 @@ # Changelog ## [Unreleased] - ### Added +- Added `streamIdleTimeoutMs`, `supportsLongPromptCacheRetention`, `requiresToolResultId`, and `replayUnsignedThinking` to the OpenAI `compat` schema so custom model entries can configure those provider-specific capabilities - New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("need: 5h → 3 of 5 accounts"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing. - Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently. - npm installs now execute a prebundled single-file entry: `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks @@ -43,11 +43,11 @@ ### Fixed +- Fixed model-provider detection for append-only mode, authoritative Vertex endpoint checks, and upstream-routing selection by switching from URL substring checks to catalog host-matching helpers - Fixed pasting into the ask tool's "Other (type your own)" text box (and hook input/editor dialogs) on terminals with OSC 5522 enhanced paste (kitty protocol): the enhanced-paste focus routing only targets components exposing a `pasteText` hook, and the dialog wrappers had none, so the payload was stuffed into the main prompt editor hidden behind the dialog. `HookEditorComponent` and `HookInputComponent` now forward `pasteText` to their inner editor/input (pasting also resets the input dialog's timeout countdown like any keystroke). - Fixed auto-retry giving up after one attempt ("Provider requested Xms wait, exceeds retry.maxDelayMs") on a usage-limit 429 when every sibling account was only momentarily blocked: the retry delay now waits for the earliest sibling unblock when that comes sooner than the provider's multi-hour retry-after, so the next attempt picks up the recovered account instead of failing fast. - Fixed Hindsight `per-project-tagged` mental-model seeding so each project gets its own conventions/decisions models and session context only injects active-project or untagged models ([#2218](https://github.com/can1357/oh-my-pi/issues/2218)). -- Fixed Windows stdio MCP `.cmd` commands regressing from direct argv launches to a `cmd.exe /c` wrapper in v15.10.10, which made Codegraph MCP exit immediately with `Transport closed` ([#2220](https://github.com/can1357/oh-my-pi/issues/2220)). - +- Fixed Windows stdio MCP `.cmd` commands by wrapping batch shims with `cmd.exe /d /s /c` using the outer command quotes required by `cmd /s`, while preserving literal `%` and quoted JSON arguments for Codegraph MCP ([#2220](https://github.com/can1357/oh-my-pi/issues/2220)). - Fixed the bundled `explore` agent's `thinking-level: med` frontmatter — not a valid effort (`minimal`/`low`/`medium`/`high`/`xhigh`), so it silently parsed to undefined and the agent ran without its intended thinking level - Discovery context-file reads (`~/.claude`, `~/.cursor`, project trees, `@`-imports) now stat-gate to regular files before reading: a FIFO/socket/char device dropped where a context file is expected previously blocked startup forever on a read that can never see EOF. - Fixed the read tool's provider-visible `path` schema and docs so web URLs and internal URI targets (`omp://`, `issue://`, `pr://`, etc.) are advertised alongside local files ([#2215](https://github.com/can1357/oh-my-pi/issues/2215)). diff --git a/packages/coding-agent/src/config/append-only-context-mode.ts b/packages/coding-agent/src/config/append-only-context-mode.ts index 0efb1a8dd..71b4a53f0 100644 --- a/packages/coding-agent/src/config/append-only-context-mode.ts +++ b/packages/coding-agent/src/config/append-only-context-mode.ts @@ -1,3 +1,5 @@ +import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts"; + /** Provider metadata needed to resolve append-only context mode. */ export interface AppendOnlyContextModel { provider: string; @@ -5,19 +7,10 @@ export interface AppendOnlyContextModel { compat?: object; } -function isXiaomiHost(baseUrl: string): boolean { - try { - const host = new URL(baseUrl).hostname; - return host === "xiaomimimo.com" || host.endsWith(".xiaomimimo.com"); - } catch { - return false; - } -} - function shouldAutoEnableAppendOnlyContext(model: AppendOnlyContextModel | null | undefined): boolean { if (!model) return false; if (model.provider === "deepseek") return true; - if (isXiaomiHost(model.baseUrl)) return true; + if (hostMatchesUrl(model.baseUrl, "xiaomi")) return true; return !!model.compat && "supportsStore" in model.compat && model.compat.supportsStore === true; } diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 6ad89a15c..56ea0d020 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -2,6 +2,7 @@ import * as path from "node:path"; import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai/api-registry"; import type { Api, Context, Model, SimpleStreamOptions, ThinkingConfig } from "@oh-my-pi/pi-ai/types"; import type { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { isVertexExpressOpenAIUrl } from "@oh-my-pi/pi-catalog/hosts"; import { readModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { createModelManager, @@ -136,7 +137,7 @@ function isAuthoritativeProjectCatalogModel(model: Model): boolean { return ( model.provider === "google-vertex" && model.api === "openai-completions" && - model.baseUrl.includes("/endpoints/openapi") + isVertexExpressOpenAIUrl(model.baseUrl) ); } diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index 27bca36c9..7e6484692 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -16,6 +16,7 @@ import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import type { Api, Effort, KnownProvider, Model } from "@oh-my-pi/pi-ai"; +import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts"; import { buildModelProviderPriorityRank } from "@oh-my-pi/pi-catalog/identity"; import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking"; import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; @@ -169,14 +170,14 @@ function splitUpstreamRouting(pattern: string): { base: string; upstream: string /** OpenRouter and Vercel AI Gateway are the aggregators that honor per-request upstream routing. */ function supportsUpstreamRouting(model: Model): boolean { - return model.baseUrl.includes("openrouter.ai") || model.baseUrl.includes("ai-gateway.vercel.sh"); + return modelMatchesHost(model, "openrouter") || modelMatchesHost(model, "vercelAIGateway"); } /** Pin a resolved aggregator model to a single upstream provider via its compat routing block. */ function applyUpstreamRouting(model: Model, upstream: string): Model { const aggregatorModel = model as Model<"openai-completions">; const routing = { only: [upstream] }; - const compat = model.baseUrl.includes("ai-gateway.vercel.sh") + const compat = modelMatchesHost(model, "vercelAIGateway") ? { ...aggregatorModel.compat, vercelGatewayRouting: routing } : { ...aggregatorModel.compat, openRouterRouting: routing }; return { ...model, compat } as Model; diff --git a/packages/coding-agent/src/config/models-config-schema.ts b/packages/coding-agent/src/config/models-config-schema.ts index 1911651bb..0d83a263f 100644 --- a/packages/coding-agent/src/config/models-config-schema.ts +++ b/packages/coding-agent/src/config/models-config-schema.ts @@ -44,6 +44,11 @@ export const OpenAICompatSchema = z.object({ cacheControlFormat: z.enum(["anthropic"]).optional(), supportsStrictMode: z.boolean().optional(), toolStrictMode: z.enum(["all_strict", "none"]).optional(), + streamIdleTimeoutMs: z.number().positive().optional(), + supportsLongPromptCacheRetention: z.boolean().optional(), + // anthropic-messages compat flags (same `compat` slot, per-api interpretation) + requiresToolResultId: z.boolean().optional(), + replayUnsignedThinking: z.boolean().optional(), }); const EffortSchema = z.enum(["minimal", "low", "medium", "high", "xhigh"]); diff --git a/packages/coding-agent/test/usage-cli.test.ts b/packages/coding-agent/test/usage-cli.test.ts index 3f9efc915..65ee527a1 100644 --- a/packages/coding-agent/test/usage-cli.test.ts +++ b/packages/coding-agent/test/usage-cli.test.ts @@ -45,8 +45,8 @@ function makeReport(provider: string, email: string, limits: UsageReport["limits describe("buildRedactionMap", () => { it("masks everything past a two-char anchor when the anchor is unique", () => { const map = buildRedactionMap(["alpha@example.test", "bravo@example.test"]); - expect(map.get("alpha@example.test")).toBe("an*"); - expect(map.get("bravo@example.test")).toBe("ha*"); + expect(map.get("alpha@example.test")).toBe("al*"); + expect(map.get("bravo@example.test")).toBe("br*"); }); it("reveals a minimal middle-out differentiator instead of growing the prefix", () => { @@ -56,32 +56,32 @@ describe("buildRedactionMap", () => { // Masks must be pairwise distinct so accounts stay tellable-apart. expect(new Set(masks).size).toBe(masks.length); for (const mask of masks) { - // Never leak the local part the way prefix growth would ("can.boluk@*"). - expect(mask).not.toContain("boluk"); + // Never leak the whole local part the way prefix growth would ("dummy@*"). + expect(mask).not.toContain("dummy"); // anchor + at most a two-char differentiator. - expect(mask).toMatch(/^ca\*(.{1,2}\*)?$/); + expect(mask).toMatch(/^du\*(.{1,2}\*)?$/); } // The "89" account is distinguished by a digit only it contains. - expect(map.get("dum.my9@example.net")).toBe("ca*9*"); + expect(map.get("dum.my9@example.net")).toBe("du*9*"); }); it("gives duplicate identities the same mask", () => { const map = buildRedactionMap(["user@example.test", "user@example.test"]); expect(map.size).toBe(1); - expect(map.get("user@example.test")).toBe("me*"); + expect(map.get("user@example.test")).toBe("us*"); }); }); describe("computeProviderWindowStats", () => { - it("buckets by window duration, binds each account to its worst meter, and ceils the need", () => { + it("buckets by window duration, binds each account to its worst meter, and reports remaining capacity", () => { const reports = [ - makeReport("anthropic", "a@x", [ + makeReport("anthropic", "account-a@example.test", [ makeLimit({ id: "5h", usedFraction: 0.9, durationMs: FIVE_HOURS, windowId: "5h" }), makeLimit({ id: "7d", usedFraction: 0.1, durationMs: SEVEN_DAYS, windowId: "7d" }), // Tiered meter on the same window: higher burn must bind. makeLimit({ id: "7d-opus", usedFraction: 0.4, durationMs: SEVEN_DAYS, windowId: "7d", tier: "opus" }), ]), - makeReport("anthropic", "b@x", [ + makeReport("anthropic", "account-b@example.test", [ makeLimit({ id: "5h", usedFraction: 0.4, durationMs: FIVE_HOURS, windowId: "5h" }), makeLimit({ id: "7d", usedFraction: 0.2, durationMs: SEVEN_DAYS, windowId: "7d" }), ]), @@ -93,15 +93,15 @@ describe("computeProviderWindowStats", () => { expect(fiveHour.window).toBe("5h"); expect(fiveHour.accounts).toBe(2); expect(fiveHour.usedAccounts).toBeCloseTo(1.3); - expect(fiveHour.needed).toBe(2); + expect(fiveHour.remainingAccounts).toBeCloseTo(0.7); expect(sevenDay.window).toBe("7d"); expect(sevenDay.usedAccounts).toBeCloseTo(0.6); // 0.4 (opus binds) + 0.2 - expect(sevenDay.needed).toBe(1); + expect(sevenDay.remainingAccounts).toBeCloseTo(1.4); }); it("ignores limits without a resolvable fraction", () => { const reports = [ - makeReport("anthropic", "a@x", [ + makeReport("anthropic", "account-a@example.test", [ { id: "mystery", label: "mystery", @@ -116,23 +116,23 @@ describe("computeProviderWindowStats", () => { describe("collectUnreportedAccounts", () => { const accounts: UsageAccountIdentity[] = [ - { provider: "anthropic", type: "oauth", email: "seen@x.com" }, - { provider: "anthropic", type: "oauth", email: "missing@x.com" }, + { provider: "anthropic", type: "oauth", email: "seen@example.test" }, + { provider: "anthropic", type: "oauth", email: "missing@example.test" }, { provider: "anthropic", type: "api_key" }, { provider: "cerebras", type: "api_key" }, ]; - const reports = [makeReport("anthropic", "seen@x.com", [])]; + const reports = [makeReport("anthropic", "seen@example.test", [])]; it("flags providers without reports and identified accounts missing from reports", () => { const unreported = collectUnreportedAccounts(reports, accounts); expect(unreported).toEqual([ - { provider: "anthropic", type: "oauth", email: "missing@x.com" }, + { provider: "anthropic", type: "oauth", email: "missing@example.test" }, { provider: "cerebras", type: "api_key" }, ]); }); it("does not claim unattributable credentials are missing when reports carry no identity", () => { - const anonymous = [{ ...makeReport("anthropic", "seen@x.com", []), metadata: {} }]; + const anonymous = [{ ...makeReport("anthropic", "seen@example.test", []), metadata: {} }]; const unreported = collectUnreportedAccounts(anonymous, accounts); expect(unreported).toEqual([{ provider: "cerebras", type: "api_key" }]); }); @@ -140,33 +140,47 @@ describe("collectUnreportedAccounts", () => { describe("formatUsageBreakdown", () => { const reports = [ - makeReport("anthropic", "dum.my9@example.net", [ + makeReport("anthropic", "dummy.primary@example.test", [ makeLimit({ id: "Claude 5 Hour", usedFraction: 0.84, durationMs: FIVE_HOURS, windowId: "5h" }), ]), - makeReport("anthropic", "dummy@example.net", [ + makeReport("anthropic", "dummy.secondary@example.test", [ makeLimit({ id: "Claude 5 Hour", usedFraction: 0.5, durationMs: FIVE_HOURS, windowId: "5h" }), ]), ]; const accounts: UsageAccountIdentity[] = [ - { provider: "anthropic", type: "oauth", email: "dum.my9@example.net" }, - { provider: "anthropic", type: "oauth", email: "dummy@example.net" }, + { provider: "anthropic", type: "oauth", email: "dummy.primary@example.test" }, + { provider: "anthropic", type: "oauth", email: "dummy.secondary@example.test" }, { provider: "cerebras", type: "api_key" }, ]; it("renders every account: reported ones with limits, credential-only ones as no-data rows", () => { const text = stripVTControlCharacters(formatUsageBreakdown(reports, accounts, Date.now())); - expect(text).toContain("dum.my9@example.net"); + expect(text).toContain("dummy.primary@example.test"); expect(text).toContain("84.0% used"); expect(text).toContain("Cerebras"); expect(text).toContain("API key — no usage data"); - expect(text).toContain("need: 5h → 2 of 2 accounts"); + expect(text).toContain("capacity: 5h → 1.34/2 accounts used (0.66× quota left)"); + }); + + it("keeps near-exhausted capacity fractional instead of rounding it to an exact need", () => { + const nearReports = [ + makeReport("anthropic", "near-a@example.test", [ + makeLimit({ id: "Claude 5 Hour", usedFraction: 1, durationMs: FIVE_HOURS, windowId: "5h" }), + ]), + makeReport("anthropic", "near-b@example.test", [ + makeLimit({ id: "Claude 5 Hour", usedFraction: 0.99, durationMs: FIVE_HOURS, windowId: "5h" }), + ]), + ]; + const text = stripVTControlCharacters(formatUsageBreakdown(nearReports, [], Date.now())); + expect(text).toContain("capacity: 5h → 1.99/2 accounts used (0.01× quota left)"); + expect(text).not.toContain("need:"); }); it("redacts account labels through the provided map without leaking the originals", () => { - const redaction = buildRedactionMap(["dum.my9@example.net", "dummy@example.net"]); + const redaction = buildRedactionMap(["dummy.primary@example.test", "dummy.secondary@example.test"]); const text = stripVTControlCharacters(formatUsageBreakdown(reports, accounts, Date.now(), redaction)); - expect(text).not.toContain("dum.my9@example.net"); - expect(text).not.toContain("dummy@example.net"); + expect(text).not.toContain("dummy.primary@example.test"); + expect(text).not.toContain("dummy.secondary@example.test"); for (const mask of redaction.values()) expect(text).toContain(mask); }); }); diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index 92e942a66..c28c3f474 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -1,6 +1,10 @@ # Changelog ## [Unreleased] +### Fixed + +- Fixed embedding provider detection to match `openrouter` by URL host, so custom embedding endpoints are now recognized correctly instead of being misclassified by substring matching +- Fixed the check for OpenRouter base URLs so only true `openrouter` hosts are treated as non-custom ## [15.10.8] - 2026-06-09 ### Added diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index f4449e631..e13f574c8 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -40,6 +40,7 @@ }, "dependencies": { "@oh-my-pi/pi-ai": "catalog:", + "@oh-my-pi/pi-catalog": "catalog:", "@oh-my-pi/pi-utils": "catalog:", "fastembed": "catalog:", "lru-cache": "catalog:", diff --git a/packages/mnemopi/src/config.ts b/packages/mnemopi/src/config.ts index f4b3c19b0..458fb65b3 100644 --- a/packages/mnemopi/src/config.ts +++ b/packages/mnemopi/src/config.ts @@ -1,5 +1,6 @@ import { homedir } from "node:os"; import { join } from "node:path"; +import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts"; import { type Env, envBool, @@ -102,7 +103,7 @@ export function isApiEmbeddingModel(model = embeddingModel(), env: Env = process if (model.startsWith("openai/") || model.includes("text-embedding") || model.startsWith("text-embedding")) return true; const baseUrl = envString("MNEMOPI_EMBEDDING_API_URL", envString("OPENROUTER_BASE_URL", "", env), env); - if (baseUrl && !baseUrl.includes("openrouter.ai")) return true; + if (baseUrl && !hostMatchesUrl(baseUrl, "openrouter")) return true; return embeddingsViaApi(env); } @@ -110,7 +111,7 @@ export function apiEmbeddingsAvailable(env: Env = process.env): boolean { if (embeddingsDisabled(env)) return false; if (!isApiEmbeddingModel(embeddingModel(env), env)) return false; const baseUrl = envString("MNEMOPI_EMBEDDING_API_URL", envString("OPENROUTER_BASE_URL", "", env), env); - return Boolean(baseUrl && !baseUrl.includes("openrouter.ai")) || Boolean(embeddingApiKey(env)); + return Boolean(baseUrl && !hostMatchesUrl(baseUrl, "openrouter")) || Boolean(embeddingApiKey(env)); } export function workingMemoryMaxItems(env: Env = process.env): number { diff --git a/packages/mnemopi/src/core/embeddings.ts b/packages/mnemopi/src/core/embeddings.ts index 07756fbf7..b170f2ea1 100644 --- a/packages/mnemopi/src/core/embeddings.ts +++ b/packages/mnemopi/src/core/embeddings.ts @@ -1,4 +1,5 @@ import { mkdirSync } from "node:fs"; +import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts"; import { $env, $flag, @@ -155,7 +156,7 @@ export function isApiModel(modelName: string): boolean { } const active = activeEmbeddingOptions(); const baseUrl = active?.apiUrl ?? ($env.MNEMOPI_EMBEDDING_API_URL || $env.OPENROUTER_BASE_URL); - if (baseUrl !== undefined && baseUrl !== "" && !baseUrl.includes("openrouter.ai")) { + if (baseUrl !== undefined && baseUrl !== "" && !hostMatchesUrl(baseUrl, "openrouter")) { return true; } return $flag("MNEMOPI_EMBEDDINGS_VIA_API"); @@ -246,7 +247,7 @@ async function getLocalModel(): Promise { async function embedApi(texts: readonly string[]): Promise { const baseUrl = embeddingBaseUrl(); - const isCustom = !baseUrl.includes("openrouter.ai"); + const isCustom = !hostMatchesUrl(baseUrl, "openrouter"); const apiKey = embeddingApiKey(); if (!isCustom && apiKey === "") { return null; @@ -335,7 +336,7 @@ export async function available(): Promise { } if (isApiModel(defaultModel())) { const baseUrl = active?.apiUrl ?? ($env.MNEMOPI_EMBEDDING_API_URL || $env.OPENROUTER_BASE_URL); - if (baseUrl !== undefined && baseUrl !== "" && !baseUrl.includes("openrouter.ai")) { + if (baseUrl !== undefined && baseUrl !== "" && !hostMatchesUrl(baseUrl, "openrouter")) { return true; } return embeddingApiKey() !== "";