feat(cross-cutting): centralized model compatibility handling with catalog resolvers

- Added catalog-level host/model predicates and compat resolvers.
- Extended compatibility types and model schema with timeout and replay flags.
- Replaced provider-specific heuristics with shared resolver-based checks.
- Updated host/identity and resolver tests to validate the new behavior.
This commit is contained in:
can1357
2026-06-10 04:30:19 +02:00
parent 7572e4f04c
commit e14a63da6d
33 changed files with 740 additions and 396 deletions
+1
View File
@@ -123,6 +123,7 @@
},
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"fastembed": "catalog:",
"lru-cache": "catalog:",
+1 -16
View File
@@ -8,6 +8,7 @@
*/
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils";
@@ -852,22 +853,6 @@ function buildAdditionalModelRequestFields(
return result;
}
/**
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
* 4.6+) reject the field. Bedrock model ids are prefixed with region/inference-
* profile slugs (e.g. `eu.anthropic.claude-opus-4-7-...`); the regex matches
* the Claude model fragment regardless of prefix.
*/
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
if (!match) return false;
const major = Number(match[1]);
const minor = Number(match[2]);
return major > 4 || (major === 4 && minor >= 7);
}
/**
* Bedrock's wire format expects the image as `{ source: { bytes: <base64-string> }, format }`.
* The caller already passes base64-encoded data, so no decode/re-encode round-trip is needed.
+19 -110
View File
@@ -2,12 +2,9 @@ import * as nodeCrypto from "node:crypto";
import * as fs from "node:fs";
import { scheduler } from "node:timers/promises";
import * as tls from "node:tls";
import {
hasOpus47ApiRestrictions,
isAnthropicFableOrMythosModel,
mapEffortToAnthropicAdaptiveEffort,
supportsMidConversationSystemMessages,
} from "@oh-my-pi/pi-catalog/model-thinking";
import { isOfficialAnthropicApiUrl, resolveAnthropicCompat } from "@oh-my-pi/pi-catalog/compat/anthropic";
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import { isAnthropicOAuthToken } from "@oh-my-pi/pi-catalog/utils";
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
@@ -181,16 +178,6 @@ function isClaudeCodeClientUserAgent(userAgent: string | undefined): userAgent i
return userAgent.toLowerCase().startsWith("claude-cli");
}
export function isAnthropicApiBaseUrl(baseUrl?: string): boolean {
if (!baseUrl) return true;
try {
const url = new URL(baseUrl);
return url.protocol.toLowerCase() === "https:" && url.hostname.toLowerCase() === "api.anthropic.com";
} catch {
return false;
}
}
const sharedHeaders = {
"Accept-Encoding": "gzip, deflate, br, zstd",
Connection: "keep-alive",
@@ -263,7 +250,7 @@ export function buildAnthropicHeaders(options: AnthropicHeaderOptions): Record<s
"x-client-request-id": nodeCrypto.randomUUID(),
"User-Agent": userAgent,
};
} else if (!isAnthropicApiBaseUrl(options.baseUrl)) {
} else if (!isOfficialAnthropicApiUrl(options.baseUrl)) {
return {
...modelHeaders,
Accept: acceptHeader,
@@ -310,22 +297,6 @@ type AnthropicOutputConfig = NonNullable<MessageCreateParamsStreaming["output_co
const ANTHROPIC_STOP_SEQUENCES_MAX = 4;
let warnedStopSequencesTrim = false;
/**
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
* 4.6+) reject the field.
*/
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
// Bound the minor to non-date digits: bare dated ids like
// `claude-opus-4-20250514` (Opus 4.0) must not parse as minor=20250514.
const match = /claude-opus-(\d+)-(\d{1,2})(?!\d)/.exec(modelId);
if (!match) return false;
const major = Number(match[1]);
const minor = Number(match[2]);
return major > 4 || (major === 4 && minor >= 7);
}
const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
type AnthropicProviderSessionState = ProviderSessionState & {
@@ -450,7 +421,9 @@ function getCacheControl(
return { retention };
}
const ttl =
retention === "long" && isAnthropicApiBaseUrl(baseUrl) && getAnthropicCompat(model).supportsLongCacheRetention
retention === "long" &&
isOfficialAnthropicApiUrl(baseUrl) &&
resolveAnthropicCompat(model).supportsLongCacheRetention
? "1h"
: undefined;
return {
@@ -1146,7 +1119,7 @@ function parseAnthropicCustomHeaders(rawHeaders: string | undefined): Record<str
export function resolveAnthropicCustomHeadersForBaseUrl(
baseUrl: string | undefined,
): Record<string, string> | undefined {
if (!isFoundryEnabled() && isAnthropicApiBaseUrl(baseUrl)) return undefined;
if (!isFoundryEnabled() && isOfficialAnthropicApiUrl(baseUrl)) return undefined;
return parseAnthropicCustomHeaders($env.ANTHROPIC_CUSTOM_HEADERS);
}
@@ -1404,24 +1377,6 @@ async function* observeDecodedAnthropicSdkEvents(
}
}
function getAnthropicCompat(
model: Model<"anthropic-messages">,
): Required<NonNullable<Model<"anthropic-messages">["compat"]>> {
return {
disableStrictTools: model.compat?.disableStrictTools ?? false,
disableAdaptiveThinking: model.compat?.disableAdaptiveThinking ?? false,
supportsEagerToolInputStreaming: model.compat?.supportsEagerToolInputStreaming ?? true,
supportsLongCacheRetention: model.compat?.supportsLongCacheRetention ?? true,
supportsMidConversationSystem:
model.compat?.supportsMidConversationSystem ??
// First-party Claude API only. Bedrock/Vertex/Foundry and other
// Anthropic-compatible proxies reject the role; gate auto-detection on
// the canonical api.anthropic.com host plus a supported model id.
(isAnthropicApiBaseUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id)),
supportsForcedToolChoice: model.compat?.supportsForcedToolChoice ?? !isAnthropicFableOrMythosModel(model.id),
};
}
const PROVIDER_MAX_RETRIES = 3;
const PROVIDER_BASE_DELAY_MS = 2000;
@@ -1626,7 +1581,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
const sendsAdaptiveEffortPin =
options?.thinkingEnabled === false &&
model.thinking?.mode === "anthropic-adaptive" &&
!getAnthropicCompat(model).disableAdaptiveThinking;
!resolveAnthropicCompat(model).disableAdaptiveThinking;
if (
model.reasoning &&
(options?.thinkingEnabled || sendsAdaptiveEffortPin) &&
@@ -1635,7 +1590,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
extraBetas.push(effortBeta);
}
if (
getAnthropicCompat(model).supportsMidConversationSystem &&
resolveAnthropicCompat(model).supportsMidConversationSystem &&
!extraBetas.includes(midConversationSystemBeta)
) {
// convertAnthropicMessages may upgrade developer turns to the
@@ -2332,7 +2287,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
isOAuth,
claudeCodeSessionId,
} = args;
const compat = getAnthropicCompat(model);
const compat = resolveAnthropicCompat(model);
const needsInterleavedBeta = interleavedThinking && !supportsAdaptiveThinkingDisplay(model.id);
const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming;
const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey);
@@ -2443,7 +2398,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
const authorizationHeader = getHeaderCaseInsensitive(defaultHeaders, "Authorization");
const shouldSuppressClientApiKey =
!oauthToken &&
!isAnthropicApiBaseUrl(baseUrl) &&
!isOfficialAnthropicApiUrl(baseUrl) &&
typeof authorizationHeader === "string" &&
/^Bearer\s+/i.test(authorizationHeader);
@@ -2777,7 +2732,7 @@ function buildParams(
context.tools,
isOAuthToken,
disableStrictTools || model.provider === "github-copilot",
getAnthropicCompat(model).supportsEagerToolInputStreaming,
resolveAnthropicCompat(model).supportsEagerToolInputStreaming,
);
} else if (isOAuthToken) {
tools = [];
@@ -2800,7 +2755,7 @@ function buildParams(
if (options?.thinkingEnabled) {
const mode = model.thinking?.mode;
const effort = resolveAnthropicAdaptiveEffort(model, options);
const compat = getAnthropicCompat(model);
const compat = resolveAnthropicCompat(model);
if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" };
// Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking
@@ -2823,7 +2778,7 @@ function buildParams(
if (mode === "anthropic-budget-effort" && effort) outputConfigEffort = effort;
}
} else if (options?.thinkingEnabled === false) {
const compat = getAnthropicCompat(model);
const compat = resolveAnthropicCompat(model);
if (model.thinking?.mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
// Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject
// `thinking.type: "disabled"` — adaptive thinking cannot be switched off.
@@ -2912,7 +2867,7 @@ function buildParams(
// request succeeds; the tool stays available and the caller's prompt steers
// the model toward it.
const choiceType = params.tool_choice?.type;
if ((choiceType === "any" || choiceType === "tool") && !getAnthropicCompat(model).supportsForcedToolChoice) {
if ((choiceType === "any" || choiceType === "tool") && !resolveAnthropicCompat(model).supportsForcedToolChoice) {
params.tool_choice = { type: "auto" };
}
}
@@ -2926,52 +2881,6 @@ function buildParams(
return params;
}
/**
* Z.AI's Anthropic-compatible proxy at `api.z.ai/api/anthropic` deserializes
* tool_result blocks into a Python class that accesses `.id`, even though
* Anthropic's standard tool_result schema only carries `tool_use_id`. Detect
* that endpoint so we can emit the non-standard alias for it without
* polluting requests to api.anthropic.com or other compatible proxies.
* See: https://github.com/can1357/oh-my-pi/issues/814
*/
function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean {
if (model.provider === "zai") return true;
const baseUrl = model.baseUrl;
if (!baseUrl) return false;
try {
return new URL(baseUrl).hostname.toLowerCase() === "api.z.ai";
} catch {
return false;
}
}
/**
* Returns true when unsigned `thinking` blocks from prior assistant turns should
* be replayed as Anthropic-native thinking instead of demoted to text.
*
* Official Anthropic (matched via `isAnthropicApiBaseUrl`, which intentionally
* treats a missing baseUrl as official since `resolveAnthropicBaseUrl` routes
* it to `https://api.anthropic.com`) enforces signature-based thinking-chain
* integrity, so unsigned blocks must remain text there. Anthropic-compatible
* reasoning endpoints commonly emit unsigned thinking blocks while still
* expecting them back as `type: "thinking"` on continuation; demoting them
* loses the model's reasoning chain and can destabilize the next tool-call
* arguments (#2005). Known non-signing hosts are also preserved for
* compatibility.
*/
function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">, baseUrl: string | undefined): boolean {
if (model.provider === "zai" || model.provider === "deepseek") return true;
if (baseUrl) {
try {
const hostname = new URL(baseUrl).hostname.toLowerCase();
if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true;
} catch {
// Fall through to the protocol-level reasoning rule below.
}
}
return model.reasoning && !isAnthropicApiBaseUrl(baseUrl);
}
function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam {
const block: ContentBlockParam = {
type: "tool_result",
@@ -2979,7 +2888,7 @@ function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResul
content: convertContentBlocks(msg.content, model.input.includes("image")),
is_error: msg.isError,
};
if (isZaiAnthropicEndpoint(model)) {
if (resolveAnthropicCompat(model).requiresToolResultId) {
// Z.AI workaround (issue #814): include `id` aliased to `tool_use_id`.
(block as unknown as Record<string, unknown>).id = msg.toolCallId;
}
@@ -3092,7 +3001,7 @@ export function convertAnthropicMessages(
}
if (block.thinking.trim().length === 0) continue;
if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) {
if (shouldReplayUnsignedThinking(model, baseUrl)) {
if (resolveAnthropicCompat(model, baseUrl).replayUnsignedThinking) {
blocks.push({
type: "thinking",
thinking: block.thinking.toWellFormed(),
@@ -3170,7 +3079,7 @@ export function convertAnthropicMessages(
// never consecutive. Requiring the next param to be `assistant` (or absent)
// covers both the "followed by assistant / last" and "no consecutive system"
// constraints. Anything that does not qualify stays a `user` message.
if (developerParamIndices.length > 0 && getAnthropicCompat(model).supportsMidConversationSystem) {
if (developerParamIndices.length > 0 && resolveAnthropicCompat(model).supportsMidConversationSystem) {
for (const idx of developerParamIndices) {
const followsUser = idx > 0 && params[idx - 1]?.role === "user";
const next = params[idx + 1];
@@ -1,3 +1,4 @@
import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
import { AzureOpenAI, APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai";
import type {
@@ -31,7 +32,7 @@ import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schem
import { createSdkStreamRequestOptions } from "../utils/sdk-stream-timeout";
import { notifyRawSseEvent } from "../utils/sse-debug";
import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
import { getOpenAIResponsesCacheSessionId, supportsDeveloperRole } from "./openai-responses";
import { getOpenAIResponsesCacheSessionId } from "./openai-responses";
import {
appendResponsesToolResultMessages,
applyCommonResponsesSamplingParams,
@@ -337,7 +338,10 @@ function convertMessages(
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
if (systemPrompts.length > 0) {
const role = model.reasoning && supportsDeveloperRole(resolvedBaseUrl ?? model) ? "developer" : "system";
const role =
model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole
? "developer"
: "system";
for (const systemPrompt of systemPrompts) {
messages.push({ role, content: systemPrompt });
}
+10 -46
View File
@@ -1,6 +1,8 @@
import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id";
import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts";
import { isDeepseekModelIdOrName, isKimiModelId } from "@oh-my-pi/pi-catalog/identity";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
@@ -390,44 +392,6 @@ function getTrailingPartialDeepseekToken(text: string): string {
const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
"OpenAI completions stream timed out while waiting for the first event";
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
// DeepSeek V4 reasoning models on the official api.deepseek.com emit no SSE
// bytes while the model finishes its private chain-of-thought, which routinely
// takes longer than the generic 100s first-event floor under load (issue
// #2177). Mirror the GLM coding-plan widening: a 5-minute idle floor lifts the
// first-event watchdog (it floors at idle) without changing the runtime
// streaming behavior, so reasoning warm-ups stop aborting and retrying.
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
function isDirectDeepseekReasoningModel(model: Model<"openai-completions">): boolean {
if (!model.reasoning) return false;
if (model.provider === "deepseek") return true;
return model.baseUrl.toLowerCase().includes("api.deepseek.com");
}
/** Returns the widened OpenAI stream watchdog floor for slow reasoning models hosted on OpenAI-compatible endpoints. */
export function getOpenAICompletionsStreamIdleTimeoutFallbackMs(
model: Model<"openai-completions">,
): number | undefined {
if (GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) {
if (model.provider === "zhipu-coding-plan" || model.provider === "zai")
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
const baseUrl = model.baseUrl.toLowerCase();
if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) {
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
}
}
if (isDirectDeepseekReasoningModel(model)) {
return DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS;
}
return undefined;
}
async function* observeDecodedOpenAICompletionChunks(
chunks: AsyncIterable<ChatCompletionChunk>,
observer: (event: RawSseEvent) => void,
@@ -468,7 +432,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
try {
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
const idleTimeoutFallbackMs = getOpenAICompletionsStreamIdleTimeoutFallbackMs(model);
const idleTimeoutFallbackMs = resolveOpenAICompat(model).streamIdleTimeoutMs;
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs);
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
@@ -576,7 +540,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
// though tool calls are also surfaced structurally. Strip the leaked markers
// so users don't see raw `<|...|>` tokens.
const stripDeepseekChatTemplateTokens =
/deepseek/i.test(model.id) && (model.provider === "nvidia" || model.provider === "deepseek");
isDeepseekModelIdOrName(model.id) && (model.provider === "nvidia" || model.provider === "deepseek");
type ToolCallStreamBlock = ToolCall & {
partialArgs?: string | Record<string, unknown>;
streamIndex?: number;
@@ -1252,8 +1216,8 @@ function buildParams(
compat.allowsSyntheticReasoningContentForToolCalls = false;
compat.reasoningContentField = "reasoning_content";
}
const isKimiModelId = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
const isOpenRouter = model.baseUrl.includes("openrouter.ai");
const isKimiFamilyModel = isKimiModelId(model.id);
const isOpenRouter = modelMatchesHost(model, "openrouter");
const messages = convertMessages(model, context, compat);
maybeAddAnthropicCacheControl(compat, messages);
const supportsReasoningParams = model.provider !== "github-copilot";
@@ -1266,14 +1230,14 @@ function buildParams(
// before the final answer. Always send max_tokens — match the same
// Kimi-family regex used by the compat detector.
// Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts.
const requestedMaxTokens = options?.maxTokens ?? (isKimiModelId ? model.maxTokens : undefined);
const requestedMaxTokens = options?.maxTokens ?? (isKimiFamilyModel ? model.maxTokens : undefined);
// OpenRouter fans out to upstreams whose output caps differ from the catalog
// value (which tracks the highest-cap provider). A max_tokens above the routed
// upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras
// GLM-4.7, ~40k) for a higher-cap one, defeating `provider.order`/`only`. Omit
// it for OpenRouter so each upstream self-caps and routing is honored. Kimi is
// exempt — it derives TPM rate limits from max_tokens (see above).
const omitMaxTokensForRouting = isOpenRouter && !isKimiModelId;
const omitMaxTokensForRouting = isOpenRouter && !isKimiFamilyModel;
const effectiveMaxTokens =
requestedMaxTokens === undefined || omitMaxTokensForRouting
? undefined
@@ -1442,12 +1406,12 @@ function buildParams(
}
// OpenRouter provider routing preferences
if (model.baseUrl.includes("openrouter.ai") && compat.openRouterRouting) {
if (modelMatchesHost(model, "openrouter") && compat.openRouterRouting) {
params.provider = compat.openRouterRouting;
}
// Vercel AI Gateway provider routing preferences
if (model.baseUrl.includes("ai-gateway.vercel.sh") && model.compat?.vercelGatewayRouting) {
if (modelMatchesHost(model, "vercelAIGateway") && model.compat?.vercelGatewayRouting) {
const routing = model.compat.vercelGatewayRouting;
if (routing.only || routing.order) {
const gatewayOptions: Record<string, string[]> = {};
+14 -47
View File
@@ -1,3 +1,5 @@
import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
import OpenAI, { APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai";
@@ -10,7 +12,6 @@ import type {
import { getEnvApiKey } from "../stream";
import type {
AssistantMessage,
CacheRetention,
Context,
FetchImpl,
MessageAttribution,
@@ -69,20 +70,6 @@ import {
} from "./openai-responses-shared";
import { transformMessages } from "./transform-messages";
/**
* Get prompt cache retention based on cacheRetention and base URL.
* Only applies to direct OpenAI API calls (api.openai.com).
*/
function getPromptCacheRetention(baseUrl: string, cacheRetention: CacheRetention): "24h" | undefined {
if (cacheRetention !== "long") {
return undefined;
}
if (baseUrl.includes("api.openai.com")) {
return "24h";
}
return undefined;
}
export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined {
if (!sessionId || sessionId.length === 0) return undefined;
const wellFormed = sessionId.toWellFormed();
@@ -442,13 +429,14 @@ function buildParams(
): OpenAIResponsesSamplingParams {
const strictResponsesPairing =
options?.strictResponsesPairing ??
(isAzureOpenAIBaseUrl(model.baseUrl ?? "") || model.provider === "github-copilot");
(hostMatchesUrl(model.baseUrl ?? "", "azureOpenAI") || model.provider === "github-copilot");
const messages = convertConversationMessages(model, context, strictResponsesPairing, providerSessionState, options);
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
let systemInstructions: string | undefined;
if (systemPrompts.length > 0) {
const needsDeveloperRole = model.reasoning && supportsDeveloperRole(resolvedBaseUrl ?? model);
const needsDeveloperRole =
model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole;
if (needsDeveloperRole) {
// Reasoning models on known OpenAI-compatible endpoints require the
// `developer` role. Send all system prompts inline in `input`.
@@ -472,7 +460,10 @@ function buildParams(
stream: true,
prompt_cache_key: promptCacheKey,
prompt_cache_retention: promptCacheKey
? getPromptCacheRetention(resolvedBaseUrl ?? model.baseUrl, cacheRetention)
? cacheRetention === "long" &&
resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsLongPromptCacheRetention
? "24h"
: undefined
: undefined,
store: false,
stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined,
@@ -485,7 +476,11 @@ function buildParams(
// `StreamOptions.frequencyPenalty` is intentionally dropped for this provider.
if (context.tools) {
params.tools = convertTools(context.tools, supportsStrictMode(model), model);
params.tools = convertTools(
context.tools,
resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsStrictMode,
model,
);
if (options?.toolChoice) {
params.tool_choice = mapOpenAIResponsesToolChoiceForTools(options.toolChoice, context.tools, model);
}
@@ -528,34 +523,6 @@ function mapReasoningEffort(
return reasoningEffortMap?.[effort] ?? effort;
}
function isAzureOpenAIBaseUrl(baseUrl: string): boolean {
return baseUrl.includes(".openai.azure.com") || baseUrl.includes("azure.com/openai");
}
function supportsStrictMode(model: Model<"openai-responses">): boolean {
if (model.provider === "openai" || model.provider === "azure" || model.provider === "github-copilot") return true;
const baseUrl = model.baseUrl.toLowerCase();
return (
baseUrl.includes("api.openai.com") ||
baseUrl.includes(".openai.azure.com") ||
baseUrl.includes("models.inference.ai.azure.com")
);
}
export function supportsDeveloperRole(modelOrBaseUrl: Pick<Model, "provider" | "baseUrl"> | string): boolean {
const baseUrl =
typeof modelOrBaseUrl === "string" ? modelOrBaseUrl.toLowerCase() : (modelOrBaseUrl.baseUrl ?? "").toLowerCase();
return (
baseUrl.includes("api.openai.com") ||
baseUrl.includes(".openai.azure.com") ||
baseUrl.includes("azure.com/openai") ||
baseUrl.includes("models.inference.ai.azure.com") ||
baseUrl.includes("githubcopilot.com") ||
baseUrl.includes("copilot-api.")
);
}
function convertConversationMessages(
model: Model<"openai-responses">,
context: Context,
+5 -3
View File
@@ -1,3 +1,6 @@
import { isDashscopeCompatibleModeUrl } from "@oh-my-pi/pi-catalog/hosts";
import { isQwenModelId } from "@oh-my-pi/pi-catalog/identity";
import type { ImageContent, Model, TextContent } from "../types";
export const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]";
@@ -42,11 +45,10 @@ export function joinTextWithImagePlaceholder(text: string, omittedImages: boolea
* provider (issue #1859) can't drive the request into an unrecoverable 400.
*/
export function isDashscopeCompatibleModeTextOnlyQwen(model: Model<"openai-completions">): boolean {
const baseUrl = model.baseUrl.toLowerCase();
if (!baseUrl.includes("dashscope") || !baseUrl.includes("aliyuncs.com") || !baseUrl.includes("/compatible-mode")) {
if (!isDashscopeCompatibleModeUrl(model.baseUrl)) {
return false;
}
const id = model.id.toLowerCase();
if (!id.includes("qwen")) return false;
if (!isQwenModelId(model.id)) return false;
return /\bqwen(?:[\d.]+)?-max\b/.test(id) || /\bqwen(?:[\d.]+)?-coder\b/.test(id);
}
+4 -7
View File
@@ -1,4 +1,5 @@
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
import { isVertexExpressOpenAIUrl, isVertexRawPredictUrl } from "@oh-my-pi/pi-catalog/hosts";
import {
mapEffortToAnthropicAdaptiveEffort,
mapEffortToGoogleThinkingLevel,
@@ -65,8 +66,8 @@ import { withRequestDebugFetch } from "./utils/request-debug";
function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
return (
model.provider === "google-vertex" &&
((model.api === "openai-completions" && model.baseUrl.includes("/endpoints/openapi")) ||
(model.api === "anthropic-messages" && model.baseUrl.includes(":streamRawPredict")))
((model.api === "openai-completions" && isVertexExpressOpenAIUrl(model.baseUrl)) ||
(model.api === "anthropic-messages" && isVertexRawPredictUrl(model.baseUrl)))
);
}
@@ -78,7 +79,7 @@ function createVertexAuthenticatedFetch(options: StreamOptions | undefined): Fet
headers.set("Authorization", `Bearer ${token}`);
const rewritten = resolveVertexRequest(input);
const url = rewritten instanceof Request ? rewritten.url : rewritten.toString();
if (isVertexAnthropicRawPredict(url)) {
if (isVertexRawPredictUrl(url)) {
const bodyText = await readVertexRequestBody(rewritten, init);
const transformed = transformVertexAnthropicBody(bodyText);
return baseFetch(url, {
@@ -93,10 +94,6 @@ function createVertexAuthenticatedFetch(options: StreamOptions | undefined): Fet
return Object.assign(vertexFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {});
}
function isVertexAnthropicRawPredict(url: string): boolean {
return url.includes(":streamRawPredict") || url.includes(":rawPredict");
}
async function readVertexRequestBody(input: string | URL | Request, init: RequestInit | undefined): Promise<string> {
if (input instanceof Request) return input.clone().text();
const body = init?.body;
@@ -13,6 +13,8 @@
* deltas for thinking blocks, and holds partial tags across chunk boundaries.
*/
import { isDeepseekModelIdOrName } from "@oh-my-pi/pi-catalog/identity";
import { parseJsonWithRepair } from "./json-parse";
const KIMI_SECTION_BEGIN = "<|tool_calls_section_begin|>";
@@ -622,7 +624,7 @@ export function modelMayLeakKimiToolCalls(provider: string, modelId: string): bo
/** Cheap model/provider gate for DeepSeek DSML envelope leaks. */
export function modelMayLeakDsmlToolCalls(provider: string, modelId: string): boolean {
if (!/deepseek/i.test(modelId)) return false;
if (!isDeepseekModelIdOrName(modelId)) return false;
return (
provider === "ollama" ||
provider === "ollama-cloud" ||
@@ -161,7 +161,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
});
it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => {
// `isAnthropicApiBaseUrl(undefined) === true` because the actual HTTP
// `isOfficialAnthropicApiUrl(undefined) === true` because the actual HTTP
// dispatch falls back to https://api.anthropic.com. Same-id custom
// overrides that only tweak model metadata (no baseUrl override) must
// not regress to native-thinking replay against the first-party API.
@@ -1,10 +1,10 @@
import { describe, expect, it } from "bun:test";
import {
getOpenAICompletionsStreamIdleTimeoutFallbackMs,
isOpenAICompletionsProgressChunk,
streamOpenAICompletions,
} from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
import { resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
const openAICompletionsModel = {
@@ -78,7 +78,7 @@ function createKeepaliveOnlyCompletionsResponse(modelId: string, signal: AbortSi
});
}
describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
describe("resolveOpenAICompat stream idle timeout", () => {
it("widens GLM 5.1 coding-plan stream watchdogs", () => {
const model = {
...openAICompletionsModel,
@@ -88,7 +88,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000);
});
it("also widens custom Z.AI OpenAI-compatible GLM 5.1 endpoints", () => {
@@ -100,7 +100,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
baseUrl: "https://api.z.ai/api/coding/paas/v4",
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000);
});
it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => {
@@ -113,7 +113,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
reasoning: true,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000);
});
it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => {
@@ -126,7 +126,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
reasoning: true,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000);
});
it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => {
@@ -139,7 +139,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
reasoning: false,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined();
});
it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => {
@@ -152,11 +152,11 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
reasoning: true,
} satisfies Model<"openai-completions">;
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined();
});
it("keeps ordinary OpenAI-compatible models on the global timeout", () => {
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined();
expect(resolveOpenAICompat(openAICompletionsModel).streamIdleTimeoutMs).toBeUndefined();
});
});
@@ -1,78 +1,78 @@
import { describe, expect, it } from "bun:test";
import { supportsDeveloperRole } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { Model } from "@oh-my-pi/pi-ai/types";
describe("supportsDeveloperRole", () => {
import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
describe("resolveOpenAIResponsesCompat supportsDeveloperRole", () => {
it("returns true for openai provider with official API base URL", () => {
const model = { provider: "openai", baseUrl: "https://api.openai.com/v1" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "openai", baseUrl: "https://api.openai.com/v1" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns false for openai provider with custom proxy base URL", () => {
const model = { provider: "openai", baseUrl: "https://my-proxy.example.com/v1" } as Model;
expect(supportsDeveloperRole(model)).toBe(false);
const model = { provider: "openai", baseUrl: "https://my-proxy.example.com/v1" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
});
it("returns true for github-copilot provider", () => {
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns false for github-copilot provider with custom proxy base URL", () => {
const model = { provider: "github-copilot", baseUrl: "https://proxy.example.com/v1" } as Model;
expect(supportsDeveloperRole(model)).toBe(false);
const model = { provider: "github-copilot", baseUrl: "https://proxy.example.com/v1" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
});
it("returns true for Azure OpenAI base URL", () => {
const model = { provider: "azure-openai", baseUrl: "https://my-resource.openai.azure.com/openai" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "azure-openai", baseUrl: "https://my-resource.openai.azure.com/openai" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns true for Azure AI Inference base URL", () => {
const model = {
provider: "azure-openai",
baseUrl: "https://models.inference.ai.azure.com/v1/chat/completions",
} as Model;
expect(supportsDeveloperRole(model)).toBe(true);
};
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns true for api.openai.com base URL", () => {
const model = { provider: "custom", baseUrl: "https://api.openai.com/v1/chat/completions" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "custom", baseUrl: "https://api.openai.com/v1/chat/completions" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns false for generic third-party provider", () => {
const model = { provider: "custom", baseUrl: "https://api.example.com/v1" } as Model;
expect(supportsDeveloperRole(model)).toBe(false);
const model = { provider: "custom", baseUrl: "https://api.example.com/v1" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
});
it("returns false for local/localhost endpoints", () => {
const model = { provider: "custom", baseUrl: "http://localhost:8080/v1" } as Model;
expect(supportsDeveloperRole(model)).toBe(false);
const model = { provider: "custom", baseUrl: "http://localhost:8080/v1" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
});
it("is case-insensitive for base URL matching", () => {
const model = { provider: "custom", baseUrl: "https://API.OPENAI.COM/v1" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "custom", baseUrl: "https://API.OPENAI.COM/v1" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns true for azure.com/openai base URL", () => {
const model = { provider: "custom", baseUrl: "https://azure.com/openai/deployments/my-model" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "custom", baseUrl: "https://azure.com/openai/deployments/my-model" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns true for github-copilot provider with api.githubcopilot.com", () => {
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns true for github-copilot provider with api.enterprise.githubcopilot.com", () => {
const model = { provider: "github-copilot", baseUrl: "https://api.enterprise.githubcopilot.com" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "github-copilot", baseUrl: "https://api.enterprise.githubcopilot.com" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
it("returns true for github-copilot provider with copilot-api enterprise domain", () => {
const model = { provider: "github-copilot", baseUrl: "https://copilot-api.mycompany.com" } as Model;
expect(supportsDeveloperRole(model)).toBe(true);
const model = { provider: "github-copilot", baseUrl: "https://copilot-api.mycompany.com" };
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
});
});
+6 -1
View File
@@ -1,9 +1,12 @@
# Changelog
## [Unreleased]
### Added
- Added `hostMatchesUrl`, `modelMatchesHost`, and endpoint-shape helpers in the new `hosts` module for consistent provider/baseUrl matching
- Added `streamIdleTimeoutMs` to `OpenAICompat` and now auto-populated it for GLM coding-plan and direct DeepSeek reasoning models
- Added `supportsLongPromptCacheRetention` and the OpenAI Responses helpers `detectOpenAIResponsesCompat`/`resolveOpenAIResponsesCompat`
- Added anthropic-messages compatibility resolution with new `AnthropicCompat` fields `requiresToolResultId` and `replayUnsignedThinking`
- New `@oh-my-pi/pi-catalog` package: the model catalog extracted from `@oh-my-pi/pi-ai`. Owns the bundled `models.json` and its generation pipeline (`scripts/generate-models.ts`), the core model data types (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces), thinking metadata enrichment and generated policies (`model-thinking.ts`), the SQLite model cache and model manager, per-provider discovery factories (`provider-models/`), the discovery protocol clients (`discovery/`), and the new `CATALOG_PROVIDERS` table — the single source of truth for provider ids, default models, and discovery wiring (`KnownProvider`, `PROVIDER_DESCRIPTORS`, and `DEFAULT_MODEL_PER_PROVIDER` are derived from it).
- New `identity/` module centralizing model-identity concerns that were previously duplicated across packages: family classification and version parsing (`identity/classify.ts`, extracted from pi-ai's `model-thinking` internals), canonical model equivalence with injected reference data (`identity/equivalence.ts`, from coding-agent's `model-equivalence`), proxy/reseller reference lookup (`identity/reference.ts`, from coding-agent's `model-registry`), bracket-affix and id-segment helpers (`identity/id.ts`), a single trailing-marker vocabulary with canonical vs reference flavors (`identity/markers.ts` — `search` stays reference-only so Perplexity's `sonar-pro-search` remains canonical-distinct), and provider priority ordering (`identity/priority.ts`).
- Memoized bundled-reference accessors (`getBundledCanonicalReferenceData` / `getBundledModelReferenceIndex` in `identity/bundled.ts`): one lazy walk of the bundled catalog feeds both canonical equivalence and proxy-reference lookup, so consumers no longer hand-roll the glue.
@@ -11,6 +14,8 @@
### Changed
- Changed OpenAI compatibility detection to use shared host classifiers (`modelMatchesHost`/`hostMatchesUrl`) with normalized matching instead of raw URL substring checks
- Changed `hostMatchesUrl`/`modelMatchesHost` usage in compatibility detection to reduce mismatches across case variants and provider alias hosts
- Provider catalog entries now carry the runtime API-key env fallback as an ordered `envVars` list; `catalogDiscovery.envVars` became an optional generation-time override (only `cursor` and `vercel-ai-gateway` differ) and `PROVIDER_DESCRIPTORS` materializes the resolved list for `generate-models.ts`.
- `Model`'s api parameter now defaults to `Api` instead of `any` (`Model<TApi extends Api = Api>`), so bare `Model` no longer behaves as `Model<any>` at call sites.
+109
View File
@@ -0,0 +1,109 @@
/**
* Anthropic-messages compatibility detection and resolution — the
* anthropic-side analogue of `./openai`. Detect-time defaults come from
* provider ids, strict URL checks, and model-id classification; explicit
* `model.compat` overrides always win.
*/
import { isAnthropicFableOrMythosModel, supportsMidConversationSystemMessages } from "../model-thinking";
import type { AnthropicCompat, Model } from "../types";
/**
* Official first-party Anthropic API check (https + exact host). A missing
* baseUrl is official on purpose: request dispatch falls back to
* `https://api.anthropic.com`. Strict URL parsing (not substring) because the
* callers gate auth flows and body mutations on it.
*/
export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean {
if (!baseUrl) return true;
try {
const url = new URL(baseUrl);
return url.protocol.toLowerCase() === "https:" && url.hostname.toLowerCase() === "api.anthropic.com";
} catch {
return false;
}
}
/** Z.AI's Anthropic-compatible proxy (`api.z.ai/api/anthropic`), strict-host matched. */
function isZaiAnthropicUrl(baseUrl: string | undefined): boolean {
if (!baseUrl) return false;
try {
return new URL(baseUrl).hostname.toLowerCase() === "api.z.ai";
} catch {
return false;
}
}
/** DeepSeek-operated host, strict-host matched (`api.deepseek.com` or any `*.deepseek.com`). */
function isDeepseekHostUrl(baseUrl: string | undefined): boolean {
if (!baseUrl) return false;
try {
const hostname = new URL(baseUrl).hostname.toLowerCase();
return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com");
} catch {
return false;
}
}
export type ResolvedAnthropicCompat = Required<AnthropicCompat>;
/**
* Detect anthropic-messages compatibility defaults from provider/baseUrl/model id.
* @param resolvedBaseUrl - Effective request base URL when it differs from
* `model.baseUrl` (e.g. an options-level override).
*/
export function detectAnthropicCompat(
model: Model<"anthropic-messages">,
resolvedBaseUrl?: string,
): ResolvedAnthropicCompat {
const baseUrl = resolvedBaseUrl ?? model.baseUrl;
const isZai = model.provider === "zai" || isZaiAnthropicUrl(baseUrl);
return {
disableStrictTools: false,
disableAdaptiveThinking: false,
supportsEagerToolInputStreaming: true,
supportsLongCacheRetention: true,
// First-party Claude API only. Bedrock/Vertex/Foundry and other
// Anthropic-compatible gateways reject mid-conversation system roles, so
// detection requires the canonical api.anthropic.com host plus a
// supported model id.
supportsMidConversationSystem:
isOfficialAnthropicApiUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id),
supportsForcedToolChoice: !isAnthropicFableOrMythosModel(model.id),
// Z.AI workaround (issue #814): its proxy deserializes tool_result blocks
// into a class that reads `.id`.
requiresToolResultId: isZai,
// Official Anthropic enforces signature-based thinking-chain integrity, so
// unsigned thinking blocks must stay text there. Anthropic-compatible
// reasoning endpoints commonly emit unsigned thinking blocks while still
// expecting them back as `type: "thinking"` on continuation; demoting them
// loses the reasoning chain and can destabilize the next tool-call
// arguments (#2005). Known non-signing hosts (Z.AI, DeepSeek) are also
// preserved for compatibility.
replayUnsignedThinking:
isZai ||
model.provider === "deepseek" ||
isDeepseekHostUrl(baseUrl) ||
(model.reasoning && !isOfficialAnthropicApiUrl(baseUrl)),
};
}
/** Layer explicit `model.compat` overrides onto the detected anthropic defaults. */
export function resolveAnthropicCompat(
model: Model<"anthropic-messages">,
resolvedBaseUrl?: string,
): ResolvedAnthropicCompat {
const detected = detectAnthropicCompat(model, resolvedBaseUrl);
const compat = model.compat;
if (!compat) return detected;
return {
disableStrictTools: compat.disableStrictTools ?? detected.disableStrictTools,
disableAdaptiveThinking: compat.disableAdaptiveThinking ?? detected.disableAdaptiveThinking,
supportsEagerToolInputStreaming:
compat.supportsEagerToolInputStreaming ?? detected.supportsEagerToolInputStreaming,
supportsLongCacheRetention: compat.supportsLongCacheRetention ?? detected.supportsLongCacheRetention,
supportsMidConversationSystem: compat.supportsMidConversationSystem ?? detected.supportsMidConversationSystem,
supportsForcedToolChoice: compat.supportsForcedToolChoice ?? detected.supportsForcedToolChoice,
requiresToolResultId: compat.requiresToolResultId ?? detected.requiresToolResultId,
replayUnsignedThinking: compat.replayUnsignedThinking ?? detected.replayUnsignedThinking,
};
}
+130 -69
View File
@@ -1,3 +1,13 @@
import { hostMatchesUrl, modelMatchesHost } from "../hosts";
import {
isAnthropicNamespacedModelId,
isClaudeModelId,
isDeepseekModelIdOrName,
isKimiK26ModelId,
isKimiModelId,
isMimoModelIdOrName,
isQwenModelId,
} from "../identity/family";
import type { Model, OpenAICompat } from "../types";
type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
@@ -10,6 +20,8 @@ export type ResolvedOpenAICompat = Required<
| "vercelGatewayRouting"
| "extraBody"
| "toolStrictMode"
| "streamIdleTimeoutMs"
| "supportsLongPromptCacheRetention"
| "cacheControlFormat"
| "thinkingKeep"
>
@@ -19,9 +31,16 @@ export type ResolvedOpenAICompat = Required<
extraBody?: OpenAICompat["extraBody"];
cacheControlFormat?: OpenAICompat["cacheControlFormat"];
thinkingKeep?: OpenAICompat["thinkingKeep"];
streamIdleTimeoutMs?: number;
toolStrictMode: ResolvedToolStrictMode;
};
/** GLM coding-plan SKUs idle for minutes mid-reasoning; see `streamIdleTimeoutMs`. */
const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
/** Direct DeepSeek reasoning models stall between thinking and answer phases. */
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
if (
provider === "openai" ||
@@ -33,17 +52,13 @@ function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
) {
return true;
}
const normalizedBaseUrl = baseUrl.toLowerCase();
return (
normalizedBaseUrl.includes("api.openai.com") ||
normalizedBaseUrl.includes(".openai.azure.com") ||
normalizedBaseUrl.includes("models.inference.ai.azure.com") ||
normalizedBaseUrl.includes("api.cerebras.ai") ||
normalizedBaseUrl.includes("api.together.xyz") ||
normalizedBaseUrl.includes("openrouter.ai") ||
normalizedBaseUrl.includes("api.deepseek.com") ||
normalizedBaseUrl.includes("deepseek.com")
hostMatchesUrl(baseUrl, "openai") ||
hostMatchesUrl(baseUrl, "azureOpenAI") ||
hostMatchesUrl(baseUrl, "cerebras") ||
hostMatchesUrl(baseUrl, "together") ||
hostMatchesUrl(baseUrl, "openrouter") ||
hostMatchesUrl(baseUrl, "deepseekFamily")
);
}
@@ -87,23 +102,19 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
const provider = model.provider;
// Use resolvedBaseUrl if provided (e.g., after GitHub Copilot proxy-ep resolution)
const baseUrl = resolvedBaseUrl ?? model.baseUrl;
const hostModel = { provider, baseUrl };
const isCerebras = provider === "cerebras" || baseUrl.includes("cerebras.ai");
const isZai = provider === "zai" || baseUrl.includes("api.z.ai");
const isZhipu = provider === "zhipu-coding-plan" || baseUrl.includes("open.bigmodel.cn");
const isKilo = provider === "kilo" || baseUrl.includes("api.kilo.ai");
const isKimiModel = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
const isMoonshotNativeHost =
provider === "moonshot" || provider === "kimi-code" || /api\.moonshot\.ai|api\.kimi\.com/i.test(baseUrl);
const isMoonshotKimi = isKimiModel && isMoonshotNativeHost;
const usesMoonshotKimiPreservedThinking = isMoonshotKimi && /(^|\/)kimi-k2\.6(?:[-:]|$)/i.test(model.id);
const isCerebras = modelMatchesHost(hostModel, "cerebras");
const isZai = modelMatchesHost(hostModel, "zai");
const isZhipu = modelMatchesHost(hostModel, "zhipu");
const isKilo = modelMatchesHost(hostModel, "kilo");
const isKimiModel = isKimiModelId(model.id);
const isMoonshotKimi = isKimiModel && modelMatchesHost(hostModel, "moonshotNative");
const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(model.id);
const isAnthropicModel =
provider === "anthropic" ||
baseUrl.includes("api.anthropic.com") ||
/(^|\/)claude[-.]/i.test(model.id) ||
/(^|\/)anthropic\//i.test(model.id);
const isAlibaba = provider === "alibaba-coding-plan" || baseUrl.includes("dashscope");
const isQwen = model.id.toLowerCase().includes("qwen");
modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(model.id) || isAnthropicNamespacedModelId(model.id);
const isAlibaba = modelMatchesHost(hostModel, "alibabaDashscope");
const isQwen = isQwenModelId(model.id);
// DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
// thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
// upstream model is reachable through many OpenAI-compat hosts (api.deepseek.com, Deepinfra,
@@ -112,66 +123,52 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
// applies when thinking mode is actually engaged.
const lowerId = model.id.toLowerCase();
const lowerName = (model.name ?? "").toLowerCase();
const isXiaomiHost =
provider === "xiaomi" || provider.startsWith("xiaomi-token-plan-") || baseUrl.includes("xiaomimimo.com");
const isMimoModel = lowerId.includes("mimo") || lowerName.includes("mimo");
const isXiaomiMimo = isXiaomiHost && isMimoModel;
const isXiaomiHost = modelMatchesHost(hostModel, "xiaomi");
const isXiaomiMimo = isXiaomiHost && (isMimoModelIdOrName(model.id) || isMimoModelIdOrName(model.name ?? ""));
// OpenCode Zen's `big-pickle` is a DeepSeek reasoning alias; the upstream
// 400s come from DeepSeek and require exact reasoning_content replay.
const isOpenCodeDeepseekAlias =
provider === "opencode-zen" && (lowerId === "big-pickle" || lowerName === "big pickle");
const isDeepseekFamily =
provider === "deepseek" ||
baseUrl.includes("deepseek.com") ||
lowerId.includes("deepseek") ||
lowerName.includes("deepseek") ||
modelMatchesHost(hostModel, "deepseekFamily") ||
isDeepseekModelIdOrName(model.id) ||
isDeepseekModelIdOrName(model.name ?? "") ||
isOpenCodeDeepseekAlias;
const isDirectDeepseekApi = provider === "deepseek" || baseUrl.includes("api.deepseek.com");
const isDirectDeepseekApi = modelMatchesHost(hostModel, "deepseekDirect");
const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(model.reasoning);
const isGrok = modelMatchesHost(hostModel, "xai");
const isMistral = modelMatchesHost(hostModel, "mistral");
const isOpenCodeHost = modelMatchesHost(hostModel, "opencode");
const isNonStandard =
isCerebras ||
provider === "xai" ||
baseUrl.includes("api.x.ai") ||
provider === "mistral" ||
baseUrl.includes("mistral.ai") ||
baseUrl.includes("chutes.ai") ||
baseUrl.includes("deepseek.com") ||
baseUrl.includes("fireworks.ai") ||
isGrok ||
isMistral ||
hostMatchesUrl(baseUrl, "chutes") ||
hostMatchesUrl(baseUrl, "deepseekFamily") ||
hostMatchesUrl(baseUrl, "fireworks") ||
isAlibaba ||
isZai ||
isZhipu ||
isKilo ||
isQwen ||
isXiaomiHost ||
provider === "opencode-zen" ||
provider === "opencode-go" ||
baseUrl.includes("opencode.ai");
isOpenCodeHost;
const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
const useMaxTokens =
provider === "mistral" ||
baseUrl.includes("mistral.ai") ||
baseUrl.includes("chutes.ai") ||
baseUrl.includes("fireworks.ai") ||
isDirectDeepseekApi;
const isGrok = provider === "xai" || baseUrl.includes("api.x.ai");
const isMistral = provider === "mistral" || baseUrl.includes("mistral.ai");
isMistral || hostMatchesUrl(baseUrl, "chutes") || hostMatchesUrl(baseUrl, "fireworks") || isDirectDeepseekApi;
// Hosts whose chat-completions endpoints are known to accept multiple
// leading `system`/`developer` messages (preferred for KV-cache reuse).
// Anything outside this allowlist defaults to coalescing because
// strict chat templates (Qwen 3.5+ via vLLM, MiniMax, etc.) reject
// follow-up system messages with a 400.
const isOpenAIHost = provider === "openai" || baseUrl.includes("api.openai.com");
const isAzureHost =
provider === "azure" ||
baseUrl.includes(".openai.azure.com") ||
baseUrl.includes("models.inference.ai.azure.com") ||
baseUrl.includes("azure.com/openai");
const isOpenRouter = provider === "openrouter" || baseUrl.includes("openrouter.ai");
const isTogether = provider === "together" || baseUrl.includes("api.together.xyz");
const isFireworks = baseUrl.includes("fireworks.ai");
const isGroqHost = provider === "groq" || baseUrl.includes("api.groq.com");
const isOpenAIHost = modelMatchesHost(hostModel, "openai");
const isAzureHost = modelMatchesHost(hostModel, "azureOpenAI");
const isOpenRouter = modelMatchesHost(hostModel, "openrouter");
const isTogether = modelMatchesHost(hostModel, "together");
const isFireworks = hostMatchesUrl(baseUrl, "fireworks");
const isGroqHost = modelMatchesHost(hostModel, "groq");
const isCopilotHost = provider === "github-copilot";
const isZenmuxHost = provider === "zenmux";
// Endpoints that MUST receive a single system block. MiniMax's OpenAI
@@ -179,12 +176,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
// Dashscope and Qwen Portal serve Qwen models whose chat template
// raises "System message must be at the beginning" if any system
// message appears past index 0.
const isMiniMaxHost =
provider === "minimax-code" ||
provider === "minimax-code-cn" ||
baseUrl.includes("api.minimax.io") ||
baseUrl.includes("api.minimaxi.com");
const isQwenPortal = provider === "qwen-portal" || baseUrl.includes("portal.qwen.ai");
const isMiniMaxHost = modelMatchesHost(hostModel, "minimax");
const isQwenPortal = modelMatchesHost(hostModel, "qwenPortal");
const supportsMultipleSystemMessagesDefault =
!isMiniMaxHost &&
!isAlibaba &&
@@ -234,6 +227,16 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
: {};
// Stream-watchdog floor: GLM coding-plan SKUs and direct DeepSeek reasoning
// models idle for minutes mid-reasoning; widen the idle timeout so warm-ups
// stop aborting and retrying.
const streamIdleTimeoutMs =
GLM_CODING_PLAN_MODEL_PATTERN.test(model.id) && (isZai || isZhipu)
? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS
: model.reasoning && isDirectDeepseekApi
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
: undefined;
return {
supportsStore: !isNonStandard,
// `developer` is an OpenAI-Responses-era extension to the chat-completions schema. Almost
@@ -263,7 +266,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
thinkingFormat:
isZai || isZhipu || isMoonshotKimi || isXiaomiMimo
? "zai"
: provider === "openrouter" || baseUrl.includes("openrouter.ai")
: isOpenRouter
? "openrouter"
: isAlibaba || isQwen
? "qwen"
@@ -286,7 +289,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
(isKimiModel && !isOpenCodeProvider) ||
(isDeepseekFamily && Boolean(model.reasoning)) ||
isXiaomiMimo ||
((provider === "openrouter" || baseUrl.includes("openrouter.ai")) && Boolean(model.reasoning)),
(isOpenRouter && Boolean(model.reasoning)),
// DeepSeek V4 and Xiaomi MiMo reject synthetic reasoning_content placeholders (".") on tool-call turns.
// Kimi and OpenRouter accept them when actual reasoning is unavailable.
allowsSyntheticReasoningContentForToolCalls: (!isDeepseekFamily || !model.reasoning) && !isXiaomiMimo,
@@ -297,6 +300,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
toolStrictMode: isCerebras ? "all_strict" : "mixed",
streamIdleTimeoutMs,
};
}
@@ -350,5 +354,62 @@ export function resolveOpenAICompat(
supportsStrictMode: model.compat.supportsStrictMode ?? detected.supportsStrictMode,
extraBody: model.compat.extraBody ?? detected.extraBody,
toolStrictMode: model.compat.toolStrictMode ?? detected.toolStrictMode,
streamIdleTimeoutMs: model.compat.streamIdleTimeoutMs ?? detected.streamIdleTimeoutMs,
};
}
/** Resolved Responses-API compatibility view (see `detectOpenAIResponsesCompat`). */
export interface ResolvedOpenAIResponsesCompat {
supportsDeveloperRole: boolean;
supportsStrictMode: boolean;
supportsLongPromptCacheRetention: boolean;
}
/**
* Detect Responses-API compatibility from provider/baseUrl. The Responses
* flavor deliberately differs from chat-completions: GitHub Copilot's
* responses endpoint accepts the `developer` role, while strict tool mode is
* scoped to first-party OpenAI/Azure/Copilot providers. Developer-role and
* prompt-cache detection are URL-only on purpose — the historical call sites
* never consulted the provider id for them.
*/
export function detectOpenAIResponsesCompat(
model: { provider: string; baseUrl: string },
resolvedBaseUrl?: string,
): ResolvedOpenAIResponsesCompat {
const baseUrl = resolvedBaseUrl ?? model.baseUrl ?? "";
return {
supportsDeveloperRole:
hostMatchesUrl(baseUrl, "openai") ||
hostMatchesUrl(baseUrl, "azureOpenAI") ||
hostMatchesUrl(baseUrl, "githubCopilot"),
supportsStrictMode:
model.provider === "openai" ||
model.provider === "azure" ||
model.provider === "github-copilot" ||
hostMatchesUrl(baseUrl, "openai") ||
hostMatchesUrl(baseUrl, "azureOpenAI"),
supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"),
};
}
/**
* Resolve Responses-API compatibility by layering explicit `model.compat`
* overrides onto the detected defaults — the Responses-side analogue of
* `resolveOpenAICompat`. Models bundled with `supportsDeveloperRole: false`
* (codex-mini-style SKUs) take effect here.
*/
export function resolveOpenAIResponsesCompat(
model: { provider: string; baseUrl: string; compat?: OpenAICompat },
resolvedBaseUrl?: string,
): ResolvedOpenAIResponsesCompat {
const detected = detectOpenAIResponsesCompat(model, resolvedBaseUrl);
const compat = model.compat;
if (!compat) return detected;
return {
supportsDeveloperRole: compat.supportsDeveloperRole ?? detected.supportsDeveloperRole,
supportsStrictMode: compat.supportsStrictMode ?? detected.supportsStrictMode,
supportsLongPromptCacheRetention:
compat.supportsLongPromptCacheRetention ?? detected.supportsLongPromptCacheRetention,
};
}
+110
View File
@@ -0,0 +1,110 @@
/**
* Known model-endpoint host classification — the single vocabulary for the
* `provider === id || baseUrl.includes(marker)` idiom that gates wire-level
* behavior (compat detection, routing, header shaping, watchdog floors).
*
* Markers are case-insensitive substrings matched against the base URL, NOT
* parsed hostnames: proxies regularly embed the upstream host in a path
* segment, and the historical call sites all used substring semantics.
* Callers needing strict hostname matching (e.g. guards before request-body
* mutation) should keep their own `new URL().hostname` checks.
*/
interface HostClassSpec {
/** Provider ids that imply this host class regardless of baseUrl. */
readonly providers?: readonly string[];
/** Provider-id prefixes that imply this host class (e.g. `xiaomi-token-plan-`). */
readonly providerPrefixes?: readonly string[];
/** Case-insensitive substrings matched against the base URL. */
readonly urlMarkers: readonly string[];
}
export const KNOWN_HOSTS = {
openai: { providers: ["openai"], urlMarkers: ["api.openai.com"] },
azureOpenAI: {
providers: ["azure"],
urlMarkers: [".openai.azure.com", "azure.com/openai", "models.inference.ai.azure.com"],
},
openrouter: { providers: ["openrouter"], urlMarkers: ["openrouter.ai"] },
vercelAIGateway: { providers: ["vercel-ai-gateway"], urlMarkers: ["ai-gateway.vercel.sh"] },
githubCopilot: { providers: ["github-copilot"], urlMarkers: ["githubcopilot.com", "copilot-api."] },
anthropic: { providers: ["anthropic"], urlMarkers: ["api.anthropic.com"] },
/** DeepSeek's first-party API only — gates direct-API quirks (max_tokens field, thinking extraBody). */
deepseekDirect: { providers: ["deepseek"], urlMarkers: ["api.deepseek.com"] },
/** Any DeepSeek-operated host (first-party API, web-chat fronts). Wider than `deepseekDirect` on purpose. */
deepseekFamily: { providers: ["deepseek"], urlMarkers: ["deepseek.com"] },
cerebras: { providers: ["cerebras"], urlMarkers: ["cerebras.ai"] },
zai: { providers: ["zai"], urlMarkers: ["api.z.ai"] },
zhipu: { providers: ["zhipu-coding-plan"], urlMarkers: ["open.bigmodel.cn"] },
kilo: { providers: ["kilo"], urlMarkers: ["api.kilo.ai"] },
alibabaDashscope: { providers: ["alibaba-coding-plan"], urlMarkers: ["dashscope"] },
xiaomi: { providers: ["xiaomi"], providerPrefixes: ["xiaomi-token-plan-"], urlMarkers: ["xiaomimimo.com"] },
xai: { providers: ["xai"], urlMarkers: ["api.x.ai"] },
mistral: { providers: ["mistral"], urlMarkers: ["mistral.ai"] },
together: { providers: ["together"], urlMarkers: ["api.together.xyz"] },
/** URL-only on purpose: the `fireworks`/`firepass` providers route per-model and not every model is Fireworks-shaped. */
fireworks: { urlMarkers: ["fireworks.ai"] },
groq: { providers: ["groq"], urlMarkers: ["api.groq.com"] },
minimax: {
providers: ["minimax", "minimax-code", "minimax-code-cn"],
urlMarkers: ["api.minimax.io", "api.minimaxi.com"],
},
qwenPortal: { providers: ["qwen-portal"], urlMarkers: ["portal.qwen.ai"] },
moonshotNative: { providers: ["moonshot", "kimi-code"], urlMarkers: ["api.moonshot.ai", "api.kimi.com"] },
opencode: { providers: ["opencode-go", "opencode-zen"], urlMarkers: ["opencode.ai"] },
chutes: { urlMarkers: ["chutes.ai"] },
} as const satisfies Record<string, HostClassSpec>;
export type KnownHost = keyof typeof KNOWN_HOSTS;
/** URL-only host check (for call sites that have no provider id, e.g. raw env config). */
export function hostMatchesUrl(baseUrl: string | undefined, host: KnownHost): boolean {
if (!baseUrl) return false;
const spec: HostClassSpec = KNOWN_HOSTS[host];
const normalized = baseUrl.toLowerCase();
for (const marker of spec.urlMarkers) {
if (normalized.includes(marker)) return true;
}
return false;
}
/** Provider-or-URL host check — the canonical `provider === id || baseUrl.includes(marker)` idiom. */
export function modelMatchesHost(model: { provider: string; baseUrl: string }, host: KnownHost): boolean {
const spec: HostClassSpec = KNOWN_HOSTS[host];
if (spec.providers) {
for (const provider of spec.providers) {
if (model.provider === provider) return true;
}
}
if (spec.providerPrefixes) {
for (const prefix of spec.providerPrefixes) {
if (model.provider.startsWith(prefix)) return true;
}
}
return hostMatchesUrl(model.baseUrl, host);
}
// --- Endpoint-shape predicates (URL path/verb shapes, not vendor hosts) ---
/** Vertex AI express-mode OpenAI-compatible endpoint (`…/endpoints/openapi`). */
export function isVertexExpressOpenAIUrl(baseUrl: string): boolean {
return baseUrl.includes("/endpoints/openapi");
}
/** Vertex AI Anthropic raw-predict endpoints (`:streamRawPredict` / `:rawPredict`). */
export function isVertexRawPredictUrl(baseUrl: string): boolean {
return baseUrl.includes(":streamRawPredict") || baseUrl.includes(":rawPredict");
}
/** Azure OpenAI deployment-scoped path (`…/deployments/<name>/…`). */
export function isAzureDeploymentsUrl(baseUrl: string): boolean {
return baseUrl.includes("/deployments/");
}
/** Alibaba DashScope consumer `compatible-mode` endpoint (rejects multimodal arrays for some text-only SKUs). */
export function isDashscopeCompatibleModeUrl(baseUrl: string): boolean {
const normalized = baseUrl.toLowerCase();
return (
normalized.includes("dashscope") && normalized.includes("aliyuncs.com") && normalized.includes("/compatible-mode")
);
}
+59
View File
@@ -0,0 +1,59 @@
/**
* Model-family id predicates: the shared vocabulary for "is this id a member
* of family X" checks that gate wire-level behavior across hosts (a Kimi or
* DeepSeek model keeps its quirks no matter which OpenAI-compatible proxy
* serves it). Looser per-feature heuristics (e.g. stream-markup healing)
* deliberately keep their own patterns — only provably-shared matchers live
* here.
*/
/** Kimi family ids in any namespace form (`moonshotai/kimi-*`, `kimi-k2.6`, `vendor/kimi.x`). */
export function isKimiModelId(modelId: string): boolean {
return modelId.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(modelId);
}
/** Kimi K2.6 specifically (preserved-thinking transport on Moonshot-native hosts). */
export function isKimiK26ModelId(modelId: string): boolean {
return /(^|\/)kimi-k2\.6(?:[-:]|$)/i.test(modelId);
}
/** Claude ids in any namespace form (`claude-*`, `vendor/claude.x`). */
export function isClaudeModelId(modelId: string): boolean {
return /(^|\/)claude[-.]/i.test(modelId);
}
/** `anthropic/`-namespaced ids (aggregator catalogs like OpenRouter). */
export function isAnthropicNamespacedModelId(modelId: string): boolean {
return /(^|\/)anthropic\//i.test(modelId);
}
/** Qwen family ids (substring match — Qwen SKUs have no stable prefix shape). */
export function isQwenModelId(modelId: string): boolean {
return modelId.toLowerCase().includes("qwen");
}
/** DeepSeek family by id or display name (proxies often rename the id but keep the name). */
export function isDeepseekModelIdOrName(value: string): boolean {
return value.toLowerCase().includes("deepseek");
}
/** Xiaomi MiMo family by id or display name. */
export function isMimoModelIdOrName(value: string): boolean {
return value.toLowerCase().includes("mimo");
}
/**
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
* 4.6+) reject the field.
*/
export function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
// Bound the minor to non-date digits: bare dated ids like
// `claude-opus-4-20250514` (Opus 4.0) must not parse as minor=20250514.
const match = /claude-opus-(\d+)-(\d{1,2})(?!\d)/.exec(modelId);
if (!match) return false;
const major = Number(match[1]);
const minor = Number(match[2]);
return major > 4 || (major === 4 && minor >= 7);
}
+1
View File
@@ -1,6 +1,7 @@
export * from "./bundled";
export * from "./classify";
export * from "./equivalence";
export * from "./family";
export * from "./id";
export * from "./markers";
export * from "./priority";
+2 -1
View File
@@ -1,5 +1,6 @@
import { resolveOpenAICompat } from "./compat/openai";
import { Effort, THINKING_EFFORTS } from "./effort";
import { modelMatchesHost } from "./hosts";
import {
type AnthropicModel,
bareModelId,
@@ -353,7 +354,7 @@ function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
model: ApiModel<TApi>,
): boolean {
if (model.api !== "openai-completions") return false;
if (model.provider !== "openrouter" && !model.baseUrl.includes("openrouter.ai")) return false;
if (!modelMatchesHost(model, "openrouter")) return false;
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
}
@@ -2349,11 +2349,15 @@ export interface GithubCopilotModelManagerConfig {
fetch?: FetchImpl;
}
const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus)-4([.-]|$)/;
const isCopilotResponsesModelId = (modelId: string): boolean =>
modelId.startsWith("gpt-5") || modelId.startsWith("oswe");
function inferCopilotApi(modelId: string): Api {
if (/^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId)) {
if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) {
return "anthropic-messages";
}
if (modelId.startsWith("gpt-5") || modelId.startsWith("oswe")) {
if (isCopilotResponsesModelId(modelId)) {
return "openai-responses";
}
return "openai-completions";
@@ -2776,11 +2780,11 @@ const COPILOT_DEFAULT_RESOLUTION = {
const COPILOT_API_RESOLUTION_RULES: readonly ApiResolutionRule[] = [
{
matches: modelId => /^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId),
matches: modelId => COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId),
resolved: { api: "anthropic-messages", baseUrl: COPILOT_BASE_URL },
},
{
matches: modelId => modelId.startsWith("gpt-5") || modelId.startsWith("oswe"),
matches: isCopilotResponsesModelId,
resolved: { api: "openai-responses", baseUrl: COPILOT_BASE_URL },
},
];
+23
View File
@@ -182,6 +182,13 @@ export interface OpenAICompat {
cacheControlFormat?: "anthropic" | undefined;
/** Whether the provider supports the `strict` field in tool definitions. Default: auto-detected per provider/baseUrl (conservative for unknown providers). */
supportsStrictMode?: boolean;
/**
* Stream-watchdog idle-timeout floor in ms for slow reasoning hosts.
* Default: auto-detected (GLM coding-plan hosts, direct DeepSeek reasoning).
*/
streamIdleTimeoutMs?: number;
/** Whether the host honors `prompt_cache_retention: "24h"` on the Responses API. Default: auto-detected (api.openai.com). */
supportsLongPromptCacheRetention?: boolean;
/** Whether tool schemas must be sent either all strict or all non-strict. Undefined keeps the existing per-tool mixed behavior. */
toolStrictMode?: "all_strict" | "none";
}
@@ -225,6 +232,22 @@ export interface AnthropicCompat {
* When unset, auto-detected from the model id. Default: true.
*/
supportsForcedToolChoice?: boolean;
/**
* Include a non-standard `id` field (aliasing `tool_use_id`) on
* `tool_result` blocks. Z.AI's Anthropic-compatible proxy deserializes
* tool results into a class that reads `.id` (issue #814). Default:
* auto-detected (Z.AI hosts).
*/
requiresToolResultId?: boolean;
/**
* Replay unsigned `thinking` blocks from prior assistant turns as native
* thinking instead of demoting them to text. Official Anthropic enforces
* signature-based thinking-chain integrity, so unsigned blocks must stay
* text there; compatible reasoning endpoints (Z.AI, DeepSeek, …) emit
* unsigned blocks and expect them back as `type: "thinking"` (#2005).
* Default: auto-detected from provider/baseUrl and `model.reasoning`.
*/
replayUnsignedThinking?: boolean;
}
/**
+75
View File
@@ -0,0 +1,75 @@
import { describe, expect, test } from "bun:test";
import {
hostMatchesUrl,
isDashscopeCompatibleModeUrl,
isVertexExpressOpenAIUrl,
isVertexRawPredictUrl,
modelMatchesHost,
} from "@oh-my-pi/pi-catalog/hosts";
describe("hostMatchesUrl", () => {
test("matches OpenRouter URLs and rejects other or missing URLs", () => {
expect(hostMatchesUrl("https://openrouter.ai/api/v1", "openrouter")).toBe(true);
expect(hostMatchesUrl("https://api.openai.com/v1", "openrouter")).toBe(false);
expect(hostMatchesUrl(undefined, "openrouter")).toBe(false);
});
test("matches Z.AI URLs case-insensitively", () => {
expect(hostMatchesUrl("https://API.Z.AI/api/paas/v4", "zai")).toBe(true);
});
test("keeps DeepSeek direct host narrower than DeepSeek family", () => {
expect(hostMatchesUrl("https://api.deepseek.com/v1", "deepseekDirect")).toBe(true);
expect(hostMatchesUrl("https://api.deepseek.com/v1", "deepseekFamily")).toBe(true);
expect(hostMatchesUrl("https://chat.deepseek.com/api", "deepseekFamily")).toBe(true);
expect(hostMatchesUrl("https://chat.deepseek.com/api", "deepseekDirect")).toBe(false);
});
});
describe("modelMatchesHost", () => {
test("matches by provider id, provider prefix, and URL-only Fireworks markers", () => {
expect(modelMatchesHost({ provider: "openrouter", baseUrl: "https://example.com/v1" }, "openrouter")).toBe(true);
expect(modelMatchesHost({ provider: "xiaomi-token-plan-eu", baseUrl: "https://example.com/v1" }, "xiaomi")).toBe(
true,
);
expect(modelMatchesHost({ provider: "fireworks", baseUrl: "https://example.com/v1" }, "fireworks")).toBe(false);
expect(
modelMatchesHost({ provider: "custom", baseUrl: "https://api.fireworks.ai/inference/v1" }, "fireworks"),
).toBe(true);
});
});
describe("endpoint shape predicates", () => {
test("recognizes Vertex express OpenAI-compatible URLs", () => {
expect(
isVertexExpressOpenAIUrl(
"https://us-central1-aiplatform.googleapis.com/v1/projects/p/locations/us/endpoints/openapi",
),
).toBe(true);
expect(
isVertexExpressOpenAIUrl(
"https://us-central1-aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/google/models/gemini",
),
).toBe(false);
});
test("recognizes Vertex rawPredict and streamRawPredict URLs", () => {
expect(
isVertexRawPredictUrl(
"https://aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/anthropic/models/claude:rawPredict",
),
).toBe(true);
expect(
isVertexRawPredictUrl(
"https://aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/anthropic/models/claude:streamRawPredict",
),
).toBe(true);
});
test("requires all DashScope compatible-mode URL markers", () => {
expect(isDashscopeCompatibleModeUrl("https://dashscope.aliyuncs.com/compatible-mode/v1")).toBe(true);
expect(isDashscopeCompatibleModeUrl("https://example.aliyuncs.com/compatible-mode/v1")).toBe(false);
expect(isDashscopeCompatibleModeUrl("https://dashscope.example.com/compatible-mode/v1")).toBe(false);
expect(isDashscopeCompatibleModeUrl("https://dashscope.aliyuncs.com/api/v1")).toBe(false);
});
});
@@ -0,0 +1,44 @@
import { describe, expect, test } from "bun:test";
import {
isClaudeModelId,
isKimiK26ModelId,
isKimiModelId,
supportsAdaptiveThinkingDisplay,
} from "@oh-my-pi/pi-catalog/identity";
describe("isKimiModelId", () => {
test("matches Kimi namespace and delimiter forms", () => {
expect(isKimiModelId("moonshotai/kimi-k2")).toBe(true);
expect(isKimiModelId("kimi-k2.6")).toBe(true);
expect(isKimiModelId("vendor/kimi.x")).toBe(true);
expect(isKimiModelId("akimbo-model")).toBe(false);
});
});
describe("isKimiK26ModelId", () => {
test("matches Kimi K2.6 without accepting adjacent versions", () => {
expect(isKimiK26ModelId("kimi-k2.6")).toBe(true);
expect(isKimiK26ModelId("kimi-k2.6-thinking")).toBe(true);
expect(isKimiK26ModelId("kimi-k2.61")).toBe(false);
expect(isKimiK26ModelId("kimi-k2.5")).toBe(false);
});
});
describe("isClaudeModelId", () => {
test("matches Claude namespace and delimiter forms", () => {
expect(isClaudeModelId("claude-sonnet-4-6")).toBe(true);
expect(isClaudeModelId("anthropic/claude.3")).toBe(true);
expect(isClaudeModelId("my-claudius")).toBe(false);
});
});
describe("supportsAdaptiveThinkingDisplay", () => {
test("allows Claude Fable 5 and Opus 4.7 or newer only", () => {
expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-7")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-5-0")).toBe(true);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-6")).toBe(false);
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-20250514")).toBe(false);
expect(supportsAdaptiveThinkingDisplay("claude-sonnet-4-6")).toBe(false);
});
});
+3 -3
View File
@@ -1,9 +1,9 @@
# Changelog
## [Unreleased]
### Added
- Added `streamIdleTimeoutMs`, `supportsLongPromptCacheRetention`, `requiresToolResultId`, and `replayUnsignedThinking` to the OpenAI `compat` schema so custom model entries can configure those provider-specific capabilities
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("need: 5h → 3 of 5 accounts"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
- npm installs now execute a prebundled single-file entry: `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks
@@ -43,11 +43,11 @@
### Fixed
- Fixed model-provider detection for append-only mode, authoritative Vertex endpoint checks, and upstream-routing selection by switching from URL substring checks to catalog host-matching helpers
- Fixed pasting into the ask tool's "Other (type your own)" text box (and hook input/editor dialogs) on terminals with OSC 5522 enhanced paste (kitty protocol): the enhanced-paste focus routing only targets components exposing a `pasteText` hook, and the dialog wrappers had none, so the payload was stuffed into the main prompt editor hidden behind the dialog. `HookEditorComponent` and `HookInputComponent` now forward `pasteText` to their inner editor/input (pasting also resets the input dialog's timeout countdown like any keystroke).
- Fixed auto-retry giving up after one attempt ("Provider requested Xms wait, exceeds retry.maxDelayMs") on a usage-limit 429 when every sibling account was only momentarily blocked: the retry delay now waits for the earliest sibling unblock when that comes sooner than the provider's multi-hour retry-after, so the next attempt picks up the recovered account instead of failing fast.
- Fixed Hindsight `per-project-tagged` mental-model seeding so each project gets its own conventions/decisions models and session context only injects active-project or untagged models ([#2218](https://github.com/can1357/oh-my-pi/issues/2218)).
- Fixed Windows stdio MCP `.cmd` commands regressing from direct argv launches to a `cmd.exe /c` wrapper in v15.10.10, which made Codegraph MCP exit immediately with `Transport closed` ([#2220](https://github.com/can1357/oh-my-pi/issues/2220)).
- Fixed Windows stdio MCP `.cmd` commands by wrapping batch shims with `cmd.exe /d /s /c` using the outer command quotes required by `cmd /s`, while preserving literal `%` and quoted JSON arguments for Codegraph MCP ([#2220](https://github.com/can1357/oh-my-pi/issues/2220)).
- Fixed the bundled `explore` agent's `thinking-level: med` frontmatter — not a valid effort (`minimal`/`low`/`medium`/`high`/`xhigh`), so it silently parsed to undefined and the agent ran without its intended thinking level
- Discovery context-file reads (`~/.claude`, `~/.cursor`, project trees, `@`-imports) now stat-gate to regular files before reading: a FIFO/socket/char device dropped where a context file is expected previously blocked startup forever on a read that can never see EOF.
- Fixed the read tool's provider-visible `path` schema and docs so web URLs and internal URI targets (`omp://`, `issue://`, `pr://`, etc.) are advertised alongside local files ([#2215](https://github.com/can1357/oh-my-pi/issues/2215)).
@@ -1,3 +1,5 @@
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
/** Provider metadata needed to resolve append-only context mode. */
export interface AppendOnlyContextModel {
provider: string;
@@ -5,19 +7,10 @@ export interface AppendOnlyContextModel {
compat?: object;
}
function isXiaomiHost(baseUrl: string): boolean {
try {
const host = new URL(baseUrl).hostname;
return host === "xiaomimimo.com" || host.endsWith(".xiaomimimo.com");
} catch {
return false;
}
}
function shouldAutoEnableAppendOnlyContext(model: AppendOnlyContextModel | null | undefined): boolean {
if (!model) return false;
if (model.provider === "deepseek") return true;
if (isXiaomiHost(model.baseUrl)) return true;
if (hostMatchesUrl(model.baseUrl, "xiaomi")) return true;
return !!model.compat && "supportsStore" in model.compat && model.compat.supportsStore === true;
}
@@ -2,6 +2,7 @@ import * as path from "node:path";
import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai/api-registry";
import type { Api, Context, Model, SimpleStreamOptions, ThinkingConfig } from "@oh-my-pi/pi-ai/types";
import type { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import { isVertexExpressOpenAIUrl } from "@oh-my-pi/pi-catalog/hosts";
import { readModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import {
createModelManager,
@@ -136,7 +137,7 @@ function isAuthoritativeProjectCatalogModel(model: Model<Api>): boolean {
return (
model.provider === "google-vertex" &&
model.api === "openai-completions" &&
model.baseUrl.includes("/endpoints/openapi")
isVertexExpressOpenAIUrl(model.baseUrl)
);
}
@@ -16,6 +16,7 @@
import { ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import type { Api, Effort, KnownProvider, Model } from "@oh-my-pi/pi-ai";
import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts";
import { buildModelProviderPriorityRank } from "@oh-my-pi/pi-catalog/identity";
import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
@@ -169,14 +170,14 @@ function splitUpstreamRouting(pattern: string): { base: string; upstream: string
/** OpenRouter and Vercel AI Gateway are the aggregators that honor per-request upstream routing. */
function supportsUpstreamRouting(model: Model<Api>): boolean {
return model.baseUrl.includes("openrouter.ai") || model.baseUrl.includes("ai-gateway.vercel.sh");
return modelMatchesHost(model, "openrouter") || modelMatchesHost(model, "vercelAIGateway");
}
/** Pin a resolved aggregator model to a single upstream provider via its compat routing block. */
function applyUpstreamRouting(model: Model<Api>, upstream: string): Model<Api> {
const aggregatorModel = model as Model<"openai-completions">;
const routing = { only: [upstream] };
const compat = model.baseUrl.includes("ai-gateway.vercel.sh")
const compat = modelMatchesHost(model, "vercelAIGateway")
? { ...aggregatorModel.compat, vercelGatewayRouting: routing }
: { ...aggregatorModel.compat, openRouterRouting: routing };
return { ...model, compat } as Model<Api>;
@@ -44,6 +44,11 @@ export const OpenAICompatSchema = z.object({
cacheControlFormat: z.enum(["anthropic"]).optional(),
supportsStrictMode: z.boolean().optional(),
toolStrictMode: z.enum(["all_strict", "none"]).optional(),
streamIdleTimeoutMs: z.number().positive().optional(),
supportsLongPromptCacheRetention: z.boolean().optional(),
// anthropic-messages compat flags (same `compat` slot, per-api interpretation)
requiresToolResultId: z.boolean().optional(),
replayUnsignedThinking: z.boolean().optional(),
});
const EffortSchema = z.enum(["minimal", "low", "medium", "high", "xhigh"]);
+41 -27
View File
@@ -45,8 +45,8 @@ function makeReport(provider: string, email: string, limits: UsageReport["limits
describe("buildRedactionMap", () => {
it("masks everything past a two-char anchor when the anchor is unique", () => {
const map = buildRedactionMap(["alpha@example.test", "bravo@example.test"]);
expect(map.get("alpha@example.test")).toBe("an*");
expect(map.get("bravo@example.test")).toBe("ha*");
expect(map.get("alpha@example.test")).toBe("al*");
expect(map.get("bravo@example.test")).toBe("br*");
});
it("reveals a minimal middle-out differentiator instead of growing the prefix", () => {
@@ -56,32 +56,32 @@ describe("buildRedactionMap", () => {
// Masks must be pairwise distinct so accounts stay tellable-apart.
expect(new Set(masks).size).toBe(masks.length);
for (const mask of masks) {
// Never leak the local part the way prefix growth would ("can.boluk@*").
expect(mask).not.toContain("boluk");
// Never leak the whole local part the way prefix growth would ("dummy@*").
expect(mask).not.toContain("dummy");
// anchor + at most a two-char differentiator.
expect(mask).toMatch(/^ca\*(.{1,2}\*)?$/);
expect(mask).toMatch(/^du\*(.{1,2}\*)?$/);
}
// The "89" account is distinguished by a digit only it contains.
expect(map.get("dum.my9@example.net")).toBe("ca*9*");
expect(map.get("dum.my9@example.net")).toBe("du*9*");
});
it("gives duplicate identities the same mask", () => {
const map = buildRedactionMap(["user@example.test", "user@example.test"]);
expect(map.size).toBe(1);
expect(map.get("user@example.test")).toBe("me*");
expect(map.get("user@example.test")).toBe("us*");
});
});
describe("computeProviderWindowStats", () => {
it("buckets by window duration, binds each account to its worst meter, and ceils the need", () => {
it("buckets by window duration, binds each account to its worst meter, and reports remaining capacity", () => {
const reports = [
makeReport("anthropic", "a@x", [
makeReport("anthropic", "account-a@example.test", [
makeLimit({ id: "5h", usedFraction: 0.9, durationMs: FIVE_HOURS, windowId: "5h" }),
makeLimit({ id: "7d", usedFraction: 0.1, durationMs: SEVEN_DAYS, windowId: "7d" }),
// Tiered meter on the same window: higher burn must bind.
makeLimit({ id: "7d-opus", usedFraction: 0.4, durationMs: SEVEN_DAYS, windowId: "7d", tier: "opus" }),
]),
makeReport("anthropic", "b@x", [
makeReport("anthropic", "account-b@example.test", [
makeLimit({ id: "5h", usedFraction: 0.4, durationMs: FIVE_HOURS, windowId: "5h" }),
makeLimit({ id: "7d", usedFraction: 0.2, durationMs: SEVEN_DAYS, windowId: "7d" }),
]),
@@ -93,15 +93,15 @@ describe("computeProviderWindowStats", () => {
expect(fiveHour.window).toBe("5h");
expect(fiveHour.accounts).toBe(2);
expect(fiveHour.usedAccounts).toBeCloseTo(1.3);
expect(fiveHour.needed).toBe(2);
expect(fiveHour.remainingAccounts).toBeCloseTo(0.7);
expect(sevenDay.window).toBe("7d");
expect(sevenDay.usedAccounts).toBeCloseTo(0.6); // 0.4 (opus binds) + 0.2
expect(sevenDay.needed).toBe(1);
expect(sevenDay.remainingAccounts).toBeCloseTo(1.4);
});
it("ignores limits without a resolvable fraction", () => {
const reports = [
makeReport("anthropic", "a@x", [
makeReport("anthropic", "account-a@example.test", [
{
id: "mystery",
label: "mystery",
@@ -116,23 +116,23 @@ describe("computeProviderWindowStats", () => {
describe("collectUnreportedAccounts", () => {
const accounts: UsageAccountIdentity[] = [
{ provider: "anthropic", type: "oauth", email: "seen@x.com" },
{ provider: "anthropic", type: "oauth", email: "missing@x.com" },
{ provider: "anthropic", type: "oauth", email: "seen@example.test" },
{ provider: "anthropic", type: "oauth", email: "missing@example.test" },
{ provider: "anthropic", type: "api_key" },
{ provider: "cerebras", type: "api_key" },
];
const reports = [makeReport("anthropic", "seen@x.com", [])];
const reports = [makeReport("anthropic", "seen@example.test", [])];
it("flags providers without reports and identified accounts missing from reports", () => {
const unreported = collectUnreportedAccounts(reports, accounts);
expect(unreported).toEqual([
{ provider: "anthropic", type: "oauth", email: "missing@x.com" },
{ provider: "anthropic", type: "oauth", email: "missing@example.test" },
{ provider: "cerebras", type: "api_key" },
]);
});
it("does not claim unattributable credentials are missing when reports carry no identity", () => {
const anonymous = [{ ...makeReport("anthropic", "seen@x.com", []), metadata: {} }];
const anonymous = [{ ...makeReport("anthropic", "seen@example.test", []), metadata: {} }];
const unreported = collectUnreportedAccounts(anonymous, accounts);
expect(unreported).toEqual([{ provider: "cerebras", type: "api_key" }]);
});
@@ -140,33 +140,47 @@ describe("collectUnreportedAccounts", () => {
describe("formatUsageBreakdown", () => {
const reports = [
makeReport("anthropic", "dum.my9@example.net", [
makeReport("anthropic", "dummy.primary@example.test", [
makeLimit({ id: "Claude 5 Hour", usedFraction: 0.84, durationMs: FIVE_HOURS, windowId: "5h" }),
]),
makeReport("anthropic", "dummy@example.net", [
makeReport("anthropic", "dummy.secondary@example.test", [
makeLimit({ id: "Claude 5 Hour", usedFraction: 0.5, durationMs: FIVE_HOURS, windowId: "5h" }),
]),
];
const accounts: UsageAccountIdentity[] = [
{ provider: "anthropic", type: "oauth", email: "dum.my9@example.net" },
{ provider: "anthropic", type: "oauth", email: "dummy@example.net" },
{ provider: "anthropic", type: "oauth", email: "dummy.primary@example.test" },
{ provider: "anthropic", type: "oauth", email: "dummy.secondary@example.test" },
{ provider: "cerebras", type: "api_key" },
];
it("renders every account: reported ones with limits, credential-only ones as no-data rows", () => {
const text = stripVTControlCharacters(formatUsageBreakdown(reports, accounts, Date.now()));
expect(text).toContain("dum.my9@example.net");
expect(text).toContain("dummy.primary@example.test");
expect(text).toContain("84.0% used");
expect(text).toContain("Cerebras");
expect(text).toContain("API key — no usage data");
expect(text).toContain("need: 5h → 2 of 2 accounts");
expect(text).toContain("capacity: 5h → 1.34/2 accounts used (0.66× quota left)");
});
it("keeps near-exhausted capacity fractional instead of rounding it to an exact need", () => {
const nearReports = [
makeReport("anthropic", "near-a@example.test", [
makeLimit({ id: "Claude 5 Hour", usedFraction: 1, durationMs: FIVE_HOURS, windowId: "5h" }),
]),
makeReport("anthropic", "near-b@example.test", [
makeLimit({ id: "Claude 5 Hour", usedFraction: 0.99, durationMs: FIVE_HOURS, windowId: "5h" }),
]),
];
const text = stripVTControlCharacters(formatUsageBreakdown(nearReports, [], Date.now()));
expect(text).toContain("capacity: 5h → 1.99/2 accounts used (0.01× quota left)");
expect(text).not.toContain("need:");
});
it("redacts account labels through the provided map without leaking the originals", () => {
const redaction = buildRedactionMap(["dum.my9@example.net", "dummy@example.net"]);
const redaction = buildRedactionMap(["dummy.primary@example.test", "dummy.secondary@example.test"]);
const text = stripVTControlCharacters(formatUsageBreakdown(reports, accounts, Date.now(), redaction));
expect(text).not.toContain("dum.my9@example.net");
expect(text).not.toContain("dummy@example.net");
expect(text).not.toContain("dummy.primary@example.test");
expect(text).not.toContain("dummy.secondary@example.test");
for (const mask of redaction.values()) expect(text).toContain(mask);
});
});
+4
View File
@@ -1,6 +1,10 @@
# Changelog
## [Unreleased]
### Fixed
- Fixed embedding provider detection to match `openrouter` by URL host, so custom embedding endpoints are now recognized correctly instead of being misclassified by substring matching
- Fixed the check for OpenRouter base URLs so only true `openrouter` hosts are treated as non-custom
## [15.10.8] - 2026-06-09
### Added
+1
View File
@@ -40,6 +40,7 @@
},
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"fastembed": "catalog:",
"lru-cache": "catalog:",
+3 -2
View File
@@ -1,5 +1,6 @@
import { homedir } from "node:os";
import { join } from "node:path";
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
import {
type Env,
envBool,
@@ -102,7 +103,7 @@ export function isApiEmbeddingModel(model = embeddingModel(), env: Env = process
if (model.startsWith("openai/") || model.includes("text-embedding") || model.startsWith("text-embedding"))
return true;
const baseUrl = envString("MNEMOPI_EMBEDDING_API_URL", envString("OPENROUTER_BASE_URL", "", env), env);
if (baseUrl && !baseUrl.includes("openrouter.ai")) return true;
if (baseUrl && !hostMatchesUrl(baseUrl, "openrouter")) return true;
return embeddingsViaApi(env);
}
@@ -110,7 +111,7 @@ export function apiEmbeddingsAvailable(env: Env = process.env): boolean {
if (embeddingsDisabled(env)) return false;
if (!isApiEmbeddingModel(embeddingModel(env), env)) return false;
const baseUrl = envString("MNEMOPI_EMBEDDING_API_URL", envString("OPENROUTER_BASE_URL", "", env), env);
return Boolean(baseUrl && !baseUrl.includes("openrouter.ai")) || Boolean(embeddingApiKey(env));
return Boolean(baseUrl && !hostMatchesUrl(baseUrl, "openrouter")) || Boolean(embeddingApiKey(env));
}
export function workingMemoryMaxItems(env: Env = process.env): number {
+4 -3
View File
@@ -1,4 +1,5 @@
import { mkdirSync } from "node:fs";
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
import {
$env,
$flag,
@@ -155,7 +156,7 @@ export function isApiModel(modelName: string): boolean {
}
const active = activeEmbeddingOptions();
const baseUrl = active?.apiUrl ?? ($env.MNEMOPI_EMBEDDING_API_URL || $env.OPENROUTER_BASE_URL);
if (baseUrl !== undefined && baseUrl !== "" && !baseUrl.includes("openrouter.ai")) {
if (baseUrl !== undefined && baseUrl !== "" && !hostMatchesUrl(baseUrl, "openrouter")) {
return true;
}
return $flag("MNEMOPI_EMBEDDINGS_VIA_API");
@@ -246,7 +247,7 @@ async function getLocalModel(): Promise<LocalEmbeddingModel | null> {
async function embedApi(texts: readonly string[]): Promise<EmbeddingMatrix | null> {
const baseUrl = embeddingBaseUrl();
const isCustom = !baseUrl.includes("openrouter.ai");
const isCustom = !hostMatchesUrl(baseUrl, "openrouter");
const apiKey = embeddingApiKey();
if (!isCustom && apiKey === "") {
return null;
@@ -335,7 +336,7 @@ export async function available(): Promise<boolean> {
}
if (isApiModel(defaultModel())) {
const baseUrl = active?.apiUrl ?? ($env.MNEMOPI_EMBEDDING_API_URL || $env.OPENROUTER_BASE_URL);
if (baseUrl !== undefined && baseUrl !== "" && !baseUrl.includes("openrouter.ai")) {
if (baseUrl !== undefined && baseUrl !== "" && !hostMatchesUrl(baseUrl, "openrouter")) {
return true;
}
return embeddingApiKey() !== "";