feat(cross-cutting): centralized model compatibility handling with catalog resolvers
- Added catalog-level host/model predicates and compat resolvers. - Extended compatibility types and model schema with timeout and replay flags. - Replaced provider-specific heuristics with shared resolver-based checks. - Updated host/identity and resolver tests to validate the new behavior.
This commit is contained in:
@@ -123,6 +123,7 @@
|
||||
},
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-catalog": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"fastembed": "catalog:",
|
||||
"lru-cache": "catalog:",
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
*/
|
||||
|
||||
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils";
|
||||
@@ -852,22 +853,6 @@ function buildAdditionalModelRequestFields(
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
|
||||
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
|
||||
* 4.6+) reject the field. Bedrock model ids are prefixed with region/inference-
|
||||
* profile slugs (e.g. `eu.anthropic.claude-opus-4-7-...`); the regex matches
|
||||
* the Claude model fragment regardless of prefix.
|
||||
*/
|
||||
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
||||
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
|
||||
const match = /claude-opus-(\d+)-(\d+)/.exec(modelId);
|
||||
if (!match) return false;
|
||||
const major = Number(match[1]);
|
||||
const minor = Number(match[2]);
|
||||
return major > 4 || (major === 4 && minor >= 7);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bedrock's wire format expects the image as `{ source: { bytes: <base64-string> }, format }`.
|
||||
* The caller already passes base64-encoded data, so no decode/re-encode round-trip is needed.
|
||||
|
||||
@@ -2,12 +2,9 @@ import * as nodeCrypto from "node:crypto";
|
||||
import * as fs from "node:fs";
|
||||
import { scheduler } from "node:timers/promises";
|
||||
import * as tls from "node:tls";
|
||||
import {
|
||||
hasOpus47ApiRestrictions,
|
||||
isAnthropicFableOrMythosModel,
|
||||
mapEffortToAnthropicAdaptiveEffort,
|
||||
supportsMidConversationSystemMessages,
|
||||
} from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { isOfficialAnthropicApiUrl, resolveAnthropicCompat } from "@oh-my-pi/pi-catalog/compat/anthropic";
|
||||
import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import { isAnthropicOAuthToken } from "@oh-my-pi/pi-catalog/utils";
|
||||
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
|
||||
@@ -181,16 +178,6 @@ function isClaudeCodeClientUserAgent(userAgent: string | undefined): userAgent i
|
||||
return userAgent.toLowerCase().startsWith("claude-cli");
|
||||
}
|
||||
|
||||
export function isAnthropicApiBaseUrl(baseUrl?: string): boolean {
|
||||
if (!baseUrl) return true;
|
||||
try {
|
||||
const url = new URL(baseUrl);
|
||||
return url.protocol.toLowerCase() === "https:" && url.hostname.toLowerCase() === "api.anthropic.com";
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
const sharedHeaders = {
|
||||
"Accept-Encoding": "gzip, deflate, br, zstd",
|
||||
Connection: "keep-alive",
|
||||
@@ -263,7 +250,7 @@ export function buildAnthropicHeaders(options: AnthropicHeaderOptions): Record<s
|
||||
"x-client-request-id": nodeCrypto.randomUUID(),
|
||||
"User-Agent": userAgent,
|
||||
};
|
||||
} else if (!isAnthropicApiBaseUrl(options.baseUrl)) {
|
||||
} else if (!isOfficialAnthropicApiUrl(options.baseUrl)) {
|
||||
return {
|
||||
...modelHeaders,
|
||||
Accept: acceptHeader,
|
||||
@@ -310,22 +297,6 @@ type AnthropicOutputConfig = NonNullable<MessageCreateParamsStreaming["output_co
|
||||
const ANTHROPIC_STOP_SEQUENCES_MAX = 4;
|
||||
let warnedStopSequencesTrim = false;
|
||||
|
||||
/**
|
||||
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
|
||||
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
|
||||
* 4.6+) reject the field.
|
||||
*/
|
||||
function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
||||
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
|
||||
// Bound the minor to non-date digits: bare dated ids like
|
||||
// `claude-opus-4-20250514` (Opus 4.0) must not parse as minor=20250514.
|
||||
const match = /claude-opus-(\d+)-(\d{1,2})(?!\d)/.exec(modelId);
|
||||
if (!match) return false;
|
||||
const major = Number(match[1]);
|
||||
const minor = Number(match[2]);
|
||||
return major > 4 || (major === 4 && minor >= 7);
|
||||
}
|
||||
|
||||
const ANTHROPIC_PROVIDER_SESSION_STATE_KEY = "anthropic-messages";
|
||||
|
||||
type AnthropicProviderSessionState = ProviderSessionState & {
|
||||
@@ -450,7 +421,9 @@ function getCacheControl(
|
||||
return { retention };
|
||||
}
|
||||
const ttl =
|
||||
retention === "long" && isAnthropicApiBaseUrl(baseUrl) && getAnthropicCompat(model).supportsLongCacheRetention
|
||||
retention === "long" &&
|
||||
isOfficialAnthropicApiUrl(baseUrl) &&
|
||||
resolveAnthropicCompat(model).supportsLongCacheRetention
|
||||
? "1h"
|
||||
: undefined;
|
||||
return {
|
||||
@@ -1146,7 +1119,7 @@ function parseAnthropicCustomHeaders(rawHeaders: string | undefined): Record<str
|
||||
export function resolveAnthropicCustomHeadersForBaseUrl(
|
||||
baseUrl: string | undefined,
|
||||
): Record<string, string> | undefined {
|
||||
if (!isFoundryEnabled() && isAnthropicApiBaseUrl(baseUrl)) return undefined;
|
||||
if (!isFoundryEnabled() && isOfficialAnthropicApiUrl(baseUrl)) return undefined;
|
||||
return parseAnthropicCustomHeaders($env.ANTHROPIC_CUSTOM_HEADERS);
|
||||
}
|
||||
|
||||
@@ -1404,24 +1377,6 @@ async function* observeDecodedAnthropicSdkEvents(
|
||||
}
|
||||
}
|
||||
|
||||
function getAnthropicCompat(
|
||||
model: Model<"anthropic-messages">,
|
||||
): Required<NonNullable<Model<"anthropic-messages">["compat"]>> {
|
||||
return {
|
||||
disableStrictTools: model.compat?.disableStrictTools ?? false,
|
||||
disableAdaptiveThinking: model.compat?.disableAdaptiveThinking ?? false,
|
||||
supportsEagerToolInputStreaming: model.compat?.supportsEagerToolInputStreaming ?? true,
|
||||
supportsLongCacheRetention: model.compat?.supportsLongCacheRetention ?? true,
|
||||
supportsMidConversationSystem:
|
||||
model.compat?.supportsMidConversationSystem ??
|
||||
// First-party Claude API only. Bedrock/Vertex/Foundry and other
|
||||
// Anthropic-compatible proxies reject the role; gate auto-detection on
|
||||
// the canonical api.anthropic.com host plus a supported model id.
|
||||
(isAnthropicApiBaseUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id)),
|
||||
supportsForcedToolChoice: model.compat?.supportsForcedToolChoice ?? !isAnthropicFableOrMythosModel(model.id),
|
||||
};
|
||||
}
|
||||
|
||||
const PROVIDER_MAX_RETRIES = 3;
|
||||
const PROVIDER_BASE_DELAY_MS = 2000;
|
||||
|
||||
@@ -1626,7 +1581,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
||||
const sendsAdaptiveEffortPin =
|
||||
options?.thinkingEnabled === false &&
|
||||
model.thinking?.mode === "anthropic-adaptive" &&
|
||||
!getAnthropicCompat(model).disableAdaptiveThinking;
|
||||
!resolveAnthropicCompat(model).disableAdaptiveThinking;
|
||||
if (
|
||||
model.reasoning &&
|
||||
(options?.thinkingEnabled || sendsAdaptiveEffortPin) &&
|
||||
@@ -1635,7 +1590,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
||||
extraBetas.push(effortBeta);
|
||||
}
|
||||
if (
|
||||
getAnthropicCompat(model).supportsMidConversationSystem &&
|
||||
resolveAnthropicCompat(model).supportsMidConversationSystem &&
|
||||
!extraBetas.includes(midConversationSystemBeta)
|
||||
) {
|
||||
// convertAnthropicMessages may upgrade developer turns to the
|
||||
@@ -2332,7 +2287,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
isOAuth,
|
||||
claudeCodeSessionId,
|
||||
} = args;
|
||||
const compat = getAnthropicCompat(model);
|
||||
const compat = resolveAnthropicCompat(model);
|
||||
const needsInterleavedBeta = interleavedThinking && !supportsAdaptiveThinkingDisplay(model.id);
|
||||
const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming;
|
||||
const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey);
|
||||
@@ -2443,7 +2398,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
const authorizationHeader = getHeaderCaseInsensitive(defaultHeaders, "Authorization");
|
||||
const shouldSuppressClientApiKey =
|
||||
!oauthToken &&
|
||||
!isAnthropicApiBaseUrl(baseUrl) &&
|
||||
!isOfficialAnthropicApiUrl(baseUrl) &&
|
||||
typeof authorizationHeader === "string" &&
|
||||
/^Bearer\s+/i.test(authorizationHeader);
|
||||
|
||||
@@ -2777,7 +2732,7 @@ function buildParams(
|
||||
context.tools,
|
||||
isOAuthToken,
|
||||
disableStrictTools || model.provider === "github-copilot",
|
||||
getAnthropicCompat(model).supportsEagerToolInputStreaming,
|
||||
resolveAnthropicCompat(model).supportsEagerToolInputStreaming,
|
||||
);
|
||||
} else if (isOAuthToken) {
|
||||
tools = [];
|
||||
@@ -2800,7 +2755,7 @@ function buildParams(
|
||||
if (options?.thinkingEnabled) {
|
||||
const mode = model.thinking?.mode;
|
||||
const effort = resolveAnthropicAdaptiveEffort(model, options);
|
||||
const compat = getAnthropicCompat(model);
|
||||
const compat = resolveAnthropicCompat(model);
|
||||
if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
|
||||
const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" };
|
||||
// Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking
|
||||
@@ -2823,7 +2778,7 @@ function buildParams(
|
||||
if (mode === "anthropic-budget-effort" && effort) outputConfigEffort = effort;
|
||||
}
|
||||
} else if (options?.thinkingEnabled === false) {
|
||||
const compat = getAnthropicCompat(model);
|
||||
const compat = resolveAnthropicCompat(model);
|
||||
if (model.thinking?.mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) {
|
||||
// Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject
|
||||
// `thinking.type: "disabled"` — adaptive thinking cannot be switched off.
|
||||
@@ -2912,7 +2867,7 @@ function buildParams(
|
||||
// request succeeds; the tool stays available and the caller's prompt steers
|
||||
// the model toward it.
|
||||
const choiceType = params.tool_choice?.type;
|
||||
if ((choiceType === "any" || choiceType === "tool") && !getAnthropicCompat(model).supportsForcedToolChoice) {
|
||||
if ((choiceType === "any" || choiceType === "tool") && !resolveAnthropicCompat(model).supportsForcedToolChoice) {
|
||||
params.tool_choice = { type: "auto" };
|
||||
}
|
||||
}
|
||||
@@ -2926,52 +2881,6 @@ function buildParams(
|
||||
return params;
|
||||
}
|
||||
|
||||
/**
|
||||
* Z.AI's Anthropic-compatible proxy at `api.z.ai/api/anthropic` deserializes
|
||||
* tool_result blocks into a Python class that accesses `.id`, even though
|
||||
* Anthropic's standard tool_result schema only carries `tool_use_id`. Detect
|
||||
* that endpoint so we can emit the non-standard alias for it without
|
||||
* polluting requests to api.anthropic.com or other compatible proxies.
|
||||
* See: https://github.com/can1357/oh-my-pi/issues/814
|
||||
*/
|
||||
function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean {
|
||||
if (model.provider === "zai") return true;
|
||||
const baseUrl = model.baseUrl;
|
||||
if (!baseUrl) return false;
|
||||
try {
|
||||
return new URL(baseUrl).hostname.toLowerCase() === "api.z.ai";
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Returns true when unsigned `thinking` blocks from prior assistant turns should
|
||||
* be replayed as Anthropic-native thinking instead of demoted to text.
|
||||
*
|
||||
* Official Anthropic (matched via `isAnthropicApiBaseUrl`, which intentionally
|
||||
* treats a missing baseUrl as official since `resolveAnthropicBaseUrl` routes
|
||||
* it to `https://api.anthropic.com`) enforces signature-based thinking-chain
|
||||
* integrity, so unsigned blocks must remain text there. Anthropic-compatible
|
||||
* reasoning endpoints commonly emit unsigned thinking blocks while still
|
||||
* expecting them back as `type: "thinking"` on continuation; demoting them
|
||||
* loses the model's reasoning chain and can destabilize the next tool-call
|
||||
* arguments (#2005). Known non-signing hosts are also preserved for
|
||||
* compatibility.
|
||||
*/
|
||||
function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">, baseUrl: string | undefined): boolean {
|
||||
if (model.provider === "zai" || model.provider === "deepseek") return true;
|
||||
if (baseUrl) {
|
||||
try {
|
||||
const hostname = new URL(baseUrl).hostname.toLowerCase();
|
||||
if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true;
|
||||
} catch {
|
||||
// Fall through to the protocol-level reasoning rule below.
|
||||
}
|
||||
}
|
||||
return model.reasoning && !isAnthropicApiBaseUrl(baseUrl);
|
||||
}
|
||||
|
||||
function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam {
|
||||
const block: ContentBlockParam = {
|
||||
type: "tool_result",
|
||||
@@ -2979,7 +2888,7 @@ function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResul
|
||||
content: convertContentBlocks(msg.content, model.input.includes("image")),
|
||||
is_error: msg.isError,
|
||||
};
|
||||
if (isZaiAnthropicEndpoint(model)) {
|
||||
if (resolveAnthropicCompat(model).requiresToolResultId) {
|
||||
// Z.AI workaround (issue #814): include `id` aliased to `tool_use_id`.
|
||||
(block as unknown as Record<string, unknown>).id = msg.toolCallId;
|
||||
}
|
||||
@@ -3092,7 +3001,7 @@ export function convertAnthropicMessages(
|
||||
}
|
||||
if (block.thinking.trim().length === 0) continue;
|
||||
if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) {
|
||||
if (shouldReplayUnsignedThinking(model, baseUrl)) {
|
||||
if (resolveAnthropicCompat(model, baseUrl).replayUnsignedThinking) {
|
||||
blocks.push({
|
||||
type: "thinking",
|
||||
thinking: block.thinking.toWellFormed(),
|
||||
@@ -3170,7 +3079,7 @@ export function convertAnthropicMessages(
|
||||
// never consecutive. Requiring the next param to be `assistant` (or absent)
|
||||
// covers both the "followed by assistant / last" and "no consecutive system"
|
||||
// constraints. Anything that does not qualify stays a `user` message.
|
||||
if (developerParamIndices.length > 0 && getAnthropicCompat(model).supportsMidConversationSystem) {
|
||||
if (developerParamIndices.length > 0 && resolveAnthropicCompat(model).supportsMidConversationSystem) {
|
||||
for (const idx of developerParamIndices) {
|
||||
const followsUser = idx > 0 && params[idx - 1]?.role === "user";
|
||||
const next = params[idx + 1];
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
|
||||
import { AzureOpenAI, APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai";
|
||||
import type {
|
||||
@@ -31,7 +32,7 @@ import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schem
|
||||
import { createSdkStreamRequestOptions } from "../utils/sdk-stream-timeout";
|
||||
import { notifyRawSseEvent } from "../utils/sse-debug";
|
||||
import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
|
||||
import { getOpenAIResponsesCacheSessionId, supportsDeveloperRole } from "./openai-responses";
|
||||
import { getOpenAIResponsesCacheSessionId } from "./openai-responses";
|
||||
import {
|
||||
appendResponsesToolResultMessages,
|
||||
applyCommonResponsesSamplingParams,
|
||||
@@ -337,7 +338,10 @@ function convertMessages(
|
||||
|
||||
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
||||
if (systemPrompts.length > 0) {
|
||||
const role = model.reasoning && supportsDeveloperRole(resolvedBaseUrl ?? model) ? "developer" : "system";
|
||||
const role =
|
||||
model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole
|
||||
? "developer"
|
||||
: "system";
|
||||
for (const systemPrompt of systemPrompts) {
|
||||
messages.push({ role, content: systemPrompt });
|
||||
}
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id";
|
||||
import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import { isDeepseekModelIdOrName, isKimiModelId } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
|
||||
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
|
||||
@@ -390,44 +392,6 @@ function getTrailingPartialDeepseekToken(text: string): string {
|
||||
const OPENAI_COMPLETIONS_FIRST_EVENT_TIMEOUT_MESSAGE =
|
||||
"OpenAI completions stream timed out while waiting for the first event";
|
||||
|
||||
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
|
||||
const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
|
||||
|
||||
// DeepSeek V4 reasoning models on the official api.deepseek.com emit no SSE
|
||||
// bytes while the model finishes its private chain-of-thought, which routinely
|
||||
// takes longer than the generic 100s first-event floor under load (issue
|
||||
// #2177). Mirror the GLM coding-plan widening: a 5-minute idle floor lifts the
|
||||
// first-event watchdog (it floors at idle) without changing the runtime
|
||||
// streaming behavior, so reasoning warm-ups stop aborting and retrying.
|
||||
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
|
||||
|
||||
function isDirectDeepseekReasoningModel(model: Model<"openai-completions">): boolean {
|
||||
if (!model.reasoning) return false;
|
||||
if (model.provider === "deepseek") return true;
|
||||
return model.baseUrl.toLowerCase().includes("api.deepseek.com");
|
||||
}
|
||||
|
||||
/** Returns the widened OpenAI stream watchdog floor for slow reasoning models hosted on OpenAI-compatible endpoints. */
|
||||
export function getOpenAICompletionsStreamIdleTimeoutFallbackMs(
|
||||
model: Model<"openai-completions">,
|
||||
): number | undefined {
|
||||
if (GLM_CODING_PLAN_MODEL_PATTERN.test(model.id)) {
|
||||
if (model.provider === "zhipu-coding-plan" || model.provider === "zai")
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
|
||||
const baseUrl = model.baseUrl.toLowerCase();
|
||||
if (baseUrl.includes("open.bigmodel.cn") || baseUrl.includes("api.z.ai")) {
|
||||
return GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS;
|
||||
}
|
||||
}
|
||||
|
||||
if (isDirectDeepseekReasoningModel(model)) {
|
||||
return DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
async function* observeDecodedOpenAICompletionChunks(
|
||||
chunks: AsyncIterable<ChatCompletionChunk>,
|
||||
observer: (event: RawSseEvent) => void,
|
||||
@@ -468,7 +432,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
|
||||
try {
|
||||
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
||||
const idleTimeoutFallbackMs = getOpenAICompletionsStreamIdleTimeoutFallbackMs(model);
|
||||
const idleTimeoutFallbackMs = resolveOpenAICompat(model).streamIdleTimeoutMs;
|
||||
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs);
|
||||
const firstEventTimeoutMs =
|
||||
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
|
||||
@@ -576,7 +540,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
// though tool calls are also surfaced structurally. Strip the leaked markers
|
||||
// so users don't see raw `<|...|>` tokens.
|
||||
const stripDeepseekChatTemplateTokens =
|
||||
/deepseek/i.test(model.id) && (model.provider === "nvidia" || model.provider === "deepseek");
|
||||
isDeepseekModelIdOrName(model.id) && (model.provider === "nvidia" || model.provider === "deepseek");
|
||||
type ToolCallStreamBlock = ToolCall & {
|
||||
partialArgs?: string | Record<string, unknown>;
|
||||
streamIndex?: number;
|
||||
@@ -1252,8 +1216,8 @@ function buildParams(
|
||||
compat.allowsSyntheticReasoningContentForToolCalls = false;
|
||||
compat.reasoningContentField = "reasoning_content";
|
||||
}
|
||||
const isKimiModelId = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
|
||||
const isOpenRouter = model.baseUrl.includes("openrouter.ai");
|
||||
const isKimiFamilyModel = isKimiModelId(model.id);
|
||||
const isOpenRouter = modelMatchesHost(model, "openrouter");
|
||||
const messages = convertMessages(model, context, compat);
|
||||
maybeAddAnthropicCacheControl(compat, messages);
|
||||
const supportsReasoningParams = model.provider !== "github-copilot";
|
||||
@@ -1266,14 +1230,14 @@ function buildParams(
|
||||
// before the final answer. Always send max_tokens — match the same
|
||||
// Kimi-family regex used by the compat detector.
|
||||
// Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts.
|
||||
const requestedMaxTokens = options?.maxTokens ?? (isKimiModelId ? model.maxTokens : undefined);
|
||||
const requestedMaxTokens = options?.maxTokens ?? (isKimiFamilyModel ? model.maxTokens : undefined);
|
||||
// OpenRouter fans out to upstreams whose output caps differ from the catalog
|
||||
// value (which tracks the highest-cap provider). A max_tokens above the routed
|
||||
// upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras
|
||||
// GLM-4.7, ~40k) for a higher-cap one, defeating `provider.order`/`only`. Omit
|
||||
// it for OpenRouter so each upstream self-caps and routing is honored. Kimi is
|
||||
// exempt — it derives TPM rate limits from max_tokens (see above).
|
||||
const omitMaxTokensForRouting = isOpenRouter && !isKimiModelId;
|
||||
const omitMaxTokensForRouting = isOpenRouter && !isKimiFamilyModel;
|
||||
const effectiveMaxTokens =
|
||||
requestedMaxTokens === undefined || omitMaxTokensForRouting
|
||||
? undefined
|
||||
@@ -1442,12 +1406,12 @@ function buildParams(
|
||||
}
|
||||
|
||||
// OpenRouter provider routing preferences
|
||||
if (model.baseUrl.includes("openrouter.ai") && compat.openRouterRouting) {
|
||||
if (modelMatchesHost(model, "openrouter") && compat.openRouterRouting) {
|
||||
params.provider = compat.openRouterRouting;
|
||||
}
|
||||
|
||||
// Vercel AI Gateway provider routing preferences
|
||||
if (model.baseUrl.includes("ai-gateway.vercel.sh") && model.compat?.vercelGatewayRouting) {
|
||||
if (modelMatchesHost(model, "vercelAIGateway") && model.compat?.vercelGatewayRouting) {
|
||||
const routing = model.compat.vercelGatewayRouting;
|
||||
if (routing.only || routing.order) {
|
||||
const gatewayOptions: Record<string, string[]> = {};
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot";
|
||||
import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
|
||||
import OpenAI, { APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai";
|
||||
@@ -10,7 +12,6 @@ import type {
|
||||
import { getEnvApiKey } from "../stream";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
CacheRetention,
|
||||
Context,
|
||||
FetchImpl,
|
||||
MessageAttribution,
|
||||
@@ -69,20 +70,6 @@ import {
|
||||
} from "./openai-responses-shared";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
|
||||
/**
|
||||
* Get prompt cache retention based on cacheRetention and base URL.
|
||||
* Only applies to direct OpenAI API calls (api.openai.com).
|
||||
*/
|
||||
function getPromptCacheRetention(baseUrl: string, cacheRetention: CacheRetention): "24h" | undefined {
|
||||
if (cacheRetention !== "long") {
|
||||
return undefined;
|
||||
}
|
||||
if (baseUrl.includes("api.openai.com")) {
|
||||
return "24h";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function normalizeOpenAIResponsesPromptCacheKey(sessionId: string | undefined): string | undefined {
|
||||
if (!sessionId || sessionId.length === 0) return undefined;
|
||||
const wellFormed = sessionId.toWellFormed();
|
||||
@@ -442,13 +429,14 @@ function buildParams(
|
||||
): OpenAIResponsesSamplingParams {
|
||||
const strictResponsesPairing =
|
||||
options?.strictResponsesPairing ??
|
||||
(isAzureOpenAIBaseUrl(model.baseUrl ?? "") || model.provider === "github-copilot");
|
||||
(hostMatchesUrl(model.baseUrl ?? "", "azureOpenAI") || model.provider === "github-copilot");
|
||||
const messages = convertConversationMessages(model, context, strictResponsesPairing, providerSessionState, options);
|
||||
|
||||
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
||||
let systemInstructions: string | undefined;
|
||||
if (systemPrompts.length > 0) {
|
||||
const needsDeveloperRole = model.reasoning && supportsDeveloperRole(resolvedBaseUrl ?? model);
|
||||
const needsDeveloperRole =
|
||||
model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole;
|
||||
if (needsDeveloperRole) {
|
||||
// Reasoning models on known OpenAI-compatible endpoints require the
|
||||
// `developer` role. Send all system prompts inline in `input`.
|
||||
@@ -472,7 +460,10 @@ function buildParams(
|
||||
stream: true,
|
||||
prompt_cache_key: promptCacheKey,
|
||||
prompt_cache_retention: promptCacheKey
|
||||
? getPromptCacheRetention(resolvedBaseUrl ?? model.baseUrl, cacheRetention)
|
||||
? cacheRetention === "long" &&
|
||||
resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsLongPromptCacheRetention
|
||||
? "24h"
|
||||
: undefined
|
||||
: undefined,
|
||||
store: false,
|
||||
stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined,
|
||||
@@ -485,7 +476,11 @@ function buildParams(
|
||||
// `StreamOptions.frequencyPenalty` is intentionally dropped for this provider.
|
||||
|
||||
if (context.tools) {
|
||||
params.tools = convertTools(context.tools, supportsStrictMode(model), model);
|
||||
params.tools = convertTools(
|
||||
context.tools,
|
||||
resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsStrictMode,
|
||||
model,
|
||||
);
|
||||
if (options?.toolChoice) {
|
||||
params.tool_choice = mapOpenAIResponsesToolChoiceForTools(options.toolChoice, context.tools, model);
|
||||
}
|
||||
@@ -528,34 +523,6 @@ function mapReasoningEffort(
|
||||
return reasoningEffortMap?.[effort] ?? effort;
|
||||
}
|
||||
|
||||
function isAzureOpenAIBaseUrl(baseUrl: string): boolean {
|
||||
return baseUrl.includes(".openai.azure.com") || baseUrl.includes("azure.com/openai");
|
||||
}
|
||||
|
||||
function supportsStrictMode(model: Model<"openai-responses">): boolean {
|
||||
if (model.provider === "openai" || model.provider === "azure" || model.provider === "github-copilot") return true;
|
||||
|
||||
const baseUrl = model.baseUrl.toLowerCase();
|
||||
return (
|
||||
baseUrl.includes("api.openai.com") ||
|
||||
baseUrl.includes(".openai.azure.com") ||
|
||||
baseUrl.includes("models.inference.ai.azure.com")
|
||||
);
|
||||
}
|
||||
|
||||
export function supportsDeveloperRole(modelOrBaseUrl: Pick<Model, "provider" | "baseUrl"> | string): boolean {
|
||||
const baseUrl =
|
||||
typeof modelOrBaseUrl === "string" ? modelOrBaseUrl.toLowerCase() : (modelOrBaseUrl.baseUrl ?? "").toLowerCase();
|
||||
return (
|
||||
baseUrl.includes("api.openai.com") ||
|
||||
baseUrl.includes(".openai.azure.com") ||
|
||||
baseUrl.includes("azure.com/openai") ||
|
||||
baseUrl.includes("models.inference.ai.azure.com") ||
|
||||
baseUrl.includes("githubcopilot.com") ||
|
||||
baseUrl.includes("copilot-api.")
|
||||
);
|
||||
}
|
||||
|
||||
function convertConversationMessages(
|
||||
model: Model<"openai-responses">,
|
||||
context: Context,
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
import { isDashscopeCompatibleModeUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import { isQwenModelId } from "@oh-my-pi/pi-catalog/identity";
|
||||
|
||||
import type { ImageContent, Model, TextContent } from "../types";
|
||||
|
||||
export const NON_VISION_IMAGE_PLACEHOLDER = "[image omitted: model does not support vision]";
|
||||
@@ -42,11 +45,10 @@ export function joinTextWithImagePlaceholder(text: string, omittedImages: boolea
|
||||
* provider (issue #1859) can't drive the request into an unrecoverable 400.
|
||||
*/
|
||||
export function isDashscopeCompatibleModeTextOnlyQwen(model: Model<"openai-completions">): boolean {
|
||||
const baseUrl = model.baseUrl.toLowerCase();
|
||||
if (!baseUrl.includes("dashscope") || !baseUrl.includes("aliyuncs.com") || !baseUrl.includes("/compatible-mode")) {
|
||||
if (!isDashscopeCompatibleModeUrl(model.baseUrl)) {
|
||||
return false;
|
||||
}
|
||||
const id = model.id.toLowerCase();
|
||||
if (!id.includes("qwen")) return false;
|
||||
if (!isQwenModelId(model.id)) return false;
|
||||
return /\bqwen(?:[\d.]+)?-max\b/.test(id) || /\bqwen(?:[\d.]+)?-coder\b/.test(id);
|
||||
}
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import type { Effort } from "@oh-my-pi/pi-catalog/effort";
|
||||
import { isVertexExpressOpenAIUrl, isVertexRawPredictUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import {
|
||||
mapEffortToAnthropicAdaptiveEffort,
|
||||
mapEffortToGoogleThinkingLevel,
|
||||
@@ -65,8 +66,8 @@ import { withRequestDebugFetch } from "./utils/request-debug";
|
||||
function isGoogleVertexAuthenticatedModel(model: Model<Api>): boolean {
|
||||
return (
|
||||
model.provider === "google-vertex" &&
|
||||
((model.api === "openai-completions" && model.baseUrl.includes("/endpoints/openapi")) ||
|
||||
(model.api === "anthropic-messages" && model.baseUrl.includes(":streamRawPredict")))
|
||||
((model.api === "openai-completions" && isVertexExpressOpenAIUrl(model.baseUrl)) ||
|
||||
(model.api === "anthropic-messages" && isVertexRawPredictUrl(model.baseUrl)))
|
||||
);
|
||||
}
|
||||
|
||||
@@ -78,7 +79,7 @@ function createVertexAuthenticatedFetch(options: StreamOptions | undefined): Fet
|
||||
headers.set("Authorization", `Bearer ${token}`);
|
||||
const rewritten = resolveVertexRequest(input);
|
||||
const url = rewritten instanceof Request ? rewritten.url : rewritten.toString();
|
||||
if (isVertexAnthropicRawPredict(url)) {
|
||||
if (isVertexRawPredictUrl(url)) {
|
||||
const bodyText = await readVertexRequestBody(rewritten, init);
|
||||
const transformed = transformVertexAnthropicBody(bodyText);
|
||||
return baseFetch(url, {
|
||||
@@ -93,10 +94,6 @@ function createVertexAuthenticatedFetch(options: StreamOptions | undefined): Fet
|
||||
return Object.assign(vertexFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {});
|
||||
}
|
||||
|
||||
function isVertexAnthropicRawPredict(url: string): boolean {
|
||||
return url.includes(":streamRawPredict") || url.includes(":rawPredict");
|
||||
}
|
||||
|
||||
async function readVertexRequestBody(input: string | URL | Request, init: RequestInit | undefined): Promise<string> {
|
||||
if (input instanceof Request) return input.clone().text();
|
||||
const body = init?.body;
|
||||
|
||||
@@ -13,6 +13,8 @@
|
||||
* deltas for thinking blocks, and holds partial tags across chunk boundaries.
|
||||
*/
|
||||
|
||||
import { isDeepseekModelIdOrName } from "@oh-my-pi/pi-catalog/identity";
|
||||
|
||||
import { parseJsonWithRepair } from "./json-parse";
|
||||
|
||||
const KIMI_SECTION_BEGIN = "<|tool_calls_section_begin|>";
|
||||
@@ -622,7 +624,7 @@ export function modelMayLeakKimiToolCalls(provider: string, modelId: string): bo
|
||||
|
||||
/** Cheap model/provider gate for DeepSeek DSML envelope leaks. */
|
||||
export function modelMayLeakDsmlToolCalls(provider: string, modelId: string): boolean {
|
||||
if (!/deepseek/i.test(modelId)) return false;
|
||||
if (!isDeepseekModelIdOrName(modelId)) return false;
|
||||
return (
|
||||
provider === "ollama" ||
|
||||
provider === "ollama-cloud" ||
|
||||
|
||||
@@ -161,7 +161,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
|
||||
});
|
||||
|
||||
it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => {
|
||||
// `isAnthropicApiBaseUrl(undefined) === true` because the actual HTTP
|
||||
// `isOfficialAnthropicApiUrl(undefined) === true` because the actual HTTP
|
||||
// dispatch falls back to https://api.anthropic.com. Same-id custom
|
||||
// overrides that only tweak model metadata (no baseUrl override) must
|
||||
// not regress to native-thinking replay against the first-party API.
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import {
|
||||
getOpenAICompletionsStreamIdleTimeoutFallbackMs,
|
||||
isOpenAICompletionsProgressChunk,
|
||||
streamOpenAICompletions,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
const openAICompletionsModel = {
|
||||
@@ -78,7 +78,7 @@ function createKeepaliveOnlyCompletionsResponse(modelId: string, signal: AbortSi
|
||||
});
|
||||
}
|
||||
|
||||
describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
describe("resolveOpenAICompat stream idle timeout", () => {
|
||||
it("widens GLM 5.1 coding-plan stream watchdogs", () => {
|
||||
const model = {
|
||||
...openAICompletionsModel,
|
||||
@@ -88,7 +88,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4",
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
|
||||
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000);
|
||||
});
|
||||
|
||||
it("also widens custom Z.AI OpenAI-compatible GLM 5.1 endpoints", () => {
|
||||
@@ -100,7 +100,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
baseUrl: "https://api.z.ai/api/coding/paas/v4",
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(600_000);
|
||||
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000);
|
||||
});
|
||||
|
||||
it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => {
|
||||
@@ -113,7 +113,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
reasoning: true,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
|
||||
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000);
|
||||
});
|
||||
|
||||
it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => {
|
||||
@@ -126,7 +126,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
reasoning: true,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBe(300_000);
|
||||
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000);
|
||||
});
|
||||
|
||||
it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => {
|
||||
@@ -139,7 +139,7 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
reasoning: false,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
|
||||
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined();
|
||||
});
|
||||
|
||||
it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => {
|
||||
@@ -152,11 +152,11 @@ describe("getOpenAICompletionsStreamIdleTimeoutFallbackMs", () => {
|
||||
reasoning: true,
|
||||
} satisfies Model<"openai-completions">;
|
||||
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)).toBeUndefined();
|
||||
expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined();
|
||||
});
|
||||
|
||||
it("keeps ordinary OpenAI-compatible models on the global timeout", () => {
|
||||
expect(getOpenAICompletionsStreamIdleTimeoutFallbackMs(openAICompletionsModel)).toBeUndefined();
|
||||
expect(resolveOpenAICompat(openAICompletionsModel).streamIdleTimeoutMs).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
@@ -1,78 +1,78 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { supportsDeveloperRole } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import type { Model } from "@oh-my-pi/pi-ai/types";
|
||||
|
||||
describe("supportsDeveloperRole", () => {
|
||||
import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai";
|
||||
|
||||
describe("resolveOpenAIResponsesCompat supportsDeveloperRole", () => {
|
||||
it("returns true for openai provider with official API base URL", () => {
|
||||
const model = { provider: "openai", baseUrl: "https://api.openai.com/v1" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "openai", baseUrl: "https://api.openai.com/v1" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns false for openai provider with custom proxy base URL", () => {
|
||||
const model = { provider: "openai", baseUrl: "https://my-proxy.example.com/v1" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(false);
|
||||
const model = { provider: "openai", baseUrl: "https://my-proxy.example.com/v1" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
|
||||
});
|
||||
|
||||
it("returns true for github-copilot provider", () => {
|
||||
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns false for github-copilot provider with custom proxy base URL", () => {
|
||||
const model = { provider: "github-copilot", baseUrl: "https://proxy.example.com/v1" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(false);
|
||||
const model = { provider: "github-copilot", baseUrl: "https://proxy.example.com/v1" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
|
||||
});
|
||||
|
||||
it("returns true for Azure OpenAI base URL", () => {
|
||||
const model = { provider: "azure-openai", baseUrl: "https://my-resource.openai.azure.com/openai" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "azure-openai", baseUrl: "https://my-resource.openai.azure.com/openai" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for Azure AI Inference base URL", () => {
|
||||
const model = {
|
||||
provider: "azure-openai",
|
||||
baseUrl: "https://models.inference.ai.azure.com/v1/chat/completions",
|
||||
} as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
};
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for api.openai.com base URL", () => {
|
||||
const model = { provider: "custom", baseUrl: "https://api.openai.com/v1/chat/completions" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "custom", baseUrl: "https://api.openai.com/v1/chat/completions" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns false for generic third-party provider", () => {
|
||||
const model = { provider: "custom", baseUrl: "https://api.example.com/v1" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(false);
|
||||
const model = { provider: "custom", baseUrl: "https://api.example.com/v1" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false for local/localhost endpoints", () => {
|
||||
const model = { provider: "custom", baseUrl: "http://localhost:8080/v1" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(false);
|
||||
const model = { provider: "custom", baseUrl: "http://localhost:8080/v1" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false);
|
||||
});
|
||||
|
||||
it("is case-insensitive for base URL matching", () => {
|
||||
const model = { provider: "custom", baseUrl: "https://API.OPENAI.COM/v1" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "custom", baseUrl: "https://API.OPENAI.COM/v1" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for azure.com/openai base URL", () => {
|
||||
const model = { provider: "custom", baseUrl: "https://azure.com/openai/deployments/my-model" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "custom", baseUrl: "https://azure.com/openai/deployments/my-model" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for github-copilot provider with api.githubcopilot.com", () => {
|
||||
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for github-copilot provider with api.enterprise.githubcopilot.com", () => {
|
||||
const model = { provider: "github-copilot", baseUrl: "https://api.enterprise.githubcopilot.com" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "github-copilot", baseUrl: "https://api.enterprise.githubcopilot.com" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for github-copilot provider with copilot-api enterprise domain", () => {
|
||||
const model = { provider: "github-copilot", baseUrl: "https://copilot-api.mycompany.com" } as Model;
|
||||
expect(supportsDeveloperRole(model)).toBe(true);
|
||||
const model = { provider: "github-copilot", baseUrl: "https://copilot-api.mycompany.com" };
|
||||
expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `hostMatchesUrl`, `modelMatchesHost`, and endpoint-shape helpers in the new `hosts` module for consistent provider/baseUrl matching
|
||||
- Added `streamIdleTimeoutMs` to `OpenAICompat` and now auto-populated it for GLM coding-plan and direct DeepSeek reasoning models
|
||||
- Added `supportsLongPromptCacheRetention` and the OpenAI Responses helpers `detectOpenAIResponsesCompat`/`resolveOpenAIResponsesCompat`
|
||||
- Added anthropic-messages compatibility resolution with new `AnthropicCompat` fields `requiresToolResultId` and `replayUnsignedThinking`
|
||||
- New `@oh-my-pi/pi-catalog` package: the model catalog extracted from `@oh-my-pi/pi-ai`. Owns the bundled `models.json` and its generation pipeline (`scripts/generate-models.ts`), the core model data types (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces), thinking metadata enrichment and generated policies (`model-thinking.ts`), the SQLite model cache and model manager, per-provider discovery factories (`provider-models/`), the discovery protocol clients (`discovery/`), and the new `CATALOG_PROVIDERS` table — the single source of truth for provider ids, default models, and discovery wiring (`KnownProvider`, `PROVIDER_DESCRIPTORS`, and `DEFAULT_MODEL_PER_PROVIDER` are derived from it).
|
||||
- New `identity/` module centralizing model-identity concerns that were previously duplicated across packages: family classification and version parsing (`identity/classify.ts`, extracted from pi-ai's `model-thinking` internals), canonical model equivalence with injected reference data (`identity/equivalence.ts`, from coding-agent's `model-equivalence`), proxy/reseller reference lookup (`identity/reference.ts`, from coding-agent's `model-registry`), bracket-affix and id-segment helpers (`identity/id.ts`), a single trailing-marker vocabulary with canonical vs reference flavors (`identity/markers.ts` — `search` stays reference-only so Perplexity's `sonar-pro-search` remains canonical-distinct), and provider priority ordering (`identity/priority.ts`).
|
||||
- Memoized bundled-reference accessors (`getBundledCanonicalReferenceData` / `getBundledModelReferenceIndex` in `identity/bundled.ts`): one lazy walk of the bundled catalog feeds both canonical equivalence and proxy-reference lookup, so consumers no longer hand-roll the glue.
|
||||
@@ -11,6 +14,8 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed OpenAI compatibility detection to use shared host classifiers (`modelMatchesHost`/`hostMatchesUrl`) with normalized matching instead of raw URL substring checks
|
||||
- Changed `hostMatchesUrl`/`modelMatchesHost` usage in compatibility detection to reduce mismatches across case variants and provider alias hosts
|
||||
- Provider catalog entries now carry the runtime API-key env fallback as an ordered `envVars` list; `catalogDiscovery.envVars` became an optional generation-time override (only `cursor` and `vercel-ai-gateway` differ) and `PROVIDER_DESCRIPTORS` materializes the resolved list for `generate-models.ts`.
|
||||
- `Model`'s api parameter now defaults to `Api` instead of `any` (`Model<TApi extends Api = Api>`), so bare `Model` no longer behaves as `Model<any>` at call sites.
|
||||
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
/**
|
||||
* Anthropic-messages compatibility detection and resolution — the
|
||||
* anthropic-side analogue of `./openai`. Detect-time defaults come from
|
||||
* provider ids, strict URL checks, and model-id classification; explicit
|
||||
* `model.compat` overrides always win.
|
||||
*/
|
||||
import { isAnthropicFableOrMythosModel, supportsMidConversationSystemMessages } from "../model-thinking";
|
||||
import type { AnthropicCompat, Model } from "../types";
|
||||
|
||||
/**
|
||||
* Official first-party Anthropic API check (https + exact host). A missing
|
||||
* baseUrl is official on purpose: request dispatch falls back to
|
||||
* `https://api.anthropic.com`. Strict URL parsing (not substring) because the
|
||||
* callers gate auth flows and body mutations on it.
|
||||
*/
|
||||
export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean {
|
||||
if (!baseUrl) return true;
|
||||
try {
|
||||
const url = new URL(baseUrl);
|
||||
return url.protocol.toLowerCase() === "https:" && url.hostname.toLowerCase() === "api.anthropic.com";
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Z.AI's Anthropic-compatible proxy (`api.z.ai/api/anthropic`), strict-host matched. */
|
||||
function isZaiAnthropicUrl(baseUrl: string | undefined): boolean {
|
||||
if (!baseUrl) return false;
|
||||
try {
|
||||
return new URL(baseUrl).hostname.toLowerCase() === "api.z.ai";
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/** DeepSeek-operated host, strict-host matched (`api.deepseek.com` or any `*.deepseek.com`). */
|
||||
function isDeepseekHostUrl(baseUrl: string | undefined): boolean {
|
||||
if (!baseUrl) return false;
|
||||
try {
|
||||
const hostname = new URL(baseUrl).hostname.toLowerCase();
|
||||
return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com");
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
export type ResolvedAnthropicCompat = Required<AnthropicCompat>;
|
||||
|
||||
/**
|
||||
* Detect anthropic-messages compatibility defaults from provider/baseUrl/model id.
|
||||
* @param resolvedBaseUrl - Effective request base URL when it differs from
|
||||
* `model.baseUrl` (e.g. an options-level override).
|
||||
*/
|
||||
export function detectAnthropicCompat(
|
||||
model: Model<"anthropic-messages">,
|
||||
resolvedBaseUrl?: string,
|
||||
): ResolvedAnthropicCompat {
|
||||
const baseUrl = resolvedBaseUrl ?? model.baseUrl;
|
||||
const isZai = model.provider === "zai" || isZaiAnthropicUrl(baseUrl);
|
||||
return {
|
||||
disableStrictTools: false,
|
||||
disableAdaptiveThinking: false,
|
||||
supportsEagerToolInputStreaming: true,
|
||||
supportsLongCacheRetention: true,
|
||||
// First-party Claude API only. Bedrock/Vertex/Foundry and other
|
||||
// Anthropic-compatible gateways reject mid-conversation system roles, so
|
||||
// detection requires the canonical api.anthropic.com host plus a
|
||||
// supported model id.
|
||||
supportsMidConversationSystem:
|
||||
isOfficialAnthropicApiUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id),
|
||||
supportsForcedToolChoice: !isAnthropicFableOrMythosModel(model.id),
|
||||
// Z.AI workaround (issue #814): its proxy deserializes tool_result blocks
|
||||
// into a class that reads `.id`.
|
||||
requiresToolResultId: isZai,
|
||||
// Official Anthropic enforces signature-based thinking-chain integrity, so
|
||||
// unsigned thinking blocks must stay text there. Anthropic-compatible
|
||||
// reasoning endpoints commonly emit unsigned thinking blocks while still
|
||||
// expecting them back as `type: "thinking"` on continuation; demoting them
|
||||
// loses the reasoning chain and can destabilize the next tool-call
|
||||
// arguments (#2005). Known non-signing hosts (Z.AI, DeepSeek) are also
|
||||
// preserved for compatibility.
|
||||
replayUnsignedThinking:
|
||||
isZai ||
|
||||
model.provider === "deepseek" ||
|
||||
isDeepseekHostUrl(baseUrl) ||
|
||||
(model.reasoning && !isOfficialAnthropicApiUrl(baseUrl)),
|
||||
};
|
||||
}
|
||||
|
||||
/** Layer explicit `model.compat` overrides onto the detected anthropic defaults. */
|
||||
export function resolveAnthropicCompat(
|
||||
model: Model<"anthropic-messages">,
|
||||
resolvedBaseUrl?: string,
|
||||
): ResolvedAnthropicCompat {
|
||||
const detected = detectAnthropicCompat(model, resolvedBaseUrl);
|
||||
const compat = model.compat;
|
||||
if (!compat) return detected;
|
||||
return {
|
||||
disableStrictTools: compat.disableStrictTools ?? detected.disableStrictTools,
|
||||
disableAdaptiveThinking: compat.disableAdaptiveThinking ?? detected.disableAdaptiveThinking,
|
||||
supportsEagerToolInputStreaming:
|
||||
compat.supportsEagerToolInputStreaming ?? detected.supportsEagerToolInputStreaming,
|
||||
supportsLongCacheRetention: compat.supportsLongCacheRetention ?? detected.supportsLongCacheRetention,
|
||||
supportsMidConversationSystem: compat.supportsMidConversationSystem ?? detected.supportsMidConversationSystem,
|
||||
supportsForcedToolChoice: compat.supportsForcedToolChoice ?? detected.supportsForcedToolChoice,
|
||||
requiresToolResultId: compat.requiresToolResultId ?? detected.requiresToolResultId,
|
||||
replayUnsignedThinking: compat.replayUnsignedThinking ?? detected.replayUnsignedThinking,
|
||||
};
|
||||
}
|
||||
@@ -1,3 +1,13 @@
|
||||
import { hostMatchesUrl, modelMatchesHost } from "../hosts";
|
||||
import {
|
||||
isAnthropicNamespacedModelId,
|
||||
isClaudeModelId,
|
||||
isDeepseekModelIdOrName,
|
||||
isKimiK26ModelId,
|
||||
isKimiModelId,
|
||||
isMimoModelIdOrName,
|
||||
isQwenModelId,
|
||||
} from "../identity/family";
|
||||
import type { Model, OpenAICompat } from "../types";
|
||||
|
||||
type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
|
||||
@@ -10,6 +20,8 @@ export type ResolvedOpenAICompat = Required<
|
||||
| "vercelGatewayRouting"
|
||||
| "extraBody"
|
||||
| "toolStrictMode"
|
||||
| "streamIdleTimeoutMs"
|
||||
| "supportsLongPromptCacheRetention"
|
||||
| "cacheControlFormat"
|
||||
| "thinkingKeep"
|
||||
>
|
||||
@@ -19,9 +31,16 @@ export type ResolvedOpenAICompat = Required<
|
||||
extraBody?: OpenAICompat["extraBody"];
|
||||
cacheControlFormat?: OpenAICompat["cacheControlFormat"];
|
||||
thinkingKeep?: OpenAICompat["thinkingKeep"];
|
||||
streamIdleTimeoutMs?: number;
|
||||
toolStrictMode: ResolvedToolStrictMode;
|
||||
};
|
||||
|
||||
/** GLM coding-plan SKUs idle for minutes mid-reasoning; see `streamIdleTimeoutMs`. */
|
||||
const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i;
|
||||
const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000;
|
||||
/** Direct DeepSeek reasoning models stall between thinking and answer phases. */
|
||||
const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000;
|
||||
|
||||
function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
|
||||
if (
|
||||
provider === "openai" ||
|
||||
@@ -33,17 +52,13 @@ function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const normalizedBaseUrl = baseUrl.toLowerCase();
|
||||
return (
|
||||
normalizedBaseUrl.includes("api.openai.com") ||
|
||||
normalizedBaseUrl.includes(".openai.azure.com") ||
|
||||
normalizedBaseUrl.includes("models.inference.ai.azure.com") ||
|
||||
normalizedBaseUrl.includes("api.cerebras.ai") ||
|
||||
normalizedBaseUrl.includes("api.together.xyz") ||
|
||||
normalizedBaseUrl.includes("openrouter.ai") ||
|
||||
normalizedBaseUrl.includes("api.deepseek.com") ||
|
||||
normalizedBaseUrl.includes("deepseek.com")
|
||||
hostMatchesUrl(baseUrl, "openai") ||
|
||||
hostMatchesUrl(baseUrl, "azureOpenAI") ||
|
||||
hostMatchesUrl(baseUrl, "cerebras") ||
|
||||
hostMatchesUrl(baseUrl, "together") ||
|
||||
hostMatchesUrl(baseUrl, "openrouter") ||
|
||||
hostMatchesUrl(baseUrl, "deepseekFamily")
|
||||
);
|
||||
}
|
||||
|
||||
@@ -87,23 +102,19 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
const provider = model.provider;
|
||||
// Use resolvedBaseUrl if provided (e.g., after GitHub Copilot proxy-ep resolution)
|
||||
const baseUrl = resolvedBaseUrl ?? model.baseUrl;
|
||||
const hostModel = { provider, baseUrl };
|
||||
|
||||
const isCerebras = provider === "cerebras" || baseUrl.includes("cerebras.ai");
|
||||
const isZai = provider === "zai" || baseUrl.includes("api.z.ai");
|
||||
const isZhipu = provider === "zhipu-coding-plan" || baseUrl.includes("open.bigmodel.cn");
|
||||
const isKilo = provider === "kilo" || baseUrl.includes("api.kilo.ai");
|
||||
const isKimiModel = model.id.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(model.id);
|
||||
const isMoonshotNativeHost =
|
||||
provider === "moonshot" || provider === "kimi-code" || /api\.moonshot\.ai|api\.kimi\.com/i.test(baseUrl);
|
||||
const isMoonshotKimi = isKimiModel && isMoonshotNativeHost;
|
||||
const usesMoonshotKimiPreservedThinking = isMoonshotKimi && /(^|\/)kimi-k2\.6(?:[-:]|$)/i.test(model.id);
|
||||
const isCerebras = modelMatchesHost(hostModel, "cerebras");
|
||||
const isZai = modelMatchesHost(hostModel, "zai");
|
||||
const isZhipu = modelMatchesHost(hostModel, "zhipu");
|
||||
const isKilo = modelMatchesHost(hostModel, "kilo");
|
||||
const isKimiModel = isKimiModelId(model.id);
|
||||
const isMoonshotKimi = isKimiModel && modelMatchesHost(hostModel, "moonshotNative");
|
||||
const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(model.id);
|
||||
const isAnthropicModel =
|
||||
provider === "anthropic" ||
|
||||
baseUrl.includes("api.anthropic.com") ||
|
||||
/(^|\/)claude[-.]/i.test(model.id) ||
|
||||
/(^|\/)anthropic\//i.test(model.id);
|
||||
const isAlibaba = provider === "alibaba-coding-plan" || baseUrl.includes("dashscope");
|
||||
const isQwen = model.id.toLowerCase().includes("qwen");
|
||||
modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(model.id) || isAnthropicNamespacedModelId(model.id);
|
||||
const isAlibaba = modelMatchesHost(hostModel, "alibabaDashscope");
|
||||
const isQwen = isQwenModelId(model.id);
|
||||
// DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in
|
||||
// thinking mode unless prior assistant tool-call turns include `reasoning_content`. The
|
||||
// upstream model is reachable through many OpenAI-compat hosts (api.deepseek.com, Deepinfra,
|
||||
@@ -112,66 +123,52 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
// applies when thinking mode is actually engaged.
|
||||
const lowerId = model.id.toLowerCase();
|
||||
const lowerName = (model.name ?? "").toLowerCase();
|
||||
const isXiaomiHost =
|
||||
provider === "xiaomi" || provider.startsWith("xiaomi-token-plan-") || baseUrl.includes("xiaomimimo.com");
|
||||
const isMimoModel = lowerId.includes("mimo") || lowerName.includes("mimo");
|
||||
const isXiaomiMimo = isXiaomiHost && isMimoModel;
|
||||
const isXiaomiHost = modelMatchesHost(hostModel, "xiaomi");
|
||||
const isXiaomiMimo = isXiaomiHost && (isMimoModelIdOrName(model.id) || isMimoModelIdOrName(model.name ?? ""));
|
||||
// OpenCode Zen's `big-pickle` is a DeepSeek reasoning alias; the upstream
|
||||
// 400s come from DeepSeek and require exact reasoning_content replay.
|
||||
const isOpenCodeDeepseekAlias =
|
||||
provider === "opencode-zen" && (lowerId === "big-pickle" || lowerName === "big pickle");
|
||||
const isDeepseekFamily =
|
||||
provider === "deepseek" ||
|
||||
baseUrl.includes("deepseek.com") ||
|
||||
lowerId.includes("deepseek") ||
|
||||
lowerName.includes("deepseek") ||
|
||||
modelMatchesHost(hostModel, "deepseekFamily") ||
|
||||
isDeepseekModelIdOrName(model.id) ||
|
||||
isDeepseekModelIdOrName(model.name ?? "") ||
|
||||
isOpenCodeDeepseekAlias;
|
||||
const isDirectDeepseekApi = provider === "deepseek" || baseUrl.includes("api.deepseek.com");
|
||||
const isDirectDeepseekApi = modelMatchesHost(hostModel, "deepseekDirect");
|
||||
const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(model.reasoning);
|
||||
const isGrok = modelMatchesHost(hostModel, "xai");
|
||||
const isMistral = modelMatchesHost(hostModel, "mistral");
|
||||
const isOpenCodeHost = modelMatchesHost(hostModel, "opencode");
|
||||
const isNonStandard =
|
||||
isCerebras ||
|
||||
provider === "xai" ||
|
||||
baseUrl.includes("api.x.ai") ||
|
||||
provider === "mistral" ||
|
||||
baseUrl.includes("mistral.ai") ||
|
||||
baseUrl.includes("chutes.ai") ||
|
||||
baseUrl.includes("deepseek.com") ||
|
||||
baseUrl.includes("fireworks.ai") ||
|
||||
isGrok ||
|
||||
isMistral ||
|
||||
hostMatchesUrl(baseUrl, "chutes") ||
|
||||
hostMatchesUrl(baseUrl, "deepseekFamily") ||
|
||||
hostMatchesUrl(baseUrl, "fireworks") ||
|
||||
isAlibaba ||
|
||||
isZai ||
|
||||
isZhipu ||
|
||||
isKilo ||
|
||||
isQwen ||
|
||||
isXiaomiHost ||
|
||||
provider === "opencode-zen" ||
|
||||
provider === "opencode-go" ||
|
||||
baseUrl.includes("opencode.ai");
|
||||
isOpenCodeHost;
|
||||
const isOpenCodeProvider = provider === "opencode-go" || provider === "opencode-zen";
|
||||
|
||||
const useMaxTokens =
|
||||
provider === "mistral" ||
|
||||
baseUrl.includes("mistral.ai") ||
|
||||
baseUrl.includes("chutes.ai") ||
|
||||
baseUrl.includes("fireworks.ai") ||
|
||||
isDirectDeepseekApi;
|
||||
const isGrok = provider === "xai" || baseUrl.includes("api.x.ai");
|
||||
const isMistral = provider === "mistral" || baseUrl.includes("mistral.ai");
|
||||
isMistral || hostMatchesUrl(baseUrl, "chutes") || hostMatchesUrl(baseUrl, "fireworks") || isDirectDeepseekApi;
|
||||
|
||||
// Hosts whose chat-completions endpoints are known to accept multiple
|
||||
// leading `system`/`developer` messages (preferred for KV-cache reuse).
|
||||
// Anything outside this allowlist defaults to coalescing because
|
||||
// strict chat templates (Qwen 3.5+ via vLLM, MiniMax, etc.) reject
|
||||
// follow-up system messages with a 400.
|
||||
const isOpenAIHost = provider === "openai" || baseUrl.includes("api.openai.com");
|
||||
const isAzureHost =
|
||||
provider === "azure" ||
|
||||
baseUrl.includes(".openai.azure.com") ||
|
||||
baseUrl.includes("models.inference.ai.azure.com") ||
|
||||
baseUrl.includes("azure.com/openai");
|
||||
const isOpenRouter = provider === "openrouter" || baseUrl.includes("openrouter.ai");
|
||||
const isTogether = provider === "together" || baseUrl.includes("api.together.xyz");
|
||||
const isFireworks = baseUrl.includes("fireworks.ai");
|
||||
const isGroqHost = provider === "groq" || baseUrl.includes("api.groq.com");
|
||||
const isOpenAIHost = modelMatchesHost(hostModel, "openai");
|
||||
const isAzureHost = modelMatchesHost(hostModel, "azureOpenAI");
|
||||
const isOpenRouter = modelMatchesHost(hostModel, "openrouter");
|
||||
const isTogether = modelMatchesHost(hostModel, "together");
|
||||
const isFireworks = hostMatchesUrl(baseUrl, "fireworks");
|
||||
const isGroqHost = modelMatchesHost(hostModel, "groq");
|
||||
const isCopilotHost = provider === "github-copilot";
|
||||
const isZenmuxHost = provider === "zenmux";
|
||||
// Endpoints that MUST receive a single system block. MiniMax's OpenAI
|
||||
@@ -179,12 +176,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
// Dashscope and Qwen Portal serve Qwen models whose chat template
|
||||
// raises "System message must be at the beginning" if any system
|
||||
// message appears past index 0.
|
||||
const isMiniMaxHost =
|
||||
provider === "minimax-code" ||
|
||||
provider === "minimax-code-cn" ||
|
||||
baseUrl.includes("api.minimax.io") ||
|
||||
baseUrl.includes("api.minimaxi.com");
|
||||
const isQwenPortal = provider === "qwen-portal" || baseUrl.includes("portal.qwen.ai");
|
||||
const isMiniMaxHost = modelMatchesHost(hostModel, "minimax");
|
||||
const isQwenPortal = modelMatchesHost(hostModel, "qwenPortal");
|
||||
const supportsMultipleSystemMessagesDefault =
|
||||
!isMiniMaxHost &&
|
||||
!isAlibaba &&
|
||||
@@ -234,6 +227,16 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
||||
: {};
|
||||
|
||||
// Stream-watchdog floor: GLM coding-plan SKUs and direct DeepSeek reasoning
|
||||
// models idle for minutes mid-reasoning; widen the idle timeout so warm-ups
|
||||
// stop aborting and retrying.
|
||||
const streamIdleTimeoutMs =
|
||||
GLM_CODING_PLAN_MODEL_PATTERN.test(model.id) && (isZai || isZhipu)
|
||||
? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS
|
||||
: model.reasoning && isDirectDeepseekApi
|
||||
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
|
||||
: undefined;
|
||||
|
||||
return {
|
||||
supportsStore: !isNonStandard,
|
||||
// `developer` is an OpenAI-Responses-era extension to the chat-completions schema. Almost
|
||||
@@ -263,7 +266,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
thinkingFormat:
|
||||
isZai || isZhipu || isMoonshotKimi || isXiaomiMimo
|
||||
? "zai"
|
||||
: provider === "openrouter" || baseUrl.includes("openrouter.ai")
|
||||
: isOpenRouter
|
||||
? "openrouter"
|
||||
: isAlibaba || isQwen
|
||||
? "qwen"
|
||||
@@ -286,7 +289,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
(isKimiModel && !isOpenCodeProvider) ||
|
||||
(isDeepseekFamily && Boolean(model.reasoning)) ||
|
||||
isXiaomiMimo ||
|
||||
((provider === "openrouter" || baseUrl.includes("openrouter.ai")) && Boolean(model.reasoning)),
|
||||
(isOpenRouter && Boolean(model.reasoning)),
|
||||
// DeepSeek V4 and Xiaomi MiMo reject synthetic reasoning_content placeholders (".") on tool-call turns.
|
||||
// Kimi and OpenRouter accept them when actual reasoning is unavailable.
|
||||
allowsSyntheticReasoningContentForToolCalls: (!isDeepseekFamily || !model.reasoning) && !isXiaomiMimo,
|
||||
@@ -297,6 +300,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
supportsStrictMode: detectStrictModeSupport(provider, baseUrl),
|
||||
extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined,
|
||||
toolStrictMode: isCerebras ? "all_strict" : "mixed",
|
||||
streamIdleTimeoutMs,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -350,5 +354,62 @@ export function resolveOpenAICompat(
|
||||
supportsStrictMode: model.compat.supportsStrictMode ?? detected.supportsStrictMode,
|
||||
extraBody: model.compat.extraBody ?? detected.extraBody,
|
||||
toolStrictMode: model.compat.toolStrictMode ?? detected.toolStrictMode,
|
||||
streamIdleTimeoutMs: model.compat.streamIdleTimeoutMs ?? detected.streamIdleTimeoutMs,
|
||||
};
|
||||
}
|
||||
|
||||
/** Resolved Responses-API compatibility view (see `detectOpenAIResponsesCompat`). */
|
||||
export interface ResolvedOpenAIResponsesCompat {
|
||||
supportsDeveloperRole: boolean;
|
||||
supportsStrictMode: boolean;
|
||||
supportsLongPromptCacheRetention: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect Responses-API compatibility from provider/baseUrl. The Responses
|
||||
* flavor deliberately differs from chat-completions: GitHub Copilot's
|
||||
* responses endpoint accepts the `developer` role, while strict tool mode is
|
||||
* scoped to first-party OpenAI/Azure/Copilot providers. Developer-role and
|
||||
* prompt-cache detection are URL-only on purpose — the historical call sites
|
||||
* never consulted the provider id for them.
|
||||
*/
|
||||
export function detectOpenAIResponsesCompat(
|
||||
model: { provider: string; baseUrl: string },
|
||||
resolvedBaseUrl?: string,
|
||||
): ResolvedOpenAIResponsesCompat {
|
||||
const baseUrl = resolvedBaseUrl ?? model.baseUrl ?? "";
|
||||
return {
|
||||
supportsDeveloperRole:
|
||||
hostMatchesUrl(baseUrl, "openai") ||
|
||||
hostMatchesUrl(baseUrl, "azureOpenAI") ||
|
||||
hostMatchesUrl(baseUrl, "githubCopilot"),
|
||||
supportsStrictMode:
|
||||
model.provider === "openai" ||
|
||||
model.provider === "azure" ||
|
||||
model.provider === "github-copilot" ||
|
||||
hostMatchesUrl(baseUrl, "openai") ||
|
||||
hostMatchesUrl(baseUrl, "azureOpenAI"),
|
||||
supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve Responses-API compatibility by layering explicit `model.compat`
|
||||
* overrides onto the detected defaults — the Responses-side analogue of
|
||||
* `resolveOpenAICompat`. Models bundled with `supportsDeveloperRole: false`
|
||||
* (codex-mini-style SKUs) take effect here.
|
||||
*/
|
||||
export function resolveOpenAIResponsesCompat(
|
||||
model: { provider: string; baseUrl: string; compat?: OpenAICompat },
|
||||
resolvedBaseUrl?: string,
|
||||
): ResolvedOpenAIResponsesCompat {
|
||||
const detected = detectOpenAIResponsesCompat(model, resolvedBaseUrl);
|
||||
const compat = model.compat;
|
||||
if (!compat) return detected;
|
||||
return {
|
||||
supportsDeveloperRole: compat.supportsDeveloperRole ?? detected.supportsDeveloperRole,
|
||||
supportsStrictMode: compat.supportsStrictMode ?? detected.supportsStrictMode,
|
||||
supportsLongPromptCacheRetention:
|
||||
compat.supportsLongPromptCacheRetention ?? detected.supportsLongPromptCacheRetention,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
/**
|
||||
* Known model-endpoint host classification — the single vocabulary for the
|
||||
* `provider === id || baseUrl.includes(marker)` idiom that gates wire-level
|
||||
* behavior (compat detection, routing, header shaping, watchdog floors).
|
||||
*
|
||||
* Markers are case-insensitive substrings matched against the base URL, NOT
|
||||
* parsed hostnames: proxies regularly embed the upstream host in a path
|
||||
* segment, and the historical call sites all used substring semantics.
|
||||
* Callers needing strict hostname matching (e.g. guards before request-body
|
||||
* mutation) should keep their own `new URL().hostname` checks.
|
||||
*/
|
||||
|
||||
interface HostClassSpec {
|
||||
/** Provider ids that imply this host class regardless of baseUrl. */
|
||||
readonly providers?: readonly string[];
|
||||
/** Provider-id prefixes that imply this host class (e.g. `xiaomi-token-plan-`). */
|
||||
readonly providerPrefixes?: readonly string[];
|
||||
/** Case-insensitive substrings matched against the base URL. */
|
||||
readonly urlMarkers: readonly string[];
|
||||
}
|
||||
|
||||
export const KNOWN_HOSTS = {
|
||||
openai: { providers: ["openai"], urlMarkers: ["api.openai.com"] },
|
||||
azureOpenAI: {
|
||||
providers: ["azure"],
|
||||
urlMarkers: [".openai.azure.com", "azure.com/openai", "models.inference.ai.azure.com"],
|
||||
},
|
||||
openrouter: { providers: ["openrouter"], urlMarkers: ["openrouter.ai"] },
|
||||
vercelAIGateway: { providers: ["vercel-ai-gateway"], urlMarkers: ["ai-gateway.vercel.sh"] },
|
||||
githubCopilot: { providers: ["github-copilot"], urlMarkers: ["githubcopilot.com", "copilot-api."] },
|
||||
anthropic: { providers: ["anthropic"], urlMarkers: ["api.anthropic.com"] },
|
||||
/** DeepSeek's first-party API only — gates direct-API quirks (max_tokens field, thinking extraBody). */
|
||||
deepseekDirect: { providers: ["deepseek"], urlMarkers: ["api.deepseek.com"] },
|
||||
/** Any DeepSeek-operated host (first-party API, web-chat fronts). Wider than `deepseekDirect` on purpose. */
|
||||
deepseekFamily: { providers: ["deepseek"], urlMarkers: ["deepseek.com"] },
|
||||
cerebras: { providers: ["cerebras"], urlMarkers: ["cerebras.ai"] },
|
||||
zai: { providers: ["zai"], urlMarkers: ["api.z.ai"] },
|
||||
zhipu: { providers: ["zhipu-coding-plan"], urlMarkers: ["open.bigmodel.cn"] },
|
||||
kilo: { providers: ["kilo"], urlMarkers: ["api.kilo.ai"] },
|
||||
alibabaDashscope: { providers: ["alibaba-coding-plan"], urlMarkers: ["dashscope"] },
|
||||
xiaomi: { providers: ["xiaomi"], providerPrefixes: ["xiaomi-token-plan-"], urlMarkers: ["xiaomimimo.com"] },
|
||||
xai: { providers: ["xai"], urlMarkers: ["api.x.ai"] },
|
||||
mistral: { providers: ["mistral"], urlMarkers: ["mistral.ai"] },
|
||||
together: { providers: ["together"], urlMarkers: ["api.together.xyz"] },
|
||||
/** URL-only on purpose: the `fireworks`/`firepass` providers route per-model and not every model is Fireworks-shaped. */
|
||||
fireworks: { urlMarkers: ["fireworks.ai"] },
|
||||
groq: { providers: ["groq"], urlMarkers: ["api.groq.com"] },
|
||||
minimax: {
|
||||
providers: ["minimax", "minimax-code", "minimax-code-cn"],
|
||||
urlMarkers: ["api.minimax.io", "api.minimaxi.com"],
|
||||
},
|
||||
qwenPortal: { providers: ["qwen-portal"], urlMarkers: ["portal.qwen.ai"] },
|
||||
moonshotNative: { providers: ["moonshot", "kimi-code"], urlMarkers: ["api.moonshot.ai", "api.kimi.com"] },
|
||||
opencode: { providers: ["opencode-go", "opencode-zen"], urlMarkers: ["opencode.ai"] },
|
||||
chutes: { urlMarkers: ["chutes.ai"] },
|
||||
} as const satisfies Record<string, HostClassSpec>;
|
||||
|
||||
export type KnownHost = keyof typeof KNOWN_HOSTS;
|
||||
|
||||
/** URL-only host check (for call sites that have no provider id, e.g. raw env config). */
|
||||
export function hostMatchesUrl(baseUrl: string | undefined, host: KnownHost): boolean {
|
||||
if (!baseUrl) return false;
|
||||
const spec: HostClassSpec = KNOWN_HOSTS[host];
|
||||
const normalized = baseUrl.toLowerCase();
|
||||
for (const marker of spec.urlMarkers) {
|
||||
if (normalized.includes(marker)) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Provider-or-URL host check — the canonical `provider === id || baseUrl.includes(marker)` idiom. */
|
||||
export function modelMatchesHost(model: { provider: string; baseUrl: string }, host: KnownHost): boolean {
|
||||
const spec: HostClassSpec = KNOWN_HOSTS[host];
|
||||
if (spec.providers) {
|
||||
for (const provider of spec.providers) {
|
||||
if (model.provider === provider) return true;
|
||||
}
|
||||
}
|
||||
if (spec.providerPrefixes) {
|
||||
for (const prefix of spec.providerPrefixes) {
|
||||
if (model.provider.startsWith(prefix)) return true;
|
||||
}
|
||||
}
|
||||
return hostMatchesUrl(model.baseUrl, host);
|
||||
}
|
||||
|
||||
// --- Endpoint-shape predicates (URL path/verb shapes, not vendor hosts) ---
|
||||
|
||||
/** Vertex AI express-mode OpenAI-compatible endpoint (`…/endpoints/openapi`). */
|
||||
export function isVertexExpressOpenAIUrl(baseUrl: string): boolean {
|
||||
return baseUrl.includes("/endpoints/openapi");
|
||||
}
|
||||
|
||||
/** Vertex AI Anthropic raw-predict endpoints (`:streamRawPredict` / `:rawPredict`). */
|
||||
export function isVertexRawPredictUrl(baseUrl: string): boolean {
|
||||
return baseUrl.includes(":streamRawPredict") || baseUrl.includes(":rawPredict");
|
||||
}
|
||||
|
||||
/** Azure OpenAI deployment-scoped path (`…/deployments/<name>/…`). */
|
||||
export function isAzureDeploymentsUrl(baseUrl: string): boolean {
|
||||
return baseUrl.includes("/deployments/");
|
||||
}
|
||||
|
||||
/** Alibaba DashScope consumer `compatible-mode` endpoint (rejects multimodal arrays for some text-only SKUs). */
|
||||
export function isDashscopeCompatibleModeUrl(baseUrl: string): boolean {
|
||||
const normalized = baseUrl.toLowerCase();
|
||||
return (
|
||||
normalized.includes("dashscope") && normalized.includes("aliyuncs.com") && normalized.includes("/compatible-mode")
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
/**
|
||||
* Model-family id predicates: the shared vocabulary for "is this id a member
|
||||
* of family X" checks that gate wire-level behavior across hosts (a Kimi or
|
||||
* DeepSeek model keeps its quirks no matter which OpenAI-compatible proxy
|
||||
* serves it). Looser per-feature heuristics (e.g. stream-markup healing)
|
||||
* deliberately keep their own patterns — only provably-shared matchers live
|
||||
* here.
|
||||
*/
|
||||
|
||||
/** Kimi family ids in any namespace form (`moonshotai/kimi-*`, `kimi-k2.6`, `vendor/kimi.x`). */
|
||||
export function isKimiModelId(modelId: string): boolean {
|
||||
return modelId.includes("moonshotai/kimi") || /(^|\/)kimi[-.]/i.test(modelId);
|
||||
}
|
||||
|
||||
/** Kimi K2.6 specifically (preserved-thinking transport on Moonshot-native hosts). */
|
||||
export function isKimiK26ModelId(modelId: string): boolean {
|
||||
return /(^|\/)kimi-k2\.6(?:[-:]|$)/i.test(modelId);
|
||||
}
|
||||
|
||||
/** Claude ids in any namespace form (`claude-*`, `vendor/claude.x`). */
|
||||
export function isClaudeModelId(modelId: string): boolean {
|
||||
return /(^|\/)claude[-.]/i.test(modelId);
|
||||
}
|
||||
|
||||
/** `anthropic/`-namespaced ids (aggregator catalogs like OpenRouter). */
|
||||
export function isAnthropicNamespacedModelId(modelId: string): boolean {
|
||||
return /(^|\/)anthropic\//i.test(modelId);
|
||||
}
|
||||
|
||||
/** Qwen family ids (substring match — Qwen SKUs have no stable prefix shape). */
|
||||
export function isQwenModelId(modelId: string): boolean {
|
||||
return modelId.toLowerCase().includes("qwen");
|
||||
}
|
||||
|
||||
/** DeepSeek family by id or display name (proxies often rename the id but keep the name). */
|
||||
export function isDeepseekModelIdOrName(value: string): boolean {
|
||||
return value.toLowerCase().includes("deepseek");
|
||||
}
|
||||
|
||||
/** Xiaomi MiMo family by id or display name. */
|
||||
export function isMimoModelIdOrName(value: string): boolean {
|
||||
return value.toLowerCase().includes("mimo");
|
||||
}
|
||||
|
||||
/**
|
||||
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
|
||||
* Claude Fable/Mythos 5. Older adaptive-thinking models (Opus 4.6, Sonnet
|
||||
* 4.6+) reject the field.
|
||||
*/
|
||||
export function supportsAdaptiveThinkingDisplay(modelId: string): boolean {
|
||||
if (/claude-(?:fable|mythos)-5\b/.test(modelId)) return true;
|
||||
// Bound the minor to non-date digits: bare dated ids like
|
||||
// `claude-opus-4-20250514` (Opus 4.0) must not parse as minor=20250514.
|
||||
const match = /claude-opus-(\d+)-(\d{1,2})(?!\d)/.exec(modelId);
|
||||
if (!match) return false;
|
||||
const major = Number(match[1]);
|
||||
const minor = Number(match[2]);
|
||||
return major > 4 || (major === 4 && minor >= 7);
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
export * from "./bundled";
|
||||
export * from "./classify";
|
||||
export * from "./equivalence";
|
||||
export * from "./family";
|
||||
export * from "./id";
|
||||
export * from "./markers";
|
||||
export * from "./priority";
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { resolveOpenAICompat } from "./compat/openai";
|
||||
import { Effort, THINKING_EFFORTS } from "./effort";
|
||||
import { modelMatchesHost } from "./hosts";
|
||||
import {
|
||||
type AnthropicModel,
|
||||
bareModelId,
|
||||
@@ -353,7 +354,7 @@ function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
|
||||
model: ApiModel<TApi>,
|
||||
): boolean {
|
||||
if (model.api !== "openai-completions") return false;
|
||||
if (model.provider !== "openrouter" && !model.baseUrl.includes("openrouter.ai")) return false;
|
||||
if (!modelMatchesHost(model, "openrouter")) return false;
|
||||
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
|
||||
}
|
||||
|
||||
|
||||
@@ -2349,11 +2349,15 @@ export interface GithubCopilotModelManagerConfig {
|
||||
fetch?: FetchImpl;
|
||||
}
|
||||
|
||||
const COPILOT_ANTHROPIC_MODEL_PATTERN = /^claude-(haiku|sonnet|opus)-4([.-]|$)/;
|
||||
const isCopilotResponsesModelId = (modelId: string): boolean =>
|
||||
modelId.startsWith("gpt-5") || modelId.startsWith("oswe");
|
||||
|
||||
function inferCopilotApi(modelId: string): Api {
|
||||
if (/^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId)) {
|
||||
if (COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId)) {
|
||||
return "anthropic-messages";
|
||||
}
|
||||
if (modelId.startsWith("gpt-5") || modelId.startsWith("oswe")) {
|
||||
if (isCopilotResponsesModelId(modelId)) {
|
||||
return "openai-responses";
|
||||
}
|
||||
return "openai-completions";
|
||||
@@ -2776,11 +2780,11 @@ const COPILOT_DEFAULT_RESOLUTION = {
|
||||
|
||||
const COPILOT_API_RESOLUTION_RULES: readonly ApiResolutionRule[] = [
|
||||
{
|
||||
matches: modelId => /^claude-(haiku|sonnet|opus)-4([.-]|$)/.test(modelId),
|
||||
matches: modelId => COPILOT_ANTHROPIC_MODEL_PATTERN.test(modelId),
|
||||
resolved: { api: "anthropic-messages", baseUrl: COPILOT_BASE_URL },
|
||||
},
|
||||
{
|
||||
matches: modelId => modelId.startsWith("gpt-5") || modelId.startsWith("oswe"),
|
||||
matches: isCopilotResponsesModelId,
|
||||
resolved: { api: "openai-responses", baseUrl: COPILOT_BASE_URL },
|
||||
},
|
||||
];
|
||||
|
||||
@@ -182,6 +182,13 @@ export interface OpenAICompat {
|
||||
cacheControlFormat?: "anthropic" | undefined;
|
||||
/** Whether the provider supports the `strict` field in tool definitions. Default: auto-detected per provider/baseUrl (conservative for unknown providers). */
|
||||
supportsStrictMode?: boolean;
|
||||
/**
|
||||
* Stream-watchdog idle-timeout floor in ms for slow reasoning hosts.
|
||||
* Default: auto-detected (GLM coding-plan hosts, direct DeepSeek reasoning).
|
||||
*/
|
||||
streamIdleTimeoutMs?: number;
|
||||
/** Whether the host honors `prompt_cache_retention: "24h"` on the Responses API. Default: auto-detected (api.openai.com). */
|
||||
supportsLongPromptCacheRetention?: boolean;
|
||||
/** Whether tool schemas must be sent either all strict or all non-strict. Undefined keeps the existing per-tool mixed behavior. */
|
||||
toolStrictMode?: "all_strict" | "none";
|
||||
}
|
||||
@@ -225,6 +232,22 @@ export interface AnthropicCompat {
|
||||
* When unset, auto-detected from the model id. Default: true.
|
||||
*/
|
||||
supportsForcedToolChoice?: boolean;
|
||||
/**
|
||||
* Include a non-standard `id` field (aliasing `tool_use_id`) on
|
||||
* `tool_result` blocks. Z.AI's Anthropic-compatible proxy deserializes
|
||||
* tool results into a class that reads `.id` (issue #814). Default:
|
||||
* auto-detected (Z.AI hosts).
|
||||
*/
|
||||
requiresToolResultId?: boolean;
|
||||
/**
|
||||
* Replay unsigned `thinking` blocks from prior assistant turns as native
|
||||
* thinking instead of demoting them to text. Official Anthropic enforces
|
||||
* signature-based thinking-chain integrity, so unsigned blocks must stay
|
||||
* text there; compatible reasoning endpoints (Z.AI, DeepSeek, …) emit
|
||||
* unsigned blocks and expect them back as `type: "thinking"` (#2005).
|
||||
* Default: auto-detected from provider/baseUrl and `model.reasoning`.
|
||||
*/
|
||||
replayUnsignedThinking?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import {
|
||||
hostMatchesUrl,
|
||||
isDashscopeCompatibleModeUrl,
|
||||
isVertexExpressOpenAIUrl,
|
||||
isVertexRawPredictUrl,
|
||||
modelMatchesHost,
|
||||
} from "@oh-my-pi/pi-catalog/hosts";
|
||||
|
||||
describe("hostMatchesUrl", () => {
|
||||
test("matches OpenRouter URLs and rejects other or missing URLs", () => {
|
||||
expect(hostMatchesUrl("https://openrouter.ai/api/v1", "openrouter")).toBe(true);
|
||||
expect(hostMatchesUrl("https://api.openai.com/v1", "openrouter")).toBe(false);
|
||||
expect(hostMatchesUrl(undefined, "openrouter")).toBe(false);
|
||||
});
|
||||
|
||||
test("matches Z.AI URLs case-insensitively", () => {
|
||||
expect(hostMatchesUrl("https://API.Z.AI/api/paas/v4", "zai")).toBe(true);
|
||||
});
|
||||
|
||||
test("keeps DeepSeek direct host narrower than DeepSeek family", () => {
|
||||
expect(hostMatchesUrl("https://api.deepseek.com/v1", "deepseekDirect")).toBe(true);
|
||||
expect(hostMatchesUrl("https://api.deepseek.com/v1", "deepseekFamily")).toBe(true);
|
||||
expect(hostMatchesUrl("https://chat.deepseek.com/api", "deepseekFamily")).toBe(true);
|
||||
expect(hostMatchesUrl("https://chat.deepseek.com/api", "deepseekDirect")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("modelMatchesHost", () => {
|
||||
test("matches by provider id, provider prefix, and URL-only Fireworks markers", () => {
|
||||
expect(modelMatchesHost({ provider: "openrouter", baseUrl: "https://example.com/v1" }, "openrouter")).toBe(true);
|
||||
expect(modelMatchesHost({ provider: "xiaomi-token-plan-eu", baseUrl: "https://example.com/v1" }, "xiaomi")).toBe(
|
||||
true,
|
||||
);
|
||||
expect(modelMatchesHost({ provider: "fireworks", baseUrl: "https://example.com/v1" }, "fireworks")).toBe(false);
|
||||
expect(
|
||||
modelMatchesHost({ provider: "custom", baseUrl: "https://api.fireworks.ai/inference/v1" }, "fireworks"),
|
||||
).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("endpoint shape predicates", () => {
|
||||
test("recognizes Vertex express OpenAI-compatible URLs", () => {
|
||||
expect(
|
||||
isVertexExpressOpenAIUrl(
|
||||
"https://us-central1-aiplatform.googleapis.com/v1/projects/p/locations/us/endpoints/openapi",
|
||||
),
|
||||
).toBe(true);
|
||||
expect(
|
||||
isVertexExpressOpenAIUrl(
|
||||
"https://us-central1-aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/google/models/gemini",
|
||||
),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
test("recognizes Vertex rawPredict and streamRawPredict URLs", () => {
|
||||
expect(
|
||||
isVertexRawPredictUrl(
|
||||
"https://aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/anthropic/models/claude:rawPredict",
|
||||
),
|
||||
).toBe(true);
|
||||
expect(
|
||||
isVertexRawPredictUrl(
|
||||
"https://aiplatform.googleapis.com/v1/projects/p/locations/us/publishers/anthropic/models/claude:streamRawPredict",
|
||||
),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
test("requires all DashScope compatible-mode URL markers", () => {
|
||||
expect(isDashscopeCompatibleModeUrl("https://dashscope.aliyuncs.com/compatible-mode/v1")).toBe(true);
|
||||
expect(isDashscopeCompatibleModeUrl("https://example.aliyuncs.com/compatible-mode/v1")).toBe(false);
|
||||
expect(isDashscopeCompatibleModeUrl("https://dashscope.example.com/compatible-mode/v1")).toBe(false);
|
||||
expect(isDashscopeCompatibleModeUrl("https://dashscope.aliyuncs.com/api/v1")).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,44 @@
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import {
|
||||
isClaudeModelId,
|
||||
isKimiK26ModelId,
|
||||
isKimiModelId,
|
||||
supportsAdaptiveThinkingDisplay,
|
||||
} from "@oh-my-pi/pi-catalog/identity";
|
||||
|
||||
describe("isKimiModelId", () => {
|
||||
test("matches Kimi namespace and delimiter forms", () => {
|
||||
expect(isKimiModelId("moonshotai/kimi-k2")).toBe(true);
|
||||
expect(isKimiModelId("kimi-k2.6")).toBe(true);
|
||||
expect(isKimiModelId("vendor/kimi.x")).toBe(true);
|
||||
expect(isKimiModelId("akimbo-model")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isKimiK26ModelId", () => {
|
||||
test("matches Kimi K2.6 without accepting adjacent versions", () => {
|
||||
expect(isKimiK26ModelId("kimi-k2.6")).toBe(true);
|
||||
expect(isKimiK26ModelId("kimi-k2.6-thinking")).toBe(true);
|
||||
expect(isKimiK26ModelId("kimi-k2.61")).toBe(false);
|
||||
expect(isKimiK26ModelId("kimi-k2.5")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isClaudeModelId", () => {
|
||||
test("matches Claude namespace and delimiter forms", () => {
|
||||
expect(isClaudeModelId("claude-sonnet-4-6")).toBe(true);
|
||||
expect(isClaudeModelId("anthropic/claude.3")).toBe(true);
|
||||
expect(isClaudeModelId("my-claudius")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("supportsAdaptiveThinkingDisplay", () => {
|
||||
test("allows Claude Fable 5 and Opus 4.7 or newer only", () => {
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-fable-5")).toBe(true);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-7")).toBe(true);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-5-0")).toBe(true);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-6")).toBe(false);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-opus-4-20250514")).toBe(false);
|
||||
expect(supportsAdaptiveThinkingDisplay("claude-sonnet-4-6")).toBe(false);
|
||||
});
|
||||
});
|
||||
@@ -1,9 +1,9 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `streamIdleTimeoutMs`, `supportsLongPromptCacheRetention`, `requiresToolResultId`, and `replayUnsignedThinking` to the OpenAI `compat` schema so custom model entries can configure those provider-specific capabilities
|
||||
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("need: 5h → 3 of 5 accounts"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
|
||||
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
|
||||
- npm installs now execute a prebundled single-file entry: `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks
|
||||
@@ -43,11 +43,11 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed model-provider detection for append-only mode, authoritative Vertex endpoint checks, and upstream-routing selection by switching from URL substring checks to catalog host-matching helpers
|
||||
- Fixed pasting into the ask tool's "Other (type your own)" text box (and hook input/editor dialogs) on terminals with OSC 5522 enhanced paste (kitty protocol): the enhanced-paste focus routing only targets components exposing a `pasteText` hook, and the dialog wrappers had none, so the payload was stuffed into the main prompt editor hidden behind the dialog. `HookEditorComponent` and `HookInputComponent` now forward `pasteText` to their inner editor/input (pasting also resets the input dialog's timeout countdown like any keystroke).
|
||||
- Fixed auto-retry giving up after one attempt ("Provider requested Xms wait, exceeds retry.maxDelayMs") on a usage-limit 429 when every sibling account was only momentarily blocked: the retry delay now waits for the earliest sibling unblock when that comes sooner than the provider's multi-hour retry-after, so the next attempt picks up the recovered account instead of failing fast.
|
||||
- Fixed Hindsight `per-project-tagged` mental-model seeding so each project gets its own conventions/decisions models and session context only injects active-project or untagged models ([#2218](https://github.com/can1357/oh-my-pi/issues/2218)).
|
||||
- Fixed Windows stdio MCP `.cmd` commands regressing from direct argv launches to a `cmd.exe /c` wrapper in v15.10.10, which made Codegraph MCP exit immediately with `Transport closed` ([#2220](https://github.com/can1357/oh-my-pi/issues/2220)).
|
||||
|
||||
- Fixed Windows stdio MCP `.cmd` commands by wrapping batch shims with `cmd.exe /d /s /c` using the outer command quotes required by `cmd /s`, while preserving literal `%` and quoted JSON arguments for Codegraph MCP ([#2220](https://github.com/can1357/oh-my-pi/issues/2220)).
|
||||
- Fixed the bundled `explore` agent's `thinking-level: med` frontmatter — not a valid effort (`minimal`/`low`/`medium`/`high`/`xhigh`), so it silently parsed to undefined and the agent ran without its intended thinking level
|
||||
- Discovery context-file reads (`~/.claude`, `~/.cursor`, project trees, `@`-imports) now stat-gate to regular files before reading: a FIFO/socket/char device dropped where a context file is expected previously blocked startup forever on a read that can never see EOF.
|
||||
- Fixed the read tool's provider-visible `path` schema and docs so web URLs and internal URI targets (`omp://`, `issue://`, `pr://`, etc.) are advertised alongside local files ([#2215](https://github.com/can1357/oh-my-pi/issues/2215)).
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
|
||||
/** Provider metadata needed to resolve append-only context mode. */
|
||||
export interface AppendOnlyContextModel {
|
||||
provider: string;
|
||||
@@ -5,19 +7,10 @@ export interface AppendOnlyContextModel {
|
||||
compat?: object;
|
||||
}
|
||||
|
||||
function isXiaomiHost(baseUrl: string): boolean {
|
||||
try {
|
||||
const host = new URL(baseUrl).hostname;
|
||||
return host === "xiaomimimo.com" || host.endsWith(".xiaomimimo.com");
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function shouldAutoEnableAppendOnlyContext(model: AppendOnlyContextModel | null | undefined): boolean {
|
||||
if (!model) return false;
|
||||
if (model.provider === "deepseek") return true;
|
||||
if (isXiaomiHost(model.baseUrl)) return true;
|
||||
if (hostMatchesUrl(model.baseUrl, "xiaomi")) return true;
|
||||
return !!model.compat && "supportsStore" in model.compat && model.compat.supportsStore === true;
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@ import * as path from "node:path";
|
||||
import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai/api-registry";
|
||||
import type { Api, Context, Model, SimpleStreamOptions, ThinkingConfig } from "@oh-my-pi/pi-ai/types";
|
||||
import type { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
||||
import { isVertexExpressOpenAIUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import { readModelCache } from "@oh-my-pi/pi-catalog/model-cache";
|
||||
import {
|
||||
createModelManager,
|
||||
@@ -136,7 +137,7 @@ function isAuthoritativeProjectCatalogModel(model: Model<Api>): boolean {
|
||||
return (
|
||||
model.provider === "google-vertex" &&
|
||||
model.api === "openai-completions" &&
|
||||
model.baseUrl.includes("/endpoints/openapi")
|
||||
isVertexExpressOpenAIUrl(model.baseUrl)
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -16,6 +16,7 @@
|
||||
|
||||
import { ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Api, Effort, KnownProvider, Model } from "@oh-my-pi/pi-ai";
|
||||
import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import { buildModelProviderPriorityRank } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
|
||||
@@ -169,14 +170,14 @@ function splitUpstreamRouting(pattern: string): { base: string; upstream: string
|
||||
|
||||
/** OpenRouter and Vercel AI Gateway are the aggregators that honor per-request upstream routing. */
|
||||
function supportsUpstreamRouting(model: Model<Api>): boolean {
|
||||
return model.baseUrl.includes("openrouter.ai") || model.baseUrl.includes("ai-gateway.vercel.sh");
|
||||
return modelMatchesHost(model, "openrouter") || modelMatchesHost(model, "vercelAIGateway");
|
||||
}
|
||||
|
||||
/** Pin a resolved aggregator model to a single upstream provider via its compat routing block. */
|
||||
function applyUpstreamRouting(model: Model<Api>, upstream: string): Model<Api> {
|
||||
const aggregatorModel = model as Model<"openai-completions">;
|
||||
const routing = { only: [upstream] };
|
||||
const compat = model.baseUrl.includes("ai-gateway.vercel.sh")
|
||||
const compat = modelMatchesHost(model, "vercelAIGateway")
|
||||
? { ...aggregatorModel.compat, vercelGatewayRouting: routing }
|
||||
: { ...aggregatorModel.compat, openRouterRouting: routing };
|
||||
return { ...model, compat } as Model<Api>;
|
||||
|
||||
@@ -44,6 +44,11 @@ export const OpenAICompatSchema = z.object({
|
||||
cacheControlFormat: z.enum(["anthropic"]).optional(),
|
||||
supportsStrictMode: z.boolean().optional(),
|
||||
toolStrictMode: z.enum(["all_strict", "none"]).optional(),
|
||||
streamIdleTimeoutMs: z.number().positive().optional(),
|
||||
supportsLongPromptCacheRetention: z.boolean().optional(),
|
||||
// anthropic-messages compat flags (same `compat` slot, per-api interpretation)
|
||||
requiresToolResultId: z.boolean().optional(),
|
||||
replayUnsignedThinking: z.boolean().optional(),
|
||||
});
|
||||
|
||||
const EffortSchema = z.enum(["minimal", "low", "medium", "high", "xhigh"]);
|
||||
|
||||
@@ -45,8 +45,8 @@ function makeReport(provider: string, email: string, limits: UsageReport["limits
|
||||
describe("buildRedactionMap", () => {
|
||||
it("masks everything past a two-char anchor when the anchor is unique", () => {
|
||||
const map = buildRedactionMap(["alpha@example.test", "bravo@example.test"]);
|
||||
expect(map.get("alpha@example.test")).toBe("an*");
|
||||
expect(map.get("bravo@example.test")).toBe("ha*");
|
||||
expect(map.get("alpha@example.test")).toBe("al*");
|
||||
expect(map.get("bravo@example.test")).toBe("br*");
|
||||
});
|
||||
|
||||
it("reveals a minimal middle-out differentiator instead of growing the prefix", () => {
|
||||
@@ -56,32 +56,32 @@ describe("buildRedactionMap", () => {
|
||||
// Masks must be pairwise distinct so accounts stay tellable-apart.
|
||||
expect(new Set(masks).size).toBe(masks.length);
|
||||
for (const mask of masks) {
|
||||
// Never leak the local part the way prefix growth would ("can.boluk@*").
|
||||
expect(mask).not.toContain("boluk");
|
||||
// Never leak the whole local part the way prefix growth would ("dummy@*").
|
||||
expect(mask).not.toContain("dummy");
|
||||
// anchor + at most a two-char differentiator.
|
||||
expect(mask).toMatch(/^ca\*(.{1,2}\*)?$/);
|
||||
expect(mask).toMatch(/^du\*(.{1,2}\*)?$/);
|
||||
}
|
||||
// The "89" account is distinguished by a digit only it contains.
|
||||
expect(map.get("dum.my9@example.net")).toBe("ca*9*");
|
||||
expect(map.get("dum.my9@example.net")).toBe("du*9*");
|
||||
});
|
||||
|
||||
it("gives duplicate identities the same mask", () => {
|
||||
const map = buildRedactionMap(["user@example.test", "user@example.test"]);
|
||||
expect(map.size).toBe(1);
|
||||
expect(map.get("user@example.test")).toBe("me*");
|
||||
expect(map.get("user@example.test")).toBe("us*");
|
||||
});
|
||||
});
|
||||
|
||||
describe("computeProviderWindowStats", () => {
|
||||
it("buckets by window duration, binds each account to its worst meter, and ceils the need", () => {
|
||||
it("buckets by window duration, binds each account to its worst meter, and reports remaining capacity", () => {
|
||||
const reports = [
|
||||
makeReport("anthropic", "a@x", [
|
||||
makeReport("anthropic", "account-a@example.test", [
|
||||
makeLimit({ id: "5h", usedFraction: 0.9, durationMs: FIVE_HOURS, windowId: "5h" }),
|
||||
makeLimit({ id: "7d", usedFraction: 0.1, durationMs: SEVEN_DAYS, windowId: "7d" }),
|
||||
// Tiered meter on the same window: higher burn must bind.
|
||||
makeLimit({ id: "7d-opus", usedFraction: 0.4, durationMs: SEVEN_DAYS, windowId: "7d", tier: "opus" }),
|
||||
]),
|
||||
makeReport("anthropic", "b@x", [
|
||||
makeReport("anthropic", "account-b@example.test", [
|
||||
makeLimit({ id: "5h", usedFraction: 0.4, durationMs: FIVE_HOURS, windowId: "5h" }),
|
||||
makeLimit({ id: "7d", usedFraction: 0.2, durationMs: SEVEN_DAYS, windowId: "7d" }),
|
||||
]),
|
||||
@@ -93,15 +93,15 @@ describe("computeProviderWindowStats", () => {
|
||||
expect(fiveHour.window).toBe("5h");
|
||||
expect(fiveHour.accounts).toBe(2);
|
||||
expect(fiveHour.usedAccounts).toBeCloseTo(1.3);
|
||||
expect(fiveHour.needed).toBe(2);
|
||||
expect(fiveHour.remainingAccounts).toBeCloseTo(0.7);
|
||||
expect(sevenDay.window).toBe("7d");
|
||||
expect(sevenDay.usedAccounts).toBeCloseTo(0.6); // 0.4 (opus binds) + 0.2
|
||||
expect(sevenDay.needed).toBe(1);
|
||||
expect(sevenDay.remainingAccounts).toBeCloseTo(1.4);
|
||||
});
|
||||
|
||||
it("ignores limits without a resolvable fraction", () => {
|
||||
const reports = [
|
||||
makeReport("anthropic", "a@x", [
|
||||
makeReport("anthropic", "account-a@example.test", [
|
||||
{
|
||||
id: "mystery",
|
||||
label: "mystery",
|
||||
@@ -116,23 +116,23 @@ describe("computeProviderWindowStats", () => {
|
||||
|
||||
describe("collectUnreportedAccounts", () => {
|
||||
const accounts: UsageAccountIdentity[] = [
|
||||
{ provider: "anthropic", type: "oauth", email: "seen@x.com" },
|
||||
{ provider: "anthropic", type: "oauth", email: "missing@x.com" },
|
||||
{ provider: "anthropic", type: "oauth", email: "seen@example.test" },
|
||||
{ provider: "anthropic", type: "oauth", email: "missing@example.test" },
|
||||
{ provider: "anthropic", type: "api_key" },
|
||||
{ provider: "cerebras", type: "api_key" },
|
||||
];
|
||||
const reports = [makeReport("anthropic", "seen@x.com", [])];
|
||||
const reports = [makeReport("anthropic", "seen@example.test", [])];
|
||||
|
||||
it("flags providers without reports and identified accounts missing from reports", () => {
|
||||
const unreported = collectUnreportedAccounts(reports, accounts);
|
||||
expect(unreported).toEqual([
|
||||
{ provider: "anthropic", type: "oauth", email: "missing@x.com" },
|
||||
{ provider: "anthropic", type: "oauth", email: "missing@example.test" },
|
||||
{ provider: "cerebras", type: "api_key" },
|
||||
]);
|
||||
});
|
||||
|
||||
it("does not claim unattributable credentials are missing when reports carry no identity", () => {
|
||||
const anonymous = [{ ...makeReport("anthropic", "seen@x.com", []), metadata: {} }];
|
||||
const anonymous = [{ ...makeReport("anthropic", "seen@example.test", []), metadata: {} }];
|
||||
const unreported = collectUnreportedAccounts(anonymous, accounts);
|
||||
expect(unreported).toEqual([{ provider: "cerebras", type: "api_key" }]);
|
||||
});
|
||||
@@ -140,33 +140,47 @@ describe("collectUnreportedAccounts", () => {
|
||||
|
||||
describe("formatUsageBreakdown", () => {
|
||||
const reports = [
|
||||
makeReport("anthropic", "dum.my9@example.net", [
|
||||
makeReport("anthropic", "dummy.primary@example.test", [
|
||||
makeLimit({ id: "Claude 5 Hour", usedFraction: 0.84, durationMs: FIVE_HOURS, windowId: "5h" }),
|
||||
]),
|
||||
makeReport("anthropic", "dummy@example.net", [
|
||||
makeReport("anthropic", "dummy.secondary@example.test", [
|
||||
makeLimit({ id: "Claude 5 Hour", usedFraction: 0.5, durationMs: FIVE_HOURS, windowId: "5h" }),
|
||||
]),
|
||||
];
|
||||
const accounts: UsageAccountIdentity[] = [
|
||||
{ provider: "anthropic", type: "oauth", email: "dum.my9@example.net" },
|
||||
{ provider: "anthropic", type: "oauth", email: "dummy@example.net" },
|
||||
{ provider: "anthropic", type: "oauth", email: "dummy.primary@example.test" },
|
||||
{ provider: "anthropic", type: "oauth", email: "dummy.secondary@example.test" },
|
||||
{ provider: "cerebras", type: "api_key" },
|
||||
];
|
||||
|
||||
it("renders every account: reported ones with limits, credential-only ones as no-data rows", () => {
|
||||
const text = stripVTControlCharacters(formatUsageBreakdown(reports, accounts, Date.now()));
|
||||
expect(text).toContain("dum.my9@example.net");
|
||||
expect(text).toContain("dummy.primary@example.test");
|
||||
expect(text).toContain("84.0% used");
|
||||
expect(text).toContain("Cerebras");
|
||||
expect(text).toContain("API key — no usage data");
|
||||
expect(text).toContain("need: 5h → 2 of 2 accounts");
|
||||
expect(text).toContain("capacity: 5h → 1.34/2 accounts used (0.66× quota left)");
|
||||
});
|
||||
|
||||
it("keeps near-exhausted capacity fractional instead of rounding it to an exact need", () => {
|
||||
const nearReports = [
|
||||
makeReport("anthropic", "near-a@example.test", [
|
||||
makeLimit({ id: "Claude 5 Hour", usedFraction: 1, durationMs: FIVE_HOURS, windowId: "5h" }),
|
||||
]),
|
||||
makeReport("anthropic", "near-b@example.test", [
|
||||
makeLimit({ id: "Claude 5 Hour", usedFraction: 0.99, durationMs: FIVE_HOURS, windowId: "5h" }),
|
||||
]),
|
||||
];
|
||||
const text = stripVTControlCharacters(formatUsageBreakdown(nearReports, [], Date.now()));
|
||||
expect(text).toContain("capacity: 5h → 1.99/2 accounts used (0.01× quota left)");
|
||||
expect(text).not.toContain("need:");
|
||||
});
|
||||
|
||||
it("redacts account labels through the provided map without leaking the originals", () => {
|
||||
const redaction = buildRedactionMap(["dum.my9@example.net", "dummy@example.net"]);
|
||||
const redaction = buildRedactionMap(["dummy.primary@example.test", "dummy.secondary@example.test"]);
|
||||
const text = stripVTControlCharacters(formatUsageBreakdown(reports, accounts, Date.now(), redaction));
|
||||
expect(text).not.toContain("dum.my9@example.net");
|
||||
expect(text).not.toContain("dummy@example.net");
|
||||
expect(text).not.toContain("dummy.primary@example.test");
|
||||
expect(text).not.toContain("dummy.secondary@example.test");
|
||||
for (const mask of redaction.values()) expect(text).toContain(mask);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,6 +1,10 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Fixed
|
||||
|
||||
- Fixed embedding provider detection to match `openrouter` by URL host, so custom embedding endpoints are now recognized correctly instead of being misclassified by substring matching
|
||||
- Fixed the check for OpenRouter base URLs so only true `openrouter` hosts are treated as non-custom
|
||||
|
||||
## [15.10.8] - 2026-06-09
|
||||
### Added
|
||||
|
||||
@@ -40,6 +40,7 @@
|
||||
},
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-catalog": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
"fastembed": "catalog:",
|
||||
"lru-cache": "catalog:",
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { homedir } from "node:os";
|
||||
import { join } from "node:path";
|
||||
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import {
|
||||
type Env,
|
||||
envBool,
|
||||
@@ -102,7 +103,7 @@ export function isApiEmbeddingModel(model = embeddingModel(), env: Env = process
|
||||
if (model.startsWith("openai/") || model.includes("text-embedding") || model.startsWith("text-embedding"))
|
||||
return true;
|
||||
const baseUrl = envString("MNEMOPI_EMBEDDING_API_URL", envString("OPENROUTER_BASE_URL", "", env), env);
|
||||
if (baseUrl && !baseUrl.includes("openrouter.ai")) return true;
|
||||
if (baseUrl && !hostMatchesUrl(baseUrl, "openrouter")) return true;
|
||||
return embeddingsViaApi(env);
|
||||
}
|
||||
|
||||
@@ -110,7 +111,7 @@ export function apiEmbeddingsAvailable(env: Env = process.env): boolean {
|
||||
if (embeddingsDisabled(env)) return false;
|
||||
if (!isApiEmbeddingModel(embeddingModel(env), env)) return false;
|
||||
const baseUrl = envString("MNEMOPI_EMBEDDING_API_URL", envString("OPENROUTER_BASE_URL", "", env), env);
|
||||
return Boolean(baseUrl && !baseUrl.includes("openrouter.ai")) || Boolean(embeddingApiKey(env));
|
||||
return Boolean(baseUrl && !hostMatchesUrl(baseUrl, "openrouter")) || Boolean(embeddingApiKey(env));
|
||||
}
|
||||
|
||||
export function workingMemoryMaxItems(env: Env = process.env): number {
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { mkdirSync } from "node:fs";
|
||||
import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts";
|
||||
import {
|
||||
$env,
|
||||
$flag,
|
||||
@@ -155,7 +156,7 @@ export function isApiModel(modelName: string): boolean {
|
||||
}
|
||||
const active = activeEmbeddingOptions();
|
||||
const baseUrl = active?.apiUrl ?? ($env.MNEMOPI_EMBEDDING_API_URL || $env.OPENROUTER_BASE_URL);
|
||||
if (baseUrl !== undefined && baseUrl !== "" && !baseUrl.includes("openrouter.ai")) {
|
||||
if (baseUrl !== undefined && baseUrl !== "" && !hostMatchesUrl(baseUrl, "openrouter")) {
|
||||
return true;
|
||||
}
|
||||
return $flag("MNEMOPI_EMBEDDINGS_VIA_API");
|
||||
@@ -246,7 +247,7 @@ async function getLocalModel(): Promise<LocalEmbeddingModel | null> {
|
||||
|
||||
async function embedApi(texts: readonly string[]): Promise<EmbeddingMatrix | null> {
|
||||
const baseUrl = embeddingBaseUrl();
|
||||
const isCustom = !baseUrl.includes("openrouter.ai");
|
||||
const isCustom = !hostMatchesUrl(baseUrl, "openrouter");
|
||||
const apiKey = embeddingApiKey();
|
||||
if (!isCustom && apiKey === "") {
|
||||
return null;
|
||||
@@ -335,7 +336,7 @@ export async function available(): Promise<boolean> {
|
||||
}
|
||||
if (isApiModel(defaultModel())) {
|
||||
const baseUrl = active?.apiUrl ?? ($env.MNEMOPI_EMBEDDING_API_URL || $env.OPENROUTER_BASE_URL);
|
||||
if (baseUrl !== undefined && baseUrl !== "" && !baseUrl.includes("openrouter.ai")) {
|
||||
if (baseUrl !== undefined && baseUrl !== "" && !hostMatchesUrl(baseUrl, "openrouter")) {
|
||||
return true;
|
||||
}
|
||||
return embeddingApiKey() !== "";
|
||||
|
||||
Reference in New Issue
Block a user