diff --git a/packages/agent/test/compaction-telemetry.test.ts b/packages/agent/test/compaction-telemetry.test.ts index 87e3b8cad..27887206b 100644 --- a/packages/agent/test/compaction-telemetry.test.ts +++ b/packages/agent/test/compaction-telemetry.test.ts @@ -27,6 +27,7 @@ import { import type { AgentMessage } from "@oh-my-pi/pi-agent-core/types"; import type { AssistantMessage, Model, Usage } from "@oh-my-pi/pi-ai"; import * as ai from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { SpanStatusCode } from "@opentelemetry/api"; import { BasicTracerProvider, @@ -35,7 +36,7 @@ import { SimpleSpanProcessor, } from "@opentelemetry/sdk-trace-base"; -const MODEL: Model = { +const MODEL: Model = buildModel({ id: "mock-model", name: "mock-model", api: "mock", @@ -46,7 +47,7 @@ const MODEL: Model = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 32_768, -}; +}); let exporter: InMemorySpanExporter; let provider: BasicTracerProvider; diff --git a/packages/agent/test/proxy-stream-disconnect.test.ts b/packages/agent/test/proxy-stream-disconnect.test.ts index 325fe5ca2..5fb66ad7f 100644 --- a/packages/agent/test/proxy-stream-disconnect.test.ts +++ b/packages/agent/test/proxy-stream-disconnect.test.ts @@ -10,8 +10,9 @@ import { describe, expect, it } from "bun:test"; import type { ProxyAssistantMessageEvent } from "@oh-my-pi/pi-agent-core/proxy"; import { type ProxyMessageEventStream, streamProxy } from "@oh-my-pi/pi-agent-core/proxy"; import type { AssistantMessageEvent, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; -const mockModel: Model = { +const mockModel: Model = buildModel({ id: "test-model", name: "Test Model", api: "openai", @@ -22,7 +23,7 @@ const mockModel: Model = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 4096, maxTokens: 1024, -}; +}); const mockContext: Context = { messages: [{ role: "user", content: "hello", timestamp: Date.now() }], diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index 693366cc2..0afec42af 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -1,9 +1,11 @@ import { describe, expect, test } from "bun:test"; import { buildOpenAiNativeHistory, requestOpenAiRemoteCompaction } from "@oh-my-pi/pi-agent-core/compaction/openai"; import type { AssistantMessage, FetchImpl, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; -function makeOpenAiModel(overrides: Partial> = {}): Model<"openai-responses"> { - return { +function makeOpenAiModel(overrides: Partial> = {}): Model<"openai-responses"> { + return buildModel({ id: "gpt-5", name: "GPT-5", api: "openai-responses", @@ -15,7 +17,7 @@ function makeOpenAiModel(overrides: Partial> = {}): Mo contextWindow: 400000, maxTokens: 128000, ...overrides, - }; + }); } describe("buildOpenAiNativeHistory custom tool calls", () => { diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 5dfcd1e00..066feee1e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -14,6 +14,7 @@ - Tool-argument validation errors now truncate embedded argument strings at 256 chars per field — a failed `write`-class call no longer echoes hundreds of KB of payload back to the model as the error message. - Auth storage no longer issues per-boot no-op writes: the schema-version row is only rewritten when the recorded version actually changes, and the credential identity-key backfill skips rows whose derived identity is null — reopening a current-schema database now performs zero write transactions - Plain provider env-var names moved to the catalog table: registry defs dropped their 48 `envKeys` literals (including the pure `$pickenv` pickers for `huggingface`/`qwen-portal`/`xai-oauth`), `getEnvApiKey` now derives those fallbacks from `CATALOG_PROVIDERS[].envVars`, and `envKeys` remains only for computed resolvers (Anthropic Foundry, Vertex ADC, Bedrock credential chains) and non-catalog providers (`kagi`, `tavily`, `parallel`, `perplexity`) +- Protocol handlers are now pure `model.compat` readers — the per-request `resolve*Compat`/`detect*Compat` calls (anthropic ×11, responses ×3, completions wrappers), inline `strictResponsesPairing` host detection, the OpenCode `reasoning_content` mutation block, and all `resolvedBaseUrl` threading are gone. Compat is materialized once at model build time (`@oh-my-pi/pi-catalog` `buildModel`); the OpenCode thinking-mode quirk is a precomputed `compat.whenThinking` pointer swap, and request-time base-URL overrides only feed the HTTP client. Behavior is unchanged (the Anthropic `supportsLongCacheRetention` official-endpoint gate is folded into detection). ### Fixed diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 8b05244c6..27aa2d719 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2,7 +2,7 @@ import * as nodeCrypto from "node:crypto"; import * as fs from "node:fs"; import { scheduler } from "node:timers/promises"; import * as tls from "node:tls"; -import { isOfficialAnthropicApiUrl, resolveAnthropicCompat } from "@oh-my-pi/pi-catalog/compat/anthropic"; +import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic"; import { supportsAdaptiveThinkingDisplay } from "@oh-my-pi/pi-catalog/identity"; import { hasOpus47ApiRestrictions, mapEffortToAnthropicAdaptiveEffort } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; @@ -412,7 +412,6 @@ function dropAnthropicStrictTools(params: MessageCreateParamsStreaming): void { function getCacheControl( model: Model<"anthropic-messages">, - baseUrl: string, cacheRetention: CacheRetention | undefined, isOAuthToken: boolean, ): { retention: CacheRetention; cacheControl?: AnthropicCacheControl } { @@ -420,12 +419,7 @@ function getCacheControl( if (retention === "none") { return { retention }; } - const ttl = - retention === "long" && - isOfficialAnthropicApiUrl(baseUrl) && - resolveAnthropicCompat(model).supportsLongCacheRetention - ? "1h" - : undefined; + const ttl = retention === "long" && model.compat.supportsLongCacheRetention ? "1h" : undefined; return { retention, cacheControl: { type: "ephemeral", ...(ttl && { ttl }) }, @@ -1581,7 +1575,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( const sendsAdaptiveEffortPin = options?.thinkingEnabled === false && model.thinking?.mode === "anthropic-adaptive" && - !resolveAnthropicCompat(model).disableAdaptiveThinking; + !model.compat.disableAdaptiveThinking; if ( model.reasoning && (options?.thinkingEnabled || sendsAdaptiveEffortPin) && @@ -1589,10 +1583,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( ) { extraBetas.push(effortBeta); } - if ( - resolveAnthropicCompat(model).supportsMidConversationSystem && - !extraBetas.includes(midConversationSystemBeta) - ) { + if (model.compat.supportsMidConversationSystem && !extraBetas.includes(midConversationSystemBeta)) { // convertAnthropicMessages may upgrade developer turns to the // mid-conversation `system` role on these models; API-key requests // need the beta alongside the role (OAuth agent requests already @@ -1620,7 +1611,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( } const preparedContext = await prepareAnthropicManyImageContext(context, model.input.includes("image")); const prepareParams = async (): Promise => { - let nextParams = buildParams(model, baseUrl, preparedContext, isOAuthToken, options, disableStrictTools); + let nextParams = buildParams(model, preparedContext, isOAuthToken, options, disableStrictTools); if (disableStrictTools) { dropAnthropicStrictTools(nextParams); } @@ -2287,7 +2278,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A isOAuth, claudeCodeSessionId, } = args; - const compat = resolveAnthropicCompat(model); + const compat = model.compat; const needsInterleavedBeta = interleavedThinking && !supportsAdaptiveThinkingDisplay(model.id); const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming; const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey); @@ -2398,7 +2389,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A const authorizationHeader = getHeaderCaseInsensitive(defaultHeaders, "Authorization"); const shouldSuppressClientApiKey = !oauthToken && - !isOfficialAnthropicApiUrl(baseUrl) && + !model.compat.officialEndpoint && typeof authorizationHeader === "string" && /^Bearer\s+/i.test(authorizationHeader); @@ -2707,13 +2698,12 @@ function extractClaudeCodeFirstUserMessageText(messages: readonly Message[]): st function buildParams( model: Model<"anthropic-messages">, - baseUrl: string, context: Context, isOAuthToken: boolean, options?: AnthropicOptions, disableStrictTools = false, ): MessageCreateParamsStreaming { - const { cacheControl } = getCacheControl(model, baseUrl, options?.cacheRetention, isOAuthToken); + const { cacheControl } = getCacheControl(model, options?.cacheRetention, isOAuthToken); // Pre-compute system blocks so they occupy the right slot in the serialized body. const shouldInjectClaudeCodeInstruction = isOAuthToken && !model.id.startsWith("claude-3-5-haiku"); @@ -2732,7 +2722,7 @@ function buildParams( context.tools, isOAuthToken, disableStrictTools || model.provider === "github-copilot", - resolveAnthropicCompat(model).supportsEagerToolInputStreaming, + model.compat.supportsEagerToolInputStreaming, ); } else if (isOAuthToken) { tools = []; @@ -2755,7 +2745,7 @@ function buildParams( if (options?.thinkingEnabled) { const mode = model.thinking?.mode; const effort = resolveAnthropicAdaptiveEffort(model, options); - const compat = resolveAnthropicCompat(model); + const compat = model.compat; if (mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { const adaptive: { type: "adaptive"; display?: AnthropicThinkingDisplay } = { type: "adaptive" }; // Starting with Claude Opus 4.7 and Claude Fable/Mythos 5, adaptive thinking @@ -2778,7 +2768,7 @@ function buildParams( if (mode === "anthropic-budget-effort" && effort) outputConfigEffort = effort; } } else if (options?.thinkingEnabled === false) { - const compat = resolveAnthropicCompat(model); + const compat = model.compat; if (model.thinking?.mode === "anthropic-adaptive" && !compat.disableAdaptiveThinking) { // Adaptive-only Claude models (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5) reject // `thinking.type: "disabled"` — adaptive thinking cannot be switched off. @@ -2813,7 +2803,7 @@ function buildParams( // metadata → max_tokens → thinking → context_management → output_config → stream. const params: MessageCreateParamsStreaming = { model: model.id, - messages: convertAnthropicMessages(context.messages, model, isOAuthToken, baseUrl), + messages: convertAnthropicMessages(context.messages, model, isOAuthToken), ...(systemBlocks && { system: systemBlocks }), ...(tools !== undefined && { tools }), ...(metadata && { metadata }), @@ -2867,7 +2857,7 @@ function buildParams( // request succeeds; the tool stays available and the caller's prompt steers // the model toward it. const choiceType = params.tool_choice?.type; - if ((choiceType === "any" || choiceType === "tool") && !resolveAnthropicCompat(model).supportsForcedToolChoice) { + if ((choiceType === "any" || choiceType === "tool") && !model.compat.supportsForcedToolChoice) { params.tool_choice = { type: "auto" }; } } @@ -2888,7 +2878,7 @@ function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResul content: convertContentBlocks(msg.content, model.input.includes("image")), is_error: msg.isError, }; - if (resolveAnthropicCompat(model).requiresToolResultId) { + if (model.compat.requiresToolResultId) { // Z.AI workaround (issue #814): include `id` aliased to `tool_use_id`. (block as unknown as Record).id = msg.toolCallId; } @@ -2936,7 +2926,6 @@ export function convertAnthropicMessages( messages: Message[], model: Model<"anthropic-messages">, isOAuthToken: boolean, - baseUrl = resolveAnthropicBaseUrl(model), ): AnthropicMessageParam[] { // Indices of params emitted from `developer` messages. After the main pass, // the ones whose placement satisfies Anthropic's mid-conversation rules are @@ -3001,7 +2990,7 @@ export function convertAnthropicMessages( } if (block.thinking.trim().length === 0) continue; if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) { - if (resolveAnthropicCompat(model, baseUrl).replayUnsignedThinking) { + if (model.compat.replayUnsignedThinking) { blocks.push({ type: "thinking", thinking: block.thinking.toWellFormed(), @@ -3079,7 +3068,7 @@ export function convertAnthropicMessages( // never consecutive. Requiring the next param to be `assistant` (or absent) // covers both the "followed by assistant / last" and "no consecutive system" // constraints. Anything that does not qualify stays a `user` message. - if (developerParamIndices.length > 0 && resolveAnthropicCompat(model).supportsMidConversationSystem) { + if (developerParamIndices.length > 0 && model.compat.supportsMidConversationSystem) { for (const idx of developerParamIndices) { const followsUser = idx > 0 && params[idx - 1]?.role === "user"; const next = params[idx + 1]; diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 15506057c..d352133cb 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -1,4 +1,3 @@ -import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import { AzureOpenAI, APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai"; import type { @@ -137,7 +136,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses" const apiKey = options?.apiKey || getEnvApiKey(model.provider) || ""; const client = createClient(model, apiKey, options); const { baseUrl } = resolveAzureConfig(model, options); - const params = buildParams(model, context, options, deploymentName, baseUrl); + const params = buildParams(model, context, options, deploymentName); options?.onPayload?.(params); const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); const firstEventTimeoutMs = @@ -297,9 +296,8 @@ function buildParams( context: Context, options: AzureOpenAIResponsesOptions | undefined, deploymentName: string, - resolvedBaseUrl?: string, ) { - const messages = convertMessages(model, context, true, resolvedBaseUrl); + const messages = convertMessages(model, context, true); const params: AzureOpenAIResponsesSamplingParams = { model: deploymentName, @@ -329,7 +327,6 @@ function convertMessages( model: Model<"azure-openai-responses">, context: Context, strictResponsesPairing: boolean, - resolvedBaseUrl?: string, ): ResponseInput { const messages: ResponseInput = []; const transformedMessages = transformMessages(context.messages, model, normalizeResponsesToolCallIdForTransform); @@ -338,10 +335,7 @@ function convertMessages( const systemPrompts = normalizeSystemPrompts(context.systemPrompt); if (systemPrompts.length > 0) { - const role = - model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole - ? "developer" - : "system"; + const role = model.reasoning && model.compat.supportsDeveloperRole ? "developer" : "system"; for (const systemPrompt of systemPrompts) { messages.push({ role, content: systemPrompt }); } diff --git a/packages/ai/src/providers/gitlab-duo.ts b/packages/ai/src/providers/gitlab-duo.ts index 382ce56c9..8d812c153 100644 --- a/packages/ai/src/providers/gitlab-duo.ts +++ b/packages/ai/src/providers/gitlab-duo.ts @@ -1,5 +1,6 @@ +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { ANTHROPIC_THINKING, mapAnthropicToolChoice } from "../stream"; -import type { Api, Context, FetchImpl, Model, SimpleStreamOptions } from "../types"; +import type { Api, Context, FetchImpl, Model, ModelSpec, SimpleStreamOptions } from "../types"; import { AssistantMessageEventStream } from "../utils/event-stream"; import { createProviderErrorMessage } from "./error-message"; import type { OpenAICompletionsOptions } from "./openai-completions"; @@ -145,23 +146,25 @@ export function getModelMapping(modelId: string): GitLabModelMapping | undefined } export function getGitLabDuoModels(): Model[] { - return Object.entries(MODEL_MAPPINGS).map(([id, mapping]) => ({ - id, - name: mapping.name, - api: - mapping.provider === "anthropic" - ? "anthropic-messages" - : mapping.openaiApiType === "responses" - ? "openai-responses" - : "openai-completions", - provider: "gitlab-duo", - baseUrl: mapping.provider === "anthropic" ? ANTHROPIC_PROXY_URL : OPENAI_PROXY_URL, - reasoning: mapping.reasoning, - input: [...mapping.input], - cost: { ...mapping.cost }, - contextWindow: mapping.contextWindow, - maxTokens: mapping.maxTokens, - })); + return Object.entries(MODEL_MAPPINGS).map(([id, mapping]) => + buildModel({ + id, + name: mapping.name, + api: + mapping.provider === "anthropic" + ? "anthropic-messages" + : mapping.openaiApiType === "responses" + ? "openai-responses" + : "openai-completions", + provider: "gitlab-duo", + baseUrl: mapping.provider === "anthropic" ? ANTHROPIC_PROXY_URL : OPENAI_PROXY_URL, + reasoning: mapping.reasoning, + input: [...mapping.input], + cost: { ...mapping.cost }, + contextWindow: mapping.contextWindow, + maxTokens: mapping.maxTokens, + } as ModelSpec), + ); } interface DirectAccessToken { @@ -255,12 +258,13 @@ export function streamGitLabDuo( const inner = mapping.provider === "anthropic" ? streamAnthropic( - { + buildModel({ ...model, id: mapping.model, api: "anthropic-messages", baseUrl: ANTHROPIC_PROXY_URL, - } as Model<"anthropic-messages">, + compat: model.compatConfig, + } as ModelSpec<"anthropic-messages">), context, { apiKey: directAccess.token, @@ -293,12 +297,13 @@ export function streamGitLabDuo( ) : mapping.openaiApiType === "responses" ? streamOpenAIResponses( - { + buildModel({ ...model, id: mapping.model, api: "openai-responses", baseUrl: OPENAI_PROXY_URL, - } as Model<"openai-responses">, + compat: model.compatConfig, + } as ModelSpec<"openai-responses">), context, { apiKey: directAccess.token, @@ -325,12 +330,13 @@ export function streamGitLabDuo( } satisfies OpenAIResponsesOptions, ) : streamOpenAICompletions( - { + buildModel({ ...model, id: mapping.model, api: "openai-completions", baseUrl: OPENAI_PROXY_URL, - } as Model<"openai-completions">, + compat: model.compatConfig, + } as ModelSpec<"openai-completions">), context, { apiKey: directAccess.token, diff --git a/packages/ai/src/providers/mock.ts b/packages/ai/src/providers/mock.ts index cc18c0d96..5e9c87cfc 100644 --- a/packages/ai/src/providers/mock.ts +++ b/packages/ai/src/providers/mock.ts @@ -168,6 +168,7 @@ export class MockModel implements Model { readonly cost: Model["cost"]; readonly contextWindow: number; readonly maxTokens: number; + readonly compat = undefined; /** Recorded calls in invocation order. */ readonly calls: MockCall[] = []; diff --git a/packages/ai/src/providers/openai-anthropic-shim.ts b/packages/ai/src/providers/openai-anthropic-shim.ts index a4f9b8fac..6d587d71f 100644 --- a/packages/ai/src/providers/openai-anthropic-shim.ts +++ b/packages/ai/src/providers/openai-anthropic-shim.ts @@ -8,8 +8,9 @@ * here once. */ +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { ANTHROPIC_THINKING } from "../stream"; -import type { Context, Model, SimpleStreamOptions } from "../types"; +import type { Context, Model, ModelSpec, SimpleStreamOptions } from "../types"; import { AssistantMessageEventStream } from "../utils/event-stream"; import { createProviderErrorMessage } from "./error-message"; import { streamAnthropic, streamOpenAICompletions } from "./register-builtins"; @@ -56,7 +57,7 @@ export function streamOpenAIAnthropicShim( }; if (format === "anthropic") { - const anthropicModel: Model<"anthropic-messages"> = { + const anthropicModel = buildModel({ id: model.id, name: model.name, api: "anthropic-messages", @@ -68,7 +69,7 @@ export function streamOpenAIAnthropicShim( reasoning: model.reasoning, input: model.input, cost: model.cost, - }; + } as ModelSpec<"anthropic-messages">); const reasoningEffort = options?.reasoning; const thinkingEnabled = !!reasoningEffort && model.reasoning; @@ -101,7 +102,12 @@ export function streamOpenAIAnthropicShim( } } else { const openaiModel: Model<"openai-completions"> = config.openaiBaseUrl - ? { ...model, baseUrl: config.openaiBaseUrl, headers: mergedHeaders } + ? buildModel({ + ...model, + baseUrl: config.openaiBaseUrl, + headers: mergedHeaders, + compat: model.compatConfig, + } as ModelSpec<"openai-completions">) : model; const reasoningEffort = options?.reasoning; diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 3acdc904b..0a1d83eb3 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -1,10 +1,9 @@ -import { detectOpenAICompat, type ResolvedOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; import type { Effort } from "@oh-my-pi/pi-catalog/effort"; import { toFirepassWireModelId, toFireworksWireModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id"; -import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts"; -import { isDeepseekModelIdOrName, isKimiModelId } from "@oh-my-pi/pi-catalog/identity"; +import { isDeepseekModelIdOrName } from "@oh-my-pi/pi-catalog/identity"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; +import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot"; import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import OpenAI, { APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai"; @@ -432,7 +431,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( try { const apiKey = options?.apiKey || getEnvApiKey(model.provider) || ""; - const idleTimeoutFallbackMs = resolveOpenAICompat(model).streamIdleTimeoutMs; + const idleTimeoutFallbackMs = model.compat.streamIdleTimeoutMs; const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs); const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs); @@ -459,13 +458,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( const createCompletionsStream = async (toolStrictModeOverride?: ToolStrictModeOverride) => { clearCapturedErrorResponse(); const effectiveToolStrictModeOverride = disableStrictTools ? "none" : toolStrictModeOverride; - const { params, toolStrictMode } = buildParams( - model, - context, - options, - baseUrl, - effectiveToolStrictModeOverride, - ); + const { params, toolStrictMode } = buildParams(model, context, options, effectiveToolStrictModeOverride); appliedToolStrictMode = toolStrictMode; options?.onPayload?.(params); rawRequestDump = { @@ -1180,64 +1173,32 @@ function buildParams( model: Model<"openai-completions">, context: Context, options: OpenAICompletionsOptions | undefined, - resolvedBaseUrl?: string, toolStrictModeOverride?: ToolStrictModeOverride, ): { params: OpenAICompletionsParams; toolStrictMode: AppliedToolStrictMode } { - const compat = getCompat(model, resolvedBaseUrl); - // Opencode Zen's gateway (https://opencode.ai/zen/go/v1) gates - // `reasoning_content` on the request's thinking state for every model it - // fronts (Kimi K2.x, DeepSeek V4, GLM-5.x, Qwen3.x, MiMo, MiniMax, …): it - // 400s with `Extra inputs are not permitted` when thinking is off but the - // field is supplied (#1071), and 400s with `thinking is enabled but - // reasoning_content is missing in assistant tool call message at index N` - // (#1484) when thinking is on and the field is absent. `detectOpenAICompat` - // only set `requiresReasoningContentForToolCalls` for the DeepSeek family - // (and previously for Kimi until #1071 carved out opencode); reactivate it - // per request for every opencode model whenever this turn is in thinking - // mode so prior tool-call turns replay reasoning_content. Forced-tool - // turns are excluded because the later `disableReasoningOnForcedToolChoice` - // guard at the bottom of `buildParams` strips thinking from the wire body - // for Kimi-style models — keeping the replay on under those conditions - // would resurrect the #1071 failure. - // - // `allowsSyntheticReasoningContentForToolCalls` is forced to `false` on - // the same path: the gateway specifically requires `reasoning_content`, - // and the default synthetic-friendly behavior would echo whichever field - // the upstream streamed (e.g. `reasoning` for many opencode turns), - // landing the replay in the wrong key and re-triggering the 400. - const isOpenCodeProvider = model.provider === "opencode-go" || model.provider === "opencode-zen"; + let compat = model.compat; const thinkingEnabledForRequest = Boolean(options?.reasoning) && !options?.disableReasoning && Boolean(model.reasoning); const forcedToolChoiceSuppressesThinking = compat.disableReasoningOnForcedToolChoice && isForcedToolChoice(mapToOpenAICompletionsToolChoice(options?.toolChoice)); - if (isOpenCodeProvider && thinkingEnabledForRequest && !forcedToolChoiceSuppressesThinking) { - compat.requiresReasoningContentForToolCalls = true; - compat.allowsSyntheticReasoningContentForToolCalls = false; - compat.reasoningContentField = "reasoning_content"; + if (compat.whenThinking && thinkingEnabledForRequest && !forcedToolChoiceSuppressesThinking) { + compat = compat.whenThinking; // precomputed at model build — pointer swap, no allocation } - const isKimiFamilyModel = isKimiModelId(model.id); - const isOpenRouter = modelMatchesHost(model, "openrouter"); const messages = convertMessages(model, context, compat); maybeAddAnthropicCacheControl(compat, messages); - const supportsReasoningParams = model.provider !== "github-copilot"; + const supportsReasoningParams = compat.supportsReasoningParams; - // Kimi (including via OpenRouter and Fireworks router-form IDs such as - // `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on - // max_tokens, not actual output. The official Kimi K2 model guidance - // (https://docs.fireworks.ai/models/kimi-k2) also requires `max_tokens` for - // every call since the family can otherwise emit very long reasoning traces - // before the final answer. Always send max_tokens — match the same - // Kimi-family regex used by the compat detector. - // Note: Direct kimi-code provider is handled by the dedicated Kimi provider in kimi.ts. - const requestedMaxTokens = options?.maxTokens ?? (isKimiFamilyModel ? model.maxTokens : undefined); + // Kimi-family models calculate TPM rate limits from max_tokens (not actual + // output) and the official guidance requires sending it on every call — + // `compat.alwaysSendMaxTokens` carries that detection. + const requestedMaxTokens = options?.maxTokens ?? (compat.alwaysSendMaxTokens ? model.maxTokens : undefined); // OpenRouter fans out to upstreams whose output caps differ from the catalog // value (which tracks the highest-cap provider). A max_tokens above the routed // upstream's cap makes OpenRouter silently skip that provider (e.g. Cerebras // GLM-4.7, ~40k) for a higher-cap one, defeating `provider.order`/`only`. Omit - // it for OpenRouter so each upstream self-caps and routing is honored. Kimi is - // exempt — it derives TPM rate limits from max_tokens (see above). - const omitMaxTokensForRouting = isOpenRouter && !isKimiFamilyModel; + // it for OpenRouter so each upstream self-caps and routing is honored — unless + // the model always requires max_tokens (Kimi TPM accounting, see above). + const omitMaxTokensForRouting = compat.isOpenRouterHost && !compat.alwaysSendMaxTokens; const effectiveMaxTokens = requestedMaxTokens === undefined || omitMaxTokensForRouting ? undefined @@ -1406,13 +1367,13 @@ function buildParams( } // OpenRouter provider routing preferences - if (modelMatchesHost(model, "openrouter") && compat.openRouterRouting) { + if (compat.isOpenRouterHost && compat.openRouterRouting) { params.provider = compat.openRouterRouting; } // Vercel AI Gateway provider routing preferences - if (modelMatchesHost(model, "vercelAIGateway") && model.compat?.vercelGatewayRouting) { - const routing = model.compat.vercelGatewayRouting; + if (compat.isVercelGatewayHost && compat.vercelGatewayRouting) { + const routing = compat.vercelGatewayRouting; if (routing.only || routing.order) { const gatewayOptions: Record = {}; if (routing.only) gatewayOptions.only = routing.only; @@ -2085,22 +2046,3 @@ function mapStopReason(reason: ChatCompletionChunk.Choice["finish_reason"] | str }; } } - -/** - * Detect compatibility settings from provider and baseUrl for known providers. - * Provider takes precedence over URL-based detection since it's explicitly configured. - * Returns a fully resolved OpenAICompat object with all fields set. - */ -export function detectCompat(model: Model<"openai-completions">): ResolvedOpenAICompat { - return detectOpenAICompat(model); -} - -/** - * Get resolved compatibility settings for a model. - * Uses explicit model.compat if provided, otherwise auto-detects from provider/URL. - * @param model - The model configuration - * @param resolvedBaseUrl - Optional resolved base URL (e.g., after GitHub Copilot proxy-ep resolution). - */ -function getCompat(model: Model<"openai-completions">, resolvedBaseUrl?: string): ResolvedOpenAICompat { - return resolveOpenAICompat(model, resolvedBaseUrl); -} diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index ef7e0653c..cb18790fa 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -1,5 +1,3 @@ -import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; -import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts"; import { parseGitHubCopilotApiKey } from "@oh-my-pi/pi-catalog/wire/github-copilot"; import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import OpenAI, { APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai"; @@ -229,7 +227,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = ( ); const premiumRequestsTotal = copilotPremiumRequests; const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState); - const params = buildParams(model, context, options, providerSessionState, baseUrl); + const params = buildParams(model, context, options, providerSessionState); const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs); @@ -382,7 +380,7 @@ function createClient( copilotPremiumRequests = copilot.premiumRequests; baseUrl = resolveGitHubCopilotBaseUrl(model.baseUrl, rawApiKey) ?? model.baseUrl; } - if (sessionId && model.provider === "openai" && (baseUrl ?? "").toLowerCase().includes("api.openai.com")) { + if (sessionId && model.provider === "openai") { headers.session_id ??= sessionId; headers["x-client-request-id"] ??= sessionId; } @@ -425,18 +423,14 @@ function buildParams( context: Context, options: OpenAIResponsesOptions | undefined, providerSessionState: OpenAIResponsesProviderSessionState | undefined, - resolvedBaseUrl?: string, ): OpenAIResponsesSamplingParams { - const strictResponsesPairing = - options?.strictResponsesPairing ?? - (hostMatchesUrl(model.baseUrl ?? "", "azureOpenAI") || model.provider === "github-copilot"); + const strictResponsesPairing = options?.strictResponsesPairing ?? model.compat.strictResponsesPairing; const messages = convertConversationMessages(model, context, strictResponsesPairing, providerSessionState, options); const systemPrompts = normalizeSystemPrompts(context.systemPrompt); let systemInstructions: string | undefined; if (systemPrompts.length > 0) { - const needsDeveloperRole = - model.reasoning && resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsDeveloperRole; + const needsDeveloperRole = model.reasoning && model.compat.supportsDeveloperRole; if (needsDeveloperRole) { // Reasoning models on known OpenAI-compatible endpoints require the // `developer` role. Send all system prompts inline in `input`. @@ -460,8 +454,7 @@ function buildParams( stream: true, prompt_cache_key: promptCacheKey, prompt_cache_retention: promptCacheKey - ? cacheRetention === "long" && - resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsLongPromptCacheRetention + ? cacheRetention === "long" && model.compat.supportsLongPromptCacheRetention ? "24h" : undefined : undefined, @@ -476,11 +469,7 @@ function buildParams( // `StreamOptions.frequencyPenalty` is intentionally dropped for this provider. if (context.tools) { - params.tools = convertTools( - context.tools, - resolveOpenAIResponsesCompat(model, resolvedBaseUrl).supportsStrictMode, - model, - ); + params.tools = convertTools(context.tools, model.compat.supportsStrictMode, model); if (options?.toolChoice) { params.tool_choice = mapOpenAIResponsesToolChoiceForTools(options.toolChoice, context.tools, model); } @@ -503,7 +492,7 @@ function buildParams( effort => mapReasoningEffort( effort as NonNullable, - model.compat?.reasoningEffortMap, + model.compat.reasoningEffortMap, ), options?.includeEncryptedReasoning ?? true, options?.omitReasoningEffort ?? false, diff --git a/packages/ai/test/anthropic-abandoned-tooluse-replay.test.ts b/packages/ai/test/anthropic-abandoned-tooluse-replay.test.ts index 8e774971e..91da4797e 100644 --- a/packages/ai/test/anthropic-abandoned-tooluse-replay.test.ts +++ b/packages/ai/test/anthropic-abandoned-tooluse-replay.test.ts @@ -1,6 +1,14 @@ import { describe, expect, it } from "bun:test"; import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; -import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; +import type { + AssistantMessage, + Message, + Model, + ModelSpec, + ToolResultMessage, + UserMessage, +} from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; // These tests pin the wire-validity contract that was verified end-to-end against the // live Anthropic Messages API (claude-opus-4-8): @@ -25,7 +33,7 @@ import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } // continues. That continuation is valid only when the transform preserves latest signed // thinking and downgrades historical/invalid signed thinking. -const model: Model<"anthropic-messages"> = { +const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-opus-4-8", @@ -36,7 +44,7 @@ const model: Model<"anthropic-messages"> = { maxTokens: 8_192, contextWindow: 200_000, reasoning: true, -}; +}); const emptyUsage = { input: 0, @@ -158,10 +166,11 @@ describe("Anthropic abandoned/aborted tool-use replay", () => { // The whole signature must replay as native signed thinking even when the first-party // provider is routed through an LLM gateway baseUrl, which still reaches signature-enforcing // Anthropic. Dropping it would emit signature:"" and 400 the gateway. - const gatewayModel: Model<"anthropic-messages"> = { + const gatewayModel: Model<"anthropic-messages"> = buildModel({ ...model, baseUrl: "https://llm2.example.com/abc/v1/messages", - }; + compat: model.compatConfig, + } as ModelSpec<"anthropic-messages">); const user: UserMessage = { role: "user", content: "deploy the update", timestamp: 1 }; const aborted: AssistantMessage = { role: "assistant", diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index 526b416f2..d23b6a8bf 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -22,11 +22,20 @@ import { stripClaudeToolPrefix, } from "@oh-my-pi/pi-ai/providers/anthropic"; import { getEnvApiKey } from "@oh-my-pi/pi-ai/stream"; -import type { AssistantMessage, Context, Model, TJsonSchema, TokenTaskBudget, Tool } from "@oh-my-pi/pi-ai/types"; +import type { + AssistantMessage, + Context, + Model, + ModelSpec, + TJsonSchema, + TokenTaskBudget, + Tool, +} from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import * as z from "zod/v4"; import { withEnv } from "./helpers"; -const ANTHROPIC_MODEL: Model<"anthropic-messages"> = { +const ANTHROPIC_MODEL_SPEC: ModelSpec<"anthropic-messages"> = { id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -39,13 +48,15 @@ const ANTHROPIC_MODEL: Model<"anthropic-messages"> = { maxTokens: 8_192, }; -const CLOUDFLARE_ANTHROPIC_MODEL: Model<"anthropic-messages"> = { - ...ANTHROPIC_MODEL, +const ANTHROPIC_MODEL: Model<"anthropic-messages"> = buildModel(ANTHROPIC_MODEL_SPEC); + +const CLOUDFLARE_ANTHROPIC_MODEL: Model<"anthropic-messages"> = buildModel({ + ...ANTHROPIC_MODEL_SPEC, id: "anthropic/claude-sonnet-4-5", name: "Claude Sonnet 4.5 via Cloudflare", provider: "cloudflare-ai-gateway", baseUrl: "https://gateway.ai.cloudflare.com/v1/account/gateway/anthropic", -}; +}); function createAbortedSignal(): AbortSignal { const controller = new AbortController(); @@ -284,7 +295,7 @@ describe("Anthropic request fingerprint alignment", () => { it("clamps requested max_tokens to Claude Code's 64k cap when the model ceiling is higher", async () => { const payload = (await captureAnthropicPayload( - { ...ANTHROPIC_MODEL, id: "claude-opus-4-8", name: "Claude Opus 4.8", maxTokens: 128_000 }, + buildModel({ ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-8", name: "Claude Opus 4.8", maxTokens: 128_000 }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -303,7 +314,7 @@ describe("Anthropic request fingerprint alignment", () => { it("keeps the full model output ceiling for API-key requests", async () => { const payload = (await captureAnthropicPayload( - { ...ANTHROPIC_MODEL, id: "claude-opus-4-8", name: "Claude Opus 4.8", maxTokens: 128_000 }, + buildModel({ ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-8", name: "Claude Opus 4.8", maxTokens: 128_000 }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -361,12 +372,12 @@ describe("Anthropic request fingerprint alignment", () => { { status: 400, headers: { "Content-Type": "application/json" } }, ); }) as typeof fetch; - const adaptiveModel: Model<"anthropic-messages"> = { - ...ANTHROPIC_MODEL, + const adaptiveModel: Model<"anthropic-messages"> = buildModel({ + ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-8-20260528", name: "Claude Opus 4.8", thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh }, - }; + }); await streamAnthropic( adaptiveModel, @@ -518,7 +529,7 @@ describe("Anthropic request fingerprint alignment", () => { it("skips Claude Code instruction injection for claude-3-5-haiku models", async () => { const payload = (await captureAnthropicPayload( - { ...ANTHROPIC_MODEL, id: "claude-3-5-haiku", name: "Claude 3.5 Haiku" }, + buildModel({ ...ANTHROPIC_MODEL_SPEC, id: "claude-3-5-haiku", name: "Claude 3.5 Haiku" }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1263,7 +1274,7 @@ describe("Anthropic request fingerprint alignment", () => { it("keeps the interleaved-thinking beta for dated Opus 4.0 ids", () => { const legacy = buildAnthropicClientOptions({ - model: { ...ANTHROPIC_MODEL, id: "claude-opus-4-20250514", name: "Claude Opus 4" }, + model: buildModel({ ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-20250514", name: "Claude Opus 4" }), apiKey: "sk-ant-api-test", extraBetas: [], stream: true, @@ -1274,7 +1285,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(legacy.defaultHeaders["anthropic-beta"]).toContain("interleaved-thinking-2025-05-14"); const modern = buildAnthropicClientOptions({ - model: { ...ANTHROPIC_MODEL, id: "claude-opus-4-7", name: "Claude Opus 4.7" }, + model: buildModel({ ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-7", name: "Claude Opus 4.7" }), apiKey: "sk-ant-api-test", extraBetas: [], stream: true, @@ -1285,10 +1296,10 @@ describe("Anthropic request fingerprint alignment", () => { }); it("adds legacy fine-grained tool-streaming beta only for tool requests on incompatible models", () => { - const incompatibleModel: Model<"anthropic-messages"> = { - ...ANTHROPIC_MODEL, + const incompatibleModel: Model<"anthropic-messages"> = buildModel({ + ...ANTHROPIC_MODEL_SPEC, compat: { supportsEagerToolInputStreaming: false }, - }; + }); const withoutTools = buildAnthropicClientOptions({ model: incompatibleModel, @@ -1417,10 +1428,10 @@ describe("Anthropic request fingerprint alignment", () => { }); it("forwards ANTHROPIC_CUSTOM_HEADERS to an enterprise gateway base URL without Foundry mode", async () => { - const gatewayModel: Model<"anthropic-messages"> = { - ...ANTHROPIC_MODEL, + const gatewayModel: Model<"anthropic-messages"> = buildModel({ + ...ANTHROPIC_MODEL_SPEC, baseUrl: "https://gateway.example.com", - }; + }); await withEnv( { CLAUDE_CODE_USE_FOUNDRY: undefined, @@ -1604,7 +1615,7 @@ describe("Anthropic request fingerprint alignment", () => { it("drops temperature and sampling params for Opus 4.7 without enabled thinking", async () => { const payload = (await captureAnthropicPayload( - { ...ANTHROPIC_MODEL, id: "claude-opus-4-7", name: "Claude Opus 4.7" }, + buildModel({ ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-7", name: "Claude Opus 4.7" }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1630,11 +1641,11 @@ describe("Anthropic request fingerprint alignment", () => { it("drops sampling params for Claude Fable/Mythos 5 without enabled thinking", async () => { for (const id of ["claude-fable-5", "claude-mythos-5"] as const) { const payload = (await captureAnthropicPayload( - { - ...ANTHROPIC_MODEL, + buildModel({ + ...ANTHROPIC_MODEL_SPEC, id, name: id === "claude-fable-5" ? "Claude Fable 5" : "Claude Mythos 5", - }, + }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1660,8 +1671,8 @@ describe("Anthropic request fingerprint alignment", () => { it("drops sampling params and keeps summarized adaptive thinking for OAuth Opus 4.7+", async () => { const payload = (await captureAnthropicPayload( - { - ...ANTHROPIC_MODEL, + buildModel({ + ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-7", name: "Claude Opus 4.7", thinking: { @@ -1669,7 +1680,7 @@ describe("Anthropic request fingerprint alignment", () => { minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - }, + }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1700,8 +1711,8 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.output_config).toEqual({ effort: "xhigh" }); const maxPayload = (await captureAnthropicPayload( - { - ...ANTHROPIC_MODEL, + buildModel({ + ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-7", name: "Claude Opus 4.7", thinking: { @@ -1709,7 +1720,7 @@ describe("Anthropic request fingerprint alignment", () => { minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - }, + }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1728,8 +1739,8 @@ describe("Anthropic request fingerprint alignment", () => { it("keeps summarized adaptive thinking by default for API-key Opus 4.7+ requests", async () => { const payload = (await captureAnthropicPayload( - { - ...ANTHROPIC_MODEL, + buildModel({ + ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-7", name: "Claude Opus 4.7", thinking: { @@ -1737,7 +1748,7 @@ describe("Anthropic request fingerprint alignment", () => { minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - }, + }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }], @@ -1760,8 +1771,8 @@ describe("Anthropic request fingerprint alignment", () => { it("sends task budgets through Anthropic output_config without dropping adaptive effort", async () => { const payload = (await captureAnthropicPayload( - { - ...ANTHROPIC_MODEL, + buildModel({ + ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-7", name: "Claude Opus 4.7", thinking: { @@ -1769,7 +1780,7 @@ describe("Anthropic request fingerprint alignment", () => { minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - }, + }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Review this repo", timestamp: Date.now() }], @@ -1794,8 +1805,8 @@ describe("Anthropic request fingerprint alignment", () => { it("preserves task budget when forced tool choice disables thinking", async () => { const payload = (await captureAnthropicPayload( - { - ...ANTHROPIC_MODEL, + buildModel({ + ...ANTHROPIC_MODEL_SPEC, id: "claude-opus-4-7", name: "Claude Opus 4.7", thinking: { @@ -1803,7 +1814,7 @@ describe("Anthropic request fingerprint alignment", () => { minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - }, + }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Use the tool", timestamp: Date.now() }], @@ -1838,8 +1849,8 @@ describe("Anthropic request fingerprint alignment", () => { it("downgrades forced tool choice for Claude Fable/Mythos without deleting adaptive thinking", async () => { for (const id of ["claude-fable-5", "claude-mythos-5"] as const) { const payload = (await captureAnthropicPayload( - { - ...ANTHROPIC_MODEL, + buildModel({ + ...ANTHROPIC_MODEL_SPEC, id, name: id === "claude-fable-5" ? "Claude Fable 5" : "Claude Mythos 5", contextWindow: 1_000_000, @@ -1849,7 +1860,7 @@ describe("Anthropic request fingerprint alignment", () => { minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - }, + }), { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Use the tool", timestamp: Date.now() }], diff --git a/packages/ai/test/anthropic-fable-request-shaping.test.ts b/packages/ai/test/anthropic-fable-request-shaping.test.ts index b844303b3..fa53096d9 100644 --- a/packages/ai/test/anthropic-fable-request-shaping.test.ts +++ b/packages/ai/test/anthropic-fable-request-shaping.test.ts @@ -1,10 +1,11 @@ import { describe, expect, it } from "bun:test"; import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; -import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import type { Context, Model, ModelSpec } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; function makeAnthropicModel(id: string): Model<"anthropic-messages"> { - return { + return buildModel({ id, name: id, api: "anthropic-messages", @@ -15,15 +16,17 @@ function makeAnthropicModel(id: string): Model<"anthropic-messages"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 128_000, - }; + }); } /** Adaptive-thinking model (Opus 4.6+, Sonnet 4.6+, Fable/Mythos 5). */ function adaptiveModel(id: string): Model<"anthropic-messages"> { - return { - ...makeAnthropicModel(id), + const base = makeAnthropicModel(id); + return buildModel({ + ...base, thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh }, - }; + compat: base.compatConfig, + } as ModelSpec<"anthropic-messages">); } const CONTEXT: Context = { diff --git a/packages/ai/test/anthropic-fast-mode.test.ts b/packages/ai/test/anthropic-fast-mode.test.ts index 24df5a433..e33f8dabe 100644 --- a/packages/ai/test/anthropic-fast-mode.test.ts +++ b/packages/ai/test/anthropic-fast-mode.test.ts @@ -5,9 +5,10 @@ import { streamAnthropic, } from "@oh-my-pi/pi-ai/providers/anthropic"; import type { Context, Model, ProviderSessionState, ServiceTier } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function makeAnthropicModel(id: string): Model<"anthropic-messages"> { - return { + return buildModel({ id, name: id, api: "anthropic-messages", @@ -18,7 +19,7 @@ function makeAnthropicModel(id: string): Model<"anthropic-messages"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, - }; + }); } const CONTEXT: Context = { diff --git a/packages/ai/test/anthropic-many-image-resize.test.ts b/packages/ai/test/anthropic-many-image-resize.test.ts index 493f34464..153c32278 100644 --- a/packages/ai/test/anthropic-many-image-resize.test.ts +++ b/packages/ai/test/anthropic-many-image-resize.test.ts @@ -1,11 +1,12 @@ import { describe, expect, it } from "bun:test"; import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import type { AssistantMessage, Context, ImageContent, Model, TextContent, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; const RED_1X1_PNG_BASE64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC"; -const model: Model<"anthropic-messages"> = { +const model: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -16,7 +17,7 @@ const model: Model<"anthropic-messages"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); const emptyUsage: Usage = { input: 0, diff --git a/packages/ai/test/anthropic-mid-conversation-system.test.ts b/packages/ai/test/anthropic-mid-conversation-system.test.ts index 3358a4086..9dd44133f 100644 --- a/packages/ai/test/anthropic-mid-conversation-system.test.ts +++ b/packages/ai/test/anthropic-mid-conversation-system.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "bun:test"; import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; -import type { AssistantMessage, DeveloperMessage, Message, Model, UserMessage } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessage, DeveloperMessage, Message, Model, ModelSpec, UserMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; /** * Claude Opus 4.8 and the Fable/Mythos 5 generation support mid-conversation @@ -11,8 +12,8 @@ import type { AssistantMessage, DeveloperMessage, Message, Model, UserMessage } * @see https://platform.claude.com/docs/en/build-with-claude/mid-conversation-system-messages */ -function makeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { - return { +function makeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { + return buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-opus-4-8-20260528", @@ -24,7 +25,7 @@ function makeModel(overrides: Partial> = {}): Model< contextWindow: 1000000, reasoning: true, ...overrides, - }; + } as ModelSpec<"anthropic-messages">); } function user(text: string): UserMessage { diff --git a/packages/ai/test/anthropic-prefill.test.ts b/packages/ai/test/anthropic-prefill.test.ts index b64110b4e..68a4bb767 100644 --- a/packages/ai/test/anthropic-prefill.test.ts +++ b/packages/ai/test/anthropic-prefill.test.ts @@ -1,7 +1,8 @@ import { describe, expect, it } from "bun:test"; import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; -import type { AssistantMessage, Model, UserMessage } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessage, Model, ModelSpec, UserMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; /** * Regression: some Anthropic-routed models reject "assistant prefill" requests @@ -9,7 +10,7 @@ import type { AssistantMessage, Model, UserMessage } from "@oh-my-pi/pi-ai/types * synthetic user message to keep the request valid. */ describe("Anthropic assistant-prefill fallback", () => { - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-3-5-sonnet-20241022", @@ -20,7 +21,7 @@ describe("Anthropic assistant-prefill fallback", () => { maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); it("appends a user Continue. message when the last turn is assistant", () => { const user: UserMessage = { @@ -121,7 +122,7 @@ describe("Anthropic assistant-prefill fallback", () => { }); it("preserves redacted thinking blocks in assistant replay payloads", () => { - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-3-5-sonnet-20241022", @@ -132,7 +133,7 @@ it("preserves redacted thinking blocks in assistant replay payloads", () => { maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); const user: UserMessage = { role: "user", content: "continue", @@ -171,7 +172,7 @@ it("preserves redacted thinking blocks in assistant replay payloads", () => { }); it("preserves latest Anthropic thinking blocks even when model id changes", () => { - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-3-5-sonnet-20241022", @@ -182,8 +183,12 @@ it("preserves latest Anthropic thinking blocks even when model id changes", () = maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; - const switchedModel: Model<"anthropic-messages"> = { ...model, id: "claude-opus-4-6-20251201" }; + }); + const switchedModel: Model<"anthropic-messages"> = buildModel({ + ...model, + id: "claude-opus-4-6-20251201", + compat: model.compatConfig, + } as ModelSpec<"anthropic-messages">); const assistant: AssistantMessage = { role: "assistant", content: [ @@ -223,7 +228,7 @@ it("preserves a completed thinking signature on an aborted turn interrupted duri // signature is whole and must survive transform. Interrupting during the visible text output // after thinking finished is the common case; dropping the valid signature and replaying it // empty makes Anthropic reject the request with 400 "Invalid `signature` in `thinking` block". - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-3-5-sonnet-20241022", @@ -234,7 +239,7 @@ it("preserves a completed thinking signature on an aborted turn interrupted duri maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); const assistant: AssistantMessage = { role: "assistant", content: [ diff --git a/packages/ai/test/anthropic-stream-envelope.test.ts b/packages/ai/test/anthropic-stream-envelope.test.ts index a7d185eb0..c597332ba 100644 --- a/packages/ai/test/anthropic-stream-envelope.test.ts +++ b/packages/ai/test/anthropic-stream-envelope.test.ts @@ -2,9 +2,10 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { scheduler } from "node:timers/promises"; import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import { AnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic-client"; -import type { AssistantMessageEvent, Context, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessageEvent, Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; -const model: Model<"anthropic-messages"> = { +const model: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -15,7 +16,7 @@ const model: Model<"anthropic-messages"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); const context: Context = { messages: [{ role: "user", content: "Say hi", timestamp: Date.now() }], @@ -839,7 +840,10 @@ describe("anthropic stream envelope handling", () => { await eagerStream.result(); const disabledStream = streamAnthropic( - { ...model, compat: { supportsEagerToolInputStreaming: false } }, + buildModel({ + ...model, + compat: { ...model.compatConfig, supportsEagerToolInputStreaming: false }, + } as ModelSpec<"anthropic-messages">), toolContext, { apiKey: "sk-ant-test" }, ); @@ -863,8 +867,15 @@ describe("anthropic stream envelope handling", () => { for (const testModel of [ model, - { ...model, compat: { supportsLongCacheRetention: false } }, - { ...model, baseUrl: "https://proxy.example.com/anthropic" }, + buildModel({ + ...model, + compat: { ...model.compatConfig, supportsLongCacheRetention: false }, + } as ModelSpec<"anthropic-messages">), + buildModel({ + ...model, + baseUrl: "https://proxy.example.com/anthropic", + compat: model.compatConfig, + } as ModelSpec<"anthropic-messages">), ]) { const stream = streamAnthropic(testModel, context, { apiKey: "sk-ant-test", diff --git a/packages/ai/test/anthropic-stream-timeout.test.ts b/packages/ai/test/anthropic-stream-timeout.test.ts index debac66e8..230118e35 100644 --- a/packages/ai/test/anthropic-stream-timeout.test.ts +++ b/packages/ai/test/anthropic-stream-timeout.test.ts @@ -2,9 +2,10 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import { AnthropicApiError, type AnthropicMessagesClientLike } from "@oh-my-pi/pi-ai/providers/anthropic-client"; import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { waitForDelayOrAbort } from "./helpers"; -const model: Model<"anthropic-messages"> = { +const model: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -15,7 +16,7 @@ const model: Model<"anthropic-messages"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); const context: Context = { messages: [{ role: "user", content: "Say hi", timestamp: Date.now() }], diff --git a/packages/ai/test/anthropic-thinking-immutability.test.ts b/packages/ai/test/anthropic-thinking-immutability.test.ts index 9854d6c43..1d07df7a4 100644 --- a/packages/ai/test/anthropic-thinking-immutability.test.ts +++ b/packages/ai/test/anthropic-thinking-immutability.test.ts @@ -1,8 +1,9 @@ import { describe, expect, it } from "bun:test"; import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; import type { AssistantMessage, Model, UserMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; -const model: Model<"anthropic-messages"> = { +const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-sonnet-4-6", @@ -13,7 +14,7 @@ const model: Model<"anthropic-messages"> = { maxTokens: 8_192, contextWindow: 200_000, reasoning: true, -}; +}); describe("Anthropic thinking replay immutability", () => { it("preserves signed-thinking blocks while normalizing non-thinking content", () => { diff --git a/packages/ai/test/anthropic-thinking-only-length-truncated.test.ts b/packages/ai/test/anthropic-thinking-only-length-truncated.test.ts index af7ca7557..d8be7f028 100644 --- a/packages/ai/test/anthropic-thinking-only-length-truncated.test.ts +++ b/packages/ai/test/anthropic-thinking-only-length-truncated.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; import type { AssistantMessage, Message, Model, UserMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; /** * Regression test for: "messages.X.content.Y: `thinking` or `redacted_thinking` blocks in @@ -19,7 +20,7 @@ import type { AssistantMessage, Message, Model, UserMessage } from "@oh-my-pi/pi * keeps proper `user` / `assistant` alternation regardless of which provider is sending it. */ describe("transformMessages drops thinking-only assistant turns", () => { - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-opus-4-7", @@ -30,7 +31,7 @@ describe("transformMessages drops thinking-only assistant turns", () => { maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); const makeThinkingOnlyAssistant = ( thinking: string, diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts index 4284b2e9f..6a0c37e2d 100644 --- a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -1,6 +1,14 @@ import { describe, expect, it } from "bun:test"; import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; -import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; +import type { + AssistantMessage, + Message, + Model, + ModelSpec, + ToolResultMessage, + UserMessage, +} from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; /** * Regression: Anthropic-compatible reasoning endpoints often emit `thinking` @@ -13,8 +21,8 @@ import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } * Official Anthropic remains conservative: unsigned thinking is demoted to text * there because the first-party API enforces signature-based integrity. */ -function makeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { - return { +function makeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { + return buildModel({ api: "anthropic-messages", provider: "custom-anthropic", id: "reasoning-model", @@ -26,7 +34,7 @@ function makeModel(overrides: Partial> = {}): Model< contextWindow: 200_000, reasoning: true, ...overrides, - }; + } as ModelSpec<"anthropic-messages">); } function makeUser(text = "continue"): UserMessage { @@ -165,7 +173,7 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { // dispatch falls back to https://api.anthropic.com. Same-id custom // overrides that only tweak model metadata (no baseUrl override) must // not regress to native-thinking replay against the first-party API. - const model = { ...makeModel(), provider: "anthropic", baseUrl: "" }; + const model = makeModel({ provider: "anthropic", baseUrl: "" }); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); expect(blocks[0]?.type).toBe("text"); expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); diff --git a/packages/ai/test/apply-patch-freeform.test.ts b/packages/ai/test/apply-patch-freeform.test.ts index ad19ad7e4..a3c2b4a53 100644 --- a/packages/ai/test/apply-patch-freeform.test.ts +++ b/packages/ai/test/apply-patch-freeform.test.ts @@ -13,7 +13,8 @@ import { convertResponsesAssistantMessage, processResponsesStream, } from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; -import type { AssistantMessage, Model, Tool, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessage, Model, ModelSpec, Tool, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { ResponseStreamEvent } from "openai/resources/responses/responses"; import * as z from "zod/v4"; @@ -27,8 +28,8 @@ const GRAMMAR = [ ].join("\n"); const COMPACT_GRAMMAR = 'start: "*** Begin Patch" LF\nPATH: /https?:\\/\\/[^\\n]+/\nLITERAL: "//"'; -function makeModel(overrides: Partial> = {}): Model<"openai-responses"> { - return { +function makeModel(overrides: Partial> = {}): Model<"openai-responses"> { + return buildModel({ id: "gpt-5", name: "GPT-5", api: "openai-responses", @@ -40,11 +41,11 @@ function makeModel(overrides: Partial> = {}): Model<"o contextWindow: 400000, maxTokens: 128000, ...overrides, - }; + } as ModelSpec<"openai-responses">); } -function makeCodexModel(overrides: Partial> = {}): Model<"openai-codex-responses"> { - return { +function makeCodexModel(overrides: Partial> = {}): Model<"openai-codex-responses"> { + return buildModel({ id: "gpt-5", name: "GPT-5", api: "openai-codex-responses", @@ -56,7 +57,7 @@ function makeCodexModel(overrides: Partial> = {} contextWindow: 272000, maxTokens: 128000, ...overrides, - }; + } as ModelSpec<"openai-codex-responses">); } const editTool: Tool = { diff --git a/packages/ai/test/azure-openai-responses-stream.test.ts b/packages/ai/test/azure-openai-responses-stream.test.ts index 7d463bc9f..ef9cb87c7 100644 --- a/packages/ai/test/azure-openai-responses-stream.test.ts +++ b/packages/ai/test/azure-openai-responses-stream.test.ts @@ -3,9 +3,10 @@ import { type AzureOpenAIResponsesOptions, streamAzureOpenAIResponses, } from "@oh-my-pi/pi-ai/providers/azure-openai-responses"; -import type { Context, FetchImpl, Model, Tool } from "@oh-my-pi/pi-ai/types"; +import type { Context, FetchImpl, Model, ModelSpec, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; -const azureModel: Model<"azure-openai-responses"> = { +const azureModel: Model<"azure-openai-responses"> = buildModel({ id: "gpt-5-mini", name: "GPT-5 Mini", api: "azure-openai-responses", @@ -16,7 +17,7 @@ const azureModel: Model<"azure-openai-responses"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, -}; +}); function createAbortedSignal(): AbortSignal { const controller = new AbortController(); @@ -95,10 +96,11 @@ describe("azure openai responses streaming", () => { }); it("uses developer role for Azure Responses reasoning model system prompts", async () => { - const reasoningModel: Model<"azure-openai-responses"> = { + const reasoningModel: Model<"azure-openai-responses"> = buildModel({ ...azureModel, reasoning: true, - }; + compat: azureModel.compatConfig, + } as ModelSpec<"azure-openai-responses">); const payload = await captureAzurePayload( { systemPrompt: ["Reasoning instruction", "Second instruction"], diff --git a/packages/ai/test/context-overflow.test.ts b/packages/ai/test/context-overflow.test.ts index 7aad9b774..013488a8d 100644 --- a/packages/ai/test/context-overflow.test.ts +++ b/packages/ai/test/context-overflow.test.ts @@ -17,6 +17,7 @@ import { execSync, spawn } from "node:child_process"; import { complete } from "@oh-my-pi/pi-ai/stream"; import type { AssistantMessage, Context, Model, Usage } from "@oh-my-pi/pi-ai/types"; import { isContextOverflow } from "@oh-my-pi/pi-ai/utils/overflow"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { $which } from "@oh-my-pi/pi-utils"; import { e2eApiKey, resolveApiKey } from "./oauth"; @@ -593,7 +594,7 @@ describe("Context overflow error handling", () => { setTimeout(checkServer, 1000); }); - model = { + model = buildModel({ id: "gpt-oss:20b", api: "openai-completions", provider: "ollama", @@ -604,7 +605,7 @@ describe("Context overflow error handling", () => { maxTokens: 16000, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, name: "Ollama GPT-OSS 20B", - }; + }); }, 60000); afterAll(() => { @@ -640,7 +641,7 @@ describe("Context overflow error handling", () => { describe.skipIf(lmStudioModel === undefined)("LM Studio (local)", () => { it("should detect overflow via isContextOverflow", async () => { if (!lmStudioModel) return; - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ id: lmStudioModel.id, api: "openai-completions", provider: "lm-studio", @@ -651,7 +652,7 @@ describe("Context overflow error handling", () => { maxTokens: 2048, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, name: lmStudioModel.name, - }; + }); const result = await testContextOverflow(model, Bun.env.LM_STUDIO_API_KEY || "lm-studio"); logResult(result); @@ -676,7 +677,7 @@ describe("Context overflow error handling", () => { describe.skipIf(!llamaCppRunning)("llama.cpp (local)", () => { it("should detect overflow via isContextOverflow", async () => { // Using small context (4096) to match server --ctx-size setting - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ id: "local-model", api: "openai-completions", provider: "llama.cpp", @@ -687,7 +688,7 @@ describe("Context overflow error handling", () => { maxTokens: 2048, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, name: "llama.cpp Local Model", - }; + }); const result = await testContextOverflow(model, "llama.cpp"); logResult(result); diff --git a/packages/ai/test/cursor-exec-handlers.test.ts b/packages/ai/test/cursor-exec-handlers.test.ts index 3735d811d..e57d409c9 100644 --- a/packages/ai/test/cursor-exec-handlers.test.ts +++ b/packages/ai/test/cursor-exec-handlers.test.ts @@ -6,9 +6,10 @@ import { streamCursor, } from "@oh-my-pi/pi-ai/providers/cursor"; import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { AgentRunRequest } from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb"; -const cursorModel: Model<"cursor-agent"> = { +const cursorModel: Model<"cursor-agent"> = buildModel({ id: "cursor-composer-2.5", name: "Cursor Composer 2.5", api: "cursor-agent", @@ -19,7 +20,7 @@ const cursorModel: Model<"cursor-agent"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1, maxTokens: 1, -}; +}); function captureCursorPayload(context: Context): Promise { const { promise, resolve, reject } = Promise.withResolvers(); diff --git a/packages/ai/test/deepseek-reasoning-content.test.ts b/packages/ai/test/deepseek-reasoning-content.test.ts index bbee3e754..9ebb77609 100644 --- a/packages/ai/test/deepseek-reasoning-content.test.ts +++ b/packages/ai/test/deepseek-reasoning-content.test.ts @@ -1,15 +1,18 @@ import { describe, expect, it } from "bun:test"; -import { convertMessages, detectCompat } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { AssistantMessage, Model, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types"; +import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import type { AssistantMessage, Model, ModelSpec, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; -function deepseekModel(overrides: Partial>): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), +function deepseekModel(overrides: Partial>): Model<"openai-completions"> { + const base = getBundledModel("openai", "gpt-4o-mini"); + return buildModel({ + ...base, api: "openai-completions", reasoning: true, + compat: base.compatConfig, ...overrides, - }; + } as ModelSpec<"openai-completions">); } function assistantToolCall( @@ -48,13 +51,11 @@ describe("DeepSeek reasoning_content tool-call replay", () => { // ---------------------------------------------------------------- describe("reasoningEffortMap (Fix 1)", () => { it("maps unsupported lower DeepSeek efforts to high on opencode-go", () => { - const compat = detectCompat( - deepseekModel({ - provider: "opencode-go", - baseUrl: "https://opencode.ai/zen/go/v1", - id: "deepseek-v4-flash", - }), - ); + const compat = deepseekModel({ + provider: "opencode-go", + baseUrl: "https://opencode.ai/zen/go/v1", + id: "deepseek-v4-flash", + }).compat; expect(compat.reasoningEffortMap).toMatchObject({ minimal: "high", low: "high", @@ -65,13 +66,11 @@ describe("DeepSeek reasoning_content tool-call replay", () => { }); it("maps unsupported lower DeepSeek efforts to high on NVIDIA", () => { - const compat = detectCompat( - deepseekModel({ - provider: "nvidia", - baseUrl: "https://integrate.api.nvidia.com/v1", - id: "deepseek-ai/deepseek-v4-flash", - }), - ); + const compat = deepseekModel({ + provider: "nvidia", + baseUrl: "https://integrate.api.nvidia.com/v1", + id: "deepseek-ai/deepseek-v4-flash", + }).compat; expect(compat.reasoningEffortMap).toMatchObject({ minimal: "high", low: "high", @@ -82,13 +81,11 @@ describe("DeepSeek reasoning_content tool-call replay", () => { }); it("maps unsupported lower DeepSeek efforts to high on the official endpoint", () => { - const compat = detectCompat( - deepseekModel({ - provider: "deepseek", - baseUrl: "https://api.deepseek.com/v1", - id: "deepseek-v4-pro", - }), - ); + const compat = deepseekModel({ + provider: "deepseek", + baseUrl: "https://api.deepseek.com/v1", + id: "deepseek-v4-pro", + }).compat; expect(compat.reasoningEffortMap).toMatchObject({ minimal: "high", low: "high", @@ -99,14 +96,12 @@ describe("DeepSeek reasoning_content tool-call replay", () => { }); it("does NOT map xhigh for non-DeepSeek models", () => { - const compat = detectCompat( - deepseekModel({ - provider: "openai", - baseUrl: "https://api.openai.com/v1", - id: "gpt-4o-mini", - reasoning: false, - }), - ); + const compat = deepseekModel({ + provider: "openai", + baseUrl: "https://api.openai.com/v1", + id: "gpt-4o-mini", + reasoning: false, + }).compat; expect(compat.reasoningEffortMap.xhigh).toBeUndefined(); }); }); @@ -116,36 +111,34 @@ describe("DeepSeek reasoning_content tool-call replay", () => { // ---------------------------------------------------------------- describe("allowsSyntheticReasoningContentForToolCalls flag", () => { it("is false for DeepSeek-family reasoning models", () => { - const compat = detectCompat( - deepseekModel({ - provider: "deepseek", - baseUrl: "https://api.deepseek.com/v1", - id: "deepseek-v4-pro", - }), - ); + const compat = deepseekModel({ + provider: "deepseek", + baseUrl: "https://api.deepseek.com/v1", + id: "deepseek-v4-pro", + }).compat; expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(false); }); it("is false for DeepSeek-family on NVIDIA", () => { - const compat = detectCompat( - deepseekModel({ - provider: "nvidia", - baseUrl: "https://integrate.api.nvidia.com/v1", - id: "deepseek-ai/deepseek-v4-flash", - }), - ); + const compat = deepseekModel({ + provider: "nvidia", + baseUrl: "https://integrate.api.nvidia.com/v1", + id: "deepseek-ai/deepseek-v4-flash", + }).compat; expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(false); }); it("is true for non-DeepSeek reasoning models on OpenRouter", () => { - const compat = detectCompat({ - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const compat = buildModel({ + ...base, api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", id: "qwen/qwq-32b", reasoning: true, - }); + compat: base.compatConfig, + } as ModelSpec<"openai-completions">).compat; // Qwen is not isDeepseekFamily, so synthetic is allowed expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(true); }); @@ -161,7 +154,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - const compat = detectCompat(model); + const compat = model.compat; // Simulate a tool-call turn with an empty thinking block that has a valid // signature — this happens when reasoning text was lost but the signature // (field name) is preserved. @@ -207,7 +200,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - const compat = detectCompat(model); + const compat = model.compat; const msg: AssistantMessage = { role: "assistant", content: [ @@ -245,7 +238,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { it("normalizes OpenRouter reasoning deltas to DeepSeek reasoning_content on replay", () => { const model = getBundledModel("openrouter", "deepseek/deepseek-v4-pro") as Model<"openai-completions">; - const compat = detectCompat(model); + const compat = model.compat; expect(compat.requiresReasoningContentForToolCalls).toBe(true); expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(false); @@ -273,7 +266,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - const compat = detectCompat(model); + const compat = model.compat; // Simulate a thinking block with an opaque signature from another provider // (e.g. Anthropic encrypted signature, OpenAI Responses JSON item). // The code should NOT write to a property named after the opaque signature. @@ -322,7 +315,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - const compat = detectCompat(model); + const compat = model.compat; // Empty-text thinking block with opaque signature — Tier 1 should reject the // opaque signature, nonEmptyThinkingBlocks won't include it, and the openai path // won't set anything. Tier 2 should then emit empty reasoning_content. @@ -374,7 +367,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - const compat = detectCompat(model); + const compat = model.compat; // Tool-call turn with NO thinking blocks at all — matches the actual // observed 400 error pattern where proxy stripped reasoning_content. const msg = assistantToolCall(model, [ @@ -396,7 +389,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { it("sets reasoning_content to empty string for OpenCode Zen big-pickle tool-call turns", () => { const model = getBundledModel("opencode-zen", "big-pickle") as Model<"openai-completions">; - const compat = detectCompat(model); + const compat = model.compat; expect(compat.requiresReasoningContentForToolCalls).toBe(true); expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(false); @@ -421,7 +414,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://integrate.api.nvidia.com/v1", id: "deepseek-ai/deepseek-v4-flash", }); - const compat = detectCompat(model); + const compat = model.compat; const msg = assistantToolCall(model, [ { type: "toolCall", @@ -449,7 +442,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://api.deepseek.com/v1", id: "deepseek-v4-pro", }); - const compat = detectCompat(model); + const compat = model.compat; // Plain text assistant response — no tool calls, no thinking blocks. // This is the exact pattern from the observed 400 error. const msg: AssistantMessage = { @@ -484,7 +477,7 @@ describe("DeepSeek reasoning_content tool-call replay", () => { baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", }); - const compat = detectCompat(model); + const compat = model.compat; const msg: AssistantMessage = { role: "assistant", content: [ @@ -517,15 +510,17 @@ describe("DeepSeek reasoning_content tool-call replay", () => { }); it("does NOT inject reasoning_content on non-tool-call turn for non-DeepSeek providers", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const model: Model<"openai-completions"> = buildModel({ + ...base, api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", id: "qwen/qwq-32b", reasoning: true, - }; - const compat = detectCompat(model); + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); + const compat = model.compat; const msg: AssistantMessage = { role: "assistant", content: [{ type: "text", text: "Plain answer." }], @@ -556,15 +551,17 @@ describe("DeepSeek reasoning_content tool-call replay", () => { // ---------------------------------------------------------------- describe("synthetic placeholder for non-DeepSeek providers (Tier 3)", () => { it('still uses "." placeholder for Kimi models that accept it', () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const model: Model<"openai-completions"> = buildModel({ + ...base, api: "openai-completions", provider: "moonshot", baseUrl: "https://api.moonshot.ai/v1", id: "kimi-k2.5", reasoning: true, - }; - const compat = detectCompat(model); + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); + const compat = model.compat; expect(compat.requiresReasoningContentForToolCalls).toBe(true); expect(compat.allowsSyntheticReasoningContentForToolCalls).toBe(true); const msg = assistantToolCall(model, [ diff --git a/packages/ai/test/duplicate-tool-results.test.ts b/packages/ai/test/duplicate-tool-results.test.ts index 22a70f234..584d6d6b5 100644 --- a/packages/ai/test/duplicate-tool-results.test.ts +++ b/packages/ai/test/duplicate-tool-results.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { convertMessages, detectCompat } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; import type { Api, @@ -12,6 +12,7 @@ import type { ToolResultMessage, UserMessage, } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { ChatCompletionAssistantMessageParam, ChatCompletionMessageParam, @@ -26,7 +27,7 @@ import type { * transformMessages should NOT add duplicate synthetic tool results. */ describe("Duplicate Tool Results Regression", () => { - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-3-5-sonnet-20241022", @@ -37,7 +38,7 @@ describe("Duplicate Tool Results Regression", () => { maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); const makeEvalAssistantMessage = (id: string, timestamp: number): AssistantMessage => ({ role: "assistant", @@ -516,7 +517,7 @@ describe("Duplicate Tool Results Regression", () => { expectedDuplicateId: string; }> = [ { - model: { + model: buildModel({ api: "openai-completions", provider: "openai", id: "gpt-4o-mini", @@ -527,12 +528,12 @@ describe("Duplicate Tool Results Regression", () => { maxTokens: 8192, contextWindow: 128000, reasoning: false, - }, + }), duplicateId: `call_${"a".repeat(35)}`, expectedDuplicateId: `${`call_${"a".repeat(35)}`.slice(0, 35)}_dup1`, }, { - model: { + model: buildModel({ api: "openai-completions", provider: "mistral", id: "mistral-large-latest", @@ -543,7 +544,7 @@ describe("Duplicate Tool Results Regression", () => { maxTokens: 8192, contextWindow: 128000, reasoning: false, - }, + }), duplicateId: "ABCDEF123", expectedDuplicateId: "ABCDEdup1", }, @@ -557,7 +558,7 @@ describe("Duplicate Tool Results Regression", () => { makeEvalToolResult(duplicateId, "second", 4), ]; const context: Context = { messages }; - const wireMessages = convertMessages(providerModel, context, detectCompat(providerModel)); + const wireMessages = convertMessages(providerModel, context, providerModel.compat); const assistantIds = assistantWireMessages(wireMessages).flatMap( message => message.tool_calls?.map(toolCall => toolCall.id) ?? [], ); @@ -581,7 +582,7 @@ describe("Duplicate Tool Results Regression", () => { * request is rejected. */ describe("Orphan Tool Result (handoff/compaction) Regression", () => { - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-3-5-sonnet-20241022", @@ -592,7 +593,7 @@ describe("Orphan Tool Result (handoff/compaction) Regression", () => { maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); const makeAssistantWithToolCall = ( id: string, @@ -1021,7 +1022,7 @@ describe("Orphan Tool Result (handoff/compaction) Regression", () => { { role: "user", content: "Resume work.", timestamp: 4 }, ]; - const openaiModel: Model<"openai-responses"> = { + const openaiModel: Model<"openai-responses"> = buildModel({ api: "openai-responses", provider: "openai", id: "gpt-5", @@ -1032,7 +1033,7 @@ describe("Orphan Tool Result (handoff/compaction) Regression", () => { maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); for (const m of [model, openaiModel] as Model[]) { const transformed = transformMessages(buildMessages(), m); @@ -1079,7 +1080,7 @@ describe("Orphan Tool Result (handoff/compaction) Regression", () => { * - Synthetic "aborted" tool results are injected */ describe("Codex-style Abort Handling", () => { - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ api: "anthropic-messages", provider: "anthropic", id: "claude-3-5-sonnet-20241022", @@ -1090,7 +1091,7 @@ describe("Codex-style Abort Handling", () => { maxTokens: 8192, contextWindow: 200000, reasoning: true, - }; + }); it("should preserve tool call structure in aborted messages", () => { const toolCallId = "toolu_preserve_test"; diff --git a/packages/ai/test/github-copilot-anthropic-auth.test.ts b/packages/ai/test/github-copilot-anthropic-auth.test.ts index 1b2751433..3e3a3fe90 100644 --- a/packages/ai/test/github-copilot-anthropic-auth.test.ts +++ b/packages/ai/test/github-copilot-anthropic-auth.test.ts @@ -2,6 +2,7 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { buildAnthropicClientOptions, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import type { Context, Model } from "@oh-my-pi/pi-ai/types"; import { buildAnthropicUrl } from "@oh-my-pi/pi-ai/utils/anthropic-auth"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { OPENCODE_HEADERS } from "@oh-my-pi/pi-catalog/wire/github-copilot"; afterEach(() => { @@ -9,7 +10,7 @@ afterEach(() => { }); function makeCopilotClaudeModel(): Model<"anthropic-messages"> { - return { + return buildModel({ id: "claude-sonnet-4", name: "Claude Sonnet 4", api: "anthropic-messages", @@ -21,10 +22,10 @@ function makeCopilotClaudeModel(): Model<"anthropic-messages"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 16000, - }; + }); } function makeOpenCodeGoQwen37Model(): Model<"anthropic-messages"> { - return { + return buildModel({ id: "qwen3.7-max", name: "Qwen3.7 Max", api: "anthropic-messages", @@ -35,7 +36,7 @@ function makeOpenCodeGoQwen37Model(): Model<"anthropic-messages"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 65_536, - }; + }); } const testContext: Context = { diff --git a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts index a46f323a1..3d068cafc 100644 --- a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts +++ b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts @@ -2,7 +2,7 @@ import { describe, expect, it } from "bun:test"; import { Effort, type FetchImpl } from "@oh-my-pi/pi-ai"; import { streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { Context, Model } from "@oh-my-pi/pi-ai/types"; -import { enrichModelThinking } from "@oh-my-pi/pi-catalog/model-thinking"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; interface GeminiCliThinkingConfig { thinkingLevel?: string; @@ -18,7 +18,7 @@ interface CapturedRequestBody { } function createModel(id: string): Model<"google-gemini-cli"> { - return enrichModelThinking({ + return buildModel({ id, name: id, api: "google-gemini-cli", diff --git a/packages/ai/test/google-gemini-cli-alignment.test.ts b/packages/ai/test/google-gemini-cli-alignment.test.ts index 38a0ca844..513475758 100644 --- a/packages/ai/test/google-gemini-cli-alignment.test.ts +++ b/packages/ai/test/google-gemini-cli-alignment.test.ts @@ -9,9 +9,11 @@ import { } from "@oh-my-pi/pi-ai/providers/google-gemini-cli"; import { getOAuthApiKey } from "@oh-my-pi/pi-ai/registry/oauth"; import type { Context, FetchImpl, Model, TJsonSchema } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; function createModel(provider: "google-gemini-cli" | "google-antigravity"): Model<"google-gemini-cli"> { - return { + return buildModel({ id: provider === "google-antigravity" ? "gemini-3-flash" : "gemini-2.5-flash", name: provider, api: "google-gemini-cli", @@ -27,7 +29,7 @@ function createModel(provider: "google-gemini-cli" | "google-antigravity"): Mode }, contextWindow: 200000, maxTokens: 8192, - }; + }); } function createContext(): Context { @@ -205,10 +207,10 @@ describe("Google Gemini CLI alignment", () => { // "gemini-3-pro-high" (hyphen) but the deployed model IDs use "gemini-3.1-pro-high" (dot), // so the injection was silently skipped and the Cloud Code Assist API returned HTTP 400. for (const modelId of ["gemini-3.1-pro-high", "gemini-3.1-pro-low"] as const) { - const model: Model<"google-gemini-cli"> = { + const model: Model<"google-gemini-cli"> = buildModel({ ...createModel("google-antigravity"), id: modelId, - }; + } as ModelSpec<"google-gemini-cli">); const context: Context = { systemPrompt: ["my instructions"], messages: [{ role: "user", content: "hi", timestamp: Date.now() }], @@ -231,12 +233,12 @@ describe("Google Gemini CLI alignment", () => { return new Response('{"error":{"message":"bad request"}}', { status: 400 }); }; - const model: Model<"google-gemini-cli"> = { + const model: Model<"google-gemini-cli"> = buildModel({ ...createModel("google-antigravity"), id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6", reasoning: true, - }; + } as ModelSpec<"google-gemini-cli">); const result = await streamGoogleGeminiCli(model, createContext(), { apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }), diff --git a/packages/ai/test/google-system-prompt.test.ts b/packages/ai/test/google-system-prompt.test.ts index a381d0e6c..3298fa389 100644 --- a/packages/ai/test/google-system-prompt.test.ts +++ b/packages/ai/test/google-system-prompt.test.ts @@ -1,8 +1,9 @@ import { describe, expect, it } from "bun:test"; import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; -const model: Model<"google-generative-ai"> = { +const model: Model<"google-generative-ai"> = buildModel({ id: "gemini-3-pro-preview", name: "Gemini 3 Pro Preview", api: "google-generative-ai", @@ -13,7 +14,7 @@ const model: Model<"google-generative-ai"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 32_000, -}; +}); async function captureGooglePayload( context: Context, diff --git a/packages/ai/test/google-tool-schema.test.ts b/packages/ai/test/google-tool-schema.test.ts index e82286879..735355c9b 100644 --- a/packages/ai/test/google-tool-schema.test.ts +++ b/packages/ai/test/google-tool-schema.test.ts @@ -2,9 +2,10 @@ import { describe, expect, it } from "bun:test"; import { convertTools } from "@oh-my-pi/pi-ai/providers/google-shared"; import type { Model, TJsonSchema, Tool } from "@oh-my-pi/pi-ai/types"; import { normalizeSchemaForCCA, normalizeSchemaForGoogle } from "@oh-my-pi/pi-ai/utils/schema"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function createModel(id: string): Model<"google-gemini-cli"> { - return { + return buildModel({ id, name: id, api: "google-gemini-cli", @@ -20,7 +21,7 @@ function createModel(id: string): Model<"google-gemini-cli"> { }, contextWindow: 200000, maxTokens: 8192, - }; + }); } describe("Cloud Code Assist Claude tool schema conversion", () => { diff --git a/packages/ai/test/helpers/index.ts b/packages/ai/test/helpers/index.ts index e0bb4d7e7..fe987fcf1 100644 --- a/packages/ai/test/helpers/index.ts +++ b/packages/ai/test/helpers/index.ts @@ -1,7 +1,7 @@ import * as os from "node:os"; import * as path from "node:path"; import type { Model } from "@oh-my-pi/pi-ai/types"; -import { enrichModelThinking } from "@oh-my-pi/pi-catalog/model-thinking"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { isEnoent } from "@oh-my-pi/pi-utils"; export async function withEnv( @@ -55,7 +55,7 @@ export async function waitForDelayOrAbort(delayMs: number, signal: AbortSignal | } export function createCodexModel(id: string): Model<"openai-codex-responses"> { - return enrichModelThinking({ + return buildModel({ id, name: id, api: "openai-codex-responses", diff --git a/packages/ai/test/issue-1207-repro.test.ts b/packages/ai/test/issue-1207-repro.test.ts index b983b41ae..ae0a85ad6 100644 --- a/packages/ai/test/issue-1207-repro.test.ts +++ b/packages/ai/test/issue-1207-repro.test.ts @@ -1,7 +1,7 @@ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types"; -import { detectOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import type { Context, Model, ModelSpec, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import * as z from "zod/v4"; @@ -36,7 +36,7 @@ async function capturePayload(model: Model<"openai-completions">): Promise { - return { + return buildModel({ ...getBundledModel("openai", "gpt-4o-mini"), api: "openai-completions", id: "deepseek-v4-flash", @@ -48,13 +48,13 @@ function customDeepseekFlash(): Model<"openai-completions"> { supportsReasoningEffort: true, reasoningEffortMap: { xhigh: "max" }, }, - }; + } as ModelSpec<"openai-completions">); } describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { it("detects the documented direct DeepSeek V4 compat shape", () => { const model = getBundledModel("deepseek", "deepseek-v4-flash") as Model<"openai-completions">; - const compat = detectOpenAICompat(model); + const compat = model.compat; expect(compat.supportsToolChoice).toBe(false); expect(compat.maxTokensField).toBe("max_tokens"); @@ -69,7 +69,7 @@ describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { }); it("merges partial user reasoning maps with DeepSeek defaults", () => { - const compat = resolveOpenAICompat(customDeepseekFlash()); + const compat = customDeepseekFlash().compat; expect(compat.supportsToolChoice).toBe(false); expect(compat.reasoningEffortMap).toMatchObject({ @@ -93,7 +93,7 @@ describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { it("does not mix Fireworks DeepSeek effort with the native thinking toggle", async () => { const model = getBundledModel("fireworks", "deepseek-v4-pro") as Model<"openai-completions">; - const compat = resolveOpenAICompat(model); + const compat = model.compat; const body = await capturePayload(model); expect(compat.extraBody).toBeUndefined(); @@ -106,7 +106,7 @@ describe("issue #1207 — DeepSeek V4 keeps reasoning with tools", () => { it("preserves OpenRouter reasoning when tool_choice auto is present", async () => { const model = getBundledModel("openrouter", "deepseek/deepseek-v4-flash") as Model<"openai-completions">; - const compat = detectOpenAICompat(model); + const compat = model.compat; const body = await capturePayload(model); expect(compat.disableReasoningOnToolChoice).toBe(false); diff --git a/packages/ai/test/issue-1227-repro.test.ts b/packages/ai/test/issue-1227-repro.test.ts index 70a7cd7f3..f17a7f281 100644 --- a/packages/ai/test/issue-1227-repro.test.ts +++ b/packages/ai/test/issue-1227-repro.test.ts @@ -17,7 +17,8 @@ */ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { AssistantMessage, Context, Model, Tool } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessage, Context, Model, ModelSpec, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import * as z from "zod/v4"; @@ -28,14 +29,16 @@ function abortedSignal(): AbortSignal { } function bedrockModel(): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + return buildModel({ + ...base, api: "openai-completions", id: "bedrock-claude-sonnet-4-6", name: "Bedrock Claude Sonnet 4.6 (LiteLLM)", provider: "litellm-bedrock", baseUrl: "https://example.test/v1", - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); } async function capturePayload( diff --git a/packages/ai/test/issue-1270-repro.test.ts b/packages/ai/test/issue-1270-repro.test.ts index 1169b6b0c..5d3582254 100644 --- a/packages/ai/test/issue-1270-repro.test.ts +++ b/packages/ai/test/issue-1270-repro.test.ts @@ -2,6 +2,7 @@ import { afterEach, describe, expect, it } from "bun:test"; import { __resetVertexTokenCache } from "@oh-my-pi/pi-ai/providers/google-auth"; import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex"; import type { Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; const OAUTH_TOKEN_URL = "https://oauth2.googleapis.com/token"; const METADATA_TOKEN_URL = "http://metadata.google.internal/computeMetadata/v1/instance/service-accounts/default/token"; @@ -10,7 +11,7 @@ const context = { messages: [{ role: "user" as const, content: "hello", timestamp: 0 }], }; -const model: Model<"google-vertex"> = { +const model: Model<"google-vertex"> = buildModel({ id: "gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview", api: "google-vertex", @@ -21,7 +22,7 @@ const model: Model<"google-vertex"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 65_536, -}; +}); describe("issue #1270: Vertex AI global endpoint", () => { const originalApiKey = Bun.env.GOOGLE_CLOUD_API_KEY; diff --git a/packages/ai/test/issue-1373-repro.test.ts b/packages/ai/test/issue-1373-repro.test.ts index 213498421..9a9d2d2c6 100644 --- a/packages/ai/test/issue-1373-repro.test.ts +++ b/packages/ai/test/issue-1373-repro.test.ts @@ -1,6 +1,7 @@ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; import { streamBedrock } from "@oh-my-pi/pi-ai/providers/amazon-bedrock"; import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; const originalSkipAuth = process.env.AWS_BEDROCK_SKIP_AUTH; @@ -15,7 +16,7 @@ afterAll(() => { }); function adaptiveModel(id: string): Model<"bedrock-converse-stream"> { - return { + return buildModel({ id, name: id, api: "bedrock-converse-stream", @@ -27,11 +28,11 @@ function adaptiveModel(id: string): Model<"bedrock-converse-stream"> { contextWindow: 1_000_000, maxTokens: 128_000, thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh }, - }; + }); } function budgetModel(id: string): Model<"bedrock-converse-stream"> { - return { + return buildModel({ id, name: id, api: "bedrock-converse-stream", @@ -43,7 +44,7 @@ function budgetModel(id: string): Model<"bedrock-converse-stream"> { contextWindow: 200_000, maxTokens: 64_000, thinking: { mode: "budget", minLevel: Effort.Minimal, maxLevel: Effort.High }, - }; + }); } const baseContext: Context = { diff --git a/packages/ai/test/issue-1399-repro.test.ts b/packages/ai/test/issue-1399-repro.test.ts index fd6eb4115..c2544156a 100644 --- a/packages/ai/test/issue-1399-repro.test.ts +++ b/packages/ai/test/issue-1399-repro.test.ts @@ -5,8 +5,9 @@ import * as path from "node:path"; import { streamBedrock } from "@oh-my-pi/pi-ai/providers/amazon-bedrock"; import { clearAwsCredentialCache } from "@oh-my-pi/pi-ai/providers/aws-credentials"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; -const model: Model<"bedrock-converse-stream"> = { +const model: Model<"bedrock-converse-stream"> = buildModel({ id: "zai.glm-5", name: "GLM-5", api: "bedrock-converse-stream", @@ -17,7 +18,7 @@ const model: Model<"bedrock-converse-stream"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 131_072, maxTokens: 16_384, -}; +}); const context: Context = { systemPrompt: [], diff --git a/packages/ai/test/issue-1417-repro.test.ts b/packages/ai/test/issue-1417-repro.test.ts index 2e4b6af9c..273165863 100644 --- a/packages/ai/test/issue-1417-repro.test.ts +++ b/packages/ai/test/issue-1417-repro.test.ts @@ -2,13 +2,13 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; -import type { Model } from "@oh-my-pi/pi-ai/types"; +import type { ModelSpec } from "@oh-my-pi/pi-ai/types"; import { readModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager"; const TTL_MS = 24 * 60 * 60 * 1000; -function syntheticModel(id: string): Model<"openai-completions"> { +function syntheticModel(id: string): ModelSpec<"openai-completions"> { return { id, name: id, diff --git a/packages/ai/test/issue-1838-repro.test.ts b/packages/ai/test/issue-1838-repro.test.ts index 2f2b1a000..cfc5b9d44 100644 --- a/packages/ai/test/issue-1838-repro.test.ts +++ b/packages/ai/test/issue-1838-repro.test.ts @@ -34,7 +34,8 @@ */ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { AssistantMessage, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessage, Context, FetchImpl, Model, ModelSpec } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; function abortedSignal(): AbortSignal { @@ -56,25 +57,29 @@ function mockFetch(): FetchImpl { } function moonshotKimiModel(id: string, reasoning = true): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + return buildModel({ + ...base, api: "openai-completions", provider: "moonshot", baseUrl: "https://api.moonshot.ai/v1", id, reasoning, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); } function openRouterKimiModel(id: string): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + return buildModel({ + ...base, api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", id, reasoning: true, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); } function basicContext(): Context { @@ -113,14 +118,16 @@ describe("issue #1838 — kimi-k2.6 preserves historical reasoning across tool c // Sanity: the Moonshot-native gate is provider+baseUrl driven, not id-only. // A made-up host with `kimi-k2.6` in the id but a non-Moonshot baseUrl must // never get the Moonshot-only `keep` parameter on the wire. - const customModel: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const customModel: Model<"openai-completions"> = buildModel({ + ...base, api: "openai-completions", provider: "openrouter", baseUrl: "https://example.com/v1", id: "kimi-k2.6", reasoning: true, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); const payload = (await capturePayload(customModel, { reasoning: "high" })) as CompletionBody; expect(payload.thinking).toBeUndefined(); }); @@ -184,14 +191,16 @@ describe("issue #1838 — kimi-k2.6 preserves historical reasoning across tool c // Fireworks publishes Kimi K2.6 under the `accounts/fireworks/routers/` // namespace. The `keep` flag is Moonshot-specific, so a Fireworks-hosted // K2.6 (which never speaks the Moonshot wire) must not see it. - const fireworksModel: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const fireworksModel: Model<"openai-completions"> = buildModel({ + ...base, api: "openai-completions", provider: "fireworks", baseUrl: "https://api.fireworks.ai/inference/v1", id: "accounts/fireworks/routers/kimi-k2.6", reasoning: true, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); const payload = (await capturePayload(fireworksModel, { reasoning: "high" })) as CompletionBody; // Fireworks → reasoning_effort path; thinking object never set. expect(payload.thinking).toBeUndefined(); diff --git a/packages/ai/test/issue-2123-repro.test.ts b/packages/ai/test/issue-2123-repro.test.ts index fca94ca80..129488b2a 100644 --- a/packages/ai/test/issue-2123-repro.test.ts +++ b/packages/ai/test/issue-2123-repro.test.ts @@ -22,9 +22,10 @@ import { describe, expect, it } from "bun:test"; import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; -const OPUS_46_OAUTH: Model<"anthropic-messages"> = { +const OPUS_46_OAUTH: Model<"anthropic-messages"> = buildModel({ id: "claude-opus-4-6", name: "Claude Opus 4.6", api: "anthropic-messages", @@ -36,7 +37,7 @@ const OPUS_46_OAUTH: Model<"anthropic-messages"> = { contextWindow: 1_000_000, maxTokens: 128_000, thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh }, -}; +}); const todoTool: Tool = { name: "todo", diff --git a/packages/ai/test/issue-814-repro.test.ts b/packages/ai/test/issue-814-repro.test.ts index 17009bc48..35df594e3 100644 --- a/packages/ai/test/issue-814-repro.test.ts +++ b/packages/ai/test/issue-814-repro.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "bun:test"; import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; -import type { AssistantMessage, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessage, Model, ModelSpec, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; /** * Issue #814: Z.AI returns 500 @@ -14,7 +15,7 @@ import type { AssistantMessage, Model, ToolResultMessage, UserMessage } from "@o * endpoints must remain unchanged (no `id` field). */ -const baseModel: Omit, "provider" | "baseUrl"> = { +const baseModel: Omit, "provider" | "baseUrl"> = { api: "anthropic-messages", id: "glm-4.6", name: "GLM-4.6", @@ -25,19 +26,19 @@ const baseModel: Omit, "provider" | "baseUrl"> = { reasoning: false, }; -const zaiModel: Model<"anthropic-messages"> = { +const zaiModel: Model<"anthropic-messages"> = buildModel({ ...baseModel, provider: "zai", baseUrl: "https://api.z.ai/api/anthropic", -}; +}); -const anthropicModel: Model<"anthropic-messages"> = { +const anthropicModel: Model<"anthropic-messages"> = buildModel({ ...baseModel, id: "claude-3-5-sonnet-20241022", name: "Claude 3.5 Sonnet", provider: "anthropic", baseUrl: "https://api.anthropic.com", -}; +}); const user: UserMessage = { role: "user", diff --git a/packages/ai/test/issue-826-repro.test.ts b/packages/ai/test/issue-826-repro.test.ts index 607820b6f..ff73562e1 100644 --- a/packages/ai/test/issue-826-repro.test.ts +++ b/packages/ai/test/issue-826-repro.test.ts @@ -1,9 +1,10 @@ import { describe, expect, it } from "bun:test"; import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; -import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types"; +import type { Context, Model, ModelSpec, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; -const baseModel: Model<"anthropic-messages"> = { +const baseModel: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -15,7 +16,7 @@ const baseModel: Model<"anthropic-messages"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); const bashTool: Tool = { name: "bash", @@ -64,17 +65,19 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies", }); it("omits strict on tool defs when compat.disableStrictTools is set", async () => { - const params = await captureParams({ - ...baseModel, - compat: { disableStrictTools: true }, - }); + const params = await captureParams( + buildModel({ + ...baseModel, + compat: { ...baseModel.compatConfig, disableStrictTools: true }, + } as ModelSpec<"anthropic-messages">), + ); const bash = params.tools?.find(t => t.name === "bash"); expect(bash).toBeDefined(); expect(bash?.strict).toBeUndefined(); }); it("preserves adaptive thinking by default", async () => { - const adaptiveModel: Model<"anthropic-messages"> = { + const adaptiveModel: Model<"anthropic-messages"> = buildModel({ ...baseModel, id: "claude-opus-4-7", reasoning: true, @@ -83,7 +86,8 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies", minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - }; + compat: baseModel.compatConfig, + } as ModelSpec<"anthropic-messages">); const { promise, resolve } = Promise.withResolvers<{ thinking?: { type?: string } }>(); void streamAnthropic(adaptiveModel, baseContext, { apiKey: "sk-ant-api-test", @@ -100,7 +104,7 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies", }); it("maps adaptive thinking to enabled when compat.disableAdaptiveThinking is set", async () => { - const adaptiveModel: Model<"anthropic-messages"> = { + const adaptiveModel: Model<"anthropic-messages"> = buildModel({ ...baseModel, id: "claude-opus-4-7", reasoning: true, @@ -109,8 +113,8 @@ describe("issue #826: Anthropic strict-tools opt-out for Vertex-style proxies", minLevel: Effort.Minimal, maxLevel: Effort.XHigh, }, - compat: { disableAdaptiveThinking: true }, - }; + compat: { ...baseModel.compatConfig, disableAdaptiveThinking: true }, + } as ModelSpec<"anthropic-messages">); const { promise, resolve } = Promise.withResolvers<{ thinking?: { type?: string; budget_tokens?: number } }>(); void streamAnthropic(adaptiveModel, baseContext, { apiKey: "sk-ant-api-test", diff --git a/packages/ai/test/issue-827-repro.test.ts b/packages/ai/test/issue-827-repro.test.ts index 73cedc1fe..acb6888df 100644 --- a/packages/ai/test/issue-827-repro.test.ts +++ b/packages/ai/test/issue-827-repro.test.ts @@ -9,7 +9,8 @@ */ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types"; +import type { Context, Model, ModelSpec, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import * as z from "zod/v4"; @@ -31,27 +32,31 @@ function abortedSignal(): AbortSignal { } function kimiOpencodeGoModel(): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + return buildModel({ + ...base, api: "openai-completions", provider: "opencode-go", baseUrl: "https://opencode.ai/zen/v1", id: "kimi-k2.6", name: "Kimi K2.6", reasoning: true, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); } function kimiOpenRouterModel(): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + return buildModel({ + ...base, api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", id: "moonshotai/kimi-k2", name: "Kimi K2 (OpenRouter)", reasoning: true, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); } function captureBody( @@ -110,15 +115,17 @@ describe("issue #827 — kimi reasoning models drop reasoning under forced tool_ expect(body.reasoning_effort).toBeUndefined(); }); it("sends explicit thinking disabled for Moonshot Kimi K2.6 when a named tool is forced", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const model: Model<"openai-completions"> = buildModel({ + ...base, api: "openai-completions", provider: "moonshot", baseUrl: "https://api.moonshot.ai/v1", id: "kimi-k2.6", name: "Kimi K2.6", reasoning: false, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); const body = (await captureBody(model, { toolChoice: { type: "tool", name: "echo" }, })) as CompletionsBody; @@ -133,15 +140,17 @@ describe("issue #827 — kimi reasoning models drop reasoning under forced tool_ // LiteLLM / Vertex proxies often expose Claude through chat-completions; Anthropic // itself rejects reasoning + forced tool_choice (see anthropic.ts:disableThinkingIfToolChoiceForced), // so the same constraint must follow the model when it's reached through the OpenAI shape. - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const model: Model<"openai-completions"> = buildModel({ + ...base, api: "openai-completions", provider: "litellm", baseUrl: "http://localhost:4000/v1", id: "claude-sonnet-4-6", name: "Claude Sonnet 4.6 (LiteLLM)", reasoning: true, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); const body = (await captureBody(model, { reasoning: "high", @@ -153,12 +162,14 @@ describe("issue #827 — kimi reasoning models drop reasoning under forced tool_ }); it("does not strip reasoning on non-Kimi models even with forced tool_choice", async () => { // Non-kimi reasoning model — OpenAI itself accepts forced tool_choice with reasoning. - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const base = getBundledModel("openai", "gpt-4o-mini"); + const model: Model<"openai-completions"> = buildModel({ + ...base, api: "openai-completions", id: "gpt-5-mini", reasoning: true, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); const body = (await captureBody(model, { reasoning: "high", diff --git a/packages/ai/test/issue-883-repro.test.ts b/packages/ai/test/issue-883-repro.test.ts index 197c25b69..a1b5c1da5 100644 --- a/packages/ai/test/issue-883-repro.test.ts +++ b/packages/ai/test/issue-883-repro.test.ts @@ -1,15 +1,18 @@ import { describe, expect, it } from "bun:test"; -import { convertMessages, detectCompat } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types"; +import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import type { AssistantMessage, Model, ModelSpec } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; -function deepseekModel(overrides: Partial>): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), +function deepseekModel(overrides: Partial>): Model<"openai-completions"> { + const base = getBundledModel("openai", "gpt-4o-mini"); + return buildModel({ + ...base, api: "openai-completions", reasoning: true, + compat: base.compatConfig, ...overrides, - }; + } as ModelSpec<"openai-completions">); } function assistantWithToolCall(model: Model<"openai-completions">): AssistantMessage { @@ -42,24 +45,20 @@ function assistantWithToolCall(model: Model<"openai-completions">): AssistantMes describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay", () => { it("flags requiresReasoningContentForToolCalls for deepseek-v4-pro on the official endpoint", () => { - const compat = detectCompat( - deepseekModel({ - provider: "deepseek", - baseUrl: "https://api.deepseek.com/v1", - id: "deepseek-v4-pro", - }), - ); + const compat = deepseekModel({ + provider: "deepseek", + baseUrl: "https://api.deepseek.com/v1", + id: "deepseek-v4-pro", + }).compat; expect(compat.requiresReasoningContentForToolCalls).toBe(true); }); it("flags requiresReasoningContentForToolCalls for deepseek-v4 served by a non-deepseek host (e.g. Deepinfra)", () => { - const compat = detectCompat( - deepseekModel({ - provider: "deepinfra", - baseUrl: "https://api.deepinfra.com/v1/openai", - id: "deepseek-ai/DeepSeek-V4-Flash", - }), - ); + const compat = deepseekModel({ + provider: "deepinfra", + baseUrl: "https://api.deepinfra.com/v1/openai", + id: "deepseek-ai/DeepSeek-V4-Flash", + }).compat; expect(compat.requiresReasoningContentForToolCalls).toBe(true); }); @@ -69,7 +68,7 @@ describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay", baseUrl: "https://api.deepseek.com/v1", id: "deepseek-v4-pro", }); - const compat = detectCompat(model); + const compat = model.compat; const messages = convertMessages(model, { messages: [assistantWithToolCall(model)] }, compat); const assistant = messages.find(m => m.role === "assistant"); expect(assistant).toBeDefined(); @@ -85,7 +84,7 @@ describe("issue #883 / #810 — DeepSeek V4 reasoning_content tool-call replay", baseUrl: "https://api.deepinfra.com/v1/openai", id: "deepseek-ai/DeepSeek-V4-Pro", }); - const compat = detectCompat(model); + const compat = model.compat; // Assistant turn whose only content is a tool call (no text) - matches what the SDK // produces after a pure tool-use turn. content must end up "" (not null) because // DeepSeek rejects null content alongside reasoning_content. diff --git a/packages/ai/test/issue-912-repro.test.ts b/packages/ai/test/issue-912-repro.test.ts index 60507b796..3f62b7ccd 100644 --- a/packages/ai/test/issue-912-repro.test.ts +++ b/packages/ai/test/issue-912-repro.test.ts @@ -1,9 +1,10 @@ import { describe, expect, it } from "bun:test"; import { stream } from "@oh-my-pi/pi-ai/stream"; import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function makeCopilotResponsesModel(baseUrl: string): Model<"openai-responses"> { - return { + return buildModel({ id: "gpt-5-mini", name: "GPT-5 Mini", api: "openai-responses", @@ -15,7 +16,7 @@ function makeCopilotResponsesModel(baseUrl: string): Model<"openai-responses"> { contextWindow: 128000, maxTokens: 64000, headers: { "User-Agent": "opencode/1.3.15" }, - }; + }); } function makeContext(): Context { diff --git a/packages/ai/test/issue-967-vision-guard.test.ts b/packages/ai/test/issue-967-vision-guard.test.ts index 928740d4a..d105dee9c 100644 --- a/packages/ai/test/issue-967-vision-guard.test.ts +++ b/packages/ai/test/issue-967-vision-guard.test.ts @@ -8,8 +8,9 @@ import { convertResponsesInputContent, } from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard"; -import type { Api, AssistantMessage, Context, Model, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types"; -import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import type { Api, AssistantMessage, Context, Model, ModelSpec, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; const emptyUsage: Usage = { input: 0, @@ -25,6 +26,10 @@ const compat: ResolvedOpenAICompat = { supportsDeveloperRole: true, supportsMultipleSystemMessages: true, supportsReasoningEffort: true, + supportsReasoningParams: true, + alwaysSendMaxTokens: false, + isOpenRouterHost: false, + isVercelGatewayHost: false, reasoningEffortMap: {}, supportsUsageInStreaming: true, supportsToolChoice: true, @@ -48,7 +53,7 @@ const compat: ResolvedOpenAICompat = { }; function makeModel(api: TApi, provider: Model["provider"]): Model { - return { + return buildModel({ id: `${provider}-${api}-text-only`, name: `${provider} ${api}`, api, @@ -59,7 +64,7 @@ function makeModel(api: TApi, provider: Model["provider"]): Mo cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128_000, maxTokens: 8_192, - }; + } as ModelSpec); } function makeAssistant(api: Model["api"], provider: Model["provider"], modelId: string): AssistantMessage { diff --git a/packages/ai/test/issue-969-repro.test.ts b/packages/ai/test/issue-969-repro.test.ts index 5228b4625..be68361d8 100644 --- a/packages/ai/test/issue-969-repro.test.ts +++ b/packages/ai/test/issue-969-repro.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking"; @@ -17,7 +18,7 @@ function createSseResponse(events: unknown[]): Response { } function customOpenAICompatModel(): Model<"openai-completions"> { - return { + return buildModel({ id: "gpt-5.1", name: "GPT-5.1 proxy", api: "openai-completions", @@ -33,7 +34,7 @@ function customOpenAICompatModel(): Model<"openai-completions"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128_000, maxTokens: 16_384, - }; + }); } describe("issue #969 — custom thinking metadata must preserve explicit xhigh", () => { diff --git a/packages/ai/test/issue-976-repro.test.ts b/packages/ai/test/issue-976-repro.test.ts index 743358a06..1c185477b 100644 --- a/packages/ai/test/issue-976-repro.test.ts +++ b/packages/ai/test/issue-976-repro.test.ts @@ -1,9 +1,10 @@ import { describe, expect, it } from "bun:test"; import { buildRequest } from "@oh-my-pi/pi-ai/providers/google-gemini-cli"; import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function createModel(): Model<"google-gemini-cli"> { - return { + return buildModel({ id: "gemini-2.5-flash", name: "gemini", api: "google-gemini-cli", @@ -19,7 +20,7 @@ function createModel(): Model<"google-gemini-cli"> { }, contextWindow: 200000, maxTokens: 8192, - }; + }); } describe("issue #976 — legacy string systemPrompt", () => { diff --git a/packages/ai/test/model-cache.test.ts b/packages/ai/test/model-cache.test.ts index d63586799..cc27cab62 100644 --- a/packages/ai/test/model-cache.test.ts +++ b/packages/ai/test/model-cache.test.ts @@ -4,12 +4,13 @@ import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; import type { Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { readModelCache, writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; const TTL_MS = 24 * 60 * 60 * 1000; function createModel(id: string, name: string): Model<"openai-completions"> { - return { + return buildModel({ id, name, api: "openai-completions", @@ -25,7 +26,7 @@ function createModel(id: string, name: string): Model<"openai-completions"> { }, contextWindow: 4096, maxTokens: 1024, - }; + }); } describe("model cache migrations", () => { diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index f903bc3eb..a7b6f3cf8 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -5,8 +5,8 @@ import { prewarmOpenAICodexResponses, streamOpenAICodexResponses, } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; -import type { Context, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; -import { enrichModelThinking } from "@oh-my-pi/pi-catalog/model-thinking"; +import type { Context, FetchImpl, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getAgentDir, setAgentDir, TempDir } from "@oh-my-pi/pi-utils"; const originalAgentDir = getAgentDir(); @@ -53,7 +53,7 @@ function createCodexTestToken(accountId = "acc_test"): string { } function createCodexTestModel(baseUrl?: string): Model<"openai-codex-responses"> { - return { + return buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -65,7 +65,7 @@ function createCodexTestModel(baseUrl?: string): Model<"openai-codex-responses"> cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); } function createCodexTestContext(): Context { @@ -679,7 +679,7 @@ describe("openai-codex streaming", () => { return new Response("not found", { status: 404 }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -690,7 +690,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], @@ -741,7 +741,7 @@ describe("openai-codex streaming", () => { }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -752,7 +752,7 @@ describe("openai-codex streaming", () => { cost: { input: 1, output: 2, cacheRead: 0.5, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], @@ -811,7 +811,7 @@ describe("openai-codex streaming", () => { return new Response("not found", { status: 404 }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -822,7 +822,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], @@ -859,7 +859,7 @@ describe("openai-codex streaming", () => { async () => new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }), ); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -870,7 +870,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -916,7 +916,7 @@ describe("openai-codex streaming", () => { return new Response("not found", { status: 404 }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -927,7 +927,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], @@ -982,7 +982,7 @@ describe("openai-codex streaming", () => { return new Response("not found", { status: 404 }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -993,7 +993,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], @@ -1086,7 +1086,7 @@ describe("openai-codex streaming", () => { return new Response("not found", { status: 404 }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -1097,7 +1097,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], @@ -1224,7 +1224,7 @@ describe("openai-codex streaming", () => { return new Response("not found", { status: 404 }); }); - const model = enrichModelThinking({ + const model = buildModel({ id: "gpt-5.3-codex", name: "GPT-5.3 Codex", api: "openai-codex-responses", @@ -1315,7 +1315,7 @@ describe("openai-codex streaming", () => { return new Response("not found", { status: 404 }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -1326,7 +1326,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], @@ -1380,7 +1380,7 @@ describe("openai-codex streaming", () => { } global.WebSocket = FailingWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -1392,7 +1392,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -1449,7 +1449,7 @@ describe("openai-codex streaming", () => { global.WebSocket = FailingConnectWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -1461,7 +1461,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -1528,7 +1528,7 @@ describe("openai-codex streaming", () => { global.WebSocket = HandshakeWebSocket as unknown as typeof WebSocket; - const websocketModel: Model<"openai-codex-responses"> = { + const websocketModel: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -1540,11 +1540,12 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; - const sseModel: Model<"openai-codex-responses"> = { + }); + const sseModel: Model<"openai-codex-responses"> = buildModel({ ...websocketModel, preferWebsockets: false, - }; + compat: websocketModel.compatConfig, + } as ModelSpec<"openai-codex-responses">); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -1614,7 +1615,7 @@ describe("openai-codex streaming", () => { global.WebSocket = ServiceTierWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -1626,7 +1627,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -1679,7 +1680,7 @@ describe("openai-codex streaming", () => { } global.WebSocket = DeltaWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -1691,7 +1692,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const providerSessionState = new Map(); const firstContext: Context = { systemPrompt: ["You are a helpful assistant.", "Use concise answers."], @@ -1884,7 +1885,7 @@ describe("openai-codex streaming", () => { capturedBodies.push(JSON.parse(String(init?.body)) as Record); return new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -1895,7 +1896,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -1943,7 +1944,7 @@ describe("openai-codex streaming", () => { global.WebSocket = WebSocketV2HeaderProbe as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -1955,7 +1956,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -2001,7 +2002,7 @@ describe("openai-codex streaming", () => { global.WebSocket = IdleWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -2013,7 +2014,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -2669,7 +2670,7 @@ describe("openai-codex streaming", () => { global.WebSocket = FlakyCloseWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -2681,7 +2682,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -2743,7 +2744,7 @@ describe("openai-codex streaming", () => { global.WebSocket = UnavailableBeforeStreamWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -2755,7 +2756,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -2841,7 +2842,7 @@ describe("openai-codex streaming", () => { } global.WebSocket = AbortResetWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -2853,7 +2854,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const firstContext: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -2960,7 +2961,7 @@ describe("openai-codex streaming", () => { } global.WebSocket = ErrorResetWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -2972,7 +2973,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const firstContext: Context = { systemPrompt: ["You are a helpful assistant."], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -3058,7 +3059,7 @@ describe("openai-codex streaming", () => { } global.WebSocket = MalformedMessageWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -3070,7 +3071,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const result = await streamOpenAICodexResponses( model, { @@ -3138,7 +3139,7 @@ describe("openai-codex streaming", () => { } global.WebSocket = BufferedCloseWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -3150,7 +3151,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const result = await streamOpenAICodexResponses( model, { @@ -3224,7 +3225,7 @@ describe("openai-codex streaming", () => { global.WebSocket = DivergedAppendWebSocket as unknown as typeof WebSocket; - const websocketModel: Model<"openai-codex-responses"> = { + const websocketModel: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -3236,11 +3237,12 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; - const sseModel: Model<"openai-codex-responses"> = { + }); + const sseModel: Model<"openai-codex-responses"> = buildModel({ ...websocketModel, preferWebsockets: false, - }; + compat: websocketModel.compatConfig, + } as ModelSpec<"openai-codex-responses">); const firstContext: Context = { systemPrompt: ["Prompt A"], messages: [{ role: "user", content: "Say hello", timestamp: Date.now() }], @@ -3313,7 +3315,7 @@ describe("openai-codex streaming", () => { global.WebSocket = ReusableWebSocket as unknown as typeof WebSocket; - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "openai-codex-responses", @@ -3325,7 +3327,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 128000, - }; + }); const providerSessionState = new Map(); await prewarmOpenAICodexResponses(model, { @@ -3402,7 +3404,7 @@ describe("openai-codex streaming", () => { return new Response(sse, { status: 200, headers: responseHeaders }); }); - const model: Model<"openai-codex-responses"> = { + const model: Model<"openai-codex-responses"> = buildModel({ id: "gpt-5.1-codex", name: "GPT-5.1 Codex", api: "openai-codex-responses", @@ -3413,7 +3415,7 @@ describe("openai-codex streaming", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, - }; + }); const context: Context = { systemPrompt: ["You are a helpful assistant."], diff --git a/packages/ai/test/openai-completions-compat.test.ts b/packages/ai/test/openai-completions-compat.test.ts index 3427aaa1b..46fa4ed0e 100644 --- a/packages/ai/test/openai-completions-compat.test.ts +++ b/packages/ai/test/openai-completions-compat.test.ts @@ -2,7 +2,6 @@ import { describe, expect, it } from "bun:test"; import { applyOpenRouterRoutingVariant, convertMessages, - detectCompat, streamOpenAICompletions, } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { @@ -10,11 +9,22 @@ import type { Context, FetchImpl, Model, + ModelSpec, OpenAICompat, ToolResultMessage, } from "@oh-my-pi/pi-ai/types"; -import { type ResolvedOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; + +const gpt4oMiniSpec: ModelSpec<"openai-completions"> = (() => { + const { + compat: _resolved, + compatConfig, + ...rest + } = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">; + return { ...rest, compat: compatConfig }; +})(); function createAbortedSignal(): AbortSignal { const controller = new AbortController(); @@ -112,10 +122,10 @@ function getLastTextPart(content: unknown): object | undefined { describe("openai-completions compatibility", () => { it("serializes assistant text content as a plain string", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const compat = { supportsStore: true, supportsDeveloperRole: true, @@ -141,6 +151,10 @@ describe("openai-completions compatibility", () => { extraBody: {}, supportsStrictMode: true, toolStrictMode: "none", + supportsReasoningParams: true, + alwaysSendMaxTokens: false, + isOpenRouterHost: false, + isVercelGatewayHost: false, } satisfies ResolvedOpenAICompat; const assistantMessage: AssistantMessage = { role: "assistant", @@ -173,10 +187,10 @@ describe("openai-completions compatibility", () => { }); it("prepends thinking text to string assistant content when requiresThinkingAsText is set", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const assistantMessage: AssistantMessage = { role: "assistant", content: [ @@ -201,7 +215,7 @@ describe("openai-completions compatibility", () => { model, { messages: [assistantMessage] }, { - ...detectCompat(model), + ...model.compat, requiresThinkingAsText: true, }, ); @@ -215,10 +229,10 @@ describe("openai-completions compatibility", () => { }); it("emits thinking-only assistant content as a plain string when requiresThinkingAsText is set", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const assistantMessage: AssistantMessage = { role: "assistant", content: [{ type: "thinking", thinking: "only thoughts" }], @@ -240,7 +254,7 @@ describe("openai-completions compatibility", () => { model, { messages: [assistantMessage] }, { - ...detectCompat(model), + ...model.compat, requiresThinkingAsText: true, }, ); @@ -251,10 +265,10 @@ describe("openai-completions compatibility", () => { }); it("preserves multiple system prompts as leading system messages for chat completions", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const messages = convertMessages( model, @@ -262,7 +276,7 @@ describe("openai-completions compatibility", () => { systemPrompt: ["stable instructions", "cacheable policy"], messages: [{ role: "user", content: "hello", timestamp: Date.now() }], }, - detectCompat(model), + model.compat, ); expect(messages.slice(0, 3)).toEqual([ @@ -273,11 +287,11 @@ describe("openai-completions compatibility", () => { }); it("uses developer messages for reasoning chat models only when the target supports them", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", reasoning: true, - }; + } as ModelSpec<"openai-completions">); const supportedMessages = convertMessages( model, @@ -285,7 +299,7 @@ describe("openai-completions compatibility", () => { systemPrompt: ["stable instructions", "cacheable policy"], messages: [{ role: "user", content: "hello", timestamp: Date.now() }], }, - detectCompat(model), + model.compat, ); expect(supportedMessages.slice(0, 3)).toEqual([ @@ -300,7 +314,7 @@ describe("openai-completions compatibility", () => { systemPrompt: ["stable instructions", "cacheable policy"], messages: [{ role: "user", content: "hello", timestamp: Date.now() }], }, - { ...detectCompat(model), supportsDeveloperRole: false }, + { ...model.compat, supportsDeveloperRole: false }, ); expect(unsupportedMessages.slice(0, 3)).toEqual([ @@ -325,26 +339,26 @@ describe("openai-completions compatibility", () => { { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com", expected: false }, ]; for (const { provider, baseUrl, expected } of cases) { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: provider as Model["provider"], baseUrl, reasoning: true, - }; - expect(detectCompat(model).supportsDeveloperRole).toBe(expected); + } as ModelSpec<"openai-completions">); + expect(model.compat.supportsDeveloperRole).toBe(expected); } }); it("emits system role for reasoning models on Moonshot (kimi tokenization rejects developer)", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "moonshot", baseUrl: "https://api.moonshot.ai/v1", id: "kimi-k2.5", reasoning: true, - }; + } as ModelSpec<"openai-completions">); const messages = convertMessages( model, @@ -352,7 +366,7 @@ describe("openai-completions compatibility", () => { systemPrompt: ["you are a helpful assistant"], messages: [{ role: "user", content: "hi", timestamp: Date.now() }], }, - detectCompat(model), + model.compat, ); expect(messages.slice(0, 2)).toEqual([ @@ -362,10 +376,10 @@ describe("openai-completions compatibility", () => { }); it("coalesces ordered system prompts when the host disables multi-system support", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const messages = convertMessages( model, @@ -373,7 +387,7 @@ describe("openai-completions compatibility", () => { systemPrompt: ["stable instructions", "cacheable policy"], messages: [{ role: "user", content: "hello", timestamp: Date.now() }], }, - { ...detectCompat(model), supportsMultipleSystemMessages: false }, + { ...model.compat, supportsMultipleSystemMessages: false }, ); expect(messages.slice(0, 2)).toEqual([ @@ -383,11 +397,11 @@ describe("openai-completions compatibility", () => { }); it("coalesces system prompts on a developer-role reasoning model when multi-system is disabled", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", reasoning: true, - }; + } as ModelSpec<"openai-completions">); const messages = convertMessages( model, @@ -395,7 +409,7 @@ describe("openai-completions compatibility", () => { systemPrompt: ["stable instructions", "cacheable policy"], messages: [{ role: "user", content: "hello", timestamp: Date.now() }], }, - { ...detectCompat(model), supportsMultipleSystemMessages: false }, + { ...model.compat, supportsMultipleSystemMessages: false }, ); expect(messages.slice(0, 2)).toEqual([ @@ -405,14 +419,14 @@ describe("openai-completions compatibility", () => { }); it("emits separate system prompts for an unknown OpenAI-compatible host when explicitly enabled", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "custom" as Model["provider"], baseUrl: "https://example.invalid/v1", - }; + } as ModelSpec<"openai-completions">); - const detected = detectCompat(model); + const detected = model.compat; expect(detected.supportsMultipleSystemMessages).toBe(false); const overridden = convertMessages( @@ -432,14 +446,14 @@ describe("openai-completions compatibility", () => { }); it("auto-detects MiniMax OpenAI hosts as single-system to satisfy error 2013", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "minimax-code" as Model["provider"], baseUrl: "https://api.minimax.io/v1", - }; + } as ModelSpec<"openai-completions">); - const detected = detectCompat(model); + const detected = model.compat; expect(detected.supportsMultipleSystemMessages).toBe(false); const messages = convertMessages( @@ -458,8 +472,8 @@ describe("openai-completions compatibility", () => { }); it("respects an explicit compat override for strict-template local providers", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "custom" as Model["provider"], baseUrl: "https://my-vllm.local/v1", @@ -467,7 +481,7 @@ describe("openai-completions compatibility", () => { supportsDeveloperRole: false, supportsMultipleSystemMessages: false, }, - }; + } as ModelSpec<"openai-completions">); const messages = convertMessages( model, @@ -475,7 +489,7 @@ describe("openai-completions compatibility", () => { systemPrompt: ["stable instructions", "cacheable policy"], messages: [{ role: "user", content: "hello", timestamp: Date.now() }], }, - resolveOpenAICompat(model), + model.compat, ); expect(messages.slice(0, 2)).toEqual([ @@ -485,10 +499,10 @@ describe("openai-completions compatibility", () => { }); it("reads usage from choice usage fallback", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const fetchMock = createMockFetch([ { id: "chatcmpl-test", @@ -529,14 +543,14 @@ describe("openai-completions compatibility", () => { }); it("maps qwen chat template reasoning into chat_template_kwargs", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", reasoning: true, compat: { thinkingFormat: "qwen-chat-template", }, - }; + } as ModelSpec<"openai-completions">); const { promise, resolve } = Promise.withResolvers(); streamOpenAICompletions(model, baseContext(), { apiKey: "test-key", @@ -550,10 +564,10 @@ describe("openai-completions compatibility", () => { }); it("treats finish_reason end as stop", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const fetchMock = createMockFetch([ { id: "chatcmpl-end", @@ -581,8 +595,8 @@ describe("openai-completions compatibility", () => { }); it("injects compat.extraBody into OpenAI payload", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", compat: { extraBody: { @@ -590,7 +604,7 @@ describe("openai-completions compatibility", () => { controller: "mlx", }, }, - }; + } as ModelSpec<"openai-completions">); const { promise, resolve } = Promise.withResolvers(); const fetchMock = createMockFetch(["[DONE]"]); @@ -611,10 +625,10 @@ describe("openai-completions compatibility", () => { }); it("preserves the streamed reasoning field name when replay requires reasoning content", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const fetchMock = createMockFetch([ { id: "chatcmpl-reasoning-text", @@ -648,7 +662,7 @@ describe("openai-completions compatibility", () => { thinkingSignature: "reasoning_text", }); - const compat = { ...detectCompat(model), requiresReasoningContentForToolCalls: true }; + const compat = { ...model.compat, requiresReasoningContentForToolCalls: true }; const messages = convertMessages(model, { messages: [result] }, compat); const assistant = messages.find(message => message.role === "assistant"); expect(assistant).toBeDefined(); @@ -661,25 +675,25 @@ describe("openai-completions compatibility", () => { describe("kimi model detection via detectCompat", () => { function kimiOpenCodeModel(id: string): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + return buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", id, reasoning: true, - }; + } as ModelSpec<"openai-completions">); } function kimiMoonshotModel(id: string): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + return buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "moonshot", baseUrl: "https://api.moonshot.ai/v1", id, reasoning: true, - }; + } as ModelSpec<"openai-completions">); } // The z.ai binary `thinking: { type }` field is Kimi's *native* surface // (Moonshot / Kimi-code, matched by isMoonshotKimi). Kimi reached through an @@ -693,49 +707,49 @@ describe("kimi model detection via detectCompat", () => { // `compat.thinkingFormat` per catalog entry (e.g. kimi-code, wafer-serverless). it("reserves zai for native Kimi hosts and defaults proxies to OpenAI reasoning_effort", () => { // Native Moonshot surface → z.ai binary thinking. - const moonshotK25 = detectCompat(kimiMoonshotModel("kimi-k2.5")); + const moonshotK25 = kimiMoonshotModel("kimi-k2.5").compat; expect(moonshotK25.thinkingFormat).toBe("zai"); expect(moonshotK25.thinkingKeep).toBeUndefined(); - const moonshotK26 = detectCompat(kimiMoonshotModel("kimi-k2.6")); + const moonshotK26 = kimiMoonshotModel("kimi-k2.6").compat; expect(moonshotK26.thinkingFormat).toBe("zai"); expect(moonshotK26.thinkingKeep).toBe("all"); // OpenAI-compatible proxies → reasoning_effort ("openai"). - const opencodeK26 = detectCompat(kimiOpenCodeModel("kimi-k2.6")); + const opencodeK26 = kimiOpenCodeModel("kimi-k2.6").compat; expect(opencodeK26.thinkingFormat).toBe("openai"); expect(opencodeK26.thinkingKeep).toBeUndefined(); - const kiloKimi: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const kiloKimi: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "kilo", baseUrl: "https://api.kilo.ai/api/gateway", id: "moonshotai/kimi-k2.6", reasoning: true, - }; - expect(detectCompat(kiloKimi).thinkingFormat).toBe("openai"); + } as ModelSpec<"openai-completions">); + expect(kiloKimi.compat.thinkingFormat).toBe("openai"); // OpenRouter normalizes reasoning via its own object and keeps precedence // over the generic Kimi id match. - const openRouterKimi: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const openRouterKimi: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", id: "moonshotai/kimi-k2.6", reasoning: true, - }; - expect(detectCompat(openRouterKimi).thinkingFormat).toBe("openrouter"); + } as ModelSpec<"openai-completions">); + expect(openRouterKimi.compat.thinkingFormat).toBe("openrouter"); }); it("maps OpenRouter Anthropic adaptive reasoning efforts to the Anthropic scale", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", id: "anthropic/claude-fable-5", reasoning: true, - }; + } as ModelSpec<"openai-completions">); const highPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "high" }); const xhighPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "xhigh" }); @@ -749,7 +763,7 @@ describe("kimi model detection via detectCompat", () => { // permitted"). Kimi on opencode-* MUST NOT have reasoning_content injected, // even though it's still recognized as a Kimi model for other quirks. it("does not require reasoning_content for tool calls on kimi-k2.5 (opencode-go)", () => { - const compat = detectCompat(kimiOpenCodeModel("kimi-k2.5")); + const compat = kimiOpenCodeModel("kimi-k2.5").compat; expect(compat.requiresReasoningContentForToolCalls).toBe(false); // Kimi-specific quirks still apply even on opencode hosts. expect(compat.requiresAssistantContentForToolCalls).toBe(true); @@ -757,7 +771,7 @@ describe("kimi model detection via detectCompat", () => { it("does not inject reasoning_content placeholder for kimi on opencode-go", () => { const model = kimiOpenCodeModel("kimi-k2.5"); - const compat = detectCompat(model); + const compat = model.compat; const toolCallMessage: AssistantMessage = { role: "assistant", content: [ @@ -791,7 +805,7 @@ describe("kimi model detection via detectCompat", () => { it("does not replay streamed reasoning fields for kimi on opencode-go", () => { const model = kimiOpenCodeModel("kimi-k2.6"); - const compat = detectCompat(model); + const compat = model.compat; const toolCallMessage: AssistantMessage = { role: "assistant", content: [ @@ -1067,14 +1081,14 @@ describe("kimi model detection via detectCompat", () => { // `allowsSyntheticReasoningContentForToolCalls=false`, so DeepSeek V4 // payloads carry only `reasoning_content`. it("emits only reasoning_content on deepseek-v4-flash opencode-go tool-call replays", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", id: "deepseek-v4-flash", reasoning: true, - }; + } as ModelSpec<"openai-completions">); const priorAssistant: AssistantMessage = { role: "assistant", content: [ @@ -1151,14 +1165,14 @@ describe("kimi model detection via detectCompat", () => { { id: "qwen3.7-max", reasoning: "high" as const, expectReplay: true }, { id: "mimo-v2-pro", reasoning: "high" as const, expectReplay: true }, ])("opencode-go/%s reasoning=%s → replay=%s", async ({ id, reasoning, expectReplay }) => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", id, reasoning: true, - }; + } as ModelSpec<"openai-completions">); const priorAssistant: AssistantMessage = { role: "assistant", content: [ @@ -1232,7 +1246,7 @@ describe("kimi model detection via detectCompat", () => { it("injects reasoning_content placeholder when kimi-on-moonshot has tool calls without reasoning field", () => { const model = kimiMoonshotModel("kimi-k2.5"); - const compat = detectCompat(model); + const compat = model.compat; const toolCallMessage: AssistantMessage = { role: "assistant", content: [ @@ -1268,15 +1282,15 @@ describe("kimi model detection via detectCompat", () => { }); it("injects reasoning_content placeholder for direct Moonshot Kimi after thinking-disabled forced tool calls", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "moonshot", baseUrl: "https://api.moonshot.ai/v1", id: "kimi-k2.6", reasoning: false, - }; - const compat = detectCompat(model); + } as ModelSpec<"openai-completions">); + const compat = model.compat; const toolCallMessage: AssistantMessage = { role: "assistant", content: [ @@ -1311,14 +1325,14 @@ describe("kimi model detection via detectCompat", () => { }); it("does not inject reasoning_content when model is not kimi", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "opencode-go", baseUrl: "https://opencode.ai/zen/go/v1", id: "some-other-model", - }; - const compat = detectCompat(model); + } as ModelSpec<"openai-completions">); + const compat = model.compat; expect(compat.requiresReasoningContentForToolCalls).toBe(false); expect(compat.requiresAssistantContentForToolCalls).toBe(false); }); @@ -1327,35 +1341,35 @@ describe("kimi model detection via detectCompat", () => { // is provider-agnostic, so it's the cleanest signal that the id-pattern // match recognizes every Kimi variant. it.each(["kimi-k2.5", "kimi-k1.5", "kimi-k2-5"])("matches kimi model id: %s", id => { - const compat = detectCompat(kimiMoonshotModel(id)); + const compat = kimiMoonshotModel(id).compat; expect(compat.requiresAssistantContentForToolCalls).toBe(true); expect(compat.requiresReasoningContentForToolCalls).toBe(true); }); it("still matches moonshotai/kimi via openrouter", () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", id: "moonshotai/kimi-k2-5", reasoning: true, - }; - const compat = detectCompat(model); + } as ModelSpec<"openai-completions">); + const compat = model.compat; expect(compat.requiresReasoningContentForToolCalls).toBe(true); }); }); describe("NVIDIA NIM DeepSeek special-token stripping", () => { function nvidiaDeepseekModel(): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + return buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "nvidia", baseUrl: "https://integrate.api.nvidia.com/v1", id: "deepseek-ai/deepseek-v4-flash", reasoning: true, - }; + } as ModelSpec<"openai-completions">); } it("strips leaked <\uff5cDSML\uff5c...\uff5c> markers from visible content", async () => { @@ -1468,13 +1482,13 @@ describe("NVIDIA NIM DeepSeek special-token stripping", () => { }); it("leaves visible content alone for non-deepseek nvidia models", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "nvidia", baseUrl: "https://integrate.api.nvidia.com/v1", id: "meta/llama-3.3-70b-instruct", - }; + } as ModelSpec<"openai-completions">); const fetchMock = createMockFetch([ { id: "chatcmpl-nim-4", @@ -1538,14 +1552,14 @@ describe("applyOpenRouterRoutingVariant", () => { describe("anthropic cache control for OpenAI-compatible chat completions", () => { function claudeProxyModel(compat?: OpenAICompat): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + return buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", provider: "litellm", baseUrl: "https://litellm.example/v1", id: "claude-opus-4-8", compat, - }; + } as ModelSpec<"openai-completions">); } function cacheContext(): Context { @@ -1648,10 +1662,11 @@ describe("openrouterVariant request integration", () => { it("does not override an explicit variant in the model id", async () => { const base = getBundledModel("openrouter", "anthropic/claude-sonnet-4") as Model<"openai-completions">; - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ ...base, id: `${base.id}:online`, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); const { promise, resolve } = Promise.withResolvers(); const fetchMock = createMockFetch(["[DONE]"]); streamOpenAICompletions(model, baseContext(), { @@ -1666,10 +1681,10 @@ describe("openrouterVariant request integration", () => { }); it("leaves params.model unchanged for non-OpenRouter providers", async () => { - const model: Model<"openai-completions"> = { - ...getBundledModel("openai", "gpt-4o-mini"), + const model: Model<"openai-completions"> = buildModel({ + ...gpt4oMiniSpec, api: "openai-completions", - }; + } as ModelSpec<"openai-completions">); const { promise, resolve } = Promise.withResolvers(); const fetchMock = createMockFetch(["[DONE]"]); streamOpenAICompletions(model, baseContext(), { diff --git a/packages/ai/test/openai-completions-disable-reasoning.test.ts b/packages/ai/test/openai-completions-disable-reasoning.test.ts index ba6399d5b..70682a81f 100644 --- a/packages/ai/test/openai-completions-disable-reasoning.test.ts +++ b/packages/ai/test/openai-completions-disable-reasoning.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import type { Context, FetchImpl, Model, ModelSpec } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; const testContext: Context = { @@ -16,7 +17,7 @@ function createSseResponse(events: unknown[]): Response { } function createReasoningEffortModel(): Model<"openai-completions"> { - return { + return buildModel({ id: "minimal-reasoner", name: "Minimal Reasoner", api: "openai-completions", @@ -32,17 +33,19 @@ function createReasoningEffortModel(): Model<"openai-completions"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128_000, maxTokens: 16_384, - }; + }); } function createFireworksReasoningEffortModel(): Model<"openai-completions"> { - return { - ...createReasoningEffortModel(), + const base = createReasoningEffortModel(); + return buildModel({ + ...base, id: "glm-5.1", name: "GLM 5.1", provider: "fireworks", baseUrl: "https://api.fireworks.ai/inference/v1", - }; + compat: base.compatConfig, + } as ModelSpec<"openai-completions">); } async function captureDisableReasoningPayload(model: Model<"openai-completions">): Promise> { diff --git a/packages/ai/test/openai-completions-progress-chunk.test.ts b/packages/ai/test/openai-completions-progress-chunk.test.ts index 831fc9ab5..cbb0f7d72 100644 --- a/packages/ai/test/openai-completions-progress-chunk.test.ts +++ b/packages/ai/test/openai-completions-progress-chunk.test.ts @@ -3,8 +3,8 @@ import { isOpenAICompletionsProgressChunk, streamOpenAICompletions, } from "@oh-my-pi/pi-ai/providers/openai-completions"; -import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; -import { resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import type { Context, FetchImpl, Model, ModelSpec } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; const openAICompletionsModel = { @@ -80,83 +80,89 @@ function createKeepaliveOnlyCompletionsResponse(modelId: string, signal: AbortSi describe("resolveOpenAICompat stream idle timeout", () => { it("widens GLM 5.1 coding-plan stream watchdogs", () => { - const model = { + const model = buildModel({ ...openAICompletionsModel, id: "glm-5.1", name: "GLM-5.1", provider: "zhipu-coding-plan", baseUrl: "https://open.bigmodel.cn/api/coding/paas/v4", - } satisfies Model<"openai-completions">; + compat: openAICompletionsModel.compatConfig, + } as ModelSpec<"openai-completions">); - expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000); + expect(model.compat.streamIdleTimeoutMs).toBe(600_000); }); it("also widens custom Z.AI OpenAI-compatible GLM 5.1 endpoints", () => { - const model = { + const model = buildModel({ ...openAICompletionsModel, id: "glm-5.1", name: "GLM-5.1", provider: "openai", baseUrl: "https://api.z.ai/api/coding/paas/v4", - } satisfies Model<"openai-completions">; + compat: openAICompletionsModel.compatConfig, + } as ModelSpec<"openai-completions">); - expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(600_000); + expect(model.compat.streamIdleTimeoutMs).toBe(600_000); }); it("widens DeepSeek V4 reasoning streams on the official DeepSeek API", () => { - const model = { + const model = buildModel({ ...openAICompletionsModel, id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", provider: "deepseek", baseUrl: "https://api.deepseek.com", reasoning: true, - } satisfies Model<"openai-completions">; + compat: openAICompletionsModel.compatConfig, + } as ModelSpec<"openai-completions">); - expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000); + expect(model.compat.streamIdleTimeoutMs).toBe(300_000); }); it("widens DeepSeek reasoning streams routed through an aliased OpenAI-compatible provider id", () => { - const model = { + const model = buildModel({ ...openAICompletionsModel, id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", provider: "openai", baseUrl: "https://api.deepseek.com/v1", reasoning: true, - } satisfies Model<"openai-completions">; + compat: openAICompletionsModel.compatConfig, + } as ModelSpec<"openai-completions">); - expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBe(300_000); + expect(model.compat.streamIdleTimeoutMs).toBe(300_000); }); it("leaves non-reasoning DeepSeek-hosted models on the global timeout", () => { - const model = { + const model = buildModel({ ...openAICompletionsModel, id: "deepseek-chat", name: "DeepSeek Chat", provider: "deepseek", baseUrl: "https://api.deepseek.com", reasoning: false, - } satisfies Model<"openai-completions">; + compat: openAICompletionsModel.compatConfig, + } as ModelSpec<"openai-completions">); - expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined(); + expect(model.compat.streamIdleTimeoutMs).toBeUndefined(); }); it("does not widen DeepSeek V4 reasoning models hosted on third-party OpenAI-compatible proxies", () => { - const model = { + const model = buildModel({ ...openAICompletionsModel, id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", provider: "aimlapi", baseUrl: "https://api.aimlapi.com/v1", reasoning: true, - } satisfies Model<"openai-completions">; + compat: openAICompletionsModel.compatConfig, + } as ModelSpec<"openai-completions">); - expect(resolveOpenAICompat(model).streamIdleTimeoutMs).toBeUndefined(); + expect(model.compat.streamIdleTimeoutMs).toBeUndefined(); }); it("keeps ordinary OpenAI-compatible models on the global timeout", () => { - expect(resolveOpenAICompat(openAICompletionsModel).streamIdleTimeoutMs).toBeUndefined(); + expect(openAICompletionsModel.compat.streamIdleTimeoutMs).toBeUndefined(); }); }); diff --git a/packages/ai/test/openai-completions-tool-result-images.test.ts b/packages/ai/test/openai-completions-tool-result-images.test.ts index 3ed9c0812..7ebdb45bf 100644 --- a/packages/ai/test/openai-completions-tool-result-images.test.ts +++ b/packages/ai/test/openai-completions-tool-result-images.test.ts @@ -2,8 +2,8 @@ import { describe, expect, it } from "bun:test"; import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard"; import type { AssistantMessage, Context, Model, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai/types"; -import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import type { ResolvedOpenAICompat } from "@oh-my-pi/pi-catalog/types"; const emptyUsage: Usage = { input: 0, @@ -39,6 +39,10 @@ const compat: ResolvedOpenAICompat = { extraBody: {}, supportsStrictMode: true, toolStrictMode: "none", + supportsReasoningParams: true, + alwaysSendMaxTokens: false, + isOpenRouterHost: false, + isVercelGatewayHost: false, }; function buildToolResult(toolCallId: string, timestamp: number): ToolResultMessage { diff --git a/packages/ai/test/openai-first-event-timeout.test.ts b/packages/ai/test/openai-first-event-timeout.test.ts index fbaebe063..cf822d7aa 100644 --- a/packages/ai/test/openai-first-event-timeout.test.ts +++ b/packages/ai/test/openai-first-event-timeout.test.ts @@ -4,6 +4,7 @@ import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-comple import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; import { streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { Context, FetchImpl, Model, TextContent } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { waitForDelayOrAbort } from "./helpers"; @@ -12,7 +13,7 @@ const openAICompletionsModel = { ...(getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">), api: "openai-completions", } satisfies Model<"openai-completions">; -const azureOpenAIResponsesModel: Model<"azure-openai-responses"> = { +const azureOpenAIResponsesModel: Model<"azure-openai-responses"> = buildModel({ id: "gpt-5-mini", name: "GPT-5 Mini", api: "azure-openai-responses", @@ -23,8 +24,8 @@ const azureOpenAIResponsesModel: Model<"azure-openai-responses"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400000, maxTokens: 128000, -}; -const ollamaChatModel: Model<"ollama-chat"> = { +}); +const ollamaChatModel: Model<"ollama-chat"> = buildModel({ id: "llama-local", name: "llama-local", api: "ollama-chat", @@ -35,7 +36,7 @@ const ollamaChatModel: Model<"ollama-chat"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 8192, -}; +}); function baseContext(): Context { return { diff --git a/packages/ai/test/openai-max-output-tokens-cap.test.ts b/packages/ai/test/openai-max-output-tokens-cap.test.ts index 8e862ca0f..cb932ac58 100644 --- a/packages/ai/test/openai-max-output-tokens-cap.test.ts +++ b/packages/ai/test/openai-max-output-tokens-cap.test.ts @@ -2,7 +2,8 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; -import { type Context, type Model, OPENAI_MAX_OUTPUT_TOKENS } from "@oh-my-pi/pi-ai/types"; +import { type Context, type Model, type ModelSpec, OPENAI_MAX_OUTPUT_TOKENS } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; // Output-token wire policy for OpenAI-family providers: @@ -92,7 +93,7 @@ async function captureCompletionsBody( // The OpenRouter z-ai/glm-4.7 entry that triggered the report. function glmCompletionsModel(maxTokens: number): Model<"openai-completions"> { - return { + return buildModel({ id: "z-ai/glm-4.7", name: "GLM 4.7", api: "openai-completions", @@ -103,12 +104,12 @@ function glmCompletionsModel(maxTokens: number): Model<"openai-completions"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 202_752, maxTokens, - }; + }); } // Non-aggregator completions model: the 64k clamp applies (max_tokens is sent). function directCompletionsModel(maxTokens: number): Model<"openai-completions"> { - return { + return buildModel({ id: "glm-4.7", name: "GLM 4.7 (direct)", api: "openai-completions", @@ -119,12 +120,12 @@ function directCompletionsModel(maxTokens: number): Model<"openai-completions"> cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 131_072, maxTokens, - }; + }); } // Kimi via OpenRouter stays exempt from the omit (TPM rate limits need max_tokens). function kimiOpenRouterModel(maxTokens: number): Model<"openai-completions"> { - return { + return buildModel({ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5", api: "openai-completions", @@ -135,16 +136,18 @@ function kimiOpenRouterModel(maxTokens: number): Model<"openai-completions"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 131_072, maxTokens, - }; + }); } describe("OpenAI-family output-token cap", () => { it("clamps openai-responses max_output_tokens to the 64k ceiling", async () => { - const model: Model<"openai-responses"> = { - ...(getBundledModel("openai", "gpt-4o-mini") as Model<"openai-responses">), + const base = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-responses">; + const model: Model<"openai-responses"> = buildModel({ + ...base, reasoning: false, maxTokens: 200_000, - }; + compat: base.compatConfig, + } as ModelSpec<"openai-responses">); const body = await drainResponses(model); expect(body.max_output_tokens).toBe(OPENAI_MAX_OUTPUT_TOKENS); }); diff --git a/packages/ai/test/openai-responses-developer-role.test.ts b/packages/ai/test/openai-responses-developer-role.test.ts index 5129a0048..42cd9293c 100644 --- a/packages/ai/test/openai-responses-developer-role.test.ts +++ b/packages/ai/test/openai-responses-developer-role.test.ts @@ -1,31 +1,31 @@ import { describe, expect, it } from "bun:test"; -import { resolveOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import { buildOpenAIResponsesCompat } from "@oh-my-pi/pi-catalog/compat/openai"; describe("resolveOpenAIResponsesCompat supportsDeveloperRole", () => { it("returns true for openai provider with official API base URL", () => { const model = { provider: "openai", baseUrl: "https://api.openai.com/v1" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns false for openai provider with custom proxy base URL", () => { const model = { provider: "openai", baseUrl: "https://my-proxy.example.com/v1" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("returns true for github-copilot provider", () => { const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns false for github-copilot provider with custom proxy base URL", () => { const model = { provider: "github-copilot", baseUrl: "https://proxy.example.com/v1" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("returns true for Azure OpenAI base URL", () => { const model = { provider: "azure-openai", baseUrl: "https://my-resource.openai.azure.com/openai" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for Azure AI Inference base URL", () => { @@ -33,46 +33,46 @@ describe("resolveOpenAIResponsesCompat supportsDeveloperRole", () => { provider: "azure-openai", baseUrl: "https://models.inference.ai.azure.com/v1/chat/completions", }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for api.openai.com base URL", () => { const model = { provider: "custom", baseUrl: "https://api.openai.com/v1/chat/completions" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns false for generic third-party provider", () => { const model = { provider: "custom", baseUrl: "https://api.example.com/v1" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("returns false for local/localhost endpoints", () => { const model = { provider: "custom", baseUrl: "http://localhost:8080/v1" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(false); }); it("is case-insensitive for base URL matching", () => { const model = { provider: "custom", baseUrl: "https://API.OPENAI.COM/v1" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for azure.com/openai base URL", () => { const model = { provider: "custom", baseUrl: "https://azure.com/openai/deployments/my-model" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for github-copilot provider with api.githubcopilot.com", () => { const model = { provider: "github-copilot", baseUrl: "https://api.githubcopilot.com" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for github-copilot provider with api.enterprise.githubcopilot.com", () => { const model = { provider: "github-copilot", baseUrl: "https://api.enterprise.githubcopilot.com" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); it("returns true for github-copilot provider with copilot-api enterprise domain", () => { const model = { provider: "github-copilot", baseUrl: "https://copilot-api.mycompany.com" }; - expect(resolveOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); + expect(buildOpenAIResponsesCompat(model).supportsDeveloperRole).toBe(true); }); }); diff --git a/packages/ai/test/openai-responses-history-payload.test.ts b/packages/ai/test/openai-responses-history-payload.test.ts index a894542b5..a10c5d6c0 100644 --- a/packages/ai/test/openai-responses-history-payload.test.ts +++ b/packages/ai/test/openai-responses-history-payload.test.ts @@ -1,8 +1,9 @@ import { describe, expect, it } from "bun:test"; import { streamOpenAICodexResponses } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; import { type OpenAIResponsesOptions, streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; -import type { Context, Model, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; +import type { Context, Model, ModelSpec, ProviderSessionState } from "@oh-my-pi/pi-ai/types"; import { createOpenAIResponsesHistoryPayload, truncateResponseItemId } from "@oh-my-pi/pi-ai/utils"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; function createAbortedSignal(): AbortSignal { @@ -323,10 +324,12 @@ describe("OpenAI responses history payload", () => { }); it("uses canonical instructions field for endpoints without developer-role support", async () => { - const model = { - ...getOpenAIReasoningModel("openai", "gpt-5-mini"), + const baseModel = getOpenAIReasoningModel("openai", "gpt-5-mini"); + const model = buildModel({ + ...baseModel, baseUrl: "https://proxy.example.com/v1", - }; + compat: baseModel.compatConfig, + } as ModelSpec<"openai-responses">); const payload = (await captureResponsesPayload(model, { systemPrompt: ["stable instructions", "second instructions"], messages: [{ role: "user", content: "hi", timestamp: Date.now() }], diff --git a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts index dbff9a657..2233db71e 100644 --- a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts +++ b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts @@ -15,10 +15,11 @@ import { describe, expect, test } from "bun:test"; import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { ResponseStreamEvent } from "openai/resources/responses/responses"; function makeModel(): Model<"openai-responses"> { - return { + return buildModel({ api: "openai-responses", name: "Llama", id: "llama-3", @@ -29,7 +30,7 @@ function makeModel(): Model<"openai-responses"> { input: ["text"], reasoning: false, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - }; + }); } function makeOutput(): AssistantMessage { diff --git a/packages/ai/test/openai-responses-stream-terminal.test.ts b/packages/ai/test/openai-responses-stream-terminal.test.ts index 597f85acf..4ff09bdfa 100644 --- a/packages/ai/test/openai-responses-stream-terminal.test.ts +++ b/packages/ai/test/openai-responses-stream-terminal.test.ts @@ -10,10 +10,11 @@ import { describe, expect, test } from "bun:test"; import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { ResponseStreamEvent } from "openai/resources/responses/responses"; function makeModel(): Model<"openai-responses"> { - return { + return buildModel({ api: "openai-responses", name: "GPT Test", id: "gpt-test", @@ -24,7 +25,7 @@ function makeModel(): Model<"openai-responses"> { input: ["text"], reasoning: false, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - }; + }); } function makeOutput(): AssistantMessage { diff --git a/packages/ai/test/openai-responses-system-prompt.test.ts b/packages/ai/test/openai-responses-system-prompt.test.ts index 2f447d9a8..f09888189 100644 --- a/packages/ai/test/openai-responses-system-prompt.test.ts +++ b/packages/ai/test/openai-responses-system-prompt.test.ts @@ -1,6 +1,7 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; -import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import type { Context, FetchImpl, Model, ModelSpec } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; // Non-reasoning model on api.openai.com (canonical path) @@ -96,10 +97,12 @@ describe("openai-responses system prompt routing", () => { }); it("uses instructions for custom proxy base URL (third-party /v1/responses compatibility)", async () => { - const proxyModel: Model<"openai-responses"> = { + const proxyModel: Model<"openai-responses"> = buildModel({ ...gpt4oMiniModel, + api: "openai-responses", baseUrl: "https://proxy.example.com/v1", - }; + compat: gpt4oMiniModel.compatConfig, + } as ModelSpec<"openai-responses">); const context: Context = { systemPrompt: ["You are a proxy assistant."], messages: [{ role: "user", content: "hi", timestamp: Date.now() }], @@ -145,10 +148,12 @@ describe("openai-responses system prompt routing", () => { describe("reasoning model on custom proxy (instructions path)", () => { it("uses instructions for reasoning model on non-official endpoint", async () => { - const proxyModel: Model<"openai-responses"> = { + const proxyModel: Model<"openai-responses"> = buildModel({ ...o4MiniModel, + api: "openai-responses", baseUrl: "https://proxy.example.com/v1", - }; + compat: o4MiniModel.compatConfig, + } as ModelSpec<"openai-responses">); const context: Context = { systemPrompt: ["Proxy reasoning prompt."], messages: [{ role: "user", content: "hi", timestamp: Date.now() }], diff --git a/packages/ai/test/openai-tool-strict-mode.test.ts b/packages/ai/test/openai-tool-strict-mode.test.ts index 6f0491f1c..6724ceda4 100644 --- a/packages/ai/test/openai-tool-strict-mode.test.ts +++ b/packages/ai/test/openai-tool-strict-mode.test.ts @@ -1,7 +1,16 @@ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; -import type { Context, FetchImpl, Model, OpenAICompat, ProviderSessionState, Tool } from "@oh-my-pi/pi-ai/types"; +import type { + Context, + FetchImpl, + Model, + ModelSpec, + OpenAICompat, + ProviderSessionState, + Tool, +} from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import * as z from "zod/v4"; @@ -118,11 +127,11 @@ describe("OpenAI tool strict mode", () => { }); it("omits strict for openai-completions when compatibility disables strict mode", async () => { - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ ...(getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">), api: "openai-completions", compat: { supportsStrictMode: false } satisfies OpenAICompat, - }; + } as ModelSpec<"openai-completions">); const payload = (await captureCompletionsPayload(model)) as { tools?: Array<{ function?: { strict?: boolean } }>; @@ -175,11 +184,11 @@ describe("OpenAI tool strict mode", () => { }); it("uses uniformly non-strict tool schemas when provider requires all-or-none strictness", async () => { - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ ...(getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">), api: "openai-completions", compat: { toolStrictMode: "all_strict" } satisfies OpenAICompat, - }; + } as ModelSpec<"openai-completions">); const context: Context = { ...testContext, tools: [ @@ -234,11 +243,11 @@ describe("OpenAI tool strict mode", () => { }); it("retries with non-strict tool schemas after strict-mode request errors", async () => { - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ ...(getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">), api: "openai-completions", compat: { toolStrictMode: "all_strict" } satisfies OpenAICompat, - }; + } as ModelSpec<"openai-completions">); const strictFlags: boolean[][] = []; const fetchMock: FetchImpl = Object.assign( async (_input: string | URL | Request, init?: RequestInit): Promise => { diff --git a/packages/ai/test/pi-native-client.test.ts b/packages/ai/test/pi-native-client.test.ts index eb1052542..2034b4068 100644 --- a/packages/ai/test/pi-native-client.test.ts +++ b/packages/ai/test/pi-native-client.test.ts @@ -1,6 +1,14 @@ import { afterEach, describe, expect, it, mock, spyOn } from "bun:test"; import { streamPiNative } from "@oh-my-pi/pi-ai/providers/pi-native-client"; -import type { AssistantMessage, AssistantMessageEvent, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import type { + AssistantMessage, + AssistantMessageEvent, + Context, + FetchImpl, + Model, + ModelSpec, +} from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function sseBytes(events: AssistantMessageEvent[]): Uint8Array { const encoder = new TextEncoder(); @@ -58,7 +66,7 @@ function baseAssistant(overrides: Partial = {}): AssistantMess } function fakeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { - return { + return buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -71,7 +79,7 @@ function fakeModel(overrides: Partial> = {}): Model< maxTokens: 64000, transport: "pi-native", ...overrides, - }; + } as ModelSpec<"anthropic-messages">); } const baseContext: Context = { diff --git a/packages/ai/test/raw-sse-sdk-capture.test.ts b/packages/ai/test/raw-sse-sdk-capture.test.ts index 35f389a7a..6399592ae 100644 --- a/packages/ai/test/raw-sse-sdk-capture.test.ts +++ b/packages/ai/test/raw-sse-sdk-capture.test.ts @@ -6,6 +6,7 @@ import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-open import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses"; import type { Context, FetchImpl, Model, RawSseEvent } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; const context: Context = { @@ -17,7 +18,7 @@ const openAICompletionsModel = { ...(getBundledModel("openai", "gpt-4o-mini") as Model<"openai-completions">), api: "openai-completions", } satisfies Model<"openai-completions">; -const azureOpenAIResponsesModel: Model<"azure-openai-responses"> = { +const azureOpenAIResponsesModel: Model<"azure-openai-responses"> = buildModel({ id: "gpt-5-mini", name: "GPT-5 Mini", api: "azure-openai-responses", @@ -28,8 +29,8 @@ const azureOpenAIResponsesModel: Model<"azure-openai-responses"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400_000, maxTokens: 128_000, -}; -const anthropicModel: Model<"anthropic-messages"> = { +}); +const anthropicModel: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -40,7 +41,7 @@ const anthropicModel: Model<"anthropic-messages"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); const openAIResponsesEvents = [ { type: "response.created", response: { id: "resp_raw_sse", status: "in_progress" } }, diff --git a/packages/ai/test/register-builtins.test.ts b/packages/ai/test/register-builtins.test.ts index 9f7a6a389..23326aa75 100644 --- a/packages/ai/test/register-builtins.test.ts +++ b/packages/ai/test/register-builtins.test.ts @@ -2,9 +2,10 @@ import { describe, expect, it } from "bun:test"; import { setBedrockProviderModule, streamBedrock } from "@oh-my-pi/pi-ai/providers/register-builtins"; import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types"; import type { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function createModel(): Model<"bedrock-converse-stream"> { - return { + return buildModel({ id: "mock-bedrock", name: "Mock Bedrock", api: "bedrock-converse-stream", @@ -15,7 +16,7 @@ function createModel(): Model<"bedrock-converse-stream"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 8192, maxTokens: 2048, - }; + }); } function createAssistantMessage( diff --git a/packages/ai/test/request-debug.test.ts b/packages/ai/test/request-debug.test.ts index 389812218..95332136b 100644 --- a/packages/ai/test/request-debug.test.ts +++ b/packages/ai/test/request-debug.test.ts @@ -4,9 +4,10 @@ import * as os from "node:os"; import * as path from "node:path"; import { clearCustomApis, registerCustomApi } from "@oh-my-pi/pi-ai/api-registry"; import { stream } from "@oh-my-pi/pi-ai/stream"; -import type { AssistantMessage, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessage, FetchImpl, Model, ModelSpec } from "@oh-my-pi/pi-ai/types"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { wrapFetchForRequestDebug } from "@oh-my-pi/pi-ai/utils/request-debug"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; const enc = new TextEncoder(); @@ -179,7 +180,7 @@ describe("PI_REQ_DEBUG request/response recording", () => { return events; }); - const model: Model = { + const model: Model = buildModel({ id: "debug-model", name: "Debug Model", api: "req-debug-test", @@ -190,7 +191,7 @@ describe("PI_REQ_DEBUG request/response recording", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 4096, maxTokens: 1024, - }; + } as ModelSpec); const events = stream( model, { messages: [{ role: "user", content: "hi", timestamp: Date.now() }] }, diff --git a/packages/ai/test/schema-normalization.test.ts b/packages/ai/test/schema-normalization.test.ts index f85b35bb6..caa79a80d 100644 --- a/packages/ai/test/schema-normalization.test.ts +++ b/packages/ai/test/schema-normalization.test.ts @@ -15,9 +15,10 @@ import { tryEnforceStrictSchema, upgradeJsonSchemaTo202012, } from "@oh-my-pi/pi-ai/utils/schema"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function createGoogleCliModel(id: string): Model<"google-gemini-cli"> { - return { + return buildModel({ id, name: id, api: "google-gemini-cli", @@ -33,7 +34,7 @@ function createGoogleCliModel(id: string): Model<"google-gemini-cli"> { }, contextWindow: 200000, maxTokens: 8192, - }; + }); } // --------------------------------------------------------------------------- diff --git a/packages/ai/test/stream-markup-healing.test.ts b/packages/ai/test/stream-markup-healing.test.ts index 1265293d8..2cf04edb4 100644 --- a/packages/ai/test/stream-markup-healing.test.ts +++ b/packages/ai/test/stream-markup-healing.test.ts @@ -3,6 +3,7 @@ import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-comple import { stream } from "@oh-my-pi/pi-ai/stream"; import type { Context, FetchImpl, Model, Tool, ToolCall } from "@oh-my-pi/pi-ai/types"; import { getStreamMarkupHealingPattern, StreamMarkupHealing } from "@oh-my-pi/pi-ai/utils/stream-markup-healing"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; interface SseToolCallDelta { @@ -102,7 +103,7 @@ const readTool: Tool = { additionalProperties: false, }, }; -const deepseekCloudModel: Model<"ollama-chat"> = { +const deepseekCloudModel: Model<"ollama-chat"> = buildModel({ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", api: "ollama-chat", @@ -113,7 +114,7 @@ const deepseekCloudModel: Model<"ollama-chat"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 131_072, maxTokens: 8_192, -}; +}); function ndjsonResponse(lines: ReadonlyArray): Response { const body = `${lines.map(line => JSON.stringify(line)).join("\n")}\n`; @@ -602,7 +603,7 @@ describe("Ollama provider DSML envelope healing", () => { describe("OpenAI completions MiniMax thinking healing", () => { it("parses OpenCode Zen MiniMax think tags into a thinking block", async () => { - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ id: "minimax-m3", name: "MiniMax M3", api: "openai-completions", @@ -613,7 +614,7 @@ describe("OpenAI completions MiniMax thinking healing", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, - }; + }); const fetchMock = mockFetch([ chunk(model.id, { content: "visible hidden reasoning { describe("OpenAI completions provider DSML envelope healing", () => { it("heals the envelope into a structured tool call and suppresses leaked text", async () => { - const model: Model<"openai-completions"> = { + const model: Model<"openai-completions"> = buildModel({ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", api: "openai-completions", @@ -649,7 +650,7 @@ describe("OpenAI completions provider DSML envelope healing", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 131_072, maxTokens: 8_192, - }; + }); const fetchMock = mockFetch([ chunk(model.id, { content: "I'll check.\n" }), chunk(model.id, { content: `${REPORTED_DSML_LEAK}\nThat should give us the package list.` }), diff --git a/packages/ai/test/stream.test.ts b/packages/ai/test/stream.test.ts index 6c972dbdf..f98198605 100644 --- a/packages/ai/test/stream.test.ts +++ b/packages/ai/test/stream.test.ts @@ -7,6 +7,7 @@ import { Effort } from "@oh-my-pi/pi-ai"; import { __resetVertexTokenCache } from "@oh-my-pi/pi-ai/providers/google-auth"; import { complete, getEnvApiKey, stream } from "@oh-my-pi/pi-ai/stream"; import type { Api, Context, ImageContent, Model, OptionsForApi, Tool, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { $which } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; @@ -566,7 +567,7 @@ describe("Generate E2E Tests", () => { const homedirSpy = spyOn(os, "homedir").mockReturnValue( path.join(os.tmpdir(), `vertex-adc-absent-${Date.now()}`), ); - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4@20250514", name: "Claude Sonnet 4", api: "anthropic-messages", @@ -578,7 +579,7 @@ describe("Generate E2E Tests", () => { cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }, contextWindow: 200_000, maxTokens: 64_000, - }; + }); const captured = Promise.withResolvers<{ url: string; authorization: string | null; body: unknown }>(); try { @@ -679,7 +680,7 @@ describe("Generate E2E Tests", () => { delegates: ["projects/-/serviceAccounts/delegate@project.iam.gserviceaccount.com"], }), ); - const model: Model<"anthropic-messages"> = { + const model: Model<"anthropic-messages"> = buildModel({ id: "claude-sonnet-4@20250514", name: "Claude Sonnet 4", api: "anthropic-messages", @@ -691,7 +692,7 @@ describe("Generate E2E Tests", () => { cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }, contextWindow: 200_000, maxTokens: 64_000, - }; + }); const callOrder: string[] = []; let iamRequest: { url: string; authorization: string | null; body: unknown } | undefined; const captured = Promise.withResolvers<{ url: string; authorization: string | null }>(); @@ -1824,7 +1825,7 @@ describe("Generate E2E Tests", () => { setTimeout(checkServer, 1000); // Initial delay }); - llm = { + llm = buildModel({ id: "gpt-oss:20b", api: "openai-completions", provider: "ollama", @@ -1840,7 +1841,7 @@ describe("Generate E2E Tests", () => { cacheWrite: 0, }, name: "Ollama GPT-OSS 20B", - }; + }); }, 30000); // 30 second timeout for setup afterAll(() => { diff --git a/packages/ai/test/transform-messages-dedup.test.ts b/packages/ai/test/transform-messages-dedup.test.ts index 65634e090..a0026ac88 100644 --- a/packages/ai/test/transform-messages-dedup.test.ts +++ b/packages/ai/test/transform-messages-dedup.test.ts @@ -8,9 +8,10 @@ import { describe, expect, it } from "bun:test"; import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages"; import type { AssistantMessage, Message, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; import { normalizeResponsesToolCallId } from "@oh-my-pi/pi-ai/utils"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; function makeModel(): Model<"openai-responses"> { - return { + return buildModel({ api: "openai-responses", name: "GPT Test", id: "gpt-test", @@ -21,7 +22,7 @@ function makeModel(): Model<"openai-responses"> { input: ["text"], reasoning: false, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - }; + }); } function assistantWithCall(id: string): AssistantMessage { diff --git a/packages/ai/test/usage-attribution.test.ts b/packages/ai/test/usage-attribution.test.ts index 081a0366b..a1c61e7f2 100644 --- a/packages/ai/test/usage-attribution.test.ts +++ b/packages/ai/test/usage-attribution.test.ts @@ -2,8 +2,9 @@ import { describe, expect, it } from "bun:test"; import { applyAnthropicUsageExtras } from "@oh-my-pi/pi-ai/providers/anthropic"; import { parseChunkUsage } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { Model, Usage } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; -const OPENAI_MODEL: Model<"openai-completions"> = { +const OPENAI_MODEL: Model<"openai-completions"> = buildModel({ id: "gpt-5", name: "GPT-5", api: "openai-completions", @@ -14,7 +15,7 @@ const OPENAI_MODEL: Model<"openai-completions"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); function blankUsage(): Usage { return { diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 823a04d88..0ed2e1275 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -1,12 +1,12 @@ # Changelog ## [Unreleased] + ### Added - Added `hostMatchesUrl`, `modelMatchesHost`, and endpoint-shape helpers in the new `hosts` module for consistent provider/baseUrl matching -- Added `streamIdleTimeoutMs` to `OpenAICompat` and now auto-populated it for GLM coding-plan and direct DeepSeek reasoning models -- Added `supportsLongPromptCacheRetention` and the OpenAI Responses helpers `detectOpenAIResponsesCompat`/`resolveOpenAIResponsesCompat` -- Added anthropic-messages compatibility resolution with new `AnthropicCompat` fields `requiresToolResultId` and `replayUnsignedThinking` +- `buildModel(spec)` (`build.ts`) is now the single Model constructor: it runs thinking enrichment and materializes the fully-resolved compat record exactly once, so `Model.compat` is a required, complete `CompatOf` (`ResolvedOpenAICompat`/`ResolvedOpenAIResponsesCompat`/`ResolvedAnthropicCompat`) and request-path code reads fields with zero URL parsing and zero per-request allocation. Sparse user/config overrides live on the new `ModelSpec` input shape and survive on `Model.compatConfig` for introspection. +- Compat detection gained model-time flags so handlers stop sniffing baseUrl: completions `supportsReasoningParams`, `alwaysSendMaxTokens`, `isOpenRouterHost`, `isVercelGatewayHost`, `streamIdleTimeoutMs`, and a precomputed `whenThinking` alternate view (OpenCode `reasoning_content` gating, #1071/#1484); responses `strictResponsesPairing`, `supportsLongPromptCacheRetention`, `supportsReasoningEffort`; anthropic `officialEndpoint`, `requiresToolResultId`, `replayUnsignedThinking`. - New `@oh-my-pi/pi-catalog` package: the model catalog extracted from `@oh-my-pi/pi-ai`. Owns the bundled `models.json` and its generation pipeline (`scripts/generate-models.ts`), the core model data types (`Model`, `Api`, `ThinkingConfig`, `Effort`, `Usage`, compat interfaces), thinking metadata enrichment and generated policies (`model-thinking.ts`), the SQLite model cache and model manager, per-provider discovery factories (`provider-models/`), the discovery protocol clients (`discovery/`), and the new `CATALOG_PROVIDERS` table — the single source of truth for provider ids, default models, and discovery wiring (`KnownProvider`, `PROVIDER_DESCRIPTORS`, and `DEFAULT_MODEL_PER_PROVIDER` are derived from it). - New `identity/` module centralizing model-identity concerns that were previously duplicated across packages: family classification and version parsing (`identity/classify.ts`, extracted from pi-ai's `model-thinking` internals), canonical model equivalence with injected reference data (`identity/equivalence.ts`, from coding-agent's `model-equivalence`), proxy/reseller reference lookup (`identity/reference.ts`, from coding-agent's `model-registry`), bracket-affix and id-segment helpers (`identity/id.ts`), a single trailing-marker vocabulary with canonical vs reference flavors (`identity/markers.ts` — `search` stays reference-only so Perplexity's `sonar-pro-search` remains canonical-distinct), and provider priority ordering (`identity/priority.ts`). - Memoized bundled-reference accessors (`getBundledCanonicalReferenceData` / `getBundledModelReferenceIndex` in `identity/bundled.ts`): one lazy walk of the bundled catalog feeds both canonical equivalence and proxy-reference lookup, so consumers no longer hand-roll the glue. @@ -21,4 +21,5 @@ ### Fixed +- Fixed Anthropic official-endpoint detection to require strict HTTPS hostname matching so non-official or lookalike URLs are no longer treated as official Anthropic hosts - Wired `@oh-my-pi/pi-catalog` into the release publish package list, tarball install smoke test, and root `bun generate-models` script. diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index a1de0a3c4..f4bb37dff 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -23,6 +23,7 @@ import { linkOpenAIPromotionTargets, } from "../src/model-thinking"; import prevModelsJson from "../src/models.json" with { type: "json" }; +import { toModelSpec } from "../src/provider-models/bundled-references"; import { allowsUnauthenticatedCatalogDiscovery, type CatalogDiscoveryConfig, @@ -41,7 +42,7 @@ import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS, } from "../src/provider-models/openai-compat"; -import type { Model } from "../src/types"; +import type { ModelSpec } from "../src/types"; import { JWT_CLAIM_PATH } from "../src/wire/codex"; const packageRoot = path.join(import.meta.dir, ".."); @@ -93,7 +94,7 @@ async function resolveProviderApiKey(providerId: string, catalog: CatalogDiscove return undefined; } -async function fetchProviderModelsFromCatalog(descriptor: CatalogProviderDescriptor): Promise { +async function fetchProviderModelsFromCatalog(descriptor: CatalogProviderDescriptor): Promise { const apiKey = await resolveProviderApiKey(descriptor.providerId, descriptor.catalogDiscovery); if (!apiKey && !allowsUnauthenticatedCatalogDiscovery(descriptor)) { @@ -111,14 +112,15 @@ async function fetchProviderModelsFromCatalog(descriptor: CatalogProviderDescrip return []; } console.log(`Fetched ${models.length} models from ${descriptor.catalogDiscovery.label} model manager`); - return models; + // The manager returns built models; models.json stores specs (sparse compat). + return models.map(model => toModelSpec(model)); } catch (error) { console.error(`Failed to fetch ${descriptor.catalogDiscovery.label} models:`, error); return []; } } -async function loadModelsDevData(): Promise { +async function loadModelsDevData(): Promise { try { console.log("Fetching models from models.dev API..."); const response = await fetch("https://models.dev/api.json"); @@ -133,8 +135,8 @@ async function loadModelsDevData(): Promise { } } -function createGlobalModelsDevReferenceMap(modelsDevModels: readonly Model[]): Map { - const references = new Map(); +function createGlobalModelsDevReferenceMap(modelsDevModels: readonly ModelSpec[]): Map { + const references = new Map(); for (const model of modelsDevModels) { const existing = references.get(model.id); if (!existing) { @@ -156,7 +158,10 @@ function inheritModelsDevLimit(value: number, referenceValue: number, unspecifie return value === unspecifiedValue ? referenceValue : value; } -function applyGlobalModelsDevFallback(models: readonly Model[], modelsDevModels: readonly Model[]): Model[] { +function applyGlobalModelsDevFallback( + models: readonly ModelSpec[], + modelsDevModels: readonly ModelSpec[], +): ModelSpec[] { const providerScopedKeys = new Set(modelsDevModels.map(model => `${model.provider}/${model.id}`)); const globalReferences = createGlobalModelsDevReferenceMap(modelsDevModels); return models.map(model => { @@ -180,7 +185,7 @@ function applyGlobalModelsDevFallback(models: readonly Model[], modelsDevModels: }); } -function applyPremiumMultiplierOverrides(models: readonly Model[]): Model[] { +function applyPremiumMultiplierOverrides(models: readonly ModelSpec[]): ModelSpec[] { return models.map(model => { const premiumMultiplier = COPILOT_PREMIUM_MULTIPLIERS[`${model.provider}/${model.id}`]; if (premiumMultiplier === undefined) { @@ -195,11 +200,11 @@ function applyPremiumMultiplierOverrides(models: readonly Model[]): Model[] { }; }); } -function hasBillableCost(cost: Model["cost"]): boolean { +function hasBillableCost(cost: ModelSpec["cost"]): boolean { return cost.input !== 0 || cost.output !== 0 || cost.cacheRead !== 0 || cost.cacheWrite !== 0; } -function applyCodexPricingFallback(models: readonly Model[]): Model[] { +function applyCodexPricingFallback(models: readonly ModelSpec[]): ModelSpec[] { const openAIModels = new Map( models .filter(model => model.provider === "openai" && hasBillableCost(model.cost)) @@ -234,7 +239,7 @@ function applyCodexPricingFallback(models: readonly Model[]): Model[] { * stale or inflated upstream value through. The resolver applies the same * cap when discovery runs at runtime; this is the bundle-time safety net. */ -function applyFireworksKimiMaxTokensCap(models: readonly Model[]): Model[] { +function applyFireworksKimiMaxTokensCap(models: readonly ModelSpec[]): ModelSpec[] { const FIREWORKS_KIMI_PROVIDERS = new Set(["fireworks", "firepass"]); return models.map(model => { if (!FIREWORKS_KIMI_PROVIDERS.has(model.provider)) return model; @@ -250,11 +255,11 @@ function applyFireworksKimiMaxTokensCap(models: readonly Model[]): Model[] { * `reasoning_effort` and rejects the DeepSeek-native binary `thinking` toggle * when both are present. Strip stale reference metadata from generated fallbacks. */ -function applyFireworksDeepSeekReasoningShape(models: readonly Model[]): Model[] { +function applyFireworksDeepSeekReasoningShape(models: readonly ModelSpec[]): ModelSpec[] { return models.map(model => { if (model.provider !== "fireworks" || model.api !== "openai-completions") return model; // `.api` equality doesn't narrow the generic; the guard makes this cast sound. - return stripFireworksDeepSeekThinkingToggle(model as Model<"openai-completions">, model.id); + return stripFireworksDeepSeekThinkingToggle(model as ModelSpec<"openai-completions">, model.id); }); } @@ -283,7 +288,7 @@ async function getOAuthAccessFromStorage(provider: OAuthProvider): Promise[]> { +async function fetchAntigravityModels(): Promise[]> { const access = await getOAuthAccessFromStorage("google-antigravity"); if (!access) { console.log("No Antigravity credentials found, will use previous models"); @@ -327,7 +332,7 @@ function extractCodexAccountId(accessToken: string): string | null { } } -async function fetchCodexDiscoveryModels(): Promise[]> { +async function fetchCodexDiscoveryModels(): Promise[]> { const access = await getOAuthAccessFromStorage("openai-codex"); if (!access) { return []; @@ -365,7 +370,8 @@ async function generateModels() { ).map(descriptor => fetchProviderModelsFromCatalog(descriptor as CatalogProviderDescriptor)), ) ).flat(); - const gitLabDuoModels = getGitLabDuoModels(); + // getGitLabDuoModels returns built models; project back to spec stage for the bundle. + const gitLabDuoModels = getGitLabDuoModels().map(model => toModelSpec(model)); // Combine models (models.dev has priority) let allModels = applyGlobalModelsDevFallback( [...modelsDevModels, ...catalogProviderModels, ...gitLabDuoModels], @@ -373,7 +379,7 @@ async function generateModels() { ); if (!allModels.some(model => model.provider === "cloudflare-ai-gateway")) { - allModels.push(CLOUDFLARE_FALLBACK_MODEL); + allModels.push(CLOUDFLARE_FALLBACK_MODEL as ModelSpec<"anthropic-messages">); } // xai-oauth has no upstream catalog source (not in models.dev or @@ -423,7 +429,7 @@ async function generateModels() { // Discovery-only providers (local inference servers) — never bundle static models. const fetchedKeys = new Set(allModels.map(model => `${model.provider}/${model.id}`)); - for (const models of Object.values(prevModelsJson as Record>)) { + for (const models of Object.values(prevModelsJson as Record>)) { for (const model of Object.values(models)) { if ( !fetchedKeys.has(`${model.provider}/${model.id}`) && @@ -444,7 +450,7 @@ async function generateModels() { linkOpenAIPromotionTargets(allModels); // Group by provider and sort each provider's models - const providers: Record> = {}; + const providers: Record> = {}; for (const model of allModels) { if (DISCOVERY_ONLY_PROVIDERS.has(model.provider)) continue; if (!providers[model.provider]) { @@ -466,7 +472,7 @@ async function generateModels() { ); }; - const MODELS: Record> = sortObj(providers); + const MODELS: Record> = sortObj(providers); for (const key in MODELS) { MODELS[key] = sortObj(MODELS[key]); } diff --git a/packages/catalog/src/build.ts b/packages/catalog/src/build.ts new file mode 100644 index 000000000..185240994 --- /dev/null +++ b/packages/catalog/src/build.ts @@ -0,0 +1,34 @@ +/** + * The single Model constructor. Thinking metadata and the resolved compat + * record are materialized here, exactly once per spec — request handlers read + * `model.compat` fields and perform zero URL parsing and zero compat + * allocation per request. + */ +import { buildAnthropicCompat } from "./compat/anthropic"; +import { buildOpenAICompat, buildOpenAIResponsesCompat } from "./compat/openai"; +import { enrichModelThinking } from "./model-thinking"; +import type { Api, CompatOf, Model, ModelSpec } from "./types"; + +export function buildModel(spec: ModelSpec): Model { + const enriched = enrichModelThinking(spec); + return { + ...enriched, + compat: buildCompat(enriched) as CompatOf, + compatConfig: enriched.compat, + } as Model; +} + +function buildCompat(spec: ModelSpec): CompatOf { + switch (spec.api) { + case "openai-completions": + return buildOpenAICompat(spec as ModelSpec<"openai-completions">); + case "openai-responses": + case "azure-openai-responses": + case "openai-codex-responses": + return buildOpenAIResponsesCompat(spec as ModelSpec<"openai-responses">); + case "anthropic-messages": + return buildAnthropicCompat(spec as ModelSpec<"anthropic-messages">); + default: + return undefined; + } +} diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts index 2e06c66ea..5c46545dc 100644 --- a/packages/catalog/src/compat/anthropic.ts +++ b/packages/catalog/src/compat/anthropic.ts @@ -1,74 +1,49 @@ /** - * Anthropic-messages compatibility detection and resolution — the - * anthropic-side analogue of `./openai`. Detect-time defaults come from - * provider ids, strict URL checks, and model-id classification; explicit - * `model.compat` overrides always win. + * Anthropic-messages compat builder — the anthropic-side analogue of + * `./openai`. Runs exactly once per model (from `buildModel`); detect-time + * defaults come from provider ids, strict host checks, and model-id + * classification, with explicit spec overrides assigned on top. */ +import { modelMatchesHost } from "../hosts"; import { isAnthropicFableOrMythosModel, supportsMidConversationSystemMessages } from "../model-thinking"; -import type { AnthropicCompat, Model } from "../types"; +import type { ModelSpec, ResolvedAnthropicCompat } from "../types"; +import { applyCompatOverrides } from "./apply"; + +const OFFICIAL_ANTHROPIC_URL = "https://api.anthropic.com"; /** - * Official first-party Anthropic API check (https + exact host). A missing - * baseUrl is official on purpose: request dispatch falls back to - * `https://api.anthropic.com`. Strict URL parsing (not substring) because the - * callers gate auth flows and body mutations on it. + * Official first-party Anthropic API. A missing baseUrl is official on purpose: + * request dispatch falls back to `https://api.anthropic.com`. This is the one + * auth-sensitive host check — OAuth credentials are attached based on it — so + * it requires the exact origin or a path boundary (`/`) after it; a bare + * prefix check would accept lookalikes like `https://api.anthropic.com.evil.com`. */ export function isOfficialAnthropicApiUrl(baseUrl?: string): boolean { if (!baseUrl) return true; - try { - const url = new URL(baseUrl); - return url.protocol.toLowerCase() === "https:" && url.hostname.toLowerCase() === "api.anthropic.com"; - } catch { - return false; - } + const lower = baseUrl.toLowerCase(); + return lower === OFFICIAL_ANTHROPIC_URL || lower.startsWith(`${OFFICIAL_ANTHROPIC_URL}/`); } -/** Z.AI's Anthropic-compatible proxy (`api.z.ai/api/anthropic`), strict-host matched. */ -function isZaiAnthropicUrl(baseUrl: string | undefined): boolean { - if (!baseUrl) return false; - try { - return new URL(baseUrl).hostname.toLowerCase() === "api.z.ai"; - } catch { - return false; - } -} - -/** DeepSeek-operated host, strict-host matched (`api.deepseek.com` or any `*.deepseek.com`). */ -function isDeepseekHostUrl(baseUrl: string | undefined): boolean { - if (!baseUrl) return false; - try { - const hostname = new URL(baseUrl).hostname.toLowerCase(); - return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com"); - } catch { - return false; - } -} - -export type ResolvedAnthropicCompat = Required; - -/** - * Detect anthropic-messages compatibility defaults from provider/baseUrl/model id. - * @param resolvedBaseUrl - Effective request base URL when it differs from - * `model.baseUrl` (e.g. an options-level override). - */ -export function detectAnthropicCompat( - model: Model<"anthropic-messages">, - resolvedBaseUrl?: string, -): ResolvedAnthropicCompat { - const baseUrl = resolvedBaseUrl ?? model.baseUrl; - const isZai = model.provider === "zai" || isZaiAnthropicUrl(baseUrl); - return { +/** Build the resolved anthropic-messages compat record for a model spec. */ +export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): ResolvedAnthropicCompat { + const baseUrl = spec.baseUrl; + const official = isOfficialAnthropicApiUrl(baseUrl); + // Z.AI's Anthropic-compatible proxy lives at `api.z.ai/api/anthropic`. + const isZai = modelMatchesHost(spec, "zai"); + const compat: ResolvedAnthropicCompat = { + officialEndpoint: official, disableStrictTools: false, disableAdaptiveThinking: false, supportsEagerToolInputStreaming: true, - supportsLongCacheRetention: true, + // Long cache retention is only sent to the official API by default; + // proxies opt in explicitly via `compat.supportsLongCacheRetention: true`. + supportsLongCacheRetention: official, // First-party Claude API only. Bedrock/Vertex/Foundry and other // Anthropic-compatible gateways reject mid-conversation system roles, so // detection requires the canonical api.anthropic.com host plus a // supported model id. - supportsMidConversationSystem: - isOfficialAnthropicApiUrl(model.baseUrl) && supportsMidConversationSystemMessages(model.id), - supportsForcedToolChoice: !isAnthropicFableOrMythosModel(model.id), + supportsMidConversationSystem: official && supportsMidConversationSystemMessages(spec.id), + supportsForcedToolChoice: !isAnthropicFableOrMythosModel(spec.id), // Z.AI workaround (issue #814): its proxy deserializes tool_result blocks // into a class that reads `.id`. requiresToolResultId: isZai, @@ -79,31 +54,8 @@ export function detectAnthropicCompat( // loses the reasoning chain and can destabilize the next tool-call // arguments (#2005). Known non-signing hosts (Z.AI, DeepSeek) are also // preserved for compatibility. - replayUnsignedThinking: - isZai || - model.provider === "deepseek" || - isDeepseekHostUrl(baseUrl) || - (model.reasoning && !isOfficialAnthropicApiUrl(baseUrl)), - }; -} - -/** Layer explicit `model.compat` overrides onto the detected anthropic defaults. */ -export function resolveAnthropicCompat( - model: Model<"anthropic-messages">, - resolvedBaseUrl?: string, -): ResolvedAnthropicCompat { - const detected = detectAnthropicCompat(model, resolvedBaseUrl); - const compat = model.compat; - if (!compat) return detected; - return { - disableStrictTools: compat.disableStrictTools ?? detected.disableStrictTools, - disableAdaptiveThinking: compat.disableAdaptiveThinking ?? detected.disableAdaptiveThinking, - supportsEagerToolInputStreaming: - compat.supportsEagerToolInputStreaming ?? detected.supportsEagerToolInputStreaming, - supportsLongCacheRetention: compat.supportsLongCacheRetention ?? detected.supportsLongCacheRetention, - supportsMidConversationSystem: compat.supportsMidConversationSystem ?? detected.supportsMidConversationSystem, - supportsForcedToolChoice: compat.supportsForcedToolChoice ?? detected.supportsForcedToolChoice, - requiresToolResultId: compat.requiresToolResultId ?? detected.requiresToolResultId, - replayUnsignedThinking: compat.replayUnsignedThinking ?? detected.replayUnsignedThinking, + replayUnsignedThinking: isZai || modelMatchesHost(spec, "deepseekFamily") || (spec.reasoning && !official), }; + applyCompatOverrides(compat, spec.compat); + return compat; } diff --git a/packages/catalog/src/compat/apply.ts b/packages/catalog/src/compat/apply.ts new file mode 100644 index 000000000..4655435e2 --- /dev/null +++ b/packages/catalog/src/compat/apply.ts @@ -0,0 +1,15 @@ +/** + * Assign defined override values onto a freshly-built resolved compat record, + * in place. Keys the record doesn't declare are ignored (loosely-typed config + * may carry junk). `buildModel` is the only intended caller — the record being + * mutated is the single per-model allocation; nothing here runs per request. + */ +export function applyCompatOverrides(compat: object, overrides: object | undefined): void { + if (!overrides) return; + for (const key in overrides) { + const value = (overrides as Record)[key]; + if (value !== undefined && key in compat) { + (compat as Record)[key] = value; + } + } +} diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index 8215d9074..9541c52e8 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -1,3 +1,12 @@ +/** + * OpenAI-API compat builders — chat-completions and Responses flavors. + * + * `buildOpenAICompat`/`buildOpenAIResponsesCompat` run exactly once per model + * (from `buildModel`): detection writes a fresh record, sparse spec overrides + * are assigned onto it in place, and conditional policies are materialized as + * complete alternate views. Request handlers read `model.compat` fields and + * never detect, resolve, or allocate. + */ import { hostMatchesUrl, modelMatchesHost } from "../hosts"; import { isAnthropicNamespacedModelId, @@ -8,32 +17,10 @@ import { isMimoModelIdOrName, isQwenModelId, } from "../identity/family"; -import type { Model, OpenAICompat } from "../types"; +import type { ModelSpec, OpenAICompat, ResolvedOpenAICompat, ResolvedOpenAIResponsesCompat } from "../types"; +import { applyCompatOverrides } from "./apply"; type OpenAIReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh"; -type ResolvedToolStrictMode = NonNullable | "mixed"; - -export type ResolvedOpenAICompat = Required< - Omit< - OpenAICompat, - | "openRouterRouting" - | "vercelGatewayRouting" - | "extraBody" - | "toolStrictMode" - | "streamIdleTimeoutMs" - | "supportsLongPromptCacheRetention" - | "cacheControlFormat" - | "thinkingKeep" - > -> & { - openRouterRouting?: OpenAICompat["openRouterRouting"]; - vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"]; - extraBody?: OpenAICompat["extraBody"]; - cacheControlFormat?: OpenAICompat["cacheControlFormat"]; - thinkingKeep?: OpenAICompat["thinkingKeep"]; - streamIdleTimeoutMs?: number; - toolStrictMode: ResolvedToolStrictMode; -}; /** GLM coding-plan SKUs idle for minutes mid-reasoning; see `streamIdleTimeoutMs`. */ const GLM_CODING_PLAN_MODEL_PATTERN = /^glm-5(?:[.-]|$)/i; @@ -41,6 +28,27 @@ const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; /** Direct DeepSeek reasoning models stall between thinking and answer phases. */ const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; +/** + * OpenCode's gateways (https://opencode.ai/zen|go) gate `reasoning_content` + * on the request's thinking state for every model they front (Kimi K2.x, + * DeepSeek V4, GLM-5.x, Qwen3.x, MiMo, MiniMax, …): they 400 with `Extra + * inputs are not permitted` when thinking is off but the field is supplied + * (#1071), and 400 with `thinking is enabled but reasoning_content is missing + * in assistant tool call message at index N` (#1484) when thinking is on and + * the field is absent. The base compat therefore leaves the replay off, and + * this `whenThinking` policy reactivates it for thinking-engaged requests. + * `allowsSyntheticReasoningContentForToolCalls` is forced to `false` on the + * same path: the gateway specifically requires `reasoning_content`, and the + * synthetic-friendly default would echo whichever field the upstream streamed + * (e.g. `reasoning` for many opencode turns), landing the replay in the wrong + * key and re-triggering the 400. + */ +const OPENCODE_WHEN_THINKING: NonNullable = { + requiresReasoningContentForToolCalls: true, + allowsSyntheticReasoningContentForToolCalls: false, + reasoningContentField: "reasoning_content", +}; + function detectStrictModeSupport(provider: string, baseUrl: string): boolean { if ( provider === "openai" || @@ -92,50 +100,46 @@ function getOpenRouterAnthropicReasoningEffortMap( } /** - * Detect compatibility settings from provider and baseUrl for known providers. + * Build the resolved chat-completions compat record for a model spec. * Provider takes precedence over URL-based detection since it's explicitly configured. - * @param model - The model configuration - * @param resolvedBaseUrl - Optional resolved base URL (e.g., after GitHub Copilot proxy-ep resolution). - * If provided, this takes precedence over model.baseUrl for URL-based checks. */ -export function detectOpenAICompat(model: Model<"openai-completions">, resolvedBaseUrl?: string): ResolvedOpenAICompat { - const provider = model.provider; - // Use resolvedBaseUrl if provided (e.g., after GitHub Copilot proxy-ep resolution) - const baseUrl = resolvedBaseUrl ?? model.baseUrl; +export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): ResolvedOpenAICompat { + const provider = spec.provider; + const baseUrl = spec.baseUrl; const hostModel = { provider, baseUrl }; const isCerebras = modelMatchesHost(hostModel, "cerebras"); const isZai = modelMatchesHost(hostModel, "zai"); const isZhipu = modelMatchesHost(hostModel, "zhipu"); const isKilo = modelMatchesHost(hostModel, "kilo"); - const isKimiModel = isKimiModelId(model.id); + const isKimiModel = isKimiModelId(spec.id); const isMoonshotKimi = isKimiModel && modelMatchesHost(hostModel, "moonshotNative"); - const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(model.id); + const usesMoonshotKimiPreservedThinking = isMoonshotKimi && isKimiK26ModelId(spec.id); const isAnthropicModel = - modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(model.id) || isAnthropicNamespacedModelId(model.id); + modelMatchesHost(hostModel, "anthropic") || isClaudeModelId(spec.id) || isAnthropicNamespacedModelId(spec.id); const isAlibaba = modelMatchesHost(hostModel, "alibabaDashscope"); - const isQwen = isQwenModelId(model.id); + const isQwen = isQwenModelId(spec.id); // DeepSeek V4 (and other reasoning-capable DeepSeek models) reject follow-up requests in // thinking mode unless prior assistant tool-call turns include `reasoning_content`. The // upstream model is reachable through many OpenAI-compat hosts (api.deepseek.com, Deepinfra, // Kilo, NVIDIA NIM, Zenmux, OpenRouter, …), so we match by model id/name as well as by - // provider/baseUrl. The flag is gated by `model.reasoning` because the invariant only + // provider/baseUrl. The flag is gated by `spec.reasoning` because the invariant only // applies when thinking mode is actually engaged. - const lowerId = model.id.toLowerCase(); - const lowerName = (model.name ?? "").toLowerCase(); + const lowerId = spec.id.toLowerCase(); + const lowerName = (spec.name ?? "").toLowerCase(); const isXiaomiHost = modelMatchesHost(hostModel, "xiaomi"); - const isXiaomiMimo = isXiaomiHost && (isMimoModelIdOrName(model.id) || isMimoModelIdOrName(model.name ?? "")); + const isXiaomiMimo = isXiaomiHost && (isMimoModelIdOrName(spec.id) || isMimoModelIdOrName(spec.name ?? "")); // OpenCode Zen's `big-pickle` is a DeepSeek reasoning alias; the upstream // 400s come from DeepSeek and require exact reasoning_content replay. const isOpenCodeDeepseekAlias = provider === "opencode-zen" && (lowerId === "big-pickle" || lowerName === "big pickle"); const isDeepseekFamily = modelMatchesHost(hostModel, "deepseekFamily") || - isDeepseekModelIdOrName(model.id) || - isDeepseekModelIdOrName(model.name ?? "") || + isDeepseekModelIdOrName(spec.id) || + isDeepseekModelIdOrName(spec.name ?? "") || isOpenCodeDeepseekAlias; const isDirectDeepseekApi = modelMatchesHost(hostModel, "deepseekDirect"); - const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(model.reasoning); + const isDirectDeepseekReasoning = isDirectDeepseekApi && isDeepseekFamily && Boolean(spec.reasoning); const isGrok = modelMatchesHost(hostModel, "xai"); const isMistral = modelMatchesHost(hostModel, "mistral"); const isOpenCodeHost = modelMatchesHost(hostModel, "opencode"); @@ -166,6 +170,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB const isOpenAIHost = modelMatchesHost(hostModel, "openai"); const isAzureHost = modelMatchesHost(hostModel, "azureOpenAI"); const isOpenRouter = modelMatchesHost(hostModel, "openrouter"); + const isVercelGateway = modelMatchesHost(hostModel, "vercelAIGateway"); const isTogether = modelMatchesHost(hostModel, "together"); const isFireworks = hostMatchesUrl(baseUrl, "fireworks"); const isGroqHost = modelMatchesHost(hostModel, "groq"); @@ -200,8 +205,8 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB const openRouterAnthropicReasoningEffortMap = isOpenRouter ? getOpenRouterAnthropicReasoningEffortMap(lowerId) : undefined; - const reasoningEffortMap: NonNullable = - provider === "groq" && model.id === "qwen/qwen3-32b" + const detectedReasoningEffortMap: NonNullable = + provider === "groq" && spec.id === "qwen/qwen3-32b" ? ({ minimal: "default", low: "default", @@ -209,7 +214,7 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB high: "default", xhigh: "default", } satisfies Partial>) - : isDeepseekFamily && model.reasoning + : isDeepseekFamily && spec.reasoning ? ({ minimal: "high", low: "high", @@ -231,13 +236,13 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB // models idle for minutes mid-reasoning; widen the idle timeout so warm-ups // stop aborting and retrying. const streamIdleTimeoutMs = - GLM_CODING_PLAN_MODEL_PATTERN.test(model.id) && (isZai || isZhipu) + GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu) ? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS - : model.reasoning && isDirectDeepseekApi + : spec.reasoning && isDirectDeepseekApi ? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS : undefined; - return { + const compat: ResolvedOpenAICompat = { supportsStore: !isNonStandard, // `developer` is an OpenAI-Responses-era extension to the chat-completions schema. Almost // every OpenAI-compatible host other than OpenAI itself (and Azure OpenAI, which mirrors @@ -248,10 +253,19 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB supportsDeveloperRole: isOpenAIHost || isAzureHost, supportsMultipleSystemMessages: supportsMultipleSystemMessagesDefault, supportsReasoningEffort: !isGrok && !isZai && !isZhipu && !isXiaomiMimo, - reasoningEffortMap, + // GitHub Copilot's chat-completions endpoint rejects reasoning params wholesale. + supportsReasoningParams: provider !== "github-copilot", + reasoningEffortMap: detectedReasoningEffortMap, supportsUsageInStreaming: !isCerebras, + // Kimi (including via OpenRouter and Fireworks router-form IDs such as + // `accounts/fireworks/routers/kimi-*`) calculates TPM rate limits based on + // max_tokens, not actual output. The official Kimi K2 model guidance + // (https://docs.fireworks.ai/models/kimi-k2) also requires `max_tokens` for + // every call since the family can otherwise emit very long reasoning traces + // before the final answer. + alwaysSendMaxTokens: isKimiModel, disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel, - disableReasoningOnToolChoice: isDeepseekFamily && Boolean(model.reasoning) && !isOpenRouter, + disableReasoningOnToolChoice: isDeepseekFamily && Boolean(spec.reasoning) && !isOpenRouter, supportsToolChoice: !isDirectDeepseekReasoning, maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens", requiresToolResultName: isMistral, @@ -284,132 +298,79 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB // Anthropic's redacted/encrypted reasoning into provider-native plaintext, // so cross-provider continuations rely on a placeholder. // OpenCode Kimi aliases handle reasoning content internally and reject - // client-sent `reasoning_content`, so exclude only that Kimi-on-OpenCode path. + // client-sent `reasoning_content`, so exclude only that Kimi-on-OpenCode path + // (the `whenThinking` policy below re-enables the replay for thinking turns). requiresReasoningContentForToolCalls: (isKimiModel && !isOpenCodeProvider) || - (isDeepseekFamily && Boolean(model.reasoning)) || + (isDeepseekFamily && Boolean(spec.reasoning)) || isXiaomiMimo || - (isOpenRouter && Boolean(model.reasoning)), + (isOpenRouter && Boolean(spec.reasoning)), // DeepSeek V4 and Xiaomi MiMo reject synthetic reasoning_content placeholders (".") on tool-call turns. // Kimi and OpenRouter accept them when actual reasoning is unavailable. - allowsSyntheticReasoningContentForToolCalls: (!isDeepseekFamily || !model.reasoning) && !isXiaomiMimo, + allowsSyntheticReasoningContentForToolCalls: (!isDeepseekFamily || !spec.reasoning) && !isXiaomiMimo, requiresAssistantContentForToolCalls: isKimiModel || isDirectDeepseekReasoning, - cacheControlFormat: isOpenRouter && model.id.startsWith("anthropic/") ? "anthropic" : undefined, + cacheControlFormat: isOpenRouter && spec.id.startsWith("anthropic/") ? "anthropic" : undefined, openRouterRouting: undefined, vercelGatewayRouting: undefined, + isOpenRouterHost: isOpenRouter, + isVercelGatewayHost: isVercelGateway, supportsStrictMode: detectStrictModeSupport(provider, baseUrl), extraBody: isDirectDeepseekReasoning ? { thinking: { type: "enabled" } } : undefined, toolStrictMode: isCerebras ? "all_strict" : "mixed", streamIdleTimeoutMs, }; -} -/** - * Resolve compatibility settings by layering explicit model.compat overrides onto - * the detected defaults. This is the canonical compat view for both metadata and transport. - * @param model - The model configuration - * @param resolvedBaseUrl - Optional resolved base URL (e.g., after GitHub Copilot proxy-ep resolution). - * If provided, this takes precedence over model.baseUrl for URL-based checks. - */ -export function resolveOpenAICompat( - model: Model<"openai-completions">, - resolvedBaseUrl?: string, -): ResolvedOpenAICompat { - const detected = detectOpenAICompat(model, resolvedBaseUrl); - if (!model.compat) { - return detected; + applyCompatOverrides(compat, spec.compat); + if (spec.compat?.reasoningEffortMap) { + // Effort maps merge per level instead of replacing wholesale. + compat.reasoningEffortMap = { ...detectedReasoningEffortMap, ...spec.compat.reasoningEffortMap }; } - return { - supportsStore: model.compat.supportsStore ?? detected.supportsStore, - supportsDeveloperRole: model.compat.supportsDeveloperRole ?? detected.supportsDeveloperRole, - supportsMultipleSystemMessages: - model.compat.supportsMultipleSystemMessages ?? detected.supportsMultipleSystemMessages, - supportsReasoningEffort: model.compat.supportsReasoningEffort ?? detected.supportsReasoningEffort, - reasoningEffortMap: { ...detected.reasoningEffortMap, ...(model.compat.reasoningEffortMap ?? {}) }, - supportsUsageInStreaming: model.compat.supportsUsageInStreaming ?? detected.supportsUsageInStreaming, - supportsToolChoice: model.compat.supportsToolChoice ?? detected.supportsToolChoice, - maxTokensField: model.compat.maxTokensField ?? detected.maxTokensField, - requiresToolResultName: model.compat.requiresToolResultName ?? detected.requiresToolResultName, - requiresAssistantAfterToolResult: - model.compat.requiresAssistantAfterToolResult ?? detected.requiresAssistantAfterToolResult, - requiresThinkingAsText: model.compat.requiresThinkingAsText ?? detected.requiresThinkingAsText, - requiresMistralToolIds: model.compat.requiresMistralToolIds ?? detected.requiresMistralToolIds, - thinkingFormat: model.compat.thinkingFormat ?? detected.thinkingFormat, - thinkingKeep: model.compat.thinkingKeep ?? detected.thinkingKeep, - reasoningContentField: model.compat.reasoningContentField ?? detected.reasoningContentField, - requiresReasoningContentForToolCalls: - model.compat.requiresReasoningContentForToolCalls ?? detected.requiresReasoningContentForToolCalls, - allowsSyntheticReasoningContentForToolCalls: - model.compat.allowsSyntheticReasoningContentForToolCalls ?? - detected.allowsSyntheticReasoningContentForToolCalls, - requiresAssistantContentForToolCalls: - model.compat.requiresAssistantContentForToolCalls ?? detected.requiresAssistantContentForToolCalls, - cacheControlFormat: model.compat.cacheControlFormat ?? detected.cacheControlFormat, - disableReasoningOnForcedToolChoice: - model.compat.disableReasoningOnForcedToolChoice ?? detected.disableReasoningOnForcedToolChoice, - disableReasoningOnToolChoice: model.compat.disableReasoningOnToolChoice ?? detected.disableReasoningOnToolChoice, - openRouterRouting: model.compat.openRouterRouting ?? detected.openRouterRouting, - vercelGatewayRouting: model.compat.vercelGatewayRouting ?? detected.vercelGatewayRouting, - supportsStrictMode: model.compat.supportsStrictMode ?? detected.supportsStrictMode, - extraBody: model.compat.extraBody ?? detected.extraBody, - toolStrictMode: model.compat.toolStrictMode ?? detected.toolStrictMode, - streamIdleTimeoutMs: model.compat.streamIdleTimeoutMs ?? detected.streamIdleTimeoutMs, - }; + const whenThinkingPolicy = + spec.compat?.whenThinking ?? (isOpenCodeProvider && spec.reasoning ? OPENCODE_WHEN_THINKING : undefined); + if (whenThinkingPolicy) { + const variant: ResolvedOpenAICompat = { ...compat }; + applyCompatOverrides(variant, whenThinkingPolicy); + compat.whenThinking = variant; + } + + return compat; } -/** Resolved Responses-API compatibility view (see `detectOpenAIResponsesCompat`). */ -export interface ResolvedOpenAIResponsesCompat { - supportsDeveloperRole: boolean; - supportsStrictMode: boolean; - supportsLongPromptCacheRetention: boolean; +interface OpenAIResponsesSpecLike { + provider: string; + baseUrl: string; + compat?: OpenAICompat; } /** - * Detect Responses-API compatibility from provider/baseUrl. The Responses - * flavor deliberately differs from chat-completions: GitHub Copilot's - * responses endpoint accepts the `developer` role, while strict tool mode is - * scoped to first-party OpenAI/Azure/Copilot providers. Developer-role and - * prompt-cache detection are URL-only on purpose — the historical call sites - * never consulted the provider id for them. + * Build the resolved Responses-API compat record. The Responses flavor + * deliberately differs from chat-completions: GitHub Copilot's responses + * endpoint accepts the `developer` role, while strict tool mode is scoped to + * first-party OpenAI/Azure/Copilot providers. Developer-role and prompt-cache + * detection are URL-only on purpose — the historical call sites never + * consulted the provider id for them. */ -export function detectOpenAIResponsesCompat( - model: { provider: string; baseUrl: string }, - resolvedBaseUrl?: string, -): ResolvedOpenAIResponsesCompat { - const baseUrl = resolvedBaseUrl ?? model.baseUrl ?? ""; - return { +export function buildOpenAIResponsesCompat(spec: OpenAIResponsesSpecLike): ResolvedOpenAIResponsesCompat { + const baseUrl = spec.baseUrl ?? ""; + const compat: ResolvedOpenAIResponsesCompat = { supportsDeveloperRole: hostMatchesUrl(baseUrl, "openai") || hostMatchesUrl(baseUrl, "azureOpenAI") || hostMatchesUrl(baseUrl, "githubCopilot"), supportsStrictMode: - model.provider === "openai" || - model.provider === "azure" || - model.provider === "github-copilot" || + spec.provider === "openai" || + spec.provider === "azure" || + spec.provider === "github-copilot" || hostMatchesUrl(baseUrl, "openai") || hostMatchesUrl(baseUrl, "azureOpenAI"), + supportsReasoningEffort: true, supportsLongPromptCacheRetention: hostMatchesUrl(baseUrl, "openai"), + // Azure OpenAI and GitHub Copilot Responses paths require tool results + // to strictly match prior tool calls when building Responses inputs. + strictResponsesPairing: hostMatchesUrl(baseUrl, "azureOpenAI") || spec.provider === "github-copilot", + reasoningEffortMap: {}, }; -} - -/** - * Resolve Responses-API compatibility by layering explicit `model.compat` - * overrides onto the detected defaults — the Responses-side analogue of - * `resolveOpenAICompat`. Models bundled with `supportsDeveloperRole: false` - * (codex-mini-style SKUs) take effect here. - */ -export function resolveOpenAIResponsesCompat( - model: { provider: string; baseUrl: string; compat?: OpenAICompat }, - resolvedBaseUrl?: string, -): ResolvedOpenAIResponsesCompat { - const detected = detectOpenAIResponsesCompat(model, resolvedBaseUrl); - const compat = model.compat; - if (!compat) return detected; - return { - supportsDeveloperRole: compat.supportsDeveloperRole ?? detected.supportsDeveloperRole, - supportsStrictMode: compat.supportsStrictMode ?? detected.supportsStrictMode, - supportsLongPromptCacheRetention: - compat.supportsLongPromptCacheRetention ?? detected.supportsLongPromptCacheRetention, - }; + applyCompatOverrides(compat, spec.compat); + return compat; } diff --git a/packages/catalog/src/discovery/antigravity.ts b/packages/catalog/src/discovery/antigravity.ts index 8fea6663a..a27fc65a8 100644 --- a/packages/catalog/src/discovery/antigravity.ts +++ b/packages/catalog/src/discovery/antigravity.ts @@ -1,5 +1,5 @@ import * as z from "zod/v4"; -import type { Model } from "../types"; +import type { ModelSpec } from "../types"; import { toPositiveNumber } from "../utils"; import { getAntigravityUserAgent } from "../wire/gemini-headers"; @@ -172,7 +172,7 @@ export interface FetchAntigravityDiscoveryModelsOptions { */ export async function fetchAntigravityDiscoveryModels( options: FetchAntigravityDiscoveryModelsOptions, -): Promise[] | null> { +): Promise[] | null> { const fetcher = options.fetcher ?? fetch; const endpoints = options.endpoint ? [trimTrailingSlashes(options.endpoint)] @@ -211,7 +211,7 @@ export async function fetchAntigravityDiscoveryModels( continue; } - const models: Model<"google-gemini-cli">[] = []; + const models: ModelSpec<"google-gemini-cli">[] = []; for (const [modelId, model] of Object.entries(parsed.models ?? {})) { if (ANTIGRAVITY_DISCOVERY_DENYLIST.has(modelId)) { diff --git a/packages/catalog/src/discovery/codex.ts b/packages/catalog/src/discovery/codex.ts index 9a61c4d5e..18a8f59db 100644 --- a/packages/catalog/src/discovery/codex.ts +++ b/packages/catalog/src/discovery/codex.ts @@ -1,5 +1,5 @@ import * as z from "zod/v4"; -import type { Model } from "../types"; +import type { ModelSpec } from "../types"; import { isRecord } from "../utils"; import { CODEX_BASE_URL, OPENAI_HEADER_VALUES, OPENAI_HEADERS } from "../wire/codex"; @@ -40,7 +40,7 @@ const codexModelsResponseSchema = z type CodexModelEntry = z.infer; interface NormalizedCodexModel { - model: Model<"openai-codex-responses">; + model: ModelSpec<"openai-codex-responses">; priority: number; } @@ -72,7 +72,7 @@ export interface CodexModelDiscoveryOptions { * Normalized Codex discovery response. */ export interface CodexModelDiscoveryResult { - models: Model<"openai-codex-responses">[]; + models: ModelSpec<"openai-codex-responses">[]; etag?: string; } @@ -215,7 +215,7 @@ function isAbortError(error: unknown): error is Error { return error instanceof Error && error.name === "AbortError"; } -function normalizeCodexModels(payload: unknown, baseUrl: string): Model<"openai-codex-responses">[] | null { +function normalizeCodexModels(payload: unknown, baseUrl: string): ModelSpec<"openai-codex-responses">[] | null { const parsedResponse = codexModelsResponseSchema.safeParse(payload); if (!parsedResponse.success) { return null; diff --git a/packages/catalog/src/discovery/cursor.ts b/packages/catalog/src/discovery/cursor.ts index 51664de50..a078cb0bc 100644 --- a/packages/catalog/src/discovery/cursor.ts +++ b/packages/catalog/src/discovery/cursor.ts @@ -2,7 +2,8 @@ import * as http2 from "node:http2"; import { create, fromBinary, toBinary } from "@bufbuild/protobuf"; import * as z from "zod/v4"; import { getBundledModels } from "../models"; -import type { Model } from "../types"; +import { toModelSpec } from "../provider-models/bundled-references"; +import type { Model, ModelSpec } from "../types"; import { GetUsableModelsRequestSchema, GetUsableModelsResponseSchema } from "./cursor-gen/agent_pb"; const CURSOR_DEFAULT_BASE_URL = "https://api2.cursor.sh"; @@ -58,7 +59,7 @@ export interface CursorModelDiscoveryOptions { */ export async function fetchCursorUsableModels( options: CursorModelDiscoveryOptions, -): Promise[] | null> { +): Promise[] | null> { const timeoutMs = options.timeoutMs ?? 5_000; try { const requestPayload = create(GetUsableModelsRequestSchema, { @@ -169,10 +170,10 @@ function normalizeCustomModelIds(customModelIds: readonly string[] | undefined): return [...normalized]; } -function createCursorReferenceMap(): Map> { - const references = new Map>(); - for (const model of getBundledModels("cursor") as Model<"cursor-agent">[]) { - references.set(model.id, model); +function createCursorReferenceMap(): Map> { + const references = new Map>(); + for (const model of getBundledModels("cursor")) { + references.set(model.id, toModelSpec(model as Model<"cursor-agent">)); } return references; } @@ -230,13 +231,13 @@ function decodeConnectUnaryBody(payload: Uint8Array): Uint8Array | null { function normalizeCursorModels( models: readonly unknown[] | undefined, baseUrlOverride: string | undefined, - references: Map>, -): Model<"cursor-agent">[] { + references: Map>, +): ModelSpec<"cursor-agent">[] { if (!models || models.length === 0) { return []; } - const byId = new Map>(); + const byId = new Map>(); for (const model of models) { const normalized = normalizeCursorModel(model, baseUrlOverride, references); if (!normalized) { @@ -251,8 +252,8 @@ function normalizeCursorModels( function normalizeCursorModel( model: unknown, baseUrlOverride: string | undefined, - references: Map>, -): Model<"cursor-agent"> | null { + references: Map>, +): ModelSpec<"cursor-agent"> | null { const parsedModel = CursorModelDetailsSchema.safeParse(model); if (!parsedModel.success) { return null; diff --git a/packages/catalog/src/discovery/gemini.ts b/packages/catalog/src/discovery/gemini.ts index 1bc5c4f0e..a7f59e2bd 100644 --- a/packages/catalog/src/discovery/gemini.ts +++ b/packages/catalog/src/discovery/gemini.ts @@ -1,7 +1,8 @@ import * as z from "zod/v4"; import { getBundledModels } from "../models"; +import { toModelSpec } from "../provider-models/bundled-references"; import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "../provider-models/discovery-constants"; -import type { FetchImpl, Model } from "../types"; +import type { FetchImpl, Model, ModelSpec } from "../types"; const GOOGLE_GENERATIVE_AI_BASE_URL = "https://generativelanguage.googleapis.com/v1beta"; const DEFAULT_PAGE_SIZE = 100; @@ -63,7 +64,7 @@ export interface GeminiDiscoveryOptions { */ export async function fetchGeminiModels( options: GeminiDiscoveryOptions, -): Promise[] | null> { +): Promise[] | null> { if (!options.apiKey.trim()) { return null; } @@ -74,9 +75,9 @@ export async function fetchGeminiModels( const maxPages = normalizePositiveInt(options.maxPages, DEFAULT_MAX_PAGES); const bundledById = new Map( - getBundledModels("google").map(model => [model.id, model as Model<"google-generative-ai">]), + getBundledModels("google").map(model => [model.id, toModelSpec(model as Model<"google-generative-ai">)]), ); - const modelsById = new Map>(); + const modelsById = new Map>(); const seenTokens = new Set(); let nextPageToken: string | undefined; @@ -166,8 +167,8 @@ function normalizePageToken(value: unknown): string | undefined { function normalizeModel( item: GeminiModelListItem, baseUrl: string, - bundledById: Map>, -): Model<"google-generative-ai"> | null { + bundledById: Map>, +): ModelSpec<"google-generative-ai"> | null { const id = normalizeModelId(item.name); if (!id) { return null; diff --git a/packages/catalog/src/discovery/openai-compatible.ts b/packages/catalog/src/discovery/openai-compatible.ts index 2f2341520..a567d7b36 100644 --- a/packages/catalog/src/discovery/openai-compatible.ts +++ b/packages/catalog/src/discovery/openai-compatible.ts @@ -1,6 +1,6 @@ import * as z from "zod/v4"; import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "../provider-models/discovery-constants"; -import type { Api, FetchImpl, Model, Provider } from "../types"; +import type { Api, FetchImpl, ModelSpec, Provider } from "../types"; const MODELS_PATH = "/models"; @@ -86,16 +86,16 @@ export interface FetchOpenAICompatibleModelsOptions { * Optional post-normalization filter. * Return false to skip a model. */ - filterModel?: (entry: OpenAICompatibleModelRecord, model: Model) => boolean; + filterModel?: (entry: OpenAICompatibleModelRecord, model: ModelSpec) => boolean; /** * Optional mapper override for provider-specific quirks. * Return null to skip a model. */ mapModel?: ( entry: OpenAICompatibleModelRecord, - defaults: Model, + defaults: ModelSpec, context: OpenAICompatibleModelMapperContext, - ) => Model | null; + ) => ModelSpec | null; } /** @@ -106,7 +106,7 @@ export interface FetchOpenAICompatibleModelsOptions { */ export async function fetchOpenAICompatibleModels( options: FetchOpenAICompatibleModelsOptions, -): Promise[] | null> { +): Promise[] | null> { const baseUrl = normalizeBaseUrl(options.baseUrl); if (!baseUrl) { return null; @@ -154,9 +154,9 @@ export async function fetchOpenAICompatibleModels( baseUrl, }; - const deduped = new Map>(); + const deduped = new Map>(); for (const entry of entries) { - const defaults: Model = { + const defaults: ModelSpec = { id: entry.id, name: typeof entry.name === "string" && entry.name.length > 0 ? entry.name : entry.id, api: options.api, diff --git a/packages/catalog/src/hosts.ts b/packages/catalog/src/hosts.ts index af79151bd..7e19a2d92 100644 --- a/packages/catalog/src/hosts.ts +++ b/packages/catalog/src/hosts.ts @@ -6,8 +6,9 @@ * Markers are case-insensitive substrings matched against the base URL, NOT * parsed hostnames: proxies regularly embed the upstream host in a path * segment, and the historical call sites all used substring semantics. - * Callers needing strict hostname matching (e.g. guards before request-body - * mutation) should keep their own `new URL().hostname` checks. + * Callers that need strict hostname matching — where a substring false + * positive is dangerous, e.g. the Anthropic official-endpoint OAuth gate — + * parse the URL and compare the hostname themselves. */ interface HostClassSpec { @@ -17,6 +18,9 @@ interface HostClassSpec { readonly providerPrefixes?: readonly string[]; /** Case-insensitive substrings matched against the base URL. */ readonly urlMarkers: readonly string[]; + // Strict hostname matching is intentionally not modeled here: the one + // auth-sensitive consumer (Anthropic official-endpoint) parses the URL + // itself; every other call site is benign and uses substring matching. } export const KNOWN_HOSTS = { diff --git a/packages/catalog/src/model-cache.ts b/packages/catalog/src/model-cache.ts index 07bd8e2cf..b6aa5597b 100644 --- a/packages/catalog/src/model-cache.ts +++ b/packages/catalog/src/model-cache.ts @@ -4,8 +4,11 @@ */ import { Database } from "bun:sqlite"; import { getModelDbPath } from "@oh-my-pi/pi-utils"; -import type { Api, Model } from "./types"; +import type { Api, Model, ModelSpec } from "./types"; +// Rows persist ModelSpec JSON (sparse `compat`, never the resolved record); +// the model manager rebuilds via `buildModel` on load. v3 rows predating the +// resolved-compat redesign already carried sparse compat, so they stay valid. const CACHE_SCHEMA_VERSION = 3; interface CacheRow { @@ -22,7 +25,7 @@ interface TableInfoRow { } interface CacheEntry { - models: Model[]; + models: ModelSpec[]; fresh: boolean; authoritative: boolean; updatedAt: number; @@ -86,7 +89,7 @@ export function readModelCache( if (!row || row.version !== CACHE_SCHEMA_VERSION) { return null; } - const models = JSON.parse(row.models) as Model[]; + const models = JSON.parse(row.models) as ModelSpec[]; const ageMs = now() - row.updated_at; const fresh = Number.isFinite(ageMs) && ageMs >= 0 && ageMs <= ttlMs; return { @@ -120,7 +123,7 @@ export function writeModelCache( updatedAt, authoritative ? 1 : 0, staticFingerprint, - JSON.stringify(models), + JSON.stringify(models.map(model => ({ ...model, compat: model.compatConfig, compatConfig: undefined }))), ], ); } catch { diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index 7d68d3f5f..111f03c91 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -1,7 +1,7 @@ +import { buildModel } from "./build"; import { readModelCache, writeModelCache } from "./model-cache"; -import { enrichModelThinking } from "./model-thinking"; import { type GeneratedProvider, getBundledModels } from "./models"; -import type { Api, Model, Provider } from "./types"; +import type { Api, Model, ModelSpec, Provider } from "./types"; import { isRecord } from "./utils"; const DEFAULT_CACHE_TTL_MS = 2 * 60 * 60 * 1000; @@ -19,7 +19,7 @@ export interface ModelsDevFallback { /** Fetches raw fallback payload (for example from models.dev). */ fetch(): Promise; /** Maps payload into provider models. */ - map(payload: TPayload, providerId: Provider): readonly Model[]; + map(payload: TPayload, providerId: Provider): readonly ModelSpec[]; } /** @@ -29,7 +29,7 @@ export interface ModelManagerOptions[]; + staticModels?: readonly ModelSpec[]; /** Optional override for the cache database path. Default: /models.db. */ cacheDbPath?: string; /** Maximum cache age in milliseconds before considered stale. Default: 24h. */ @@ -37,7 +37,7 @@ export interface ModelManagerOptions Promise[] | null>; + fetchDynamicModels?: () => Promise[] | null>; /** Optional models.dev fallback hook. */ modelsDev?: ModelsDevFallback; /** Clock override for deterministic tests. */ @@ -78,8 +78,9 @@ export function createModelManager(value: unknown): Model[] { if (!Array.isArray(value)) { @@ -90,7 +91,7 @@ function passModelList(value: unknown): Model[] { if (item === null || typeof item !== "object" || typeof (item as { id: unknown }).id !== "string") { continue; } - out.push(enrichModelThinking(item as Model)); + out.push(buildModel(item as ModelSpec)); } return out; } @@ -108,9 +109,9 @@ export async function resolveProviderModels( - options.staticModels ?? getBundledModels(options.providerId as GeneratedProvider), - ); + const staticModels = options.staticModels + ? passModelList(options.staticModels) + : (getBundledModels(options.providerId as GeneratedProvider) as Model[]); const cache = readModelCache(options.providerId, ttlMs, now, dbPath); const dynamicModelsAuthoritative = options.dynamicModelsAuthoritative ?? false; const staticFingerprint = fingerprintStatic(staticModels, dynamicModelsAuthoritative); @@ -196,7 +197,7 @@ async function fetchModelsDev( } async function fetchDynamicModels( - fetcher: () => Promise[] | null>, + fetcher: () => Promise[] | null>, ): Promise[] | null> { try { const models = await fetcher(); @@ -311,7 +312,9 @@ function fingerprintStatic( function mergeDynamicModel(existingModel: Model, dynamicModel: Model): Model { const supportsImage = existingModel.input.includes("image") || dynamicModel.input.includes("image"); - return enrichModelThinking({ + // Re-build from spec stage: sparse compat comes from `compatConfig` (the + // verbatim override vocabulary), never the resolved `compat` record. + return buildModel({ ...existingModel, ...dynamicModel, name: preferDiscoveryName(dynamicModel.name, existingModel.name, dynamicModel.id), @@ -326,9 +329,9 @@ function mergeDynamicModel(existingModel: Model, dynamic contextWindow: preferDiscoveryLimit(dynamicModel.contextWindow, existingModel.contextWindow), maxTokens: preferDiscoveryLimit(dynamicModel.maxTokens, existingModel.maxTokens), headers: dynamicModel.headers ? { ...existingModel.headers, ...dynamicModel.headers } : existingModel.headers, - compat: dynamicModel.compat ?? existingModel.compat, + compat: dynamicModel.compatConfig ?? existingModel.compatConfig, contextPromotionTarget: dynamicModel.contextPromotionTarget ?? existingModel.contextPromotionTarget, - }); + } as ModelSpec); } function preferDiscoveryCost(discoveryCost: number, fallbackCost: number): number { @@ -366,13 +369,13 @@ function normalizeModelList(value: unknown): Model[] { const models: Model[] = []; for (const item of value) { if (isModelLike(item)) { - models.push(enrichModelThinking(item as Model)); + models.push(buildModel(item as ModelSpec)); } } return models; } -function isModelLike(value: unknown): value is Model { +function isModelLike(value: unknown): value is ModelSpec { if (!isRecord(value)) { return false; } diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index 6e99501bd..690570d5a 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -1,4 +1,4 @@ -import { resolveOpenAICompat } from "./compat/openai"; +import { buildOpenAICompat } from "./compat/openai"; import { Effort, THINKING_EFFORTS } from "./effort"; import { modelMatchesHost } from "./hosts"; import { @@ -14,7 +14,13 @@ import { semverEqual, semverGte, } from "./identity/classify"; -import type { Api, Model as ApiModel, ThinkingConfig } from "./types"; +import type { Api, Model, ModelSpec, ThinkingConfig } from "./types"; + +/** + * Thinking inference reads identity fields plus sparse compat intent, so it + * accepts both pre-build specs and built models. + */ +type ApiModel = ModelSpec | Model; const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]; const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [ @@ -76,6 +82,8 @@ type ModelWithEnriched = ApiModel & { [kEnrichedModel]?: ApiModel }; * This helper belongs to catalog enrichment only. Runtime consumers should * trust `model.thinking` and avoid inferring capabilities on demand. */ +export function enrichModelThinking(model: ModelSpec): ModelSpec; +export function enrichModelThinking(model: Model): Model; export function enrichModelThinking(model: ApiModel): ApiModel { const tagged = model as ModelWithEnriched; const cached = tagged[kEnrichedModel]; @@ -592,7 +600,7 @@ function inferFallbackEfforts(model: ApiModel): readonly return DEFAULT_REASONING_EFFORTS; } if (model.api === "openai-completions") { - const compat = resolveOpenAICompat(model as ApiModel<"openai-completions">); + const compat = buildOpenAICompat(model as ModelSpec<"openai-completions">); if (compat.thinkingFormat === "openai" && compat.supportsReasoningEffort) { return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; } diff --git a/packages/catalog/src/models.ts b/packages/catalog/src/models.ts index 8794d0259..363ac1b6f 100644 --- a/packages/catalog/src/models.ts +++ b/packages/catalog/src/models.ts @@ -1,6 +1,6 @@ -import { enrichModelThinking } from "./model-thinking"; +import { buildModel } from "./build"; import MODELS from "./models.json" with { type: "json" }; -import type { Api, KnownProvider, Model, Usage } from "./types"; +import type { Api, KnownProvider, Model, ModelSpec, Usage } from "./types"; /** * Static bundled model registry loaded from `models.json`. @@ -19,7 +19,7 @@ function getModelRegistry(): Map>> { for (const [provider, models] of Object.entries(MODELS)) { const providerModels = new Map>(); for (const [id, model] of Object.entries(models)) { - providerModels.set(id, enrichModelThinking(model as Model)); + providerModels.set(id, buildModel(model as ModelSpec)); } modelRegistry.set(provider, providerModels); } diff --git a/packages/catalog/src/provider-models/bundled-references.ts b/packages/catalog/src/provider-models/bundled-references.ts index 9127fe563..824b0e9a0 100644 --- a/packages/catalog/src/provider-models/bundled-references.ts +++ b/packages/catalog/src/provider-models/bundled-references.ts @@ -1,19 +1,30 @@ import { getBundledModels, getBundledProviders } from "../models"; -import type { Api, Model } from "../types"; +import type { Api, Model, ModelSpec } from "../types"; + +/** + * Project a built `Model` back to spec stage: `compat` becomes the verbatim + * sparse override record (`compatConfig`), never the resolved view. Discovery + * mappers spread these references into the specs they hand to the model + * manager, which rebuilds via `buildModel`. + */ +export function toModelSpec(model: Model): ModelSpec { + const { compat: _compat, compatConfig, ...rest } = model; + return { ...rest, compat: compatConfig } as ModelSpec; +} export function createBundledReferenceMap( provider: Parameters[0], -): Map> { - const references = new Map>(); +): Map> { + const references = new Map>(); for (const model of getBundledModels(provider)) { - references.set(model.id, model as Model); + references.set(model.id, toModelSpec(model as Model)); } return references; } export function createReferenceResolver( - providerRefs: Map>, -): (modelId: string) => Model | undefined { + providerRefs: Map>, +): (modelId: string) => ModelSpec | undefined { const globalRefs = new Map>(); for (const provider of getBundledProviders()) { for (const model of getBundledModels(provider as Parameters[0])) { @@ -34,5 +45,10 @@ export function createReferenceResolver( } } } - return (modelId: string) => providerRefs.get(modelId) ?? (globalRefs.get(modelId) as Model | undefined); + return (modelId: string) => { + const providerRef = providerRefs.get(modelId); + if (providerRef) return providerRef; + const globalRef = globalRefs.get(modelId); + return globalRef ? toModelSpec(globalRef as Model) : undefined; + }; } diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 7fd765abf..aba4119bd 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -7,10 +7,10 @@ import { Effort } from "../effort"; import { toFireworksPublicModelId } from "../fireworks-model-id"; import type { ModelManagerOptions } from "../model-manager"; import { getBundledModels } from "../models"; -import type { Api, FetchImpl, Model, Provider, ThinkingConfig } from "../types"; +import type { Api, FetchImpl, Model, ModelSpec, Provider, ThinkingConfig } from "../types"; import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils"; import { getGitHubCopilotBaseUrl, OPENCODE_HEADERS, parseGitHubCopilotApiKey } from "../wire/github-copilot"; -import { createBundledReferenceMap, createReferenceResolver } from "./bundled-references"; +import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references"; import { UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS } from "./discovery-constants"; const MODELS_DEV_URL = "https://models.dev/api.json"; @@ -67,7 +67,7 @@ async function fetchModelsDevPayload(fetchImpl: FetchImpl = fetch): Promise[] { +function mapAnthropicModelsDev(payload: unknown, baseUrl: string): ModelSpec<"anthropic-messages">[] { if (!isRecord(payload)) { return []; } @@ -80,7 +80,7 @@ function mapAnthropicModelsDev(payload: unknown, baseUrl: string): Model<"anthro return []; } - const models: Model<"anthropic-messages">[] = []; + const models: ModelSpec<"anthropic-messages">[] = []; for (const [modelId, rawModel] of Object.entries(modelsValue)) { if (!isRecord(rawModel)) { continue; @@ -128,9 +128,9 @@ function buildAnthropicDiscoveryHeaders(apiKey: string): Record } function buildAnthropicReferenceMap( - modelsDevModels: readonly Model<"anthropic-messages">[], -): Map> { - const merged = new Map>(); + modelsDevModels: readonly ModelSpec<"anthropic-messages">[], +): Map> { + const merged = new Map>(); for (const model of modelsDevModels) { merged.set(model.id, model); } @@ -140,7 +140,7 @@ function buildAnthropicReferenceMap( (model): model is Model<"anthropic-messages"> => model.api === "anthropic-messages", ); for (const model of bundledModels) { - merged.set(model.id, model); + merged.set(model.id, toModelSpec(model)); } return merged; } @@ -155,7 +155,7 @@ function buildAnthropicReferenceMap( * `applyAnthropicCatalogPolicy`, and `thinking` is derived by * `refreshModelThinking` during generation. */ -export const ANTHROPIC_CURATED_FALLBACK_MODELS: readonly Model<"anthropic-messages">[] = [ +export const ANTHROPIC_CURATED_FALLBACK_MODELS: readonly ModelSpec<"anthropic-messages">[] = [ { id: "claude-fable-5", name: "Claude Fable 5", @@ -184,9 +184,9 @@ export const ANTHROPIC_CURATED_FALLBACK_MODELS: readonly Model<"anthropic-messag function mapWithBundledReference( entry: OpenAICompatibleModelRecord, - defaults: Model, - reference: Model | undefined, -): Model { + defaults: ModelSpec, + reference: ModelSpec | undefined, +): ModelSpec { const name = toModelName(entry.name, reference?.name ?? defaults.name); if (!reference) { return { @@ -233,7 +233,7 @@ async function fetchOllamaNativeModels( baseUrl: string, resolveMetadata: (modelId: string) => Promise, fetchImpl: FetchImpl = fetch, -): Promise[] | null> { +): Promise[] | null> { const nativeBaseUrl = toOllamaNativeBaseUrl(baseUrl); let response: Response; try { @@ -250,7 +250,7 @@ async function fetchOllamaNativeModels( const payload = (await response.json()) as { models?: Array<{ name?: string; model?: string }> }; const entries = payload.models ?? []; const resolved = await Promise.all( - entries.map(async (entry): Promise | null> => { + entries.map(async (entry): Promise | null> => { const id = entry.model ?? entry.name; if (!id) return null; const metadata = await resolveMetadata(id); @@ -269,7 +269,9 @@ async function fetchOllamaNativeModels( }; }), ); - const models: Model<"openai-responses">[] = resolved.filter((m): m is Model<"openai-responses"> => m !== null); + const models: ModelSpec<"openai-responses">[] = resolved.filter( + (m): m is ModelSpec<"openai-responses"> => m !== null, + ); return models.sort((left, right) => left.id.localeCompare(right.id)); } @@ -290,7 +292,7 @@ const OLLAMA_DEFAULT_MAX_TOKENS = 8192; const OLLAMA_REASONING_EFFORT_MAP = { minimal: "low", xhigh: "max" } as const; /** Stamp the Ollama reasoning-effort map onto a reasoning-capable model. */ -function applyOllamaReasoningCompat(model: Model<"openai-responses">): void { +function applyOllamaReasoningCompat(model: ModelSpec<"openai-responses">): void { if (!model.reasoning) return; model.compat = { ...model.compat, @@ -431,7 +433,7 @@ const OPENAI_NON_RESPONSES_PREFIXES = [ "gpt-realtime", ] as const; -function isLikelyOpenAIResponsesModelId(id: string, references: Map>): boolean { +function isLikelyOpenAIResponsesModelId(id: string, references: Map>): boolean { const trimmed = id.trim(); if (!trimmed) { return false; @@ -766,7 +768,10 @@ const XAI_REASONING_EFFORT_MAP = { minimal: "low" } as const; // The `minimal -> low` effort clamp (XAI_REASONING_EFFORT_MAP) is always // merged in so dynamic-fetched models — which arrive without curated // compat keys — still get the clamp applyResponsesReasoningParams expects. -function mergeCuratedIntoModel(base: Model<"openai-responses">, curated: XAICuratedModel): Model<"openai-responses"> { +function mergeCuratedIntoModel( + base: ModelSpec<"openai-responses">, + curated: XAICuratedModel, +): ModelSpec<"openai-responses"> { const effort = curated.supportsReasoningEffort; const compat = { ...(base.compat ?? {}), @@ -805,10 +810,10 @@ function mergeCuratedIntoModel(base: Model<"openai-responses">, curated: XAICura * Order: curated models first in declaration order; then dynamic remainder * in original order. */ -function applyXAIOAuthCuration(dynamic: readonly Model<"openai-responses">[]): Model<"openai-responses">[] { +function applyXAIOAuthCuration(dynamic: readonly ModelSpec<"openai-responses">[]): ModelSpec<"openai-responses">[] { const filtered = dynamic.filter(e => !XAI_NON_CHAT_PREFIXES.some(p => e.id.startsWith(p))); - const byId = new Map>(filtered.map(e => [e.id, e])); + const byId = new Map>(filtered.map(e => [e.id, e])); for (const curated of XAI_OAUTH_CURATED_MODELS) { const existing = byId.get(curated.id); if (existing) { @@ -823,7 +828,7 @@ function applyXAIOAuthCuration(dynamic: readonly Model<"openai-responses">[]): M // Reset id/name on the template before merging so the helper's // `curated.name ?? base.name` clause falls back to curated.id // (the inject contract), not to the unrelated template's label. - const base: Model<"openai-responses"> = { ...template, id: curated.id, name: curated.id }; + const base: ModelSpec<"openai-responses"> = { ...template, id: curated.id, name: curated.id }; byId.set(curated.id, mergeCuratedIntoModel(base, curated)); } } @@ -831,14 +836,14 @@ function applyXAIOAuthCuration(dynamic: readonly Model<"openai-responses">[]): M const curatedIds = new Set(XAI_OAUTH_CURATED_MODELS.map(c => c.id)); const curatedFirst = XAI_OAUTH_CURATED_MODELS.map(c => byId.get(c.id)).filter( - (e): e is Model<"openai-responses"> => e !== undefined, + (e): e is ModelSpec<"openai-responses"> => e !== undefined, ); const rest = filtered.filter(e => !curatedIds.has(e.id)); return [...curatedFirst, ...rest]; } /** - * Render `XAI_OAUTH_CURATED_MODELS` as full `Model<"openai-responses">` entries. + * Render `XAI_OAUTH_CURATED_MODELS` as full `ModelSpec<"openai-responses">` entries. * * Single source of truth for the curated to Model fan-in, consumed by both * - {@link xaiOAuthModelManagerOptions} (runtime static seed handed to the model @@ -854,14 +859,14 @@ function applyXAIOAuthCuration(dynamic: readonly Model<"openai-responses">[]): M * dynamic fetch merge cleanly. Mirrors * `hermes-agent/hermes_cli/models.py:_XAI_STATIC_FALLBACK`. */ -export function buildXaiOAuthStaticSeed(baseUrl?: string): Model<"openai-responses">[] { +export function buildXaiOAuthStaticSeed(baseUrl?: string): ModelSpec<"openai-responses">[] { const resolvedBaseUrl = baseUrl ?? "https://api.x.ai/v1"; return XAI_OAUTH_CURATED_MODELS.map(curated => { // Synthesise a bare base then layer curated metadata via the same helper // the dynamic overlay/inject paths use. `name: curated.id` is a sentinel // the helper rewrites to `curated.name ?? base.name`, so curated.name // wins when set. - const base: Model<"openai-responses"> = { + const base: ModelSpec<"openai-responses"> = { id: curated.id, name: curated.id, api: "openai-responses", @@ -1005,9 +1010,9 @@ export function zhipuCodingPlanModelManagerOptions( apiKey, mapModel: ( _entry: OpenAICompatibleModelRecord, - defaults: Model<"openai-completions">, + defaults: ModelSpec<"openai-completions">, _context: OpenAICompatibleModelMapperContext<"openai-completions">, - ): Model<"openai-completions"> => { + ): ModelSpec<"openai-completions"> => { const id = defaults.id; return { ...defaults, @@ -1084,9 +1089,9 @@ export function clampFireworksKimiMaxTokens(modelId: string, candidate: number): * DeepSeek-native binary `thinking` toggle when both are present. */ export function stripFireworksDeepSeekThinkingToggle( - model: Model<"openai-completions">, + model: ModelSpec<"openai-completions">, publicModelId: string, -): Model<"openai-completions"> { +): ModelSpec<"openai-completions"> { if (!publicModelId.startsWith("deepseek-v4")) return model; const compat = model.compat; if (!compat?.extraBody || !("thinking" in compat.extraBody)) return model; @@ -1121,10 +1126,12 @@ function toFireworksModelName(entry: OpenAICompatibleModelRecord, fallback: stri .join(" "); } -function createModelsDevReferenceMap(models: readonly Model[]): Map> { - const references = new Map>(); +function createModelsDevReferenceMap( + models: readonly ModelSpec[], +): Map> { + const references = new Map>(); for (const model of models) { - const candidate = model as Model; + const candidate = model as ModelSpec; const existing = references.get(candidate.id); if (!existing) { references.set(candidate.id, candidate); @@ -1141,14 +1148,14 @@ function createModelsDevReferenceMap(models: readonly Model(fetchImpl?: FetchImpl): Promise>> { +async function loadModelsDevReferences(fetchImpl?: FetchImpl): Promise>> { try { const payload = await fetchModelsDevPayload(fetchImpl); return createModelsDevReferenceMap( mapModelsDevToModels(payload as Record, MODELS_DEV_PROVIDER_DESCRIPTORS), ); } catch { - return new Map>(); + return new Map>(); } } export function fireworksModelManagerOptions( @@ -1241,7 +1248,7 @@ const WAFER_MAX_TOKENS_CAP = 65536; * * Wafer wraps each entry with a `wafer` envelope describing tier, capabilities, * and cents-per-million pricing. The mapper folds that metadata into the - * canonical `Model<"openai-completions">` shape and applies zai-family thinking + * canonical `ModelSpec<"openai-completions">` shape and applies zai-family thinking * compat when the entry advertises reasoning support (GLM-family on the Pass * SKU). Cents-per-million → dollars-per-million via /100. */ @@ -1267,8 +1274,8 @@ function mapWaferModel( providerId: "wafer-pass" | "wafer-serverless", baseUrl: string, entry: OpenAICompatibleModelRecord, - defaults: Model<"openai-completions">, -): Model<"openai-completions"> { + defaults: ModelSpec<"openai-completions">, +): ModelSpec<"openai-completions"> { const wafer = readWaferRecord(entry); const capabilities = wafer?.capabilities ?? {}; const reasoning = capabilities.reasoning === true; @@ -1299,7 +1306,7 @@ function mapWaferModel( cacheWrite: 0, }; const name = toModelName(wafer?.display_name, defaults.name); - const base: Model<"openai-completions"> = { + const base: ModelSpec<"openai-completions"> = { ...defaults, id: defaults.id, name, @@ -1561,9 +1568,9 @@ export function openrouterModelManagerOptions( }, mapModel: ( entry: OpenAICompatibleModelRecord, - defaults: Model<"openai-completions">, + defaults: ModelSpec<"openai-completions">, _context: OpenAICompatibleModelMapperContext<"openai-completions">, - ): Model<"openai-completions"> => { + ): ModelSpec<"openai-completions"> => { const pricing = entry.pricing as Record | undefined; const params = Array.isArray(entry.supported_parameters) ? (entry.supported_parameters as string[]) : []; const modality = String((entry.architecture as Record | undefined)?.modality ?? ""); @@ -1805,9 +1812,9 @@ export function vercelAiGatewayModelManagerOptions( }, mapModel: ( entry: OpenAICompatibleModelRecord, - defaults: Model<"anthropic-messages">, + defaults: ModelSpec<"anthropic-messages">, _context: OpenAICompatibleModelMapperContext<"anthropic-messages">, - ): Model<"anthropic-messages"> => { + ): ModelSpec<"anthropic-messages"> => { const pricing = entry.pricing as Record | undefined; const tags = Array.isArray(entry.tags) ? (entry.tags as string[]) : []; @@ -1862,9 +1869,9 @@ export function kimiCodeModelManagerOptions( }, mapModel: ( entry: OpenAICompatibleModelRecord, - defaults: Model<"openai-completions">, + defaults: ModelSpec<"openai-completions">, _context: OpenAICompatibleModelMapperContext<"openai-completions">, - ): Model<"openai-completions"> => { + ): ModelSpec<"openai-completions"> => { const id = defaults.id; return { ...defaults, @@ -1935,7 +1942,7 @@ export function syntheticModelManagerOptions( const apiKey = config?.apiKey; const baseUrl = config?.baseUrl ?? "https://api.synthetic.new/openai/v1"; const references = new Map( - (getBundledModels("synthetic") as Model<"openai-completions">[]).map(model => [model.id, model]), + (getBundledModels("synthetic") as Model<"openai-completions">[]).map(model => [model.id, toModelSpec(model)]), ); return { providerId: "synthetic", @@ -1949,9 +1956,9 @@ export function syntheticModelManagerOptions( apiKey, mapModel: ( entry: OpenAICompatibleModelRecord, - defaults: Model<"openai-completions">, + defaults: ModelSpec<"openai-completions">, _context: OpenAICompatibleModelMapperContext<"openai-completions">, - ): Model<"openai-completions"> => { + ): ModelSpec<"openai-completions"> => { const reference = references.get(defaults.id); const referenceSupportsImage = reference?.input.includes("image") ?? false; return { @@ -2407,9 +2414,9 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana headers: OPENCODE_HEADERS, mapModel: ( entry: OpenAICompatibleModelRecord, - defaults: Model, + defaults: ModelSpec, _context: OpenAICompatibleModelMapperContext, - ): Model => { + ): ModelSpec => { const reference = resolveReference(defaults.id); const copilotLimits = extractCopilotLimits(entry); // Copilot exposes token limits under capabilities.limits.*. @@ -2522,9 +2529,9 @@ export function anthropicModelManagerOptions( headers: buildAnthropicDiscoveryHeaders(apiKey), mapModel: ( entry: OpenAICompatibleModelRecord, - defaults: Model<"anthropic-messages">, + defaults: ModelSpec<"anthropic-messages">, _context: OpenAICompatibleModelMapperContext<"anthropic-messages">, - ): Model<"anthropic-messages"> => { + ): ModelSpec<"anthropic-messages"> => { const discoveredName = typeof entry.display_name === "string" ? entry.display_name : defaults.name; const reference = references.get(defaults.id); if (!reference) { @@ -2571,7 +2578,7 @@ export interface ModelsDevProviderDescriptor { /** Default max tokens fallback (default: UNKNNOWN_MAX_TOKENS) */ defaultMaxTokens?: number; /** Optional compat overrides applied to every model from this provider */ - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; /** Optional static headers applied to every model */ headers?: Record; /** @@ -2583,7 +2590,11 @@ export interface ModelsDevProviderDescriptor { * Optional transform: modify the mapped model before it's added. * Can return null to skip the model, or an array to emit multiple models. */ - transformModel?: (model: Model, modelId: string, raw: ModelsDevModel) => Model | Model[] | null; + transformModel?: ( + model: ModelSpec, + modelId: string, + raw: ModelsDevModel, + ) => ModelSpec | ModelSpec[] | null; /** * Optional: override the API type per-model. * Called with (modelId, raw). Return the API type to use. @@ -2596,8 +2607,8 @@ export interface ModelsDevProviderDescriptor { export function mapModelsDevToModels( data: Record, descriptors: readonly ModelsDevProviderDescriptor[], -): Model[] { - const models: Model[] = []; +): ModelSpec[] { + const models: ModelSpec[] = []; for (const desc of descriptors) { const providerData = (data as Record>)[desc.modelsDevKey]; if (!isRecord(providerData) || !isRecord(providerData.models)) continue; @@ -2617,11 +2628,11 @@ export function mapModelsDevToModels( const resolved = desc.resolveApi?.(modelId, m) ?? { api: desc.api, baseUrl: desc.baseUrl }; if (!resolved) continue; - const mapped: Model = { + const mapped: ModelSpec = { id: modelId, name: toModelName(m.name, modelId), api: resolved.api, - provider: desc.providerId as Model["provider"], + provider: desc.providerId as ModelSpec["provider"], baseUrl: resolved.baseUrl, reasoning: m.reasoning === true, input: toInputCapabilities(m.modalities?.input), @@ -2858,7 +2869,7 @@ const MODELS_DEV_PROVIDER_DESCRIPTORS_BEDROCK: readonly ModelsDevProviderDescrip }, transformModel: (model, modelId, m) => { const crossRegionId = bedrockCrossRegionId(modelId); - const bedrockModel: Model = { + const bedrockModel: ModelSpec = { ...model, id: crossRegionId, name: toModelName(m.name, crossRegionId), diff --git a/packages/catalog/src/types.ts b/packages/catalog/src/types.ts index 0b974e0cf..f841b943e 100644 --- a/packages/catalog/src/types.ts +++ b/packages/catalog/src/types.ts @@ -191,6 +191,20 @@ export interface OpenAICompat { supportsLongPromptCacheRetention?: boolean; /** Whether tool schemas must be sent either all strict or all non-strict. Undefined keeps the existing per-tool mixed behavior. */ toolStrictMode?: "all_strict" | "none"; + /** Whether request shaping may send reasoning params at all. Default: auto-detected (disabled for GitHub Copilot chat-completions). */ + supportsReasoningParams?: boolean; + /** Always send a max-token field when the caller did not provide one. Default: auto-detected (Kimi-family models derive TPM limits from max_tokens). */ + alwaysSendMaxTokens?: boolean; + /** Whether Responses-API tool-call/result history must be strictly paired. Default: auto-detected (Azure OpenAI, GitHub Copilot). */ + strictResponsesPairing?: boolean; + /** + * Compat deltas applied when a request actually engages thinking mode + * (reasoning requested and not disabled, model reasoning-capable, and not + * suppressed by a forced tool choice). `buildModel` materializes the full + * alternate view as `compat.whenThinking`; handlers pointer-swap, never + * spread. Default: auto-detected (OpenCode gateways, #1071/#1484). + */ + whenThinking?: Partial>; } /** @@ -274,6 +288,85 @@ export interface VercelGatewayRouting { order?: string[]; } +type ResolvedToolStrictMode = NonNullable | "mixed"; + +/** + * Fully-resolved chat-completions compat view: every detected default + * materialized and user overrides applied. Built once per model by + * `buildModel`; request handlers read fields and never detect, resolve, or + * allocate. + */ +export type ResolvedOpenAICompat = Required< + Omit< + OpenAICompat, + | "openRouterRouting" + | "vercelGatewayRouting" + | "extraBody" + | "toolStrictMode" + | "streamIdleTimeoutMs" + | "supportsLongPromptCacheRetention" + | "cacheControlFormat" + | "thinkingKeep" + | "strictResponsesPairing" + | "whenThinking" + > +> & { + openRouterRouting?: OpenAICompat["openRouterRouting"]; + vercelGatewayRouting?: OpenAICompat["vercelGatewayRouting"]; + extraBody?: OpenAICompat["extraBody"]; + cacheControlFormat?: OpenAICompat["cacheControlFormat"]; + thinkingKeep?: OpenAICompat["thinkingKeep"]; + streamIdleTimeoutMs?: number; + toolStrictMode: ResolvedToolStrictMode; + /** The model sits behind OpenRouter (routing prefs and max-token omission apply). */ + isOpenRouterHost: boolean; + /** The model sits behind Vercel AI Gateway. */ + isVercelGatewayHost: boolean; + /** Complete alternate view for thinking-engaged requests; swap pointers, never spread. */ + whenThinking?: ResolvedOpenAICompat; +}; + +/** Fully-resolved Responses-API compat view (same contract as `ResolvedOpenAICompat`). */ +export interface ResolvedOpenAIResponsesCompat { + supportsDeveloperRole: boolean; + supportsStrictMode: boolean; + supportsReasoningEffort: boolean; + supportsLongPromptCacheRetention: boolean; + strictResponsesPairing: boolean; + reasoningEffortMap: Partial>; +} + +/** Fully-resolved anthropic-messages compat view (same contract as `ResolvedOpenAICompat`). */ +export type ResolvedAnthropicCompat = Required & { + /** + * The configured endpoint is the official first-party Anthropic API + * (https + exact `api.anthropic.com` host; a missing baseUrl counts as + * official because dispatch defaults there). Gates OAuth framing, custom + * env headers, and cache-TTL shaping without per-request URL parsing. + */ + officialEndpoint: boolean; +}; + +/** Sparse, user-authored compat overrides for a given API (models.json / config vocabulary). */ +export type CompatConfigOf = TApi extends + | "openai-completions" + | "openai-responses" + | "azure-openai-responses" + | "openai-codex-responses" + ? OpenAICompat + : TApi extends "anthropic-messages" + ? AnthropicCompat + : undefined; + +/** Resolved compat for a given API: complete record, materialized once by `buildModel`. */ +export type CompatOf = TApi extends "openai-completions" + ? ResolvedOpenAICompat + : TApi extends "openai-responses" | "azure-openai-responses" | "openai-codex-responses" + ? ResolvedOpenAIResponsesCompat + : TApi extends "anthropic-messages" + ? ResolvedAnthropicCompat + : undefined; + // Model interface for the unified model system export interface Model { id: string; @@ -329,12 +422,13 @@ export interface Model { priority?: number; /** Canonical thinking capability metadata for this model. */ thinking?: ThinkingConfig; - /** Compatibility overrides per API. If not set, auto-detected from baseUrl. */ - compat?: TApi extends "openai-completions" | "openai-responses" - ? OpenAICompat - : TApi extends "anthropic-messages" - ? AnthropicCompat - : never; + /** + * Fully-resolved compatibility record, materialized once by `buildModel`. + * Protocol handlers read fields; they never detect, resolve, or allocate. + */ + compat: CompatOf; + /** Verbatim sparse compat from the spec (user/config intent), for introspection only. */ + compatConfig?: CompatConfigOf; /** * Which shape to use when exposing the Codex `apply_patch` tool to this model. * Generated catalog policy sets `"freeform"` for first-party GPT-5 Responses @@ -351,3 +445,13 @@ export interface Model { */ isOAuth?: boolean; } + +/** + * A model as authored by configs, bundled catalogs, and discovery — the input + * vocabulary of `buildModel`. Identical to `Model` except `compat` carries the + * sparse override shape and nothing is resolved yet. + */ +export interface ModelSpec extends Omit, "compat" | "compatConfig"> { + /** Sparse compatibility overrides; resolved into `Model.compat` by `buildModel`. */ + compat?: CompatConfigOf; +} diff --git a/packages/catalog/test/build.test.ts b/packages/catalog/test/build.test.ts new file mode 100644 index 000000000..6dbf3c0b3 --- /dev/null +++ b/packages/catalog/test/build.test.ts @@ -0,0 +1,144 @@ +import { Database } from "bun:sqlite"; +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { isOfficialAnthropicApiUrl } from "@oh-my-pi/pi-catalog/compat/anthropic"; +import { resolveProviderModels } from "@oh-my-pi/pi-catalog/model-manager"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; + +function completionsSpec(overrides: Partial> = {}): ModelSpec<"openai-completions"> { + return { + id: "some-model", + name: "Some Model", + api: "openai-completions", + provider: "custom", + baseUrl: "https://api.example.com/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 8_192, + ...overrides, + }; +} + +describe("buildModel", () => { + it("resolves a complete compat record for an openai-completions spec with no compat", () => { + const model = buildModel(completionsSpec()); + + expect(model.compat).toBeDefined(); + expect(typeof model.compat.supportsStore).toBe("boolean"); + expect(model.compat.maxTokensField).toBe("max_completion_tokens"); + expect(model.compat.thinkingFormat).toBe("openai"); + expect(typeof model.compat.isOpenRouterHost).toBe("boolean"); + expect(model.compat.isOpenRouterHost).toBe(false); + expect(model.compatConfig).toBeUndefined(); + }); + + it("lets sparse overrides win over detection and keeps the verbatim config", () => { + const sparse = { supportsDeveloperRole: true } as const; + const model = buildModel( + completionsSpec({ + provider: "groq", + baseUrl: "https://api.groq.com/openai/v1", + compat: sparse, + }), + ); + + // Detection would say false for a non-OpenAI host; the override wins. + expect(model.compat.supportsDeveloperRole).toBe(true); + // The verbatim sparse object is preserved by reference. + expect(model.compatConfig).toBe(sparse); + }); + + it("materializes the opencode whenThinking variant without mutating the base view", () => { + const model = buildModel( + completionsSpec({ + provider: "opencode-zen", + baseUrl: "https://opencode.ai/zen/v1", + reasoning: true, + }), + ); + + expect(model.compat.whenThinking).toBeDefined(); + expect(model.compat.whenThinking?.requiresReasoningContentForToolCalls).toBe(true); + expect(model.compat.whenThinking?.allowsSyntheticReasoningContentForToolCalls).toBe(false); + // Base compat stays on the thinking-off defaults. + expect(model.compat.requiresReasoningContentForToolCalls).toBe(false); + expect(model.compat.allowsSyntheticReasoningContentForToolCalls).toBe(true); + }); + + it("leaves whenThinking undefined for non-opencode reasoning specs", () => { + const model = buildModel(completionsSpec({ reasoning: true })); + expect(model.compat.whenThinking).toBeUndefined(); + }); +}); + +describe("model cache spec round trip", () => { + it("persists sparse specs and rebuilds resolved models on cache reads", async () => { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-catalog-model-cache-")); + const dbPath = path.join(tempDir, "models.db"); + const sparse = { supportsDeveloperRole: true } as const; + const spec = completionsSpec({ provider: "spec-cache-test", compat: sparse }); + try { + const online = await resolveProviderModels<"openai-completions">( + { + providerId: "spec-cache-test", + staticModels: [], + cacheDbPath: dbPath, + fetchDynamicModels: async () => [spec], + }, + "online", + ); + expect(online.models[0]?.compat.supportsDeveloperRole).toBe(true); + + // The persisted row carries the sparse spec, never the resolved record. + const db = new Database(dbPath, { readonly: true }); + const row = db + .query<{ models: string }, [string]>("SELECT models FROM model_cache WHERE provider_id = ?") + .get("spec-cache-test"); + db.close(); + expect(row).toBeDefined(); + const persisted = JSON.parse(row?.models ?? "[]") as ModelSpec<"openai-completions">[]; + expect(persisted[0]?.compat).toEqual(sparse); + expect(persisted[0]).not.toHaveProperty("compatConfig"); + expect(persisted[0]?.compat).not.toHaveProperty("isOpenRouterHost"); + + // Offline reads rebuild the row into a fully-resolved model. + const offline = await resolveProviderModels<"openai-completions">( + { + providerId: "spec-cache-test", + staticModels: [], + cacheDbPath: dbPath, + }, + "offline", + ); + const model = offline.models.find(candidate => candidate.id === spec.id); + expect(model?.compat.supportsDeveloperRole).toBe(true); + expect(model?.compat.isOpenRouterHost).toBe(false); + expect(model?.compatConfig).toEqual(sparse); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } + }); +}); + +describe("isOfficialAnthropicApiUrl", () => { + it("treats a missing baseUrl as official", () => { + expect(isOfficialAnthropicApiUrl(undefined)).toBe(true); + }); + + it("accepts the https first-party host", () => { + expect(isOfficialAnthropicApiUrl("https://api.anthropic.com/v1")).toBe(true); + }); + + it("rejects non-https schemes", () => { + expect(isOfficialAnthropicApiUrl("http://api.anthropic.com")).toBe(false); + }); + + it("rejects lookalike hostnames", () => { + expect(isOfficialAnthropicApiUrl("https://api.anthropic.com.evil.com")).toBe(false); + }); +}); diff --git a/packages/catalog/test/issue-1846-repro.test.ts b/packages/catalog/test/issue-1846-repro.test.ts index b6060e2e3..fb4211a39 100644 --- a/packages/catalog/test/issue-1846-repro.test.ts +++ b/packages/catalog/test/issue-1846-repro.test.ts @@ -2,9 +2,10 @@ import { Database } from "bun:sqlite"; import { afterEach, describe, expect, it, vi } from "bun:test"; import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai/auth-storage"; -import { convertMessages, detectCompat } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { convertMessages } from "@oh-my-pi/pi-ai/providers/openai-completions"; import { getOAuthProviders } from "@oh-my-pi/pi-ai/registry/oauth"; import type { AssistantMessage, ThinkingContent, ToolCall } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { xiaomiModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types"; @@ -16,7 +17,7 @@ afterEach(() => { }); function mimoModel(): Model<"openai-completions"> { - return { + return buildModel({ id: "mimo-v2.5-pro", name: "MiMo V2.5 Pro", api: "openai-completions", @@ -27,7 +28,7 @@ function mimoModel(): Model<"openai-completions"> { cost: { input: 1, output: 3, cacheRead: 0.2, cacheWrite: 0 }, contextWindow: 1_048_576, maxTokens: 131_072, - }; + }); } function assistantToolCall(model: Model<"openai-completions">, content: AssistantMessage["content"]): AssistantMessage { @@ -109,7 +110,7 @@ describe("issue #1846: Xiaomi Token Plan provider support", () => { it("replays MiMo reasoning_content on Token Plan tool-call turns", () => { const model = mimoModel(); - const compat = detectCompat(model); + const compat = model.compat; const thinking: ThinkingContent = { type: "thinking", thinking: "I need to inspect the file before answering.", diff --git a/packages/catalog/test/issue-2113-repro.test.ts b/packages/catalog/test/issue-2113-repro.test.ts index 400e7f7b0..0f083d55f 100644 --- a/packages/catalog/test/issue-2113-repro.test.ts +++ b/packages/catalog/test/issue-2113-repro.test.ts @@ -19,20 +19,25 @@ import { describe, expect, it } from "bun:test"; import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { AssistantMessage, Context } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { moonshotModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; -import type { Model } from "@oh-my-pi/pi-catalog/types"; +import type { Model, ModelSpec } from "@oh-my-pi/pi-catalog/types"; function moonshotKimiModel(id: string, reasoning: boolean): Model<"openai-completions"> { - return { - ...getBundledModel("openai", "gpt-4o-mini"), + const reference = getBundledModel("openai", "gpt-4o-mini"); + // Derive a variant from the built bundled model: sparse compat comes from + // `compatConfig`; `buildModel` re-resolves it for the Moonshot host. + return buildModel({ + ...reference, api: "openai-completions", provider: "moonshot", baseUrl: "https://api.moonshot.ai/v1", id, reasoning, - }; + compat: reference.compatConfig, + } as ModelSpec<"openai-completions">); } function basicContext(): Context { diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index e359bc86b..4122fee1c 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -9,14 +9,14 @@ import { mapEffortToGoogleThinkingLevel, requireSupportedEffort, } from "@oh-my-pi/pi-catalog/model-thinking"; -import type { Api, Model, Provider } from "@oh-my-pi/pi-catalog/types"; +import type { Api, ModelSpec, Provider } from "@oh-my-pi/pi-catalog/types"; function createModel(overrides: { id: string; api: TApi; provider: Provider; reasoning?: boolean; -}): Model { +}): ModelSpec { return enrichModelThinking({ id: overrides.id, name: overrides.id, @@ -158,7 +158,7 @@ describe("model thinking metadata", () => { describe("generated model policies", () => { it("refreshes thinking metadata and applies parsed catalog corrections", () => { - const models: Model[] = [ + const models: ModelSpec[] = [ { id: "claude-opus-4-5", name: "Claude Opus 4.5", @@ -238,7 +238,7 @@ describe("generated model policies", () => { }); it("pins Claude Mythos 5 first-party Anthropic catalog metadata", () => { - const models: Model[] = [ + const models: ModelSpec[] = [ { id: "claude-mythos-5", name: "Claude Mythos 5", @@ -266,7 +266,7 @@ describe("generated model policies", () => { }); it("normalizes Copilot generated fallback limits", () => { - const models: Model[] = [ + const models: ModelSpec[] = [ { ...createModel({ id: "claude-opus-4.6", @@ -332,7 +332,7 @@ describe("generated model policies", () => { }); it("sets freeform apply_patch metadata for first-party GPT-5 Responses models", () => { - const models: Model[] = [ + const models: ModelSpec[] = [ createModel({ id: "gpt-5.4", api: "openai-responses", @@ -372,7 +372,7 @@ describe("generated model policies", () => { describe("model thinking runtime helpers", () => { it("clamps from explicit metadata instead of inferring from model id", () => { - const model: Model<"openai-codex-responses"> = { + const model: ModelSpec<"openai-codex-responses"> = { id: "custom-reasoner", name: "Custom Reasoner", api: "openai-codex-responses", @@ -433,7 +433,7 @@ describe("model thinking runtime helpers", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 32000, - } satisfies Model<"openai-completions">); + } satisfies ModelSpec<"openai-completions">); expect(model.thinking).toEqual({ mode: "effort", @@ -461,7 +461,7 @@ describe("model thinking runtime helpers", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 32000, - } satisfies Model<"openai-completions">); + } satisfies ModelSpec<"openai-completions">); expect(model.thinking).toEqual({ mode: "effort", @@ -529,7 +529,7 @@ describe("model thinking runtime helpers", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200000, maxTokens: 32000, - } as Model<"openai-responses">; + } as ModelSpec<"openai-responses">; expect(() => requireSupportedEffort(model, Effort.High)).toThrow(/missing thinking metadata/); }); @@ -551,7 +551,7 @@ describe("model thinking runtime helpers", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200000, maxTokens: 32000, - } satisfies Model<"openai-responses">); + } satisfies ModelSpec<"openai-responses">); expect(model.thinking).toBeUndefined(); }); diff --git a/packages/catalog/test/ollama-cloud-provider.test.ts b/packages/catalog/test/ollama-cloud-provider.test.ts index b7c48846b..9a2290368 100644 --- a/packages/catalog/test/ollama-cloud-provider.test.ts +++ b/packages/catalog/test/ollama-cloud-provider.test.ts @@ -1,12 +1,13 @@ import { afterEach, describe, expect, test, vi } from "bun:test"; import { completeSimple, getEnvApiKey, stream, streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { Context, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { ollamaCloudModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/ollama"; import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types"; const originalApiKey = Bun.env.OLLAMA_CLOUD_API_KEY; -const cloudModel: Model<"ollama-chat"> = { +const cloudModel: Model<"ollama-chat"> = buildModel({ id: "gpt-oss:120b", name: "GPT OSS 120B", api: "ollama-chat", @@ -17,7 +18,7 @@ const cloudModel: Model<"ollama-chat"> = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 262_144, maxTokens: 8_192, -}; +}); const readFileTool = { name: "read_file", diff --git a/packages/catalog/test/ollama-provider.test.ts b/packages/catalog/test/ollama-provider.test.ts index 56cff3966..804c07b82 100644 --- a/packages/catalog/test/ollama-provider.test.ts +++ b/packages/catalog/test/ollama-provider.test.ts @@ -1,9 +1,10 @@ import { describe, expect, test, vi } from "bun:test"; import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama"; import type { Context, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { ollamaModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; -import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types"; +import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; interface OllamaRequestBody { tools?: Array<{ function: { name: string } }>; @@ -102,7 +103,7 @@ describe("ollama tool forcing", () => { }); }); - const model = { + const model = buildModel({ id: "ggml-org/gemma-3-1b-it/GGUF", name: "Gemma 3 1B", api: "ollama-chat", @@ -113,7 +114,7 @@ describe("ollama tool forcing", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 32_768, maxTokens: 8_192, - } satisfies Model<"ollama-chat">; + } satisfies ModelSpec<"ollama-chat">); const readTool = { name: "read", description: "Read a file", diff --git a/packages/catalog/test/wafer.test.ts b/packages/catalog/test/wafer.test.ts index 6ac466e48..b207a2eae 100644 --- a/packages/catalog/test/wafer.test.ts +++ b/packages/catalog/test/wafer.test.ts @@ -39,9 +39,9 @@ describe("Wafer Pass provider", () => { expect(model.baseUrl).toBe("https://pass.wafer.ai/v1"); expect(model.reasoning).toBe(true); expect(model.input).toEqual(["text"]); - expect(model.compat?.thinkingFormat).toBe("zai"); - expect(model.compat?.reasoningContentField).toBe("reasoning_content"); - expect(model.compat?.supportsDeveloperRole).toBe(false); + expect(model.compatConfig?.thinkingFormat).toBe("zai"); + expect(model.compatConfig?.reasoningContentField).toBe("reasoning_content"); + expect(model.compatConfig?.supportsDeveloperRole).toBe(false); }); it("ships a bundled Qwen3.5-397B-A17B entry with vision input and no reasoning", () => { @@ -91,7 +91,7 @@ describe("Wafer Serverless provider", () => { expect(glm).toBeDefined(); expect(glm.provider).toBe("wafer-serverless"); expect(glm.baseUrl).toBe("https://pass.wafer.ai/v1"); - expect(glm.compat?.thinkingFormat).toBe("zai"); + expect(glm.compatConfig?.thinkingFormat).toBe("zai"); const qwen35 = getBundledModel<"openai-completions">("wafer-serverless", "Qwen3.5-397B-A17B"); expect(qwen35).toBeDefined(); @@ -104,8 +104,8 @@ describe("Wafer Serverless provider", () => { // `thinking: { type: "enabled" | "disabled" }`. Locked in explicitly so a // future regen with credentials cannot silently strip it (auto-detect // would mis-pick "openai" because the Wafer baseUrl/provider doesn't match - // the api.moonshot.ai / api.kimi.com URL patterns in `detectOpenAICompat`). - expect(kimi.compat?.thinkingFormat).toBe("zai"); + // the api.moonshot.ai / api.kimi.com URL patterns in `buildOpenAICompat`). + expect(kimi.compatConfig?.thinkingFormat).toBe("zai"); // Kimi-K2.6's retail Serverless rate per wafer.ai (= API cents × 0.0125): // $1.10 in / $4.80 out / $0.1125 cached. expect(kimi.cost).toEqual({ input: 1.1, output: 4.8, cacheRead: 0.1125, cacheWrite: 0 }); @@ -123,25 +123,25 @@ describe("Wafer Serverless provider", () => { expect(qwen37max.name).toBe("Qwen3.7 Max"); expect(qwen37max.reasoning).toBe(true); // qwen3.7-max routes to Alibaba upstream; native wire format is `enable_thinking`. - // The bundled entry leaves `thinkingFormat` unset so `detectOpenAICompat` picks "qwen" - // from the lowercase id at request time. - expect(qwen37max.compat?.thinkingFormat).toBeUndefined(); + // The bundled entry leaves `thinkingFormat` unset so the build-time detection + // in `buildOpenAICompat` picks "qwen" from the lowercase id. + expect(qwen37max.compatConfig?.thinkingFormat).toBeUndefined(); const dsFlash = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-flash"); expect(dsFlash).toBeDefined(); // DeepSeek V4 family uses `reasoning_effort`, not zai's `thinking: {type}`. - // Bundled entry must NOT pin `thinkingFormat: "zai"` — `detectOpenAICompat` - // auto-picks "openai" (default) from the deepseek-* id pattern at request time. - expect(dsFlash.compat?.thinkingFormat).toBeUndefined(); + // Bundled entry must NOT pin `thinkingFormat: "zai"` — `buildOpenAICompat` + // auto-picks "openai" (default) from the deepseek-* id pattern at build time. + expect(dsFlash.compatConfig?.thinkingFormat).toBeUndefined(); expect(dsFlash.contextWindow).toBe(1000000); expect(dsFlash.reasoning).toBe(true); - expect(dsFlash.compat?.reasoningContentField).toBe("reasoning_content"); + expect(dsFlash.compatConfig?.reasoningContentField).toBe("reasoning_content"); const dsPro = getBundledModel<"openai-completions">("wafer-serverless", "deepseek-v4-pro"); expect(dsPro).toBeDefined(); expect(dsPro.contextWindow).toBe(1000000); expect(dsPro.reasoning).toBe(true); - expect(dsPro.compat?.thinkingFormat).toBeUndefined(); + expect(dsPro.compatConfig?.thinkingFormat).toBeUndefined(); }); it("does not expose Serverless-only ids on the Wafer Pass catalog", () => { @@ -208,19 +208,19 @@ describe("Wafer dynamic discovery mapper", () => { const { models } = await manager.refresh("online"); const byId = new Map(models.map(m => [m.id, m as Model<"openai-completions">])); - expect(byId.get("GLM-fake")?.compat?.thinkingFormat).toBe("zai"); - expect(byId.get("Kimi-fake")?.compat?.thinkingFormat).toBe("zai"); - expect(byId.get("qwen-fake")?.compat?.thinkingFormat).toBe("qwen"); - // deepseek and unknown upstreams: thinkingFormat unset so detectOpenAICompat - // picks from the id pattern at request time (deepseek → "openai" effort). - expect(byId.get("deepseek-fake")?.compat?.thinkingFormat).toBeUndefined(); - expect(byId.get("mystery-fake")?.compat?.thinkingFormat).toBeUndefined(); + expect(byId.get("GLM-fake")?.compatConfig?.thinkingFormat).toBe("zai"); + expect(byId.get("Kimi-fake")?.compatConfig?.thinkingFormat).toBe("zai"); + expect(byId.get("qwen-fake")?.compatConfig?.thinkingFormat).toBe("qwen"); + // deepseek and unknown upstreams: thinkingFormat unset so `buildOpenAICompat` + // picks from the id pattern at build time (deepseek → "openai" effort). + expect(byId.get("deepseek-fake")?.compatConfig?.thinkingFormat).toBeUndefined(); + expect(byId.get("mystery-fake")?.compatConfig?.thinkingFormat).toBeUndefined(); // Non-reasoning entries never receive a thinkingFormat hint regardless of upstream. - expect(byId.get("nothink-fake")?.compat?.thinkingFormat).toBeUndefined(); + expect(byId.get("nothink-fake")?.compatConfig?.thinkingFormat).toBeUndefined(); expect(byId.get("nothink-fake")?.reasoning).toBe(false); // All entries keep reasoning_content as the canonical field for reasoning models. for (const id of ["GLM-fake", "Kimi-fake", "qwen-fake", "deepseek-fake", "mystery-fake"]) { - expect(byId.get(id)?.compat?.reasoningContentField).toBe("reasoning_content"); + expect(byId.get(id)?.compatConfig?.reasoningContentField).toBe("reasoning_content"); } }); diff --git a/packages/catalog/test/xai-oauth-bundle.test.ts b/packages/catalog/test/xai-oauth-bundle.test.ts index e09d7c816..de40e3820 100644 --- a/packages/catalog/test/xai-oauth-bundle.test.ts +++ b/packages/catalog/test/xai-oauth-bundle.test.ts @@ -1,7 +1,7 @@ import { describe, expect, it } from "bun:test"; import MODELS_JSON from "@oh-my-pi/pi-catalog/models.json" with { type: "json" }; import { buildXaiOAuthStaticSeed } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; -import type { Model } from "@oh-my-pi/pi-catalog/types"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; // Pins the invariant: bundled `models.json` carries every entry the runtime // curated catalog (XAI_OAUTH_CURATED_MODELS, surfaced via @@ -14,7 +14,7 @@ import type { Model } from "@oh-my-pi/pi-catalog/types"; // Failure here means: run `bun run generate-models` and commit the diff. describe("xai-oauth bundled catalog (regression)", () => { const bundled = - (MODELS_JSON as unknown as Record>>)["xai-oauth"] ?? {}; + (MODELS_JSON as unknown as Record>>)["xai-oauth"] ?? {}; const seed = buildXaiOAuthStaticSeed(); it("bundles every curated id", () => { diff --git a/packages/catalog/test/zhipu-compat.test.ts b/packages/catalog/test/zhipu-compat.test.ts index 343ccf265..9d53ba5c5 100644 --- a/packages/catalog/test/zhipu-compat.test.ts +++ b/packages/catalog/test/zhipu-compat.test.ts @@ -1,17 +1,17 @@ import { describe, expect, it } from "bun:test"; -import { detectOpenAICompat, resolveOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; +import { buildOpenAICompat } from "@oh-my-pi/pi-catalog/compat/openai"; import { zhipuCodingPlanModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; -import type { FetchImpl, Model } from "@oh-my-pi/pi-catalog/types"; +import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; /** * Resolver-branch coverage for the `isZhipu` path added by the * `zhipu-coding-plan` provider. Mirrors the shape of existing zai/cerebras * tests: assert the contract the provider relies on (zai thinking format, * disabled `reasoning_effort`, no `developer` role) so future refactors of - * `detectOpenAICompat` cannot silently regress the BigModel SKU. + * `buildOpenAICompat` cannot silently regress the BigModel SKU. */ -const baseModel: Omit, "provider" | "baseUrl"> = { +const baseModel: Omit, "provider" | "baseUrl"> = { api: "openai-completions", id: "glm-4.7", name: "GLM-4.7", @@ -22,7 +22,7 @@ const baseModel: Omit, "provider" | "baseUrl"> = { reasoning: true, }; -function zhipuByProvider(): Model<"openai-completions"> { +function zhipuByProvider(): ModelSpec<"openai-completions"> { return { ...baseModel, provider: "zhipu-coding-plan", @@ -30,7 +30,7 @@ function zhipuByProvider(): Model<"openai-completions"> { }; } -function zhipuByBaseUrl(): Model<"openai-completions"> { +function zhipuByBaseUrl(): ModelSpec<"openai-completions"> { return { ...baseModel, // Provider intentionally not "zhipu-coding-plan" — exercises the @@ -42,7 +42,7 @@ function zhipuByBaseUrl(): Model<"openai-completions"> { describe("openai-completions compat — zhipu-coding-plan branch", () => { it("forces zai thinking format and disables reasoning_effort / developer role", () => { - const compat = detectOpenAICompat(zhipuByProvider()); + const compat = buildOpenAICompat(zhipuByProvider()); expect(compat.thinkingFormat).toBe("zai"); expect(compat.supportsReasoningEffort).toBe(false); @@ -55,14 +55,14 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => { }); it("detects zhipu by baseUrl when provider id is custom", () => { - const compat = detectOpenAICompat(zhipuByBaseUrl()); + const compat = buildOpenAICompat(zhipuByBaseUrl()); expect(compat.thinkingFormat).toBe("zai"); expect(compat.supportsReasoningEffort).toBe(false); }); it("lets explicit model.compat overrides win at the resolver layer", () => { - const model: Model<"openai-completions"> = { + const model: ModelSpec<"openai-completions"> = { ...zhipuByProvider(), compat: { supportsDeveloperRole: true, @@ -70,7 +70,7 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => { thinkingFormat: "openai", }, }; - const resolved = resolveOpenAICompat(model); + const resolved = buildOpenAICompat(model); expect(resolved.supportsDeveloperRole).toBe(true); expect(resolved.supportsReasoningEffort).toBe(true); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 4f9e2e5c8..3586ce0b6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -3,7 +3,7 @@ ## [Unreleased] ### Added -- Added `streamIdleTimeoutMs`, `supportsLongPromptCacheRetention`, `requiresToolResultId`, and `replayUnsignedThinking` to the OpenAI `compat` schema so custom model entries can configure those provider-specific capabilities +- Added `supportsReasoningParams`, `alwaysSendMaxTokens`, `strictResponsesPairing`, and a recursive `whenThinking` overlay (alongside `streamIdleTimeoutMs`/`supportsLongPromptCacheRetention`/`requiresToolResultId`/`replayUnsignedThinking`) to the OpenAI/Anthropic `compat` schema so custom model entries can configure those provider-specific capabilities - New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("capacity: 5h → 2.40/5 accounts used (2.60× quota left)"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing. - Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently. - npm installs now execute a prebundled single-file entry: the published `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks. The on-repo manifest keeps `bin.omp` at `src/cli.ts` — release rewrites it via the `publishBin` override in `scripts/ci-release-publish.ts` — so source installs (`bun link`, `install.sh --source`) keep working without a build step diff --git a/packages/coding-agent/src/config/append-only-context-mode.ts b/packages/coding-agent/src/config/append-only-context-mode.ts index 71b4a53f0..cf8b8425e 100644 --- a/packages/coding-agent/src/config/append-only-context-mode.ts +++ b/packages/coding-agent/src/config/append-only-context-mode.ts @@ -4,14 +4,15 @@ import { hostMatchesUrl } from "@oh-my-pi/pi-catalog/hosts"; export interface AppendOnlyContextModel { provider: string; baseUrl: string; - compat?: object; + /** Verbatim sparse compat config (explicit user intent), never the resolved record. */ + compatConfig?: object; } function shouldAutoEnableAppendOnlyContext(model: AppendOnlyContextModel | null | undefined): boolean { if (!model) return false; if (model.provider === "deepseek") return true; if (hostMatchesUrl(model.baseUrl, "xiaomi")) return true; - return !!model.compat && "supportsStore" in model.compat && model.compat.supportsStore === true; + return !!model.compatConfig && "supportsStore" in model.compatConfig && model.compatConfig.supportsStore === true; } /** Resolves whether append-only context should be active for a model and setting. */ diff --git a/packages/coding-agent/src/config/model-discovery.ts b/packages/coding-agent/src/config/model-discovery.ts index 68021841a..c569144b4 100644 --- a/packages/coding-agent/src/config/model-discovery.ts +++ b/packages/coding-agent/src/config/model-discovery.ts @@ -7,12 +7,13 @@ */ import type { FetchImpl } from "@oh-my-pi/pi-ai"; import type { Api, Model } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModelReferenceIndex, resolveModelReference, stripBracketedModelIdAffixes, } from "@oh-my-pi/pi-catalog/identity"; -import { enrichModelThinking } from "@oh-my-pi/pi-catalog/model-thinking"; +import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; import { isRecord } from "@oh-my-pi/pi-utils"; import type { ProviderDiscovery } from "./models-config-schema"; @@ -86,7 +87,7 @@ export interface DiscoveryProviderConfig { api: Api; baseUrl?: string; headers?: Record; - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; discovery: ProviderDiscovery; optional?: boolean; } @@ -263,7 +264,7 @@ export async function discoverOllamaModels( ); return entries.map(entry => { const metadata = metadataById.get(entry.id); - return enrichModelThinking({ + return buildModel({ id: entry.id, name: entry.name, api: providerConfig.api, @@ -275,7 +276,7 @@ export async function discoverOllamaModels( contextWindow: metadata?.contextWindow ?? 128000, maxTokens: Math.min(metadata?.contextWindow ?? Number.POSITIVE_INFINITY, DISCOVERY_DEFAULT_MAX_TOKENS), headers: providerConfig.headers, - }); + } as ModelSpec); }); } @@ -336,7 +337,7 @@ export async function discoverLlamaCppModels( const id = item.id; if (!id) continue; discovered.push( - enrichModelThinking({ + buildModel({ id, name: id, api: providerConfig.api, @@ -356,7 +357,7 @@ export async function discoverLlamaCppModels( supportsDeveloperRole: false, supportsReasoningEffort: false, }, - }), + } as ModelSpec), ); } return discovered; @@ -389,7 +390,7 @@ export async function discoverOpenAIModelsList( const id = item.id; if (!id) continue; discovered.push( - enrichModelThinking({ + buildModel({ id, name: id, api: providerConfig.api, @@ -406,7 +407,7 @@ export async function discoverOpenAIModelsList( supportsDeveloperRole: false, supportsReasoningEffort: false, }, - }), + } as ModelSpec), ); } return discovered; @@ -471,7 +472,7 @@ export async function discoverProxyModels( stripBracketedModelIdAffixes(id) ?? id; discovered.push( - enrichModelThinking({ + buildModel({ id, name: displayName, api, @@ -499,7 +500,7 @@ export async function discoverProxyModels( supportsDeveloperRole: false, supportsReasoningEffort: false, }, - }), + } as ModelSpec), ); } return discovered; diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 56ea0d020..8dffa52f2 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -1,7 +1,8 @@ import * as path from "node:path"; import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai/api-registry"; -import type { Api, Context, Model, SimpleStreamOptions, ThinkingConfig } from "@oh-my-pi/pi-ai/types"; +import type { Api, Context, Model, ModelSpec, SimpleStreamOptions, ThinkingConfig } from "@oh-my-pi/pi-ai/types"; import type { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { isVertexExpressOpenAIUrl } from "@oh-my-pi/pi-catalog/hosts"; import { readModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { @@ -9,7 +10,6 @@ import { type ModelManagerOptions, type ModelRefreshStrategy, } from "@oh-my-pi/pi-catalog/model-manager"; -import { enrichModelThinking } from "@oh-my-pi/pi-catalog/model-thinking"; import { getBundledModels, getBundledProviders } from "@oh-my-pi/pi-catalog/models"; import { googleAntigravityModelManagerOptions, @@ -84,7 +84,7 @@ interface ProviderOverride { headers?: Record; apiKey?: string; authHeader?: boolean; - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; transport?: Model["transport"]; } @@ -110,19 +110,21 @@ export function mergeDiscoveredModel( providerOverride?: Pick, ): Model { if (existing) { - return { + return buildModel({ ...model, baseUrl: providerOverride?.baseUrl ?? model.baseUrl ?? existing.baseUrl, headers: existing.headers ? { ...existing.headers, ...model.headers } : model.headers, - }; + compat: model.compatConfig, + } as ModelSpec); } if (providerOverride) { - return { + return buildModel({ ...model, baseUrl: providerOverride.baseUrl ?? model.baseUrl, headers: providerOverride.headers ? { ...model.headers, ...providerOverride.headers } : model.headers, ...(providerOverride.transport !== undefined ? { transport: providerOverride.transport } : {}), - }; + compat: model.compatConfig, + } as ModelSpec); } return model; } @@ -292,6 +294,15 @@ function mergeCompat( return merged as TBase & TOverride; } +/** + * Project a built model back to spec shape for the model-manager/cache + * boundary: sparse compat comes from `compatConfig`, never from the resolved + * record. + */ +function toModelSpec(model: Model): ModelSpec { + return { ...model, compat: model.compatConfig } as ModelSpec; +} + /** * The patchable subset of `Model` fields shared by `modelOverrides` entries, * custom model definitions, and parsed custom-model overlays. `undefined` @@ -307,7 +318,7 @@ interface ModelPatch { maxTokens?: number; omitMaxOutputTokens?: boolean; headers?: Record; - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; contextPromotionTarget?: string; premiumMultiplier?: number; } @@ -340,16 +351,17 @@ function applyModelPatch(base: Model, patch: ModelPatch, transport: ModelTr cacheWrite: patch.cost.cacheWrite ?? base.cost.cacheWrite, }; } + let compat: ModelSpec["compat"]; if (transport === "merge") { if (patch.headers) { result.headers = { ...base.headers, ...patch.headers }; } - result.compat = mergeCompat(base.compat, patch.compat); + compat = mergeCompat(base.compatConfig, patch.compat); } else { result.headers = patch.headers; - result.compat = patch.compat; + compat = patch.compat; } - return enrichModelThinking(result); + return buildModel({ ...result, compat } as ModelSpec); } function applyModelOverride(model: Model, override: ModelOverride): Model { @@ -421,7 +433,7 @@ function buildCustomModelOverlay( providerHeaders: Record | undefined, providerApiKey: string | undefined, authHeader: boolean | undefined, - providerCompat: Model["compat"] | undefined, + providerCompat: ModelSpec["compat"] | undefined, providerAuth: ProviderAuthMode | undefined, modelDef: CustomModelDefinitionLike, ): CustomModelOverlay | undefined { @@ -465,7 +477,7 @@ function finalizeCustomModel(model: CustomModelOverlay, options: CustomModelBuil reference?.cost ?? (options.useDefaults ? { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } : undefined); const input = resolvedModel.input ?? reference?.input ?? (options.useDefaults ? ["text"] : undefined); - return enrichModelThinking({ + return buildModel({ id: resolvedModel.id, name: resolvedModel.name ?? (options.useDefaults ? resolvedModel.id : undefined), api: resolvedModel.api, @@ -480,11 +492,11 @@ function finalizeCustomModel(model: CustomModelOverlay, options: CustomModelBuil maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : undefined), headers: resolvedModel.headers, omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens, - compat: mergeCompat(reference?.compat, resolvedModel.compat), + compat: mergeCompat(reference?.compatConfig, resolvedModel.compat), contextPromotionTarget: resolvedModel.contextPromotionTarget, premiumMultiplier: resolvedModel.premiumMultiplier, isOAuth: resolvedModel.isOAuth, - } as Model); + } as ModelSpec); } function normalizeSuppressedSelector(selector: string): string { @@ -745,10 +757,10 @@ export class ModelRegistry { return models.map(m => { if (!providerOverride) return m; const withTransportOverride = this.#applyProviderTransportOverride(m, providerOverride); - return { + return buildModel({ ...withTransportOverride, - compat: mergeCompat(m.compat, providerOverride.compat), - }; + compat: mergeCompat(m.compatConfig, providerOverride.compat), + } as ModelSpec); }); }); } @@ -810,8 +822,13 @@ export class ModelRegistry { ? models.map(model => this.#applyProviderTransportOverride(model, providerOverride)) : models; const withCompat = providerOverride?.compat - ? withTransport.map(model => ({ ...model, compat: mergeCompat(model.compat, providerOverride.compat) })) - : withTransport; + ? withTransport.map(model => + buildModel({ + ...model, + compat: mergeCompat(model.compat, providerOverride.compat), + } as ModelSpec), + ) + : withTransport.map(model => buildModel(model)); cachedModels.push(...this.#applyProviderModelOverrides(providerId, withCompat)); } return { models: cachedModels, authoritativeFreshProviders }; @@ -835,7 +852,10 @@ export class ModelRegistry { providerConfig.provider, this.#normalizeDiscoverableModels( providerConfig, - this.#applyProviderCompat(providerConfig.compat, cache.models), + this.#applyProviderCompat( + providerConfig.compat, + cache.models.map(model => buildModel(model)), + ), ), ); cachedModels.push(...models); @@ -851,9 +871,11 @@ export class ModelRegistry { return cachedModels; } - #applyProviderCompat(compat: Model["compat"] | undefined, models: Model[]): Model[] { + #applyProviderCompat(compat: ModelSpec["compat"] | undefined, models: Model[]): Model[] { if (!compat) return models; - return models.map(model => ({ ...model, compat: mergeCompat(model.compat, compat) })); + return models.map(model => + buildModel({ ...model, compat: mergeCompat(model.compatConfig, compat) } as ModelSpec), + ); } #normalizeDiscoverableModels(providerConfig: DiscoveryProviderConfig, models: Model[]): Model[] { @@ -863,7 +885,14 @@ export class ModelRegistry { const contextLengthOverride = getOllamaContextLengthOverride(); return models.map(model => { - const normalized = model.api === "openai-completions" ? { ...model, api: "openai-responses" as const } : model; + const normalized = + model.api === "openai-completions" + ? buildModel({ + ...model, + api: "openai-responses" as const, + compat: model.compatConfig, + } as ModelSpec) + : model; if (contextLengthOverride === undefined) { return normalized; } @@ -1086,20 +1115,20 @@ export class ModelRegistry { models: cached?.models.map(model => model.id) ?? [], }); this.#lastDiscoveryWarnings.delete(providerConfig.provider); - return cached?.models ?? []; + return cached ? cached.models.map(model => buildModel(model)) : []; } } const providerId = providerConfig.provider; let discoveryError: string | undefined; - const fetchDynamicModels = async (): Promise[] | null> => { + const fetchDynamicModels = async (): Promise[] | null> => { try { const models = this.#applyProviderModelOverrides( providerId, await discoverModelsByProviderType(providerConfig, this.#discoveryContext()), ); this.#lastDiscoveryWarnings.delete(providerId); - return models; + return models.map(toModelSpec); } catch (error) { discoveryError = error instanceof Error ? error.message : String(error); return null; @@ -1881,7 +1910,7 @@ export class ModelRegistry { ); if (overlay) results.push(finalizeCustomModel(overlay, { useDefaults: true })); } - return results; + return results.map(toModelSpec); }, }; this.#runtimeModelManagers.set(providerName, { options: managerOptions, sourceId: sourceId ?? "" }); @@ -1955,7 +1984,7 @@ export interface ProviderConfigInput { api?: Api; streamSimple?: (model: Model, context: Context, options?: SimpleStreamOptions) => AssistantMessageEventStream; headers?: Record; - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; authHeader?: boolean; /** Streaming transport override — see {@link Model.transport}. */ transport?: Model["transport"]; @@ -1987,7 +2016,7 @@ export interface ProviderConfigInput { contextWindow: number; maxTokens: number; headers?: Record; - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; contextPromotionTarget?: string; premiumMultiplier?: number; }>; diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index 7e6484692..0e1debdcb 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -15,7 +15,8 @@ */ import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; -import type { Api, Effort, KnownProvider, Model } from "@oh-my-pi/pi-ai"; +import type { Api, Effort, KnownProvider, Model, ModelSpec } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { modelMatchesHost } from "@oh-my-pi/pi-catalog/hosts"; import { buildModelProviderPriorityRank } from "@oh-my-pi/pi-catalog/identity"; import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking"; @@ -177,10 +178,12 @@ function supportsUpstreamRouting(model: Model): boolean { function applyUpstreamRouting(model: Model, upstream: string): Model { const aggregatorModel = model as Model<"openai-completions">; const routing = { only: [upstream] }; - const compat = modelMatchesHost(model, "vercelAIGateway") - ? { ...aggregatorModel.compat, vercelGatewayRouting: routing } - : { ...aggregatorModel.compat, openRouterRouting: routing }; - return { ...model, compat } as Model; + return buildModel({ + ...model, + compat: modelMatchesHost(model, "vercelAIGateway") + ? { ...aggregatorModel.compatConfig, vercelGatewayRouting: routing } + : { ...aggregatorModel.compatConfig, openRouterRouting: routing }, + } as ModelSpec); } const kProviderModelIndex = Symbol("model-resolver.providerIndex"); diff --git a/packages/coding-agent/src/config/models-config-schema.ts b/packages/coding-agent/src/config/models-config-schema.ts index 0d83a263f..fdf9e6516 100644 --- a/packages/coding-agent/src/config/models-config-schema.ts +++ b/packages/coding-agent/src/config/models-config-schema.ts @@ -18,7 +18,7 @@ const ReasoningEffortMapSchema = z.object({ xhigh: z.string().optional(), }); -export const OpenAICompatSchema = z.object({ +const OpenAICompatFieldsSchema = z.object({ supportsStore: z.boolean().optional(), supportsDeveloperRole: z.boolean().optional(), supportsMultipleSystemMessages: z.boolean().optional(), @@ -46,11 +46,18 @@ export const OpenAICompatSchema = z.object({ toolStrictMode: z.enum(["all_strict", "none"]).optional(), streamIdleTimeoutMs: z.number().positive().optional(), supportsLongPromptCacheRetention: z.boolean().optional(), + supportsReasoningParams: z.boolean().optional(), + alwaysSendMaxTokens: z.boolean().optional(), + strictResponsesPairing: z.boolean().optional(), // anthropic-messages compat flags (same `compat` slot, per-api interpretation) requiresToolResultId: z.boolean().optional(), replayUnsignedThinking: z.boolean().optional(), }); +export const OpenAICompatSchema = OpenAICompatFieldsSchema.extend({ + whenThinking: OpenAICompatFieldsSchema.optional(), +}); + const EffortSchema = z.enum(["minimal", "low", "medium", "high", "xhigh"]); const ThinkingControlModeSchema = z.enum([ diff --git a/packages/coding-agent/src/config/models-config.ts b/packages/coding-agent/src/config/models-config.ts index 198d79815..e53fa92e4 100644 --- a/packages/coding-agent/src/config/models-config.ts +++ b/packages/coding-agent/src/config/models-config.ts @@ -2,7 +2,7 @@ * models.json config file handle and provider configuration validation. */ -import type { Api, Model } from "@oh-my-pi/pi-ai/types"; +import type { Api, ModelSpec } from "@oh-my-pi/pi-ai/types"; import { ConfigFile } from "./config-file"; import { type ModelsConfig, @@ -28,7 +28,7 @@ export interface ProviderValidationConfig { auth?: ProviderAuthMode; oauthConfigured?: boolean; discovery?: ProviderDiscovery; - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; disableStrictTools?: boolean; modelOverrides?: Record; models: ProviderValidationModel[]; diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 8abde9b06..48a5d9acf 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -22,6 +22,7 @@ import type { Context, ImageContent, Model, + ModelSpec, ProviderResponseMetadata, SimpleStreamOptions, Static, @@ -1168,7 +1169,7 @@ export interface ProviderModelConfig { /** Custom headers for this model. */ headers?: Record; /** OpenAI compatibility settings. */ - compat?: Model["compat"]; + compat?: ModelSpec["compat"]; } /** Extension factory function type. Supports both sync and async initialization. */ diff --git a/packages/coding-agent/test/acp-agent.test.ts b/packages/coding-agent/test/acp-agent.test.ts index 85d56324e..6fabd6c64 100644 --- a/packages/coding-agent/test/acp-agent.test.ts +++ b/packages/coding-agent/test/acp-agent.test.ts @@ -18,6 +18,7 @@ import { zSessionNotification, } from "@agentclientprotocol/sdk/dist/schema/zod.gen.js"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { ACP_BOOTSTRAP_RACE_GUARD_MS, @@ -32,7 +33,7 @@ import { getConfigRootDir, setAgentDir } from "@oh-my-pi/pi-utils"; import { expectAcpStructure } from "./helpers/acp-schema"; const TEST_MODELS: Model[] = [ - { + buildModel({ id: "claude-sonnet-4-20250514", name: "Claude Sonnet", api: "anthropic-messages", @@ -43,8 +44,8 @@ const TEST_MODELS: Model[] = [ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, - }, - { + }), + buildModel({ id: "gpt-5.4", name: "GPT-5.4", api: "openai-responses", @@ -55,7 +56,7 @@ const TEST_MODELS: Model[] = [ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, - }, + }), ]; function makeAssistantMessage(text: string, thinking?: string) { diff --git a/packages/coding-agent/test/acp-event-mapper.test.ts b/packages/coding-agent/test/acp-event-mapper.test.ts index 1cab03a00..a937d0612 100644 --- a/packages/coding-agent/test/acp-event-mapper.test.ts +++ b/packages/coding-agent/test/acp-event-mapper.test.ts @@ -5,6 +5,7 @@ import path from "node:path"; import type { AgentSideConnection, SessionNotification } from "@agentclientprotocol/sdk"; import { zSessionNotification } from "@agentclientprotocol/sdk/dist/schema/zod.gen.js"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { AcpAgent } from "@oh-my-pi/pi-coding-agent/modes/acp/acp-agent"; import { buildToolCallStartUpdate, @@ -46,7 +47,7 @@ function expectAcpNotifications(updates: SessionNotification[]): void { } } -const TEST_MODEL: Model = { +const TEST_MODEL: Model = buildModel({ id: "claude-sonnet-4-20250514", name: "Claude Sonnet", api: "anthropic-messages", @@ -57,7 +58,7 @@ const TEST_MODEL: Model = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); class ReplayTestSession { sessionManager: SessionManager; diff --git a/packages/coding-agent/test/acp-initialize-conformance.test.ts b/packages/coding-agent/test/acp-initialize-conformance.test.ts index eddd6bee6..8df0a2a37 100644 --- a/packages/coding-agent/test/acp-initialize-conformance.test.ts +++ b/packages/coding-agent/test/acp-initialize-conformance.test.ts @@ -10,6 +10,7 @@ import * as path from "node:path"; import type { AgentSideConnection, InitializeRequest } from "@agentclientprotocol/sdk"; import { zInitializeResponse } from "@agentclientprotocol/sdk/dist/schema/zod.gen.js"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { AcpAgent } from "@oh-my-pi/pi-coding-agent/modes/acp/acp-agent"; import { ACP_TERMINAL_AUTH_FLAG, prepareAcpTerminalAuthArgs } from "@oh-my-pi/pi-coding-agent/modes/acp/terminal-auth"; import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -18,7 +19,7 @@ import { getConfigRootDir, setAgentDir, VERSION } from "@oh-my-pi/pi-utils"; import { expectAcpStructure } from "./helpers/acp-schema"; const TEST_MODELS: Model[] = [ - { + buildModel({ id: "claude-sonnet-4-20250514", name: "Claude Sonnet", api: "anthropic-messages", @@ -29,7 +30,7 @@ const TEST_MODELS: Model[] = [ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, - }, + }), ]; class FakeAgentSession { diff --git a/packages/coding-agent/test/acp-lazy-startup.test.ts b/packages/coding-agent/test/acp-lazy-startup.test.ts index afdabaee0..d4e16b743 100644 --- a/packages/coding-agent/test/acp-lazy-startup.test.ts +++ b/packages/coding-agent/test/acp-lazy-startup.test.ts @@ -11,6 +11,7 @@ import { type SessionNotification, } from "@agentclientprotocol/sdk"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { createAcpConnection } from "@oh-my-pi/pi-coding-agent/modes/acp/acp-mode"; import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -18,7 +19,7 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { TempDir } from "@oh-my-pi/pi-utils"; -const TEST_MODEL: Model = { +const TEST_MODEL: Model = buildModel({ id: "claude-sonnet-4-20250514", name: "Claude Sonnet", api: "anthropic-messages", @@ -29,7 +30,7 @@ const TEST_MODEL: Model = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); function emptyWorkspaceTree(cwd: string) { return { rootPath: cwd, rendered: ".\n", truncated: false, totalLines: 1, agentsMdFiles: [] }; diff --git a/packages/coding-agent/test/agent-session-mcp-discovery.test.ts b/packages/coding-agent/test/agent-session-mcp-discovery.test.ts index 3c5d66ddf..f8a56d49e 100644 --- a/packages/coding-agent/test/agent-session-mcp-discovery.test.ts +++ b/packages/coding-agent/test/agent-session-mcp-discovery.test.ts @@ -4,6 +4,7 @@ import * as os from "node:os"; import * as path from "node:path"; import { Agent, type AgentTool, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { Effort, type Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { CustomTool } from "@oh-my-pi/pi-coding-agent/extensibility/custom-tools/types"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -11,7 +12,7 @@ import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manage import * as z from "zod/v4"; function createModel(): Model<"openai-responses"> { - return { + return buildModel({ id: "mock", name: "mock", api: "openai-responses", @@ -22,7 +23,7 @@ function createModel(): Model<"openai-responses"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 8192, maxTokens: 2048, - }; + }); } function createBasicTool(name: string, label: string): AgentTool { diff --git a/packages/coding-agent/test/agent-session-message-pipeline.test.ts b/packages/coding-agent/test/agent-session-message-pipeline.test.ts index 8ac1f9b20..3d54a4580 100644 --- a/packages/coding-agent/test/agent-session-message-pipeline.test.ts +++ b/packages/coding-agent/test/agent-session-message-pipeline.test.ts @@ -1,14 +1,17 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { Agent, type AgentMessage } from "@oh-my-pi/pi-agent-core"; import { + type Api, clearCustomApis, type Message, type Model, + type ModelSpec, registerCustomApi, type SimpleStreamOptions, type TextContent, } from "@oh-my-pi/pi-ai"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { convertToLlm, wrapSteeringForModel } from "@oh-my-pi/pi-coding-agent/session/messages"; @@ -179,7 +182,7 @@ describe("AgentSession message pipeline", () => { return stream; }); - const model = { + const model = buildModel({ id: "side-model", name: "Side Model", api, @@ -190,7 +193,7 @@ describe("AgentSession message pipeline", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 4096, maxTokens: 1024, - } satisfies Model; + } as ModelSpec) as Model; const session = new AgentSession({ agent: new Agent({ initialState: { @@ -232,7 +235,7 @@ describe("AgentSession message pipeline", () => { return stream; }); - const model = { + const model = buildModel({ id: "anthropic/claude-sonnet-4", name: "OpenRouter Model", api, @@ -243,7 +246,7 @@ describe("AgentSession message pipeline", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 4096, maxTokens: 1024, - } satisfies Model; + } as ModelSpec) as Model; const session = new AgentSession({ agent: new Agent({ initialState: { diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 56355fb92..80571da58 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -4,6 +4,7 @@ import { scheduler } from "node:timers/promises"; import { Agent } from "@oh-my-pi/pi-agent-core"; import { type AssistantMessage, Effort, type Model } from "@oh-my-pi/pi-ai"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -826,7 +827,7 @@ describe("AgentSession retry fallback", () => { if (!primaryModel) { throw new Error("Expected bundled OpenAI test model to exist"); } - const cachedModel: Model<"ollama-chat"> = { + const cachedModel: Model<"ollama-chat"> = buildModel({ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", api: "ollama-chat", @@ -837,7 +838,7 @@ describe("AgentSession retry fallback", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 384_000, - }; + }); writeModelCache("ollama-cloud", Date.now(), [cachedModel], true, "", path.join(tempDir.path(), "models.db")); modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.json")); diff --git a/packages/coding-agent/test/agent-session-ssh-refresh.test.ts b/packages/coding-agent/test/agent-session-ssh-refresh.test.ts index ed74523a1..2934fa363 100644 --- a/packages/coding-agent/test/agent-session-ssh-refresh.test.ts +++ b/packages/coding-agent/test/agent-session-ssh-refresh.test.ts @@ -1,6 +1,7 @@ import { afterEach, describe, expect, it, spyOn } from "bun:test"; import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { reset as resetCapabilities } from "@oh-my-pi/pi-coding-agent/capability"; import { type SSHHost, sshCapability } from "@oh-my-pi/pi-coding-agent/capability/ssh"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; @@ -13,7 +14,7 @@ import { loadSshTool, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { getSSHConfigPath, TempDir } from "@oh-my-pi/pi-utils"; function createModel(): Model<"openai-responses"> { - return { + return buildModel({ id: "mock", name: "mock", api: "openai-responses", @@ -24,7 +25,7 @@ function createModel(): Model<"openai-responses"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 8192, maxTokens: 2048, - }; + }); } describe("AgentSession SSH tool refresh", () => { diff --git a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts index 18e5066fb..e70e1fe7c 100644 --- a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts +++ b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts @@ -1,6 +1,7 @@ import { afterEach, describe, expect, it, setSystemTime } from "bun:test"; import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { CustomTool } from "@oh-my-pi/pi-coding-agent/extensibility/custom-tools/types"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -13,7 +14,7 @@ import * as z from "zod/v4"; // and forces a full prefix re-encode on the next request. function createModel(): Model<"openai-responses"> { - return { + return buildModel({ id: "mock", name: "mock", api: "openai-responses", @@ -24,7 +25,7 @@ function createModel(): Model<"openai-responses"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 8192, maxTokens: 2048, - }; + }); } function createBasicTool(name: string, label: string, description = `${label} tool`): AgentTool { diff --git a/packages/coding-agent/test/append-only-context-mode.test.ts b/packages/coding-agent/test/append-only-context-mode.test.ts index 381c3215c..4593d039c 100644 --- a/packages/coding-agent/test/append-only-context-mode.test.ts +++ b/packages/coding-agent/test/append-only-context-mode.test.ts @@ -33,7 +33,7 @@ describe("shouldEnableAppendOnlyContext", () => { expect( shouldEnableAppendOnlyContext("auto", { ...GENERIC_PROXY, - compat: { supportsStore: true }, + compatConfig: { supportsStore: true }, }), ).toBe(true); }); diff --git a/packages/coding-agent/test/debug/raw-sse-buffer.test.ts b/packages/coding-agent/test/debug/raw-sse-buffer.test.ts index 510810c35..ad5e66b04 100644 --- a/packages/coding-agent/test/debug/raw-sse-buffer.test.ts +++ b/packages/coding-agent/test/debug/raw-sse-buffer.test.ts @@ -1,12 +1,13 @@ import { describe, expect, it } from "bun:test"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { RawSseDebugBuffer, rawSseRecordLines, resolveRawSseDebugBuffer, } from "@oh-my-pi/pi-coding-agent/debug/raw-sse-buffer"; -const model: Model<"anthropic-messages"> = { +const model: Model<"anthropic-messages"> = buildModel({ id: "claude-test", name: "Claude Test", api: "anthropic-messages", @@ -17,7 +18,7 @@ const model: Model<"anthropic-messages"> = { cost: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); describe("RawSseDebugBuffer", () => { it("records response metadata and raw SSE frame lines for diagnostics", () => { diff --git a/packages/coding-agent/test/debug/raw-sse-report-bundle.test.ts b/packages/coding-agent/test/debug/raw-sse-report-bundle.test.ts index 3137d9e1a..a5366213e 100644 --- a/packages/coding-agent/test/debug/raw-sse-report-bundle.test.ts +++ b/packages/coding-agent/test/debug/raw-sse-report-bundle.test.ts @@ -3,11 +3,12 @@ import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { RawSseDebugBuffer } from "@oh-my-pi/pi-coding-agent/debug/raw-sse-buffer"; import { createReportBundle } from "@oh-my-pi/pi-coding-agent/debug/report-bundle"; import { getConfigRootDir, setAgentDir } from "@oh-my-pi/pi-utils"; -const model: Model<"anthropic-messages"> = { +const model: Model<"anthropic-messages"> = buildModel({ id: "claude-test", name: "Claude Test", api: "anthropic-messages", @@ -18,7 +19,7 @@ const model: Model<"anthropic-messages"> = { cost: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200_000, maxTokens: 8_192, -}; +}); const originalAgentDir = process.env.PI_CODING_AGENT_DIR; const originalXdgStateHome = process.env.XDG_STATE_HOME; diff --git a/packages/coding-agent/test/issue-980-bedrock-priority.test.ts b/packages/coding-agent/test/issue-980-bedrock-priority.test.ts index f06f71480..116d93b35 100644 --- a/packages/coding-agent/test/issue-980-bedrock-priority.test.ts +++ b/packages/coding-agent/test/issue-980-bedrock-priority.test.ts @@ -1,5 +1,6 @@ import { describe, expect, test } from "bun:test"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { resolveCliModel, resolveModelFromSettings, @@ -8,7 +9,7 @@ import { import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; function model(provider: string, id: string): Model<"anthropic-messages"> { - return { + return buildModel({ provider, id, name: `${provider}/${id}`, @@ -19,7 +20,7 @@ function model(provider: string, id: string): Model<"anthropic-messages"> { cost: { input: 1, output: 1, cacheRead: 0, cacheWrite: 0 }, contextWindow: 200000, maxTokens: 8192, - }; + }); } describe("issue #980 provider-qualified model resolution", () => { diff --git a/packages/coding-agent/test/issue-985-subagent-auth-fallback.test.ts b/packages/coding-agent/test/issue-985-subagent-auth-fallback.test.ts index 7ccfcfadb..b17ecddb2 100644 --- a/packages/coding-agent/test/issue-985-subagent-auth-fallback.test.ts +++ b/packages/coding-agent/test/issue-985-subagent-auth-fallback.test.ts @@ -1,5 +1,6 @@ import { describe, expect, test } from "bun:test"; import type { Api, Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { kNoAuth } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { type ModelLookupRegistry, @@ -20,7 +21,7 @@ import { * definition has working auth — the parent turn is using it). */ -const parentModel: Model = { +const parentModel: Model = buildModel({ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", api: "openai-completions", @@ -31,9 +32,9 @@ const parentModel: Model = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 8192, -}; +}); -const unauthedTaskModel: Model = { +const unauthedTaskModel: Model = buildModel({ id: "qwen3.6-plus-free", name: "Qwen3.6 Plus Free", api: "openai-completions", @@ -44,9 +45,9 @@ const unauthedTaskModel: Model = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 8192, -}; +}); -const sharedModel: Model = { +const sharedModel: Model = buildModel({ id: "shared-id", name: "Shared", api: "openai-completions", @@ -57,7 +58,7 @@ const sharedModel: Model = { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 8192, -}; +}); interface MockRegistryOptions { models: Model[]; diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index a03de048a..7023fc5bb 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -3,6 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { Effort, type FetchImpl, type Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { kNoAuth, ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { resetSettingsForTest } from "@oh-my-pi/pi-coding-agent/config/settings"; @@ -283,7 +284,7 @@ describe("ModelRegistry runtime discovery", () => { }, }); writeCachedOllamaModels([ - { + buildModel({ id: "phi4-mini", name: "phi4-mini", api: "openai-completions", @@ -294,7 +295,7 @@ describe("ModelRegistry runtime discovery", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 8192, - }, + }), ]); const registry = new ModelRegistry(authStorage, modelsJsonPath); diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 89e987417..358cd3f01 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -3,6 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { Effort, type FetchImpl, type Model, type OpenAICompat, type ThinkingConfig } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; @@ -115,8 +116,8 @@ describe("ModelRegistry", () => { function getOpenAICompat(model: Model | undefined): OpenAICompat | undefined { // All custom-model compat overrides flow through OpenAICompatSchema regardless of // the underlying api ("openai-completions" vs "openai-responses"), so we can read - // the field for any model in this fixture. - return model?.compat as OpenAICompat | undefined; + // the configured (sparse) compat for any model in this fixture. + return model?.compatConfig as OpenAICompat | undefined; } /** Create a baseUrl-only override (no custom models) */ @@ -1676,7 +1677,9 @@ describe("ModelRegistry", () => { expect(models.length).toBeGreaterThan(0); for (const model of models) { - expect((model.compat as { disableStrictTools?: boolean } | undefined)?.disableStrictTools).toBeUndefined(); + expect( + (model.compatConfig as { disableStrictTools?: boolean } | undefined)?.disableStrictTools, + ).toBeUndefined(); } }); @@ -1846,7 +1849,7 @@ describe("ModelRegistry", () => { "openai", Date.now(), [ - { + buildModel({ id: "gpt-4o", name: "GPT-4o", api: "openai-completions", @@ -1857,7 +1860,7 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 222_222, // UNK_CONTEXT_WINDOW maxTokens: 8_888, // UNK_MAX_TOKENS - }, + }), ], true, cacheDbPath, @@ -1874,7 +1877,7 @@ describe("ModelRegistry", () => { }); test("loads cached standard provider discovery models on startup", () => { - const cachedModel: Model<"ollama-chat"> = { + const cachedModel: Model<"ollama-chat"> = buildModel({ id: "deepseek-v4-pro", name: "DeepSeek V4 Pro", api: "ollama-chat", @@ -1885,7 +1888,7 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 384_000, - }; + }); writeModelCache("ollama-cloud", Date.now(), [cachedModel], true, "", cacheDbPath); const registry = new ModelRegistry(authStorage, modelsJsonPath); @@ -1895,7 +1898,7 @@ describe("ModelRegistry", () => { test("loads cached special provider discovery models on startup", () => { const cachedModels: Model[] = [ - { + buildModel({ id: "gemini-3.5-flash-low", name: "Gemini 3.5 Flash Low", api: "google-gemini-cli", @@ -1906,8 +1909,8 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 8_192, - }, - { + }), + buildModel({ id: "gemini-3.5-flash", name: "Gemini 3.5 Flash", api: "google-gemini-cli", @@ -1918,8 +1921,8 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 16_384, - }, - { + }), + buildModel({ id: "gpt-5.4-codex-pro", name: "GPT-5.4 Codex Pro", api: "openai-codex-responses", @@ -1930,7 +1933,7 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 400_000, maxTokens: 128_000, - }, + }), ]; for (const cachedModel of cachedModels) { writeModelCache(cachedModel.provider, Date.now(), [cachedModel], true, "", cacheDbPath); @@ -1944,7 +1947,7 @@ describe("ModelRegistry", () => { }); test("replaces bundled google-vertex models with authoritative Vertex project discovery", () => { - const cachedModel: Model<"openai-completions"> = { + const cachedModel: Model<"openai-completions"> = buildModel({ id: "zai-org/glm-4.7-maas", name: "GLM-4.7", api: "openai-completions", @@ -1955,7 +1958,7 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 222_222, maxTokens: 8_888, - }; + }); writeModelCache("google-vertex", Date.now(), [cachedModel], true, "", cacheDbPath); const registry = new ModelRegistry(authStorage, modelsJsonPath); @@ -1966,7 +1969,7 @@ describe("ModelRegistry", () => { }); test("does not re-add bundled synthetic models after authoritative cache load", () => { - const cachedModel: Model<"openai-completions"> = { + const cachedModel: Model<"openai-completions"> = buildModel({ id: "hf:zai-org/GLM-5.1", name: "GLM 5.1", api: "openai-completions", @@ -1977,7 +1980,7 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128_000, maxTokens: 8_192, - }; + }); writeModelCache("synthetic", Date.now(), [cachedModel], true, "authoritative:test", cacheDbPath); const registry = new ModelRegistry(authStorage, modelsJsonPath); @@ -2002,7 +2005,7 @@ describe("ModelRegistry", () => { }); test("keeps bundled google-vertex fallback when cached project catalog is non-authoritative", () => { - const cachedModel: Model<"openai-completions"> = { + const cachedModel: Model<"openai-completions"> = buildModel({ id: "zai-org/glm-4.7-maas", name: "GLM-4.7", api: "openai-completions", @@ -2013,7 +2016,7 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 222_222, maxTokens: 8_888, - }; + }); writeModelCache("google-vertex", Date.now(), [cachedModel], false, "", cacheDbPath); const registry = new ModelRegistry(authStorage, modelsJsonPath); @@ -2024,7 +2027,7 @@ describe("ModelRegistry", () => { }); test("keeps bundled google-vertex fallback when cached project catalog is stale", () => { - const cachedModel: Model<"openai-completions"> = { + const cachedModel: Model<"openai-completions"> = buildModel({ id: "zai-org/glm-4.7-maas", name: "GLM-4.7", api: "openai-completions", @@ -2035,7 +2038,7 @@ describe("ModelRegistry", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 222_222, maxTokens: 8_888, - }; + }); // 25h old > 24h TTL → cache.fresh === false even though authoritative === true. const staleTimestamp = Date.now() - 25 * 60 * 60 * 1000; writeModelCache("google-vertex", staleTimestamp, [cachedModel], true, "", cacheDbPath); diff --git a/packages/coding-agent/test/model-resolver.test.ts b/packages/coding-agent/test/model-resolver.test.ts index 8dd1fea1c..c7aa3616a 100644 --- a/packages/coding-agent/test/model-resolver.test.ts +++ b/packages/coding-agent/test/model-resolver.test.ts @@ -1,5 +1,6 @@ import { describe, expect, test } from "bun:test"; import { type Api, Effort, type Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { expandRoleAlias, parseModelPattern, @@ -15,7 +16,7 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; // Mock models for testing const mockModels: Model<"anthropic-messages">[] = [ - { + buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -31,8 +32,8 @@ const mockModels: Model<"anthropic-messages">[] = [ cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }, contextWindow: 200000, maxTokens: 8192, - }, - { + }), + buildModel({ id: "gpt-4o", name: "GPT-4o", api: "anthropic-messages", // Using same type for simplicity @@ -43,12 +44,12 @@ const mockModels: Model<"anthropic-messages">[] = [ cost: { input: 5, output: 15, cacheRead: 0.5, cacheWrite: 5 }, contextWindow: 128000, maxTokens: 4096, - }, + }), ]; // Mock OpenRouter models with colons in IDs -const mockOpenRouterModels: Model<"anthropic-messages">[] = [ - { +const mockOpenRouterModels: Model[] = [ + buildModel({ id: "qwen/qwen3-coder:exacto", name: "Qwen3 Coder Exacto", api: "anthropic-messages", @@ -64,8 +65,8 @@ const mockOpenRouterModels: Model<"anthropic-messages">[] = [ cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 }, contextWindow: 128000, maxTokens: 8192, - }, - { + }), + buildModel({ id: "openai/gpt-4o:extended", name: "GPT-4o Extended", api: "anthropic-messages", @@ -76,11 +77,11 @@ const mockOpenRouterModels: Model<"anthropic-messages">[] = [ cost: { input: 5, output: 15, cacheRead: 0.5, cacheWrite: 5 }, contextWindow: 128000, maxTokens: 4096, - }, - { + }), + buildModel({ id: "z-ai/glm-4.7", name: "GLM 4.7", - api: "anthropic-messages", + api: "openai-completions", provider: "openrouter", baseUrl: "https://openrouter.ai/api/v1", reasoning: true, @@ -93,11 +94,11 @@ const mockOpenRouterModels: Model<"anthropic-messages">[] = [ cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 }, contextWindow: 128000, maxTokens: 8192, - }, + }), ]; const mockProviderOverlapModels: Model<"anthropic-messages">[] = [ - { + buildModel({ id: "kimi-k2.5", name: "Kimi K2.5", api: "anthropic-messages", @@ -108,8 +109,8 @@ const mockProviderOverlapModels: Model<"anthropic-messages">[] = [ cost: { input: 2, output: 6, cacheRead: 0.2, cacheWrite: 2 }, contextWindow: 128000, maxTokens: 8192, - }, - { + }), + buildModel({ id: "moonshotai/kimi-k2.5", name: "Kimi K2.5 (OpenRouter)", api: "anthropic-messages", @@ -120,11 +121,11 @@ const mockProviderOverlapModels: Model<"anthropic-messages">[] = [ cost: { input: 2.2, output: 6.2, cacheRead: 0.22, cacheWrite: 2.2 }, contextWindow: 128000, maxTokens: 8192, - }, + }), ]; const mockCodexOverlapModels: Model<"anthropic-messages">[] = [ - { + buildModel({ id: "gpt-5.3-codex", name: "GPT-5.3 Codex", api: "anthropic-messages", @@ -140,8 +141,8 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [ cost: { input: 1.5, output: 6, cacheRead: 0.15, cacheWrite: 1.5 }, contextWindow: 200000, maxTokens: 8192, - }, - { + }), + buildModel({ id: "gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", api: "anthropic-messages", @@ -157,11 +158,11 @@ const mockCodexOverlapModels: Model<"anthropic-messages">[] = [ cost: { input: 1, output: 4, cacheRead: 0.1, cacheWrite: 1 }, contextWindow: 200000, maxTokens: 8192, - }, + }), ]; function createOpusModel(provider: string, id: string, name: string): Model<"anthropic-messages"> { - return { + return buildModel({ id, name, api: "anthropic-messages", @@ -177,11 +178,11 @@ function createOpusModel(provider: string, id: string, name: string): Model<"ant cost: { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 }, contextWindow: 200000, maxTokens: 32000, - }; + }); } const canonicalVariantModels: Model<"anthropic-messages">[] = [ - { + buildModel({ id: "claude-sonnet-4-5", name: "Claude Sonnet 4.5", api: "anthropic-messages", @@ -197,8 +198,8 @@ const canonicalVariantModels: Model<"anthropic-messages">[] = [ cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }, contextWindow: 200000, maxTokens: 8192, - }, - { + }), + buildModel({ id: "anthropic/claude-sonnet-4.5", name: "Claude Sonnet 4.5 (Copilot)", api: "anthropic-messages", @@ -214,7 +215,7 @@ const canonicalVariantModels: Model<"anthropic-messages">[] = [ cost: { input: 3, output: 15, cacheRead: 0.3, cacheWrite: 3.75 }, contextWindow: 200000, maxTokens: 8192, - }, + }), ]; const canonicalRegistry = { @@ -791,7 +792,7 @@ describe("resolveCliModel", () => { // Simulates the zai/glm-5 bug: vercel-ai-gateway has id="zai/glm-5", // zai has id="glm-5". Input "zai/glm-5" should resolve to provider=zai. const ambiguousModels: Model<"anthropic-messages">[] = [ - { + buildModel({ id: "zai/glm-5", name: "GLM-5 (Vercel)", api: "anthropic-messages", @@ -802,8 +803,8 @@ describe("resolveCliModel", () => { cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 }, contextWindow: 128000, maxTokens: 4096, - }, - { + }), + buildModel({ id: "glm-5", name: "GLM-5", api: "anthropic-messages", @@ -814,7 +815,7 @@ describe("resolveCliModel", () => { cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 }, contextWindow: 128000, maxTokens: 4096, - }, + }), ]; const registry = { getAll: () => ambiguousModels, @@ -949,10 +950,10 @@ describe("provider routing selector (@upstream)", () => { }); test("routes Vercel AI Gateway models via vercelGatewayRouting", () => { - const gatewayModel: Model<"anthropic-messages"> = { + const gatewayModel: Model<"openai-completions"> = buildModel({ id: "zai/glm-4.7", name: "GLM 4.7 (Gateway)", - api: "anthropic-messages", + api: "openai-completions", provider: "vercel-ai-gateway", baseUrl: "https://ai-gateway.vercel.sh/v1", reasoning: true, @@ -960,7 +961,7 @@ describe("provider routing selector (@upstream)", () => { cost: { input: 1, output: 2, cacheRead: 0.1, cacheWrite: 1 }, contextWindow: 128000, maxTokens: 8192, - }; + }); const result = parseModelPattern("vercel-ai-gateway/zai/glm-4.7@cerebras", [gatewayModel]); expect(result.model?.id).toBe("zai/glm-4.7"); expect( @@ -971,7 +972,7 @@ describe("provider routing selector (@upstream)", () => { }); test("does not split a model id that legitimately ends in @ (Vertex)", () => { - const vertexModel: Model<"anthropic-messages"> = { + const vertexModel: Model<"anthropic-messages"> = buildModel({ id: "claude-opus-4-8@default", name: "Claude Opus 4.8", api: "anthropic-messages", @@ -982,7 +983,7 @@ describe("provider routing selector (@upstream)", () => { cost: { input: 15, output: 75, cacheRead: 1.5, cacheWrite: 18.75 }, contextWindow: 200000, maxTokens: 32000, - }; + }); const result = parseModelPattern("claude-opus-4-8@default", [vertexModel]); expect(result.model?.id).toBe("claude-opus-4-8@default"); expect(result.upstream).toBeUndefined(); diff --git a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts index 1c5762049..b7c4c44dd 100644 --- a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts +++ b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts @@ -1,6 +1,7 @@ import { beforeAll, describe, expect, test, vi } from "bun:test"; import { stripVTControlCharacters } from "node:util"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import type { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; @@ -34,7 +35,7 @@ function createSelector(model: Model, settings: Settings): ModelSelectorComponen } function createOllamaCloudModel(id: string): Model { - return { + return buildModel({ id, name: "DeepSeek V4 Pro", api: "ollama-chat", @@ -45,10 +46,10 @@ function createOllamaCloudModel(id: string): Model { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 1_000_000, maxTokens: 8192, - }; + }); } function createContextTestModel(id: string, contextWindow: number): Model { - return { + return buildModel({ id, name: id, api: "ollama-chat", @@ -59,7 +60,7 @@ function createContextTestModel(id: string, contextWindow: number): Model { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow, maxTokens: 1024, - }; + }); } function createScopedSelector( diff --git a/packages/coding-agent/test/sdk-mcp-discovery.test.ts b/packages/coding-agent/test/sdk-mcp-discovery.test.ts index 4f59734c0..97ecbf975 100644 --- a/packages/coding-agent/test/sdk-mcp-discovery.test.ts +++ b/packages/coding-agent/test/sdk-mcp-discovery.test.ts @@ -4,6 +4,7 @@ import * as os from "node:os"; import * as path from "node:path"; import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { AuthStorage, Effort, type Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; @@ -29,7 +30,7 @@ function createMcpCustomTool(name: string, serverName: string, mcpToolName: stri } function createReasoningModel(): Model<"openai-responses"> { - return { + return buildModel({ id: "mock-reasoning", name: "mock-reasoning", api: "openai-responses", @@ -41,7 +42,7 @@ function createReasoningModel(): Model<"openai-responses"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 8192, maxTokens: 2048, - }; + }); } const oldSessionMtime = new Date("2000-01-01T00:00:00.000Z"); diff --git a/packages/coding-agent/test/slash-commands/force.test.ts b/packages/coding-agent/test/slash-commands/force.test.ts index 3c47db41b..cf5bced02 100644 --- a/packages/coding-agent/test/slash-commands/force.test.ts +++ b/packages/coding-agent/test/slash-commands/force.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it, vi } from "bun:test"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; import { executeBuiltinSlashCommand } from "@oh-my-pi/pi-coding-agent/slash-commands/builtin-registry"; import { buildNamedToolChoice } from "@oh-my-pi/pi-coding-agent/utils/tool-choice"; @@ -98,7 +99,7 @@ describe("/force slash command", () => { }); it("builds a named Ollama choice for local forced tools", () => { - const model = { + const model = buildModel({ id: "ggml-org/gemma-3-1b-it/GGUF", name: "Gemma 3 1B", api: "ollama-chat", @@ -109,7 +110,7 @@ describe("/force slash command", () => { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 32_768, maxTokens: 8_192, - } satisfies Model<"ollama-chat">; + }) satisfies Model<"ollama-chat">; expect(buildNamedToolChoice("write", model)).toEqual({ type: "function", name: "write" }); }); diff --git a/packages/coding-agent/test/tools/inspect-image.test.ts b/packages/coding-agent/test/tools/inspect-image.test.ts index 8365ce9c2..7ffedb0fa 100644 --- a/packages/coding-agent/test/tools/inspect-image.test.ts +++ b/packages/coding-agent/test/tools/inspect-image.test.ts @@ -3,6 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import type { completeSimple, Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; @@ -14,7 +15,7 @@ import { sanitizeText } from "@oh-my-pi/pi-utils"; const TINY_PNG_BASE64 = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="; -const visionModel: Model<"openai-responses"> = { +const visionModel: Model<"openai-responses"> = buildModel({ id: "gpt-4o", name: "GPT-4o", api: "openai-responses", @@ -25,7 +26,7 @@ const visionModel: Model<"openai-responses"> = { cost: { input: 5, output: 15, cacheRead: 0.5, cacheWrite: 5 }, contextWindow: 128000, maxTokens: 4096, -}; +}); const textOnlyModel: Model<"openai-responses"> = { ...visionModel, diff --git a/packages/coding-agent/test/xiaomi-tp-discovery-merge.test.ts b/packages/coding-agent/test/xiaomi-tp-discovery-merge.test.ts index 1c90b6f27..1579c140a 100644 --- a/packages/coding-agent/test/xiaomi-tp-discovery-merge.test.ts +++ b/packages/coding-agent/test/xiaomi-tp-discovery-merge.test.ts @@ -1,5 +1,6 @@ import { describe, expect, test } from "bun:test"; import type { Model } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { mergeDiscoveredModel } from "@oh-my-pi/pi-coding-agent/config/model-registry"; /** @@ -14,7 +15,7 @@ const STANDARD = "https://api.xiaomimimo.com/v1"; const TOKEN_PLAN = "https://token-plan-sgp.xiaomimimo.com/v1"; function bundled(baseUrl: string): Model<"openai-completions"> { - return { + return buildModel({ id: "mimo-v2.5", name: "MiMo v2.5", api: "openai-completions", @@ -25,7 +26,7 @@ function bundled(baseUrl: string): Model<"openai-completions"> { cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: 128000, maxTokens: 8192, - }; + }); } describe("mergeDiscoveredModel", () => { diff --git a/packages/stats/test/db-cost.test.ts b/packages/stats/test/db-cost.test.ts index 3fa370872..dbd42b199 100644 --- a/packages/stats/test/db-cost.test.ts +++ b/packages/stats/test/db-cost.test.ts @@ -4,7 +4,7 @@ import * as os from "node:os"; import * as path from "node:path"; import { closeDb, getRecentRequests, initDb, insertMessageStats } from "@oh-my-pi/omp-stats/db"; import type { MessageStats } from "@oh-my-pi/omp-stats/types"; -import { getBundledModel } from "@oh-my-pi/pi-ai"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { getAgentDir, getStatsDbPath, setAgentDir, TempDir } from "@oh-my-pi/pi-utils"; const originalConfigDir = process.env.PI_CONFIG_DIR;