From e5541a577afa49fbbf8ed03d7a8f23634d5605bc Mon Sep 17 00:00:00 2001 From: Anatoli Tsinovoy Date: Sat, 1 Aug 2026 12:29:07 +0300 Subject: [PATCH] fix(ai): address Bedrock Mantle review feedback --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/providers/amazon-bedrock.ts | 10 +- packages/ai/src/providers/aws-credentials.ts | 145 +- packages/ai/src/providers/bedrock-mantle.ts | 54 +- packages/ai/src/registry/amazon-bedrock.ts | 14 +- packages/ai/src/registry/aws.ts | 22 +- packages/ai/src/registry/bedrock-mantle.ts | 32 +- packages/ai/src/registry/google-vertex.ts | 4 +- packages/ai/src/registry/types.ts | 27 + packages/ai/src/stream.ts | 72 +- packages/ai/src/types.ts | 11 +- packages/ai/src/utils/aws-profile.ts | 39 +- packages/ai/test/aws-credentials.test.ts | 126 +- packages/ai/test/aws-registry.test.ts | 25 + .../ai/test/bedrock-inference-profile.test.ts | 31 +- packages/ai/test/bedrock-mantle-auth.test.ts | 101 +- packages/catalog/CHANGELOG.md | 4 +- packages/catalog/scripts/generate-models.ts | 9 +- packages/catalog/src/models.json | 1815 ++++++++++++----- .../src/provider-models/descriptor-types.ts | 8 +- .../src/provider-models/descriptors.ts | 4 + .../src/provider-models/openai-compat.ts | 35 +- .../test/amazon-bedrock-openai.test.ts | 59 +- .../coding-agent/src/config/model-registry.ts | 18 +- 24 files changed, 1997 insertions(+), 670 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 40f899002..c398fa4ae 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -22,7 +22,7 @@ ### Fixed -- Added Bedrock Mantle region selection and bearer-token or SigV4 authentication for OpenAI Responses models. +- Added profile-aware Bedrock Mantle region selection, authenticated model discovery, bearer-token or SigV4 authentication, and credential refresh handling for OpenAI Responses models ([#7080](https://github.com/can1357/oh-my-pi/pull/7080) by [@anatoli-tsinovoy](https://github.com/anatoli-tsinovoy)). - Fixed Novita login rejecting valid API keys belonging to Developer and Basic team members by validating against the chat completions endpoint instead of the billing balance endpoint. - Fixed Cursor resource_exhausted errors being incorrectly classified as QUOTA_EXHAUSTED (which caused 30-minute credential blocks), mapping them to MODEL_CAPACITY_EXHAUSTED with a shorter backoff instead. - Fixed a crash in Amazon Bedrock and Devin providers when Context.systemPrompt is passed as a bare string. diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index cf9fe87a4..453d90e15 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -10,9 +10,10 @@ import type { Effort } from "@oh-my-pi/pi-catalog/effort"; import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking"; import { calculateCost } from "@oh-my-pi/pi-catalog/models"; -import { $env, $flag, fetchWithRetry, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils"; +import { $flag, fetchWithRetry, parseStreamingJson, parseStreamingJsonThrottled } from "@oh-my-pi/pi-utils"; import { renderDemotedThinking } from "../dialect/demotion"; import * as AIError from "../error"; +import { resolveAwsBearerToken } from "../registry/aws"; import type { Api, AssistantMessage, @@ -30,6 +31,7 @@ import type { ToolResultMessage, } from "../types"; import { normalizeSystemPrompts, normalizeToolCallId, resolveCacheRetention } from "../utils"; +import { resolveAwsAmbientRegion } from "../utils/aws-profile"; import { clearStreamingPartialJson, kStreamingBlockIndex, @@ -74,11 +76,9 @@ export interface BedrockOptions extends StreamOptions { */ thinkingDisplay?: BedrockThinkingDisplay; } -const AUTHENTICATED_API_KEY_SENTINEL = ""; function resolveBearerToken(options: BedrockOptions): string | undefined { - const apiKey = options.apiKey === AUTHENTICATED_API_KEY_SENTINEL ? undefined : options.apiKey; - return options.bearerToken || apiKey || $env.AWS_BEARER_TOKEN_BEDROCK; + return resolveAwsBearerToken(options.apiKey, options.bearerToken); } function inferRegionFromBedrockArn(modelId: string): string | undefined { @@ -149,7 +149,7 @@ function regionServesGeo(region: string, geo: string): boolean { function resolveBedrockRegion(modelId: string, options: BedrockOptions): string { const explicit = options.region || inferRegionFromBedrockArn(modelId); if (explicit) return explicit; - const ambient = $env.AWS_REGION || $env.AWS_DEFAULT_REGION; + const ambient = resolveAwsAmbientRegion(options.profile); const geo = inferenceProfileGeo(modelId); if (geo) { if (ambient && regionServesGeo(ambient, geo)) return ambient; diff --git a/packages/ai/src/providers/aws-credentials.ts b/packages/ai/src/providers/aws-credentials.ts index dee9cd70c..825df53cb 100644 --- a/packages/ai/src/providers/aws-credentials.ts +++ b/packages/ai/src/providers/aws-credentials.ts @@ -21,7 +21,13 @@ import { $env, isEnoent, logger } from "@oh-my-pi/pi-utils"; import * as AIError from "../error"; import type { FetchImpl } from "../types"; import { raceWithSignal } from "../utils/abort"; -import { type AwsIniFile, parseAwsIni } from "../utils/aws-profile"; +import { + type AwsIniFile, + parseAwsIni, + resolveAwsProfile, + resolveAwsRegion, + shouldLoadAwsSharedConfig, +} from "../utils/aws-profile"; import { isLocalOrMetadataHost } from "../utils/proxy"; import type { AwsCredentials } from "./aws-sigv4"; @@ -52,6 +58,23 @@ const FILE_SESSION_CREDS_TTL_MS = 5 * 60_000; */ const SHARED_RESOLVE_TIMEOUT_MS = 30_000; +function requireDynamicCredentialExpiration( + value: string | undefined, + source: "AWS web identity" | "AWS container credential", + kind: "web-identity" | "container", +): number { + const expiresAt = value ? Date.parse(value) : Number.NaN; + if (Number.isFinite(expiresAt)) return expiresAt; + throw new AIError.AwsCredentialsError(`${source} response has a missing or invalid Expiration.`, kind); +} + +/** Credential-process expiry is optional; missing/malformed values disable caching. */ +function dynamicCredentialExpiration(value: string | undefined): number { + if (!value) return Date.now(); + const expiresAt = Date.parse(value); + return Number.isFinite(expiresAt) ? expiresAt : Date.now(); +} + interface CacheEntry { creds: ResolvedCredentials; expiresAt: number; @@ -60,10 +83,15 @@ interface CacheEntry { const cache: Map = new Map(); const inflight: Map> = new Map(); +function credentialCacheKey(profile: string, region: string, loadSharedConfig: boolean): string { + return `${profile}\x00${region}\x00${loadSharedConfig ? "config" : "credentials"}`; +} + export async function resolveAwsCredentials(opts: CredentialResolveOptions = {}): Promise { - const profile = opts.profile || $env.AWS_PROFILE || "default"; - const region = opts.region || $env.AWS_REGION || $env.AWS_DEFAULT_REGION || "us-east-1"; - const cacheKey = `${profile}\x00${region}`; + const profile = resolveAwsProfile(opts.profile); + const region = resolveAwsRegion(opts.region, opts.profile); + const loadSharedConfig = shouldLoadAwsSharedConfig(opts.profile); + const cacheKey = credentialCacheKey(profile, region, loadSharedConfig); const hit = cache.get(cacheKey); if (hit && hit.expiresAt - REFRESH_SKEW_MS > Date.now()) return hit.creds; @@ -78,7 +106,13 @@ export async function resolveAwsCredentials(opts: CredentialResolveOptions = {}) const fetchImpl = opts.fetch ?? (globalThis.fetch as FetchImpl); const promise = (async () => { try { - const creds = await resolveFresh(profile, region, AbortSignal.timeout(SHARED_RESOLVE_TIMEOUT_MS), fetchImpl); + const creds = await resolveFresh( + profile, + region, + loadSharedConfig, + AbortSignal.timeout(SHARED_RESOLVE_TIMEOUT_MS), + fetchImpl, + ); cache.set(cacheKey, { creds, expiresAt: creds.expiresAt ?? Number.POSITIVE_INFINITY }); return creds; } finally { @@ -92,6 +126,7 @@ export async function resolveAwsCredentials(opts: CredentialResolveOptions = {}) async function resolveFresh( profile: string, region: string, + loadSharedConfig: boolean, signal?: AbortSignal, fetchImpl: FetchImpl = globalThis.fetch as FetchImpl, ): Promise { @@ -104,7 +139,7 @@ async function resolveFresh( if (webIdentityCreds) return webIdentityCreds; // 3. Profile (static, SSO, or credential_process). - const profileCreds = await readProfileCredentials(profile, region, signal, fetchImpl); + const profileCreds = await readProfileCredentials(profile, region, loadSharedConfig, signal, fetchImpl); if (profileCreds) return profileCreds; // 4. ECS/container credentials. @@ -149,6 +184,7 @@ async function readIniFile(p: string): Promise { async function readProfileCredentials( profile: string, region: string, + loadSharedConfig: boolean, signal: AbortSignal | undefined, fetchImpl: FetchImpl, ): Promise { @@ -157,7 +193,7 @@ async function readProfileCredentials( const configPath = $env.AWS_CONFIG_FILE || path.join(home, ".aws", "config"); const credentialsIni = await readIniFile(credentialsPath); - const configIni = await readIniFile(configPath); + const configIni = loadSharedConfig ? await readIniFile(configPath) : undefined; // Static credentials live in ~/.aws/credentials; SSO config lives in // ~/.aws/config under `[profile foo]`. Merge into a single view. @@ -380,10 +416,11 @@ async function readCredentialProcess( accessKeyId: parsed.AccessKeyId, secretAccessKey: parsed.SecretAccessKey, }; - if (parsed.SessionToken) out.sessionToken = parsed.SessionToken; - if (parsed.Expiration) { - const exp = Date.parse(parsed.Expiration); - if (!Number.isNaN(exp)) out.expiresAt = exp; + if (parsed.SessionToken) { + out.sessionToken = parsed.SessionToken; + out.expiresAt = dynamicCredentialExpiration(parsed.Expiration); + } else if (parsed.Expiration) { + out.expiresAt = dynamicCredentialExpiration(parsed.Expiration); } return out; } @@ -501,6 +538,11 @@ function xmlTag(xml: string, tag: string): string | undefined { .replaceAll("'", "'"); } +function stsEndpoint(region: string): string { + const dnsSuffix = region.startsWith("cn-") ? "amazonaws.com.cn" : "amazonaws.com"; + return `https://sts.${region}.${dnsSuffix}/`; +} + async function readWebIdentityCredentials( region: string, signal: AbortSignal | undefined, @@ -531,7 +573,7 @@ async function readWebIdentityCredentials( RoleSessionName: $env.AWS_ROLE_SESSION_NAME || `omp-${process.pid}`, WebIdentityToken: token, }); - const response = await fetchImpl(`https://sts.${region}.amazonaws.com/`, { + const response = await fetchImpl(stsEndpoint(region), { method: "POST", headers: { "content-type": "application/x-www-form-urlencoded" }, body: body.toString(), @@ -553,10 +595,13 @@ async function readWebIdentityCredentials( "web-identity", ); } - const credentials: ResolvedCredentials = { accessKeyId, secretAccessKey, sessionToken }; - const expiration = xmlTag(xml, "Expiration"); - if (expiration) credentials.expiresAt = Date.parse(expiration); - return credentials; + const expiresAt = requireDynamicCredentialExpiration(xmlTag(xml, "Expiration"), "AWS web identity", "web-identity"); + return { + accessKeyId, + secretAccessKey, + sessionToken, + expiresAt, + }; } // ---------- ECS/container credentials ---------- @@ -568,6 +613,8 @@ interface ContainerCredentialResponse { Expiration?: string; } +const ECS_TASK_CREDENTIALS_BASE_URL = new URL("http://169.254.170.2/"); + async function readContainerCredentials( signal: AbortSignal | undefined, fetchImpl: FetchImpl, @@ -577,13 +624,13 @@ async function readContainerCredentials( if (!relativeUri && !fullUri) return undefined; let endpoint: URL; if (relativeUri) { - if (!relativeUri.startsWith("/")) { + if (!relativeUri.startsWith("/") || relativeUri.startsWith("//")) { throw new AIError.AwsCredentialsError( - "AWS_CONTAINER_CREDENTIALS_RELATIVE_URI must start with '/'.", + "AWS_CONTAINER_CREDENTIALS_RELATIVE_URI must be a single-host absolute path.", "container", ); } - endpoint = new URL(`http://169.254.170.2${relativeUri}`); + endpoint = new URL(relativeUri.slice(1), ECS_TASK_CREDENTIALS_BASE_URL); } else { try { endpoint = new URL(fullUri as string); @@ -626,55 +673,66 @@ async function readContainerCredentials( ); } const body = (await response.json()) as ContainerCredentialResponse; - if (!body.AccessKeyId || !body.SecretAccessKey) { + if (!body.AccessKeyId || !body.SecretAccessKey || !body.Token) { throw new AIError.AwsCredentialsError( - "AWS container credential response is missing AccessKeyId/SecretAccessKey.", + "AWS container credential response is missing AccessKeyId/SecretAccessKey/Token.", "container", ); } - const credentials: ResolvedCredentials = { + return { accessKeyId: body.AccessKeyId, secretAccessKey: body.SecretAccessKey, + sessionToken: body.Token, + expiresAt: requireDynamicCredentialExpiration(body.Expiration, "AWS container credential", "container"), }; - if (body.Token) credentials.sessionToken = body.Token; - if (body.Expiration) credentials.expiresAt = Date.parse(body.Expiration); - return credentials; } // ---------- IMDSv2 ---------- -const IMDS_HOST = "169.254.169.254"; +const IMDS_IPV4_BASE_URL = "http://169.254.169.254/"; +const IMDS_IPV6_BASE_URL = "http://[fd00:ec2::254]/"; const IMDS_TIMEOUT_MS = 1000; +function imdsRequestSignal(parentSignal: AbortSignal | undefined): AbortSignal { + const timeout = AbortSignal.timeout(IMDS_TIMEOUT_MS); + return parentSignal ? AbortSignal.any([parentSignal, timeout]) : timeout; +} + +function imdsBaseUrl(): URL { + const mode = $env.AWS_EC2_METADATA_SERVICE_ENDPOINT_MODE?.toLowerCase(); + const fallback = mode === "ipv6" ? IMDS_IPV6_BASE_URL : IMDS_IPV4_BASE_URL; + const endpoint = new URL($env.AWS_EC2_METADATA_SERVICE_ENDPOINT || fallback); + if (!endpoint.pathname.endsWith("/")) endpoint.pathname += "/"; + return endpoint; +} + async function readImdsCredentials( parentSignal: AbortSignal | undefined, fetchImpl: FetchImpl, ): Promise { - const timeout = AbortSignal.timeout(IMDS_TIMEOUT_MS); - const signal = parentSignal ? AbortSignal.any([parentSignal, timeout]) : timeout; - const endpoint = ($env.AWS_EC2_METADATA_SERVICE_ENDPOINT || `http://${IMDS_HOST}`).replace(/\/+$/, ""); try { - const tokenRes = await fetchImpl(`${endpoint}/latest/api/token`, { + const endpoint = imdsBaseUrl(); + const tokenRes = await fetchImpl(new URL("latest/api/token", endpoint), { method: "PUT", headers: { "x-aws-ec2-metadata-token-ttl-seconds": "21600" }, - signal, + signal: imdsRequestSignal(parentSignal), }); if (!tokenRes.ok) return undefined; const token = await tokenRes.text(); - const roleRes = await fetchImpl(`${endpoint}/latest/meta-data/iam/security-credentials/`, { + const roleRes = await fetchImpl(new URL("latest/meta-data/iam/security-credentials/", endpoint), { headers: { "x-aws-ec2-metadata-token": token }, - signal, + signal: imdsRequestSignal(parentSignal), }); if (!roleRes.ok) return undefined; const role = (await roleRes.text()).trim(); if (!role) return undefined; const credsRes = await fetchImpl( - `${endpoint}/latest/meta-data/iam/security-credentials/${encodeURIComponent(role)}`, + new URL(`latest/meta-data/iam/security-credentials/${encodeURIComponent(role)}`, endpoint), { headers: { "x-aws-ec2-metadata-token": token }, - signal, + signal: imdsRequestSignal(parentSignal), }, ); if (!credsRes.ok) return undefined; @@ -684,14 +742,15 @@ async function readImdsCredentials( Token?: string; Expiration?: string; }; - if (!body.AccessKeyId || !body.SecretAccessKey) return undefined; - const out: ResolvedCredentials = { + if (!body.AccessKeyId || !body.SecretAccessKey || !body.Token || !body.Expiration) return undefined; + const expiresAt = Date.parse(body.Expiration); + if (!Number.isFinite(expiresAt)) return undefined; + return { accessKeyId: body.AccessKeyId, secretAccessKey: body.SecretAccessKey, + sessionToken: body.Token, + expiresAt, }; - if (body.Token) out.sessionToken = body.Token; - if (body.Expiration) out.expiresAt = Date.parse(body.Expiration); - return out; } catch { return undefined; } @@ -707,7 +766,7 @@ export function clearAwsCredentialCache(): void { * 401/403 responses so stale credentials are re-resolved instead of served until restart. */ export function invalidateAwsCredentialCache(opts: { profile?: string; region?: string } = {}): void { - const profile = opts.profile || $env.AWS_PROFILE || "default"; - const region = opts.region || $env.AWS_REGION || $env.AWS_DEFAULT_REGION || "us-east-1"; - cache.delete(`${profile}\x00${region}`); + const profile = resolveAwsProfile(opts.profile); + const region = resolveAwsRegion(opts.region, opts.profile); + cache.delete(credentialCacheKey(profile, region, shouldLoadAwsSharedConfig(opts.profile))); } diff --git a/packages/ai/src/providers/bedrock-mantle.ts b/packages/ai/src/providers/bedrock-mantle.ts index 2f7324108..d56b50a62 100644 --- a/packages/ai/src/providers/bedrock-mantle.ts +++ b/packages/ai/src/providers/bedrock-mantle.ts @@ -1,15 +1,15 @@ -import { $env } from "@oh-my-pi/pi-utils"; -import { AUTHENTICATED_SENTINEL } from "../registry/types"; +import { type AwsBedrockProviderOptions, resolveAwsBearerToken } from "../registry/aws"; import type { FetchImpl, Model } from "../types"; -import { resolveAwsCredentials } from "./aws-credentials"; +import { resolveAwsRegion } from "../utils/aws-profile"; +import { invalidateAwsCredentialCache, resolveAwsCredentials } from "./aws-credentials"; import { signRequest } from "./aws-sigv4"; import type { OpenAIResponsesOptions } from "./openai-responses"; +import { NO_AUTH_SENTINEL } from "./openai-shared"; + +export type BedrockMantleProviderOptions = AwsBedrockProviderOptions; export interface BedrockMantleOptions extends OpenAIResponsesOptions { - region?: string; - profile?: string; - /** Amazon Bedrock API key sent as a bearer token, ahead of SigV4 credential resolution. */ - bearerToken?: string; + providerOptions?: BedrockMantleProviderOptions; } async function requestBody(input: string | URL | Request, init?: RequestInit): Promise { @@ -33,7 +33,7 @@ function createSignedFetch(options: BedrockMantleOptions, region: string): Fetch headers.delete("authorization"); const body = await requestBody(input, init); const credentials = await resolveAwsCredentials({ - profile: options.profile, + profile: options.providerOptions?.profile, region, signal: options.signal, fetch: baseFetch, @@ -52,11 +52,38 @@ function createSignedFetch(options: BedrockMantleOptions, region: string): Fetch for (const [name, value] of Object.entries(signed)) { if (value !== undefined && name !== "host") headers.set(name, value); } - return baseFetch(url, { ...init, method, headers, body }); + const response = await baseFetch( + url, + method === "GET" || method === "HEAD" ? { ...init, method, headers } : { ...init, method, headers, body }, + ); + if (response.status === 401 || response.status === 403) { + invalidateAwsCredentialCache({ profile: options.providerOptions?.profile, region }); + } + return response; }; return Object.assign(signedFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {}); } +function resolveBearerToken(options: BedrockMantleOptions): string | undefined { + const apiKey = options.apiKey === NO_AUTH_SENTINEL ? undefined : options.apiKey; + return resolveAwsBearerToken(apiKey, options.providerOptions?.bearerToken); +} + +export function createBedrockMantleAuthenticatedFetch(options: BedrockMantleOptions = {}): FetchImpl { + const region = resolveAwsRegion(options.providerOptions?.region, options.providerOptions?.profile); + const bearerToken = resolveBearerToken(options); + if (!bearerToken) return createSignedFetch(options, region); + + const baseFetch = options.fetch ?? (globalThis.fetch as FetchImpl); + const authenticatedFetch = async (input: string | URL | Request, init?: RequestInit): Promise => { + const headers = new Headers(input instanceof Request ? input.headers : undefined); + for (const [name, value] of new Headers(init?.headers)) headers.set(name, value); + headers.set("authorization", `Bearer ${bearerToken}`); + return baseFetch(input, { ...init, headers }); + }; + return Object.assign(authenticatedFetch, baseFetch.preconnect ? { preconnect: baseFetch.preconnect } : {}); +} + export interface PreparedBedrockMantleRequest { model: Model<"openai-responses">; options: OpenAIResponsesOptions; @@ -66,10 +93,9 @@ export function prepareBedrockMantleRequest( model: Model<"openai-responses">, options: BedrockMantleOptions, ): PreparedBedrockMantleRequest { - const region = options.region || $env.AWS_REGION || $env.AWS_DEFAULT_REGION || "us-east-1"; + const region = resolveAwsRegion(options.providerOptions?.region, options.providerOptions?.profile); const resolvedModel = { ...model, baseUrl: model.baseUrl.replaceAll("{region}", encodeURIComponent(region)) }; - const apiKey = options.apiKey === AUTHENTICATED_SENTINEL || options.apiKey === "N/A" ? undefined : options.apiKey; - const bearerToken = options.bearerToken || apiKey || $env.AWS_BEARER_TOKEN_BEDROCK; + const bearerToken = resolveBearerToken(options); if (bearerToken) { return { model: resolvedModel, options: { ...options, apiKey: bearerToken } }; } @@ -77,8 +103,8 @@ export function prepareBedrockMantleRequest( model: resolvedModel, options: { ...options, - apiKey: "N/A", - fetch: createSignedFetch(options, region), + apiKey: NO_AUTH_SENTINEL, + fetch: createBedrockMantleAuthenticatedFetch(options), }, }; } diff --git a/packages/ai/src/registry/amazon-bedrock.ts b/packages/ai/src/registry/amazon-bedrock.ts index d2c30aa93..37e2c40c8 100644 --- a/packages/ai/src/registry/amazon-bedrock.ts +++ b/packages/ai/src/registry/amazon-bedrock.ts @@ -1,9 +1,17 @@ -import { hasAwsCredentialSource } from "./aws"; -import { AUTHENTICATED_SENTINEL, type ProviderDefinition } from "./types"; +import { type AwsBedrockProviderOptions, resolveAwsRegistryApiKey } from "./aws"; +import type { ProviderDefinition } from "./types"; export const amazonBedrockProvider = { id: "amazon-bedrock", name: "Amazon Bedrock", // Amazon Bedrock accepts bearer tokens, IAM keys, profiles, ECS/IRSA credential chains. - envKeys: () => (hasAwsCredentialSource() ? AUTHENTICATED_SENTINEL : undefined), + envKeys: resolveAwsRegistryApiKey, + mapSimpleOptions: options => { + const awsOptions = options.providerOptions as AwsBedrockProviderOptions | undefined; + return { + region: awsOptions?.region, + profile: awsOptions?.profile, + bearerToken: awsOptions?.bearerToken, + }; + }, } as const satisfies ProviderDefinition; diff --git a/packages/ai/src/registry/aws.ts b/packages/ai/src/registry/aws.ts index 0d4d7d633..3e23f88f0 100644 --- a/packages/ai/src/registry/aws.ts +++ b/packages/ai/src/registry/aws.ts @@ -1,9 +1,18 @@ import * as fs from "node:fs"; import { $env } from "@oh-my-pi/pi-utils"; import { hasConfiguredAwsProfile } from "../utils/aws-profile"; +import { AUTHENTICATED_SENTINEL } from "./types"; + +export interface AwsBedrockProviderOptions extends Readonly> { + /** AWS region used in the service endpoint and SigV4 credential scope. */ + region?: string; + /** Named AWS shared-credentials/config profile. */ + profile?: string; + /** Amazon Bedrock API key sent as a bearer token, ahead of SigV4 credential resolution. */ + bearerToken?: string; +} function isEc2Host(): boolean { - if ($env.AWS_EXECUTION_ENV?.includes("EC2")) return true; for (const candidate of [ "/sys/hypervisor/uuid", "/sys/devices/virtual/dmi/id/product_uuid", @@ -35,3 +44,14 @@ export function hasAwsCredentialSource(): boolean { hasInstanceRole ); } + +/** Registry key marker for AWS transports that resolve their own bearer/IAM credentials. */ +export function resolveAwsRegistryApiKey(): string | undefined { + return hasAwsCredentialSource() ? AUTHENTICATED_SENTINEL : undefined; +} + +/** Resolve a real AWS bearer token while filtering the registry's auth marker. */ +export function resolveAwsBearerToken(apiKey?: string, bearerToken?: string): string | undefined { + const resolvedApiKey = apiKey === AUTHENTICATED_SENTINEL ? undefined : apiKey; + return bearerToken || resolvedApiKey || $env.AWS_BEARER_TOKEN_BEDROCK; +} diff --git a/packages/ai/src/registry/bedrock-mantle.ts b/packages/ai/src/registry/bedrock-mantle.ts index 3991e26ea..8ba069aa2 100644 --- a/packages/ai/src/registry/bedrock-mantle.ts +++ b/packages/ai/src/registry/bedrock-mantle.ts @@ -1,8 +1,34 @@ -import { hasAwsCredentialSource } from "./aws"; -import { AUTHENTICATED_SENTINEL, type ProviderDefinition } from "./types"; +import { + type BedrockMantleOptions, + createBedrockMantleAuthenticatedFetch, + prepareBedrockMantleRequest, +} from "../providers/bedrock-mantle"; +import type { Model } from "../types"; +import { resolveAwsRegion } from "../utils/aws-profile"; +import { resolveAwsBearerToken, resolveAwsRegistryApiKey } from "./aws"; +import type { ProviderDefinition } from "./types"; export const bedrockMantleProvider = { id: "bedrock-mantle", name: "Amazon Bedrock Mantle", - envKeys: () => (hasAwsCredentialSource() ? AUTHENTICATED_SENTINEL : undefined), + envKeys: resolveAwsRegistryApiKey, + allowsMissingApiKey: true, + prepareRequest: (model, options) => + prepareBedrockMantleRequest(model as Model<"openai-responses">, options as BedrockMantleOptions), + mapSimpleOptions: options => ({ providerOptions: options.providerOptions }), + prepareModelDiscovery: config => { + const bearerToken = resolveAwsBearerToken(config.apiKey); + if (!bearerToken) { + return { ...config, apiKey: undefined, authenticated: false }; + } + const region = resolveAwsRegion(); + return { + authenticated: true, + baseUrl: `https://bedrock-mantle.${encodeURIComponent(region)}.api.aws/openai/v1`, + fetch: createBedrockMantleAuthenticatedFetch({ + fetch: config.fetch, + providerOptions: { bearerToken, region }, + }), + }; + }, } as const satisfies ProviderDefinition; diff --git a/packages/ai/src/registry/google-vertex.ts b/packages/ai/src/registry/google-vertex.ts index c58b20fb2..eae816807 100644 --- a/packages/ai/src/registry/google-vertex.ts +++ b/packages/ai/src/registry/google-vertex.ts @@ -2,7 +2,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { $env } from "@oh-my-pi/pi-utils"; -import type { ProviderDefinition } from "./types"; +import { AUTHENTICATED_SENTINEL, type ProviderDefinition } from "./types"; let cachedVertexAdcCredentialsExists: boolean | null = null; @@ -32,7 +32,7 @@ export const googleVertexProvider = { const hasProject = !!($env.GOOGLE_CLOUD_PROJECT || $env.GCP_PROJECT || $env.GCLOUD_PROJECT); const hasLocation = !!($env.GOOGLE_VERTEX_LOCATION || $env.GOOGLE_CLOUD_LOCATION || $env.VERTEX_LOCATION); if (hasCredentials && hasProject && hasLocation) { - return ""; + return AUTHENTICATED_SENTINEL; } }, } as const satisfies ProviderDefinition; diff --git a/packages/ai/src/registry/types.ts b/packages/ai/src/registry/types.ts index a23bc4836..2466ffaca 100644 --- a/packages/ai/src/registry/types.ts +++ b/packages/ai/src/registry/types.ts @@ -9,6 +9,8 @@ * (default model, model-manager factory, catalog discovery) lives in * `@oh-my-pi/pi-catalog`'s descriptor table. */ + +import type { Api, FetchImpl, Model, SimpleStreamOptions, StreamOptions } from "../types"; import type { OAuthCredentials, OAuthLoginCallbacks } from "./oauth/types"; /** @@ -21,6 +23,23 @@ export type KeyResolver = string | (() => string | undefined); /** Credentials are resolved by the provider transport rather than used as a bearer string. */ export const AUTHENTICATED_SENTINEL = ""; +export interface PreparedProviderRequest { + readonly model: Model; + readonly options: StreamOptions; +} + +export type ProviderRequestPreparer = (model: Model, options: StreamOptions) => PreparedProviderRequest; +export type ProviderSimpleOptionsMapper = (options: SimpleStreamOptions) => Readonly>; + +export interface ProviderModelDiscoveryConfig { + readonly apiKey?: string; + readonly baseUrl?: string; + readonly fetch?: FetchImpl; + readonly authenticated?: boolean; +} + +export type ProviderModelDiscoveryPreparer = (config: ProviderModelDiscoveryConfig) => ProviderModelDiscoveryConfig; + /** * Declarative description of a single provider's auth/login wiring. All * fields are optional except `id`/`name`; presence of a field opts the @@ -45,6 +64,14 @@ export interface ProviderDefinition { readonly showInLoginList?: boolean; // --- env-var fallback (the catalog table's `envVars` supplies plain names; set this only for computed resolvers) --- readonly envKeys?: KeyResolver; + /** Provider transport can authenticate without a resolved API-key string. */ + readonly allowsMissingApiKey?: boolean; + /** Provider-owned request shaping applied before generic API dispatch. */ + readonly prepareRequest?: ProviderRequestPreparer; + /** Provider-owned projection from the generic simple-stream option bag. */ + readonly mapSimpleOptions?: ProviderSimpleOptionsMapper; + /** Provider-owned authentication and endpoint setup for model discovery. */ + readonly prepareModelDiscovery?: ProviderModelDiscoveryPreparer; // --- interactive login (OAuthProviderInterface-compatible) --- readonly login?: (callbacks: OAuthLoginCallbacks) => Promise; readonly refreshToken?: (credentials: OAuthCredentials) => Promise; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 235d378c4..717206eed 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -24,7 +24,6 @@ import { isInvalidatedOAuthTokenError } from "./error/auth-classify"; import { isUsageLimitOutcome } from "./error/rate-limit"; import type { BedrockOptions } from "./providers/amazon-bedrock"; import type { AnthropicOptions } from "./providers/anthropic"; -import { type BedrockMantleOptions, prepareBedrockMantleRequest } from "./providers/bedrock-mantle"; import type { CursorOptions } from "./providers/cursor"; import type { DevinOptions } from "./providers/devin"; import { isGitLabDuoModel, streamGitLabDuo } from "./providers/gitlab-duo"; @@ -60,7 +59,7 @@ import { streamOpenAIResponses, } from "./providers/register-builtins"; import { isSyntheticModel, streamSynthetic } from "./providers/synthetic"; -import { PROVIDER_REGISTRY } from "./registry"; +import { getProviderDefinition, PROVIDER_REGISTRY } from "./registry"; import type { Api, AssistantMessage, @@ -806,39 +805,37 @@ function streamDispatch( } as GitLabDuoWorkflowOptions); } - // Vertex AI uses Application Default Credentials, not API keys + // Vertex AI and Bedrock Converse authenticate outside the generic API-key path. if (model.api === "google-vertex") { return streamGoogleVertex(model as Model<"google-vertex">, context, requestOptions as GoogleVertexOptions); - } else if (model.api === "bedrock-converse-stream") { - // Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile. + } + if (model.api === "bedrock-converse-stream") { return streamBedrock(model as Model<"bedrock-converse-stream">, context, requestOptions as BedrockOptions); - } else if (model.provider === "bedrock-mantle" && model.api === "openai-responses") { - const prepared = prepareBedrockMantleRequest( - model as Model<"openai-responses">, - requestOptions as BedrockMantleOptions, - ); - return streamOpenAIResponses(prepared.model, context, prepared.options); } - const apiKey = requestOptions.apiKey || getEnvApiKey(model.provider); + const prepareRequest = getProviderDefinition(model.provider)?.prepareRequest; + const prepared = prepareRequest?.(model as Model, requestOptions as StreamOptions); + const providerModel = prepared?.model ?? (model as Model); + const preparedOptions = prepared?.options ?? (requestOptions as StreamOptions); + const apiKey = preparedOptions.apiKey || getEnvApiKey(providerModel.provider); if (!apiKey) { - throw new AIError.MissingApiKeyError(model.provider); + throw new AIError.MissingApiKeyError(providerModel.provider); } - const providerOptions = isGoogleVertexAuthenticatedModel(model) + const providerOptions = isGoogleVertexAuthenticatedModel(providerModel) ? { - ...requestOptions, + ...preparedOptions, apiKey: "vertex-adc", - fetch: createVertexAuthenticatedFetch(requestOptions), + fetch: createVertexAuthenticatedFetch(preparedOptions), } - : { ...requestOptions, apiKey }; + : { ...preparedOptions, apiKey }; - const api: Api = model.api; + const api: Api = providerModel.api; switch (api) { case "anthropic-messages": { const anthropicOptions = providerOptions as AnthropicOptions; - return streamAnthropic(model as Model<"anthropic-messages">, context, { + return streamAnthropic(providerModel as Model<"anthropic-messages">, context, { ...anthropicOptions, - isOAuth: anthropicOptions.isOAuth ?? model.isOAuth, + isOAuth: anthropicOptions.isOAuth ?? providerModel.isOAuth, }); } @@ -846,13 +843,13 @@ function streamDispatch( const useResponses = $env.PI_OPENROUTER_RESPONSES !== "0"; if (useResponses) { return streamOpenAIResponses( - model as Model<"openai-responses">, + providerModel as Model<"openai-responses">, context, providerOptions as OptionsForApi<"openai-responses">, ); } return streamOpenAICompletions( - model as Model<"openai-completions">, + providerModel as Model<"openai-completions">, context, providerOptions as OptionsForApi<"openai-completions">, ); @@ -860,50 +857,50 @@ function streamDispatch( case "openai-completions": return streamOpenAICompletions( - model as Model<"openai-completions">, + providerModel as Model<"openai-completions">, context, providerOptions as OptionsForApi<"openai-completions">, ); case "openai-responses": return streamOpenAIResponses( - model as Model<"openai-responses">, + providerModel as Model<"openai-responses">, context, providerOptions as OptionsForApi<"openai-responses">, ); case "azure-openai-responses": return streamAzureOpenAIResponses( - model as Model<"azure-openai-responses">, + providerModel as Model<"azure-openai-responses">, context, providerOptions as OptionsForApi<"azure-openai-responses">, ); case "openai-codex-responses": return streamOpenAICodexResponses( - model as Model<"openai-codex-responses">, + providerModel as Model<"openai-codex-responses">, context, providerOptions as OptionsForApi<"openai-codex-responses">, ); case "google-generative-ai": - return streamGoogle(model as Model<"google-generative-ai">, context, providerOptions); + return streamGoogle(providerModel as Model<"google-generative-ai">, context, providerOptions); case "google-gemini-cli": return streamGoogleGeminiCli( - model as Model<"google-gemini-cli">, + providerModel as Model<"google-gemini-cli">, context, providerOptions as GoogleGeminiCliOptions, ); case "ollama-chat": - return streamOllama(model as Model<"ollama-chat">, context, providerOptions as OllamaChatOptions); + return streamOllama(providerModel as Model<"ollama-chat">, context, providerOptions as OllamaChatOptions); case "cursor-agent": - return streamCursor(model as Model<"cursor-agent">, context, providerOptions as CursorOptions); + return streamCursor(providerModel as Model<"cursor-agent">, context, providerOptions as CursorOptions); case "devin-agent": - return streamDevin(model as Model<"devin-agent">, context, providerOptions as DevinOptions); + return streamDevin(providerModel as Model<"devin-agent">, context, providerOptions as DevinOptions); default: throw new AIError.ConfigurationError(`Unhandled API: ${api}`); @@ -1104,7 +1101,7 @@ export function streamSimple( return; } if (lastKey === undefined) { - if (model.provider === "bedrock-mantle") { + if (getProviderDefinition(model.provider)?.allowsMissingApiKey) { const failure = await runAttempt(); if (failure) emitFailure(failure); return; @@ -1158,11 +1155,11 @@ export function streamSimple( // Bedrock doesn't have any API keys instead it sources credentials from standard AWS env variables or from given AWS profile. const providerOptions = mapOptionsForApi(model, requestOptions, undefined); return stream(model, context, providerOptions); - } else if (model.provider === "bedrock-mantle" && model.api === "openai-responses") { + } else if (getProviderDefinition(model.provider)?.allowsMissingApiKey) { const providerOptions = mapOptionsForApi( model, requestOptions, - typeof requestOptions.apiKey === "string" ? requestOptions.apiKey : undefined, + typeof requestOptions.apiKey === "string" ? requestOptions.apiKey : getEnvApiKey(model.provider), ); return stream(model, context, providerOptions); } @@ -1463,6 +1460,7 @@ function mapOptionsForApi( apiKey?: string, ): OptionsForApi { const options = normalizeMandatoryReasoningOptions(model, rawOptions); + const simpleProviderOptions = getProviderDefinition(model.provider)?.mapSimpleOptions?.(options ?? {}); const base = { temperature: options?.temperature, topP: options?.topP, @@ -1492,6 +1490,7 @@ function mapOptionsForApi( execHandlers: options?.execHandlers, fetch: options?.fetch, fallbacks: options?.fallbacks, + ...simpleProviderOptions, }; switch (model.api) { @@ -1677,11 +1676,6 @@ function mapOptionsForApi( textVerbosity: options?.textVerbosity, promptCache: options?.promptCache, statefulResponses: options?.statefulResponses, - ...(model.provider === "bedrock-mantle" && { - region: options?.region, - profile: options?.profile, - bearerToken: options?.bearerToken, - }), }); case "azure-openai-responses": diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index ebdefb751..2cf381a1e 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -434,6 +434,11 @@ export interface StreamOptions { * For example, Anthropic uses `user_id` for abuse tracking and rate limiting. */ metadata?: Record; + /** + * Provider-owned request configuration. Provider hooks interpret this bag; + * generic API transports do not forward its fields onto the wire. + */ + providerOptions?: Readonly>; /** OpenAI Responses/Codex response fields to include verbatim. */ include?: OpenAIResponseInclude[]; /** @@ -639,12 +644,6 @@ export interface SimpleStreamOptions extends Omit { * provider. Non-Anthropic providers ignore the field. */ fallbacks?: FallbackParam[]; - /** AWS region override for Amazon Bedrock transports. */ - region?: string; - /** AWS profile override for Amazon Bedrock transports. */ - profile?: string; - /** Amazon Bedrock API key, preferred over SigV4 credential resolution. */ - bearerToken?: string; } // Generic StreamFunction with typed options diff --git a/packages/ai/src/utils/aws-profile.ts b/packages/ai/src/utils/aws-profile.ts index ffa6ae4d8..3f256fbe7 100644 --- a/packages/ai/src/utils/aws-profile.ts +++ b/packages/ai/src/utils/aws-profile.ts @@ -40,12 +40,45 @@ function readAwsIniSync(filePath: string): AwsIniFile | undefined { } } -export function hasConfiguredAwsProfile(profile = $env.AWS_PROFILE || "default"): boolean { +/** Resolve the selected shared-credentials profile. */ +export function resolveAwsProfile(profile?: string): string { + return profile || $env.AWS_PROFILE || "default"; +} + +/** + * Whether the shared config file participates in profile/region resolution. + * Explicit profile selection enables it; the implicit default profile follows + * the AWS SDK's `AWS_SDK_LOAD_CONFIG` opt-in. + */ +export function shouldLoadAwsSharedConfig(profile?: string): boolean { + if (profile || $env.AWS_PROFILE) return true; + const value = $env.AWS_SDK_LOAD_CONFIG?.toLowerCase(); + return value === "1" || value === "true"; +} + +export function resolveAwsProfileRegion(profile?: string): string | undefined { + if (!shouldLoadAwsSharedConfig(profile)) return undefined; + const configPath = $env.AWS_CONFIG_FILE || path.join(os.homedir(), ".aws", "config"); + return readAwsIniSync(configPath)?.[resolveAwsProfile(profile)]?.region; +} + +/** Region selected by the environment or active shared-config profile. */ +export function resolveAwsAmbientRegion(profile?: string): string | undefined { + return $env.AWS_REGION || $env.AWS_DEFAULT_REGION || resolveAwsProfileRegion(profile); +} + +/** Resolve the region precedence shared by AWS transports and credential exchanges. */ +export function resolveAwsRegion(explicitRegion?: string, profile?: string): string { + return explicitRegion || resolveAwsAmbientRegion(profile) || "us-east-1"; +} + +export function hasConfiguredAwsProfile(profile?: string): boolean { + const selectedProfile = resolveAwsProfile(profile); const credentialsPath = $env.AWS_SHARED_CREDENTIALS_FILE || path.join(os.homedir(), ".aws", "credentials"); const configPath = $env.AWS_CONFIG_FILE || path.join(os.homedir(), ".aws", "config"); const credentialsIni = readAwsIniSync(credentialsPath); - const configIni = readAwsIniSync(configPath); - const merged = { ...(configIni?.[profile] ?? {}), ...(credentialsIni?.[profile] ?? {}) }; + const configIni = shouldLoadAwsSharedConfig(profile) ? readAwsIniSync(configPath) : undefined; + const merged = { ...(configIni?.[selectedProfile] ?? {}), ...(credentialsIni?.[selectedProfile] ?? {}) }; if (merged.aws_access_key_id && merged.aws_secret_access_key) return true; if (merged.credential_process) return true; if (!merged.sso_account_id || !merged.sso_role_name) return false; diff --git a/packages/ai/test/aws-credentials.test.ts b/packages/ai/test/aws-credentials.test.ts index 710a1ffa5..c3cf95f86 100644 --- a/packages/ai/test/aws-credentials.test.ts +++ b/packages/ai/test/aws-credentials.test.ts @@ -9,6 +9,7 @@ import { } from "@oh-my-pi/pi-ai/providers/aws-credentials"; import type { FetchImpl } from "@oh-my-pi/pi-ai/types"; import { removeWithRetries } from "../../utils/src/temp"; +import { waitForDelayOrAbort } from "./helpers"; // `credential_process` integration coverage. Drives a real `Bun.spawn` // against a fixture script so the JSON envelope contract, exit-code @@ -20,12 +21,14 @@ const ENV_KEYS = [ "AWS_SECRET_ACCESS_KEY", "AWS_SESSION_TOKEN", "AWS_PROFILE", + "AWS_SDK_LOAD_CONFIG", "AWS_REGION", "AWS_DEFAULT_REGION", "AWS_CONFIG_FILE", "AWS_SHARED_CREDENTIALS_FILE", "AWS_EC2_METADATA_DISABLED", "AWS_EC2_METADATA_SERVICE_ENDPOINT", + "AWS_EC2_METADATA_SERVICE_ENDPOINT_MODE", "AWS_WEB_IDENTITY_TOKEN_FILE", "AWS_ROLE_ARN", "AWS_ROLE_SESSION_NAME", @@ -221,6 +224,41 @@ describe("resolveAwsCredentials", () => { }); }); + test("rejects dynamic container credentials without expiration", async () => { + const credentialsPath = path.join(tmp, "empty-dynamic-credentials"); + const configPath = path.join(tmp, "empty-dynamic-config"); + await Promise.all([Bun.write(credentialsPath, ""), Bun.write(configPath, "")]); + Bun.env.AWS_SHARED_CREDENTIALS_FILE = credentialsPath; + Bun.env.AWS_CONFIG_FILE = configPath; + Bun.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI = "/v2/credentials/rotating"; + let calls = 0; + const fetchImpl: FetchImpl = Object.assign( + async () => { + calls++; + return Response.json({ + AccessKeyId: "AKIAECS", + SecretAccessKey: "ecs-secret", + Token: "ecs-token", + }); + }, + { preconnect: fetch.preconnect }, + ); + + await expect(resolveAwsCredentials({ fetch: fetchImpl })).rejects.toThrow(/missing or invalid Expiration/); + expect(calls).toBe(1); + }); + + test("rejects container relative URIs that can replace the metadata host", async () => { + const credentialsPath = path.join(tmp, "empty-relative-credentials"); + const configPath = path.join(tmp, "empty-relative-config"); + await Promise.all([Bun.write(credentialsPath, ""), Bun.write(configPath, "")]); + Bun.env.AWS_SHARED_CREDENTIALS_FILE = credentialsPath; + Bun.env.AWS_CONFIG_FILE = configPath; + Bun.env.AWS_CONTAINER_CREDENTIALS_RELATIVE_URI = "//attacker.invalid/credentials"; + + await expect(resolveAwsCredentials()).rejects.toThrow(/single-host absolute path/); + }); + test("honors AWS_EC2_METADATA_SERVICE_ENDPOINT for instance-role credentials", async () => { const credentialsPath = path.join(tmp, "empty-imds-credentials"); const configPath = path.join(tmp, "empty-imds-config"); @@ -257,12 +295,74 @@ describe("resolveAwsCredentials", () => { expect(credentials.sessionToken).toBe("imds-session"); }); + test("uses the IPv6 IMDS endpoint when endpoint mode requests it", async () => { + const credentialsPath = path.join(tmp, "empty-ipv6-imds-credentials"); + const configPath = path.join(tmp, "empty-ipv6-imds-config"); + await Promise.all([Bun.write(credentialsPath, ""), Bun.write(configPath, "")]); + Bun.env.AWS_SHARED_CREDENTIALS_FILE = credentialsPath; + Bun.env.AWS_CONFIG_FILE = configPath; + Bun.env.AWS_EC2_METADATA_DISABLED = "false"; + Bun.env.AWS_EC2_METADATA_SERVICE_ENDPOINT_MODE = "IPv6"; + const requestedUrls: string[] = []; + const fetchImpl: FetchImpl = Object.assign( + async (input: string | URL | Request) => { + const url = String(input); + requestedUrls.push(url); + if (url.endsWith("/latest/api/token")) return new Response("imds-token"); + if (url.endsWith("/latest/meta-data/iam/security-credentials/")) return new Response("test-role"); + return Response.json({ + AccessKeyId: "AKIAIMDS", + SecretAccessKey: "imds-secret", + Token: "imds-session", + Expiration: "2099-01-01T00:00:00Z", + }); + }, + { preconnect: fetch.preconnect }, + ); + + await resolveAwsCredentials({ fetch: fetchImpl }); + + expect(requestedUrls[0]).toBe("http://[fd00:ec2::254]/latest/api/token"); + }); + + test("gives each IMDS request its own timeout budget", async () => { + const credentialsPath = path.join(tmp, "empty-slow-imds-credentials"); + const configPath = path.join(tmp, "empty-slow-imds-config"); + await Promise.all([Bun.write(credentialsPath, ""), Bun.write(configPath, "")]); + Bun.env.AWS_SHARED_CREDENTIALS_FILE = credentialsPath; + Bun.env.AWS_CONFIG_FILE = configPath; + Bun.env.AWS_EC2_METADATA_DISABLED = "false"; + Bun.env.AWS_EC2_METADATA_SERVICE_ENDPOINT = "http://slow-imds.internal"; + let calls = 0; + const fetchImpl: FetchImpl = Object.assign( + async (_input: string | URL | Request, init?: RequestInit) => { + await waitForDelayOrAbort(450, init?.signal ?? undefined); + calls++; + if (calls === 1) return new Response("imds-token"); + if (calls === 2) return new Response("test-role"); + return Response.json({ + AccessKeyId: "AKIASLOWIMDS", + SecretAccessKey: "imds-secret", + Token: "imds-session", + Expiration: "2099-01-01T00:00:00Z", + }); + }, + { preconnect: fetch.preconnect }, + ); + + const credentials = await resolveAwsCredentials({ fetch: fetchImpl }); + + expect(calls).toBe(3); + expect(credentials.accessKeyId).toBe("AKIASLOWIMDS"); + }); + test("exchanges web identity tokens for STS credentials", async () => { const tokenPath = path.join(tmp, "web-identity-token"); await Bun.write(tokenPath, "signed-identity-token\n"); Bun.env.AWS_WEB_IDENTITY_TOKEN_FILE = tokenPath; Bun.env.AWS_ROLE_ARN = "arn:aws:iam::123456789012:role/test-role"; Bun.env.AWS_ROLE_SESSION_NAME = "test-session"; + await writeConfig("regional", "region = cn-north-1"); let requestedUrl = ""; let requestBody = ""; const fetchImpl: FetchImpl = Object.assign( @@ -280,9 +380,9 @@ describe("resolveAwsCredentials", () => { { preconnect: fetch.preconnect }, ); - const credentials = await resolveAwsCredentials({ region: "us-east-2", fetch: fetchImpl }); + const credentials = await resolveAwsCredentials({ profile: "regional", fetch: fetchImpl }); - expect(requestedUrl).toBe("https://sts.us-east-2.amazonaws.com/"); + expect(requestedUrl).toBe("https://sts.cn-north-1.amazonaws.com.cn/"); expect(new URLSearchParams(requestBody).get("WebIdentityToken")).toBe("signed-identity-token"); expect(new URLSearchParams(requestBody).get("RoleSessionName")).toBe("test-session"); expect(credentials).toEqual({ @@ -292,4 +392,26 @@ describe("resolveAwsCredentials", () => { expiresAt: Date.parse("2099-01-01T00:00:00Z"), }); }); + + test("rejects web-identity responses without a valid expiration", async () => { + const tokenPath = path.join(tmp, "web-identity-token-without-expiration"); + await Bun.write(tokenPath, "signed-identity-token\n"); + Bun.env.AWS_WEB_IDENTITY_TOKEN_FILE = tokenPath; + Bun.env.AWS_ROLE_ARN = "arn:aws:iam::123456789012:role/test-role"; + const fetchImpl: FetchImpl = Object.assign( + async () => + new Response( + ` + AKIAWEBweb-secret + web-token + `, + { headers: { "content-type": "text/xml" } }, + ), + { preconnect: fetch.preconnect }, + ); + + await expect(resolveAwsCredentials({ region: "us-east-1", fetch: fetchImpl })).rejects.toThrow( + /missing or invalid Expiration/, + ); + }); }); diff --git a/packages/ai/test/aws-registry.test.ts b/packages/ai/test/aws-registry.test.ts index 4e6509aef..edf118fba 100644 --- a/packages/ai/test/aws-registry.test.ts +++ b/packages/ai/test/aws-registry.test.ts @@ -11,6 +11,7 @@ const EMPTY_AWS_ENV = { AWS_SECRET_ACCESS_KEY: undefined, AWS_BEARER_TOKEN_BEDROCK: undefined, AWS_PROFILE: undefined, + AWS_SDK_LOAD_CONFIG: undefined, AWS_WEB_IDENTITY_TOKEN_FILE: undefined, AWS_ROLE_ARN: undefined, AWS_CONTAINER_CREDENTIALS_RELATIVE_URI: undefined, @@ -61,6 +62,30 @@ describe("AWS provider availability", () => { await removeWithRetries(tmp); } }); + + test("loads implicit default config profiles only when AWS_SDK_LOAD_CONFIG is enabled", async () => { + const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "aws-registry-load-config-")); + try { + const credentialsPath = path.join(tmp, "credentials"); + const configPath = path.join(tmp, "config"); + await Promise.all([ + Bun.write(credentialsPath, ""), + Bun.write(configPath, "[default]\ncredential_process = /bin/credential-helper\n"), + ]); + const env = { + ...EMPTY_AWS_ENV, + AWS_SHARED_CREDENTIALS_FILE: credentialsPath, + AWS_CONFIG_FILE: configPath, + AWS_EC2_METADATA_DISABLED: "true", + }; + await withEnv(env, async () => expect(getEnvApiKey("bedrock-mantle")).toBeUndefined()); + await withEnv({ ...env, AWS_SDK_LOAD_CONFIG: "1" }, async () => + expect(getEnvApiKey("bedrock-mantle")).toBeDefined(), + ); + } finally { + await removeWithRetries(tmp); + } + }); test("recognizes an explicitly configured EC2 metadata endpoint", async () => { await withEnv( { diff --git a/packages/ai/test/bedrock-inference-profile.test.ts b/packages/ai/test/bedrock-inference-profile.test.ts index 495ad7a44..3dc1ac299 100644 --- a/packages/ai/test/bedrock-inference-profile.test.ts +++ b/packages/ai/test/bedrock-inference-profile.test.ts @@ -1,8 +1,12 @@ import { describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { streamBedrock } from "@oh-my-pi/pi-ai/providers/amazon-bedrock"; import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { removeWithRetries } from "../../utils/src/temp"; import { withEnv } from "./helpers"; const profileArn = "arn:aws:bedrock:us-east-2:1234567890:application-inference-profile/company-opus-48"; @@ -137,7 +141,7 @@ function bedrockModel(id: string): Model<"bedrock-converse-stream"> { async function capturedRequestHost( model: Model<"bedrock-converse-stream">, - options: { region?: string } = {}, + options: { region?: string; profile?: string } = {}, ): Promise { const calls: string[] = []; const customFetch: FetchImpl = Object.assign( @@ -212,6 +216,31 @@ describe("Bedrock cross-region inference-profile geo routing", () => { }); }); + test("uses the selected profile region when environment regions are absent", async () => { + const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "bedrock-profile-region-")); + try { + const configPath = path.join(tmp, "config"); + await Bun.write(configPath, "[profile regional]\nregion = eu-west-2\n"); + await withEnv( + { + AWS_REGION: undefined, + AWS_DEFAULT_REGION: undefined, + AWS_PROFILE: "regional", + AWS_CONFIG_FILE: configPath, + }, + async () => { + expect( + await capturedRequestHost(bedrockModel("eu.anthropic.claude-opus-4-8"), { + profile: "regional", + }), + ).toBe("bedrock-runtime.eu-west-2.amazonaws.com"); + }, + ); + } finally { + await removeWithRetries(tmp); + } + }); + test("explicit per-request region wins over the geo prefix and ambient region", async () => { await withEnv({ AWS_REGION: "eu-central-1", AWS_DEFAULT_REGION: undefined }, async () => { expect(await capturedRequestHost(bedrockModel("eu.anthropic.claude-opus-4-8"), { region: "eu-west-3" })).toBe( diff --git a/packages/ai/test/bedrock-mantle-auth.test.ts b/packages/ai/test/bedrock-mantle-auth.test.ts index 48cc61011..df2735ba7 100644 --- a/packages/ai/test/bedrock-mantle-auth.test.ts +++ b/packages/ai/test/bedrock-mantle-auth.test.ts @@ -1,9 +1,14 @@ import { describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { clearAwsCredentialCache } from "@oh-my-pi/pi-ai/providers/aws-credentials"; import type { BedrockMantleOptions } from "@oh-my-pi/pi-ai/providers/bedrock-mantle"; +import { getProviderDefinition } from "@oh-my-pi/pi-ai/registry"; import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream"; import type { Context, FetchImpl, Model, SimpleStreamOptions } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { removeWithRetries } from "../../utils/src/temp"; import { withEnv } from "./helpers"; const mantleModel: Model<"openai-responses"> = buildModel({ @@ -27,6 +32,10 @@ const cleanAwsEnv = { AWS_SESSION_TOKEN: undefined, AWS_PROFILE: undefined, AWS_REGION: undefined, + AWS_CONFIG_FILE: undefined, + AWS_SHARED_CREDENTIALS_FILE: undefined, + AWS_EC2_METADATA_SERVICE_ENDPOINT: undefined, + AWS_EC2_METADATA_SERVICE_ENDPOINT_MODE: undefined, AWS_DEFAULT_REGION: undefined, AWS_EC2_METADATA_DISABLED: "true", }; @@ -35,6 +44,7 @@ interface Capture { url?: string; authorization?: string | null; securityToken?: string | null; + body?: RequestInit["body"]; } function captureFetch(capture: Capture): FetchImpl { @@ -44,6 +54,7 @@ function captureFetch(capture: Capture): FetchImpl { const headers = new Headers(input instanceof Request ? input.headers : init?.headers); capture.authorization = headers.get("authorization"); capture.securityToken = headers.get("x-amz-security-token"); + capture.body = init?.body; return new Response("captured", { status: 418 }); }, { preconnect: fetch.preconnect }, @@ -72,6 +83,60 @@ describe("Bedrock Mantle authentication", () => { expect(capture.authorization).toBe("Bearer test-token"); }); + test("uses the selected profile region when environment regions are absent", async () => { + const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "bedrock-mantle-region-")); + try { + const configPath = path.join(tmp, "config"); + await Bun.write(configPath, "[profile regional]\nregion = eu-west-2\n"); + const capture = await runDirect({ + AWS_BEARER_TOKEN_BEDROCK: "test-token", + AWS_PROFILE: "regional", + AWS_CONFIG_FILE: configPath, + AWS_SHARED_CREDENTIALS_FILE: path.join(tmp, "missing-credentials"), + }); + expect(capture.url).toStartWith("https://bedrock-mantle.eu-west-2.api.aws/openai/v1/responses"); + } finally { + await removeWithRetries(tmp); + } + }); + + test("prepares bearer-authenticated model discovery", async () => { + const capture: Capture = {}; + await withEnv( + { + ...cleanAwsEnv, + AWS_BEARER_TOKEN_BEDROCK: "discovery-token", + AWS_REGION: "eu-west-2", + }, + async () => { + const config = getProviderDefinition("bedrock-mantle")?.prepareModelDiscovery?.({ + fetch: captureFetch(capture), + }); + expect(config?.authenticated).toBeTrue(); + expect(config?.baseUrl).toBe("https://bedrock-mantle.eu-west-2.api.aws/openai/v1"); + await config?.fetch?.("https://bedrock-mantle.eu-west-2.api.aws/v1/models", { method: "GET" }); + }, + ); + expect(capture.authorization).toBe("Bearer discovery-token"); + expect(capture.body).toBeUndefined(); + }); + + test("does not enable account-scoped discovery for SigV4-only credentials", async () => { + await withEnv( + { + ...cleanAwsEnv, + AWS_ACCESS_KEY_ID: "AKIADISCOVERY", + AWS_SECRET_ACCESS_KEY: "discovery-secret", + AWS_REGION: "eu-west-2", + }, + async () => { + const config = getProviderDefinition("bedrock-mantle")?.prepareModelDiscovery?.({}); + expect(config?.authenticated).toBeFalse(); + expect(config?.baseUrl).toBeUndefined(); + }, + ); + }); + test("SigV4-signs with the standard AWS credential chain", async () => { const capture = await runDirect({ AWS_ACCESS_KEY_ID: "AKIAIOSFODNN7EXAMPLE", @@ -84,6 +149,36 @@ describe("Bedrock Mantle authentication", () => { expect(capture.securityToken).toBe("test-session-token"); }); + test("invalidates cached SigV4 credentials after an authentication rejection", async () => { + const authorizations: string[] = []; + const rejectingFetch: FetchImpl = Object.assign( + async (input: string | URL | Request, init?: RequestInit) => { + const headers = new Headers(input instanceof Request ? input.headers : init?.headers); + authorizations.push(headers.get("authorization") ?? ""); + return new Response("rejected", { status: 403 }); + }, + { preconnect: fetch.preconnect }, + ); + await withEnv( + { + ...cleanAwsEnv, + AWS_ACCESS_KEY_ID: "AKIAFIRST", + AWS_SECRET_ACCESS_KEY: "first-secret", + AWS_REGION: "us-west-2", + }, + async () => { + clearAwsCredentialCache(); + await stream(mantleModel, context, { fetch: rejectingFetch, maxTokens: 16 }).result(); + Bun.env.AWS_ACCESS_KEY_ID = "AKIASECOND"; + Bun.env.AWS_SECRET_ACCESS_KEY = "second-secret"; + await stream(mantleModel, context, { fetch: rejectingFetch, maxTokens: 16 }).result(); + }, + ); + expect(authorizations).toHaveLength(2); + expect(authorizations[0]).toContain("Credential=AKIAFIRST/"); + expect(authorizations[1]).toContain("Credential=AKIASECOND/"); + }); + test("streamSimple preserves AWS options and resolver-supplied keys", async () => { const capture: Capture = {}; let resolverCalls = 0; @@ -92,8 +187,10 @@ describe("Bedrock Mantle authentication", () => { resolverCalls++; return "resolved-token"; }, - region: "us-east-2", - profile: "ignored-for-bearer", + providerOptions: { + region: "us-east-2", + profile: "ignored-for-bearer", + }, fetch: captureFetch(capture), maxTokens: 16, }; diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 2d4bbb8df..a1e1e4c62 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -4,11 +4,11 @@ ### Added -- Added the `bedrock-mantle` provider for OpenAI GPT-5.4, GPT-5.5, and GPT-5.6 models served through Amazon Bedrock's Responses endpoint. +- Added the `bedrock-mantle` provider with authenticated model discovery for OpenAI GPT-5.4, GPT-5.5, and GPT-5.6 models served through Amazon Bedrock's Responses endpoint ([#7080](https://github.com/can1357/oh-my-pi/pull/7080) by [@anatoli-tsinovoy](https://github.com/anatoli-tsinovoy)). ### Fixed -- Removed unusable Converse entries for OpenAI models that Amazon Bedrock serves only through Mantle. +- Removed unusable Converse entries for OpenAI models that Amazon Bedrock serves only through Mantle and corrected GPT-5.6 Luna and Terra pricing ([#7080](https://github.com/can1357/oh-my-pi/pull/7080) by [@anatoli-tsinovoy](https://github.com/anatoli-tsinovoy)). ## [17.2.0] - 2026-07-30 diff --git a/packages/catalog/scripts/generate-models.ts b/packages/catalog/scripts/generate-models.ts index 95fb311f7..95f59483f 100644 --- a/packages/catalog/scripts/generate-models.ts +++ b/packages/catalog/scripts/generate-models.ts @@ -14,6 +14,7 @@ import { discoverAuthStorage } from "@oh-my-pi/pi-ai/auth-broker/discover"; import type { OAuthAccess } from "@oh-my-pi/pi-ai/auth-storage"; import type { OAuthProvider } from "@oh-my-pi/pi-ai/oauth/types"; import { getGitLabDuoModels } from "@oh-my-pi/pi-ai/providers/gitlab-duo"; +import { getProviderDefinition } from "@oh-my-pi/pi-ai/registry"; import { $env } from "@oh-my-pi/pi-utils"; import { ANTIGRAVITY_PRIMARY_ENDPOINT, fetchAntigravityDiscoveryModels } from "../src/discovery/antigravity"; import { fetchCodexModels } from "../src/discovery/codex"; @@ -123,7 +124,10 @@ async function fetchProviderModelsFromCatalog( try { console.log(`Fetching models from ${descriptor.catalogDiscovery.label} model manager...`); - const managerOptions = descriptor.createModelManagerOptions({ apiKey }); + const discoveryConfig = { apiKey }; + const preparedConfig = + getProviderDefinition(descriptor.providerId)?.prepareModelDiscovery?.(discoveryConfig) ?? discoveryConfig; + const managerOptions = descriptor.createModelManagerOptions(preparedConfig); const manager = createModelManager(managerOptions); const result = await manager.refresh("online"); // `stale: true` means the dynamic fetch failed and the manager fell back @@ -555,7 +559,8 @@ async function generateModels() { // Seed Meta's documented Muse model so first-run selection does not depend on // credentials or live discovery. allModels.push(...META_MUSE_STATIC_MODELS); - // Bedrock Mantle has no catalog endpoint used by generation. + // Mantle's catalog endpoint is account/API-key scoped. Keep the generated + // bundle deterministic; authenticated runtime discovery may replace this seed. allModels.push(...BEDROCK_MANTLE_STATIC_MODELS); // Seed Sakana's documented Fugu models so the provider is usable when // catalog generation has no live API key. If live `/v1/models` succeeds, diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 9eb1f4d40..de01cbc24 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -2925,8 +2925,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 131072, + "maxTokens": 16384 }, "gemma-3n-e4b-it": { "id": "gemma-3n-e4b-it", @@ -11378,9 +11378,7 @@ "max" ], "supportsDisplay": true - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + } }, "claude-opus-4-0": { "id": "claude-opus-4-0", @@ -11852,157 +11850,6 @@ } } }, - "bedrock-mantle": { - "openai.gpt-5.4": { - "id": "openai.gpt-5.4", - "name": "GPT-5.4", - "api": "openai-responses", - "provider": "bedrock-mantle", - "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.75, - "output": 16.5, - "cacheRead": 0.275, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ] - } - }, - "openai.gpt-5.5": { - "id": "openai.gpt-5.5", - "name": "GPT-5.5", - "api": "openai-responses", - "provider": "bedrock-mantle", - "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 5.5, - "output": 33, - "cacheRead": 0.55, - "cacheWrite": 0 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ] - }, - "contextPromotionTarget": "bedrock-mantle/openai.gpt-5.4" - }, - "openai.gpt-5.6-luna": { - "id": "openai.gpt-5.6-luna", - "name": "GPT-5.6 Luna", - "api": "openai-responses", - "provider": "bedrock-mantle", - "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 1.1, - "output": 6.6, - "cacheRead": 0.11, - "cacheWrite": 1.38 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ] - } - }, - "openai.gpt-5.6-sol": { - "id": "openai.gpt-5.6-sol", - "name": "GPT-5.6 Sol", - "api": "openai-responses", - "provider": "bedrock-mantle", - "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 5.5, - "output": 33, - "cacheRead": 0.55, - "cacheWrite": 6.88 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ] - } - }, - "openai.gpt-5.6-terra": { - "id": "openai.gpt-5.6-terra", - "name": "GPT-5.6 Terra", - "api": "openai-responses", - "provider": "bedrock-mantle", - "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", - "reasoning": true, - "input": [ - "text", - "image" - ], - "cost": { - "input": 2.75, - "output": 16.5, - "cacheRead": 0.28, - "cacheWrite": 3.44 - }, - "contextWindow": 272000, - "maxTokens": 128000, - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh", - "max" - ] - } - } - }, "azure": { "codex-mini": { "id": "codex-mini", @@ -12802,9 +12649,9 @@ "image" ], "cost": { - "input": 1, - "output": 6, - "cacheRead": 0.1, + "input": 0.2, + "output": 1.2, + "cacheRead": 0.02, "cacheWrite": 0 }, "contextWindow": 1050000, @@ -12862,9 +12709,9 @@ "image" ], "cost": { - "input": 2.5, - "output": 15, - "cacheRead": 0.25, + "input": 2, + "output": 12, + "cacheRead": 0.2, "cacheWrite": 0 }, "contextWindow": 1050000, @@ -13343,6 +13190,157 @@ "supportsComputerUseConfig": false } }, + "bedrock-mantle": { + "openai.gpt-5.4": { + "id": "openai.gpt-5.4", + "name": "GPT-5.4", + "api": "openai-responses", + "provider": "bedrock-mantle", + "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.75, + "output": 16.5, + "cacheRead": 0.275, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai.gpt-5.5": { + "id": "openai.gpt-5.5", + "name": "GPT-5.5", + "api": "openai-responses", + "provider": "bedrock-mantle", + "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5.5, + "output": 33, + "cacheRead": 0.55, + "cacheWrite": 0 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + }, + "contextPromotionTarget": "bedrock-mantle/openai.gpt-5.4" + }, + "openai.gpt-5.6-luna": { + "id": "openai.gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-responses", + "provider": "bedrock-mantle", + "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.22, + "output": 1.32, + "cacheRead": 0.022, + "cacheWrite": 0.275 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "openai.gpt-5.6-sol": { + "id": "openai.gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-responses", + "provider": "bedrock-mantle", + "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5.5, + "output": 33, + "cacheRead": 0.55, + "cacheWrite": 6.88 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, + "openai.gpt-5.6-terra": { + "id": "openai.gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-responses", + "provider": "bedrock-mantle", + "baseUrl": "https://bedrock-mantle.{region}.api.aws/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.2, + "output": 13.2, + "cacheRead": 0.22, + "cacheWrite": 2.75 + }, + "contextWindow": 272000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + } + }, "cerebras": { "gemma-4-31b": { "id": "gemma-4-31b", @@ -15147,6 +15145,39 @@ ] } }, + "moonshotai/Kimi-K3": { + "id": "moonshotai/Kimi-K3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "coreweave", + "baseUrl": "https://api.inference.wandb.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8": { "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", "name": "Nemotron 3 Super", @@ -22801,6 +22832,39 @@ ] } }, + "moonshotai/Kimi-K3": { + "id": "moonshotai/Kimi-K3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", @@ -23297,6 +23361,65 @@ ] } }, + "tencent/Hy3": { + "id": "tencent/Hy3", + "name": "Hy3", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.58, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "thinkingmachines/Inkling": { + "id": "thinkingmachines/Inkling", + "name": "Inkling", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 4.05, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "XiaomiMiMo/MiMo-V2-Flash": { "id": "XiaomiMiMo/MiMo-V2-Flash", "name": "MiMo-V2-Flash", @@ -25997,6 +26120,26 @@ ] } }, + "deepseek/deepseek-v4-flash-0731": { + "id": "deepseek/deepseek-v4-flash-0731", + "name": "DeepSeek V4 Flash 0731", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsComputerUse": false + }, "deepseek/deepseek-v4-flash:discounted": { "id": "deepseek/deepseek-v4-flash:discounted", "name": "DeepSeek V4 Flash (lowest price)", @@ -26798,13 +26941,14 @@ }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", - "name": "Gemma 3 12B", + "name": "Gemma 3 12B IT", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", "reasoning": false, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -26838,13 +26982,14 @@ }, "google/gemma-3-4b-it": { "id": "google/gemma-3-4b-it", - "name": "Gemma 3 4B", + "name": "Gemma 3 4B IT", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", "reasoning": false, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -26852,8 +26997,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null, + "contextWindow": 131072, + "maxTokens": 16384, "supportsComputerUse": false }, "google/gemma-3n-e4b-it": { @@ -28562,7 +28707,7 @@ }, "mistralai/mistral-7b-instruct-v0.3": { "id": "mistralai/mistral-7b-instruct-v0.3", - "name": "Mistral 7B Instruct v0.3", + "name": "Mistral-7B-Instruct-v0.3", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -28576,8 +28721,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null, + "contextWindow": 65536, + "maxTokens": 65536, "supportsComputerUse": false, "supportsComputerUseConfig": false }, @@ -29497,7 +29642,7 @@ }, "nvidia/llama-3.1-nemotron-70b-instruct": { "id": "nvidia/llama-3.1-nemotron-70b-instruct", - "name": "Llama 3.1 Nemotron 70b Instruct", + "name": "Llama 3.1 Nemotron 70B Instruct", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -29518,7 +29663,7 @@ }, "nvidia/llama-3.1-nemotron-ultra-253b-v1": { "id": "nvidia/llama-3.1-nemotron-ultra-253b-v1", - "name": "Llama-3.1-Nemotron-Ultra-253B-v1", + "name": "Llama 3.1 Nemotron Ultra 253B", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -29766,13 +29911,14 @@ }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", - "name": "Nemotron Nano 12B 2 VL", + "name": "Nemotron Nano 12B v2 VL", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -29780,10 +29926,20 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null, + "contextWindow": 128000, + "maxTokens": 128000, "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUseConfig": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "nvidia/nemotron-nano-9b-v2": { "id": "nvidia/nemotron-nano-9b-v2", @@ -30296,7 +30452,8 @@ }, "contextWindow": 128000, "maxTokens": 16384, - "supportsComputerUse": false + "supportsComputerUse": false, + "supportsComputerUseConfig": false }, "openai/gpt-5-codex": { "id": "openai/gpt-5-codex", @@ -31946,7 +32103,8 @@ "xhigh" ] }, - "supportsComputerUse": false + "supportsComputerUse": false, + "supportsComputerUseConfig": false }, "poolside/laguna-m.1:free": { "id": "poolside/laguna-m.1:free", @@ -32023,7 +32181,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -32035,7 +32193,17 @@ }, "contextWindow": 262144, "maxTokens": 32768, - "supportsComputerUse": false + "supportsComputerUse": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "poolside/laguna-xs-2.1:free": { "id": "poolside/laguna-xs-2.1:free", @@ -33464,6 +33632,26 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "qwen/qwen3.7-flash": { + "id": "qwen/qwen3.7-flash", + "name": "Qwen3.7 Flash", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsComputerUse": false + }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen3.7 Max", @@ -34272,6 +34460,37 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "supportsComputerUse": false, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "thinkingmachines/inkling-small": { + "id": "thinkingmachines/inkling-small", + "name": "Inkling Small", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", "reasoning": false, "input": [ "text" @@ -34283,7 +34502,7 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 65536, + "maxTokens": 1000000, "supportsComputerUse": false }, "tngtech/deepseek-r1t2-chimera": { @@ -48696,8 +48915,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null, + "contextWindow": 131072, + "maxTokens": 65536, "supportsComputerUse": false, "supportsComputerUseConfig": false }, @@ -52828,8 +53047,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null, + "contextWindow": 1000000, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -55320,8 +55539,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null, + "contextWindow": 131072, + "maxTokens": 16384, "supportsComputerUse": false, "supportsComputerUseConfig": false }, @@ -57508,6 +57727,35 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "deepseek/deepseek-v4-flash-0731": { + "id": "deepseek/deepseek-v4-flash-0731", + "name": "Deepseek V4 Flash 0731", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.28, + "cacheRead": 0.028, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393216, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "high", + "max" + ] + }, + "supportsComputerUse": false, + "supportsComputerUseConfig": false + }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", "name": "DeepSeek V4 Pro", @@ -57561,7 +57809,7 @@ }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", - "name": "Gemma3 12B", + "name": "Gemma 3 12B IT", "api": "openai-completions", "provider": "novita", "baseUrl": "https://api.novita.ai/openai/v1", @@ -58002,6 +58250,38 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "mindai/macaron-v1-tall": { + "id": "mindai/macaron-v1-tall", + "name": "Macaron V1 Tall", + "api": "openai-completions", + "provider": "novita", + "baseUrl": "https://api.novita.ai/openai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "supportsTools": true, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false, + "supportsComputerUseConfig": false + }, "mindai/macaron-v1-venti": { "id": "mindai/macaron-v1-venti", "name": "Macaron V1 Venti", @@ -60015,7 +60295,7 @@ }, "abacusai/dracarys-llama-3.1-70b-instruct": { "id": "abacusai/dracarys-llama-3.1-70b-instruct", - "name": "abacusai/dracarys-llama-3.1-70b-instruct", + "name": "dracarys-llama-3.1-70b-instruct", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", @@ -60029,8 +60309,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 16384 + "contextWindow": 128000, + "maxTokens": 8192 }, "adept/fuyu-8b": { "id": "adept/fuyu-8b", @@ -60456,13 +60736,14 @@ }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", - "name": "Gemma 3 12b It", + "name": "Gemma 3 12B IT", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", "reasoning": false, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -60470,8 +60751,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 128000, - "maxTokens": 4096 + "contextWindow": 131072, + "maxTokens": 16384 }, "google/gemma-3-1b-it": { "id": "google/gemma-3-1b-it", @@ -60515,13 +60796,14 @@ }, "google/gemma-3-4b-it": { "id": "google/gemma-3-4b-it", - "name": "google/gemma-3-4b-it", + "name": "Gemma 3 4B IT", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", "reasoning": false, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -60529,8 +60811,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 131072, + "maxTokens": 16384 }, "google/gemma-3n-e2b-it": { "id": "google/gemma-3n-e2b-it", @@ -61390,11 +61672,11 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144 + "maxTokens": 16384 }, "mistralai/mistral-7b-instruct-v0.3": { "id": "mistralai/mistral-7b-instruct-v0.3", - "name": "mistralai/mistral-7b-instruct-v0.3", + "name": "Mistral-7B-Instruct-v0.3", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", @@ -61408,8 +61690,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 65536, + "maxTokens": 65536 }, "mistralai/mistral-7b-instruct-v03": { "id": "mistralai/mistral-7b-instruct-v03", @@ -61490,7 +61772,7 @@ }, "mistralai/mistral-medium-3.5-128b": { "id": "mistralai/mistral-medium-3.5-128b", - "name": "Mistral Medium 3.5 128B", + "name": "Mistral Medium 3.5", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", @@ -61506,7 +61788,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -61830,13 +62112,14 @@ }, "nvidia/cosmos-reason2-8b": { "id": "nvidia/cosmos-reason2-8b", - "name": "nvidia/cosmos-reason2-8b", + "name": "Cosmos Reason2 8B", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -61844,8 +62127,18 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 131072, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "nvidia/embed-qa-4": { "id": "nvidia/embed-qa-4", @@ -62021,7 +62314,7 @@ }, "nvidia/llama-3.1-nemotron-70b-instruct": { "id": "nvidia/llama-3.1-nemotron-70b-instruct", - "name": "Llama 3.1 Nemotron 70b Instruct", + "name": "Llama 3.1 Nemotron 70B Instruct", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", @@ -62036,15 +62329,15 @@ "cacheWrite": 0 }, "contextWindow": 128000, - "maxTokens": 4096 + "maxTokens": 8192 }, "nvidia/llama-3.1-nemotron-nano-8b-v1": { "id": "nvidia/llama-3.1-nemotron-nano-8b-v1", - "name": "nvidia/llama-3.1-nemotron-nano-8b-v1", + "name": "Llama 3.1 Nemotron Nano 8B v1", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -62054,18 +62347,29 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 131072, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "nvidia/llama-3.1-nemotron-nano-vl-8b-v1": { "id": "nvidia/llama-3.1-nemotron-nano-vl-8b-v1", - "name": "nvidia/llama-3.1-nemotron-nano-vl-8b-v1", + "name": "Llama 3.1 Nemotron Nano VL 8B v1", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -62073,8 +62377,18 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 32768, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "nvidia/llama-3.1-nemotron-safety-guard-8b-v3": { "id": "nvidia/llama-3.1-nemotron-safety-guard-8b-v3", @@ -62097,7 +62411,7 @@ }, "nvidia/llama-3.1-nemotron-ultra-253b-v1": { "id": "nvidia/llama-3.1-nemotron-ultra-253b-v1", - "name": "Llama-3.1-Nemotron-Ultra-253B-v1", + "name": "Llama 3.1 Nemotron Ultra 253B", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", @@ -62111,8 +62425,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 8192, + "contextWindow": 128000, + "maxTokens": 16384, "thinking": { "mode": "effort", "efforts": [ @@ -62164,26 +62478,7 @@ }, "nvidia/llama-3.3-nemotron-super-49b-v1": { "id": "nvidia/llama-3.3-nemotron-super-49b-v1", - "name": "nvidia/llama-3.3-nemotron-super-49b-v1", - "api": "openai-completions", - "provider": "nvidia", - "baseUrl": "https://integrate.api.nvidia.com/v1", - "reasoning": false, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": null, - "maxTokens": null - }, - "nvidia/llama-3.3-nemotron-super-49b-v1.5": { - "id": "nvidia/llama-3.3-nemotron-super-49b-v1.5", - "name": "Llama 3.3 Nemotron Super 49B V1.5", + "name": "Llama 3.3 Nemotron Super 49B v1", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", @@ -62198,7 +62493,36 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 16384, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "nvidia/llama-3.3-nemotron-super-49b-v1.5": { + "id": "nvidia/llama-3.3-nemotron-super-49b-v1.5", + "name": "Llama 3.3 Nemotron Super 49B v1.5", + "api": "openai-completions", + "provider": "nvidia", + "baseUrl": "https://integrate.api.nvidia.com/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -62539,13 +62863,14 @@ }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", - "name": "nvidia/nemotron-nano-12b-v2-vl", + "name": "Nemotron Nano 12B v2 VL", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", - "reasoning": false, + "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -62553,8 +62878,18 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 128000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "nvidia/nemotron-nano-3-30b-a3b": { "id": "nvidia/nemotron-nano-3-30b-a3b", @@ -62867,6 +63202,35 @@ ] } }, + "poolside/laguna-xs-2.1": { + "id": "poolside/laguna-xs-2.1", + "name": "Laguna XS 2.1", + "api": "openai-completions", + "provider": "nvidia", + "baseUrl": "https://integrate.api.nvidia.com/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "qwen/qwen2.5-coder-32b-instruct": { "id": "qwen/qwen2.5-coder-32b-instruct", "name": "Qwen2.5 Coder 32b Instruct", @@ -63174,6 +63538,36 @@ "contextWindow": null, "maxTokens": null }, + "thinkingmachines/inkling": { + "id": "thinkingmachines/inkling", + "name": "Inkling", + "api": "openai-completions", + "provider": "nvidia", + "baseUrl": "https://integrate.api.nvidia.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "upstage/solar-10_7b-instruct": { "id": "upstage/solar-10_7b-instruct", "name": "solar-10.7b-instruct", @@ -63195,7 +63589,7 @@ }, "upstage/solar-10.7b-instruct": { "id": "upstage/solar-10.7b-instruct", - "name": "upstage/solar-10.7b-instruct", + "name": "solar-10.7b-instruct", "api": "openai-completions", "provider": "nvidia", "baseUrl": "https://integrate.api.nvidia.com/v1", @@ -63209,8 +63603,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 128000, + "maxTokens": 8192 }, "writer/palmyra-creative-122b": { "id": "writer/palmyra-creative-122b", @@ -64108,11 +64502,24 @@ }, "kimi-k3": { "id": "kimi-k3", - "name": "Kimi K3", + "name": "kimi-k3", "api": "ollama-chat", "provider": "ollama-cloud", "baseUrl": "https://ollama.com", "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "omitMaxOutputTokens": true, "thinking": { "mode": "effort", "efforts": [ @@ -64122,21 +64529,7 @@ ], "defaultLevel": "max", "requiresEffort": true - }, - "input": [ - "text", - "image" - ], - "cost": { - "input": 3, - "output": 15, - "cacheRead": 0.3, - "cacheWrite": 0 - }, - "contextWindow": 1048576, - "maxTokens": 8192, - "omitMaxOutputTokens": true, - "supportsComputerUse": false + } }, "minimax-m2": { "id": "minimax-m2", @@ -65627,10 +66020,10 @@ "image" ], "cost": { - "input": 1, - "output": 6, - "cacheRead": 0.1, - "cacheWrite": 1.25 + "input": 0.2, + "output": 1.2, + "cacheRead": 0.02, + "cacheWrite": 0.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -65658,10 +66051,10 @@ "image" ], "cost": { - "input": 1, - "output": 6, - "cacheRead": 0.1, - "cacheWrite": 1.25 + "input": 0.2, + "output": 1.2, + "cacheRead": 0.02, + "cacheWrite": 0.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -65755,10 +66148,10 @@ "image" ], "cost": { - "input": 2.5, - "output": 15, - "cacheRead": 0.25, - "cacheWrite": 3.125 + "input": 2, + "output": 12, + "cacheRead": 0.2, + "cacheWrite": 2.5 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -65786,10 +66179,10 @@ "image" ], "cost": { - "input": 2.5, - "output": 15, - "cacheRead": 0.25, - "cacheWrite": 3.125 + "input": 2, + "output": 12, + "cacheRead": 0.2, + "cacheWrite": 2.5 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -66086,43 +66479,6 @@ } }, "openai-codex": { - "gpt-5.3-codex-spark": { - "id": "gpt-5.3-codex-spark", - "name": "GPT-5.3 Codex Spark", - "api": "openai-codex-responses", - "provider": "openai-codex", - "baseUrl": "https://chatgpt.com/backend-api", - "reasoning": true, - "input": [ - "text" - ], - "cost": { - "input": 1.75, - "output": 14, - "cacheRead": 0.175, - "cacheWrite": 0 - }, - "remoteCompaction": { - "enabled": true, - "api": "openai-codex-responses", - "v2StreamingEnabled": true - }, - "contextWindow": 128000, - "maxTokens": 128000, - "preferWebsockets": true, - "priority": 26, - "applyPatchToolType": "freeform", - "thinking": { - "mode": "effort", - "efforts": [ - "low", - "medium", - "high", - "xhigh" - ] - }, - "contextPromotionTarget": "openai-codex/gpt-5.5" - }, "gpt-5.4": { "id": "gpt-5.4", "name": "GPT-5.4", @@ -66247,10 +66603,10 @@ "image" ], "cost": { - "input": 1, - "output": 6, - "cacheRead": 0.1, - "cacheWrite": 1.25 + "input": 0.2, + "output": 1.2, + "cacheRead": 0.02, + "cacheWrite": 0.25 }, "remoteCompaction": { "enabled": true, @@ -66325,10 +66681,10 @@ "image" ], "cost": { - "input": 2.5, - "output": 15, - "cacheRead": 0.25, - "cacheWrite": 3.125 + "input": 2, + "output": 12, + "cacheRead": 0.2, + "cacheWrite": 2.5 }, "remoteCompaction": { "enabled": true, @@ -66503,7 +66859,7 @@ "opencode-go": { "deepseek-v4-flash": { "id": "deepseek-v4-flash", - "name": "DeepSeek V4 Flash", + "name": "DeepSeek V4 Flash (New)", "api": "openai-completions", "provider": "opencode-go", "baseUrl": "https://opencode.ai/zen/go/v1", @@ -66652,6 +67008,36 @@ ] } }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6 Luna (2x usage)", + "api": "openai-responses", + "provider": "opencode-go", + "baseUrl": "https://opencode.ai/zen/go/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.1, + "output": 0.6, + "cacheRead": 0.01, + "cacheWrite": 0.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh", + "max" + ] + } + }, "grok-4.5": { "id": "grok-4.5", "name": "Grok 4.5", @@ -66806,7 +67192,7 @@ }, "kimi-k3": { "id": "kimi-k3", - "name": "Kimi K3 (2x usage)", + "name": "Kimi K3", "api": "openai-completions", "provider": "opencode-go", "baseUrl": "https://opencode.ai/zen/go/v1", @@ -67605,7 +67991,7 @@ }, "deepseek-v4-flash-free": { "id": "deepseek-v4-flash-free", - "name": "DeepSeek V4 Flash Free", + "name": "DeepSeek V4 Flash Free (New)", "api": "openai-completions", "provider": "opencode-zen", "baseUrl": "https://opencode.ai/zen/v1", @@ -68481,10 +68867,10 @@ "image" ], "cost": { - "input": 1, - "output": 6, - "cacheRead": 0.1, - "cacheWrite": 1.25 + "input": 0.2, + "output": 1.2, + "cacheRead": 0.02, + "cacheWrite": 0.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -69624,13 +70010,13 @@ "image" ], "cost": { - "input": 3, - "output": 15, - "cacheRead": 0.3, + "input": 2.9000000000000004, + "output": 14, + "cacheRead": 0.29, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 262144, + "maxTokens": 1048576, "thinking": { "mode": "effort", "efforts": [ @@ -71205,8 +71591,8 @@ "text" ], "cost": { - "input": 0.20020000000000002, - "output": 0.8000999999999999, + "input": 0.2574, + "output": 1.0287, "cacheRead": 0.15, "cacheWrite": 0 }, @@ -71452,6 +71838,33 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "deepseek/deepseek-v4-flash-0731": { + "id": "deepseek/deepseek-v4-flash-0731", + "name": "DeepSeek V4 Flash 0731", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.28, + "cacheRead": 0.0028, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 384000, + "thinking": { + "mode": "effort", + "efforts": [ + "high" + ] + }, + "supportsComputerUse": false, + "supportsComputerUseConfig": false + }, "deepseek/deepseek-v4-flash:free": { "id": "deepseek/deepseek-v4-flash:free", "name": "DeepSeek V4 Flash (free)", @@ -72095,7 +72508,7 @@ }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", - "name": "Gemma 3 12B", + "name": "Gemma 3 12B IT", "api": "openrouter", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -72171,13 +72584,13 @@ "image" ], "cost": { - "input": 0.14, - "output": 0.42, + "input": 0.07, + "output": 0.33999999999999997, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 16384, "thinking": { "mode": "effort", "efforts": [ @@ -72233,9 +72646,9 @@ "image" ], "cost": { - "input": 0.14, - "output": 0.39999999999999997, - "cacheRead": 0.09, + "input": 0.09999999999999999, + "output": 0.33999999999999997, + "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, "contextWindow": 262144, @@ -73591,8 +74004,8 @@ "image" ], "cost": { - "input": 0.09999999999999999, - "output": 0.3, + "input": 0.075, + "output": 0.19999999999999998, "cacheRead": 0.01, "cacheWrite": 0 }, @@ -73844,9 +74257,9 @@ "image" ], "cost": { - "input": 0.646, - "output": 2.7199999999999998, - "cacheRead": 0.1088, + "input": 0.6, + "output": 3.41, + "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, "contextWindow": 262144, @@ -74136,7 +74549,7 @@ }, "nvidia/llama-3.1-nemotron-70b-instruct": { "id": "nvidia/llama-3.1-nemotron-70b-instruct", - "name": "Llama 3.1 Nemotron 70b Instruct", + "name": "Llama 3.1 Nemotron 70B Instruct", "api": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", "provider": "openrouter", @@ -74157,7 +74570,7 @@ }, "nvidia/llama-3.3-nemotron-super-49b-v1.5": { "id": "nvidia/llama-3.3-nemotron-super-49b-v1.5", - "name": "Llama 3.3 Nemotron Super 49B V1.5", + "name": "Llama 3.3 Nemotron Super 49B v1.5", "api": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", "provider": "openrouter", @@ -74198,11 +74611,11 @@ "cost": { "input": 0.049999999999999996, "output": 0.19999999999999998, - "cacheRead": 0, + "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 228000, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -75627,10 +76040,10 @@ "image" ], "cost": { - "input": 0.5, - "output": 3, - "cacheRead": 0.049999999999999996, - "cacheWrite": 0.625 + "input": 0.09999999999999999, + "output": 0.6000000000000001, + "cacheRead": 0.01, + "cacheWrite": 0.12500000000000003 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -75659,10 +76072,10 @@ "image" ], "cost": { - "input": 0.5, - "output": 3, - "cacheRead": 0.049999999999999996, - "cacheWrite": 0.625 + "input": 0.09999999999999999, + "output": 0.6000000000000001, + "cacheRead": 0.01, + "cacheWrite": 0.12500000000000003 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -75755,10 +76168,10 @@ "image" ], "cost": { - "input": 1.25, - "output": 7.5, - "cacheRead": 0.125, - "cacheWrite": 1.5625 + "input": 1.0000000000000002, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 1.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -75787,10 +76200,10 @@ "image" ], "cost": { - "input": 1.25, - "output": 7.5, - "cacheRead": 0.125, - "cacheWrite": 1.5625 + "input": 1.0000000000000002, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 1.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -75979,7 +76392,7 @@ ], "cost": { "input": 0.03, - "output": 0.14, + "output": 0.13, "cacheRead": 0.03, "cacheWrite": 0 }, @@ -76639,9 +77052,9 @@ "text" ], "cost": { - "input": 0.09999999999999999, - "output": 0.19999999999999998, - "cacheRead": 0.01, + "input": 0.09, + "output": 0.18, + "cacheRead": 0.009, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -76870,8 +77283,8 @@ "text" ], "cost": { - "input": 0.04, - "output": 0.09999999999999999, + "input": 0.09999999999999999, + "output": 0.19999999999999998, "cacheRead": 0, "cacheWrite": 0 }, @@ -76954,10 +77367,10 @@ "text" ], "cost": { - "input": 0.26, - "output": 0.78, + "input": 0.39999999999999997, + "output": 1.2, "cacheRead": 0, - "cacheWrite": 0.325 + "cacheWrite": 0.5 }, "contextWindow": 1000000, "maxTokens": 32768, @@ -77108,8 +77521,8 @@ "text" ], "cost": { - "input": 0.3, - "output": 3, + "input": 0.22999999999999998, + "output": 2.3, "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, @@ -77190,8 +77603,8 @@ "text" ], "cost": { - "input": 0.13, - "output": 1.56, + "input": 0.19999999999999998, + "output": 2.4, "cacheRead": 0.08, "cacheWrite": 0 }, @@ -77363,12 +77776,12 @@ ], "cost": { "input": 0.07, - "output": 0.27, + "output": 0.28, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768, + "maxTokens": 262144, "supportsComputerUse": false, "supportsComputerUseConfig": false }, @@ -77404,7 +77817,7 @@ "text" ], "cost": { - "input": 0.11, + "input": 0.12, "output": 0.7999999999999999, "cacheRead": 0.07, "cacheWrite": 0 @@ -77494,7 +77907,7 @@ "cacheWrite": 0.975 }, "contextWindow": 262144, - "maxTokens": 32768, + "maxTokens": 65536, "supportsComputerUse": false, "supportsComputerUseConfig": false }, @@ -77515,7 +77928,7 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 32768, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -77582,8 +77995,8 @@ "text" ], "cost": { - "input": 0.0975, - "output": 0.78, + "input": 0.15, + "output": 1.2, "cacheRead": 0, "cacheWrite": 0 }, @@ -77636,8 +78049,8 @@ "image" ], "cost": { - "input": 0.26, - "output": 2.6, + "input": 0.39999999999999997, + "output": 4, "cacheRead": 0, "cacheWrite": 0 }, @@ -77668,13 +78081,13 @@ "image" ], "cost": { - "input": 0.15, - "output": 0.6, + "input": 0.13, + "output": 0.52, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384, + "maxTokens": 32768, "supportsComputerUse": false, "supportsComputerUseConfig": false }, @@ -77690,8 +78103,8 @@ "image" ], "cost": { - "input": 0.13, - "output": 1.56, + "input": 0.19999999999999998, + "output": 2.4, "cacheRead": 0, "cacheWrite": 0 }, @@ -77766,8 +78179,8 @@ "image" ], "cost": { - "input": 0.117, - "output": 1.365, + "input": 0.18, + "output": 2.0999999999999996, "cacheRead": 0, "cacheWrite": 0 }, @@ -78138,10 +78551,10 @@ "text" ], "cost": { - "input": 1.04, - "output": 6.24, + "input": 1.0270000000000001, + "output": 6.162, "cacheRead": 0, - "cacheWrite": 1.3 + "cacheWrite": 1.28375 }, "contextWindow": 262144, "maxTokens": 65536, @@ -78249,6 +78662,37 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "qwen/qwen3.7-flash": { + "id": "qwen/qwen3.7-flash", + "name": "Qwen3.7 Flash", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.03, + "output": 0.13, + "cacheRead": 0.006, + "cacheWrite": 0.038000000000000006 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + }, + "supportsComputerUse": false, + "supportsComputerUseConfig": false + }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen3.7 Max", @@ -78266,7 +78710,7 @@ "cacheWrite": 1.84375 }, "contextWindow": 1000000, - "maxTokens": 65536, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -78297,7 +78741,7 @@ "cacheWrite": 0.39999999999999997 }, "contextWindow": 1000000, - "maxTokens": 65536, + "maxTokens": 131072, "thinking": { "mode": "effort", "efforts": [ @@ -78735,7 +79179,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 32768, + "contextWindow": 1024000, "maxTokens": 32768, "supportsComputerUse": false, "supportsComputerUseConfig": false @@ -78771,6 +79215,37 @@ "supportsComputerUse": false, "supportsComputerUseConfig": false }, + "thinkingmachines/inkling-small": { + "id": "thinkingmachines/inkling-small", + "name": "Inkling Small", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.5, + "output": 1.2, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 524288, + "maxTokens": 1000000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + }, + "supportsComputerUse": false, + "supportsComputerUseConfig": false + }, "tngtech/deepseek-r1t2-chimera": { "id": "tngtech/deepseek-r1t2-chimera", "name": "DeepSeek R1T2 Chimera", @@ -79800,13 +80275,13 @@ "text" ], "cost": { - "input": 0.7714000000000001, - "output": 2.4244, - "cacheRead": 0.14326, + "input": 0.76006, + "output": 2.38876, + "cacheRead": 0.141154, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 128000, "thinking": { "mode": "effort", "efforts": [ @@ -80007,7 +80482,7 @@ "synthetic": { "hf:MiniMaxAI/MiniMax-M3": { "id": "hf:MiniMaxAI/MiniMax-M3", - "name": "MiniMaxAI/MiniMax-M3", + "name": "MiniMax-M3", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -80022,7 +80497,7 @@ "cacheRead": 0.6, "cacheWrite": 0 }, - "contextWindow": 262144, + "contextWindow": 524288, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -80033,13 +80508,11 @@ "high", "xhigh" ] - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + } }, "hf:moonshotai/Kimi-K2.7-Code": { "id": "hf:moonshotai/Kimi-K2.7-Code", - "name": "moonshotai/Kimi-K2.7-Code", + "name": "Kimi K2.7 Code", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -80065,13 +80538,44 @@ "high", "xhigh" ] + } + }, + "hf:moonshotai/Kimi-K3": { + "id": "hf:moonshotai/Kimi-K3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "synthetic", + "baseUrl": "https://api.synthetic.new/openai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.45, + "cacheWrite": 0 }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "contextWindow": 524288, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } }, "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4": { "id": "hf:nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", - "name": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-NVFP4", + "name": "Nemotron 3 Super 120B A12B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -80096,13 +80600,11 @@ "high", "xhigh" ] - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + } }, "hf:openai/gpt-oss-120b": { "id": "hf:openai/gpt-oss-120b", - "name": "openai/gpt-oss-120b", + "name": "GPT OSS 120B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -80125,13 +80627,11 @@ "medium", "high" ] - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + } }, "hf:Qwen/Qwen3.6-27B": { "id": "hf:Qwen/Qwen3.6-27B", - "name": "Qwen/Qwen3.6-27B", + "name": "Qwen3.6 27B", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -80156,13 +80656,11 @@ "medium", "high" ] - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + } }, "hf:zai-org/GLM-4.7-Flash": { "id": "hf:zai-org/GLM-4.7-Flash", - "name": "zai-org/GLM-4.7-Flash", + "name": "GLM-4.7-Flash", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -80187,13 +80685,11 @@ "high", "xhigh" ] - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + } }, "hf:zai-org/GLM-5.2": { "id": "hf:zai-org/GLM-5.2", - "name": "zai-org/GLM-5.2", + "name": "GLM-5.2", "api": "openai-completions", "provider": "synthetic", "baseUrl": "https://api.synthetic.new/openai/v1", @@ -80218,9 +80714,7 @@ "high", "xhigh" ] - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + } }, "syn:large:text": { "id": "syn:large:text", @@ -80726,6 +81220,39 @@ ] } }, + "moonshotai/Kimi-K3": { + "id": "moonshotai/Kimi-K3", + "name": "Kimi K3", + "api": "openai-completions", + "provider": "together", + "baseUrl": "https://api.together.xyz/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "high", + "max" + ], + "defaultLevel": "max", + "effortMap": { + "max": "max" + }, + "requiresEffort": true + } + }, "nvidia/nemotron-3-ultra-550b-a55b": { "id": "nvidia/nemotron-3-ultra-550b-a55b", "name": "Nemotron 3 Ultra 550B A55B", @@ -81174,6 +81701,40 @@ "escapeBuiltinToolNames": true } }, + "umans-deepseek-v4-flash-0731": { + "id": "umans-deepseek-v4-flash-0731", + "name": "Umans DeepSeek V4 Flash (experimental)", + "api": "anthropic-messages", + "provider": "umans", + "baseUrl": "https://api.code.umans.ai", + "reasoning": true, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 393215, + "supportsComputerUse": false, + "supportsComputerUseConfig": false, + "compat": { + "escapeBuiltinToolNames": true + } + }, "umans-flash": { "id": "umans-flash", "name": "Umans Flash", @@ -81278,7 +81839,7 @@ }, "umans-kimi-k3": { "id": "umans-kimi-k3", - "name": "Umans Kimi K3 (prerelease)", + "name": "Umans Kimi K3", "api": "anthropic-messages", "provider": "umans", "baseUrl": "https://api.code.umans.ai", @@ -81306,6 +81867,7 @@ "contextWindow": 1048576, "maxTokens": 131071, "supportsComputerUse": false, + "supportsComputerUseConfig": false, "compat": { "escapeBuiltinToolNames": true } @@ -82001,6 +82563,30 @@ ] } }, + "deepseek-v4-flash-0731": { + "id": "deepseek-v4-flash-0731", + "name": "deepseek-v4-flash-0731", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 393216, + "supportsComputerUse": false, + "supportsComputerUseConfig": false, + "compat": { + "supportsUsageInStreaming": false + } + }, "deepseek-v4-pro": { "id": "deepseek-v4-pro", "name": "DeepSeek V4 Pro", @@ -83116,7 +83702,7 @@ "cacheRead": 0.2125, "cacheWrite": 0 }, - "contextWindow": 1000000, + "contextWindow": 524288, "maxTokens": 65536, "thinking": { "mode": "effort", @@ -83654,7 +84240,9 @@ "high", "xhigh" ] - } + }, + "supportsComputerUse": false, + "supportsComputerUseConfig": false }, "olafangensan-glm-4.7-flash-heretic": { "id": "olafangensan-glm-4.7-flash-heretic", @@ -84469,9 +85057,9 @@ "image" ], "cost": { - "input": 0.15, + "input": 0.1, "output": 1, - "cacheRead": 0.05, + "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 256000, @@ -85559,6 +86147,37 @@ }, "supportsComputerUse": false }, + "alibaba/qwen3.7-flash": { + "id": "alibaba/qwen3.7-flash", + "name": "Qwen 3.7 Flash", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.03, + "output": 0.13, + "cacheRead": 0.006, + "cacheWrite": 0.038000000000000006 + }, + "contextWindow": 991000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false + }, "alibaba/qwen3.7-max": { "id": "alibaba/qwen3.7-max", "name": "Qwen 3.7 Max", @@ -86684,6 +87303,36 @@ "cost": { "input": 0.14, "output": 0.28, + "cacheRead": 0.0028, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 384000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false + }, + "deepseek/deepseek-v4-flash-0731": { + "id": "deepseek/deepseek-v4-flash-0731", + "name": "DeepSeek V4 Flash 0731", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.13, + "output": 0.26, "cacheRead": 0.028, "cacheWrite": 0 }, @@ -86986,7 +87635,8 @@ ], "requiresEffort": true }, - "supportsComputerUse": false + "supportsComputerUse": false, + "supportsComputerUseConfig": false }, "google/gemini-3.1-flash-lite": { "id": "google/gemini-3.1-flash-lite", @@ -87344,7 +87994,8 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.15, @@ -87414,7 +88065,8 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.74, @@ -88144,7 +88796,8 @@ "provider": "vercel-ai-gateway", "reasoning": false, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.09999999999999999, @@ -88164,7 +88817,8 @@ "provider": "vercel-ai-gateway", "reasoning": false, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.15, @@ -88257,7 +88911,8 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": false, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.15, @@ -88725,7 +89380,7 @@ }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", - "name": "Nvidia Nemotron Nano 12B V2 VL", + "name": "Nemotron Nano 12B v2 VL", "api": "anthropic-messages", "baseUrl": "https://ai-gateway.vercel.sh", "provider": "vercel-ai-gateway", @@ -89037,7 +89692,7 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.125, + "cacheRead": 0.13, "cacheWrite": 0 }, "contextWindow": 272000, @@ -89157,7 +89812,7 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.125, + "cacheRead": 0.13, "cacheWrite": 0 }, "contextWindow": 272000, @@ -89217,7 +89872,7 @@ "cost": { "input": 0.25, "output": 2, - "cacheRead": 0.024999999999999998, + "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 272000, @@ -89245,7 +89900,7 @@ "cost": { "input": 1.25, "output": 10, - "cacheRead": 0.125, + "cacheRead": 0.13, "cacheWrite": 0 }, "contextWindow": 128000, @@ -89649,10 +90304,10 @@ "image" ], "cost": { - "input": 1, - "output": 6, - "cacheRead": 0.09999999999999999, - "cacheWrite": 1.25 + "input": 0.19999999999999998, + "output": 1.2, + "cacheRead": 0.02, + "cacheWrite": 0.25 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -89711,10 +90366,10 @@ "image" ], "cost": { - "input": 2.5, - "output": 15, - "cacheRead": 0.25, - "cacheWrite": 3.125 + "input": 2, + "output": 12, + "cacheRead": 0.19999999999999998, + "cacheWrite": 2.5 }, "contextWindow": 1050000, "maxTokens": 128000, @@ -90299,6 +90954,37 @@ }, "supportsComputerUse": false }, + "thinkingmachines/inkling-small": { + "id": "thinkingmachines/inkling-small", + "name": "Inkling Small", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.5, + "output": 1.2, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 1000000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false + }, "vercel/v0-1.0-md": { "id": "vercel/v0-1.0-md", "name": "v0-1.0-md", @@ -91288,8 +91974,8 @@ "text" ], "cost": { - "input": 0.95, - "output": 3.15, + "input": 1, + "output": 3.1999999999999997, "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, @@ -91349,8 +92035,8 @@ "image" ], "cost": { - "input": 1.3, - "output": 4.300000000000001, + "input": 1.4, + "output": 4.4, "cacheRead": 0.26, "cacheWrite": 0 }, @@ -91379,12 +92065,12 @@ "text" ], "cost": { - "input": 1.4, - "output": 4.4, - "cacheRead": 0.26, + "input": 1.1, + "output": 3.851, + "cacheRead": 0.275, "cacheWrite": 0 }, - "contextWindow": 1040000, + "contextWindow": 1000000, "maxTokens": 128000, "thinking": { "mode": "budget", @@ -92621,8 +93307,6 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -92653,8 +93337,6 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -92684,6 +93366,16 @@ }, "contextWindow": 2000000, "maxTokens": 2000000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, + "omitReasoningEffort": false, + "supportsReasoningEffort": true + }, "thinking": { "mode": "effort", "efforts": [ @@ -92696,18 +93388,6 @@ "effortMap": { "minimal": "low" } - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "supportsImageDetailOriginal": false, - "omitReasoningEffort": false, - "supportsReasoningEffort": true } }, "grok-4.3": { @@ -92729,6 +93409,16 @@ }, "contextWindow": 1000000, "maxTokens": 1000000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, + "omitReasoningEffort": false, + "supportsReasoningEffort": true + }, "thinking": { "mode": "effort", "efforts": [ @@ -92741,18 +93431,6 @@ "effortMap": { "minimal": "low" } - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "supportsImageDetailOriginal": false, - "omitReasoningEffort": false, - "supportsReasoningEffort": true } }, "grok-4.5": { @@ -92774,6 +93452,16 @@ }, "contextWindow": 500000, "maxTokens": 500000, + "compat": { + "reasoningEffortMap": { + "minimal": "low" + }, + "includeEncryptedReasoning": false, + "filterReasoningHistory": true, + "supportsImageDetailOriginal": false, + "omitReasoningEffort": false, + "supportsReasoningEffort": true + }, "thinking": { "mode": "effort", "efforts": [ @@ -92786,18 +93474,6 @@ "effortMap": { "minimal": "low" } - }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, - "compat": { - "reasoningEffortMap": { - "minimal": "low" - }, - "includeEncryptedReasoning": false, - "filterReasoningHistory": true, - "supportsImageDetailOriginal": false, - "omitReasoningEffort": false, - "supportsReasoningEffort": true } }, "grok-build": { @@ -92819,8 +93495,6 @@ }, "contextWindow": 512000, "maxTokens": 512000, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -92851,8 +93525,6 @@ }, "contextWindow": 256000, "maxTokens": 256000, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -92882,8 +93554,6 @@ }, "contextWindow": 200000, "maxTokens": 200000, - "supportsComputerUse": false, - "supportsComputerUseConfig": false, "compat": { "reasoningEffortMap": { "minimal": "low" @@ -94969,7 +95639,7 @@ }, "deepseek/deepseek-v4-flash-free": { "id": "deepseek/deepseek-v4-flash-free", - "name": "DeepSeek V4 Flash (Free)", + "name": "DeepSeek V4 Flash 0731 (Free)", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -94992,8 +95662,7 @@ "max" ] }, - "supportsComputerUse": false, - "supportsComputerUseConfig": false + "supportsComputerUse": false }, "deepseek/deepseek-v4-pro": { "id": "deepseek/deepseek-v4-pro", @@ -95486,7 +96155,7 @@ }, "google/gemma-3-12b-it": { "id": "google/gemma-3-12b-it", - "name": "Gemma 3 12B", + "name": "Gemma 3 12B IT", "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", @@ -95982,6 +96651,37 @@ "maxTokens": 32768, "supportsComputerUse": false }, + "meta/muse-spark-1.1": { + "id": "meta/muse-spark-1.1", + "name": "Muse Spark 1.1", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.25, + "output": 4.25, + "cacheRead": 0.125, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "supportsComputerUse": false + }, "minimax/minimax-m2": { "id": "minimax/minimax-m2", "name": "MiniMax M2", @@ -97874,6 +98574,36 @@ ] } }, + "qwen/qwen3.7-flash": { + "id": "qwen/qwen3.7-flash", + "name": "Qwen3.7-Flash", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.03, + "output": 0.13, + "cacheRead": 0.003, + "cacheWrite": 0.038 + }, + "contextWindow": 1000000, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + }, + "supportsComputerUse": false + }, "qwen/qwen3.7-max": { "id": "qwen/qwen3.7-max", "name": "Qwen3.7 Max", @@ -99696,6 +100426,37 @@ ] } }, + "glm-5.2-highspeed[1m]": { + "id": "glm-5.2-highspeed[1m]", + "name": "GLM-5.2 Highspeed", + "api": "openai-completions", + "provider": "zhipu-coding-plan", + "baseUrl": "https://open.bigmodel.cn/api/coding/paas/v4", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 131072, + "compat": { + "thinkingFormat": "zai", + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false + }, + "thinking": { + "mode": "effort", + "efforts": [ + "high", + "max" + ] + } + }, "glm-5v-turbo": { "id": "glm-5v-turbo", "name": "GLM-5V-Turbo", diff --git a/packages/catalog/src/provider-models/descriptor-types.ts b/packages/catalog/src/provider-models/descriptor-types.ts index 042213319..fc769a678 100644 --- a/packages/catalog/src/provider-models/descriptor-types.ts +++ b/packages/catalog/src/provider-models/descriptor-types.ts @@ -2,7 +2,13 @@ import type { ModelManagerOptions } from "../model-manager"; import type { Api, FetchImpl } from "../types"; /** Config passed to a provider's runtime model-manager factory. */ -export type ModelManagerConfig = { apiKey?: string; baseUrl?: string; fetch?: FetchImpl }; +export type ModelManagerConfig = { + apiKey?: string; + baseUrl?: string; + fetch?: FetchImpl; + /** The supplied fetch already applies provider-specific authentication. */ + authenticated?: boolean; +}; /** Catalog discovery configuration for providers that support endpoint-based model listing. */ export interface CatalogDiscoveryConfig { diff --git a/packages/catalog/src/provider-models/descriptors.ts b/packages/catalog/src/provider-models/descriptors.ts index f9385fc93..4478458bd 100644 --- a/packages/catalog/src/provider-models/descriptors.ts +++ b/packages/catalog/src/provider-models/descriptors.ts @@ -14,6 +14,7 @@ import { alibabaTokenPlanModelManagerOptions, anthropicModelManagerOptions, basetenModelManagerOptions, + bedrockMantleModelManagerOptions, cerebrasModelManagerOptions, cloudflareAiGatewayModelManagerOptions, coreWeaveModelManagerOptions, @@ -102,6 +103,9 @@ export const CATALOG_PROVIDERS = [ { id: "bedrock-mantle", defaultModel: "openai.gpt-5.6-terra", + envVars: ["AWS_BEARER_TOKEN_BEDROCK"], + createModelManagerOptions: (config: ModelManagerConfig) => bedrockMantleModelManagerOptions(config), + dynamicModelsAuthoritative: true, }, { id: "anthropic", diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index ecce9377c..4f7e7d491 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -30,6 +30,7 @@ import { } from "../wire/github-copilot"; import { createBundledReferenceMap, createReferenceResolver, toModelSpec } from "./bundled-references"; import { getDefaultModelDiscoveryBaseUrl, resolveModelCacheProviderId } from "./cache-provider-id"; +import type { ModelManagerConfig } from "./descriptor-types"; const MODELS_DEV_URL = "https://models.dev/api.json"; @@ -3392,7 +3393,7 @@ export const BEDROCK_MANTLE_STATIC_MODELS: readonly ModelSpec<"openai-responses" baseUrl: BEDROCK_MANTLE_BASE_URL, reasoning: true, input: ["text", "image"], - cost: { input: 1.1, output: 6.6, cacheRead: 0.11, cacheWrite: 1.38 }, + cost: { input: 0.22, output: 1.32, cacheRead: 0.022, cacheWrite: 0.275 }, contextWindow: 272_000, maxTokens: 128_000, thinking: BEDROCK_MANTLE_GPT_5_6_THINKING, @@ -3418,13 +3419,43 @@ export const BEDROCK_MANTLE_STATIC_MODELS: readonly ModelSpec<"openai-responses" baseUrl: BEDROCK_MANTLE_BASE_URL, reasoning: true, input: ["text", "image"], - cost: { input: 2.75, output: 16.5, cacheRead: 0.28, cacheWrite: 3.44 }, + cost: { input: 2.2, output: 13.2, cacheRead: 0.22, cacheWrite: 2.75 }, contextWindow: 272_000, maxTokens: 128_000, thinking: BEDROCK_MANTLE_GPT_5_6_THINKING, }, ]; +const BEDROCK_MANTLE_MODEL_BY_ID: Partial>> = Object.fromEntries( + BEDROCK_MANTLE_STATIC_MODELS.map(model => [model.id, model]), +); + +export function bedrockMantleModelManagerOptions( + config: ModelManagerConfig = {}, +): ModelManagerOptions<"openai-responses"> { + const inferenceBaseUrl = config.baseUrl ?? BEDROCK_MANTLE_BASE_URL; + const discoveryBaseUrl = inferenceBaseUrl.replace(/\/openai\/v1\/?$/, "/v1"); + return { + providerId: "bedrock-mantle", + staticModels: BEDROCK_MANTLE_STATIC_MODELS, + ...(config.authenticated && { + fetchDynamicModels: () => + fetchOpenAICompatibleModels({ + api: "openai-responses", + provider: "bedrock-mantle", + baseUrl: discoveryBaseUrl, + fetch: config.fetch, + mapModel: (entry, defaults) => + mapWithBundledReference( + entry, + { ...defaults, baseUrl: BEDROCK_MANTLE_BASE_URL }, + BEDROCK_MANTLE_MODEL_BY_ID[defaults.id], + ), + }), + }), + }; +} + export interface MetaModelManagerConfig { apiKey?: string; baseUrl?: string; diff --git a/packages/catalog/test/amazon-bedrock-openai.test.ts b/packages/catalog/test/amazon-bedrock-openai.test.ts index 3c6256cc3..14391c0d9 100644 --- a/packages/catalog/test/amazon-bedrock-openai.test.ts +++ b/packages/catalog/test/amazon-bedrock-openai.test.ts @@ -1,7 +1,10 @@ import { describe, expect, test } from "bun:test"; -import { DEFAULT_MODEL_PER_PROVIDER } from "@oh-my-pi/pi-catalog/provider-models/descriptors"; -import { BEDROCK_MANTLE_STATIC_MODELS } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; -import type { ModelSpec } from "@oh-my-pi/pi-catalog/types"; +import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "@oh-my-pi/pi-catalog/provider-models/descriptors"; +import { + BEDROCK_MANTLE_STATIC_MODELS, + bedrockMantleModelManagerOptions, +} from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types"; import { dropBedrockMantleOpenAIModels } from "../scripts/generated-policies"; const MANTLE_MODEL_IDS = [ @@ -38,6 +41,56 @@ describe("Amazon Bedrock OpenAI routing", () => { expect(DEFAULT_MODEL_PER_PROVIDER["bedrock-mantle"]).toBe("openai.gpt-5.6-terra"); }); + test("uses current Luna and Terra pricing", () => { + const byId = Object.fromEntries(BEDROCK_MANTLE_STATIC_MODELS.map(model => [model.id, model])); + expect(byId["openai.gpt-5.6-luna"]?.cost).toEqual({ + input: 0.22, + output: 1.32, + cacheRead: 0.022, + cacheWrite: 0.275, + }); + expect(byId["openai.gpt-5.6-terra"]?.cost).toEqual({ + input: 2.2, + output: 13.2, + cacheRead: 0.22, + cacheWrite: 2.75, + }); + }); + + test("builds bearer-authenticated Mantle runtime discovery", async () => { + let requestedUrl = ""; + const fetchImpl: FetchImpl = Object.assign( + async (input: string | URL | Request) => { + requestedUrl = String(input); + return Response.json({ + data: [ + { id: "openai.gpt-5.6-luna", name: "GPT-5.6 Luna" }, + { id: "openai.gpt-5.7-preview", name: "GPT-5.7 Preview" }, + ], + }); + }, + { preconnect: fetch.preconnect }, + ); + const managerOptions = bedrockMantleModelManagerOptions({ + authenticated: true, + baseUrl: "https://bedrock-mantle.eu-west-2.api.aws/openai/v1", + fetch: fetchImpl, + }); + + const models = await managerOptions.fetchDynamicModels?.(); + + expect(requestedUrl).toBe("https://bedrock-mantle.eu-west-2.api.aws/v1/models"); + expect(models).toHaveLength(2); + expect(models?.[0]).toMatchObject({ + id: "openai.gpt-5.6-luna", + baseUrl: "https://bedrock-mantle.{region}.api.aws/openai/v1", + cost: { input: 0.22, output: 1.32, cacheRead: 0.022, cacheWrite: 0.275 }, + }); + const descriptor = PROVIDER_DESCRIPTORS.find(descriptor => descriptor.providerId === "bedrock-mantle"); + expect(descriptor).toMatchObject({ dynamicModelsAuthoritative: true }); + expect(descriptor?.catalogDiscovery).toBeUndefined(); + }); + test("drops only the unusable Converse rows for Mantle models", () => { const input = [ ...MANTLE_MODEL_IDS.map(id => bedrockModel("amazon-bedrock", id)), diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index f49dcd3c8..b006ec6cb 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -67,6 +67,7 @@ import type { ApiKeyResolver, FetchImpl } from "@oh-my-pi/pi-ai"; import { registerOAuthProvider, unregisterOAuthProviders } from "@oh-my-pi/pi-ai/oauth"; import type { OAuthCredentials, OAuthLoginCallbacks } from "@oh-my-pi/pi-ai/oauth/types"; import { setCodexAttestationProvider } from "@oh-my-pi/pi-ai/providers/openai-codex-responses"; +import { getProviderDefinition } from "@oh-my-pi/pi-ai/registry"; import { getBundledModelReferenceIndex, inheritReferenceThinking, @@ -1997,14 +1998,15 @@ export class ModelRegistry { this.#providerOverrides.has(descriptor.providerId) || this.#keylessProviders.has(descriptor.providerId)); if (isAuthenticated(apiKey) || descriptor.allowUnauthenticated || hasExplicitVllmConfig) { - const discoveryBaseUrl = this.#descriptorBaseUrl(descriptor.providerId); - options.push( - descriptor.createModelManagerOptions({ - apiKey: isDiscoveryBearerApiKey(apiKey) ? apiKey : undefined, - baseUrl: discoveryBaseUrl, - fetch: this.#fetch, - }), - ); + const discoveryConfig = { + apiKey: isDiscoveryBearerApiKey(apiKey) ? apiKey : undefined, + baseUrl: this.#descriptorBaseUrl(descriptor.providerId), + fetch: this.#fetch, + }; + const preparedConfig = + getProviderDefinition(descriptor.providerId)?.prepareModelDiscovery?.(discoveryConfig) ?? + discoveryConfig; + options.push(descriptor.createModelManagerOptions(preparedConfig)); } }