diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 6104a2571..b4dce027e 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Synthetic models losing their thinking selector, vision input, output cap and pricing: the discovery mapper read `supports_reasoning`, `supports_vision` and `max_tokens`, none of which Synthetic sends. It advertises `supported_features`, `reasoning_parameters.efforts`, `input_modalities`, `max_output_length` and `$`-prefixed `pricing`, so any route without a bundled reference (the `syn:*` router aliases, newly added routes such as `hf:moonshotai/Kimi-K3`) resolved to `reasoning: false` — hiding the effort dial and dropping `reasoning_effort` from every request — plus text-only input, zero cost, and an 8k output cap low enough to end turns on `length` and trigger recovery compaction. Effort ladders now come from the per-model wire vocabulary, with the router's `none` tier mapped onto `minimal`. + ## [17.2.1] - 2026-07-30 ### Fixed diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index 8926b200c..2aad905ca 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3101,6 +3101,85 @@ export interface SyntheticModelManagerConfig { fetch?: FetchImpl; } +/** + * Synthetic's `/openai/v1/models` entry shape (verified live against + * api.synthetic.new). It shares no capability field names with the generic + * OpenAI-compatible conventions: capabilities arrive in `supported_features`, + * modalities in `input_modalities`, the output cap in `max_output_length`, the + * accepted `reasoning_effort` vocabulary in `reasoning_parameters.efforts`, + * and per-token prices in `pricing` as `$`-prefixed decimal strings. + */ +interface SyntheticModelRecord extends OpenAICompatibleModelRecord { + supported_features?: unknown; + input_modalities?: unknown; + max_output_length?: unknown; + reasoning_parameters?: unknown; + pricing?: unknown; +} + +/** Synthetic's thinking-off wire tier — a router state, not a user effort. */ +const SYNTHETIC_WIRE_EFFORT_NONE = "none"; +/** Output cap for routes that advertise no `max_output_length`. */ +const SYNTHETIC_FALLBACK_MAX_TOKENS = 8192; + +function toSyntheticStringList(value: unknown): readonly string[] { + return Array.isArray(value) ? value.filter((item): item is string => typeof item === "string") : []; +} + +/** + * Translate Synthetic's per-model `reasoning_effort` vocabulary into an effort + * ladder. Every advertised value that names an OMP tier maps verbatim; `none` + * is the thinking-off state rather than a tier of its own, so it backs the + * `minimal` selector through the wire map (same shape as the Fireworks + * `minimal → none` map) and gives these routes a real no-thinking tier. + * Routes with no recognizable tier fall through to identity inference. + */ +function resolveSyntheticThinking(wireEfforts: readonly string[]): ThinkingConfig | undefined { + const efforts = THINKING_EFFORTS.filter(effort => wireEfforts.includes(effort)); + if (efforts.length === 0) { + return undefined; + } + if (!wireEfforts.includes(SYNTHETIC_WIRE_EFFORT_NONE) || efforts.includes(Effort.Minimal)) { + return { mode: "effort", efforts }; + } + return { + mode: "effort", + efforts: [Effort.Minimal, ...efforts], + effortMap: { [Effort.Minimal]: SYNTHETIC_WIRE_EFFORT_NONE }, + }; +} + +/** Synthetic quotes per-token USD as `"$0.000001"`; catalog cost is per-million. */ +function toSyntheticCostPerMillion(value: unknown): number | undefined { + const parsed = toNumber(typeof value === "string" ? value.trim().replace(/^\$/, "") : value); + if (parsed === undefined || parsed < 0) { + return undefined; + } + // Scaling a per-token decimal by 1e6 drifts (4.5e-7 → 0.44999999999999996), so + // settle on a millionth of a dollar per million tokens — finer than any real tier. + return Math.round(parsed * 1e12) / 1e6; +} + +function resolveSyntheticCost( + pricing: unknown, + fallback: ModelSpec<"openai-completions">["cost"], +): ModelSpec<"openai-completions">["cost"] { + if (!isRecord(pricing)) { + return fallback; + } + const input = toSyntheticCostPerMillion(pricing.prompt); + const output = toSyntheticCostPerMillion(pricing.completion); + if (input === undefined || output === undefined) { + return fallback; + } + return { + input, + output, + cacheRead: toSyntheticCostPerMillion(pricing.input_cache_reads) ?? fallback.cacheRead, + cacheWrite: toSyntheticCostPerMillion(pricing.input_cache_writes) ?? fallback.cacheWrite, + }; +} + export function syntheticModelManagerOptions( config?: SyntheticModelManagerConfig, ): ModelManagerOptions<"openai-completions"> { @@ -3124,18 +3203,46 @@ export function syntheticModelManagerOptions( defaults: ModelSpec<"openai-completions">, _context: OpenAICompatibleModelMapperContext<"openai-completions">, ): ModelSpec<"openai-completions"> => { + const record = entry as SyntheticModelRecord; const reference = references.get(defaults.id); const referenceSupportsImage = reference?.input.includes("image") ?? false; + const features = toSyntheticStringList(record.supported_features); + const modalities = toSyntheticStringList(record.input_modalities); + const wireEfforts = isRecord(record.reasoning_parameters) + ? toSyntheticStringList(record.reasoning_parameters.efforts) + : []; + const thinking = resolveSyntheticThinking(wireEfforts); + // The router aliases (`syn:*`) and newly added routes carry no + // bundled reference, so these advertised capabilities are the only + // truth available. Without them such a model lands non-reasoning + // (which hides the thinking selector and drops `reasoning_effort` + // from every request), text-only, priced at zero, and capped at the + // 8k placeholder — a cap low enough that verbose models stop on + // `length` each turn and trip recovery compaction. + const base = reference ? { ...reference, id: defaults.id, baseUrl } : defaults; return { - ...(reference ? { ...reference, id: defaults.id, baseUrl } : defaults), + ...base, name: toModelName(entry.name, reference?.name ?? defaults.name), - reasoning: entry.supports_reasoning === true || (reference?.reasoning ?? false), - input: entry.supports_vision === true || referenceSupportsImage ? ["text", "image"] : ["text"], + reasoning: + features.includes("reasoning") || + wireEfforts.length > 0 || + entry.supports_reasoning === true || + (reference?.reasoning ?? false), + ...(thinking ? { thinking } : {}), + input: + modalities.includes("image") || entry.supports_vision === true || referenceSupportsImage + ? ["text", "image"] + : ["text"], + ...(features.length > 0 && !features.includes("tools") ? { supportsTools: false } : {}), + cost: resolveSyntheticCost(record.pricing, base.cost), contextWindow: toPositiveNumber( entry.context_length, reference?.contextWindow ?? defaults.contextWindow, ), - maxTokens: toPositiveNumber(entry.max_tokens, reference?.maxTokens ?? 8192), + maxTokens: toPositiveNumber( + record.max_output_length ?? entry.max_tokens, + reference?.maxTokens ?? SYNTHETIC_FALLBACK_MAX_TOKENS, + ), }; }, fetch: config?.fetch, diff --git a/packages/catalog/test/synthetic-provider.test.ts b/packages/catalog/test/synthetic-provider.test.ts new file mode 100644 index 000000000..ae46bd5bc --- /dev/null +++ b/packages/catalog/test/synthetic-provider.test.ts @@ -0,0 +1,137 @@ +import { describe, expect, test } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { syntheticModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat"; +import type { FetchImpl } from "@oh-my-pi/pi-catalog/types"; + +/** + * Entries mirror live `https://api.synthetic.new/openai/v1/models` payloads: + * capabilities in `supported_features`, modalities in `input_modalities`, the + * output cap in `max_output_length`, the accepted `reasoning_effort` values in + * `reasoning_parameters.efforts`, and `$`-prefixed per-token prices. + */ +function syntheticModelsFetch(): { calls: string[]; fetch: FetchImpl } { + const calls: string[] = []; + const fetch: FetchImpl = async (input: string | URL | Request) => { + calls.push(String(input)); + return new Response( + JSON.stringify({ + data: [ + { + id: "syn:large:text", + object: "model", + name: "syn:large:text", + hugging_face_id: "zai-org/GLM-5.2", + reasoning_parameters: { efforts: ["none", "high", "max"] }, + input_modalities: ["text"], + context_length: 524288, + max_output_length: 65536, + supported_features: ["tools", "json_mode", "structured_outputs", "reasoning"], + pricing: { + prompt: "$0.000001", + completion: "$0.000003", + input_cache_reads: "$0.00000016", + input_cache_writes: "0", + }, + }, + { + id: "hf:moonshotai/Kimi-K3", + object: "model", + name: "moonshotai/Kimi-K3", + reasoning_parameters: { efforts: ["low", "high", "max"] }, + input_modalities: ["text", "image"], + context_length: 524288, + max_output_length: 65536, + supported_features: ["tools", "json_mode", "structured_outputs", "reasoning"], + pricing: { + prompt: "$0.000003", + completion: "$0.000015", + input_cache_reads: "$0.00000045", + input_cache_writes: "0", + }, + }, + { + id: "hf:example/plain-completions", + object: "model", + name: "example/plain-completions", + input_modalities: ["text"], + context_length: 131072, + supported_features: ["json_mode"], + }, + ], + }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + }; + return { calls, fetch }; +} + +describe("Synthetic provider discovery", () => { + test("reads capabilities from Synthetic's own schema instead of the bundled reference", async () => { + const { calls, fetch } = syntheticModelsFetch(); + const models = await syntheticModelManagerOptions({ apiKey: "syn-test-key", fetch }).fetchDynamicModels?.(); + + expect(calls).toEqual(["https://api.synthetic.new/openai/v1/models"]); + + // `syn:*` router aliases ship a bundled reference baked from the era when + // this mapper read field names Synthetic never sends: `reasoning: false`, + // no thinking, `maxTokens: 8192`, zero cost. The advertised metadata wins. + const large = models?.find(model => model.id === "syn:large:text"); + expect(large).toMatchObject({ + provider: "synthetic", + api: "openai-completions", + reasoning: true, + input: ["text"], + contextWindow: 524288, + maxTokens: 65536, + cost: { input: 1, output: 3, cacheRead: 0.16, cacheWrite: 0 }, + }); + // `none` is the router's thinking-off state, so it backs `minimal` + // through the wire map rather than becoming a tier of its own. + expect(large?.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Minimal, Effort.High, Effort.Max], + effortMap: { minimal: "none" }, + }); + expect(large?.supportsTools).toBeUndefined(); + }); + + test("derives reasoning, vision, and output cap for routes with no bundled reference", async () => { + const { fetch } = syntheticModelsFetch(); + const models = await syntheticModelManagerOptions({ apiKey: "syn-test-key", fetch }).fetchDynamicModels?.(); + + const kimi = models?.find(model => model.id === "hf:moonshotai/Kimi-K3"); + expect(kimi).toMatchObject({ + provider: "synthetic", + reasoning: true, + input: ["text", "image"], + contextWindow: 524288, + maxTokens: 65536, + cost: { input: 3, output: 15, cacheRead: 0.45, cacheWrite: 0 }, + }); + // No `none` tier on this route: the ladder is the advertised one verbatim. + expect(kimi?.thinking).toEqual({ mode: "effort", efforts: [Effort.Low, Effort.High, Effort.Max] }); + }); + + test("keeps non-reasoning routes non-reasoning and marks missing tool support", async () => { + const { fetch } = syntheticModelsFetch(); + const models = await syntheticModelManagerOptions({ apiKey: "syn-test-key", fetch }).fetchDynamicModels?.(); + + const plain = models?.find(model => model.id === "hf:example/plain-completions"); + expect(plain).toMatchObject({ + provider: "synthetic", + reasoning: false, + input: ["text"], + contextWindow: 131072, + supportsTools: false, + // No `max_output_length` and no bundled reference: placeholder cap. + maxTokens: 8192, + // No `pricing` block: the reference/default cost survives. + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + }); + expect(plain?.thinking).toBeUndefined(); + }); + + test("serves no dynamic models without an API key", () => { + expect(syntheticModelManagerOptions().fetchDynamicModels).toBeUndefined(); + }); +});