fix(catalog): read Synthetic's advertised model capabilities

The Synthetic discovery mapper looked for `supports_reasoning`,
`supports_vision` and `max_tokens`. Synthetic's /openai/v1/models sends
none of those: capabilities arrive in `supported_features`, the accepted
`reasoning_effort` vocabulary in `reasoning_parameters.efforts`,
modalities in `input_modalities`, the output cap in `max_output_length`,
and per-token prices in `pricing` as `$`-prefixed strings.

Every capability therefore fell back to the bundled reference, so any
route without one — the `syn:*` router aliases, newly added routes such
as `hf:moonshotai/Kimi-K3` — resolved to `reasoning: false`, text-only,
zero cost, `maxTokens: 8192`. `reasoning: false` makes
`getSupportedEfforts` return `[]`, which hides the thinking selector and
silently drops a `provider/model:high` suffix, so those models never
receive `reasoning_effort` at all; the 8k cap is low enough that verbose
models stop on `length` every turn, which reads as an incomplete
response and triggers recovery compaction at ~9% context.

Read the fields Synthetic actually sends and derive the effort ladder
from the per-model wire vocabulary. `none` is the router's thinking-off
state rather than a user tier, so it backs `minimal` through the wire
map, mirroring the existing Fireworks `minimal -> none` mapping.

Verified live against api.synthetic.new: `reasoning_effort` is accepted
on every route (`none` yields no reasoning, `high` < `max`), and the
`syn:*:vision` aliases do accept image input.
This commit is contained in:
Gareth-Rouse
2026-07-31 10:05:52 +01:00
parent 4df68d6043
commit b2880217e9
3 changed files with 252 additions and 4 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed Synthetic models losing their thinking selector, vision input, output cap and pricing: the discovery mapper read `supports_reasoning`, `supports_vision` and `max_tokens`, none of which Synthetic sends. It advertises `supported_features`, `reasoning_parameters.efforts`, `input_modalities`, `max_output_length` and `$`-prefixed `pricing`, so any route without a bundled reference (the `syn:*` router aliases, newly added routes such as `hf:moonshotai/Kimi-K3`) resolved to `reasoning: false` — hiding the effort dial and dropping `reasoning_effort` from every request — plus text-only input, zero cost, and an 8k output cap low enough to end turns on `length` and trigger recovery compaction. Effort ladders now come from the per-model wire vocabulary, with the router's `none` tier mapped onto `minimal`.
## [17.2.1] - 2026-07-30
### Fixed
@@ -3101,6 +3101,85 @@ export interface SyntheticModelManagerConfig {
fetch?: FetchImpl;
}
/**
* Synthetic's `/openai/v1/models` entry shape (verified live against
* api.synthetic.new). It shares no capability field names with the generic
* OpenAI-compatible conventions: capabilities arrive in `supported_features`,
* modalities in `input_modalities`, the output cap in `max_output_length`, the
* accepted `reasoning_effort` vocabulary in `reasoning_parameters.efforts`,
* and per-token prices in `pricing` as `$`-prefixed decimal strings.
*/
interface SyntheticModelRecord extends OpenAICompatibleModelRecord {
supported_features?: unknown;
input_modalities?: unknown;
max_output_length?: unknown;
reasoning_parameters?: unknown;
pricing?: unknown;
}
/** Synthetic's thinking-off wire tier — a router state, not a user effort. */
const SYNTHETIC_WIRE_EFFORT_NONE = "none";
/** Output cap for routes that advertise no `max_output_length`. */
const SYNTHETIC_FALLBACK_MAX_TOKENS = 8192;
function toSyntheticStringList(value: unknown): readonly string[] {
return Array.isArray(value) ? value.filter((item): item is string => typeof item === "string") : [];
}
/**
* Translate Synthetic's per-model `reasoning_effort` vocabulary into an effort
* ladder. Every advertised value that names an OMP tier maps verbatim; `none`
* is the thinking-off state rather than a tier of its own, so it backs the
* `minimal` selector through the wire map (same shape as the Fireworks
* `minimal → none` map) and gives these routes a real no-thinking tier.
* Routes with no recognizable tier fall through to identity inference.
*/
function resolveSyntheticThinking(wireEfforts: readonly string[]): ThinkingConfig | undefined {
const efforts = THINKING_EFFORTS.filter(effort => wireEfforts.includes(effort));
if (efforts.length === 0) {
return undefined;
}
if (!wireEfforts.includes(SYNTHETIC_WIRE_EFFORT_NONE) || efforts.includes(Effort.Minimal)) {
return { mode: "effort", efforts };
}
return {
mode: "effort",
efforts: [Effort.Minimal, ...efforts],
effortMap: { [Effort.Minimal]: SYNTHETIC_WIRE_EFFORT_NONE },
};
}
/** Synthetic quotes per-token USD as `"$0.000001"`; catalog cost is per-million. */
function toSyntheticCostPerMillion(value: unknown): number | undefined {
const parsed = toNumber(typeof value === "string" ? value.trim().replace(/^\$/, "") : value);
if (parsed === undefined || parsed < 0) {
return undefined;
}
// Scaling a per-token decimal by 1e6 drifts (4.5e-7 → 0.44999999999999996), so
// settle on a millionth of a dollar per million tokens — finer than any real tier.
return Math.round(parsed * 1e12) / 1e6;
}
function resolveSyntheticCost(
pricing: unknown,
fallback: ModelSpec<"openai-completions">["cost"],
): ModelSpec<"openai-completions">["cost"] {
if (!isRecord(pricing)) {
return fallback;
}
const input = toSyntheticCostPerMillion(pricing.prompt);
const output = toSyntheticCostPerMillion(pricing.completion);
if (input === undefined || output === undefined) {
return fallback;
}
return {
input,
output,
cacheRead: toSyntheticCostPerMillion(pricing.input_cache_reads) ?? fallback.cacheRead,
cacheWrite: toSyntheticCostPerMillion(pricing.input_cache_writes) ?? fallback.cacheWrite,
};
}
export function syntheticModelManagerOptions(
config?: SyntheticModelManagerConfig,
): ModelManagerOptions<"openai-completions"> {
@@ -3124,18 +3203,46 @@ export function syntheticModelManagerOptions(
defaults: ModelSpec<"openai-completions">,
_context: OpenAICompatibleModelMapperContext<"openai-completions">,
): ModelSpec<"openai-completions"> => {
const record = entry as SyntheticModelRecord;
const reference = references.get(defaults.id);
const referenceSupportsImage = reference?.input.includes("image") ?? false;
const features = toSyntheticStringList(record.supported_features);
const modalities = toSyntheticStringList(record.input_modalities);
const wireEfforts = isRecord(record.reasoning_parameters)
? toSyntheticStringList(record.reasoning_parameters.efforts)
: [];
const thinking = resolveSyntheticThinking(wireEfforts);
// The router aliases (`syn:*`) and newly added routes carry no
// bundled reference, so these advertised capabilities are the only
// truth available. Without them such a model lands non-reasoning
// (which hides the thinking selector and drops `reasoning_effort`
// from every request), text-only, priced at zero, and capped at the
// 8k placeholder — a cap low enough that verbose models stop on
// `length` each turn and trip recovery compaction.
const base = reference ? { ...reference, id: defaults.id, baseUrl } : defaults;
return {
...(reference ? { ...reference, id: defaults.id, baseUrl } : defaults),
...base,
name: toModelName(entry.name, reference?.name ?? defaults.name),
reasoning: entry.supports_reasoning === true || (reference?.reasoning ?? false),
input: entry.supports_vision === true || referenceSupportsImage ? ["text", "image"] : ["text"],
reasoning:
features.includes("reasoning") ||
wireEfforts.length > 0 ||
entry.supports_reasoning === true ||
(reference?.reasoning ?? false),
...(thinking ? { thinking } : {}),
input:
modalities.includes("image") || entry.supports_vision === true || referenceSupportsImage
? ["text", "image"]
: ["text"],
...(features.length > 0 && !features.includes("tools") ? { supportsTools: false } : {}),
cost: resolveSyntheticCost(record.pricing, base.cost),
contextWindow: toPositiveNumber(
entry.context_length,
reference?.contextWindow ?? defaults.contextWindow,
),
maxTokens: toPositiveNumber(entry.max_tokens, reference?.maxTokens ?? 8192),
maxTokens: toPositiveNumber(
record.max_output_length ?? entry.max_tokens,
reference?.maxTokens ?? SYNTHETIC_FALLBACK_MAX_TOKENS,
),
};
},
fetch: config?.fetch,
@@ -0,0 +1,137 @@
import { describe, expect, test } from "bun:test";
import { Effort } from "@oh-my-pi/pi-catalog/effort";
import { syntheticModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl } from "@oh-my-pi/pi-catalog/types";
/**
* Entries mirror live `https://api.synthetic.new/openai/v1/models` payloads:
* capabilities in `supported_features`, modalities in `input_modalities`, the
* output cap in `max_output_length`, the accepted `reasoning_effort` values in
* `reasoning_parameters.efforts`, and `$`-prefixed per-token prices.
*/
function syntheticModelsFetch(): { calls: string[]; fetch: FetchImpl } {
const calls: string[] = [];
const fetch: FetchImpl = async (input: string | URL | Request) => {
calls.push(String(input));
return new Response(
JSON.stringify({
data: [
{
id: "syn:large:text",
object: "model",
name: "syn:large:text",
hugging_face_id: "zai-org/GLM-5.2",
reasoning_parameters: { efforts: ["none", "high", "max"] },
input_modalities: ["text"],
context_length: 524288,
max_output_length: 65536,
supported_features: ["tools", "json_mode", "structured_outputs", "reasoning"],
pricing: {
prompt: "$0.000001",
completion: "$0.000003",
input_cache_reads: "$0.00000016",
input_cache_writes: "0",
},
},
{
id: "hf:moonshotai/Kimi-K3",
object: "model",
name: "moonshotai/Kimi-K3",
reasoning_parameters: { efforts: ["low", "high", "max"] },
input_modalities: ["text", "image"],
context_length: 524288,
max_output_length: 65536,
supported_features: ["tools", "json_mode", "structured_outputs", "reasoning"],
pricing: {
prompt: "$0.000003",
completion: "$0.000015",
input_cache_reads: "$0.00000045",
input_cache_writes: "0",
},
},
{
id: "hf:example/plain-completions",
object: "model",
name: "example/plain-completions",
input_modalities: ["text"],
context_length: 131072,
supported_features: ["json_mode"],
},
],
}),
{ status: 200, headers: { "content-type": "application/json" } },
);
};
return { calls, fetch };
}
describe("Synthetic provider discovery", () => {
test("reads capabilities from Synthetic's own schema instead of the bundled reference", async () => {
const { calls, fetch } = syntheticModelsFetch();
const models = await syntheticModelManagerOptions({ apiKey: "syn-test-key", fetch }).fetchDynamicModels?.();
expect(calls).toEqual(["https://api.synthetic.new/openai/v1/models"]);
// `syn:*` router aliases ship a bundled reference baked from the era when
// this mapper read field names Synthetic never sends: `reasoning: false`,
// no thinking, `maxTokens: 8192`, zero cost. The advertised metadata wins.
const large = models?.find(model => model.id === "syn:large:text");
expect(large).toMatchObject({
provider: "synthetic",
api: "openai-completions",
reasoning: true,
input: ["text"],
contextWindow: 524288,
maxTokens: 65536,
cost: { input: 1, output: 3, cacheRead: 0.16, cacheWrite: 0 },
});
// `none` is the router's thinking-off state, so it backs `minimal`
// through the wire map rather than becoming a tier of its own.
expect(large?.thinking).toEqual({
mode: "effort",
efforts: [Effort.Minimal, Effort.High, Effort.Max],
effortMap: { minimal: "none" },
});
expect(large?.supportsTools).toBeUndefined();
});
test("derives reasoning, vision, and output cap for routes with no bundled reference", async () => {
const { fetch } = syntheticModelsFetch();
const models = await syntheticModelManagerOptions({ apiKey: "syn-test-key", fetch }).fetchDynamicModels?.();
const kimi = models?.find(model => model.id === "hf:moonshotai/Kimi-K3");
expect(kimi).toMatchObject({
provider: "synthetic",
reasoning: true,
input: ["text", "image"],
contextWindow: 524288,
maxTokens: 65536,
cost: { input: 3, output: 15, cacheRead: 0.45, cacheWrite: 0 },
});
// No `none` tier on this route: the ladder is the advertised one verbatim.
expect(kimi?.thinking).toEqual({ mode: "effort", efforts: [Effort.Low, Effort.High, Effort.Max] });
});
test("keeps non-reasoning routes non-reasoning and marks missing tool support", async () => {
const { fetch } = syntheticModelsFetch();
const models = await syntheticModelManagerOptions({ apiKey: "syn-test-key", fetch }).fetchDynamicModels?.();
const plain = models?.find(model => model.id === "hf:example/plain-completions");
expect(plain).toMatchObject({
provider: "synthetic",
reasoning: false,
input: ["text"],
contextWindow: 131072,
supportsTools: false,
// No `max_output_length` and no bundled reference: placeholder cap.
maxTokens: 8192,
// No `pricing` block: the reference/default cost survives.
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
});
expect(plain?.thinking).toBeUndefined();
});
test("serves no dynamic models without an API key", () => {
expect(syntheticModelManagerOptions().fetchDynamicModels).toBeUndefined();
});
});