feat(catalog): switched Fireworks discovery to control-plane and added bundled models
- Updated Fireworks model discovery to use the serverless control-plane endpoint. - Added bundled Fireworks models including deepseek-v4-flash, kimi-k2.7-code, and qwen variants. - Implemented paginated control-plane discovery with filtering of eligible Fireworks models. - Updated github-copilot contextWindow from 222222 to 524288 tokens.
This commit is contained in:
@@ -1,6 +1,15 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Added
|
||||
|
||||
- Added bundled Fireworks models `deepseek-v4-flash`, `kimi-k2.7-code`, `minimax-m2.5`, `minimax-m3`, `nemotron-3-ultra-nvfp4`, `qwen3.6-plus`, and `qwen3.7-plus`
|
||||
- Changed
|
||||
|
||||
### Changed
|
||||
|
||||
- Changed the `github-copilot` model context window to `524288` tokens
|
||||
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
|
||||
|
||||
### Fixed
|
||||
|
||||
|
||||
@@ -13598,6 +13598,51 @@
|
||||
}
|
||||
},
|
||||
"fireworks": {
|
||||
"deepseek-v4-flash": {
|
||||
"id": "deepseek-v4-flash",
|
||||
"name": "DeepSeek V4 Flash",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.14,
|
||||
"output": 0.28,
|
||||
"cacheRead": 0.0028,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1048576,
|
||||
"maxTokens": 384000,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "high",
|
||||
"low": "high",
|
||||
"medium": "high",
|
||||
"high": "high",
|
||||
"xhigh": "max"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false,
|
||||
"supportsReasoningEffort": true,
|
||||
"maxTokensField": "max_tokens",
|
||||
"supportsToolChoice": false,
|
||||
"reasoningContentField": "reasoning_content",
|
||||
"requiresReasoningContentForToolCalls": true,
|
||||
"requiresAssistantContentForToolCalls": true
|
||||
}
|
||||
},
|
||||
"deepseek-v4-pro": {
|
||||
"id": "deepseek-v4-pro",
|
||||
"name": "DeepSeek V4 Pro",
|
||||
@@ -13800,6 +13845,67 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"kimi-k2.7-code": {
|
||||
"id": "kimi-k2.7-code",
|
||||
"name": "Kimi K2.7 Code",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.95,
|
||||
"output": 4,
|
||||
"cacheRead": 0.19,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
}
|
||||
},
|
||||
"minimax-m2.5": {
|
||||
"id": "minimax-m2.5",
|
||||
"name": "MiniMax M2.5",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.3,
|
||||
"output": 1.2,
|
||||
"cacheRead": 0.06,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 196608,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"minimax-m2.7": {
|
||||
"id": "minimax-m2.7",
|
||||
"name": "MiniMax M2.7",
|
||||
@@ -13827,6 +13933,138 @@
|
||||
],
|
||||
"requiresEffort": true
|
||||
}
|
||||
},
|
||||
"minimax-m3": {
|
||||
"id": "minimax-m3",
|
||||
"name": "MiniMax M3 (3x usage)",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.1,
|
||||
"output": 0.4,
|
||||
"cacheRead": 0.02,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 512000,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
}
|
||||
},
|
||||
"nemotron-3-ultra-nvfp4": {
|
||||
"id": "nemotron-3-ultra-nvfp4",
|
||||
"name": "NVIDIA Nemotron 3 Ultra NVFP4",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 8888,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
}
|
||||
},
|
||||
"qwen3.6-plus": {
|
||||
"id": "qwen3.6-plus",
|
||||
"name": "Qwen3.6 Plus",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0,
|
||||
"output": 0,
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
},
|
||||
"compat": {
|
||||
"supportsDeveloperRole": false
|
||||
}
|
||||
},
|
||||
"qwen3.7-plus": {
|
||||
"id": "qwen3.7-plus",
|
||||
"name": "Qwen3.7 Plus",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.4,
|
||||
"output": 1.6,
|
||||
"cacheRead": 0.04,
|
||||
"cacheWrite": 0.5
|
||||
},
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": 65536,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"github-copilot": {
|
||||
@@ -64124,7 +64362,7 @@
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 222222,
|
||||
"contextWindow": 524288,
|
||||
"maxTokens": 8888,
|
||||
"compat": {
|
||||
"supportsUsageInStreaming": false
|
||||
|
||||
@@ -1128,17 +1128,155 @@ export interface FireworksModelManagerConfig {
|
||||
fetch?: FetchImpl;
|
||||
}
|
||||
|
||||
function toFireworksModelName(entry: OpenAICompatibleModelRecord, fallback: string): string {
|
||||
const name = toModelName(entry.name, "");
|
||||
if (name) return name;
|
||||
const id = typeof entry.id === "string" ? entry.id : fallback;
|
||||
const shortName = id.split("/").at(-1) ?? fallback;
|
||||
if (fallback !== id && fallback !== shortName) return fallback;
|
||||
return shortName
|
||||
.split("-")
|
||||
.filter(Boolean)
|
||||
.map(part => part.charAt(0).toUpperCase() + part.slice(1))
|
||||
.join(" ");
|
||||
const FIREWORKS_CONTROL_PLANE_ACCOUNT = "fireworks";
|
||||
const FIREWORKS_SERVERLESS_FILTER = "supports_serverless=true";
|
||||
const FIREWORKS_CONTROL_PLANE_PAGE_SIZE = 200;
|
||||
const FIREWORKS_CONTROL_PLANE_MAX_PAGES = 25;
|
||||
|
||||
/**
|
||||
* One record from the Fireworks control-plane catalog
|
||||
* (`GET /v1/accounts/{account}/models`). This is distinct from the
|
||||
* OpenAI-compatible `/v1/models` inference envelope: the control plane
|
||||
* enumerates the full serverless catalog with camelCase capability metadata,
|
||||
* including on-demand models (e.g. `kimi-k2p7-code`) that never surface in
|
||||
* `/v1/models`. Discovering here is what keeps new serverless models appearing
|
||||
* without catalog edits — see the Fireworks docs `List Models` API.
|
||||
*/
|
||||
interface FireworksControlPlaneModel {
|
||||
/** Resource name, e.g. `accounts/fireworks/models/kimi-k2p7-code`. */
|
||||
name?: unknown;
|
||||
displayName?: unknown;
|
||||
contextLength?: unknown;
|
||||
supportsImageInput?: unknown;
|
||||
supportsTools?: unknown;
|
||||
supportsServerless?: unknown;
|
||||
state?: unknown;
|
||||
}
|
||||
|
||||
/**
|
||||
* Derive the control-plane list endpoint from the inference base URL. The
|
||||
* inference API lives under `/inference/v1` while the control plane is
|
||||
* `/v1/accounts/<account>/models` on the same origin, so we route off origin.
|
||||
* Returns null for unparseable overrides (custom gateways) so discovery falls
|
||||
* back to the cached/bundled catalog.
|
||||
*/
|
||||
function toFireworksControlPlaneModelsUrl(baseUrl: string, account: string): string | null {
|
||||
try {
|
||||
return `${new URL(baseUrl).origin}/v1/accounts/${account}/models`;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function mapFireworksControlPlaneModel(
|
||||
record: FireworksControlPlaneModel,
|
||||
publicModelId: string,
|
||||
reference: ModelSpec<"openai-completions"> | undefined,
|
||||
baseUrl: string,
|
||||
): ModelSpec<"openai-completions"> {
|
||||
const name = toModelName(record.displayName, reference?.name ?? publicModelId);
|
||||
const supportsImage = toBoolean(record.supportsImageInput) === true;
|
||||
const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? UNK_CONTEXT_WINDOW);
|
||||
// The control plane reports no max-output budget; default the Kimi family to
|
||||
// its published cap, everyone else to the discovery fallback, then clamp.
|
||||
const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : UNK_MAX_TOKENS;
|
||||
const maxTokens = clampFireworksKimiMaxTokens(publicModelId, reference?.maxTokens ?? fallbackMaxTokens);
|
||||
const base: ModelSpec<"openai-completions"> = reference ?? {
|
||||
id: publicModelId,
|
||||
name,
|
||||
api: "openai-completions",
|
||||
provider: "fireworks",
|
||||
baseUrl,
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow,
|
||||
maxTokens,
|
||||
};
|
||||
const model: ModelSpec<"openai-completions"> = {
|
||||
...base,
|
||||
id: publicModelId,
|
||||
api: "openai-completions",
|
||||
provider: "fireworks",
|
||||
baseUrl,
|
||||
name,
|
||||
// The control plane exposes capability flags but no reasoning bit. Every
|
||||
// serverless chat LLM Fireworks ships reasons, and `buildModel` derives
|
||||
// the Fireworks effort map from the id at build time — so default
|
||||
// unbundled models to reasoning while bundled references keep their value.
|
||||
reasoning: reference?.reasoning ?? true,
|
||||
input: supportsImage ? ["text", "image"] : (reference?.input ?? ["text"]),
|
||||
contextWindow,
|
||||
maxTokens,
|
||||
};
|
||||
return stripFireworksDeepSeekThinkingToggle(model, publicModelId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Discover Fireworks serverless models via the control-plane `List Models`
|
||||
* API (`supports_serverless=true`), paginating the full catalog. Returns null
|
||||
* on any transport/protocol failure so the model manager keeps the cached or
|
||||
* bundled catalog rather than caching a truncated list as authoritative.
|
||||
*/
|
||||
async function fetchFireworksServerlessModels(options: {
|
||||
baseUrl: string;
|
||||
apiKey: string;
|
||||
resolveReference: (publicModelId: string) => ModelSpec<"openai-completions"> | undefined;
|
||||
fetch?: FetchImpl;
|
||||
}): Promise<ModelSpec<"openai-completions">[] | null> {
|
||||
const listUrl = toFireworksControlPlaneModelsUrl(options.baseUrl, FIREWORKS_CONTROL_PLANE_ACCOUNT);
|
||||
if (!listUrl) return null;
|
||||
const fetchImpl = options.fetch ?? fetch;
|
||||
const collected = new Map<string, ModelSpec<"openai-completions">>();
|
||||
let pageToken = "";
|
||||
for (let page = 0; page < FIREWORKS_CONTROL_PLANE_MAX_PAGES; page++) {
|
||||
const url = new URL(listUrl);
|
||||
url.searchParams.set("filter", FIREWORKS_SERVERLESS_FILTER);
|
||||
url.searchParams.set("pageSize", String(FIREWORKS_CONTROL_PLANE_PAGE_SIZE));
|
||||
if (pageToken) url.searchParams.set("pageToken", pageToken);
|
||||
let response: Response;
|
||||
try {
|
||||
response = await fetchImpl(url.toString(), {
|
||||
method: "GET",
|
||||
headers: { Accept: "application/json", Authorization: `Bearer ${options.apiKey}` },
|
||||
});
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (!response.ok) return null;
|
||||
let payload: unknown;
|
||||
try {
|
||||
payload = await response.json();
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (!isRecord(payload)) return null;
|
||||
const models = Array.isArray(payload.models) ? payload.models : [];
|
||||
for (const entry of models) {
|
||||
if (!isRecord(entry)) continue;
|
||||
const record = entry as FireworksControlPlaneModel;
|
||||
if (toBoolean(record.supportsServerless) !== true) continue;
|
||||
if (toBoolean(record.supportsTools) !== true) continue;
|
||||
if (typeof record.state === "string" && record.state !== "READY") continue;
|
||||
const wireName = typeof record.name === "string" ? record.name : "";
|
||||
if (!wireName) continue;
|
||||
const publicModelId = toFireworksPublicModelId(wireName);
|
||||
if (!publicModelId) continue;
|
||||
collected.set(
|
||||
publicModelId,
|
||||
mapFireworksControlPlaneModel(
|
||||
record,
|
||||
publicModelId,
|
||||
options.resolveReference(publicModelId),
|
||||
options.baseUrl,
|
||||
),
|
||||
);
|
||||
}
|
||||
const next = typeof payload.nextPageToken === "string" ? payload.nextPageToken : "";
|
||||
if (!next) break;
|
||||
pageToken = next;
|
||||
}
|
||||
return Array.from(collected.values());
|
||||
}
|
||||
|
||||
function createModelsDevReferenceMap<TApi extends Api>(
|
||||
@@ -1184,35 +1322,11 @@ export function fireworksModelManagerOptions(
|
||||
...(apiKey && {
|
||||
fetchDynamicModels: async () => {
|
||||
const modelsDevReferences = await loadModelsDevReferences<"openai-completions">(config?.fetch);
|
||||
return fetchOpenAICompatibleModels({
|
||||
api: "openai-completions",
|
||||
provider: "fireworks",
|
||||
return fetchFireworksServerlessModels({
|
||||
baseUrl,
|
||||
apiKey,
|
||||
filterModel: entry =>
|
||||
toBoolean(entry.supports_chat) === true && toBoolean(entry.supports_tools) === true,
|
||||
mapModel: (entry, defaults) => {
|
||||
const publicModelId = toFireworksPublicModelId(defaults.id);
|
||||
const reference = modelsDevReferences.get(publicModelId) ?? bundledReferences(publicModelId);
|
||||
const model = stripFireworksDeepSeekThinkingToggle(
|
||||
mapWithBundledReference(entry, defaults, reference),
|
||||
publicModelId,
|
||||
);
|
||||
return {
|
||||
...model,
|
||||
id: publicModelId,
|
||||
api: "openai-completions",
|
||||
provider: "fireworks",
|
||||
baseUrl,
|
||||
name: toFireworksModelName(entry, model.name),
|
||||
input: toBoolean(entry.supports_image_input) === true ? ["text", "image"] : ["text"],
|
||||
contextWindow: toPositiveNumber(entry.context_length, model.contextWindow),
|
||||
maxTokens: clampFireworksKimiMaxTokens(
|
||||
publicModelId,
|
||||
toPositiveNumber(entry.max_completion_tokens, model.maxTokens),
|
||||
),
|
||||
};
|
||||
},
|
||||
resolveReference: publicModelId =>
|
||||
modelsDevReferences.get(publicModelId) ?? bundledReferences(publicModelId),
|
||||
fetch: config?.fetch,
|
||||
});
|
||||
},
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
/**
|
||||
* Fireworks discovery sources the **control-plane** `List Models` API
|
||||
* (`GET /v1/accounts/{account}/models?filter=supports_serverless=true`) rather
|
||||
* than the OpenAI-compatible `/v1/models` inference envelope. The inference
|
||||
* endpoint omits on-demand serverless models (e.g. `kimi-k2p7-code`), so models
|
||||
* Fireworks publishes but does not echo in `/v1/models` were invisible in the
|
||||
* picker until hand-added to the bundled catalog. The control-plane catalog
|
||||
* enumerates every serverless model with capability metadata, so new releases
|
||||
* surface automatically with no catalog edits.
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { fireworksModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
|
||||
import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types";
|
||||
|
||||
function jsonResponse(body: unknown): Response {
|
||||
return new Response(JSON.stringify(body), {
|
||||
status: 200,
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
}
|
||||
|
||||
const PAGE_1 = [
|
||||
{
|
||||
name: "accounts/fireworks/models/kimi-k2p7-code",
|
||||
displayName: "Kimi K2.7 Code",
|
||||
contextLength: 262144,
|
||||
supportsImageInput: true,
|
||||
supportsTools: true,
|
||||
supportsServerless: true,
|
||||
state: "READY",
|
||||
},
|
||||
// Image model: serverless but not tool-capable — must be filtered out.
|
||||
{
|
||||
name: "accounts/fireworks/models/flux-1-schnell-fp8",
|
||||
displayName: "FLUX.1 [schnell]",
|
||||
contextLength: 0,
|
||||
supportsImageInput: false,
|
||||
supportsTools: false,
|
||||
supportsServerless: true,
|
||||
state: "READY",
|
||||
},
|
||||
];
|
||||
|
||||
const PAGE_2 = [
|
||||
{
|
||||
name: "accounts/fireworks/models/deepseek-v4-flash",
|
||||
displayName: "DeepSeek-V4-Flash",
|
||||
contextLength: 1048576,
|
||||
supportsImageInput: false,
|
||||
supportsTools: true,
|
||||
supportsServerless: true,
|
||||
state: "READY",
|
||||
},
|
||||
// Tool-capable but not serverless — dedicated-only, must be filtered out.
|
||||
{
|
||||
name: "accounts/fireworks/models/kimi-k2-instruct-0905",
|
||||
displayName: "Kimi K2 Instruct 0905",
|
||||
contextLength: 262144,
|
||||
supportsImageInput: false,
|
||||
supportsTools: true,
|
||||
supportsServerless: false,
|
||||
state: "READY",
|
||||
},
|
||||
// Serverless + tools but still spinning up — must be filtered out.
|
||||
{
|
||||
name: "accounts/fireworks/models/some-pending-model",
|
||||
displayName: "Pending Model",
|
||||
contextLength: 131072,
|
||||
supportsImageInput: false,
|
||||
supportsTools: true,
|
||||
supportsServerless: true,
|
||||
state: "DEPLOYING",
|
||||
},
|
||||
];
|
||||
|
||||
function createMockFetch(): { fetch: FetchImpl; controlPlaneUrls: string[] } {
|
||||
const controlPlaneUrls: string[] = [];
|
||||
const fetch = (async (input: string | URL | Request): Promise<Response> => {
|
||||
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
||||
// models.dev reference fetch — return an empty catalog so the mapper relies
|
||||
// purely on control-plane + bundled references.
|
||||
if (url.startsWith("https://models.dev")) {
|
||||
return jsonResponse({});
|
||||
}
|
||||
if (url.includes("/v1/accounts/fireworks/models")) {
|
||||
controlPlaneUrls.push(url);
|
||||
const token = new URL(url).searchParams.get("pageToken");
|
||||
return token
|
||||
? jsonResponse({ models: PAGE_2, totalSize: 4 })
|
||||
: jsonResponse({ models: PAGE_1, nextPageToken: "page-2", totalSize: 4 });
|
||||
}
|
||||
return new Response("unexpected", { status: 404 });
|
||||
}) as unknown as FetchImpl;
|
||||
return { fetch, controlPlaneUrls };
|
||||
}
|
||||
|
||||
async function discover(): Promise<{ models: ModelSpec<"openai-completions">[]; controlPlaneUrls: string[] }> {
|
||||
const { fetch, controlPlaneUrls } = createMockFetch();
|
||||
const options = fireworksModelManagerOptions({ apiKey: "fw_test_key", fetch });
|
||||
const result = (await options.fetchDynamicModels?.()) ?? [];
|
||||
return { models: result as ModelSpec<"openai-completions">[], controlPlaneUrls };
|
||||
}
|
||||
|
||||
describe("Fireworks control-plane serverless discovery", () => {
|
||||
it("queries the control-plane serverless catalog, not the OpenAI-compat /v1/models", async () => {
|
||||
const { controlPlaneUrls } = await discover();
|
||||
expect(controlPlaneUrls.length).toBeGreaterThan(0);
|
||||
const first = new URL(controlPlaneUrls[0]);
|
||||
expect(first.pathname).toBe("/v1/accounts/fireworks/models");
|
||||
expect(first.searchParams.get("filter")).toBe("supports_serverless=true");
|
||||
// No request should hit the inference `/v1/models` listing.
|
||||
expect(controlPlaneUrls.some(u => u.includes("/inference/v1/models"))).toBe(false);
|
||||
});
|
||||
|
||||
it("paginates the full catalog via nextPageToken", async () => {
|
||||
const { models, controlPlaneUrls } = await discover();
|
||||
expect(controlPlaneUrls.length).toBe(2);
|
||||
// Models from both pages are collected.
|
||||
const ids = models.map(m => m.id);
|
||||
expect(ids).toContain("kimi-k2.7-code");
|
||||
expect(ids).toContain("deepseek-v4-flash");
|
||||
});
|
||||
|
||||
it("surfaces an unbundled on-demand model (kimi-k2.7-code) with live metadata", async () => {
|
||||
const { models } = await discover();
|
||||
const kimi = models.find(m => m.id === "kimi-k2.7-code");
|
||||
expect(kimi).toBeDefined();
|
||||
if (!kimi) return;
|
||||
// Wire id `kimi-k2p7-code` normalizes to the public `kimi-k2.7-code`.
|
||||
expect(kimi.name).toBe("Kimi K2.7 Code");
|
||||
expect(kimi.provider).toBe("fireworks");
|
||||
expect(kimi.baseUrl).toBe("https://api.fireworks.ai/inference/v1");
|
||||
expect(kimi.contextWindow).toBe(262144);
|
||||
// Kimi family clamps to the published 32,768 output cap.
|
||||
expect(kimi.maxTokens).toBe(32768);
|
||||
expect(kimi.input).toEqual(["text", "image"]);
|
||||
// Control plane reports no reasoning bit; serverless chat LLMs default on.
|
||||
expect(kimi.reasoning).toBe(true);
|
||||
});
|
||||
|
||||
it("filters non-serverless, non-tool, and not-ready records", async () => {
|
||||
const { models } = await discover();
|
||||
const ids = models.map(m => m.id);
|
||||
expect(ids).not.toContain("flux-1-schnell-fp8"); // tools: false
|
||||
expect(ids).not.toContain("kimi-k2-instruct-0905"); // serverless: false
|
||||
expect(ids).not.toContain("some-pending-model"); // state: DEPLOYING
|
||||
});
|
||||
|
||||
it("builds the discovered model with Fireworks effort-mode thinking", async () => {
|
||||
const { models } = await discover();
|
||||
const kimi = models.find(m => m.id === "kimi-k2.7-code");
|
||||
expect(kimi).toBeDefined();
|
||||
if (!kimi) return;
|
||||
const built = buildModel(kimi);
|
||||
expect(built.reasoning).toBe(true);
|
||||
// reasoning + Fireworks host ⇒ buildModel derives an effort-mode thinking
|
||||
// config (the Fireworks effort map), so the model is usable with thinking
|
||||
// tiers without any bundled metadata.
|
||||
expect(built.thinking?.mode).toBe("effort");
|
||||
});
|
||||
|
||||
it("returns null on a control-plane transport failure so the manager keeps its cache", async () => {
|
||||
const fetch = (async (input: string | URL | Request): Promise<Response> => {
|
||||
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
|
||||
if (url.startsWith("https://models.dev")) return jsonResponse({});
|
||||
return new Response("server error", { status: 500 });
|
||||
}) as unknown as FetchImpl;
|
||||
const options = fireworksModelManagerOptions({ apiKey: "fw_test_key", fetch });
|
||||
const result = await options.fetchDynamicModels?.();
|
||||
expect(result).toBeNull();
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user