feat(catalog): switched Fireworks discovery to control-plane and added bundled models

- Updated Fireworks model discovery to use the serverless control-plane endpoint.
- Added bundled Fireworks models including deepseek-v4-flash, kimi-k2.7-code, and qwen variants.
- Implemented paginated control-plane discovery with filtering of eligible Fireworks models.
- Updated github-copilot contextWindow from 222222 to 524288 tokens.
This commit is contained in:
can1357
2026-06-13 15:01:40 +02:00
parent 9dcaf1ae6b
commit 2a7e10cc44
4 changed files with 573 additions and 39 deletions
+9
View File
@@ -1,6 +1,15 @@
# Changelog
## [Unreleased]
### Added
- Added bundled Fireworks models `deepseek-v4-flash`, `kimi-k2.7-code`, `minimax-m2.5`, `minimax-m3`, `nemotron-3-ultra-nvfp4`, `qwen3.6-plus`, and `qwen3.7-plus`
- Changed
### Changed
- Changed the `github-copilot` model context window to `524288` tokens
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
### Fixed
+239 -1
View File
@@ -13598,6 +13598,51 @@
}
},
"fireworks": {
"deepseek-v4-flash": {
"id": "deepseek-v4-flash",
"name": "DeepSeek V4 Flash",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.14,
"output": 0.28,
"cacheRead": 0.0028,
"cacheWrite": 0
},
"contextWindow": 1048576,
"maxTokens": 384000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "high",
"low": "high",
"medium": "high",
"high": "high",
"xhigh": "max"
}
},
"compat": {
"supportsDeveloperRole": false,
"supportsReasoningEffort": true,
"maxTokensField": "max_tokens",
"supportsToolChoice": false,
"reasoningContentField": "reasoning_content",
"requiresReasoningContentForToolCalls": true,
"requiresAssistantContentForToolCalls": true
}
},
"deepseek-v4-pro": {
"id": "deepseek-v4-pro",
"name": "DeepSeek V4 Pro",
@@ -13800,6 +13845,67 @@
}
}
},
"kimi-k2.7-code": {
"id": "kimi-k2.7-code",
"name": "Kimi K2.7 Code",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.95,
"output": 4,
"cacheRead": 0.19,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "none"
}
}
},
"minimax-m2.5": {
"id": "minimax-m2.5",
"name": "MiniMax M2.5",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.3,
"output": 1.2,
"cacheRead": 0.06,
"cacheWrite": 0
},
"contextWindow": 196608,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high"
],
"requiresEffort": true
}
},
"minimax-m2.7": {
"id": "minimax-m2.7",
"name": "MiniMax M2.7",
@@ -13827,6 +13933,138 @@
],
"requiresEffort": true
}
},
"minimax-m3": {
"id": "minimax-m3",
"name": "MiniMax M3 (3x usage)",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.1,
"output": 0.4,
"cacheRead": 0.02,
"cacheWrite": 0
},
"contextWindow": 512000,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "none"
}
}
},
"nemotron-3-ultra-nvfp4": {
"id": "nemotron-3-ultra-nvfp4",
"name": "NVIDIA Nemotron 3 Ultra NVFP4",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 8888,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "none"
}
}
},
"qwen3.6-plus": {
"id": "qwen3.6-plus",
"name": "Qwen3.6 Plus",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
],
"effortMap": {
"minimal": "none"
}
},
"compat": {
"supportsDeveloperRole": false
}
},
"qwen3.7-plus": {
"id": "qwen3.7-plus",
"name": "Qwen3.7 Plus",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.4,
"output": 1.6,
"cacheRead": 0.04,
"cacheWrite": 0.5
},
"contextWindow": 1000000,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
],
"effortMap": {
"minimal": "none"
}
}
}
},
"github-copilot": {
@@ -64124,7 +64362,7 @@
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 222222,
"contextWindow": 524288,
"maxTokens": 8888,
"compat": {
"supportsUsageInStreaming": false
@@ -1128,17 +1128,155 @@ export interface FireworksModelManagerConfig {
fetch?: FetchImpl;
}
function toFireworksModelName(entry: OpenAICompatibleModelRecord, fallback: string): string {
const name = toModelName(entry.name, "");
if (name) return name;
const id = typeof entry.id === "string" ? entry.id : fallback;
const shortName = id.split("/").at(-1) ?? fallback;
if (fallback !== id && fallback !== shortName) return fallback;
return shortName
.split("-")
.filter(Boolean)
.map(part => part.charAt(0).toUpperCase() + part.slice(1))
.join(" ");
const FIREWORKS_CONTROL_PLANE_ACCOUNT = "fireworks";
const FIREWORKS_SERVERLESS_FILTER = "supports_serverless=true";
const FIREWORKS_CONTROL_PLANE_PAGE_SIZE = 200;
const FIREWORKS_CONTROL_PLANE_MAX_PAGES = 25;
/**
* One record from the Fireworks control-plane catalog
* (`GET /v1/accounts/{account}/models`). This is distinct from the
* OpenAI-compatible `/v1/models` inference envelope: the control plane
* enumerates the full serverless catalog with camelCase capability metadata,
* including on-demand models (e.g. `kimi-k2p7-code`) that never surface in
* `/v1/models`. Discovering here is what keeps new serverless models appearing
* without catalog edits — see the Fireworks docs `List Models` API.
*/
interface FireworksControlPlaneModel {
/** Resource name, e.g. `accounts/fireworks/models/kimi-k2p7-code`. */
name?: unknown;
displayName?: unknown;
contextLength?: unknown;
supportsImageInput?: unknown;
supportsTools?: unknown;
supportsServerless?: unknown;
state?: unknown;
}
/**
* Derive the control-plane list endpoint from the inference base URL. The
* inference API lives under `/inference/v1` while the control plane is
* `/v1/accounts/<account>/models` on the same origin, so we route off origin.
* Returns null for unparseable overrides (custom gateways) so discovery falls
* back to the cached/bundled catalog.
*/
function toFireworksControlPlaneModelsUrl(baseUrl: string, account: string): string | null {
try {
return `${new URL(baseUrl).origin}/v1/accounts/${account}/models`;
} catch {
return null;
}
}
function mapFireworksControlPlaneModel(
record: FireworksControlPlaneModel,
publicModelId: string,
reference: ModelSpec<"openai-completions"> | undefined,
baseUrl: string,
): ModelSpec<"openai-completions"> {
const name = toModelName(record.displayName, reference?.name ?? publicModelId);
const supportsImage = toBoolean(record.supportsImageInput) === true;
const contextWindow = toPositiveNumber(record.contextLength, reference?.contextWindow ?? UNK_CONTEXT_WINDOW);
// The control plane reports no max-output budget; default the Kimi family to
// its published cap, everyone else to the discovery fallback, then clamp.
const fallbackMaxTokens = isFireworksKimiK2ModelId(publicModelId) ? FIREWORKS_KIMI_MAX_TOKENS : UNK_MAX_TOKENS;
const maxTokens = clampFireworksKimiMaxTokens(publicModelId, reference?.maxTokens ?? fallbackMaxTokens);
const base: ModelSpec<"openai-completions"> = reference ?? {
id: publicModelId,
name,
api: "openai-completions",
provider: "fireworks",
baseUrl,
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow,
maxTokens,
};
const model: ModelSpec<"openai-completions"> = {
...base,
id: publicModelId,
api: "openai-completions",
provider: "fireworks",
baseUrl,
name,
// The control plane exposes capability flags but no reasoning bit. Every
// serverless chat LLM Fireworks ships reasons, and `buildModel` derives
// the Fireworks effort map from the id at build time — so default
// unbundled models to reasoning while bundled references keep their value.
reasoning: reference?.reasoning ?? true,
input: supportsImage ? ["text", "image"] : (reference?.input ?? ["text"]),
contextWindow,
maxTokens,
};
return stripFireworksDeepSeekThinkingToggle(model, publicModelId);
}
/**
* Discover Fireworks serverless models via the control-plane `List Models`
* API (`supports_serverless=true`), paginating the full catalog. Returns null
* on any transport/protocol failure so the model manager keeps the cached or
* bundled catalog rather than caching a truncated list as authoritative.
*/
async function fetchFireworksServerlessModels(options: {
baseUrl: string;
apiKey: string;
resolveReference: (publicModelId: string) => ModelSpec<"openai-completions"> | undefined;
fetch?: FetchImpl;
}): Promise<ModelSpec<"openai-completions">[] | null> {
const listUrl = toFireworksControlPlaneModelsUrl(options.baseUrl, FIREWORKS_CONTROL_PLANE_ACCOUNT);
if (!listUrl) return null;
const fetchImpl = options.fetch ?? fetch;
const collected = new Map<string, ModelSpec<"openai-completions">>();
let pageToken = "";
for (let page = 0; page < FIREWORKS_CONTROL_PLANE_MAX_PAGES; page++) {
const url = new URL(listUrl);
url.searchParams.set("filter", FIREWORKS_SERVERLESS_FILTER);
url.searchParams.set("pageSize", String(FIREWORKS_CONTROL_PLANE_PAGE_SIZE));
if (pageToken) url.searchParams.set("pageToken", pageToken);
let response: Response;
try {
response = await fetchImpl(url.toString(), {
method: "GET",
headers: { Accept: "application/json", Authorization: `Bearer ${options.apiKey}` },
});
} catch {
return null;
}
if (!response.ok) return null;
let payload: unknown;
try {
payload = await response.json();
} catch {
return null;
}
if (!isRecord(payload)) return null;
const models = Array.isArray(payload.models) ? payload.models : [];
for (const entry of models) {
if (!isRecord(entry)) continue;
const record = entry as FireworksControlPlaneModel;
if (toBoolean(record.supportsServerless) !== true) continue;
if (toBoolean(record.supportsTools) !== true) continue;
if (typeof record.state === "string" && record.state !== "READY") continue;
const wireName = typeof record.name === "string" ? record.name : "";
if (!wireName) continue;
const publicModelId = toFireworksPublicModelId(wireName);
if (!publicModelId) continue;
collected.set(
publicModelId,
mapFireworksControlPlaneModel(
record,
publicModelId,
options.resolveReference(publicModelId),
options.baseUrl,
),
);
}
const next = typeof payload.nextPageToken === "string" ? payload.nextPageToken : "";
if (!next) break;
pageToken = next;
}
return Array.from(collected.values());
}
function createModelsDevReferenceMap<TApi extends Api>(
@@ -1184,35 +1322,11 @@ export function fireworksModelManagerOptions(
...(apiKey && {
fetchDynamicModels: async () => {
const modelsDevReferences = await loadModelsDevReferences<"openai-completions">(config?.fetch);
return fetchOpenAICompatibleModels({
api: "openai-completions",
provider: "fireworks",
return fetchFireworksServerlessModels({
baseUrl,
apiKey,
filterModel: entry =>
toBoolean(entry.supports_chat) === true && toBoolean(entry.supports_tools) === true,
mapModel: (entry, defaults) => {
const publicModelId = toFireworksPublicModelId(defaults.id);
const reference = modelsDevReferences.get(publicModelId) ?? bundledReferences(publicModelId);
const model = stripFireworksDeepSeekThinkingToggle(
mapWithBundledReference(entry, defaults, reference),
publicModelId,
);
return {
...model,
id: publicModelId,
api: "openai-completions",
provider: "fireworks",
baseUrl,
name: toFireworksModelName(entry, model.name),
input: toBoolean(entry.supports_image_input) === true ? ["text", "image"] : ["text"],
contextWindow: toPositiveNumber(entry.context_length, model.contextWindow),
maxTokens: clampFireworksKimiMaxTokens(
publicModelId,
toPositiveNumber(entry.max_completion_tokens, model.maxTokens),
),
};
},
resolveReference: publicModelId =>
modelsDevReferences.get(publicModelId) ?? bundledReferences(publicModelId),
fetch: config?.fetch,
});
},
@@ -0,0 +1,173 @@
/**
* Fireworks discovery sources the **control-plane** `List Models` API
* (`GET /v1/accounts/{account}/models?filter=supports_serverless=true`) rather
* than the OpenAI-compatible `/v1/models` inference envelope. The inference
* endpoint omits on-demand serverless models (e.g. `kimi-k2p7-code`), so models
* Fireworks publishes but does not echo in `/v1/models` were invisible in the
* picker until hand-added to the bundled catalog. The control-plane catalog
* enumerates every serverless model with capability metadata, so new releases
* surface automatically with no catalog edits.
*/
import { describe, expect, it } from "bun:test";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { fireworksModelManagerOptions } from "@oh-my-pi/pi-catalog/provider-models/openai-compat";
import type { FetchImpl, ModelSpec } from "@oh-my-pi/pi-catalog/types";
function jsonResponse(body: unknown): Response {
return new Response(JSON.stringify(body), {
status: 200,
headers: { "content-type": "application/json" },
});
}
const PAGE_1 = [
{
name: "accounts/fireworks/models/kimi-k2p7-code",
displayName: "Kimi K2.7 Code",
contextLength: 262144,
supportsImageInput: true,
supportsTools: true,
supportsServerless: true,
state: "READY",
},
// Image model: serverless but not tool-capable — must be filtered out.
{
name: "accounts/fireworks/models/flux-1-schnell-fp8",
displayName: "FLUX.1 [schnell]",
contextLength: 0,
supportsImageInput: false,
supportsTools: false,
supportsServerless: true,
state: "READY",
},
];
const PAGE_2 = [
{
name: "accounts/fireworks/models/deepseek-v4-flash",
displayName: "DeepSeek-V4-Flash",
contextLength: 1048576,
supportsImageInput: false,
supportsTools: true,
supportsServerless: true,
state: "READY",
},
// Tool-capable but not serverless — dedicated-only, must be filtered out.
{
name: "accounts/fireworks/models/kimi-k2-instruct-0905",
displayName: "Kimi K2 Instruct 0905",
contextLength: 262144,
supportsImageInput: false,
supportsTools: true,
supportsServerless: false,
state: "READY",
},
// Serverless + tools but still spinning up — must be filtered out.
{
name: "accounts/fireworks/models/some-pending-model",
displayName: "Pending Model",
contextLength: 131072,
supportsImageInput: false,
supportsTools: true,
supportsServerless: true,
state: "DEPLOYING",
},
];
function createMockFetch(): { fetch: FetchImpl; controlPlaneUrls: string[] } {
const controlPlaneUrls: string[] = [];
const fetch = (async (input: string | URL | Request): Promise<Response> => {
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
// models.dev reference fetch — return an empty catalog so the mapper relies
// purely on control-plane + bundled references.
if (url.startsWith("https://models.dev")) {
return jsonResponse({});
}
if (url.includes("/v1/accounts/fireworks/models")) {
controlPlaneUrls.push(url);
const token = new URL(url).searchParams.get("pageToken");
return token
? jsonResponse({ models: PAGE_2, totalSize: 4 })
: jsonResponse({ models: PAGE_1, nextPageToken: "page-2", totalSize: 4 });
}
return new Response("unexpected", { status: 404 });
}) as unknown as FetchImpl;
return { fetch, controlPlaneUrls };
}
async function discover(): Promise<{ models: ModelSpec<"openai-completions">[]; controlPlaneUrls: string[] }> {
const { fetch, controlPlaneUrls } = createMockFetch();
const options = fireworksModelManagerOptions({ apiKey: "fw_test_key", fetch });
const result = (await options.fetchDynamicModels?.()) ?? [];
return { models: result as ModelSpec<"openai-completions">[], controlPlaneUrls };
}
describe("Fireworks control-plane serverless discovery", () => {
it("queries the control-plane serverless catalog, not the OpenAI-compat /v1/models", async () => {
const { controlPlaneUrls } = await discover();
expect(controlPlaneUrls.length).toBeGreaterThan(0);
const first = new URL(controlPlaneUrls[0]);
expect(first.pathname).toBe("/v1/accounts/fireworks/models");
expect(first.searchParams.get("filter")).toBe("supports_serverless=true");
// No request should hit the inference `/v1/models` listing.
expect(controlPlaneUrls.some(u => u.includes("/inference/v1/models"))).toBe(false);
});
it("paginates the full catalog via nextPageToken", async () => {
const { models, controlPlaneUrls } = await discover();
expect(controlPlaneUrls.length).toBe(2);
// Models from both pages are collected.
const ids = models.map(m => m.id);
expect(ids).toContain("kimi-k2.7-code");
expect(ids).toContain("deepseek-v4-flash");
});
it("surfaces an unbundled on-demand model (kimi-k2.7-code) with live metadata", async () => {
const { models } = await discover();
const kimi = models.find(m => m.id === "kimi-k2.7-code");
expect(kimi).toBeDefined();
if (!kimi) return;
// Wire id `kimi-k2p7-code` normalizes to the public `kimi-k2.7-code`.
expect(kimi.name).toBe("Kimi K2.7 Code");
expect(kimi.provider).toBe("fireworks");
expect(kimi.baseUrl).toBe("https://api.fireworks.ai/inference/v1");
expect(kimi.contextWindow).toBe(262144);
// Kimi family clamps to the published 32,768 output cap.
expect(kimi.maxTokens).toBe(32768);
expect(kimi.input).toEqual(["text", "image"]);
// Control plane reports no reasoning bit; serverless chat LLMs default on.
expect(kimi.reasoning).toBe(true);
});
it("filters non-serverless, non-tool, and not-ready records", async () => {
const { models } = await discover();
const ids = models.map(m => m.id);
expect(ids).not.toContain("flux-1-schnell-fp8"); // tools: false
expect(ids).not.toContain("kimi-k2-instruct-0905"); // serverless: false
expect(ids).not.toContain("some-pending-model"); // state: DEPLOYING
});
it("builds the discovered model with Fireworks effort-mode thinking", async () => {
const { models } = await discover();
const kimi = models.find(m => m.id === "kimi-k2.7-code");
expect(kimi).toBeDefined();
if (!kimi) return;
const built = buildModel(kimi);
expect(built.reasoning).toBe(true);
// reasoning + Fireworks host ⇒ buildModel derives an effort-mode thinking
// config (the Fireworks effort map), so the model is usable with thinking
// tiers without any bundled metadata.
expect(built.thinking?.mode).toBe("effort");
});
it("returns null on a control-plane transport failure so the manager keeps its cache", async () => {
const fetch = (async (input: string | URL | Request): Promise<Response> => {
const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
if (url.startsWith("https://models.dev")) return jsonResponse({});
return new Response("server error", { status: 500 });
}) as unknown as FetchImpl;
const options = fireworksModelManagerOptions({ apiKey: "fw_test_key", fetch });
const result = await options.fetchDynamicModels?.();
expect(result).toBeNull();
});
});