feat(catalog): supported fireworks fast serving path

- Added support for "Fast" serving-path variants for select Fireworks models.
- Updated compatibility logic to route `-fast` suffixes to the appropriate router wire format.
- Extended the model generation catalog to include these Fast variants with their respective pricing.
- Updated AI types to allow the `priority` service tier for Fireworks providers.
This commit is contained in:
can1357
2026-06-20 09:09:47 +02:00
parent b717a65fc3
commit 1afa6ba68a
11 changed files with 343 additions and 14 deletions
+7 -2
View File
@@ -1182,6 +1182,10 @@ async function streamAssistantResponse(
const dynamicReasoning = config.getReasoning?.();
const dynamicDisableReasoning = config.getDisableReasoning?.();
// `getServiceTier` is authoritative when present (replaces the static tier
// for both the wire request and telemetry), so callers can scope priority
// per model without touching the shared session `serviceTier`.
const effectiveServiceTier = config.getServiceTier ? config.getServiceTier(config.model) : config.serviceTier;
const harmonyMitigationEnabled = isHarmonyLeakMitigationTarget(config.model);
const harmonyAbortController = harmonyMitigationEnabled ? new AbortController() : undefined;
const requestSignal = harmonyAbortController
@@ -1229,7 +1233,7 @@ async function streamAssistantResponse(
topP: config.topP,
topK: config.topK,
presencePenalty: config.presencePenalty,
serviceTier: config.serviceTier,
serviceTier: effectiveServiceTier,
reasoningEffort: typeof effectiveReasoning === "string" ? effectiveReasoning : undefined,
toolChoice: effectiveToolChoice,
tools: llmContext.tools,
@@ -1251,7 +1255,7 @@ async function streamAssistantResponse(
const finishChat = async (message: AssistantMessage): Promise<void> => {
await finishChatSpan(telemetry, chatSpan, message, {
stepNumber: chatStepNumber,
serviceTier: config.serviceTier,
serviceTier: effectiveServiceTier,
responseHeaders: capturedHeaders,
baseUrl: config.model.baseUrl,
});
@@ -1267,6 +1271,7 @@ async function streamAssistantResponse(
reasoning: effectiveReasoning,
disableReasoning: effectiveDisableReasoning,
temperature: effectiveTemperature,
serviceTier: effectiveServiceTier,
signal: finalRequestSignal,
onResponse: captureOnResponse,
});
+18
View File
@@ -196,6 +196,13 @@ export interface AgentOptions {
presencePenalty?: number;
repetitionPenalty?: number;
serviceTier?: ServiceTier;
/**
* Per-call effective service-tier resolver. When set, it authoritatively
* supplies the request's tier (replacing the static `serviceTier` and its
* telemetry) per model — used to scope a provider/model into a priority
* serving path without mutating the shared session `serviceTier`.
*/
serviceTierResolver?: (model: Model) => ServiceTier | undefined;
/**
* If true, request that the underlying provider omit reasoning/thinking summaries
* from the response. The model still reasons internally; only the human-readable
@@ -333,6 +340,7 @@ export class Agent {
#presencePenalty?: number;
#repetitionPenalty?: number;
#serviceTier?: ServiceTier;
#serviceTierResolver?: (model: Model) => ServiceTier | undefined;
#hideThinkingSummary?: boolean;
#maxRetryDelayMs?: number;
#getToolContext?: (toolCall?: ToolCallContext) => AgentToolContext | undefined;
@@ -403,6 +411,7 @@ export class Agent {
this.#presencePenalty = opts.presencePenalty;
this.#repetitionPenalty = opts.repetitionPenalty;
this.#serviceTier = opts.serviceTier;
this.#serviceTierResolver = opts.serviceTierResolver;
this.#hideThinkingSummary = opts.hideThinkingSummary;
this.#maxRetryDelayMs = opts.maxRetryDelayMs;
this.getApiKey = opts.getApiKey;
@@ -610,6 +619,14 @@ export class Agent {
this.#serviceTier = value;
}
get serviceTierResolver(): ((model: Model) => ServiceTier | undefined) | undefined {
return this.#serviceTierResolver;
}
set serviceTierResolver(value: ((model: Model) => ServiceTier | undefined) | undefined) {
this.#serviceTierResolver = value;
}
get hideThinkingSummary(): boolean | undefined {
return this.#hideThinkingSummary;
}
@@ -1087,6 +1104,7 @@ export class Agent {
getToolChoice,
getReasoning: () => this.#state.thinkingLevel,
getDisableReasoning: () => this.#state.disableReasoning,
getServiceTier: this.#serviceTierResolver,
getSteeringMessages: async () => {
if (skipInitialSteeringPoll) {
skipInitialSteeringPoll = false;
+12
View File
@@ -8,6 +8,7 @@ import type {
ImageContent,
Message,
Model,
ServiceTier,
SimpleStreamOptions,
Static,
streamSimple,
@@ -315,6 +316,17 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
*/
getDisableReasoning?: () => boolean | undefined;
/**
* Per-call effective service-tier resolver. Unlike {@link getReasoning},
* this is *authoritative*: when set, its return value (including
* `undefined`) fully replaces the static `serviceTier` for the request and
* its telemetry. The resolver receives the model being requested so the
* caller can scope the tier per provider/model without mutating the shared
* session `serviceTier` (e.g. opting a Fireworks model into the Priority
* serving path while leaving the OpenAI/Anthropic tier untouched).
*/
getServiceTier?: (model: Model) => ServiceTier | undefined;
/**
* Called after a tool call has been validated and is about to execute.
*
+13 -6
View File
@@ -141,18 +141,25 @@ export function resolveServiceTier(
}
/**
* True when the (possibly scoped) tier should be sent as OpenAI's
* `service_tier` request field for the given provider. Non-OpenAI
* providers, unsupported tiers (`"auto"`, `"default"`), and scope
* mismatches all return false.
* True when the (possibly scoped) tier should be sent on the wire as the
* `service_tier` request field for the given provider. OpenAI / OpenAI-Codex
* accept `flex`/`scale`/`priority`; Fireworks Serverless realizes only its
* Priority serving path (`service_tier: "priority"`) on the OpenAI-compatible
* chat-completions endpoint. Unsupported tiers (`"auto"`, `"default"`), other
* providers, and scope mismatches all return false.
*/
export function shouldSendServiceTier(
serviceTier: ServiceTier | null | undefined,
provider: Provider | undefined,
): boolean {
if (provider !== "openai" && provider !== "openai-codex") return false;
const resolved = resolveServiceTier(serviceTier, provider);
return resolved === "flex" || resolved === "scale" || resolved === "priority";
if (provider === "openai" || provider === "openai-codex") {
return resolved === "flex" || resolved === "scale" || resolved === "priority";
}
if (provider === "fireworks") {
return resolved === "priority";
}
return false;
}
/**
@@ -29,6 +29,7 @@ import {
import { PROVIDER_DESCRIPTORS } from "../src/provider-models/descriptors";
import {
ANTHROPIC_CURATED_FALLBACK_MODELS,
buildFireworksFastSeed,
buildXaiOAuthStaticSeed,
clampFireworksKimiMaxTokens,
clampKimiK27CodeMaxTokens,
@@ -487,6 +488,11 @@ async function generateModels() {
// Mythos 5). Deduped behind upstream entries; metadata is pinned in
// applyAnthropicCatalogPolicy.
allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS);
// Seed Fireworks "Fast" serving-path variants (`<id>-fast`). Fast routers are
// not enumerated by the serverless control-plane list, so discovery never
// surfaces them; the seed projects each base entry into a fast variant.
// Deduped behind any identical previous-snapshot entry.
allModels.push(...buildFireworksFastSeed());
const specialDiscoverySources = [
{ label: "Antigravity", fetch: fetchAntigravityModels },
+6 -1
View File
@@ -7,6 +7,7 @@
* complete alternate views. Request handlers read `model.compat` fields and
* never detect, resolve, or allocate.
*/
import { isFireworksFastModelId } from "../fireworks-model-id";
import { hostMatchesUrl, modelMatchesHost } from "../hosts";
import {
isAnthropicNamespacedModelId,
@@ -276,8 +277,12 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
: undefined;
// Fireworks "Fast" variants (`<id>-fast`) are served from the router
// namespace (`accounts/fireworks/routers/<id>-fast`), like Fire Pass, rather
// than the `models/` namespace the rest of the `fireworks` provider uses.
const isFireworksFastRouter = provider === "fireworks" && isFireworksFastModelId(spec.id);
const wireModelIdMode: ResolvedOpenAISharedCompat["wireModelIdMode"] =
provider === "firepass"
provider === "firepass" || isFireworksFastRouter
? "firepass"
: provider === "fireworks"
? "fireworks"
@@ -28,3 +28,23 @@ export function toFirepassWireModelId(modelId: string): string {
const stripped = modelId.startsWith(FIREPASS_WIRE_PREFIX) ? modelId.slice(FIREPASS_WIRE_PREFIX.length) : modelId;
return `${FIREPASS_WIRE_PREFIX}${stripped.replace(VERSION_DOT_PATTERN, "p")}`;
}
/**
* Public-id suffix marking a Fireworks "Fast" serving-path variant. Fast is a
* higher-throughput route (100+ tok/s) exposed under a dedicated router id
* (`accounts/fireworks/routers/<id>-fast`), not a separate model — same weights,
* higher price, no Priority tier. We keep a friendly `<id>-fast` public id and
* translate it to the router wire form at request time (compat
* `wireModelIdMode: "firepass"`). See https://docs.fireworks.ai/serverless/serving-paths.
*/
export const FIREWORKS_FAST_SUFFIX = "-fast";
/** True for a Fireworks public model id that selects the Fast serving path. */
export function isFireworksFastModelId(modelId: string): boolean {
return modelId.endsWith(FIREWORKS_FAST_SUFFIX);
}
/** Strip the Fast suffix to recover the base (Standard-tier) model id. */
export function toFireworksBaseModelId(modelId: string): string {
return modelId.endsWith(FIREWORKS_FAST_SUFFIX) ? modelId.slice(0, -FIREWORKS_FAST_SUFFIX.length) : modelId;
}
+98
View File
@@ -14801,6 +14801,38 @@
}
}
},
"glm-5.1-fast": {
"id": "glm-5.1-fast",
"name": "GLM-5.1 Fast",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 2.8,
"output": 8.8,
"cacheRead": 0.52,
"cacheWrite": 0
},
"contextWindow": 202752,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "none"
}
}
},
"glm-5.2": {
"id": "glm-5.2",
"name": "GLM-5.2",
@@ -14947,6 +14979,39 @@
}
}
},
"kimi-k2.6-fast": {
"id": "kimi-k2.6-fast",
"name": "Kimi K2.6 Fast",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 2,
"output": 8,
"cacheRead": 0.3,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "none"
}
}
},
"kimi-k2.7-code": {
"id": "kimi-k2.7-code",
"name": "Kimi K2.7 Code",
@@ -14980,6 +15045,39 @@
}
}
},
"kimi-k2.7-code-fast": {
"id": "kimi-k2.7-code-fast",
"name": "Kimi K2.7 Code Fast",
"api": "openai-completions",
"provider": "fireworks",
"baseUrl": "https://api.fireworks.ai/inference/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 1.9,
"output": 8,
"cacheRead": 0.38,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "none"
}
}
},
"minimax-m2.5": {
"id": "minimax-m2.5",
"name": "MiniMax M2.5",
@@ -4,7 +4,7 @@ import {
type OpenAICompatibleModelRecord,
} from "../discovery/openai-compatible";
import { Effort } from "../effort";
import { toFireworksPublicModelId } from "../fireworks-model-id";
import { FIREWORKS_FAST_SUFFIX, toFireworksPublicModelId } from "../fireworks-model-id";
import { isGlmVisionModelId, isGrokReasoningEffortCapable, isReasoningGlmModelId } from "../identity/family";
import type { ModelManagerOptions } from "../model-manager";
import { getBundledModels } from "../models";
@@ -1258,6 +1258,51 @@ export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | n
return isKimiK27CodeModelId(modelId) ? Math.min(candidate, KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS) : candidate;
}
/**
* Fireworks Fast variants we surface. Each inherits the base model's
* limits/modalities/thinking and overrides only the cost with the Standard-column
* Fast prices from the Serverless pricing table; `cacheWrite` stays 0 (Fireworks
* bills no cache-write). Derived from the bundled base entries so metadata stays
* in lockstep, and the runtime auto-falls back to the base id on a failed fast
* request. See https://docs.fireworks.ai/serverless/pricing.
*/
const FIREWORKS_FAST_VARIANT_SPECS: ReadonlyArray<{
base: string;
name: string;
cost: { input: number; output: number; cacheRead: number };
}> = [
{ base: "kimi-k2.7-code", name: "Kimi K2.7 Code Fast", cost: { input: 1.9, output: 8, cacheRead: 0.38 } },
{ base: "kimi-k2.6", name: "Kimi K2.6 Fast", cost: { input: 2, output: 8, cacheRead: 0.3 } },
{ base: "glm-5.1", name: "GLM-5.1 Fast", cost: { input: 2.8, output: 8.8, cacheRead: 0.52 } },
];
/**
* Build the Fireworks Fast seed by projecting each base bundled spec into a
* `<id>-fast` variant. Pushed into the generated catalog (Fast routers never
* appear in the serverless control-plane list, so discovery cannot surface
* them) and deduped behind any identical previous-snapshot entry.
*/
export function buildFireworksFastSeed(): ModelSpec<"openai-completions">[] {
const bundled = createBundledReferenceMap<"openai-completions">("fireworks");
const seeds: ModelSpec<"openai-completions">[] = [];
for (const variant of FIREWORKS_FAST_VARIANT_SPECS) {
const base = bundled.get(variant.base);
if (!base) continue;
seeds.push({
...base,
id: `${variant.base}${FIREWORKS_FAST_SUFFIX}`,
name: variant.name,
cost: {
input: variant.cost.input,
output: variant.cost.output,
cacheRead: variant.cost.cacheRead,
cacheWrite: 0,
},
});
}
return seeds;
}
/**
* Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the
* DeepSeek-native binary `thinking` toggle when both are present.
@@ -134,7 +134,7 @@ export const TAB_GROUPS: Record<SettingTab, readonly string[]> = {
"Developer",
],
tasks: ["Modes", "Subagents", "Isolation", "Commands & Skills"],
providers: ["Services", "Tiny Model", "Protocol", "Privacy"],
providers: ["Services", "Fireworks", "Tiny Model", "Protocol", "Privacy"],
};
/** Status line segment identifiers */
@@ -4042,6 +4042,26 @@ export const SETTINGS_SCHEMA = {
],
},
},
"providers.fireworksTier": {
type: "enum",
values: ["standard", "priority"] as const,
default: "standard",
ui: {
tab: "providers",
group: "Fireworks",
label: "Fireworks Tier",
description:
"Serving path for Fireworks requests. Priority sends `service_tier: \"priority\"` for higher reliability during peak traffic at a higher price; Standard omits it. Fast (`-fast`) models ignore this — Fast is its own serving path.",
options: [
{ value: "standard", label: "Standard", description: "Default serving path (no service_tier)" },
{
value: "priority",
label: "Priority",
description: "Priority serving path: higher reliability, premium per-token pricing",
},
],
},
},
"providers.tts": {
type: "enum",
values: ["auto", "local", "xai"] as const,
@@ -105,6 +105,7 @@ import {
} from "@oh-my-pi/pi-ai";
import { stripToolDescriptions } from "@oh-my-pi/pi-ai/utils/schema";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { isFireworksFastModelId, toFireworksBaseModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives";
@@ -1537,6 +1538,10 @@ export class AgentSession {
this.#customCommands = config.customCommands ?? [];
this.#skillsSettings = config.skillsSettings;
this.#modelRegistry = config.modelRegistry;
// Resolve the wire service-tier per request so the Fireworks Priority
// toggle scopes priority to Fireworks alone, without mutating the shared
// session `serviceTier` that drives `/fast` and OpenAI/Anthropic priority.
this.agent.serviceTierResolver = model => this.#effectiveServiceTier(model);
this.#advisorReadOnlyTools = config.advisorReadOnlyTools;
this.#advisorWatchdogPrompt = config.advisorWatchdogPrompt;
this.#pruneToolDescriptions = config.pruneToolDescriptions === true;
@@ -2790,6 +2795,16 @@ export class AgentSession {
await emitAgentEndNotification();
return;
}
// Fireworks Fast variants degrade to their base model on a failed turn —
// including hard router errors the generic retry classifier rejects — so
// run this gate before the standard retryability check.
if (this.#isFireworksFastFallbackEligible(msg)) {
const didRetry = await this.#handleRetryableError(msg, { fireworksFastFallback: true });
if (didRetry) {
await emitAgentEndNotification();
return;
}
}
// Check for retryable errors first (overloaded, rate limit, server errors)
if (this.#isRetryableError(msg)) {
const didRetry = await this.#handleRetryableError(msg);
@@ -7279,7 +7294,25 @@ export class AgentSession {
* no model is selected.
*/
isFastModeActive(): boolean {
return resolveServiceTier(this.serviceTier, this.model?.provider) === "priority";
return resolveServiceTier(this.#effectiveServiceTier(), this.model?.provider) === "priority";
}
/**
* Effective wire service-tier for a request to `model`. Fireworks models
* take the Priority serving path only when the Providers › Fireworks Tier
* setting is `"priority"` — that toggle is the sole opt-in, so a global
* `serviceTier: "priority"` (for OpenAI/Anthropic) never silently incurs
* Fireworks priority costs — and never for `-fast` variants, whose Fast
* serving path is mutually exclusive with Priority. Every other provider
* uses the session `serviceTier` unchanged.
*/
#effectiveServiceTier(model: Model | undefined = this.model): ServiceTier | undefined {
if (model?.provider === "fireworks") {
return this.settings.get("providers.fireworksTier") === "priority" && !isFireworksFastModelId(model.id)
? "priority"
: undefined;
}
return this.serviceTier;
}
setServiceTier(serviceTier: ServiceTier | undefined): void {
@@ -10277,6 +10310,59 @@ export class AgentSession {
return false;
}
/**
* True when the current turn failed on a Fireworks Fast (`-fast`) model in a
* way that should degrade to the reliable base (Standard) model. Fast is a
* speed-optimized router with no SLA, so any *pre-content* failure — a
* transient overload/5xx or a hard "router/model not found / unsupported" —
* is worth retrying on the base id. Skips failures the base model shares:
* context overflow (compaction's job), usage limits and auth errors (same
* account/key), and turns that already emitted a tool call (replaying would
* duplicate work). Requires the base model to exist in the registry.
*/
#isFireworksFastFallbackEligible(message: AssistantMessage): boolean {
const model = this.model;
if (!model || model.provider !== "fireworks" || !isFireworksFastModelId(model.id)) return false;
if (message.stopReason !== "error" || !message.errorMessage) return false;
if (message.content.some(block => block.type === "toolCall")) return false;
if (isContextOverflow(message, model.contextWindow ?? 0)) return false;
const err = message.errorMessage;
if (isUsageLimitError(err)) return false;
if (
/\b(?:401|403|unauthorized|forbidden|authentication|auth[_ ]?unavailable|no auth available|(?:invalid|no)[_ ]?api[_ ]?key)\b/i.test(
err,
)
)
return false;
return this.#modelRegistry.find("fireworks", toFireworksBaseModelId(model.id)) !== undefined;
}
/**
* Switch the active model from a Fireworks Fast (`-fast`) variant to its base
* (Standard) id and stick there for the rest of the session — the auto
* fallback that makes Fast a safe default. Returns false when the current
* model is not a fast variant, the base id is missing, or it has no key.
*/
async #tryFireworksFastFallback(currentSelector: string): Promise<boolean> {
const model = this.model;
if (!model || model.provider !== "fireworks" || !isFireworksFastModelId(model.id)) return false;
const baseModel = this.#modelRegistry.find("fireworks", toFireworksBaseModelId(model.id));
if (!baseModel) return false;
const apiKey = await this.#modelRegistry.getApiKey(baseModel, this.sessionId);
if (!apiKey) return false;
const baseSelector = formatModelStringWithRouting(baseModel);
this.#setModelWithProviderSessionReset(baseModel);
this.sessionManager.appendModelChange(baseSelector, EPHEMERAL_MODEL_CHANGE_ROLE);
this.settings.getStorage()?.recordModelUsage(baseSelector);
await this.#emitSessionEvent({
type: "retry_fallback_applied",
from: currentSelector,
to: baseSelector,
role: "fireworks-fast",
});
return true;
}
async #maybeRestoreRetryFallbackPrimary(): Promise<void> {
if (!this.#activeRetryFallback) return;
if (this.#activeRetryFallback.pinned) return;
@@ -10379,7 +10465,7 @@ export class AgentSession {
*/
async #handleRetryableError(
message: AssistantMessage,
options?: { allowModelFallback?: boolean },
options?: { allowModelFallback?: boolean; fireworksFastFallback?: boolean },
): Promise<boolean> {
const retrySettings = this.settings.getGroup("retry");
if (!retrySettings.enabled) return false;
@@ -10477,6 +10563,13 @@ export class AgentSession {
}
switchedModel = await this.#tryRetryModelFallback(currentSelector, { pinFallback: classifierRefusal });
}
// Auto fallback from a Fireworks Fast variant to its base model. Independent
// of the role-fallback setting: it's intrinsic to the Fast contract (speed
// best-effort, degrade to Standard on failure) and triggers on hard router
// errors the generic retry classifier would otherwise reject.
if (!switchedModel && allowModelFallback && options?.fireworksFastFallback) {
switchedModel = await this.#tryFireworksFastFallback(currentSelector);
}
if (switchedModel) {
delayMs = 0;
} else if (usageLimitWaitMs === undefined && parsedRetryAfterMs && parsedRetryAfterMs > delayMs) {
@@ -11084,7 +11177,7 @@ export class AgentSession {
reasoning: toReasoningEffort(this.thinkingLevel),
disableReasoning: shouldDisableReasoning(this.thinkingLevel),
hideThinkingSummary: this.agent.hideThinkingSummary,
serviceTier: this.serviceTier,
serviceTier: this.#effectiveServiceTier(model),
signal: args.signal,
toolChoice: "none",
},