feat(catalog): supported fireworks fast serving path
- Added support for "Fast" serving-path variants for select Fireworks models. - Updated compatibility logic to route `-fast` suffixes to the appropriate router wire format. - Extended the model generation catalog to include these Fast variants with their respective pricing. - Updated AI types to allow the `priority` service tier for Fireworks providers.
This commit is contained in:
@@ -1182,6 +1182,10 @@ async function streamAssistantResponse(
|
||||
|
||||
const dynamicReasoning = config.getReasoning?.();
|
||||
const dynamicDisableReasoning = config.getDisableReasoning?.();
|
||||
// `getServiceTier` is authoritative when present (replaces the static tier
|
||||
// for both the wire request and telemetry), so callers can scope priority
|
||||
// per model without touching the shared session `serviceTier`.
|
||||
const effectiveServiceTier = config.getServiceTier ? config.getServiceTier(config.model) : config.serviceTier;
|
||||
const harmonyMitigationEnabled = isHarmonyLeakMitigationTarget(config.model);
|
||||
const harmonyAbortController = harmonyMitigationEnabled ? new AbortController() : undefined;
|
||||
const requestSignal = harmonyAbortController
|
||||
@@ -1229,7 +1233,7 @@ async function streamAssistantResponse(
|
||||
topP: config.topP,
|
||||
topK: config.topK,
|
||||
presencePenalty: config.presencePenalty,
|
||||
serviceTier: config.serviceTier,
|
||||
serviceTier: effectiveServiceTier,
|
||||
reasoningEffort: typeof effectiveReasoning === "string" ? effectiveReasoning : undefined,
|
||||
toolChoice: effectiveToolChoice,
|
||||
tools: llmContext.tools,
|
||||
@@ -1251,7 +1255,7 @@ async function streamAssistantResponse(
|
||||
const finishChat = async (message: AssistantMessage): Promise<void> => {
|
||||
await finishChatSpan(telemetry, chatSpan, message, {
|
||||
stepNumber: chatStepNumber,
|
||||
serviceTier: config.serviceTier,
|
||||
serviceTier: effectiveServiceTier,
|
||||
responseHeaders: capturedHeaders,
|
||||
baseUrl: config.model.baseUrl,
|
||||
});
|
||||
@@ -1267,6 +1271,7 @@ async function streamAssistantResponse(
|
||||
reasoning: effectiveReasoning,
|
||||
disableReasoning: effectiveDisableReasoning,
|
||||
temperature: effectiveTemperature,
|
||||
serviceTier: effectiveServiceTier,
|
||||
signal: finalRequestSignal,
|
||||
onResponse: captureOnResponse,
|
||||
});
|
||||
|
||||
@@ -196,6 +196,13 @@ export interface AgentOptions {
|
||||
presencePenalty?: number;
|
||||
repetitionPenalty?: number;
|
||||
serviceTier?: ServiceTier;
|
||||
/**
|
||||
* Per-call effective service-tier resolver. When set, it authoritatively
|
||||
* supplies the request's tier (replacing the static `serviceTier` and its
|
||||
* telemetry) per model — used to scope a provider/model into a priority
|
||||
* serving path without mutating the shared session `serviceTier`.
|
||||
*/
|
||||
serviceTierResolver?: (model: Model) => ServiceTier | undefined;
|
||||
/**
|
||||
* If true, request that the underlying provider omit reasoning/thinking summaries
|
||||
* from the response. The model still reasons internally; only the human-readable
|
||||
@@ -333,6 +340,7 @@ export class Agent {
|
||||
#presencePenalty?: number;
|
||||
#repetitionPenalty?: number;
|
||||
#serviceTier?: ServiceTier;
|
||||
#serviceTierResolver?: (model: Model) => ServiceTier | undefined;
|
||||
#hideThinkingSummary?: boolean;
|
||||
#maxRetryDelayMs?: number;
|
||||
#getToolContext?: (toolCall?: ToolCallContext) => AgentToolContext | undefined;
|
||||
@@ -403,6 +411,7 @@ export class Agent {
|
||||
this.#presencePenalty = opts.presencePenalty;
|
||||
this.#repetitionPenalty = opts.repetitionPenalty;
|
||||
this.#serviceTier = opts.serviceTier;
|
||||
this.#serviceTierResolver = opts.serviceTierResolver;
|
||||
this.#hideThinkingSummary = opts.hideThinkingSummary;
|
||||
this.#maxRetryDelayMs = opts.maxRetryDelayMs;
|
||||
this.getApiKey = opts.getApiKey;
|
||||
@@ -610,6 +619,14 @@ export class Agent {
|
||||
this.#serviceTier = value;
|
||||
}
|
||||
|
||||
get serviceTierResolver(): ((model: Model) => ServiceTier | undefined) | undefined {
|
||||
return this.#serviceTierResolver;
|
||||
}
|
||||
|
||||
set serviceTierResolver(value: ((model: Model) => ServiceTier | undefined) | undefined) {
|
||||
this.#serviceTierResolver = value;
|
||||
}
|
||||
|
||||
get hideThinkingSummary(): boolean | undefined {
|
||||
return this.#hideThinkingSummary;
|
||||
}
|
||||
@@ -1087,6 +1104,7 @@ export class Agent {
|
||||
getToolChoice,
|
||||
getReasoning: () => this.#state.thinkingLevel,
|
||||
getDisableReasoning: () => this.#state.disableReasoning,
|
||||
getServiceTier: this.#serviceTierResolver,
|
||||
getSteeringMessages: async () => {
|
||||
if (skipInitialSteeringPoll) {
|
||||
skipInitialSteeringPoll = false;
|
||||
|
||||
@@ -8,6 +8,7 @@ import type {
|
||||
ImageContent,
|
||||
Message,
|
||||
Model,
|
||||
ServiceTier,
|
||||
SimpleStreamOptions,
|
||||
Static,
|
||||
streamSimple,
|
||||
@@ -315,6 +316,17 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
|
||||
*/
|
||||
getDisableReasoning?: () => boolean | undefined;
|
||||
|
||||
/**
|
||||
* Per-call effective service-tier resolver. Unlike {@link getReasoning},
|
||||
* this is *authoritative*: when set, its return value (including
|
||||
* `undefined`) fully replaces the static `serviceTier` for the request and
|
||||
* its telemetry. The resolver receives the model being requested so the
|
||||
* caller can scope the tier per provider/model without mutating the shared
|
||||
* session `serviceTier` (e.g. opting a Fireworks model into the Priority
|
||||
* serving path while leaving the OpenAI/Anthropic tier untouched).
|
||||
*/
|
||||
getServiceTier?: (model: Model) => ServiceTier | undefined;
|
||||
|
||||
/**
|
||||
* Called after a tool call has been validated and is about to execute.
|
||||
*
|
||||
|
||||
@@ -141,18 +141,25 @@ export function resolveServiceTier(
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the (possibly scoped) tier should be sent as OpenAI's
|
||||
* `service_tier` request field for the given provider. Non-OpenAI
|
||||
* providers, unsupported tiers (`"auto"`, `"default"`), and scope
|
||||
* mismatches all return false.
|
||||
* True when the (possibly scoped) tier should be sent on the wire as the
|
||||
* `service_tier` request field for the given provider. OpenAI / OpenAI-Codex
|
||||
* accept `flex`/`scale`/`priority`; Fireworks Serverless realizes only its
|
||||
* Priority serving path (`service_tier: "priority"`) on the OpenAI-compatible
|
||||
* chat-completions endpoint. Unsupported tiers (`"auto"`, `"default"`), other
|
||||
* providers, and scope mismatches all return false.
|
||||
*/
|
||||
export function shouldSendServiceTier(
|
||||
serviceTier: ServiceTier | null | undefined,
|
||||
provider: Provider | undefined,
|
||||
): boolean {
|
||||
if (provider !== "openai" && provider !== "openai-codex") return false;
|
||||
const resolved = resolveServiceTier(serviceTier, provider);
|
||||
return resolved === "flex" || resolved === "scale" || resolved === "priority";
|
||||
if (provider === "openai" || provider === "openai-codex") {
|
||||
return resolved === "flex" || resolved === "scale" || resolved === "priority";
|
||||
}
|
||||
if (provider === "fireworks") {
|
||||
return resolved === "priority";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -29,6 +29,7 @@ import {
|
||||
import { PROVIDER_DESCRIPTORS } from "../src/provider-models/descriptors";
|
||||
import {
|
||||
ANTHROPIC_CURATED_FALLBACK_MODELS,
|
||||
buildFireworksFastSeed,
|
||||
buildXaiOAuthStaticSeed,
|
||||
clampFireworksKimiMaxTokens,
|
||||
clampKimiK27CodeMaxTokens,
|
||||
@@ -487,6 +488,11 @@ async function generateModels() {
|
||||
// Mythos 5). Deduped behind upstream entries; metadata is pinned in
|
||||
// applyAnthropicCatalogPolicy.
|
||||
allModels.push(...ANTHROPIC_CURATED_FALLBACK_MODELS);
|
||||
// Seed Fireworks "Fast" serving-path variants (`<id>-fast`). Fast routers are
|
||||
// not enumerated by the serverless control-plane list, so discovery never
|
||||
// surfaces them; the seed projects each base entry into a fast variant.
|
||||
// Deduped behind any identical previous-snapshot entry.
|
||||
allModels.push(...buildFireworksFastSeed());
|
||||
|
||||
const specialDiscoverySources = [
|
||||
{ label: "Antigravity", fetch: fetchAntigravityModels },
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
* complete alternate views. Request handlers read `model.compat` fields and
|
||||
* never detect, resolve, or allocate.
|
||||
*/
|
||||
import { isFireworksFastModelId } from "../fireworks-model-id";
|
||||
import { hostMatchesUrl, modelMatchesHost } from "../hosts";
|
||||
import {
|
||||
isAnthropicNamespacedModelId,
|
||||
@@ -276,8 +277,12 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
|
||||
? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS
|
||||
: undefined;
|
||||
|
||||
// Fireworks "Fast" variants (`<id>-fast`) are served from the router
|
||||
// namespace (`accounts/fireworks/routers/<id>-fast`), like Fire Pass, rather
|
||||
// than the `models/` namespace the rest of the `fireworks` provider uses.
|
||||
const isFireworksFastRouter = provider === "fireworks" && isFireworksFastModelId(spec.id);
|
||||
const wireModelIdMode: ResolvedOpenAISharedCompat["wireModelIdMode"] =
|
||||
provider === "firepass"
|
||||
provider === "firepass" || isFireworksFastRouter
|
||||
? "firepass"
|
||||
: provider === "fireworks"
|
||||
? "fireworks"
|
||||
|
||||
@@ -28,3 +28,23 @@ export function toFirepassWireModelId(modelId: string): string {
|
||||
const stripped = modelId.startsWith(FIREPASS_WIRE_PREFIX) ? modelId.slice(FIREPASS_WIRE_PREFIX.length) : modelId;
|
||||
return `${FIREPASS_WIRE_PREFIX}${stripped.replace(VERSION_DOT_PATTERN, "p")}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Public-id suffix marking a Fireworks "Fast" serving-path variant. Fast is a
|
||||
* higher-throughput route (100+ tok/s) exposed under a dedicated router id
|
||||
* (`accounts/fireworks/routers/<id>-fast`), not a separate model — same weights,
|
||||
* higher price, no Priority tier. We keep a friendly `<id>-fast` public id and
|
||||
* translate it to the router wire form at request time (compat
|
||||
* `wireModelIdMode: "firepass"`). See https://docs.fireworks.ai/serverless/serving-paths.
|
||||
*/
|
||||
export const FIREWORKS_FAST_SUFFIX = "-fast";
|
||||
|
||||
/** True for a Fireworks public model id that selects the Fast serving path. */
|
||||
export function isFireworksFastModelId(modelId: string): boolean {
|
||||
return modelId.endsWith(FIREWORKS_FAST_SUFFIX);
|
||||
}
|
||||
|
||||
/** Strip the Fast suffix to recover the base (Standard-tier) model id. */
|
||||
export function toFireworksBaseModelId(modelId: string): string {
|
||||
return modelId.endsWith(FIREWORKS_FAST_SUFFIX) ? modelId.slice(0, -FIREWORKS_FAST_SUFFIX.length) : modelId;
|
||||
}
|
||||
|
||||
@@ -14801,6 +14801,38 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"glm-5.1-fast": {
|
||||
"id": "glm-5.1-fast",
|
||||
"name": "GLM-5.1 Fast",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text"
|
||||
],
|
||||
"cost": {
|
||||
"input": 2.8,
|
||||
"output": 8.8,
|
||||
"cacheRead": 0.52,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 202752,
|
||||
"maxTokens": 131072,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
}
|
||||
},
|
||||
"glm-5.2": {
|
||||
"id": "glm-5.2",
|
||||
"name": "GLM-5.2",
|
||||
@@ -14947,6 +14979,39 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"kimi-k2.6-fast": {
|
||||
"id": "kimi-k2.6-fast",
|
||||
"name": "Kimi K2.6 Fast",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 2,
|
||||
"output": 8,
|
||||
"cacheRead": 0.3,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
}
|
||||
},
|
||||
"kimi-k2.7-code": {
|
||||
"id": "kimi-k2.7-code",
|
||||
"name": "Kimi K2.7 Code",
|
||||
@@ -14980,6 +15045,39 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"kimi-k2.7-code-fast": {
|
||||
"id": "kimi-k2.7-code-fast",
|
||||
"name": "Kimi K2.7 Code Fast",
|
||||
"api": "openai-completions",
|
||||
"provider": "fireworks",
|
||||
"baseUrl": "https://api.fireworks.ai/inference/v1",
|
||||
"reasoning": true,
|
||||
"input": [
|
||||
"text",
|
||||
"image"
|
||||
],
|
||||
"cost": {
|
||||
"input": 1.9,
|
||||
"output": 8,
|
||||
"cacheRead": 0.38,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 32768,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh"
|
||||
],
|
||||
"effortMap": {
|
||||
"minimal": "none"
|
||||
}
|
||||
}
|
||||
},
|
||||
"minimax-m2.5": {
|
||||
"id": "minimax-m2.5",
|
||||
"name": "MiniMax M2.5",
|
||||
|
||||
@@ -4,7 +4,7 @@ import {
|
||||
type OpenAICompatibleModelRecord,
|
||||
} from "../discovery/openai-compatible";
|
||||
import { Effort } from "../effort";
|
||||
import { toFireworksPublicModelId } from "../fireworks-model-id";
|
||||
import { FIREWORKS_FAST_SUFFIX, toFireworksPublicModelId } from "../fireworks-model-id";
|
||||
import { isGlmVisionModelId, isGrokReasoningEffortCapable, isReasoningGlmModelId } from "../identity/family";
|
||||
import type { ModelManagerOptions } from "../model-manager";
|
||||
import { getBundledModels } from "../models";
|
||||
@@ -1258,6 +1258,51 @@ export function clampKimiK27CodeMaxTokens(modelId: string, candidate: number | n
|
||||
return isKimiK27CodeModelId(modelId) ? Math.min(candidate, KIMI_K27_CODE_RECOMMENDED_MAX_TOKENS) : candidate;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fireworks Fast variants we surface. Each inherits the base model's
|
||||
* limits/modalities/thinking and overrides only the cost with the Standard-column
|
||||
* Fast prices from the Serverless pricing table; `cacheWrite` stays 0 (Fireworks
|
||||
* bills no cache-write). Derived from the bundled base entries so metadata stays
|
||||
* in lockstep, and the runtime auto-falls back to the base id on a failed fast
|
||||
* request. See https://docs.fireworks.ai/serverless/pricing.
|
||||
*/
|
||||
const FIREWORKS_FAST_VARIANT_SPECS: ReadonlyArray<{
|
||||
base: string;
|
||||
name: string;
|
||||
cost: { input: number; output: number; cacheRead: number };
|
||||
}> = [
|
||||
{ base: "kimi-k2.7-code", name: "Kimi K2.7 Code Fast", cost: { input: 1.9, output: 8, cacheRead: 0.38 } },
|
||||
{ base: "kimi-k2.6", name: "Kimi K2.6 Fast", cost: { input: 2, output: 8, cacheRead: 0.3 } },
|
||||
{ base: "glm-5.1", name: "GLM-5.1 Fast", cost: { input: 2.8, output: 8.8, cacheRead: 0.52 } },
|
||||
];
|
||||
|
||||
/**
|
||||
* Build the Fireworks Fast seed by projecting each base bundled spec into a
|
||||
* `<id>-fast` variant. Pushed into the generated catalog (Fast routers never
|
||||
* appear in the serverless control-plane list, so discovery cannot surface
|
||||
* them) and deduped behind any identical previous-snapshot entry.
|
||||
*/
|
||||
export function buildFireworksFastSeed(): ModelSpec<"openai-completions">[] {
|
||||
const bundled = createBundledReferenceMap<"openai-completions">("fireworks");
|
||||
const seeds: ModelSpec<"openai-completions">[] = [];
|
||||
for (const variant of FIREWORKS_FAST_VARIANT_SPECS) {
|
||||
const base = bundled.get(variant.base);
|
||||
if (!base) continue;
|
||||
seeds.push({
|
||||
...base,
|
||||
id: `${variant.base}${FIREWORKS_FAST_SUFFIX}`,
|
||||
name: variant.name,
|
||||
cost: {
|
||||
input: variant.cost.input,
|
||||
output: variant.cost.output,
|
||||
cacheRead: variant.cost.cacheRead,
|
||||
cacheWrite: 0,
|
||||
},
|
||||
});
|
||||
}
|
||||
return seeds;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fireworks DeepSeek V4 accepts effort via `reasoning_effort` but rejects the
|
||||
* DeepSeek-native binary `thinking` toggle when both are present.
|
||||
|
||||
@@ -134,7 +134,7 @@ export const TAB_GROUPS: Record<SettingTab, readonly string[]> = {
|
||||
"Developer",
|
||||
],
|
||||
tasks: ["Modes", "Subagents", "Isolation", "Commands & Skills"],
|
||||
providers: ["Services", "Tiny Model", "Protocol", "Privacy"],
|
||||
providers: ["Services", "Fireworks", "Tiny Model", "Protocol", "Privacy"],
|
||||
};
|
||||
|
||||
/** Status line segment identifiers */
|
||||
@@ -4042,6 +4042,26 @@ export const SETTINGS_SCHEMA = {
|
||||
],
|
||||
},
|
||||
},
|
||||
"providers.fireworksTier": {
|
||||
type: "enum",
|
||||
values: ["standard", "priority"] as const,
|
||||
default: "standard",
|
||||
ui: {
|
||||
tab: "providers",
|
||||
group: "Fireworks",
|
||||
label: "Fireworks Tier",
|
||||
description:
|
||||
"Serving path for Fireworks requests. Priority sends `service_tier: \"priority\"` for higher reliability during peak traffic at a higher price; Standard omits it. Fast (`-fast`) models ignore this — Fast is its own serving path.",
|
||||
options: [
|
||||
{ value: "standard", label: "Standard", description: "Default serving path (no service_tier)" },
|
||||
{
|
||||
value: "priority",
|
||||
label: "Priority",
|
||||
description: "Priority serving path: higher reliability, premium per-token pricing",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
"providers.tts": {
|
||||
type: "enum",
|
||||
values: ["auto", "local", "xai"] as const,
|
||||
|
||||
@@ -105,6 +105,7 @@ import {
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { stripToolDescriptions } from "@oh-my-pi/pi-ai/utils/schema";
|
||||
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
|
||||
import { isFireworksFastModelId, toFireworksBaseModelId } from "@oh-my-pi/pi-catalog/fireworks-model-id";
|
||||
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
|
||||
import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives";
|
||||
@@ -1537,6 +1538,10 @@ export class AgentSession {
|
||||
this.#customCommands = config.customCommands ?? [];
|
||||
this.#skillsSettings = config.skillsSettings;
|
||||
this.#modelRegistry = config.modelRegistry;
|
||||
// Resolve the wire service-tier per request so the Fireworks Priority
|
||||
// toggle scopes priority to Fireworks alone, without mutating the shared
|
||||
// session `serviceTier` that drives `/fast` and OpenAI/Anthropic priority.
|
||||
this.agent.serviceTierResolver = model => this.#effectiveServiceTier(model);
|
||||
this.#advisorReadOnlyTools = config.advisorReadOnlyTools;
|
||||
this.#advisorWatchdogPrompt = config.advisorWatchdogPrompt;
|
||||
this.#pruneToolDescriptions = config.pruneToolDescriptions === true;
|
||||
@@ -2790,6 +2795,16 @@ export class AgentSession {
|
||||
await emitAgentEndNotification();
|
||||
return;
|
||||
}
|
||||
// Fireworks Fast variants degrade to their base model on a failed turn —
|
||||
// including hard router errors the generic retry classifier rejects — so
|
||||
// run this gate before the standard retryability check.
|
||||
if (this.#isFireworksFastFallbackEligible(msg)) {
|
||||
const didRetry = await this.#handleRetryableError(msg, { fireworksFastFallback: true });
|
||||
if (didRetry) {
|
||||
await emitAgentEndNotification();
|
||||
return;
|
||||
}
|
||||
}
|
||||
// Check for retryable errors first (overloaded, rate limit, server errors)
|
||||
if (this.#isRetryableError(msg)) {
|
||||
const didRetry = await this.#handleRetryableError(msg);
|
||||
@@ -7279,7 +7294,25 @@ export class AgentSession {
|
||||
* no model is selected.
|
||||
*/
|
||||
isFastModeActive(): boolean {
|
||||
return resolveServiceTier(this.serviceTier, this.model?.provider) === "priority";
|
||||
return resolveServiceTier(this.#effectiveServiceTier(), this.model?.provider) === "priority";
|
||||
}
|
||||
|
||||
/**
|
||||
* Effective wire service-tier for a request to `model`. Fireworks models
|
||||
* take the Priority serving path only when the Providers › Fireworks Tier
|
||||
* setting is `"priority"` — that toggle is the sole opt-in, so a global
|
||||
* `serviceTier: "priority"` (for OpenAI/Anthropic) never silently incurs
|
||||
* Fireworks priority costs — and never for `-fast` variants, whose Fast
|
||||
* serving path is mutually exclusive with Priority. Every other provider
|
||||
* uses the session `serviceTier` unchanged.
|
||||
*/
|
||||
#effectiveServiceTier(model: Model | undefined = this.model): ServiceTier | undefined {
|
||||
if (model?.provider === "fireworks") {
|
||||
return this.settings.get("providers.fireworksTier") === "priority" && !isFireworksFastModelId(model.id)
|
||||
? "priority"
|
||||
: undefined;
|
||||
}
|
||||
return this.serviceTier;
|
||||
}
|
||||
|
||||
setServiceTier(serviceTier: ServiceTier | undefined): void {
|
||||
@@ -10277,6 +10310,59 @@ export class AgentSession {
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the current turn failed on a Fireworks Fast (`-fast`) model in a
|
||||
* way that should degrade to the reliable base (Standard) model. Fast is a
|
||||
* speed-optimized router with no SLA, so any *pre-content* failure — a
|
||||
* transient overload/5xx or a hard "router/model not found / unsupported" —
|
||||
* is worth retrying on the base id. Skips failures the base model shares:
|
||||
* context overflow (compaction's job), usage limits and auth errors (same
|
||||
* account/key), and turns that already emitted a tool call (replaying would
|
||||
* duplicate work). Requires the base model to exist in the registry.
|
||||
*/
|
||||
#isFireworksFastFallbackEligible(message: AssistantMessage): boolean {
|
||||
const model = this.model;
|
||||
if (!model || model.provider !== "fireworks" || !isFireworksFastModelId(model.id)) return false;
|
||||
if (message.stopReason !== "error" || !message.errorMessage) return false;
|
||||
if (message.content.some(block => block.type === "toolCall")) return false;
|
||||
if (isContextOverflow(message, model.contextWindow ?? 0)) return false;
|
||||
const err = message.errorMessage;
|
||||
if (isUsageLimitError(err)) return false;
|
||||
if (
|
||||
/\b(?:401|403|unauthorized|forbidden|authentication|auth[_ ]?unavailable|no auth available|(?:invalid|no)[_ ]?api[_ ]?key)\b/i.test(
|
||||
err,
|
||||
)
|
||||
)
|
||||
return false;
|
||||
return this.#modelRegistry.find("fireworks", toFireworksBaseModelId(model.id)) !== undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Switch the active model from a Fireworks Fast (`-fast`) variant to its base
|
||||
* (Standard) id and stick there for the rest of the session — the auto
|
||||
* fallback that makes Fast a safe default. Returns false when the current
|
||||
* model is not a fast variant, the base id is missing, or it has no key.
|
||||
*/
|
||||
async #tryFireworksFastFallback(currentSelector: string): Promise<boolean> {
|
||||
const model = this.model;
|
||||
if (!model || model.provider !== "fireworks" || !isFireworksFastModelId(model.id)) return false;
|
||||
const baseModel = this.#modelRegistry.find("fireworks", toFireworksBaseModelId(model.id));
|
||||
if (!baseModel) return false;
|
||||
const apiKey = await this.#modelRegistry.getApiKey(baseModel, this.sessionId);
|
||||
if (!apiKey) return false;
|
||||
const baseSelector = formatModelStringWithRouting(baseModel);
|
||||
this.#setModelWithProviderSessionReset(baseModel);
|
||||
this.sessionManager.appendModelChange(baseSelector, EPHEMERAL_MODEL_CHANGE_ROLE);
|
||||
this.settings.getStorage()?.recordModelUsage(baseSelector);
|
||||
await this.#emitSessionEvent({
|
||||
type: "retry_fallback_applied",
|
||||
from: currentSelector,
|
||||
to: baseSelector,
|
||||
role: "fireworks-fast",
|
||||
});
|
||||
return true;
|
||||
}
|
||||
|
||||
async #maybeRestoreRetryFallbackPrimary(): Promise<void> {
|
||||
if (!this.#activeRetryFallback) return;
|
||||
if (this.#activeRetryFallback.pinned) return;
|
||||
@@ -10379,7 +10465,7 @@ export class AgentSession {
|
||||
*/
|
||||
async #handleRetryableError(
|
||||
message: AssistantMessage,
|
||||
options?: { allowModelFallback?: boolean },
|
||||
options?: { allowModelFallback?: boolean; fireworksFastFallback?: boolean },
|
||||
): Promise<boolean> {
|
||||
const retrySettings = this.settings.getGroup("retry");
|
||||
if (!retrySettings.enabled) return false;
|
||||
@@ -10477,6 +10563,13 @@ export class AgentSession {
|
||||
}
|
||||
switchedModel = await this.#tryRetryModelFallback(currentSelector, { pinFallback: classifierRefusal });
|
||||
}
|
||||
// Auto fallback from a Fireworks Fast variant to its base model. Independent
|
||||
// of the role-fallback setting: it's intrinsic to the Fast contract (speed
|
||||
// best-effort, degrade to Standard on failure) and triggers on hard router
|
||||
// errors the generic retry classifier would otherwise reject.
|
||||
if (!switchedModel && allowModelFallback && options?.fireworksFastFallback) {
|
||||
switchedModel = await this.#tryFireworksFastFallback(currentSelector);
|
||||
}
|
||||
if (switchedModel) {
|
||||
delayMs = 0;
|
||||
} else if (usageLimitWaitMs === undefined && parsedRetryAfterMs && parsedRetryAfterMs > delayMs) {
|
||||
@@ -11084,7 +11177,7 @@ export class AgentSession {
|
||||
reasoning: toReasoningEffort(this.thinkingLevel),
|
||||
disableReasoning: shouldDisableReasoning(this.thinkingLevel),
|
||||
hideThinkingSummary: this.agent.hideThinkingSummary,
|
||||
serviceTier: this.serviceTier,
|
||||
serviceTier: this.#effectiveServiceTier(model),
|
||||
signal: args.signal,
|
||||
toolChoice: "none",
|
||||
},
|
||||
|
||||
Reference in New Issue
Block a user