feat: migrated service tier settings to a per-model-family architecture
- Migrated global service tier settings to a per-model-family architecture (OpenAI, Anthropic, Google). - Implemented `ServiceTierByFamily` mapping to allow independent configuration and resolution per provider. - Added automatic migration logic for legacy service tier and fast-mode application settings. - Updated telemetry, session management, and task execution to support provider-specific tier resolution.
This commit is contained in:
@@ -179,7 +179,7 @@ Lifecycle/state transition:
|
||||
15. restore model via `getRestorableSessionModels(sessionContext.models, lastModelChangeRole)` — tries the recorded models in fallback order and uses the first one present in the model registry
|
||||
16. restore thinking level and service tier:
|
||||
- thinking uses persisted `thinking_level_change`, otherwise the configured default clamped to model capability
|
||||
- service tier uses persisted `service_tier_change`, otherwise the configured `serviceTier` setting (`"none"` becomes unset)
|
||||
- service tier uses persisted `service_tier_change`, otherwise the configured per-family `tier.openai`/`tier.anthropic`/`tier.google` settings (`"none"` becomes unset)
|
||||
17. reconnect agent listeners, run the registered session-switch reconciler if any (interactive mode re-enters persisted modes; errors logged, not fatal), and return `true`
|
||||
|
||||
## UI state rebuild after interactive switch
|
||||
|
||||
+2
-2
@@ -177,11 +177,11 @@ Stores an `AgentMessage` directly.
|
||||
"id": "c1d2e3f4",
|
||||
"parentId": "b1c2d3e4",
|
||||
"timestamp": "2026-02-16T10:21:45.000Z",
|
||||
"serviceTier": "flex"
|
||||
"serviceTier": { "openai": "priority", "google": "flex" }
|
||||
}
|
||||
```
|
||||
|
||||
`serviceTier` can also be `null`.
|
||||
`serviceTier` is a per-family map keyed by `openai`/`anthropic`/`google` (each value `auto`/`default`/`flex`/`scale`/`priority`), or `null` when no tier is active. Legacy entries that stored a single string (`"flex"`, `"openai-only"`, `"claude-only"`, …) are normalized to this map on read.
|
||||
|
||||
### `thinking_level_change`
|
||||
|
||||
|
||||
+5
-1
@@ -366,7 +366,11 @@ A value of `-1` means "use the provider/model default" — `omp` does not send t
|
||||
| `minP` | number | `-1` | Minimum-probability cutoff. |
|
||||
| `presencePenalty` | number | `-1` | Presence penalty. |
|
||||
| `repetitionPenalty` | number | `-1` | Repetition penalty. |
|
||||
| `serviceTier` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`, `openai-only`, `claude-only`. |
|
||||
| `tier.openai` | enum | `none` | `none`, `auto`, `default`, `flex`, `scale`, `priority`. Sent as `service_tier` for OpenAI / OpenAI-Codex and OpenAI-family OpenRouter models. |
|
||||
| `tier.anthropic` | enum | `none` | `none`, `priority`. `priority` realizes fast mode on supported direct Claude models (ignored on Bedrock/Vertex and via OpenRouter). |
|
||||
| `tier.google` | enum | `none` | `none`, `flex`, `priority`. Gemini API sends it in the body; Vertex sends `priority` via header (`flex` is a no-op on Vertex). |
|
||||
| `tier.subagent` | enum | `inherit` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the spawned model's family; `inherit` tracks the main agent. |
|
||||
| `tier.advisor` | enum | `none` | `inherit`, `none`, `auto`, `default`, `flex`, `scale`, `priority`. Applied to the advisor model's family. |
|
||||
| `personality` | enum | `default` | `default`, `friendly`, `pragmatic`, `none`. |
|
||||
|
||||
### Retry and fallback
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import type { ImageContent, MessageAttribution, ServiceTier, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import type { AgentMessage } from "../types";
|
||||
|
||||
export interface SessionEntryBase {
|
||||
@@ -28,7 +28,7 @@ export interface ModelChangeEntry extends SessionEntryBase {
|
||||
|
||||
export interface ServiceTierChangeEntry extends SessionEntryBase {
|
||||
type: "service_tier_change";
|
||||
serviceTier: ServiceTier | null;
|
||||
serviceTier: ServiceTierByFamily | null;
|
||||
}
|
||||
|
||||
export interface CompactionEntry<T = unknown> extends SessionEntryBase {
|
||||
|
||||
@@ -30,7 +30,6 @@ import {
|
||||
completeSimple,
|
||||
type Message,
|
||||
type Model,
|
||||
resolveServiceTier,
|
||||
type ServiceTier,
|
||||
type SimpleStreamOptions,
|
||||
type StopReason,
|
||||
@@ -752,8 +751,7 @@ function buildChatRequestAttributes(stepNumber: number, request: ChatRequestSnap
|
||||
attrs[GenAIAttr.RequestStopSequences] = [...request.stopSequences];
|
||||
}
|
||||
if (request.serviceTier && shouldSendServiceTier(request.serviceTier, provider)) {
|
||||
const resolved = resolveServiceTier(request.serviceTier, provider);
|
||||
if (resolved) attrs[OpenAIAttr.RequestServiceTier] = resolved;
|
||||
attrs[OpenAIAttr.RequestServiceTier] = request.serviceTier;
|
||||
}
|
||||
if (request.reasoningEffort) attrs[PiGenAIAttr.RequestReasoningEffort] = request.reasoningEffort;
|
||||
const toolChoice = serializeToolChoice(request.toolChoice);
|
||||
|
||||
@@ -2,6 +2,23 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added service tier support for Google Gemini and Vertex AI
|
||||
- Introduced `ServiceTierByFamily` to allow model-specific service tier configurations
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated service tier logic to avoid global scopes in favor of per-provider configurations
|
||||
- Refactored priority request billing to better align with specific provider capabilities
|
||||
- Updated internal `coerceServiceTierByFamily` helper to facilitate migration from legacy settings
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed safety setting application for Google Vertex AI models
|
||||
- Ensured Gemini service tier is correctly passed through to the API
|
||||
- Corrected priority request accounting for supported providers
|
||||
|
||||
## [16.2.6] - 2026-06-29
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -43,7 +43,6 @@ import type {
|
||||
ToolResultMessage,
|
||||
Usage,
|
||||
} from "../types";
|
||||
import { resolveServiceTier } from "../types";
|
||||
import { isRecord, normalizeSystemPrompts, normalizeToolCallId, resolveCacheRetention } from "../utils";
|
||||
import { createAbortSourceTracker } from "../utils/abort";
|
||||
import {
|
||||
@@ -1595,7 +1594,7 @@ const streamAnthropicOnce = (
|
||||
isOAuthToken = false;
|
||||
} else {
|
||||
const extraBetas = normalizeExtraBetas(options?.betas);
|
||||
const wantsAnthropicPriority = resolveServiceTier(options?.serviceTier, model.provider) === "priority";
|
||||
const wantsAnthropicPriority = model.provider === "anthropic" && options?.serviceTier === "priority";
|
||||
// Skip the fast-mode beta when this session already learned the
|
||||
// endpoint+model rejects fast mode; `speed` is dropped from the params
|
||||
// too (dropFastMode), so the request stays a faithful non-fast request.
|
||||
@@ -2191,7 +2190,8 @@ const streamAnthropicOnce = (
|
||||
}
|
||||
if (
|
||||
!dropFastMode &&
|
||||
resolveServiceTier(options?.serviceTier, model.provider) === "priority" &&
|
||||
model.provider === "anthropic" &&
|
||||
options?.serviceTier === "priority" &&
|
||||
firstTokenTime === undefined &&
|
||||
AIError.isFastModeUnsupported(streamFailure)
|
||||
) {
|
||||
@@ -2259,7 +2259,7 @@ const streamAnthropicOnce = (
|
||||
}
|
||||
output.duration = performance.now() - startTime;
|
||||
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
||||
if (dropFastMode && resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
|
||||
if (dropFastMode && model.provider === "anthropic" && options?.serviceTier === "priority") {
|
||||
output.disabledFeatures = [...(output.disabledFeatures ?? []), "priority"];
|
||||
}
|
||||
stream.push({ type: "done", reason: output.stopReason, message: output });
|
||||
@@ -2996,7 +2996,7 @@ function buildParams(
|
||||
seqs.length > ANTHROPIC_STOP_SEQUENCES_MAX ? seqs.slice(0, ANTHROPIC_STOP_SEQUENCES_MAX) : seqs;
|
||||
}
|
||||
|
||||
if (resolveServiceTier(options?.serviceTier, model.provider) === "priority") {
|
||||
if (model.provider === "anthropic" && options?.serviceTier === "priority") {
|
||||
params.speed = "fast";
|
||||
}
|
||||
|
||||
|
||||
@@ -14,6 +14,7 @@ import type {
|
||||
FetchImpl,
|
||||
ImageContent,
|
||||
Model,
|
||||
ServiceTier,
|
||||
StopReason,
|
||||
StreamOptions,
|
||||
TextContent,
|
||||
@@ -21,6 +22,7 @@ import type {
|
||||
Tool,
|
||||
ToolCall,
|
||||
} from "../types";
|
||||
import { shouldSendServiceTier } from "../types";
|
||||
import { normalizeSystemPrompts } from "../utils";
|
||||
import { AssistantMessageEventStream } from "../utils/event-stream";
|
||||
import type { RawHttpRequestDump } from "../utils/http-inspector";
|
||||
@@ -73,6 +75,8 @@ export interface GoogleSharedStreamOptions extends StreamOptions {
|
||||
budgetTokens?: number;
|
||||
level?: GoogleThinkingLevel;
|
||||
};
|
||||
/** Gemini/Vertex serving tier (`flex`/`priority`); other values are omitted. */
|
||||
serviceTier?: ServiceTier;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -791,6 +795,14 @@ export function buildGoogleGenerateContentParams<T extends "google-generative-ai
|
||||
...(context.tools && context.tools.length > 0 && { tools: convertTools(context.tools, model) }),
|
||||
};
|
||||
|
||||
// Gemini API (google-generative-ai) reads the tier from the request body;
|
||||
// Vertex AI ignores a body field and requires the
|
||||
// `X-Vertex-AI-LLM-Shared-Request-Type` header instead (added in
|
||||
// streamGoogleVertex), so only emit the body field for the direct API.
|
||||
if (model.provider === "google" && shouldSendServiceTier(options.serviceTier, model.provider)) {
|
||||
config.serviceTier = options.serviceTier;
|
||||
}
|
||||
|
||||
if (context.tools && context.tools.length > 0 && options.toolChoice) {
|
||||
const choice = options.toolChoice;
|
||||
if (typeof choice === "string") {
|
||||
@@ -1007,6 +1019,7 @@ function paramsToWireBody(params: GenerateContentParameters): Record<string, unk
|
||||
if (config.toolConfig !== undefined) body.toolConfig = config.toolConfig;
|
||||
if (config.safetySettings !== undefined) body.safetySettings = config.safetySettings;
|
||||
if (config.cachedContent !== undefined) body.cachedContent = config.cachedContent;
|
||||
if (config.serviceTier !== undefined) body.serviceTier = config.serviceTier;
|
||||
|
||||
const gen: Record<string, unknown> = {};
|
||||
if (config.temperature !== undefined) gen.temperature = config.temperature;
|
||||
|
||||
@@ -10,6 +10,8 @@
|
||||
* - The Cloud Code Assist endpoint used by `google-gemini-cli.ts`
|
||||
*/
|
||||
|
||||
import type { ServiceTier } from "../types";
|
||||
|
||||
/** Mirror of `@google/genai`'s `FinishReason` string enum. */
|
||||
export type FinishReason =
|
||||
| "FINISH_REASON_UNSPECIFIED"
|
||||
@@ -131,6 +133,11 @@ export interface GenerateContentConfig {
|
||||
safetySettings?: Array<Record<string, unknown>>;
|
||||
cachedContent?: string;
|
||||
thinkingConfig?: ThinkingConfig;
|
||||
/**
|
||||
* Gemini/Vertex serving tier. Serialized to the request body root as
|
||||
* `serviceTier` (camelCase) by the transformer in `google-shared.ts`.
|
||||
*/
|
||||
serviceTier?: ServiceTier;
|
||||
abortSignal?: AbortSignal;
|
||||
}
|
||||
|
||||
|
||||
@@ -30,17 +30,47 @@ export const streamGoogleVertex: StreamFunction<"google-vertex"> = (
|
||||
prepare: async (): Promise<GoogleGenAIRequestPlan> => {
|
||||
const apiKey = resolveApiKey(options);
|
||||
const params = buildGoogleGenerateContentParams(model, context, options ?? {});
|
||||
params.config ||= {};
|
||||
if (!params.config.safetySettings) {
|
||||
params.config.safetySettings = [
|
||||
{
|
||||
category: "HARM_CATEGORY_HATE_SPEECH",
|
||||
threshold: "OFF",
|
||||
},
|
||||
{
|
||||
category: "HARM_CATEGORY_DANGEROUS_CONTENT",
|
||||
threshold: "OFF",
|
||||
},
|
||||
{
|
||||
category: "HARM_CATEGORY_SEXUALLY_EXPLICIT",
|
||||
threshold: "OFF",
|
||||
},
|
||||
{
|
||||
category: "HARM_CATEGORY_HARASSMENT",
|
||||
threshold: "OFF",
|
||||
},
|
||||
];
|
||||
}
|
||||
const baseHeaders: Record<string, string> = {
|
||||
...(model.headers ?? {}),
|
||||
...(options?.headers ?? {}),
|
||||
};
|
||||
// Vertex AI ignores a `serviceTier` request-body field (unlike the direct
|
||||
// Gemini API); priority must travel as a request header. Only `priority`
|
||||
// has a documented Vertex request control — `flex` has none, so it's a no-op.
|
||||
if (options?.serviceTier === "priority") {
|
||||
baseHeaders["X-Vertex-AI-LLM-Shared-Request-Type"] = "priority";
|
||||
}
|
||||
|
||||
if (apiKey) {
|
||||
const url = `https://aiplatform.googleapis.com/${API_VERSION}/publishers/google/models/${model.id}:streamGenerateContent?alt=sse`;
|
||||
return {
|
||||
params,
|
||||
url,
|
||||
headers: { ...baseHeaders, "x-goog-api-key": apiKey },
|
||||
headers: {
|
||||
...baseHeaders,
|
||||
"x-goog-api-key": apiKey,
|
||||
},
|
||||
fetch: options?.fetch,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -13,7 +13,7 @@ import type {
|
||||
Context,
|
||||
ImageContent,
|
||||
Message,
|
||||
ResolvedServiceTier,
|
||||
ServiceTier,
|
||||
StopReason,
|
||||
TextContent,
|
||||
Tool,
|
||||
@@ -38,7 +38,7 @@ function isReasoningEffort(value: unknown): value is ReasoningEffort {
|
||||
return value === "minimal" || value === "low" || value === "medium" || value === "high" || value === "xhigh";
|
||||
}
|
||||
|
||||
function isServiceTier(value: unknown): value is ResolvedServiceTier {
|
||||
function isServiceTier(value: unknown): value is ServiceTier {
|
||||
return value === "auto" || value === "default" || value === "flex" || value === "scale" || value === "priority";
|
||||
}
|
||||
|
||||
|
||||
@@ -39,8 +39,6 @@ import {
|
||||
type Model,
|
||||
OPENAI_MAX_OUTPUT_TOKENS,
|
||||
type Provider,
|
||||
type ResolvedServiceTier,
|
||||
resolveServiceTier,
|
||||
type ServiceTier,
|
||||
type StopReason,
|
||||
type StreamOptions,
|
||||
@@ -269,14 +267,13 @@ export function resolveOpenAIRequestSetup(
|
||||
}
|
||||
|
||||
export function applyOpenAIServiceTier(
|
||||
params: { service_tier?: ResolvedServiceTier | "auto" | "default" | null | undefined },
|
||||
params: { service_tier?: ServiceTier | null | undefined },
|
||||
serviceTier: ServiceTier | null | undefined,
|
||||
provider: Provider | undefined,
|
||||
): void {
|
||||
if (!shouldSendServiceTier(serviceTier, provider)) return;
|
||||
const resolved = resolveServiceTier(serviceTier, provider);
|
||||
if (resolved === "flex" || resolved === "scale" || resolved === "priority") {
|
||||
params.service_tier = resolved;
|
||||
if (serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority") {
|
||||
params.service_tier = serviceTier;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -315,10 +312,7 @@ export function applyOpenAIResponsesServiceTierCost(
|
||||
// The response echo is authoritative when present (OpenAI may downgrade a
|
||||
// requested priority/flex turn to default under load); only fall back to the
|
||||
// requested tier when the response omits the echo entirely.
|
||||
const served =
|
||||
typeof responseServiceTier === "string"
|
||||
? responseServiceTier
|
||||
: resolveServiceTier(requestServiceTier, model.provider);
|
||||
const served = typeof responseServiceTier === "string" ? responseServiceTier : (requestServiceTier ?? undefined);
|
||||
const multiplier = getOpenAIResponsesServiceTierCostMultiplier(served);
|
||||
if (multiplier === 1) return;
|
||||
usage.cost.input *= multiplier;
|
||||
@@ -623,7 +617,7 @@ export type OpenAICompletionsParams = Omit<ChatCompletionCreateParamsStreaming,
|
||||
chat_template_kwargs?: { enable_thinking?: boolean; preserve_thinking?: boolean };
|
||||
reasoning?: { effort?: string } | { enabled: false };
|
||||
reasoning_effort?: string | null;
|
||||
service_tier?: ResolvedServiceTier;
|
||||
service_tier?: ServiceTier;
|
||||
tool_stream?: boolean;
|
||||
provider?: OpenAICompat["openRouterRouting"];
|
||||
providerOptions?: { gateway?: { only?: string[]; order?: string[] } };
|
||||
|
||||
@@ -1565,6 +1565,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
if (!reasoning || !model.reasoning) {
|
||||
return castApi<"google-generative-ai">({
|
||||
...base,
|
||||
serviceTier: options?.serviceTier,
|
||||
thinking: { enabled: false },
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
@@ -1578,6 +1579,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
if (googleModel.thinking?.mode === "google-level") {
|
||||
return castApi<"google-generative-ai">({
|
||||
...base,
|
||||
serviceTier: options?.serviceTier,
|
||||
thinking: {
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
@@ -1661,6 +1663,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
if (!reasoning || !model.reasoning) {
|
||||
return castApi<"google-vertex">({
|
||||
...base,
|
||||
serviceTier: options?.serviceTier,
|
||||
thinking: { enabled: false },
|
||||
toolChoice: mapGoogleToolChoice(options?.toolChoice),
|
||||
});
|
||||
@@ -1673,6 +1676,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
if (geminiModel.thinking?.mode === "google-level") {
|
||||
return castApi<"google-vertex">({
|
||||
...base,
|
||||
serviceTier: options?.serviceTier,
|
||||
thinking: {
|
||||
enabled: true,
|
||||
level: mapEffortToGoogleThinkingLevel(effort),
|
||||
@@ -1683,6 +1687,7 @@ function mapOptionsForApi<TApi extends Api>(
|
||||
|
||||
return castApi<"google-vertex">({
|
||||
...base,
|
||||
serviceTier: options?.serviceTier,
|
||||
thinking: {
|
||||
enabled: true,
|
||||
budgetTokens: getGoogleBudget(geminiModel, effort, options?.thinkingBudgets),
|
||||
|
||||
+148
-51
@@ -105,84 +105,181 @@ export type ToolChoice =
|
||||
export type CacheRetention = "none" | "short" | "long";
|
||||
|
||||
/**
|
||||
* Service tier hint for processing priority / cost control.
|
||||
* Service tier hint for processing priority / cost control. These are the
|
||||
* values providers consume on the wire:
|
||||
*
|
||||
* The unscoped values (`"auto"`, `"default"`, `"flex"`, `"scale"`,
|
||||
* `"priority"`) are passed through to providers that understand them
|
||||
* (OpenAI's `service_tier` field directly; Anthropic translates
|
||||
* `"priority"` into `speed: "fast"` on supported Opus models).
|
||||
* - OpenAI / OpenAI-Codex: sent verbatim as the `service_tier` field
|
||||
* (`flex`/`scale`/`priority`).
|
||||
* - Google (Gemini API + Vertex AI): sent as the top-level `serviceTier`
|
||||
* field (`flex`/`priority`).
|
||||
* - OpenRouter: passed through as `service_tier`; OpenRouter realizes it for
|
||||
* the OpenAI- and Google-family upstreams it supports and ignores it
|
||||
* otherwise.
|
||||
* - Direct Anthropic: `"priority"` is translated into `speed: "fast"` plus the
|
||||
* fast-mode beta on supported Opus models. Other tiers are ignored.
|
||||
*
|
||||
* The scoped values target a specific provider family and behave as the
|
||||
* unscoped value on the matching provider, or `undefined` everywhere else.
|
||||
* They let users opt into priority on one family without paying premium
|
||||
* costs on the other when switching models mid-session.
|
||||
*
|
||||
* - `"openai-only"` → `"priority"` on `openai` and `openai-codex`; ignored elsewhere.
|
||||
* - `"claude-only"` → `"priority"` on direct `anthropic` (not Bedrock/Vertex Claude).
|
||||
* Per-family scoping is expressed by {@link ServiceTierByFamily}, not by
|
||||
* scoped sentinel values — see {@link serviceTierFamily}.
|
||||
*/
|
||||
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority" | "openai-only" | "claude-only";
|
||||
export type ServiceTier = "auto" | "default" | "flex" | "scale" | "priority";
|
||||
|
||||
/** Resolved tier — one of the values that providers actually consume on the wire. */
|
||||
export type ResolvedServiceTier = Exclude<ServiceTier, "openai-only" | "claude-only">;
|
||||
/** Provider families that expose an independent service-tier knob. */
|
||||
export type ServiceTierFamily = "openai" | "anthropic" | "google";
|
||||
|
||||
/**
|
||||
* Resolves a possibly scoped `ServiceTier` to the effective tier for the
|
||||
* given provider. Scoped values match their target family and otherwise
|
||||
* collapse to `undefined`; unscoped values pass through unchanged.
|
||||
* Per-family service-tier selection. A request consults only the entry for the
|
||||
* family its model belongs to (see {@link resolveModelServiceTier}), so a user
|
||||
* can opt one family into priority without affecting the others when switching
|
||||
* models mid-session.
|
||||
*/
|
||||
export function resolveServiceTier(
|
||||
serviceTier: ServiceTier | null | undefined,
|
||||
provider: Provider | undefined,
|
||||
): ResolvedServiceTier | undefined {
|
||||
if (!serviceTier) return undefined;
|
||||
switch (serviceTier) {
|
||||
case "openai-only":
|
||||
return provider === "openai" || provider === "openai-codex" ? "priority" : undefined;
|
||||
case "claude-only":
|
||||
return provider === "anthropic" ? "priority" : undefined;
|
||||
default:
|
||||
return serviceTier;
|
||||
export type ServiceTierByFamily = Partial<Record<ServiceTierFamily, ServiceTier>>;
|
||||
|
||||
/**
|
||||
* Classify a model into the service-tier family whose knob governs it, or
|
||||
* `undefined` when the model exposes no serving-priority control.
|
||||
*
|
||||
* OpenRouter models are classified by id namespace (`anthropic/`, `google/`,
|
||||
* `openai/`); Claude on Bedrock/Vertex (api `anthropic-messages`) is the
|
||||
* anthropic family even though its provider is `amazon-bedrock`/`google-vertex`.
|
||||
*/
|
||||
export function serviceTierFamily(model: Pick<Model, "provider" | "api" | "id">): ServiceTierFamily | undefined {
|
||||
const provider = model.provider;
|
||||
if (provider === "openrouter") {
|
||||
const id = model.id.toLowerCase();
|
||||
if (id.startsWith("anthropic/")) return "anthropic";
|
||||
if (id.startsWith("google/")) return "google";
|
||||
if (id.startsWith("openai/")) return "openai";
|
||||
return undefined;
|
||||
}
|
||||
if (provider === "openai" || provider === "openai-codex") return "openai";
|
||||
if (model.api === "anthropic-messages") return "anthropic";
|
||||
if (provider === "google" || provider === "google-vertex") return "google";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the (possibly scoped) tier should be sent on the wire as the
|
||||
* `service_tier` request field for the given provider. OpenAI / OpenAI-Codex
|
||||
* accept `flex`/`scale`/`priority`; Fireworks Serverless realizes only its
|
||||
* Priority serving path (`service_tier: "priority"`) on the OpenAI-compatible
|
||||
* chat-completions endpoint. Unsupported tiers (`"auto"`, `"default"`), other
|
||||
* providers, and scope mismatches all return false.
|
||||
* Reduce a per-family tier map to the single wire tier for `model` — the entry
|
||||
* for the model's family, or `undefined` when the model has no family.
|
||||
*/
|
||||
export function resolveModelServiceTier(
|
||||
tiers: ServiceTierByFamily | null | undefined,
|
||||
model: Pick<Model, "provider" | "api" | "id">,
|
||||
): ServiceTier | undefined {
|
||||
if (!tiers) return undefined;
|
||||
const family = serviceTierFamily(model);
|
||||
return family ? tiers[family] : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the tier should be sent on the wire as the provider's service-tier
|
||||
* request field. OpenAI / OpenAI-Codex accept `flex`/`scale`/`priority`; Google
|
||||
* (Gemini API + Vertex) and OpenRouter accept `flex`/`priority`; Fireworks
|
||||
* Serverless realizes only its Priority serving path. Anthropic is absent — it
|
||||
* realizes `priority` via `speed: "fast"`, not a service-tier field.
|
||||
*/
|
||||
export function shouldSendServiceTier(
|
||||
serviceTier: ServiceTier | null | undefined,
|
||||
provider: Provider | undefined,
|
||||
): boolean {
|
||||
const resolved = resolveServiceTier(serviceTier, provider);
|
||||
if (provider === "openai" || provider === "openai-codex") {
|
||||
return resolved === "flex" || resolved === "scale" || resolved === "priority";
|
||||
if (!serviceTier) return false;
|
||||
if (provider === "openai" || provider === "openai-codex" || provider === "openrouter") {
|
||||
return serviceTier === "flex" || serviceTier === "scale" || serviceTier === "priority";
|
||||
}
|
||||
if (provider === "fireworks") {
|
||||
return resolved === "priority";
|
||||
if (provider === "google") {
|
||||
return serviceTier === "flex" || serviceTier === "priority";
|
||||
}
|
||||
// Vertex realizes only priority (via header); flex has no documented control.
|
||||
if (provider === "google-vertex" || provider === "fireworks") {
|
||||
return serviceTier === "priority";
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Premium-request weight contributed by sending priority to a provider
|
||||
* that supports it. Mirrors GitHub Copilot's `premiumRequests` accounting
|
||||
* so the "premium requests" stat aggregates priority traffic across the
|
||||
* OpenAI family and Anthropic fast-mode realizations.
|
||||
* True when `priority` will actually be realized on the wire for `model`.
|
||||
* Direct Anthropic realizes fast mode; OpenAI/Google/Fireworks emit the
|
||||
* service-tier field; OpenRouter realizes it only for its OpenAI- and
|
||||
* Google-family upstreams. Bedrock/Vertex Claude and OpenRouter Anthropic
|
||||
* models do not realize priority and return `false`.
|
||||
*/
|
||||
export function realizesPriorityServiceTier(
|
||||
serviceTier: ServiceTier | null | undefined,
|
||||
model: Pick<Model, "provider" | "api" | "id">,
|
||||
): boolean {
|
||||
if (serviceTier !== "priority") return false;
|
||||
if (model.provider === "anthropic") return true;
|
||||
if (model.provider === "openrouter") {
|
||||
const family = serviceTierFamily(model);
|
||||
return family === "openai" || family === "google";
|
||||
}
|
||||
if (model.api === "anthropic-messages") return false;
|
||||
return shouldSendServiceTier(serviceTier, model.provider);
|
||||
}
|
||||
|
||||
/**
|
||||
* Premium-request weight contributed by a priority request to a provider that
|
||||
* realizes it and bills extra. Mirrors GitHub Copilot's `premiumRequests`
|
||||
* accounting so the "premium requests" stat aggregates priority traffic across
|
||||
* the OpenAI family, direct Anthropic fast mode, and Google priority.
|
||||
*
|
||||
* Returns 1 per resolved priority request, 0 otherwise.
|
||||
* Returns 1 only when priority is actually realized on the wire for `model`
|
||||
* (see {@link realizesPriorityServiceTier}) and the provider bills it as a
|
||||
* premium request. OpenRouter is excluded — it bills per its own pricing, not
|
||||
* Copilot-premium semantics — as are Bedrock/Vertex Claude, where priority is
|
||||
* silently dropped.
|
||||
*/
|
||||
export function getPriorityPremiumRequests(
|
||||
serviceTier: ServiceTier | null | undefined,
|
||||
provider: Provider | undefined,
|
||||
model: Pick<Model, "provider" | "api" | "id">,
|
||||
): number {
|
||||
if (resolveServiceTier(serviceTier, provider) !== "priority") return 0;
|
||||
// Only providers that realize `priority` on the wire bill the user.
|
||||
// Everywhere else, the field is silently dropped and nothing is charged.
|
||||
return provider === "openai" || provider === "openai-codex" || provider === "anthropic" ? 1 : 0;
|
||||
if (!realizesPriorityServiceTier(serviceTier, model)) return 0;
|
||||
const provider = model.provider;
|
||||
return provider === "openai" ||
|
||||
provider === "openai-codex" ||
|
||||
provider === "anthropic" ||
|
||||
provider === "google" ||
|
||||
provider === "google-vertex"
|
||||
? 1
|
||||
: 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Coerce a persisted service-tier value to a {@link ServiceTierByFamily}. Newer
|
||||
* sessions store the family map directly; legacy sessions stored a single
|
||||
* scalar — `"priority"` applied everywhere, `"openai-only"`/`"claude-only"`
|
||||
* scoped to one family, and the remaining values were OpenAI-only semantics.
|
||||
*/
|
||||
export function coerceServiceTierByFamily(value: unknown): ServiceTierByFamily | undefined {
|
||||
if (value === null || value === undefined) return undefined;
|
||||
if (typeof value === "object") {
|
||||
const src = value as Record<string, unknown>;
|
||||
const out: ServiceTierByFamily = {};
|
||||
for (const family of ["openai", "anthropic", "google"] as const) {
|
||||
const tier = src[family];
|
||||
if (tier === "auto" || tier === "default" || tier === "flex" || tier === "scale" || tier === "priority") {
|
||||
out[family] = tier;
|
||||
}
|
||||
}
|
||||
return Object.keys(out).length > 0 ? out : undefined;
|
||||
}
|
||||
switch (value) {
|
||||
case "priority":
|
||||
return { openai: "priority", anthropic: "priority", google: "priority" };
|
||||
case "openai-only":
|
||||
return { openai: "priority" };
|
||||
case "claude-only":
|
||||
return { anthropic: "priority" };
|
||||
case "auto":
|
||||
return { openai: "auto" };
|
||||
case "default":
|
||||
return { openai: "default" };
|
||||
case "flex":
|
||||
return { openai: "flex" };
|
||||
case "scale":
|
||||
return { openai: "scale" };
|
||||
default:
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
export interface ProviderSessionState {
|
||||
|
||||
@@ -88,22 +88,6 @@ describe("Anthropic priority service tier → speed='fast'", () => {
|
||||
expect(payload.speed).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
it("sets speed='fast' on direct anthropic when serviceTier='claude-only'", async () => {
|
||||
const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), {
|
||||
serviceTier: "claude-only",
|
||||
})) as { speed?: string };
|
||||
expect(payload.speed).toBe("fast");
|
||||
});
|
||||
|
||||
it("omits speed when serviceTier='openai-only' on an anthropic model", async () => {
|
||||
// Scoped to OpenAI — on this anthropic request, the scope doesn't match,
|
||||
// so `speed` must not be set on the wire.
|
||||
const payload = (await capturePayload(makeAnthropicModel("claude-opus-4-7"), {
|
||||
serviceTier: "openai-only",
|
||||
})) as Record<string, unknown>;
|
||||
expect(payload.speed).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("clearAnthropicFastModeFallback", () => {
|
||||
|
||||
@@ -0,0 +1,105 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google";
|
||||
import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex";
|
||||
import type { AssistantMessageEvent, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
const context: Context = { messages: [{ role: "user", content: "hi", timestamp: 1 }] };
|
||||
|
||||
function sseStop(): Response {
|
||||
const chunk = {
|
||||
candidates: [{ content: { parts: [{ text: "ok" }] }, finishReason: "STOP" }],
|
||||
usageMetadata: { promptTokenCount: 1, candidatesTokenCount: 1, totalTokenCount: 2 },
|
||||
};
|
||||
return new Response(`data: ${JSON.stringify(chunk)}\n\n`, {
|
||||
status: 200,
|
||||
headers: { "content-type": "text/event-stream" },
|
||||
});
|
||||
}
|
||||
|
||||
async function drain(stream: AsyncIterable<AssistantMessageEvent>): Promise<void> {
|
||||
for await (const _ of stream) {
|
||||
// consume
|
||||
}
|
||||
}
|
||||
|
||||
interface Captured {
|
||||
headers: Headers;
|
||||
body: Record<string, unknown>;
|
||||
}
|
||||
|
||||
function capturingFetch(): { fetch: FetchImpl; captured: () => Captured } {
|
||||
let cap: Captured | undefined;
|
||||
const fetch: FetchImpl = async (_url, init) => {
|
||||
cap = {
|
||||
headers: new Headers(init?.headers),
|
||||
body: JSON.parse(String(init?.body ?? "{}")),
|
||||
};
|
||||
return sseStop();
|
||||
};
|
||||
return {
|
||||
fetch,
|
||||
captured: () => {
|
||||
if (!cap) throw new Error("fetch was not called");
|
||||
return cap;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
const geminiModel: Model<"google-generative-ai"> = buildModel({
|
||||
id: "gemini-3-flash",
|
||||
name: "Gemini 3 Flash",
|
||||
api: "google-generative-ai",
|
||||
provider: "google",
|
||||
baseUrl: "https://generativelanguage.googleapis.com/v1beta",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 32_000,
|
||||
});
|
||||
|
||||
const vertexModel: Model<"google-vertex"> = buildModel({
|
||||
id: "gemini-3-flash",
|
||||
name: "Gemini 3 Flash (Vertex)",
|
||||
api: "google-vertex",
|
||||
provider: "google-vertex",
|
||||
baseUrl: "",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 32_000,
|
||||
});
|
||||
|
||||
describe("Google service tier wire encoding", () => {
|
||||
it("Gemini API sends the tier in the request body, not a header", async () => {
|
||||
const { fetch, captured } = capturingFetch();
|
||||
await drain(streamGoogle(geminiModel, context, { apiKey: "k", serviceTier: "priority", fetch }));
|
||||
const { headers, body } = captured();
|
||||
expect(body.serviceTier).toBe("priority");
|
||||
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull();
|
||||
});
|
||||
|
||||
it("Vertex sends priority via header and omits the body tier field", async () => {
|
||||
const { fetch, captured } = capturingFetch();
|
||||
await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "priority", fetch }));
|
||||
const { headers, body } = captured();
|
||||
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBe("priority");
|
||||
expect(body.serviceTier).toBeUndefined();
|
||||
});
|
||||
|
||||
it("Vertex omits both header and body for flex (no documented control)", async () => {
|
||||
const { fetch, captured } = capturingFetch();
|
||||
await drain(streamGoogleVertex(vertexModel, context, { apiKey: "k", serviceTier: "flex", fetch }));
|
||||
const { headers, body } = captured();
|
||||
expect(headers.get("X-Vertex-AI-LLM-Shared-Request-Type")).toBeNull();
|
||||
expect(body.serviceTier).toBeUndefined();
|
||||
});
|
||||
|
||||
it("omits the tier entirely when unset", async () => {
|
||||
const { fetch, captured } = capturingFetch();
|
||||
await drain(streamGoogle(geminiModel, context, { apiKey: "k", fetch }));
|
||||
expect(captured().body.serviceTier).toBeUndefined();
|
||||
});
|
||||
});
|
||||
@@ -1,138 +1,149 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { getPriorityPremiumRequests, resolveServiceTier, shouldSendServiceTier } from "@oh-my-pi/pi-ai/types";
|
||||
import type { Api } from "@oh-my-pi/pi-ai/types";
|
||||
import {
|
||||
coerceServiceTierByFamily,
|
||||
getPriorityPremiumRequests,
|
||||
realizesPriorityServiceTier,
|
||||
resolveModelServiceTier,
|
||||
serviceTierFamily,
|
||||
shouldSendServiceTier,
|
||||
} from "@oh-my-pi/pi-ai/types";
|
||||
|
||||
describe("getPriorityPremiumRequests", () => {
|
||||
it("counts priority tier as one premium request on OpenAI", () => {
|
||||
expect(getPriorityPremiumRequests("priority", "openai")).toBe(1);
|
||||
const m = (provider: string, api: Api, id: string): { provider: string; api: Api; id: string } => ({
|
||||
provider,
|
||||
api,
|
||||
id,
|
||||
});
|
||||
|
||||
it("counts priority tier as one premium request on OpenAI Codex", () => {
|
||||
expect(getPriorityPremiumRequests("priority", "openai-codex")).toBe(1);
|
||||
const openai = m("openai", "openai-responses", "gpt-5");
|
||||
const codex = m("openai-codex", "openai-codex-responses", "gpt-5.5");
|
||||
const anthropic = m("anthropic", "anthropic-messages", "claude-opus-4-6");
|
||||
const vertexClaude = m("google-vertex", "anthropic-messages", "claude-opus-4-6");
|
||||
const gemini = m("google", "google-generative-ai", "gemini-3-flash");
|
||||
const vertexGemini = m("google-vertex", "google-vertex", "gemini-3-flash");
|
||||
const fireworks = m("fireworks", "openai-completions", "qwen3");
|
||||
const orOpenAI = m("openrouter", "openai-responses", "openai/gpt-5.5");
|
||||
const orGoogle = m("openrouter", "openai-completions", "google/gemini-3-flash");
|
||||
const orAnthropic = m("openrouter", "openai-completions", "anthropic/claude-opus-4-6");
|
||||
|
||||
describe("serviceTierFamily", () => {
|
||||
it("classifies first-party providers by provider/api", () => {
|
||||
expect(serviceTierFamily(openai)).toBe("openai");
|
||||
expect(serviceTierFamily(codex)).toBe("openai");
|
||||
expect(serviceTierFamily(anthropic)).toBe("anthropic");
|
||||
expect(serviceTierFamily(vertexClaude)).toBe("anthropic"); // Claude on Vertex is the anthropic family
|
||||
expect(serviceTierFamily(gemini)).toBe("google");
|
||||
expect(serviceTierFamily(vertexGemini)).toBe("google");
|
||||
expect(serviceTierFamily(fireworks)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("ignores non-priority paid tiers", () => {
|
||||
expect(getPriorityPremiumRequests("flex", "openai")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("scale", "openai")).toBe(0);
|
||||
});
|
||||
|
||||
it("ignores default and auto tiers", () => {
|
||||
expect(getPriorityPremiumRequests("default", "openai")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("auto", "openai")).toBe(0);
|
||||
});
|
||||
|
||||
it("ignores priority tier on providers that drop service_tier", () => {
|
||||
// `priority` is realized on `openai`, `openai-codex`, and direct `anthropic`
|
||||
// (as fast mode). Everywhere else it's silently dropped, so it must not
|
||||
// be billed as premium.
|
||||
expect(getPriorityPremiumRequests("priority", "github-copilot")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("priority", "azure")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("priority", "bedrock")).toBe(0);
|
||||
});
|
||||
|
||||
it("counts priority on direct Anthropic as one premium request (fast mode)", () => {
|
||||
expect(getPriorityPremiumRequests("priority", "anthropic")).toBe(1);
|
||||
});
|
||||
|
||||
it("returns zero when service tier is unset", () => {
|
||||
expect(getPriorityPremiumRequests(undefined, "openai")).toBe(0);
|
||||
expect(getPriorityPremiumRequests(null, "openai")).toBe(0);
|
||||
});
|
||||
|
||||
describe("scoped tiers", () => {
|
||||
it("treats `openai-only` as priority on OpenAI and OpenAI-Codex", () => {
|
||||
expect(getPriorityPremiumRequests("openai-only", "openai")).toBe(1);
|
||||
expect(getPriorityPremiumRequests("openai-only", "openai-codex")).toBe(1);
|
||||
});
|
||||
|
||||
it("treats `openai-only` as inactive on Anthropic and everywhere else", () => {
|
||||
expect(getPriorityPremiumRequests("openai-only", "anthropic")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("openai-only", "github-copilot")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("openai-only", "bedrock")).toBe(0);
|
||||
});
|
||||
|
||||
it("treats `claude-only` as priority on direct Anthropic", () => {
|
||||
expect(getPriorityPremiumRequests("claude-only", "anthropic")).toBe(1);
|
||||
});
|
||||
|
||||
it("treats `claude-only` as inactive on OpenAI, Bedrock/Vertex, and elsewhere", () => {
|
||||
expect(getPriorityPremiumRequests("claude-only", "openai")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("claude-only", "openai-codex")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("claude-only", "bedrock")).toBe(0);
|
||||
expect(getPriorityPremiumRequests("claude-only", "vertex")).toBe(0);
|
||||
});
|
||||
it("classifies OpenRouter models by id namespace", () => {
|
||||
expect(serviceTierFamily(orOpenAI)).toBe("openai");
|
||||
expect(serviceTierFamily(orGoogle)).toBe("google");
|
||||
expect(serviceTierFamily(orAnthropic)).toBe("anthropic");
|
||||
expect(serviceTierFamily(m("openrouter", "openai-completions", "z-ai/glm-4.7"))).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("resolveServiceTier", () => {
|
||||
it("passes unscoped tiers through unchanged for any provider", () => {
|
||||
expect(resolveServiceTier("flex", "openai")).toBe("flex");
|
||||
expect(resolveServiceTier("priority", "anthropic")).toBe("priority");
|
||||
expect(resolveServiceTier("auto", "openai-codex")).toBe("auto");
|
||||
expect(resolveServiceTier("default", "github-copilot")).toBe("default");
|
||||
});
|
||||
|
||||
it("scopes `openai-only` to OpenAI providers", () => {
|
||||
expect(resolveServiceTier("openai-only", "openai")).toBe("priority");
|
||||
expect(resolveServiceTier("openai-only", "openai-codex")).toBe("priority");
|
||||
expect(resolveServiceTier("openai-only", "anthropic")).toBeUndefined();
|
||||
expect(resolveServiceTier("openai-only", "bedrock")).toBeUndefined();
|
||||
expect(resolveServiceTier("openai-only", undefined)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("scopes `claude-only` to direct Anthropic", () => {
|
||||
expect(resolveServiceTier("claude-only", "anthropic")).toBe("priority");
|
||||
expect(resolveServiceTier("claude-only", "openai")).toBeUndefined();
|
||||
expect(resolveServiceTier("claude-only", "bedrock")).toBeUndefined();
|
||||
expect(resolveServiceTier("claude-only", "vertex")).toBeUndefined();
|
||||
});
|
||||
|
||||
it("returns undefined for null/undefined input", () => {
|
||||
expect(resolveServiceTier(undefined, "openai")).toBeUndefined();
|
||||
expect(resolveServiceTier(null, "openai")).toBeUndefined();
|
||||
describe("resolveModelServiceTier", () => {
|
||||
it("reduces a per-family map to the model's family entry", () => {
|
||||
const tiers = { openai: "priority", anthropic: "priority", google: "flex" } as const;
|
||||
expect(resolveModelServiceTier(tiers, openai)).toBe("priority");
|
||||
expect(resolveModelServiceTier(tiers, gemini)).toBe("flex");
|
||||
expect(resolveModelServiceTier(tiers, orAnthropic)).toBe("priority");
|
||||
expect(resolveModelServiceTier(tiers, fireworks)).toBeUndefined(); // no family
|
||||
expect(resolveModelServiceTier(undefined, openai)).toBeUndefined();
|
||||
expect(resolveModelServiceTier({ google: "priority" }, openai)).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("shouldSendServiceTier", () => {
|
||||
it("returns false for non-OpenAI/non-Fireworks providers", () => {
|
||||
expect(shouldSendServiceTier("flex", "azure-openai-responses")).toBe(false);
|
||||
expect(shouldSendServiceTier("scale", "firepass")).toBe(false);
|
||||
it("sends flex/scale/priority on the OpenAI family and OpenRouter", () => {
|
||||
for (const p of ["openai", "openai-codex", "openrouter"]) {
|
||||
expect(shouldSendServiceTier("flex", p)).toBe(true);
|
||||
expect(shouldSendServiceTier("scale", p)).toBe(true);
|
||||
expect(shouldSendServiceTier("priority", p)).toBe(true);
|
||||
expect(shouldSendServiceTier("default", p)).toBe(false);
|
||||
expect(shouldSendServiceTier("auto", p)).toBe(false);
|
||||
}
|
||||
});
|
||||
|
||||
it("sends flex/priority on direct Google, priority-only on Vertex (no scale)", () => {
|
||||
expect(shouldSendServiceTier("flex", "google")).toBe(true);
|
||||
expect(shouldSendServiceTier("priority", "google")).toBe(true);
|
||||
expect(shouldSendServiceTier("scale", "google")).toBe(false);
|
||||
expect(shouldSendServiceTier("priority", "google-vertex")).toBe(true);
|
||||
expect(shouldSendServiceTier("flex", "google-vertex")).toBe(false); // Vertex flex has no wire control
|
||||
});
|
||||
|
||||
it("sends only priority on Fireworks, nothing on Anthropic", () => {
|
||||
expect(shouldSendServiceTier("priority", "fireworks")).toBe(true);
|
||||
expect(shouldSendServiceTier("flex", "fireworks")).toBe(false);
|
||||
expect(shouldSendServiceTier("priority", "anthropic")).toBe(false);
|
||||
});
|
||||
|
||||
it("returns true for fireworks only with the priority tier", () => {
|
||||
expect(shouldSendServiceTier("priority", "fireworks")).toBe(true);
|
||||
// Fireworks realizes only the Priority serving path — flex/scale are OpenAI-only.
|
||||
expect(shouldSendServiceTier("flex", "fireworks")).toBe(false);
|
||||
expect(shouldSendServiceTier("scale", "fireworks")).toBe(false);
|
||||
expect(shouldSendServiceTier("auto", "fireworks")).toBe(false);
|
||||
expect(shouldSendServiceTier("default", "fireworks")).toBe(false);
|
||||
expect(shouldSendServiceTier(undefined, "fireworks")).toBe(false);
|
||||
});
|
||||
|
||||
it("returns true for openai with priority/flex/scale tiers", () => {
|
||||
expect(shouldSendServiceTier("priority", "openai")).toBe(true);
|
||||
expect(shouldSendServiceTier("flex", "openai")).toBe(true);
|
||||
expect(shouldSendServiceTier("scale", "openai")).toBe(true);
|
||||
});
|
||||
|
||||
it("returns true for openai-codex with priority/flex/scale tiers", () => {
|
||||
expect(shouldSendServiceTier("priority", "openai-codex")).toBe(true);
|
||||
expect(shouldSendServiceTier("flex", "openai-codex")).toBe(true);
|
||||
expect(shouldSendServiceTier("scale", "openai-codex")).toBe(true);
|
||||
});
|
||||
|
||||
it("returns false for default tier on OpenAI providers", () => {
|
||||
expect(shouldSendServiceTier("default", "openai")).toBe(false);
|
||||
expect(shouldSendServiceTier("default", "openai-codex")).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false for auto tier on OpenAI providers", () => {
|
||||
expect(shouldSendServiceTier("auto", "openai")).toBe(false);
|
||||
expect(shouldSendServiceTier("auto", "openai-codex")).toBe(false);
|
||||
});
|
||||
|
||||
it("returns false for undefined/null tier", () => {
|
||||
it("returns false for unset tiers", () => {
|
||||
expect(shouldSendServiceTier(undefined, "openai")).toBe(false);
|
||||
expect(shouldSendServiceTier(null, "openai")).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("realizesPriorityServiceTier", () => {
|
||||
it("realizes priority where the wire actually applies it", () => {
|
||||
expect(realizesPriorityServiceTier("priority", openai)).toBe(true);
|
||||
expect(realizesPriorityServiceTier("priority", anthropic)).toBe(true); // direct fast mode
|
||||
expect(realizesPriorityServiceTier("priority", gemini)).toBe(true);
|
||||
expect(realizesPriorityServiceTier("priority", vertexGemini)).toBe(true);
|
||||
expect(realizesPriorityServiceTier("priority", fireworks)).toBe(true);
|
||||
expect(realizesPriorityServiceTier("priority", orOpenAI)).toBe(true);
|
||||
expect(realizesPriorityServiceTier("priority", orGoogle)).toBe(true);
|
||||
});
|
||||
|
||||
it("does not realize priority where the wire drops it", () => {
|
||||
expect(realizesPriorityServiceTier("priority", vertexClaude)).toBe(false); // no fast mode on Vertex
|
||||
expect(realizesPriorityServiceTier("priority", orAnthropic)).toBe(false); // OpenRouter Anthropic
|
||||
expect(realizesPriorityServiceTier("flex", openai)).toBe(false);
|
||||
expect(realizesPriorityServiceTier(undefined, openai)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("getPriorityPremiumRequests", () => {
|
||||
it("counts one premium request per realized priority on billing providers", () => {
|
||||
expect(getPriorityPremiumRequests("priority", openai)).toBe(1);
|
||||
expect(getPriorityPremiumRequests("priority", codex)).toBe(1);
|
||||
expect(getPriorityPremiumRequests("priority", anthropic)).toBe(1);
|
||||
expect(getPriorityPremiumRequests("priority", gemini)).toBe(1);
|
||||
expect(getPriorityPremiumRequests("priority", vertexGemini)).toBe(1);
|
||||
});
|
||||
|
||||
it("does not bill OpenRouter, unrealized, or non-priority traffic", () => {
|
||||
expect(getPriorityPremiumRequests("priority", orOpenAI)).toBe(0); // OpenRouter bills its own way
|
||||
expect(getPriorityPremiumRequests("priority", vertexClaude)).toBe(0); // not realized
|
||||
expect(getPriorityPremiumRequests("priority", fireworks)).toBe(0); // realized but not Copilot-premium
|
||||
expect(getPriorityPremiumRequests("flex", openai)).toBe(0);
|
||||
expect(getPriorityPremiumRequests(undefined, openai)).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe("coerceServiceTierByFamily", () => {
|
||||
it("migrates legacy scalar values to a per-family map", () => {
|
||||
expect(coerceServiceTierByFamily("priority")).toEqual({
|
||||
openai: "priority",
|
||||
anthropic: "priority",
|
||||
google: "priority",
|
||||
});
|
||||
expect(coerceServiceTierByFamily("openai-only")).toEqual({ openai: "priority" });
|
||||
expect(coerceServiceTierByFamily("claude-only")).toEqual({ anthropic: "priority" });
|
||||
expect(coerceServiceTierByFamily("flex")).toEqual({ openai: "flex" });
|
||||
expect(coerceServiceTierByFamily("none")).toBeUndefined();
|
||||
expect(coerceServiceTierByFamily(null)).toBeUndefined();
|
||||
});
|
||||
|
||||
it("passes a per-family map through, dropping invalid entries", () => {
|
||||
expect(coerceServiceTierByFamily({ openai: "priority", google: "flex" })).toEqual({
|
||||
openai: "priority",
|
||||
google: "flex",
|
||||
});
|
||||
expect(coerceServiceTierByFamily({ openai: "bogus" })).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -4,6 +4,11 @@
|
||||
|
||||
### Changed
|
||||
|
||||
- Replaced the global `serviceTier` setting with `tier.openai`, `tier.anthropic`, and `tier.google` for granular control
|
||||
- Updated `/fast` to target the service-tier family of the currently selected model
|
||||
- Updated subagent and advisor tier configuration to use the new per-family setting structure
|
||||
- Removed `fastModeScope` setting, as per-family scoping is now natively supported via the `tier.*` settings
|
||||
|
||||
- Improved binary file detection and terminal handling to prevent corruption from non-UTF-8 content, and updated file summaries to explicitly note skipped binary files.
|
||||
- Enhanced context compaction (snapcompact) to resolve shapes contextually based on rendered text content.
|
||||
|
||||
|
||||
@@ -10,9 +10,10 @@ import type {
|
||||
Model,
|
||||
ProviderSessionState,
|
||||
ServiceTier,
|
||||
ServiceTierByFamily,
|
||||
SimpleStreamOptions,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai";
|
||||
import { resolveModelServiceTier, streamSimple } from "@oh-my-pi/pi-ai";
|
||||
import { buildModelProviderPriorityRank, type CanonicalModelVariant } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui";
|
||||
import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils";
|
||||
@@ -25,7 +26,7 @@ import {
|
||||
getModelMatchPreferences,
|
||||
resolveCliModel,
|
||||
} from "../config/model-resolver";
|
||||
import { resolveServiceTierSetting } from "../config/service-tier";
|
||||
import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier";
|
||||
import { Settings } from "../config/settings";
|
||||
import benchPrompt from "../prompts/bench.md" with { type: "text" };
|
||||
import { discoverAuthStorage, loadCliExtensionProviders } from "../sdk";
|
||||
@@ -106,8 +107,8 @@ export interface BenchSummary {
|
||||
maxTokens: number;
|
||||
models: BenchModelReport[];
|
||||
failures: number;
|
||||
/** Requested service tier passed to every request; absent when none was requested. Scoped tiers (`openai-only`/`claude-only`) may be dropped per-provider downstream. */
|
||||
serviceTier?: ServiceTier;
|
||||
/** Requested per-family service tiers, resolved per model before reaching the wire. */
|
||||
serviceTierByFamily?: ServiceTierByFamily;
|
||||
}
|
||||
|
||||
type BenchStreamSimple = (
|
||||
@@ -518,12 +519,18 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
|
||||
const runtime = await (deps.createRuntime ?? createDefaultRuntime)();
|
||||
try {
|
||||
const targets = resolveBenchModels(command.models, runtime.modelRegistry, runtime.settings, writeStderr);
|
||||
// Explicit `--service-tier` wins; otherwise fall back to the configured
|
||||
// `serviceTier` setting (`none`/unset omits the wire field). Scope-aware
|
||||
// gating to the model's provider happens downstream in the provider layer.
|
||||
const serviceTierValue = command.flags.serviceTier ?? runtime.settings?.get("serviceTier");
|
||||
const serviceTier = serviceTierValue ? resolveServiceTierSetting(serviceTierValue, undefined) : undefined;
|
||||
if (!json && serviceTier) writeStdout(`${chalk.dim(`service tier: ${serviceTier}`)}\n`);
|
||||
// Explicit `--service-tier` (a single value broadcast across families) wins;
|
||||
// otherwise fall back to the configured per-family `tier.*` settings. Each
|
||||
// model resolves its own family's tier below before reaching the wire.
|
||||
const flagTier = command.flags.serviceTier ? serviceTierSettingToTier(command.flags.serviceTier) : undefined;
|
||||
const serviceTierByFamily = command.flags.serviceTier
|
||||
? serviceTierForAllFamilies(flagTier)
|
||||
: buildServiceTierByFamily(
|
||||
runtime.settings?.get("tier.openai") ?? "none",
|
||||
runtime.settings?.get("tier.anthropic") ?? "none",
|
||||
runtime.settings?.get("tier.google") ?? "none",
|
||||
);
|
||||
if (!json && flagTier) writeStdout(`${chalk.dim(`service tier: ${flagTier}`)}\n`);
|
||||
const reports: BenchModelReport[] = [];
|
||||
for (const { selector, model, thinking } of targets) {
|
||||
if (!json) {
|
||||
@@ -564,7 +571,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
|
||||
maxTokens,
|
||||
reasoning: toReasoningEffort(thinking),
|
||||
disableReasoning: shouldDisableReasoning(thinking) ? true : undefined,
|
||||
serviceTier,
|
||||
serviceTier: resolveModelServiceTier(serviceTierByFamily, model),
|
||||
},
|
||||
streamFn,
|
||||
now,
|
||||
@@ -606,7 +613,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
|
||||
reports.push(buildModelReport(selector, model, thinking, results));
|
||||
}
|
||||
const failures = reports.reduce((sum, report) => sum + report.results.filter(result => !result.ok).length, 0);
|
||||
const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTier };
|
||||
const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTierByFamily };
|
||||
if (json) {
|
||||
writeStdout(`${JSON.stringify(summary, null, 2)}\n`);
|
||||
} else if (reports.length > 1 || runs > 1) {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
|
||||
import { runBenchCommand } from "../cli/bench-cli";
|
||||
import { SERVICE_TIER_SETTING_VALUES } from "../config/service-tier";
|
||||
import { SERVICE_TIER_OPENAI_VALUES } from "../config/service-tier";
|
||||
|
||||
export default class Bench extends Command {
|
||||
static description =
|
||||
@@ -19,8 +19,8 @@ export default class Bench extends Command {
|
||||
"max-tokens": Flags.integer({ description: "Max output tokens per request", default: 512 }),
|
||||
prompt: Flags.string({ description: "Custom prompt text (default: bundled bench prompt)" }),
|
||||
"service-tier": Flags.string({
|
||||
description: "Service tier hint (default: configured `serviceTier` setting; `none` omits it)",
|
||||
options: SERVICE_TIER_SETTING_VALUES,
|
||||
description: "Service tier applied per model family (default: configured `tier.*` settings; `none` omits it)",
|
||||
options: SERVICE_TIER_OPENAI_VALUES,
|
||||
}),
|
||||
json: Flags.boolean({ description: "Output JSON" }),
|
||||
par: Flags.integer({ description: "Execute runs with N parallel queries/requests", default: 4 }),
|
||||
|
||||
@@ -1,87 +1,116 @@
|
||||
import type { ServiceTier } from "@oh-my-pi/pi-ai";
|
||||
import type { ServiceTier, ServiceTierByFamily } from "@oh-my-pi/pi-ai";
|
||||
import type { SubmenuOption } from "./settings-schema";
|
||||
|
||||
/**
|
||||
* Service-tier setting values shared by every "Service Tier" setting. `"none"`
|
||||
* is the omit-the-parameter sentinel; the remaining values mirror
|
||||
* {@link ServiceTier}.
|
||||
* Per-family service-tier setting values. `"none"` is the omit-the-parameter
|
||||
* sentinel; the rest mirror the wire {@link ServiceTier} values each provider
|
||||
* family actually realizes. OpenAI accepts the full set; Anthropic realizes
|
||||
* only `priority` (fast mode); Google (Gemini API + Vertex) realizes
|
||||
* `flex`/`priority`.
|
||||
*/
|
||||
export const SERVICE_TIER_SETTING_VALUES = [
|
||||
export const SERVICE_TIER_OPENAI_VALUES = ["none", "auto", "default", "flex", "scale", "priority"] as const;
|
||||
export const SERVICE_TIER_ANTHROPIC_VALUES = ["none", "priority"] as const;
|
||||
export const SERVICE_TIER_GOOGLE_VALUES = ["none", "flex", "priority"] as const;
|
||||
|
||||
export type ServiceTierOpenAISettingValue = (typeof SERVICE_TIER_OPENAI_VALUES)[number];
|
||||
export type ServiceTierAnthropicSettingValue = (typeof SERVICE_TIER_ANTHROPIC_VALUES)[number];
|
||||
export type ServiceTierGoogleSettingValue = (typeof SERVICE_TIER_GOOGLE_VALUES)[number];
|
||||
|
||||
/**
|
||||
* Inherit-capable single value for the subagent/advisor tiers. The chosen tier
|
||||
* is broadcast across families and applied to whichever family the spawned
|
||||
* model belongs to (clamped to what that family realizes); `"inherit"` defers
|
||||
* to the main agent's live per-family selection.
|
||||
*/
|
||||
export const SERVICE_TIER_INHERIT_SETTING_VALUES = [
|
||||
"inherit",
|
||||
"none",
|
||||
"auto",
|
||||
"default",
|
||||
"flex",
|
||||
"scale",
|
||||
"priority",
|
||||
"openai-only",
|
||||
"claude-only",
|
||||
] as const;
|
||||
|
||||
export type ServiceTierSettingValue = (typeof SERVICE_TIER_SETTING_VALUES)[number];
|
||||
|
||||
/** Variant value set for scoped service-tier settings (subagent/advisor) that can defer to the main agent. */
|
||||
export const SERVICE_TIER_INHERIT_SETTING_VALUES = ["inherit", ...SERVICE_TIER_SETTING_VALUES] as const;
|
||||
|
||||
export type ServiceTierInheritSettingValue = (typeof SERVICE_TIER_INHERIT_SETTING_VALUES)[number];
|
||||
|
||||
/** Submenu descriptions shared by the base `serviceTier` setting. */
|
||||
export const SERVICE_TIER_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierSettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Omit service_tier parameter" },
|
||||
{ value: "auto", label: "Auto", description: "Use provider default tier selection (OpenAI)" },
|
||||
{ value: "default", label: "Default", description: "Standard priority processing (OpenAI)" },
|
||||
{ value: "flex", label: "Flex", description: "Flexible capacity tier when available (OpenAI)" },
|
||||
{ value: "scale", label: "Scale", description: "Scale Tier credits when available (OpenAI)" },
|
||||
export const SERVICE_TIER_OPENAI_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierOpenAISettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Omit service_tier (standard processing)" },
|
||||
{ value: "auto", label: "Auto", description: "Provider default tier selection" },
|
||||
{ value: "default", label: "Default", description: "Standard priority processing" },
|
||||
{ value: "flex", label: "Flex", description: "Lower cost, higher latency when available" },
|
||||
{ value: "scale", label: "Scale", description: "Scale Tier credits when available" },
|
||||
{ value: "priority", label: "Priority", description: "Faster, higher cost (premium request)" },
|
||||
];
|
||||
|
||||
export const SERVICE_TIER_ANTHROPIC_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierAnthropicSettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Standard processing" },
|
||||
{
|
||||
value: "priority",
|
||||
label: "Priority",
|
||||
description: "Priority on every supported provider (OpenAI `service_tier`, Anthropic fast mode)",
|
||||
},
|
||||
{
|
||||
value: "openai-only",
|
||||
label: "Priority (OpenAI only)",
|
||||
description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere",
|
||||
},
|
||||
{
|
||||
value: "claude-only",
|
||||
label: "Priority (Claude only)",
|
||||
description: "Anthropic fast mode on direct Claude requests; ignored elsewhere (incl. Bedrock/Vertex)",
|
||||
description: 'Fast mode (`speed: "fast"`) on supported direct Claude models; ignored on Bedrock/Vertex',
|
||||
},
|
||||
];
|
||||
|
||||
/** Submenu descriptions for inherit-capable service-tier settings. */
|
||||
export const SERVICE_TIER_GOOGLE_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierGoogleSettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Standard processing" },
|
||||
{ value: "flex", label: "Flex", description: "Lower cost, higher latency (Gemini API + Vertex)" },
|
||||
{ value: "priority", label: "Priority", description: "Faster, higher reliability (Gemini API + Vertex)" },
|
||||
];
|
||||
|
||||
export const SERVICE_TIER_INHERIT_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierInheritSettingValue>> = [
|
||||
{ value: "inherit", label: "Inherit", description: "Use the main agent's Service Tier" },
|
||||
...SERVICE_TIER_OPTIONS,
|
||||
{ value: "inherit", label: "Inherit", description: "Match the main agent's live per-family tiers" },
|
||||
{ value: "none", label: "None", description: "Standard processing" },
|
||||
{ value: "auto", label: "Auto", description: "Provider default tier selection (OpenAI family)" },
|
||||
{ value: "default", label: "Default", description: "Standard priority processing (OpenAI family)" },
|
||||
{ value: "flex", label: "Flex", description: "Flexible capacity tier (OpenAI/Google families)" },
|
||||
{ value: "scale", label: "Scale", description: "Scale Tier credits (OpenAI family)" },
|
||||
{ value: "priority", label: "Priority", description: "Priority on every supported family of the spawned model" },
|
||||
];
|
||||
|
||||
/**
|
||||
* Resolve a service-tier setting value to the wire {@link ServiceTier} (or
|
||||
* `undefined` to omit). `"inherit"` defers to `inherited`; `"none"` omits.
|
||||
*/
|
||||
export function resolveServiceTierSetting(value: string, inherited: ServiceTier | undefined): ServiceTier | undefined {
|
||||
if (value === "inherit") return inherited;
|
||||
if (value === "none" || value === "") return undefined;
|
||||
/** Map a per-family setting value to a wire {@link ServiceTier}, or `undefined` to omit. */
|
||||
export function serviceTierSettingToTier(value: string): ServiceTier | undefined {
|
||||
if (value === "none" || value === "" || value === "inherit") return undefined;
|
||||
return value as ServiceTier;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the `serviceTier` *setting value* to stamp onto a subagent's settings
|
||||
* snapshot.
|
||||
*
|
||||
* - A concrete `subagentSetting` (`"none"` or a tier) wins outright.
|
||||
* - `"inherit"` defers to the parent's live effective tier when the caller has a
|
||||
* live session (`inherited` passed as `ServiceTier | null`, where `null` means
|
||||
* the parent explicitly has no tier — e.g. `/fast off`). When no live session
|
||||
* is available (`inherited === undefined`, e.g. cold subagent revive) it falls
|
||||
* back to the parent's configured `serviceTier` setting so behavior matches a
|
||||
* plain settings snapshot.
|
||||
*/
|
||||
export function resolveSubagentServiceTier(
|
||||
subagentSetting: string,
|
||||
configuredTier: ServiceTierSettingValue,
|
||||
inherited: ServiceTier | null | undefined,
|
||||
): ServiceTierSettingValue {
|
||||
if (subagentSetting !== "inherit") return subagentSetting as ServiceTierSettingValue;
|
||||
if (inherited === undefined) return configuredTier;
|
||||
return inherited ?? "none";
|
||||
/** Assemble the live per-family tier map from the three `tier.*` setting values. */
|
||||
export function buildServiceTierByFamily(openai: string, anthropic: string, google: string): ServiceTierByFamily {
|
||||
const out: ServiceTierByFamily = {};
|
||||
const o = serviceTierSettingToTier(openai);
|
||||
if (o) out.openai = o;
|
||||
const a = serviceTierSettingToTier(anthropic);
|
||||
if (a) out.anthropic = a;
|
||||
const g = serviceTierSettingToTier(google);
|
||||
if (g) out.google = g;
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Broadcast a single chosen tier across families, clamped to what each family
|
||||
* realizes: OpenAI takes any tier, Anthropic only `priority`, Google only
|
||||
* `flex`/`priority`. Used by the subagent/advisor single-value settings and the
|
||||
* `omp bench --service-tier` flag, which apply one tier to whatever family the
|
||||
* target model belongs to.
|
||||
*/
|
||||
export function serviceTierForAllFamilies(tier: ServiceTier | undefined): ServiceTierByFamily {
|
||||
if (!tier) return {};
|
||||
const out: ServiceTierByFamily = { openai: tier };
|
||||
if (tier === "priority") out.anthropic = "priority";
|
||||
if (tier === "flex" || tier === "priority") out.google = tier;
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a subagent/advisor service-tier setting to a per-family map.
|
||||
*
|
||||
* - A concrete tier is broadcast across families (see
|
||||
* {@link serviceTierForAllFamilies}).
|
||||
* - `"none"` yields an empty map.
|
||||
* - `"inherit"` defers to `inherited` — the parent's live per-family tiers when
|
||||
* a live session supplied them, else the empty map.
|
||||
*/
|
||||
export function resolveSubagentServiceTier(setting: string, inherited: ServiceTierByFamily): ServiceTierByFamily {
|
||||
if (setting === "inherit") return inherited;
|
||||
return serviceTierForAllFamilies(serviceTierSettingToTier(setting));
|
||||
}
|
||||
|
||||
@@ -36,10 +36,14 @@ import {
|
||||
import { EDIT_MODES } from "../utils/edit-mode";
|
||||
import { SEARCH_PROVIDER_OPTIONS, SEARCH_PROVIDER_PREFERENCES, type SearchProviderId } from "../web/search/types";
|
||||
import {
|
||||
SERVICE_TIER_ANTHROPIC_OPTIONS,
|
||||
SERVICE_TIER_ANTHROPIC_VALUES,
|
||||
SERVICE_TIER_GOOGLE_OPTIONS,
|
||||
SERVICE_TIER_GOOGLE_VALUES,
|
||||
SERVICE_TIER_INHERIT_OPTIONS,
|
||||
SERVICE_TIER_INHERIT_SETTING_VALUES,
|
||||
SERVICE_TIER_OPTIONS,
|
||||
SERVICE_TIER_SETTING_VALUES,
|
||||
SERVICE_TIER_OPENAI_OPTIONS,
|
||||
SERVICE_TIER_OPENAI_VALUES,
|
||||
} from "./service-tier";
|
||||
|
||||
/** Unified settings schema - single source of truth for all settings.
|
||||
@@ -1214,75 +1218,77 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
serviceTier: {
|
||||
"tier.openai": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_SETTING_VALUES,
|
||||
values: SERVICE_TIER_OPENAI_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier",
|
||||
label: "Service Tier — OpenAI",
|
||||
description:
|
||||
'Processing priority hint (none = omit). OpenAI accepts the tier values directly; Anthropic realizes `priority` as `speed: "fast"` on supported Opus models. Scoped values target one family.',
|
||||
options: SERVICE_TIER_OPTIONS,
|
||||
"Processing tier for OpenAI / OpenAI-Codex requests, and OpenAI-family models routed via OpenRouter (none = omit). Sent as `service_tier`.",
|
||||
options: SERVICE_TIER_OPENAI_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
serviceTierSubagent: {
|
||||
"tier.anthropic": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_ANTHROPIC_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier — Anthropic",
|
||||
description:
|
||||
'Processing tier for Claude requests. `priority` realizes fast mode (`speed: "fast"`) on supported direct Anthropic models; ignored on Bedrock/Vertex Claude and via OpenRouter.',
|
||||
options: SERVICE_TIER_ANTHROPIC_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
"tier.google": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_GOOGLE_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier — Google",
|
||||
description:
|
||||
"Processing tier for Gemini (Google AI Studio + Vertex) requests, and Google-family models routed via OpenRouter (none = omit). Sent as the top-level `serviceTier` field.",
|
||||
options: SERVICE_TIER_GOOGLE_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
"tier.subagent": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_INHERIT_SETTING_VALUES,
|
||||
default: "inherit",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier - Subagent",
|
||||
label: "Service Tier — Subagent",
|
||||
description:
|
||||
"Service Tier for spawned task/eval subagents. Inherit = match the main agent's live tier (tracks /fast); pick a value to scope subagents independently.",
|
||||
"Service Tier for spawned task/eval subagents. Inherit = match the main agent's live per-family tiers (tracks /fast); pick a value to apply it to whichever family the subagent's model belongs to.",
|
||||
options: SERVICE_TIER_INHERIT_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
serviceTierAdvisor: {
|
||||
"tier.advisor": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_INHERIT_SETTING_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier - Advisor",
|
||||
label: "Service Tier — Advisor",
|
||||
description:
|
||||
"Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live tier; pick a value (e.g. Priority) to run the advisor on a faster serving path.",
|
||||
"Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live per-family tiers; pick a value to apply it to the advisor model's family.",
|
||||
options: SERVICE_TIER_INHERIT_OPTIONS,
|
||||
condition: "advisorEnabled",
|
||||
},
|
||||
},
|
||||
|
||||
fastModeScope: {
|
||||
type: "enum",
|
||||
values: ["both", "openai", "claude"] as const,
|
||||
default: "both",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Fast Mode Scope",
|
||||
description:
|
||||
'Which providers `/fast on` (and the fast-mode toggle) target. "both" = priority on every supported provider; "openai"/"claude" scope it to one family (mirrors serviceTier openai-only/claude-only).',
|
||||
options: [
|
||||
{ value: "both", label: "Both", description: "Priority on every supported provider" },
|
||||
{
|
||||
value: "openai",
|
||||
label: "OpenAI only",
|
||||
description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere",
|
||||
},
|
||||
{
|
||||
value: "claude",
|
||||
label: "Claude only",
|
||||
description: "Anthropic fast mode on direct Claude requests; ignored elsewhere",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
// Retries
|
||||
"retry.enabled": { type: "boolean", default: true },
|
||||
|
||||
|
||||
@@ -1148,6 +1148,53 @@ export class Settings {
|
||||
// the incoherent "hashline edits without addressable anchors" state.
|
||||
delete raw.readHashLines;
|
||||
|
||||
// serviceTier (single enum with scoped openai-only/claude-only sentinels)
|
||||
// → per-family tier.openai/tier.anthropic/tier.google; serviceTierSubagent
|
||||
// → tier.subagent; serviceTierAdvisor → tier.advisor. `fastModeScope` is
|
||||
// dropped — per-family scoping is now expressed by the three tier settings.
|
||||
const tierObj = isRecord(raw.tier) ? raw.tier : {};
|
||||
let tierTouched = false;
|
||||
const setTier = (family: string, value: unknown): void => {
|
||||
if (value !== undefined && !(family in tierObj)) {
|
||||
tierObj[family] = value;
|
||||
tierTouched = true;
|
||||
}
|
||||
};
|
||||
if (typeof raw.serviceTier === "string") {
|
||||
switch (raw.serviceTier) {
|
||||
case "priority":
|
||||
setTier("openai", "priority");
|
||||
setTier("anthropic", "priority");
|
||||
setTier("google", "priority");
|
||||
break;
|
||||
case "openai-only":
|
||||
setTier("openai", "priority");
|
||||
break;
|
||||
case "claude-only":
|
||||
setTier("anthropic", "priority");
|
||||
break;
|
||||
case "auto":
|
||||
case "default":
|
||||
case "flex":
|
||||
case "scale":
|
||||
setTier("openai", raw.serviceTier);
|
||||
break;
|
||||
}
|
||||
delete raw.serviceTier;
|
||||
}
|
||||
const mapInheritTier = (value: unknown): unknown =>
|
||||
value === "openai-only" || value === "claude-only" ? "priority" : value;
|
||||
if ("serviceTierSubagent" in raw) {
|
||||
setTier("subagent", mapInheritTier(raw.serviceTierSubagent));
|
||||
delete raw.serviceTierSubagent;
|
||||
}
|
||||
if ("serviceTierAdvisor" in raw) {
|
||||
setTier("advisor", mapInheritTier(raw.serviceTierAdvisor));
|
||||
delete raw.serviceTierAdvisor;
|
||||
}
|
||||
if (tierTouched) raw.tier = tierObj;
|
||||
delete raw.fastModeScope;
|
||||
|
||||
return raw;
|
||||
}
|
||||
|
||||
|
||||
@@ -405,8 +405,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
|
||||
parentMnemopiSessionState: options.session.getMnemopiSessionState?.(),
|
||||
parentTelemetry: options.session.getTelemetry?.(),
|
||||
parentAgentId: options.session.getAgentId?.() ?? MAIN_AGENT_ID,
|
||||
// Live source of truth for `serviceTierSubagent: inherit` (null = explicit none).
|
||||
parentServiceTier: options.session.getServiceTier ? (options.session.getServiceTier() ?? null) : undefined,
|
||||
// Live source of truth for `tier.subagent: inherit` (null = explicit none).
|
||||
parentServiceTier: options.session.getServiceTierByFamily
|
||||
? (options.session.getServiceTierByFamily() ?? null)
|
||||
: undefined,
|
||||
// Deliberately omit parentEvalSessionId: the parent's Python kernel is
|
||||
// blocked on this bridge call, so sharing the eval session would deadlock
|
||||
// (subagent queues behind the parent's in-flight execution, parent waits
|
||||
|
||||
@@ -140,7 +140,7 @@ const HOST_DEFAULTED_SETTING_PATHS: SettingPath[] = [
|
||||
"advisor.subagents",
|
||||
"advisor.syncBacklog",
|
||||
"advisor.immuneTurns",
|
||||
"serviceTierAdvisor",
|
||||
"tier.advisor",
|
||||
];
|
||||
|
||||
const RPC_BACKGROUND_DEFAULTED_SETTING_PATHS: SettingPath[] = [
|
||||
|
||||
@@ -42,6 +42,7 @@ import {
|
||||
resolveModelRoleValue,
|
||||
} from "./config/model-resolver";
|
||||
import { loadPromptTemplates as loadPromptTemplatesInternal, type PromptTemplate } from "./config/prompt-templates";
|
||||
import { buildServiceTierByFamily } from "./config/service-tier";
|
||||
import { Settings, type SkillsSettings } from "./config/settings";
|
||||
import { CursorExecHandlers } from "./cursor";
|
||||
import "./discovery";
|
||||
@@ -1537,7 +1538,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
getModelString: () => (hasExplicitModel && model ? formatModelString(model) : undefined),
|
||||
getActiveModelString,
|
||||
getActiveModel: () => agent?.state.model ?? model,
|
||||
getServiceTier: () => session?.serviceTier,
|
||||
getServiceTierByFamily: () => session?.serviceTierByFamily,
|
||||
getImageAttachments: () => session?.getImageAttachments() ?? [],
|
||||
getPlanModeState: () => session?.getPlanModeState(),
|
||||
getPlanReferencePath: () => session?.getPlanReferencePath() ?? "local://PLAN.md",
|
||||
@@ -2525,13 +2526,13 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
const openaiWebsocketSetting = settings.get("providers.openaiWebsockets") ?? "off";
|
||||
const preferOpenAICodexWebsockets =
|
||||
openaiWebsocketSetting === "on" ? true : openaiWebsocketSetting === "off" ? false : undefined;
|
||||
const serviceTierSetting = settings.get("serviceTier");
|
||||
|
||||
const initialServiceTier = hasServiceTierEntry
|
||||
? existingSession.serviceTier
|
||||
: serviceTierSetting === "none"
|
||||
? undefined
|
||||
: serviceTierSetting;
|
||||
const initialServiceTierByFamily = hasServiceTierEntry
|
||||
? (existingSession.serviceTier ?? {})
|
||||
: buildServiceTierByFamily(
|
||||
settings.get("tier.openai"),
|
||||
settings.get("tier.anthropic"),
|
||||
settings.get("tier.google"),
|
||||
);
|
||||
|
||||
// One-shot launch-latency marker: fired the first time the loop dispatches
|
||||
// a chat request to the provider transport. See onFirstChatDispatch.
|
||||
@@ -2579,7 +2580,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
minP: settings.get("minP") >= 0 ? settings.get("minP") : undefined,
|
||||
presencePenalty: settings.get("presencePenalty") >= 0 ? settings.get("presencePenalty") : undefined,
|
||||
repetitionPenalty: settings.get("repetitionPenalty") >= 0 ? settings.get("repetitionPenalty") : undefined,
|
||||
serviceTier: initialServiceTier,
|
||||
hideThinkingSummary: settings.get("omitThinking"),
|
||||
kimiApiFormat: settings.get("providers.kimiApiFormat") ?? "anthropic",
|
||||
preferWebsockets: preferOpenAICodexWebsockets,
|
||||
@@ -2639,8 +2639,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
// classification persists its concrete effort once a real user turn runs.
|
||||
sessionManager.appendThinkingLevelChange(effectiveThinkingLevel);
|
||||
}
|
||||
if (initialServiceTier) {
|
||||
sessionManager.appendServiceTierChange(initialServiceTier);
|
||||
if (Object.keys(initialServiceTierByFamily).length > 0) {
|
||||
sessionManager.appendServiceTierChange(initialServiceTierByFamily);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2691,6 +2691,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
agent,
|
||||
pruneToolDescriptions: inlineToolDescriptors,
|
||||
thinkingLevel: autoThinking ? AUTO_THINKING : effectiveThinkingLevel,
|
||||
serviceTierByFamily: initialServiceTierByFamily,
|
||||
sessionManager,
|
||||
settings,
|
||||
autoApprove: options.autoApprove,
|
||||
|
||||
@@ -92,6 +92,8 @@ import type {
|
||||
ResetCreditRedeemOutcome,
|
||||
ResetCreditTarget,
|
||||
ServiceTier,
|
||||
ServiceTierByFamily,
|
||||
ServiceTierFamily,
|
||||
SimpleStreamOptions,
|
||||
TextContent,
|
||||
ToolCall,
|
||||
@@ -105,7 +107,9 @@ import {
|
||||
deriveClaudeDeviceId,
|
||||
Effort,
|
||||
parseRateLimitReason,
|
||||
resolveServiceTier,
|
||||
realizesPriorityServiceTier,
|
||||
resolveModelServiceTier,
|
||||
serviceTierFamily,
|
||||
streamSimple,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import * as AIError from "@oh-my-pi/pi-ai/error";
|
||||
@@ -168,7 +172,7 @@ import {
|
||||
} from "../config/model-resolver";
|
||||
import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles";
|
||||
import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates";
|
||||
import { resolveServiceTierSetting } from "../config/service-tier";
|
||||
import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier";
|
||||
import type { Settings, SkillsSettings } from "../config/settings";
|
||||
import { getDefault, onAppendOnlyModeChanged, validateProviderMaxInFlightRequests } from "../config/settings";
|
||||
import { RawSseDebugBuffer } from "../debug/raw-sse-buffer";
|
||||
@@ -506,6 +510,8 @@ export interface AgentSessionConfig {
|
||||
scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>;
|
||||
/** Initial session thinking selector. */
|
||||
thinkingLevel?: ConfiguredThinkingLevel;
|
||||
/** Initial per-family service tiers (OpenAI / Anthropic / Google) for the live session. */
|
||||
serviceTierByFamily?: ServiceTierByFamily;
|
||||
/** Prompt templates for expansion */
|
||||
promptTemplates?: PromptTemplate[];
|
||||
/** File-based slash commands for expansion */
|
||||
@@ -1787,6 +1793,7 @@ export class AgentSession {
|
||||
// toggle scopes priority to Fireworks alone, without mutating the shared
|
||||
// session `serviceTier` that drives `/fast` and OpenAI/Anthropic priority.
|
||||
this.agent.serviceTierResolver = model => this.#effectiveServiceTier(model);
|
||||
this.#serviceTierByFamily = config.serviceTierByFamily ?? {};
|
||||
this.#advisorTools = config.advisorTools;
|
||||
this.#advisorWatchdogPrompt = config.advisorWatchdogPrompt;
|
||||
this.#advisorSharedInstructions = config.advisorSharedInstructions;
|
||||
@@ -2045,15 +2052,20 @@ export class AgentSession {
|
||||
const legacy = !this.#advisorConfigs?.length;
|
||||
const roster: AdvisorConfig[] = legacy ? [{ name: "default" }] : this.#advisorConfigs!;
|
||||
|
||||
// Advisor service tier (`serviceTierAdvisor`): "none" (default) runs the
|
||||
// advisor on standard processing; "inherit" tracks the session's live tier
|
||||
// per request (like the main agent, including /fast toggles) via a resolver;
|
||||
// a concrete value pins the advisor to that tier. One value for all advisors.
|
||||
const advisorTierSetting = this.settings.get("serviceTierAdvisor");
|
||||
const advisorServiceTier =
|
||||
advisorTierSetting === "inherit" ? undefined : resolveServiceTierSetting(advisorTierSetting, undefined);
|
||||
const advisorServiceTierResolver =
|
||||
advisorTierSetting === "inherit" ? (model: Model) => this.#effectiveServiceTier(model) : undefined;
|
||||
// Advisor service tier (`tier.advisor`): "none" (default) runs the advisor
|
||||
// on standard processing; "inherit" tracks the session's live per-family
|
||||
// tiers per request (like the main agent, including /fast toggles); a
|
||||
// concrete value is broadcast across families and applied to the advisor
|
||||
// model's family. One value for all advisors.
|
||||
const advisorTierSetting = this.settings.get("tier.advisor");
|
||||
const advisorTierMap =
|
||||
advisorTierSetting === "inherit"
|
||||
? undefined
|
||||
: serviceTierForAllFamilies(serviceTierSettingToTier(advisorTierSetting));
|
||||
const advisorServiceTierResolver = (model: Model): ServiceTier | undefined =>
|
||||
advisorTierSetting === "inherit"
|
||||
? this.#effectiveServiceTier(model)
|
||||
: resolveModelServiceTier(advisorTierMap, model);
|
||||
|
||||
const usedSlugs = new Set<string>();
|
||||
for (const config of roster) {
|
||||
@@ -2152,7 +2164,7 @@ export class AgentSession {
|
||||
transformProviderContext: this.#transformProviderContext,
|
||||
intentTracing: false,
|
||||
telemetry: advisorTelemetry,
|
||||
serviceTier: advisorServiceTier,
|
||||
serviceTier: undefined,
|
||||
serviceTierResolver: advisorServiceTierResolver,
|
||||
});
|
||||
advisorAgent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel));
|
||||
@@ -3215,10 +3227,11 @@ export class AgentSession {
|
||||
if (event.message.role === "assistant") {
|
||||
this.#lastAssistantMessage = event.message;
|
||||
const assistantMsg = event.message as AssistantMessage;
|
||||
const currentGrantsAnthropicPriority =
|
||||
this.serviceTier === "priority" || this.serviceTier === "claude-only";
|
||||
if (assistantMsg.disabledFeatures?.includes("priority") && currentGrantsAnthropicPriority) {
|
||||
this.setServiceTier(undefined);
|
||||
if (
|
||||
assistantMsg.disabledFeatures?.includes("priority") &&
|
||||
this.#serviceTierByFamily.anthropic === "priority"
|
||||
) {
|
||||
this.setServiceTierFamily("anthropic", undefined);
|
||||
this.emitNotice(
|
||||
"warning",
|
||||
"Priority/fast mode rejected for this model; retried without it. Fast mode is now off.",
|
||||
@@ -5203,8 +5216,11 @@ export class AgentSession {
|
||||
return this.#autoResolvedLevel;
|
||||
}
|
||||
|
||||
get serviceTier(): ServiceTier | undefined {
|
||||
return this.agent.serviceTier;
|
||||
#serviceTierByFamily: ServiceTierByFamily = {};
|
||||
|
||||
/** Live per-family service tiers (OpenAI / Anthropic / Google). */
|
||||
get serviceTierByFamily(): ServiceTierByFamily {
|
||||
return this.#serviceTierByFamily;
|
||||
}
|
||||
|
||||
/** Whether agent is currently streaming a response */
|
||||
@@ -7892,7 +7908,7 @@ export class AgentSession {
|
||||
this.#scheduledHiddenNextTurnGeneration = undefined;
|
||||
|
||||
this.sessionManager.appendThinkingLevelChange(this.thinkingLevel, this.configuredThinkingLevel());
|
||||
this.sessionManager.appendServiceTierChange(this.serviceTier ?? null);
|
||||
this.sessionManager.appendServiceTierChange(this.#serviceTierEntry());
|
||||
if (nextDiscoverySessionToolNames) {
|
||||
await this.#applyActiveToolsByName(nextDiscoverySessionToolNames, { persistMCPSelection: false });
|
||||
if (this.getSelectedMCPToolNames().length > 0) {
|
||||
@@ -8434,38 +8450,36 @@ export class AgentSession {
|
||||
}
|
||||
|
||||
/**
|
||||
* True when *any* fast-mode-granting service tier is configured, regardless
|
||||
* of whether the active model's provider actually realizes it. Used by the
|
||||
* toggle (`/fast on|off`) so re-toggling a scoped tier (`openai-only`,
|
||||
* `claude-only`) doesn't silently broaden it to unscoped `priority`.
|
||||
* True when the currently selected model's family is set to `priority` — the
|
||||
* `/fast` on/off state for the active model. Returns false when no model is
|
||||
* selected or the model exposes no service-tier family (e.g. Fireworks, which
|
||||
* has its own Providers › Fireworks Tier toggle).
|
||||
*
|
||||
* For "is fast mode actually applied to the next request?" use
|
||||
* {@link isFastModeActive} instead — that one respects the model's provider.
|
||||
* For "is priority actually applied to the next request?" use
|
||||
* {@link isFastModeActive} instead.
|
||||
*/
|
||||
isFastModeEnabled(): boolean {
|
||||
return (
|
||||
this.serviceTier === "priority" || this.serviceTier === "claude-only" || this.serviceTier === "openai-only"
|
||||
);
|
||||
const family = this.model ? serviceTierFamily(this.model) : undefined;
|
||||
return family ? this.#serviceTierByFamily[family] === "priority" : false;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the configured `serviceTier` resolves to `"priority"` for the
|
||||
* *currently selected model's provider*. Returns false for scoped tiers
|
||||
* that don't match (e.g. `"openai-only"` on an anthropic model) and when
|
||||
* no model is selected.
|
||||
* True when `priority` is actually realized on the wire for the currently
|
||||
* selected model (OpenAI/Google `service_tier`, direct Anthropic fast mode,
|
||||
* or Fireworks priority). Returns false for tiers the active model can't
|
||||
* realize and when no model is selected.
|
||||
*/
|
||||
isFastModeActive(): boolean {
|
||||
return resolveServiceTier(this.#effectiveServiceTier(), this.model?.provider) === "priority";
|
||||
const model = this.model;
|
||||
return !!model && realizesPriorityServiceTier(this.#effectiveServiceTier(model), model);
|
||||
}
|
||||
|
||||
/**
|
||||
* Effective wire service-tier for a request to `model`. Fireworks models
|
||||
* take the Priority serving path only when the Providers › Fireworks Tier
|
||||
* setting is `"priority"` — that toggle is the sole opt-in, so a global
|
||||
* `serviceTier: "priority"` (for OpenAI/Anthropic) never silently incurs
|
||||
* Fireworks priority costs — and never for `-fast` variants, whose Fast
|
||||
* serving path is mutually exclusive with Priority. Every other provider
|
||||
* uses the session `serviceTier` unchanged.
|
||||
* Effective wire service-tier for a request to `model`. Fireworks models take
|
||||
* the Priority serving path only when the Providers › Fireworks Tier setting
|
||||
* is `"priority"` (and never for `-fast` variants, whose Fast serving path is
|
||||
* mutually exclusive with Priority). Every other model resolves the live
|
||||
* per-family tier map down to the entry for its family.
|
||||
*/
|
||||
#effectiveServiceTier(model: Model | undefined = this.model): ServiceTier | undefined {
|
||||
if (model?.provider === "fireworks") {
|
||||
@@ -8473,40 +8487,56 @@ export class AgentSession {
|
||||
? "priority"
|
||||
: undefined;
|
||||
}
|
||||
return this.serviceTier;
|
||||
if (!model) return undefined;
|
||||
return resolveModelServiceTier(this.#serviceTierByFamily, model);
|
||||
}
|
||||
|
||||
setServiceTier(serviceTier: ServiceTier | undefined): void {
|
||||
if (this.serviceTier === serviceTier) return;
|
||||
// Re-arming priority on Anthropic? Clear the per-session auto-fallback
|
||||
// sticky disable so the next request actually carries `speed: "fast"`
|
||||
// again. Without this, `/fast on` (or user switching to a tier that
|
||||
// grants anthropic priority) after an auto-disable is a silent no-op
|
||||
// and the warning notice fires every turn.
|
||||
if (serviceTier === "priority" || serviceTier === "claude-only") {
|
||||
/** The live per-family tier map, or `null` when empty (for session persistence). */
|
||||
#serviceTierEntry(): ServiceTierByFamily | null {
|
||||
return Object.keys(this.#serviceTierByFamily).length > 0 ? this.#serviceTierByFamily : null;
|
||||
}
|
||||
|
||||
/** Set one family's tier (or clear it with `undefined`); persists the change. */
|
||||
setServiceTierFamily(family: ServiceTierFamily, tier: ServiceTier | undefined): void {
|
||||
if (this.#serviceTierByFamily[family] === tier) return;
|
||||
const next: ServiceTierByFamily = { ...this.#serviceTierByFamily };
|
||||
if (tier) next[family] = tier;
|
||||
else delete next[family];
|
||||
this.#applyServiceTierByFamily(next);
|
||||
}
|
||||
|
||||
/** Replace the whole per-family tier map; persists + re-arms Anthropic fast mode. */
|
||||
#applyServiceTierByFamily(next: ServiceTierByFamily): void {
|
||||
// Re-arming Anthropic priority clears the per-session fast-mode auto-disable
|
||||
// so the next request actually carries `speed: "fast"` again.
|
||||
if (next.anthropic === "priority" && this.#serviceTierByFamily.anthropic !== "priority") {
|
||||
clearAnthropicFastModeFallback(this.#providerSessionState);
|
||||
}
|
||||
this.agent.serviceTier = serviceTier;
|
||||
this.sessionManager.appendServiceTierChange(serviceTier ?? null);
|
||||
this.#serviceTierByFamily = next;
|
||||
this.sessionManager.appendServiceTierChange(this.#serviceTierEntry());
|
||||
}
|
||||
|
||||
/**
|
||||
* `/fast on|off` targets the family of the currently selected model: it sets
|
||||
* (or clears) that family's `priority` tier. Models without a service-tier
|
||||
* family (Fireworks, or providers with no tier knob) have nothing to toggle.
|
||||
*/
|
||||
setFastMode(enabled: boolean): void {
|
||||
if (enabled && this.isFastModeEnabled()) {
|
||||
// Already on under any scope — keep the user's scoped value.
|
||||
const family = this.model ? serviceTierFamily(this.model) : undefined;
|
||||
if (!family) {
|
||||
this.emitNotice("info", "The current model has no service-tier control for /fast to toggle.", "priority");
|
||||
return;
|
||||
}
|
||||
if (!enabled) {
|
||||
this.setServiceTier(undefined);
|
||||
if (this.#serviceTierByFamily[family] === "priority") this.setServiceTierFamily(family, undefined);
|
||||
return;
|
||||
}
|
||||
const scope = this.settings.get("fastModeScope");
|
||||
this.setServiceTier(scope === "openai" ? "openai-only" : scope === "claude" ? "claude-only" : "priority");
|
||||
this.setServiceTierFamily(family, "priority");
|
||||
}
|
||||
|
||||
toggleFastMode(): boolean {
|
||||
const enabled = !this.isFastModeEnabled();
|
||||
this.setFastMode(enabled);
|
||||
return enabled;
|
||||
this.setFastMode(!this.isFastModeEnabled());
|
||||
return this.isFastModeEnabled();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -13236,7 +13266,7 @@ export class AgentSession {
|
||||
const previousThinkingLevel = this.#thinkingLevel;
|
||||
const previousAutoThinking = this.#autoThinking;
|
||||
const previousAutoResolvedLevel = this.#autoResolvedLevel;
|
||||
const previousServiceTier = this.agent.serviceTier;
|
||||
const previousServiceTierByFamily = this.#serviceTierByFamily;
|
||||
const previousSelectedMCPToolNames = new Set(this.#selectedMCPToolNames);
|
||||
const previousTools = [...this.agent.state.tools];
|
||||
const previousBaseSystemPrompt = this.#baseSystemPrompt;
|
||||
@@ -13322,7 +13352,11 @@ export class AgentSession {
|
||||
.getBranch()
|
||||
.some(entry => entry.type === "service_tier_change");
|
||||
const defaultThinkingLevel = parseConfiguredThinkingLevel(this.settings.get("defaultThinkingLevel"));
|
||||
const configuredServiceTier = this.settings.get("serviceTier");
|
||||
const configuredServiceTierByFamily = buildServiceTierByFamily(
|
||||
this.settings.get("tier.openai"),
|
||||
this.settings.get("tier.anthropic"),
|
||||
this.settings.get("tier.google"),
|
||||
);
|
||||
// Restore the thinking selector. Each change persists the configured
|
||||
// selector (`auto` or a concrete level), so prefer it: an `auto` session
|
||||
// resumes in auto mode (reclassifying the next turn) instead of freezing at
|
||||
@@ -13351,11 +13385,9 @@ export class AgentSession {
|
||||
this.#thinkingLevel = resolveThinkingLevelForModel(this.model, restoredThinkingLevel);
|
||||
}
|
||||
this.#applyThinkingLevelToAgent(this.#thinkingLevel);
|
||||
this.agent.serviceTier = hasServiceTierEntry
|
||||
? sessionContext.serviceTier
|
||||
: configuredServiceTier === "none"
|
||||
? undefined
|
||||
: configuredServiceTier;
|
||||
this.#serviceTierByFamily = hasServiceTierEntry
|
||||
? (sessionContext.serviceTier ?? {})
|
||||
: configuredServiceTierByFamily;
|
||||
|
||||
if (switchingToDifferentSession) {
|
||||
await this.#resetMemoryContextForNewTranscript();
|
||||
@@ -13412,7 +13444,7 @@ export class AgentSession {
|
||||
this.#autoThinking = previousAutoThinking;
|
||||
this.#autoResolvedLevel = previousAutoResolvedLevel;
|
||||
this.#applyThinkingLevelToAgent(previousThinkingLevel);
|
||||
this.agent.serviceTier = previousServiceTier;
|
||||
this.#serviceTierByFamily = previousServiceTierByFamily;
|
||||
this.#syncTodoPhasesFromBranch();
|
||||
this.#resetAllAdvisorRuntimes();
|
||||
this.#reconnectToAgent();
|
||||
@@ -14366,7 +14398,7 @@ export class AgentSession {
|
||||
const payload = {
|
||||
model: this.agent.state.model ?? null,
|
||||
thinkingLevel: this.#thinkingLevel ?? null,
|
||||
serviceTier: this.agent.serviceTier ?? null,
|
||||
serviceTier: this.#serviceTierEntry(),
|
||||
systemPrompt: this.agent.state.systemPrompt,
|
||||
tools: this.agent.state.tools.map(tool => ({
|
||||
name: tool.name,
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ProviderPayload, ServiceTier } from "@oh-my-pi/pi-ai";
|
||||
import { coerceServiceTierByFamily, type ProviderPayload, type ServiceTierByFamily } from "@oh-my-pi/pi-ai";
|
||||
import * as snapcompact from "@oh-my-pi/snapcompact";
|
||||
import { createBranchSummaryMessage, createCompactionSummaryMessage, createCustomMessage } from "./messages";
|
||||
import { type CompactionEntry, EPHEMERAL_MODEL_CHANGE_ROLE, type SessionEntry } from "./session-entries";
|
||||
@@ -9,7 +9,7 @@ export interface SessionContext {
|
||||
thinkingLevel?: string;
|
||||
/** Configured thinking selector (`"auto"` or a concrete level) from the latest change. */
|
||||
configuredThinkingLevel?: string;
|
||||
serviceTier?: ServiceTier;
|
||||
serviceTier?: ServiceTierByFamily;
|
||||
/** Model roles: { default: "provider/modelId", small: "provider/modelId", ... } */
|
||||
models: Record<string, string>;
|
||||
/** Names of TTSR rules that have been injected this session */
|
||||
@@ -138,7 +138,7 @@ export function buildSessionContext(
|
||||
// Extract settings and find compaction
|
||||
let thinkingLevel: string | undefined = "off";
|
||||
let configuredThinkingLevel: string | undefined;
|
||||
let serviceTier: ServiceTier | undefined;
|
||||
let serviceTier: ServiceTierByFamily | undefined;
|
||||
const models: Record<string, string> = {};
|
||||
let compaction: CompactionEntry | null = null;
|
||||
const injectedTtsrRulesSet = new Set<string>();
|
||||
@@ -169,7 +169,7 @@ export function buildSessionContext(
|
||||
}
|
||||
}
|
||||
} else if (entry.type === "service_tier_change") {
|
||||
serviceTier = entry.serviceTier ?? undefined;
|
||||
serviceTier = coerceServiceTierByFamily(entry.serviceTier);
|
||||
} else if (entry.type === "message" && entry.message.role === "assistant") {
|
||||
// Legacy fallback: infer default model from assistant messages only
|
||||
// when no explicit `model_change` (role=default) entry has been
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ImageContent, MessageAttribution, ServiceTier, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai";
|
||||
|
||||
export const CURRENT_SESSION_VERSION = 3;
|
||||
|
||||
@@ -73,7 +73,7 @@ export interface ModelChangeEntry extends SessionEntryBase {
|
||||
|
||||
export interface ServiceTierChangeEntry extends SessionEntryBase {
|
||||
type: "service_tier_change";
|
||||
serviceTier: ServiceTier | null;
|
||||
serviceTier: ServiceTierByFamily | null;
|
||||
}
|
||||
|
||||
export interface CompactionEntry<T = unknown> extends SessionEntryBase {
|
||||
|
||||
@@ -1,6 +1,13 @@
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import type { ImageContent, Message, MessageAttribution, ServiceTier, TextContent, Usage } from "@oh-my-pi/pi-ai";
|
||||
import type {
|
||||
ImageContent,
|
||||
Message,
|
||||
MessageAttribution,
|
||||
ServiceTierByFamily,
|
||||
TextContent,
|
||||
Usage,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
directoryExists,
|
||||
getBlobsDir,
|
||||
@@ -1286,7 +1293,7 @@ export class SessionManager {
|
||||
return entry.id;
|
||||
}
|
||||
|
||||
appendServiceTierChange(serviceTier: ServiceTier | null): string {
|
||||
appendServiceTierChange(serviceTier: ServiceTierByFamily | null): string {
|
||||
const entry: ServiceTierChangeEntry = { type: "service_tier_change", ...this.#freshEntryFields(), serviceTier };
|
||||
this.#recordEntry(entry);
|
||||
return entry.id;
|
||||
|
||||
@@ -73,17 +73,9 @@ function refreshStatusLine(ctx: InteractiveModeContext): void {
|
||||
ctx.ui.requestRender();
|
||||
}
|
||||
|
||||
/** `/fast status` label: "off", "on", or scope-qualified "on (… only)". */
|
||||
/** `/fast status` label for the active model: "on" when its family is priority, else "off". */
|
||||
function formatFastModeStatus(session: AgentSession): string {
|
||||
if (!session.isFastModeEnabled()) return "off";
|
||||
switch (session.serviceTier) {
|
||||
case "openai-only":
|
||||
return "on (OpenAI only)";
|
||||
case "claude-only":
|
||||
return "on (Claude only)";
|
||||
default:
|
||||
return "on";
|
||||
}
|
||||
return session.isFastModeEnabled() ? "on" : "off";
|
||||
}
|
||||
|
||||
const AUTOCOMPLETE_DETAIL_LIMIT = 48;
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
import path from "node:path";
|
||||
import type { AgentEvent, AgentIdentity, AgentTelemetryConfig, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import { recordHandoff, resolveTelemetry } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Api, Model, ServiceTier, Usage } from "@oh-my-pi/pi-ai";
|
||||
import type { Api, Model, ServiceTierByFamily, Usage } from "@oh-my-pi/pi-ai";
|
||||
import { logger, popLoopPhase, prompt, pushLoopPhase, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import type { Rule } from "../capability/rule";
|
||||
import { ModelRegistry } from "../config/model-registry";
|
||||
@@ -18,7 +18,7 @@ import {
|
||||
resolveModelOverrideWithAuthFallback,
|
||||
} from "../config/model-resolver";
|
||||
import type { PromptTemplate } from "../config/prompt-templates";
|
||||
import { resolveSubagentServiceTier } from "../config/service-tier";
|
||||
import { buildServiceTierByFamily, resolveSubagentServiceTier } from "../config/service-tier";
|
||||
import { Settings } from "../config/settings";
|
||||
import { SETTINGS_SCHEMA, type SettingPath } from "../config/settings-schema";
|
||||
import type { ToolPathWithSource } from "../extensibility/custom-tools";
|
||||
@@ -344,12 +344,12 @@ export interface ExecutorOptions {
|
||||
modelRegistry?: ModelRegistry;
|
||||
settings?: Settings;
|
||||
/**
|
||||
* Parent session's live effective service tier, the source of truth for a
|
||||
* subagent whose `serviceTierSubagent` is `"inherit"`. `null` = the parent
|
||||
* Parent session's live per-family service tiers, the source of truth for a
|
||||
* subagent whose `tier.subagent` is `"inherit"`. `null` = the parent
|
||||
* explicitly has no tier (e.g. `/fast off`); omitted = no live session, so
|
||||
* inherit falls back to the configured `serviceTier` setting.
|
||||
* inherit falls back to the subagent's configured `tier.*` settings.
|
||||
*/
|
||||
parentServiceTier?: ServiceTier | null;
|
||||
parentServiceTier?: ServiceTierByFamily | null;
|
||||
/** Override local:// protocol options so subagent shares parent's local:// root */
|
||||
localProtocolOptions?: LocalProtocolOptions;
|
||||
/**
|
||||
@@ -739,21 +739,28 @@ export function createMCPProxyTools(mcpManager: MCPManager): CustomTool[] {
|
||||
export function createSubagentSettings(
|
||||
baseSettings: Settings,
|
||||
overrides?: Partial<Record<SettingPath, unknown>>,
|
||||
inheritedServiceTier?: ServiceTier | null,
|
||||
inheritedServiceTier?: ServiceTierByFamily | null,
|
||||
): Settings {
|
||||
const snapshot: Partial<Record<SettingPath, unknown>> = {};
|
||||
for (const key of Object.keys(SETTINGS_SCHEMA) as SettingPath[]) {
|
||||
snapshot[key] = baseSettings.get(key);
|
||||
}
|
||||
// Resolve the subagent's service tier from `serviceTierSubagent` ("inherit" =
|
||||
// match the parent's live tier when a live session supplied one, else the
|
||||
// configured `serviceTier`). The result is stamped back onto the snapshot so
|
||||
// createAgentSession's `settings.get("serviceTier")` read picks it up.
|
||||
snapshot.serviceTier = resolveSubagentServiceTier(
|
||||
baseSettings.get("serviceTierSubagent"),
|
||||
baseSettings.get("serviceTier"),
|
||||
inheritedServiceTier,
|
||||
);
|
||||
// Resolve the subagent's per-family tiers from `tier.subagent` ("inherit" =
|
||||
// match the parent's live tiers when a live session supplied them, else the
|
||||
// subagent's own configured tier.* settings). The result is stamped back onto
|
||||
// the snapshot so createAgentSession's tier.* reads pick it up.
|
||||
const inheritedTiers =
|
||||
inheritedServiceTier === undefined
|
||||
? buildServiceTierByFamily(
|
||||
baseSettings.get("tier.openai"),
|
||||
baseSettings.get("tier.anthropic"),
|
||||
baseSettings.get("tier.google"),
|
||||
)
|
||||
: (inheritedServiceTier ?? {});
|
||||
const subagentTiers = resolveSubagentServiceTier(baseSettings.get("tier.subagent"), inheritedTiers);
|
||||
snapshot["tier.openai"] = subagentTiers.openai ?? "none";
|
||||
snapshot["tier.anthropic"] = subagentTiers.anthropic ?? "none";
|
||||
snapshot["tier.google"] = subagentTiers.google ?? "none";
|
||||
return Settings.isolated({
|
||||
...snapshot,
|
||||
"async.enabled": false,
|
||||
|
||||
@@ -1296,11 +1296,13 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
|
||||
parentTelemetry: this.session.getTelemetry?.(),
|
||||
parentEvalSessionId,
|
||||
parentAgentId: this.session.getAgentId?.() ?? MAIN_AGENT_ID,
|
||||
// Live source of truth for `serviceTierSubagent: inherit`. When the
|
||||
// session exposes a tier accessor, pass tier-or-null (null = explicit
|
||||
// none, e.g. /fast off); otherwise leave undefined so inherit falls
|
||||
// back to the configured serviceTier setting.
|
||||
parentServiceTier: this.session.getServiceTier ? (this.session.getServiceTier() ?? null) : undefined,
|
||||
// Live source of truth for `tier.subagent: inherit`. When the session
|
||||
// exposes a tier accessor, pass the per-family map or null (null =
|
||||
// explicit none, e.g. /fast off); otherwise leave undefined so inherit
|
||||
// falls back to the subagent's configured tier.* settings.
|
||||
parentServiceTier: this.session.getServiceTierByFamily
|
||||
? (this.session.getServiceTierByFamily() ?? null)
|
||||
: undefined,
|
||||
};
|
||||
|
||||
const runTask = async (): Promise<SingleResult> => {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import type { InMemorySnapshotStore } from "@oh-my-pi/hashline";
|
||||
import type { AgentTelemetryConfig, AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import type { FetchImpl, ImageContent, Model, ServiceTier, ToolChoice } from "@oh-my-pi/pi-ai";
|
||||
import type { FetchImpl, ImageContent, Model, ServiceTierByFamily, ToolChoice } from "@oh-my-pi/pi-ai";
|
||||
import { logger } from "@oh-my-pi/pi-utils";
|
||||
import type { AsyncJobManager } from "../async/job-manager";
|
||||
import type { Rule } from "../capability/rule";
|
||||
@@ -240,8 +240,8 @@ export interface ToolSession {
|
||||
getActiveModelString?: () => string | undefined;
|
||||
/** Get the current session model object (provider/api capabilities), regardless of how it was chosen. */
|
||||
getActiveModel?: () => Model | undefined;
|
||||
/** Get the session's live effective service tier (undefined = none). Source of truth for subagent `serviceTierSubagent: inherit`. */
|
||||
getServiceTier?: () => ServiceTier | undefined;
|
||||
/** Get the session's live per-family service tiers (undefined = none). Source of truth for subagent `tier.subagent: inherit`. */
|
||||
getServiceTierByFamily?: () => ServiceTierByFamily | undefined;
|
||||
/** Auth storage for passing to subagents (avoids re-discovery) */
|
||||
authStorage?: import("../session/auth-storage").AuthStorage;
|
||||
/** Model registry for passing to subagents (avoids re-discovery) */
|
||||
|
||||
@@ -452,7 +452,11 @@ describe("AgentSession handoff", () => {
|
||||
const fixedPreparation: compactionModule.CompactionPreparation = {
|
||||
firstKeptEntryId: lastEntryId,
|
||||
messagesToSummarize: [
|
||||
{ role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 },
|
||||
{
|
||||
role: "user",
|
||||
content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }],
|
||||
timestamp: 1,
|
||||
},
|
||||
],
|
||||
turnPrefixMessages: [],
|
||||
recentMessages: [],
|
||||
@@ -479,7 +483,11 @@ describe("AgentSession handoff", () => {
|
||||
const fixedPreparation: compactionModule.CompactionPreparation = {
|
||||
firstKeptEntryId: lastEntryId,
|
||||
messagesToSummarize: [
|
||||
{ role: "user", content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }], timestamp: 1 },
|
||||
{
|
||||
role: "user",
|
||||
content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }],
|
||||
timestamp: 1,
|
||||
},
|
||||
],
|
||||
turnPrefixMessages: [],
|
||||
recentMessages: [],
|
||||
|
||||
@@ -774,7 +774,7 @@ describe("AgentSession MCP discovery", () => {
|
||||
settings: Settings.isolated({
|
||||
"mcp.discoveryMode": true,
|
||||
defaultThinkingLevel: "high",
|
||||
serviceTier: "priority",
|
||||
"tier.openai": "priority",
|
||||
}),
|
||||
modelRegistry: {} as never,
|
||||
toolRegistry,
|
||||
@@ -789,10 +789,10 @@ describe("AgentSession MCP discovery", () => {
|
||||
|
||||
expect(session.getSelectedMCPToolNames()).toEqual(["mcp__docs_search"]);
|
||||
sessionManager.appendThinkingLevelChange(ThinkingLevel.High);
|
||||
sessionManager.appendServiceTierChange("flex");
|
||||
sessionManager.appendServiceTierChange({ openai: "flex" });
|
||||
sessionManager.appendMCPToolSelection(["mcp__docs_search"]);
|
||||
expect(sessionManager.buildSessionContext().thinkingLevel).toBe(ThinkingLevel.High);
|
||||
expect(sessionManager.buildSessionContext().serviceTier).toBe("flex");
|
||||
expect(sessionManager.buildSessionContext().serviceTier).toEqual({ openai: "flex" });
|
||||
expect(sessionManager.buildSessionContext().selectedMCPToolNames).toEqual(["mcp__docs_search"]);
|
||||
expect(sessionManager.buildSessionContext().hasPersistedMCPToolSelection).toBe(true);
|
||||
await sessionManager.rewriteEntries();
|
||||
@@ -803,7 +803,7 @@ describe("AgentSession MCP discovery", () => {
|
||||
await session.switchSession(olderSessionFile!);
|
||||
expect(session.sessionFile).toBe(olderSessionFile);
|
||||
expect(session.thinkingLevel).toBe(ThinkingLevel.Medium);
|
||||
expect(session.serviceTier).toBe("priority");
|
||||
expect(session.serviceTierByFamily).toEqual({ openai: "priority" });
|
||||
expect(session.getSelectedMCPToolNames()).toEqual([]);
|
||||
expect(session.getActiveToolNames()).toEqual(["read"]);
|
||||
expect(session.systemPrompt).toEqual(["tools:read"]);
|
||||
@@ -813,7 +813,7 @@ describe("AgentSession MCP discovery", () => {
|
||||
await session.switchSession(originalSessionFile!);
|
||||
expect(session.sessionFile).toBe(originalSessionFile);
|
||||
expect(session.thinkingLevel).toBe(ThinkingLevel.Medium);
|
||||
expect(session.serviceTier).toBe("flex");
|
||||
expect(session.serviceTierByFamily).toEqual({ openai: "flex" });
|
||||
expect(session.getSelectedMCPToolNames()).toEqual(["mcp__docs_search"]);
|
||||
expect(session.getActiveToolNames()).toEqual(["read", "mcp__docs_search"]);
|
||||
expect(session.systemPrompt).toEqual(["tools:read,mcp__docs_search"]);
|
||||
|
||||
@@ -173,13 +173,16 @@ describe("bench empty-output guard", () => {
|
||||
|
||||
function settingsStub(serviceTier: string | undefined): Settings | undefined {
|
||||
if (serviceTier === undefined) return undefined;
|
||||
return { get: (key: string) => (key === "serviceTier" ? serviceTier : undefined) } as unknown as Settings;
|
||||
return {
|
||||
get: (key: string) =>
|
||||
key === "tier.openai" ? serviceTier : key === "tier.anthropic" || key === "tier.google" ? "none" : undefined,
|
||||
} as unknown as Settings;
|
||||
}
|
||||
|
||||
async function captureServiceTier(opts: {
|
||||
flag?: string;
|
||||
setting?: string;
|
||||
}): Promise<{ wire: SimpleStreamOptions["serviceTier"]; summary: BenchSummary["serviceTier"] }> {
|
||||
}): Promise<{ wire: SimpleStreamOptions["serviceTier"]; summary: BenchSummary["serviceTierByFamily"] }> {
|
||||
const registry = fakeRegistry({ models: [fakeModel("openai-codex", "gpt-5.5")], authedProviders: ["openai-codex"] });
|
||||
let captured: SimpleStreamOptions | undefined;
|
||||
const summary = await runBenchCommand(
|
||||
@@ -205,7 +208,7 @@ async function captureServiceTier(opts: {
|
||||
stdoutIsTTY: false,
|
||||
},
|
||||
);
|
||||
return { wire: captured?.serviceTier, summary: summary.serviceTier };
|
||||
return { wire: captured?.serviceTier, summary: summary.serviceTierByFamily };
|
||||
}
|
||||
|
||||
describe("bench provider session state and websocket preference", () => {
|
||||
@@ -245,24 +248,24 @@ describe("bench service tier", () => {
|
||||
it("sends the configured serviceTier setting when no flag is passed", async () => {
|
||||
const { wire, summary } = await captureServiceTier({ setting: "flex" });
|
||||
expect(wire).toBe("flex");
|
||||
expect(summary).toBe("flex");
|
||||
expect(summary).toEqual({ openai: "flex" });
|
||||
});
|
||||
|
||||
it("lets an explicit --service-tier override the configured setting", async () => {
|
||||
const { wire, summary } = await captureServiceTier({ flag: "priority", setting: "flex" });
|
||||
expect(wire).toBe("priority");
|
||||
expect(summary).toBe("priority");
|
||||
expect(summary).toEqual({ openai: "priority", anthropic: "priority", google: "priority" });
|
||||
});
|
||||
|
||||
it("omits service_tier when the setting is none and no flag is passed", async () => {
|
||||
const { wire, summary } = await captureServiceTier({ setting: "none" });
|
||||
expect(wire).toBeUndefined();
|
||||
expect(summary).toBeUndefined();
|
||||
expect(summary).toEqual({});
|
||||
});
|
||||
|
||||
it("omits service_tier when neither flag nor settings are present", async () => {
|
||||
const { wire, summary } = await captureServiceTier({});
|
||||
expect(wire).toBeUndefined();
|
||||
expect(summary).toBeUndefined();
|
||||
expect(summary).toEqual({});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -9,9 +9,7 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
type FastModeScope = "both" | "openai" | "claude";
|
||||
|
||||
describe("fast mode scope", () => {
|
||||
describe("/fast targets the current model's service-tier family", () => {
|
||||
let tempDir: TempDir;
|
||||
let authStorage: AuthStorage;
|
||||
let session: AgentSession;
|
||||
@@ -29,77 +27,54 @@ describe("fast mode scope", () => {
|
||||
tempDir.removeSync();
|
||||
});
|
||||
|
||||
async function createSession(fastModeScope?: FastModeScope): Promise<AgentSession> {
|
||||
const model = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
async function createSession(provider: "anthropic" | "openai", modelId: string): Promise<AgentSession> {
|
||||
const model = getBundledModel(provider, modelId);
|
||||
if (!model) {
|
||||
throw new Error("Expected bundled test model to exist");
|
||||
throw new Error(`Expected bundled test model ${provider}/${modelId} to exist`);
|
||||
}
|
||||
|
||||
const settings = fastModeScope === undefined ? Settings.isolated() : Settings.isolated({ fastModeScope });
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] },
|
||||
});
|
||||
|
||||
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
|
||||
authStorage.setRuntimeApiKey(model.provider, "anthropic-token");
|
||||
authStorage.setRuntimeApiKey(model.provider, "token");
|
||||
modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
settings: Settings.isolated(),
|
||||
modelRegistry,
|
||||
});
|
||||
session.subscribe(() => {});
|
||||
return session;
|
||||
}
|
||||
|
||||
it("scopes enabled fast mode to OpenAI when configured", async () => {
|
||||
const session = await createSession("openai");
|
||||
|
||||
it("enables priority on the Anthropic family for a Claude model", async () => {
|
||||
const session = await createSession("anthropic", "claude-sonnet-4-5");
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("openai-only");
|
||||
expect(session.serviceTierByFamily).toEqual({ anthropic: "priority" });
|
||||
expect(session.isFastModeEnabled()).toBe(true);
|
||||
});
|
||||
|
||||
it("scopes enabled fast mode to Claude when configured", async () => {
|
||||
const session = await createSession("claude");
|
||||
|
||||
it("enables priority on the OpenAI family for an OpenAI model", async () => {
|
||||
const session = await createSession("openai", "gpt-5.2");
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("claude-only");
|
||||
expect(session.serviceTierByFamily).toEqual({ openai: "priority" });
|
||||
expect(session.isFastModeEnabled()).toBe(true);
|
||||
});
|
||||
|
||||
it("defaults enabled fast mode to priority for both providers", async () => {
|
||||
const session = await createSession();
|
||||
|
||||
it("clears only the current model's family when disabled", async () => {
|
||||
const session = await createSession("anthropic", "claude-sonnet-4-5");
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("priority");
|
||||
});
|
||||
|
||||
it("clears the service tier when disabled", async () => {
|
||||
const session = await createSession("openai");
|
||||
session.setFastMode(true);
|
||||
|
||||
session.setFastMode(false);
|
||||
|
||||
expect(session.serviceTier).toBeUndefined();
|
||||
expect(session.serviceTierByFamily).toEqual({});
|
||||
expect(session.isFastModeEnabled()).toBe(false);
|
||||
});
|
||||
|
||||
it("does not broaden an already enabled scoped tier", async () => {
|
||||
const session = await createSession("claude");
|
||||
session.setFastMode(true);
|
||||
expect(session.serviceTier).toBe("claude-only");
|
||||
session.settings.set("fastModeScope", "both");
|
||||
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("claude-only");
|
||||
it("toggle reports the resulting state", async () => {
|
||||
const session = await createSession("anthropic", "claude-sonnet-4-5");
|
||||
expect(session.toggleFastMode()).toBe(true);
|
||||
expect(session.serviceTierByFamily.anthropic).toBe("priority");
|
||||
expect(session.toggleFastMode()).toBe(false);
|
||||
expect(session.serviceTierByFamily.anthropic).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -331,7 +331,7 @@ describe("createAgentSession MCP discovery prompt gating", () => {
|
||||
settings: Settings.isolated({
|
||||
"mcp.discoveryMode": true,
|
||||
defaultThinkingLevel: "high",
|
||||
serviceTier: "priority",
|
||||
"tier.openai": "priority",
|
||||
}),
|
||||
model: createReasoningModel(),
|
||||
disableExtensionDiscovery: true,
|
||||
@@ -349,7 +349,7 @@ describe("createAgentSession MCP discovery prompt gating", () => {
|
||||
});
|
||||
await firstSession.activateDiscoveredMCPTools(["mcp__slack_post_message"]);
|
||||
firstSession.sessionManager.appendThinkingLevelChange(ThinkingLevel.Off);
|
||||
firstSession.sessionManager.appendServiceTierChange("priority");
|
||||
firstSession.sessionManager.appendServiceTierChange({ openai: "priority" });
|
||||
expect(firstSession.sessionManager.buildSessionContext().thinkingLevel).toBe(ThinkingLevel.Off);
|
||||
expect(firstSession.getSelectedMCPToolNames()).toEqual(["mcp__slack_post_message"]);
|
||||
const sessionFile = firstSession.sessionFile;
|
||||
@@ -368,7 +368,7 @@ describe("createAgentSession MCP discovery prompt gating", () => {
|
||||
settings: Settings.isolated({
|
||||
"mcp.discoveryMode": true,
|
||||
defaultThinkingLevel: "high",
|
||||
serviceTier: "none",
|
||||
"tier.openai": "none",
|
||||
}),
|
||||
model: createReasoningModel(),
|
||||
disableExtensionDiscovery: true,
|
||||
@@ -386,7 +386,7 @@ describe("createAgentSession MCP discovery prompt gating", () => {
|
||||
});
|
||||
try {
|
||||
expect(resumedSession.thinkingLevel).toBe(ThinkingLevel.Off);
|
||||
expect(resumedSession.serviceTier).toBe("priority");
|
||||
expect(resumedSession.serviceTierByFamily).toEqual({ openai: "priority" });
|
||||
expect(resumedSession.getSelectedMCPToolNames()).toEqual(["mcp__slack_post_message"]);
|
||||
expect(resumedSession.getActiveToolNames()).toEqual(
|
||||
expect.arrayContaining(["read", "search_tool_bm25", "mcp__slack_post_message"]),
|
||||
@@ -422,7 +422,7 @@ describe("createAgentSession MCP discovery prompt gating", () => {
|
||||
"mcp.discoveryMode": true,
|
||||
"mcp.discoveryDefaultServers": ["github"],
|
||||
defaultThinkingLevel: "high",
|
||||
serviceTier: "priority",
|
||||
"tier.openai": "priority",
|
||||
}),
|
||||
model: createReasoningModel(),
|
||||
disableExtensionDiscovery: true,
|
||||
@@ -440,7 +440,7 @@ describe("createAgentSession MCP discovery prompt gating", () => {
|
||||
});
|
||||
try {
|
||||
expect(session.thinkingLevel).toBe(ThinkingLevel.High);
|
||||
expect(session.serviceTier).toBe("priority");
|
||||
expect(session.serviceTierByFamily).toEqual({ openai: "priority" });
|
||||
expect(session.getSelectedMCPToolNames()).toEqual(["mcp__github_create_issue"]);
|
||||
expect(session.getActiveToolNames()).toEqual(
|
||||
expect.arrayContaining(["read", "search_tool_bm25", "mcp__github_create_issue"]),
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage";
|
||||
import { getProjectAgentDir, TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { YAML } from "bun";
|
||||
import { beginSettingsTest, restoreSettingsTestState, type SettingsTestState } from "./helpers/settings-test-state";
|
||||
|
||||
// Locks the back-compat migration of the legacy single `serviceTier` enum (with
|
||||
// scoped `openai-only`/`claude-only` sentinels) plus `serviceTierSubagent`/
|
||||
// `serviceTierAdvisor`/`fastModeScope` into the per-family `tier.*` settings.
|
||||
describe("serviceTier → tier.* settings migration", () => {
|
||||
let settingsState: SettingsTestState | undefined;
|
||||
let tempDir: TempDir;
|
||||
let agentDir: string;
|
||||
let projectDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
settingsState = beginSettingsTest();
|
||||
tempDir = TempDir.createSync("@test-service-tier-migration-");
|
||||
agentDir = path.join(tempDir.path(), "agent");
|
||||
projectDir = path.join(tempDir.path(), "project");
|
||||
fs.mkdirSync(agentDir, { recursive: true });
|
||||
fs.mkdirSync(getProjectAgentDir(projectDir), { recursive: true });
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
AgentStorage.resetInstance();
|
||||
restoreSettingsTestState(settingsState);
|
||||
settingsState = undefined;
|
||||
try {
|
||||
await tempDir.remove();
|
||||
} catch {}
|
||||
});
|
||||
|
||||
async function loadWith(raw: Record<string, unknown>): Promise<Settings> {
|
||||
await Bun.write(path.join(agentDir, "config.yml"), YAML.stringify(raw, null, 2));
|
||||
resetSettingsForTest();
|
||||
return Settings.init({ cwd: projectDir, agentDir });
|
||||
}
|
||||
|
||||
it("expands unscoped priority to every family", async () => {
|
||||
const settings = await loadWith({ serviceTier: "priority" });
|
||||
expect(settings.get("tier.openai")).toBe("priority");
|
||||
expect(settings.get("tier.anthropic")).toBe("priority");
|
||||
expect(settings.get("tier.google")).toBe("priority");
|
||||
});
|
||||
|
||||
it("scopes openai-only/claude-only to a single family", async () => {
|
||||
const openai = await loadWith({ serviceTier: "openai-only" });
|
||||
expect(openai.get("tier.openai")).toBe("priority");
|
||||
expect(openai.get("tier.anthropic")).toBe("none");
|
||||
expect(openai.get("tier.google")).toBe("none");
|
||||
|
||||
const claude = await loadWith({ serviceTier: "claude-only" });
|
||||
expect(claude.get("tier.anthropic")).toBe("priority");
|
||||
expect(claude.get("tier.openai")).toBe("none");
|
||||
});
|
||||
|
||||
it("maps plain OpenAI tiers onto the OpenAI family", async () => {
|
||||
const settings = await loadWith({ serviceTier: "flex" });
|
||||
expect(settings.get("tier.openai")).toBe("flex");
|
||||
expect(settings.get("tier.anthropic")).toBe("none");
|
||||
});
|
||||
|
||||
it("carries subagent/advisor over and drops scoped sentinels", async () => {
|
||||
const settings = await loadWith({
|
||||
serviceTierSubagent: "claude-only",
|
||||
serviceTierAdvisor: "flex",
|
||||
});
|
||||
expect(settings.get("tier.subagent")).toBe("priority"); // claude-only → priority
|
||||
expect(settings.get("tier.advisor")).toBe("flex");
|
||||
});
|
||||
|
||||
it("leaves a fresh config on the per-family defaults", async () => {
|
||||
const settings = await loadWith({});
|
||||
expect(settings.get("tier.openai")).toBe("none");
|
||||
expect(settings.get("tier.subagent")).toBe("inherit");
|
||||
expect(settings.get("tier.advisor")).toBe("none");
|
||||
});
|
||||
});
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Improved premium request calculation logic to account for specific model families
|
||||
|
||||
## [16.2.6] - 2026-06-29
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -1,6 +1,12 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import { type AssistantMessage, getPriorityPremiumRequests, type ServiceTier } from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
type AssistantMessage,
|
||||
coerceServiceTierByFamily,
|
||||
getPriorityPremiumRequests,
|
||||
resolveModelServiceTier,
|
||||
type ServiceTierByFamily,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { getSessionsDir, isEnoent } from "@oh-my-pi/pi-utils";
|
||||
import type {
|
||||
AgentType,
|
||||
@@ -130,7 +136,7 @@ function extractStats(
|
||||
sessionFile: string,
|
||||
folder: string,
|
||||
entry: SessionMessageEntry,
|
||||
currentServiceTier: ServiceTier | undefined,
|
||||
currentServiceTier: ServiceTierByFamily | undefined,
|
||||
agentType: AgentType,
|
||||
): MessageStats | null {
|
||||
const msg = entry.message as AssistantMessage;
|
||||
@@ -143,7 +149,9 @@ function extractStats(
|
||||
// non-zero value already in `usage.premiumRequests` (Copilot multipliers or
|
||||
// the new AI code path) and only synthesise when the field is missing/zero.
|
||||
const recorded = msg.usage.premiumRequests ?? 0;
|
||||
const derived = recorded > 0 ? recorded : getPriorityPremiumRequests(currentServiceTier, msg.provider);
|
||||
const model = { provider: msg.provider, api: msg.api, id: msg.model };
|
||||
const tier = resolveModelServiceTier(currentServiceTier, model);
|
||||
const derived = recorded > 0 ? recorded : getPriorityPremiumRequests(tier, model);
|
||||
const usage = derived === recorded ? msg.usage : { ...msg.usage, premiumRequests: derived };
|
||||
|
||||
return {
|
||||
@@ -206,10 +214,10 @@ function parseSessionEntriesLenient(bytes: Uint8Array): { entries: SessionEntry[
|
||||
return { entries, read };
|
||||
}
|
||||
|
||||
function scanLastServiceTier(bytes: Uint8Array): ServiceTier | undefined {
|
||||
let currentServiceTier: ServiceTier | undefined;
|
||||
function scanLastServiceTier(bytes: Uint8Array): ServiceTierByFamily | undefined {
|
||||
let currentServiceTier: ServiceTierByFamily | undefined;
|
||||
visitSessionEntriesLenient(bytes, entry => {
|
||||
if (isServiceTierChange(entry)) currentServiceTier = entry.serviceTier ?? undefined;
|
||||
if (isServiceTierChange(entry)) currentServiceTier = coerceServiceTierByFamily(entry.serviceTier);
|
||||
});
|
||||
return currentServiceTier;
|
||||
}
|
||||
@@ -253,13 +261,13 @@ export async function parseSessionFile(sessionPath: string, fromOffset = 0): Pro
|
||||
const start = Math.max(0, Math.min(fromOffset, bytes.length));
|
||||
const unprocessed = bytes.subarray(start);
|
||||
const { entries, read } = parseSessionEntriesLenient(unprocessed);
|
||||
let currentServiceTier: ServiceTier | undefined;
|
||||
let currentServiceTier: ServiceTierByFamily | undefined;
|
||||
if (start > 0) {
|
||||
currentServiceTier = scanLastServiceTier(bytes.subarray(0, start));
|
||||
}
|
||||
for (const entry of entries) {
|
||||
if (isServiceTierChange(entry)) {
|
||||
currentServiceTier = entry.serviceTier ?? undefined;
|
||||
currentServiceTier = coerceServiceTierByFamily(entry.serviceTier);
|
||||
continue;
|
||||
}
|
||||
if (isUserMessage(entry)) {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import type { AssistantMessage, ServiceTier, StopReason, Usage } from "@oh-my-pi/pi-ai";
|
||||
import type { AssistantMessage, ServiceTier, ServiceTierByFamily, StopReason, Usage } from "@oh-my-pi/pi-ai";
|
||||
import type { AgentType } from "./shared-types";
|
||||
|
||||
export * from "./shared-types";
|
||||
@@ -72,7 +72,7 @@ export interface SessionServiceTierChangeEntry {
|
||||
id: string;
|
||||
parentId?: string | null;
|
||||
timestamp: string;
|
||||
serviceTier: ServiceTier | null;
|
||||
serviceTier: ServiceTierByFamily | ServiceTier | null;
|
||||
}
|
||||
|
||||
export type SessionEntry = SessionHeader | SessionMessageEntry | SessionServiceTierChangeEntry | { type: string };
|
||||
|
||||
Reference in New Issue
Block a user