feat: improved context usage display and update model configurations
- Updated status line to display token usage with an unknown context marker (" 5K/? ") when the model context window is unavailable.
- Updated `fugu` model specifications in `models.json` and catalog constants with corrected pricing, increased context windows, and disabled stream idle timeouts.
- Corrected OpenAI usage accounting by excluding redundant orchestration input tokens in `openai-shared` logic.
This commit is contained in:
@@ -2419,7 +2419,7 @@ export function populateResponsesUsageFromResponse(
|
||||
const orchestrationInputCachedTokens = details?.orchestration_input_cached_tokens ?? 0;
|
||||
const orchestrationOutputTokens = outputDetails?.orchestration_output_tokens ?? 0;
|
||||
const accounting = calculateOpenAIUsageAccounting({
|
||||
promptTokens: (usage.input_tokens ?? 0) + orchestrationInputTokens + orchestrationInputCachedTokens,
|
||||
promptTokens: (usage.input_tokens ?? 0) + orchestrationInputTokens,
|
||||
outputTokens: (usage.output_tokens ?? 0) + orchestrationOutputTokens,
|
||||
cachedTokens: (details?.cached_tokens ?? usage.prompt_cache_hit_tokens ?? 0) + orchestrationInputCachedTokens,
|
||||
reasoningTokens: outputDetails?.reasoning_tokens ?? 0,
|
||||
|
||||
@@ -1,7 +1,10 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { applyAnthropicUsageExtras } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import { parseChunkUsage } from "@oh-my-pi/pi-ai/providers/openai-completions";
|
||||
import { calculateOpenAIUsageAccounting, populateResponsesUsageFromResponse } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import {
|
||||
calculateOpenAIUsageAccounting,
|
||||
populateResponsesUsageFromResponse,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import type { AssistantMessage, Model, Usage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
@@ -273,7 +276,7 @@ describe("openai-responses usage attribution", () => {
|
||||
populateResponsesUsageFromResponse(output, {
|
||||
input_tokens: 120,
|
||||
output_tokens: 80,
|
||||
total_tokens: 200,
|
||||
total_tokens: 270,
|
||||
input_tokens_details: {
|
||||
cached_tokens: 10,
|
||||
orchestration_input_tokens: 30,
|
||||
|
||||
@@ -60466,12 +60466,12 @@
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.66,
|
||||
"output": 3.5,
|
||||
"cacheRead": 0.33,
|
||||
"output": 3.41,
|
||||
"cacheRead": 0.144,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 65535,
|
||||
"maxTokens": 262144,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
@@ -64132,12 +64132,12 @@
|
||||
],
|
||||
"cost": {
|
||||
"input": 0.66,
|
||||
"output": 3.5,
|
||||
"cacheRead": 0.33,
|
||||
"output": 3.41,
|
||||
"cacheRead": 0.144,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": 262144,
|
||||
"maxTokens": 65535,
|
||||
"maxTokens": 262144,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"efforts": [
|
||||
@@ -69259,7 +69259,7 @@
|
||||
"cacheRead": 0,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": null,
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": null,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -69273,7 +69273,7 @@
|
||||
},
|
||||
"compat": {
|
||||
"includeEncryptedReasoning": false,
|
||||
"streamIdleTimeoutMs": 300000
|
||||
"streamIdleTimeoutMs": 0
|
||||
}
|
||||
},
|
||||
"fugu-ultra": {
|
||||
@@ -69292,7 +69292,7 @@
|
||||
"cacheRead": 0.5,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": null,
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": null,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -69306,7 +69306,7 @@
|
||||
},
|
||||
"compat": {
|
||||
"includeEncryptedReasoning": false,
|
||||
"streamIdleTimeoutMs": 300000
|
||||
"streamIdleTimeoutMs": 0
|
||||
}
|
||||
},
|
||||
"fugu-ultra-20260615": {
|
||||
@@ -69325,7 +69325,7 @@
|
||||
"cacheRead": 0.5,
|
||||
"cacheWrite": 0
|
||||
},
|
||||
"contextWindow": null,
|
||||
"contextWindow": 1000000,
|
||||
"maxTokens": null,
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
@@ -69339,7 +69339,7 @@
|
||||
},
|
||||
"compat": {
|
||||
"includeEncryptedReasoning": false,
|
||||
"streamIdleTimeoutMs": 300000
|
||||
"streamIdleTimeoutMs": 0
|
||||
}
|
||||
}
|
||||
},
|
||||
|
||||
@@ -2527,6 +2527,7 @@ export function moonshotModelManagerOptions(
|
||||
const SAKANA_DEFAULT_BASE_URL = "https://api.sakana.ai/v1";
|
||||
const SAKANA_FREE_ROUTER_COST = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 } as const;
|
||||
const SAKANA_FUGU_ULTRA_COST = { input: 5, output: 30, cacheRead: 0.5, cacheWrite: 0 } as const;
|
||||
const SAKANA_FUGU_ULTRA_CONTEXT_WINDOW = 1_000_000;
|
||||
const SAKANA_FUGU_THINKING: ThinkingConfig = {
|
||||
mode: "effort",
|
||||
efforts: [Effort.High, Effort.XHigh],
|
||||
@@ -2534,7 +2535,7 @@ const SAKANA_FUGU_THINKING: ThinkingConfig = {
|
||||
};
|
||||
const SAKANA_RESPONSES_COMPAT: ModelSpec<"openai-responses">["compat"] = {
|
||||
includeEncryptedReasoning: false,
|
||||
streamIdleTimeoutMs: 300_000,
|
||||
streamIdleTimeoutMs: 0,
|
||||
};
|
||||
|
||||
function normalizeSakanaBaseUrl(baseUrl: string | undefined): string {
|
||||
@@ -2551,6 +2552,7 @@ function createSakanaFuguStaticModel(
|
||||
id: string,
|
||||
name: string,
|
||||
cost: ModelSpec<"openai-responses">["cost"],
|
||||
contextWindow: number | null,
|
||||
): ModelSpec<"openai-responses"> {
|
||||
return {
|
||||
id,
|
||||
@@ -2561,7 +2563,7 @@ function createSakanaFuguStaticModel(
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { ...cost },
|
||||
contextWindow: null,
|
||||
contextWindow,
|
||||
maxTokens: null,
|
||||
thinking: { ...SAKANA_FUGU_THINKING },
|
||||
compat: { ...SAKANA_RESPONSES_COMPAT },
|
||||
@@ -2569,9 +2571,14 @@ function createSakanaFuguStaticModel(
|
||||
}
|
||||
|
||||
export const SAKANA_FUGU_STATIC_MODELS: readonly ModelSpec<"openai-responses">[] = [
|
||||
createSakanaFuguStaticModel("fugu", "Fugu", SAKANA_FREE_ROUTER_COST),
|
||||
createSakanaFuguStaticModel("fugu-ultra", "Fugu Ultra", SAKANA_FUGU_ULTRA_COST),
|
||||
createSakanaFuguStaticModel("fugu-ultra-20260615", "Fugu Ultra 20260615", SAKANA_FUGU_ULTRA_COST),
|
||||
createSakanaFuguStaticModel("fugu", "Fugu", SAKANA_FREE_ROUTER_COST, SAKANA_FUGU_ULTRA_CONTEXT_WINDOW),
|
||||
createSakanaFuguStaticModel("fugu-ultra", "Fugu Ultra", SAKANA_FUGU_ULTRA_COST, SAKANA_FUGU_ULTRA_CONTEXT_WINDOW),
|
||||
createSakanaFuguStaticModel(
|
||||
"fugu-ultra-20260615",
|
||||
"Fugu Ultra 20260615",
|
||||
SAKANA_FUGU_ULTRA_COST,
|
||||
SAKANA_FUGU_ULTRA_CONTEXT_WINDOW,
|
||||
),
|
||||
];
|
||||
|
||||
const SAKANA_FUGU_STATIC_MODEL_BY_ID = new Map(SAKANA_FUGU_STATIC_MODELS.map(model => [model.id, model] as const));
|
||||
|
||||
@@ -51,11 +51,15 @@ describe("Sakana AI provider support", () => {
|
||||
|
||||
const bundled = getBundledModels("sakana");
|
||||
expect(bundled.map(model => model.id).sort()).toEqual(["fugu", "fugu-ultra", "fugu-ultra-20260615"]);
|
||||
expect(bundled.find(model => model.id === "fugu")?.contextWindow).toBe(1_000_000);
|
||||
expect(bundled.find(model => model.id === "fugu-ultra")?.contextWindow).toBe(1_000_000);
|
||||
expect(bundled.find(model => model.id === "fugu-ultra-20260615")?.contextWindow).toBe(1_000_000);
|
||||
for (const model of bundled) {
|
||||
expect(model.api).toBe("openai-responses");
|
||||
expect(model.thinking?.efforts).toEqual([Effort.High, Effort.XHigh]);
|
||||
expect(model.thinking?.effortMap?.[Effort.XHigh]).toBe("max");
|
||||
expect((model.compat as ResolvedOpenAIResponsesCompat).includeEncryptedReasoning).toBe(false);
|
||||
expect((model.compat as ResolvedOpenAIResponsesCompat).streamIdleTimeoutMs).toBe(0);
|
||||
}
|
||||
|
||||
const provider = getOAuthProviders().find(item => item.id === "sakana");
|
||||
|
||||
@@ -142,7 +142,8 @@ export class FooterComponent implements Component {
|
||||
// After compaction, tokens are unknown until the next LLM response.
|
||||
const contextUsage = this.session.getContextUsage();
|
||||
const contextWindow = contextUsage?.contextWindow ?? state.model?.contextWindow ?? 0;
|
||||
const contextPercentValue = contextUsage?.percent ?? 0;
|
||||
const contextTokens = contextUsage?.tokens ?? 0;
|
||||
const contextPercentValue = contextWindow > 0 ? (contextUsage?.percent ?? 0) : null;
|
||||
|
||||
// Replace home directory with ~
|
||||
let pwd = shortenPath(getProjectDir());
|
||||
@@ -186,8 +187,8 @@ export class FooterComponent implements Component {
|
||||
// Colorize context percentage based on usage
|
||||
let contextPercentStr: string;
|
||||
const autoIndicator = this.#autoCompactEnabled ? " (auto)" : "";
|
||||
const contextPercentDisplay = `${formatContextUsage(contextPercentValue, contextWindow)}${autoIndicator}`;
|
||||
if (contextUsage) {
|
||||
const contextPercentDisplay = `${formatContextUsage(contextPercentValue, contextWindow, contextTokens)}${autoIndicator}`;
|
||||
if (contextUsage && contextPercentValue !== null) {
|
||||
const color = getContextUsageThemeColor(getContextUsageLevel(contextPercentValue, contextWindow));
|
||||
contextPercentStr =
|
||||
color === "statusLineContext" ? contextPercentDisplay : theme.fg(color, contextPercentDisplay);
|
||||
|
||||
@@ -727,10 +727,12 @@ export class StatusLineComponent implements Component {
|
||||
|
||||
let contextWindow = state.model?.contextWindow ?? this.session.model?.contextWindow ?? 0;
|
||||
let contextPercent: number | null = 0;
|
||||
let contextTokens = 0;
|
||||
if (includeContext) {
|
||||
const breakdown = this.getCachedContextBreakdown();
|
||||
contextTokens = breakdown.usedTokens;
|
||||
contextWindow = breakdown.contextWindow || contextWindow;
|
||||
contextPercent = contextWindow > 0 ? (breakdown.usedTokens / contextWindow) * 100 : 0;
|
||||
contextPercent = contextWindow > 0 ? (breakdown.usedTokens / contextWindow) * 100 : null;
|
||||
}
|
||||
|
||||
// Collab guest: context comes from the host's state frames — the local
|
||||
@@ -738,6 +740,7 @@ export class StatusLineComponent implements Component {
|
||||
const collabState = this.#collabStatus?.stateOverride;
|
||||
if (collabState?.contextUsage) {
|
||||
contextWindow = collabState.contextUsage.contextWindow || contextWindow;
|
||||
contextTokens = collabState.contextUsage.tokens ?? contextTokens;
|
||||
contextPercent = collabState.contextUsage.percent ?? contextPercent;
|
||||
}
|
||||
|
||||
@@ -756,6 +759,7 @@ export class StatusLineComponent implements Component {
|
||||
collab: this.#collabStatus,
|
||||
usageStats,
|
||||
contextPercent,
|
||||
contextTokens,
|
||||
contextWindow,
|
||||
autoCompactEnabled: this.#autoCompactEnabled,
|
||||
subagentCount: this.#subagentCount,
|
||||
|
||||
@@ -56,10 +56,18 @@ export function getContextUsageLevel(contextPercent: number, contextWindow: numb
|
||||
}
|
||||
|
||||
/**
|
||||
* Format context usage as `<percent>%/<window>` (e.g. `5.1%/1M`), matching the
|
||||
* status line's context gauge so subagent and footer renderers stay in sync.
|
||||
* Format context usage as `<percent>%/<window>` when the model window is known.
|
||||
* Unknown windows render as `<tokens>/?`, because `0.0%/0` suggests a real
|
||||
* empty context instead of missing provider metadata.
|
||||
*/
|
||||
export function formatContextUsage(contextPercent: number | null | undefined, contextWindow: number): string {
|
||||
export function formatContextUsage(
|
||||
contextPercent: number | null | undefined,
|
||||
contextWindow: number,
|
||||
usedTokens?: number,
|
||||
): string {
|
||||
if (!Number.isFinite(contextWindow) || contextWindow <= 0) {
|
||||
return `${formatNumber(usedTokens ?? 0)}/?`;
|
||||
}
|
||||
const pct = contextPercent === null || contextPercent === undefined ? "?" : `${contextPercent.toFixed(1)}%`;
|
||||
return `${pct}/${formatNumber(contextWindow)}`;
|
||||
}
|
||||
|
||||
@@ -375,7 +375,7 @@ const contextPctSegment: StatusLineSegment = {
|
||||
const window = ctx.contextWindow;
|
||||
|
||||
const autoIcon = ctx.autoCompactEnabled && theme.icon.auto ? ` ${theme.icon.auto}` : "";
|
||||
const text = `${formatContextUsage(pct, window)}${autoIcon}`;
|
||||
const text = `${formatContextUsage(pct, window, ctx.contextTokens)}${autoIcon}`;
|
||||
|
||||
const color = getContextUsageThemeColor(getContextUsageLevel(pct ?? 0, window));
|
||||
const content = withIcon(theme.icon.context, theme.fg(color, text));
|
||||
|
||||
@@ -73,6 +73,7 @@ export interface SegmentContext {
|
||||
};
|
||||
/** Context usage percent, or null when unknown (e.g. right after compaction). */
|
||||
contextPercent: number | null;
|
||||
contextTokens: number;
|
||||
contextWindow: number;
|
||||
autoCompactEnabled: boolean;
|
||||
subagentCount: number;
|
||||
|
||||
@@ -12168,8 +12168,8 @@ export class AgentSession {
|
||||
pendingMessages?: AgentMessage[];
|
||||
}): ContextUsageBreakdown | undefined {
|
||||
const model = this.model;
|
||||
const contextWindow = options?.contextWindow ?? model?.contextWindow ?? 0;
|
||||
if (!Number.isFinite(contextWindow) || contextWindow <= 0) return undefined;
|
||||
const rawContextWindow = options?.contextWindow ?? model?.contextWindow ?? 0;
|
||||
const contextWindow = Number.isFinite(rawContextWindow) && rawContextWindow > 0 ? rawContextWindow : 0;
|
||||
|
||||
const { skillsTokens, toolsTokens, systemContextTokens, systemPromptTokens } = computeNonMessageBreakdown(this);
|
||||
const categoryNonMessageTokens = skillsTokens + toolsTokens + systemContextTokens + systemPromptTokens;
|
||||
|
||||
@@ -32,6 +32,7 @@ function createCtx(usage: Partial<SegmentContext["usageStats"]>): SegmentContext
|
||||
...usage,
|
||||
},
|
||||
contextPercent: 0,
|
||||
contextTokens: 0,
|
||||
contextWindow: 0,
|
||||
autoCompactEnabled: false,
|
||||
subagentCount: 0,
|
||||
|
||||
@@ -248,4 +248,23 @@ describe("StatusLineComponent context breakdown", () => {
|
||||
const plain = comp.getTopBorder(80).content.replaceAll(/\x1b\[[0-9;]*m/g, "");
|
||||
expect(plain).toContain("0.5%/272K");
|
||||
});
|
||||
|
||||
it("renders token usage with an unknown marker when the model window is unavailable", () => {
|
||||
const { session } = makeSession({
|
||||
messages: [userMessage("hi")],
|
||||
contextWindow: 0,
|
||||
usage: { tokens: 5000, contextWindow: 0, percent: 0 },
|
||||
});
|
||||
const comp = new StatusLineComponent(session);
|
||||
comp.updateSettings({
|
||||
preset: "custom",
|
||||
leftSegments: ["context_pct"],
|
||||
rightSegments: [],
|
||||
separator: "powerline-thin",
|
||||
});
|
||||
|
||||
const plain = comp.getTopBorder(80).content.replaceAll(/\x1b\[[0-9;]*m/g, "");
|
||||
expect(plain).toContain("5K/?");
|
||||
expect(plain).not.toContain("0.0%/0");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -32,6 +32,7 @@ function createModelContext(advisorActive: boolean): SegmentContext {
|
||||
tokensPerSecond: null,
|
||||
},
|
||||
contextPercent: 0,
|
||||
contextTokens: 0,
|
||||
contextWindow: 0,
|
||||
autoCompactEnabled: false,
|
||||
subagentCount: 0,
|
||||
|
||||
@@ -56,6 +56,7 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null
|
||||
tokensPerSecond: null,
|
||||
},
|
||||
contextPercent: 0,
|
||||
contextTokens: 0,
|
||||
contextWindow: 0,
|
||||
autoCompactEnabled: false,
|
||||
subagentCount: 0,
|
||||
|
||||
@@ -42,6 +42,7 @@ function createPathContext(): SegmentContext {
|
||||
tokensPerSecond: null,
|
||||
},
|
||||
contextPercent: 0,
|
||||
contextTokens: 0,
|
||||
contextWindow: 0,
|
||||
autoCompactEnabled: false,
|
||||
subagentCount: 0,
|
||||
|
||||
Reference in New Issue
Block a user