Merge PR #4471: fix(ai): separate Codex orchestration usage (@roboomp)

This commit is contained in:
can1357
2026-07-05 13:03:08 +02:00
38 changed files with 399 additions and 30 deletions
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed `calculateContextTokens` including provider orchestration tokens in context sizing, which could trigger premature auto-compaction and context promotion on Codex/Fugu turns with sizable provider-side orchestration. ([#4469](https://github.com/can1357/oh-my-pi/issues/4469))
## [16.3.3] - 2026-07-02
### Changed
+9 -1
View File
@@ -204,9 +204,17 @@ export const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {
/**
* Calculate total context tokens from usage.
* Uses the native totalTokens field when available, falls back to computing from components.
* Provider-side orchestration tokens are billable but never replay into the
* conversation prefix, so they are excluded from context sizing to keep
* auto-compaction and context-promotion thresholds honest.
*/
export function calculateContextTokens(usage: Usage): number {
return usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;
const orchestration = usage.orchestration;
const orchestrationTotal = orchestration
? (orchestration.input ?? 0) + (orchestration.output ?? 0) + (orchestration.cacheRead ?? 0)
: 0;
const raw = usage.totalTokens || usage.input + usage.output + usage.cacheRead + usage.cacheWrite;
return Math.max(0, raw - orchestrationTotal);
}
export function calculatePromptTokens(usage: Usage): number {
@@ -0,0 +1,37 @@
import { describe, expect, it } from "bun:test";
import { calculateContextTokens, calculatePromptTokens } from "@oh-my-pi/pi-agent-core/compaction";
import type { Usage } from "@oh-my-pi/pi-ai";
function usage(overrides: Partial<Usage>): Usage {
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
...overrides,
};
}
describe("calculateContextTokens", () => {
it("excludes provider orchestration tokens from context sizing", () => {
// Codex-style turn: conversation prefix is ~186k, orchestration adds 5.5k;
// context sizing must stay on the conversation, not the billable total.
const u = usage({
input: 5_517,
output: 29,
cacheRead: 181_248,
cacheWrite: 0,
totalTokens: 186_794 + 5_629,
orchestration: { input: 5_629 },
});
expect(calculateContextTokens(u)).toBe(186_794);
expect(calculatePromptTokens(u)).toBe(5_517 + 181_248);
});
it("keeps native totalTokens when no orchestration sidecar is present", () => {
const u = usage({ input: 10, output: 5, cacheRead: 100, cacheWrite: 0, totalTokens: 115 });
expect(calculateContextTokens(u)).toBe(115);
});
});
+1
View File
@@ -32,6 +32,7 @@
### Fixed
- Fixed tool-call validation to strip stray trailing line terminators on schema-matching enum values and on well-known identifier fields (`path`, `paths`, `file`, `file_path`, `url`, `uri`, `title`, `label`) before dispatch, keeping ordinary trailing spaces and content-carrying fields (`content`, `input`, `code`, `command`, etc.) intact ([#4461](https://github.com/can1357/oh-my-pi/issues/4461)).
- Fixed OpenAI Responses/Codex orchestration token accounting so provider-side orchestration tokens stay billable and included in totals without appearing as ordinary uncached prompt input. ([#4469](https://github.com/can1357/oh-my-pi/issues/4469))
## [16.3.4] - 2026-07-03
+10 -3
View File
@@ -446,10 +446,17 @@ function mergeUsage(partial?: Partial<Omit<Usage, "cost">> & { cost?: Partial<Us
if (costProvided) {
merged.cost = { ...base.cost, ...partial.cost } as Usage["cost"];
}
// Recompute totalTokens when not explicitly provided (canonical formula matches types.ts:
// input + output + cacheRead + cacheWrite).
// Recompute totalTokens when not explicitly provided (canonical formula matches types.ts).
if (partial.totalTokens === undefined) {
merged.totalTokens = merged.input + merged.output + merged.cacheRead + merged.cacheWrite;
const orchestration = merged.orchestration;
merged.totalTokens =
merged.input +
merged.output +
merged.cacheRead +
merged.cacheWrite +
(orchestration?.input ?? 0) +
(orchestration?.output ?? 0) +
(orchestration?.cacheRead ?? 0);
}
// Recompute cost.total when cost components were supplied without an explicit total.
if (costProvided && partial.cost?.total === undefined) {
@@ -1533,8 +1533,15 @@ class CodexStreamProcessor {
input_tokens?: number;
output_tokens?: number;
total_tokens?: number;
input_tokens_details?: { cached_tokens?: number };
output_tokens_details?: { reasoning_tokens?: number };
input_tokens_details?: {
cached_tokens?: number;
orchestration_input_tokens?: number;
orchestration_input_cached_tokens?: number;
};
output_tokens_details?: {
reasoning_tokens?: number;
orchestration_output_tokens?: number;
};
};
status?: string;
service_tier?: ServiceTier | "default";
+28 -3
View File
@@ -50,6 +50,7 @@ import {
type Tool,
type ToolCall,
type ToolResultMessage,
type Usage,
} from "../types";
import {
getOpenAIResponsesHistoryItems,
@@ -343,6 +344,7 @@ export interface OpenAIUsageAccounting {
cacheWrite: number;
totalTokens: number;
reasoningTokens?: number;
orchestration?: Usage["orchestration"];
}
export function calculateOpenAIUsageAccounting(accounting: OpenAIUsageAccountingInput): OpenAIUsageAccounting {
@@ -2536,19 +2538,42 @@ export function populateResponsesUsageFromResponse(
if (!usage) return;
const details = usage.input_tokens_details;
const outputDetails = usage.output_tokens_details;
const reportedInputTokens = usage.input_tokens ?? 0;
const reportedOutputTokens = usage.output_tokens ?? 0;
const reportedCachedTokens = details?.cached_tokens ?? usage.prompt_cache_hit_tokens ?? 0;
const orchestrationInputTokens = details?.orchestration_input_tokens ?? 0;
const orchestrationInputCachedTokens = details?.orchestration_input_cached_tokens ?? 0;
const orchestrationOutputTokens = outputDetails?.orchestration_output_tokens ?? 0;
const reportedTotalTokens = typeof usage.total_tokens === "number" ? usage.total_tokens : undefined;
const reportedPrimaryTokens = reportedInputTokens + reportedOutputTokens;
const reportedWithSeparateOrchestration =
reportedPrimaryTokens + orchestrationInputTokens + orchestrationOutputTokens;
const primaryIncludesOrchestration =
reportedTotalTokens !== undefined &&
orchestrationInputTokens + orchestrationOutputTokens > 0 &&
Math.abs(reportedTotalTokens - reportedPrimaryTokens) <=
Math.abs(reportedTotalTokens - reportedWithSeparateOrchestration);
const orchestrationInputCached = Math.min(orchestrationInputTokens, orchestrationInputCachedTokens);
const orchestrationInput = Math.max(0, orchestrationInputTokens - orchestrationInputCached);
const accounting = calculateOpenAIUsageAccounting({
promptTokens: (usage.input_tokens ?? 0) + orchestrationInputTokens,
outputTokens: (usage.output_tokens ?? 0) + orchestrationOutputTokens,
cachedTokens: (details?.cached_tokens ?? usage.prompt_cache_hit_tokens ?? 0) + orchestrationInputCachedTokens,
promptTokens: Math.max(0, reportedInputTokens - (primaryIncludesOrchestration ? orchestrationInputTokens : 0)),
outputTokens: Math.max(0, reportedOutputTokens - (primaryIncludesOrchestration ? orchestrationOutputTokens : 0)),
cachedTokens: Math.max(0, reportedCachedTokens - (primaryIncludesOrchestration ? orchestrationInputCached : 0)),
reasoningTokens: outputDetails?.reasoning_tokens ?? 0,
cacheWriteOpenRouter: details?.cache_write_tokens ?? undefined,
cacheWriteDeepSeek: usage.prompt_cache_miss_tokens ?? undefined,
hasDeepSeekCacheHitAndMiss:
usage.prompt_cache_hit_tokens !== undefined && usage.prompt_cache_miss_tokens !== undefined,
});
const orchestrationTotal = orchestrationInput + orchestrationInputCached + orchestrationOutputTokens;
if (orchestrationTotal > 0) {
accounting.orchestration = {
...(orchestrationInput > 0 ? { input: orchestrationInput } : {}),
...(orchestrationInputCached > 0 ? { cacheRead: orchestrationInputCached } : {}),
...(orchestrationOutputTokens > 0 ? { output: orchestrationOutputTokens } : {}),
};
accounting.totalTokens = reportedTotalTokens ?? accounting.totalTokens + orchestrationTotal;
}
// Wholesale replacement must not drop provider-annotated extras (Copilot
// premium-request accounting): the failed/cancelled paths throw right after
+32
View File
@@ -71,6 +71,38 @@ describe("calculateCost", () => {
expect(usage.cost.total).toBeCloseTo(2.18, 8);
});
it("prices provider orchestration tokens without changing visible usage buckets", () => {
const model = {
...getBundledModel("openai", "gpt-4o-mini"),
cost: {
input: 1000,
output: 2000,
cacheRead: 500,
cacheWrite: 800,
},
};
const usage: Usage = {
input: 100,
output: 20,
cacheRead: 50,
cacheWrite: 10,
totalTokens: 250,
orchestration: { input: 25, output: 40, cacheRead: 5 },
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
calculateCost(model, usage);
expect(usage.input).toBe(100);
expect(usage.output).toBe(20);
expect(usage.cacheRead).toBe(50);
expect(usage.cost.input).toBeCloseTo(0.125, 8);
expect(usage.cost.output).toBeCloseTo(0.12, 8);
expect(usage.cost.cacheRead).toBeCloseTo(0.0275, 8);
expect(usage.cost.cacheWrite).toBeCloseTo(0.008, 8);
expect(usage.cost.total).toBeCloseTo(0.2805, 8);
});
it("prices OpenAI Codex GPT models from the matching OpenAI catalog entry", () => {
const openAIModel = getBundledModel("openai", "gpt-5.4");
const codexModel = getBundledModel("openai-codex", "gpt-5.4");
@@ -952,6 +952,58 @@ describe("openai-codex streaming", () => {
}
});
it("separates websocket terminal orchestration usage from prompt cache buckets", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
const token = createCodexTestToken();
class UsageWebSocket extends MockWebSocket {
constructor(url: string, options?: { headers?: WsHeaders }) {
super(url, options);
this.scheduleOpen();
}
send(): void {
this.sendJson({
type: "response.done",
response: {
id: "resp_usage",
status: "completed",
usage: {
input_tokens: 185_853,
output_tokens: 29,
total_tokens: 185_882,
input_tokens_details: {
cached_tokens: 180_224,
orchestration_input_tokens: 5_629,
orchestration_input_cached_tokens: 0,
},
},
},
});
}
}
global.WebSocket = UsageWebSocket as unknown as typeof WebSocket;
const model = {
...createCodexTestModel("https://chatgpt.com/backend-api"),
cost: { input: 1000, output: 2000, cacheRead: 500, cacheWrite: 0 },
};
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
apiKey: token,
sessionId: "ws-orchestration-usage-session",
providerSessionState: new Map<string, ProviderSessionState>(),
}).result();
expect(result.usage.input).toBe(0);
expect(result.usage.cacheRead).toBe(180_224);
expect(result.usage.output).toBe(29);
expect(result.usage.orchestration).toEqual({ input: 5_629 });
expect(result.usage.totalTokens).toBe(185_882);
expect(result.usage.cost.input).toBeCloseTo(5.629, 8);
expect(result.usage.cost.cacheRead).toBeCloseTo(90.112, 8);
});
it("omits request-body headers and replaces stale beta headers for websocket handshakes", async () => {
const tempDir = TempDir.createSync("@pi-codex-stream-");
setAgentDir(tempDir.path());
+35 -4
View File
@@ -261,7 +261,7 @@ describe("shared OpenAI usage accounting", () => {
});
describe("openai-responses usage attribution", () => {
it("folds Fugu Ultra orchestration token details into billable usage", () => {
it("separates Responses orchestration tokens from conversation usage", () => {
const output: AssistantMessage = {
role: "assistant",
content: [],
@@ -287,11 +287,42 @@ describe("openai-responses usage attribution", () => {
},
});
expect(output.usage.input).toBe(135);
expect(output.usage.cacheRead).toBe(15);
expect(output.usage.output).toBe(120);
expect(output.usage.input).toBe(110);
expect(output.usage.cacheRead).toBe(10);
expect(output.usage.output).toBe(80);
expect(output.usage.orchestration).toEqual({ input: 25, cacheRead: 5, output: 40 });
expect(output.usage.totalTokens).toBe(270);
});
it("does not label Codex orchestration input as an uncached prompt miss when primary totals include it", () => {
const output: AssistantMessage = {
role: "assistant",
content: [],
api: "openai-codex-responses",
provider: "openai-codex",
model: "gpt-5.5",
usage: blankUsage(),
stopReason: "toolUse",
timestamp: 0,
};
populateResponsesUsageFromResponse(output, {
input_tokens: 185_853,
output_tokens: 29,
total_tokens: 185_882,
input_tokens_details: {
cached_tokens: 180_224,
orchestration_input_tokens: 5_629,
orchestration_input_cached_tokens: 0,
},
});
expect(output.usage.input).toBe(0);
expect(output.usage.cacheRead).toBe(180_224);
expect(output.usage.output).toBe(29);
expect(output.usage.orchestration).toEqual({ input: 5_629 });
expect(output.usage.totalTokens).toBe(185_882);
});
});
describe("anthropic applyAnthropicUsageExtras", () => {
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased]
### Fixed
- Fixed usage cost calculation to include provider orchestration token sidecars without forcing those tokens into normal input/output/cache buckets. ([#4469](https://github.com/can1357/oh-my-pi/issues/4469))
## [16.3.4] - 2026-07-03
### Added
+4 -3
View File
@@ -44,9 +44,10 @@ export function getBundledModels(provider: GeneratedProvider): Model<Api>[] {
}
export function calculateCost<TApi extends Api>(model: Model<TApi>, usage: Usage): Usage["cost"] {
usage.cost.input = (model.cost.input / 1000000) * usage.input;
usage.cost.output = (model.cost.output / 1000000) * usage.output;
usage.cost.cacheRead = (model.cost.cacheRead / 1000000) * usage.cacheRead;
const orchestration = usage.orchestration;
usage.cost.input = (model.cost.input / 1000000) * (usage.input + (orchestration?.input ?? 0));
usage.cost.output = (model.cost.output / 1000000) * (usage.output + (orchestration?.output ?? 0));
usage.cost.cacheRead = (model.cost.cacheRead / 1000000) * (usage.cacheRead + (orchestration?.cacheRead ?? 0));
usage.cost.cacheWrite = (model.cost.cacheWrite / 1000000) * usage.cacheWrite;
usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite;
return usage.cost;
+14 -5
View File
@@ -93,16 +93,25 @@ export type Provider = string;
export type ThinkingBudgets = { [key in Effort]?: number };
export interface Usage {
/** Non-cached input tokens (matches the bucket the provider bills as new input). */
/** Non-cached conversation input tokens (matches the bucket the provider bills as new input). */
input: number;
/** Total output tokens for the turn, including thinking, assistant text, and tool-call argument tokens. */
/** Total conversation output tokens for the turn, including thinking, assistant text, and tool-call argument tokens. */
output: number;
/** Tokens read from the prompt cache. */
/** Conversation tokens read from the prompt cache. */
cacheRead: number;
/** Tokens written to the prompt cache (cache creation). */
/** Conversation tokens written to the prompt cache (cache creation). */
cacheWrite: number;
/** Sum of input + output + cacheRead + cacheWrite. */
/** Sum of input + output + cacheRead + cacheWrite plus provider-side orchestration tokens when reported. */
totalTokens: number;
/** Provider-side orchestration tokens, billed but not part of the conversation prompt/cache buckets. */
orchestration?: {
/** Non-cached orchestration input tokens. */
input?: number;
/** Orchestration tokens read from provider-side cache. */
cacheRead?: number;
/** Orchestration output tokens. */
output?: number;
};
/** Copilot premium-request counter, when applicable. */
premiumRequests?: number;
/**
+1
View File
@@ -65,6 +65,7 @@
- Fixed ACP `terminal/create` sending the bash tool's full shell line in `command` with no `args`, which broke spec-conformant clients that spawn `command`+`args` directly (no implicit shell) — any command containing a space, pipe, `&&`, redirect, or `$(...)` failed with `ENOENT` and the agent silently degraded to read-only tools. The bash tool now wraps the shell line before calling `clientBridge.createTerminal`, reusing the same shell binary + args the local `bash-executor` resolves via `settings.getShellConfig()` (Git Bash / `bash.exe` on Windows, `$SHELL` with `sh` fallback on POSIX) so bash semantics — `$VAR`, `$(...)`, `source`, POSIX quoting, `-l` — are preserved on both platforms. ([#4333](https://github.com/can1357/oh-my-pi/issues/4333))
- Fixed inference worker subprocesses (TTS, STT, tiny-model, mnemopi embeddings) discarding stderr, which left every unexpected exit — most visibly the local Kokoro TTS worker's recurring `exit code 7` crash loop — undiagnosable from the parent's logs. `createWorkerSubprocess` now pipes stderr without starting a live read while the worker is idle, then drains the stream after `onExit`, emits captured lines to `logger.debug` under an `<exitLabel> stderr` message, and keeps the last 16 KiB in a bounded ring that gets appended to the `Error` surfaced through `onError`. The exit surface is synchronized with the post-exit drain via `SpawnedSubprocess.stderrDrained`, so the full native trace shows up on the `tts: worker error` line without reintroducing event-loop liveness from unref'd workers. ([#4324](https://github.com/can1357/oh-my-pi/issues/4324))
- Fixed Windows session tail loss after atomic compaction rewrites by fencing append writers during full-file replacement and gating the atomic publish on a `commitGuard` that the storage backend checks synchronously before rename, so a concurrent `flushSync` (Ctrl+C / session-exit) is not overwritten by the stale body serialized before it ran. Covers post-compaction prompts, tool results, title changes, and exit diagnostics on the current JSONL path ([#4338](https://github.com/can1357/oh-my-pi/issues/4338)).
- Fixed session/status usage totals to preserve provider-reported orchestration tokens separately from ordinary input and cache-hit buckets. ([#4469](https://github.com/can1357/oh-my-pi/issues/4469))
## [16.3.4] - 2026-07-03
@@ -23,7 +23,18 @@ function goalState(extra: Partial<GoalModeState["goal"]>): GoalModeState {
}
function usage(output: number): UsageStatistics {
return { input: 0, output, cacheRead: 0, cacheWrite: 0, premiumRequests: 0, cost: 0 };
return {
input: 0,
output,
cacheRead: 0,
cacheWrite: 0,
totalTokens: output,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
};
}
describe("runEvalBudget", () => {
@@ -1823,6 +1823,10 @@ export class AcpAgent implements Agent {
output: usage.output,
cacheRead: usage.cacheRead,
cacheWrite: usage.cacheWrite,
totalTokens: usage.totalTokens,
orchestrationInput: usage.orchestrationInput,
orchestrationOutput: usage.orchestrationOutput,
orchestrationCacheRead: usage.orchestrationCacheRead,
premiumRequests: usage.premiumRequests,
cost: usage.cost,
};
@@ -1833,7 +1837,7 @@ export class AcpAgent implements Agent {
const outputTokens = Math.max(0, current.output - previous.output);
const cachedReadTokens = Math.max(0, current.cacheRead - previous.cacheRead);
const cachedWriteTokens = Math.max(0, current.cacheWrite - previous.cacheWrite);
const totalTokens = inputTokens + outputTokens + cachedReadTokens + cachedWriteTokens;
const totalTokens = Math.max(0, current.totalTokens - previous.totalTokens);
if (totalTokens === 0) {
return undefined;
@@ -998,6 +998,10 @@ export class StatusLineComponent implements Component {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
};
@@ -362,10 +362,11 @@ const tokenTotalSegment: StatusLineSegment = {
id: "token_total",
render(ctx) {
// Excludes cacheRead: that field re-reads the full cached context every
// turn, making the cumulative sum N×context_size. The dedicated cache_read
// segment handles cache monitoring; the cost segment handles billing.
const { input, output, cacheWrite } = ctx.usageStats;
const total = input + output + cacheWrite;
// turn, making the cumulative sum N×context_size. Orchestration cache read
// follows the same rule; orchestration input/output remain in the total so
// provider-side service work is preserved without labeling it prompt input.
const { input, output, cacheWrite, orchestrationInput, orchestrationOutput } = ctx.usageStats;
const total = input + output + cacheWrite + orchestrationInput + orchestrationOutput;
if (!total) return { content: "", visible: false };
const content = withIcon(theme.icon.tokens, formatNumber(total));
@@ -74,6 +74,10 @@ export interface SegmentContext {
output: number;
cacheRead: number;
cacheWrite: number;
totalTokens: number;
orchestrationInput: number;
orchestrationOutput: number;
orchestrationCacheRead: number;
premiumRequests: number;
cost: number;
tokensPerSecond: number | null;
@@ -15028,6 +15028,7 @@ export class AgentSession {
let totalCacheRead = 0;
let totalReasoning = 0;
let totalCacheWrite = 0;
let totalTokens = 0;
let totalCost = 0;
let totalPremiumRequests = 0;
@@ -15048,6 +15049,7 @@ export class AgentSession {
totalReasoning += assistantMsg.usage.reasoningTokens ?? 0;
totalCacheRead += assistantMsg.usage.cacheRead;
totalCacheWrite += assistantMsg.usage.cacheWrite;
totalTokens += assistantMsg.usage.totalTokens;
totalPremiumRequests += assistantMsg.usage.premiumRequests ?? 0;
totalCost += assistantMsg.usage.cost.total;
}
@@ -15060,6 +15062,7 @@ export class AgentSession {
totalReasoning += usage.reasoningTokens ?? 0;
totalCacheRead += usage.cacheRead;
totalCacheWrite += usage.cacheWrite;
totalTokens += usage.totalTokens;
totalPremiumRequests += usage.premiumRequests ?? 0;
totalCost += usage.cost.total;
}
@@ -15080,7 +15083,7 @@ export class AgentSession {
reasoning: totalReasoning,
cacheRead: totalCacheRead,
cacheWrite: totalCacheWrite,
total: totalInput + totalOutput + totalCacheRead + totalCacheWrite,
total: totalTokens,
},
cost: totalCost,
premiumRequests: totalPremiumRequests,
@@ -15726,6 +15729,7 @@ export class AgentSession {
let reasoning = 0;
let cacheRead = 0;
let cacheWrite = 0;
let totalTokens = 0;
let cost = 0;
let user = 0;
let assistant = 0;
@@ -15739,6 +15743,7 @@ export class AgentSession {
reasoning += assistantMsg.usage.reasoningTokens ?? 0;
cacheRead += assistantMsg.usage.cacheRead;
cacheWrite += assistantMsg.usage.cacheWrite;
totalTokens += assistantMsg.usage.totalTokens;
cost += assistantMsg.usage.cost.total;
}
}
@@ -15747,7 +15752,7 @@ export class AgentSession {
model,
contextWindow: model.contextWindow ?? 0,
contextTokens,
tokens: { input, output, reasoning, cacheRead, cacheWrite, total: input + output + cacheRead + cacheWrite },
tokens: { input, output, reasoning, cacheRead, cacheWrite, total: totalTokens },
cost,
messages: { user, assistant, total: messages.length },
};
@@ -236,6 +236,10 @@ export interface UsageStatistics {
output: number;
cacheRead: number;
cacheWrite: number;
totalTokens: number;
orchestrationInput: number;
orchestrationOutput: number;
orchestrationCacheRead: number;
premiumRequests: number;
cost: number;
}
@@ -115,7 +115,18 @@ function resolveBreadcrumbToInteractiveRoot(sessionFile: string): string {
}
function emptyUsageStatistics(): UsageStatistics {
return { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, premiumRequests: 0, cost: 0 };
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
};
}
function taskUsageFrom(details: unknown): Usage | undefined {
@@ -138,6 +149,10 @@ function addUsage(target: UsageStatistics, usage: Usage | undefined): void {
target.output += usage.output;
target.cacheRead += usage.cacheRead;
target.cacheWrite += usage.cacheWrite;
target.totalTokens += usage.totalTokens;
target.orchestrationInput += usage.orchestration?.input ?? 0;
target.orchestrationOutput += usage.orchestration?.output ?? 0;
target.orchestrationCacheRead += usage.orchestration?.cacheRead ?? 0;
target.premiumRequests += usage.premiumRequests ?? 0;
target.cost += usage.cost.total;
}
@@ -150,12 +150,15 @@ export async function buildUsageReportText(runtime: SlashCommandRuntime): Promis
}
const stats = runtime.session.sessionManager.getUsageStatistics();
const orchestrationTokens = stats.orchestrationInput + stats.orchestrationOutput + stats.orchestrationCacheRead;
return [
"Usage",
`Input tokens: ${stats.input}`,
`Output tokens: ${stats.output}`,
`Cache read tokens: ${stats.cacheRead}`,
`Cache write tokens: ${stats.cacheWrite}`,
`Total tokens: ${stats.totalTokens}`,
...(orchestrationTokens > 0 ? [`Orchestration tokens: ${orchestrationTokens}`] : []),
`Premium requests: ${stats.premiumRequests}`,
`Cost: $${stats.cost.toFixed(6)}`,
].join("\n");
@@ -73,6 +73,10 @@ function makeSession(): ConstructorParameters<typeof StatusLineComponent>[0] {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -89,6 +89,10 @@ describe("executeJs workflow helpers", () => {
output: 777,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 787,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -50,6 +50,37 @@ describe("SessionManager usage statistics", () => {
expect(usage.premiumRequests).toBe(3);
});
it("keeps orchestration usage out of ordinary input while preserving total tokens", () => {
const session = SessionManager.inMemory();
session.appendMessage({ role: "user", content: "hello", timestamp: 1 });
session.appendMessage({
role: "assistant",
content: [{ type: "text", text: "" }],
api: "openai-codex-responses",
provider: "openai-codex",
model: "gpt-5.5",
usage: {
input: 0,
output: 29,
cacheRead: 180_224,
cacheWrite: 0,
totalTokens: 185_882,
orchestration: { input: 5_629 },
cost: { input: 5.629, output: 0, cacheRead: 0, cacheWrite: 0, total: 5.629 },
},
stopReason: "toolUse",
timestamp: 2,
});
const usage = session.getUsageStatistics();
expect(usage.input).toBe(0);
expect(usage.cacheRead).toBe(180_224);
expect(usage.totalTokens).toBe(185_882);
expect(usage.orchestrationInput).toBe(5_629);
expect(usage.cost).toBeCloseTo(5.629, 8);
});
it("preserves fractional premium request multipliers", () => {
const session = SessionManager.inMemory();
@@ -15,6 +15,10 @@ function ctxWith(usage: Partial<SegmentContext["usageStats"]>): SegmentContext {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
tokensPerSecond: null,
@@ -58,6 +58,10 @@ function makeSession(opts: { messages: unknown[]; contextWindow?: number; usage?
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -66,6 +66,10 @@ function makeSession() {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -29,6 +29,10 @@ function createModelContext(advisorActive: boolean): SegmentContext {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
tokensPerSecond: null,
@@ -52,6 +52,10 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
tokensPerSecond: null,
@@ -87,6 +91,10 @@ function createStatusLineSession(sessionName: string) {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -38,6 +38,10 @@ function createPathContext(): SegmentContext {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
tokensPerSecond: null,
@@ -69,6 +69,10 @@ function makeSession() {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -57,6 +57,10 @@ function makeSession(sessionName = "Cache Session") {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -48,6 +48,10 @@ function createCtx(activeMs: number): SegmentContext {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
tokensPerSecond: null,
@@ -95,6 +99,10 @@ function makeSession(
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -40,6 +40,10 @@ function makeSession() {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -24,6 +24,10 @@ function makeSession(fetchUsageReports: (signal?: AbortSignal) => Promise<unknow
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -29,6 +29,10 @@ function makeComponent(
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),
@@ -192,6 +196,10 @@ describe("usage status-line segment", () => {
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
orchestrationInput: 0,
orchestrationOutput: 0,
orchestrationCacheRead: 0,
premiumRequests: 0,
cost: 0,
}),