fix(ai): separated codex orchestration usage
- Added a Usage.orchestration sidecar for provider-side service tokens so Responses/Codex totals and costs stay accurate without inflating visible prompt input/cache buckets. - Updated Codex/WebSocket usage, session/status aggregates, and usage reporting to preserve orchestration-aware totals. - Added regressions for OpenAI Responses accounting, Codex WebSocket terminal usage, cost calculation, and session aggregation. Fixes #4469
This commit is contained in:
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed OpenAI Responses/Codex orchestration token accounting so provider-side orchestration tokens stay billable and included in totals without appearing as ordinary uncached prompt input. ([#4469](https://github.com/can1357/oh-my-pi/issues/4469))
|
||||
|
||||
## [16.3.4] - 2026-07-03
|
||||
|
||||
### Added
|
||||
|
||||
@@ -446,10 +446,17 @@ function mergeUsage(partial?: Partial<Omit<Usage, "cost">> & { cost?: Partial<Us
|
||||
if (costProvided) {
|
||||
merged.cost = { ...base.cost, ...partial.cost } as Usage["cost"];
|
||||
}
|
||||
// Recompute totalTokens when not explicitly provided (canonical formula matches types.ts:
|
||||
// input + output + cacheRead + cacheWrite).
|
||||
// Recompute totalTokens when not explicitly provided (canonical formula matches types.ts).
|
||||
if (partial.totalTokens === undefined) {
|
||||
merged.totalTokens = merged.input + merged.output + merged.cacheRead + merged.cacheWrite;
|
||||
const orchestration = merged.orchestration;
|
||||
merged.totalTokens =
|
||||
merged.input +
|
||||
merged.output +
|
||||
merged.cacheRead +
|
||||
merged.cacheWrite +
|
||||
(orchestration?.input ?? 0) +
|
||||
(orchestration?.output ?? 0) +
|
||||
(orchestration?.cacheRead ?? 0);
|
||||
}
|
||||
// Recompute cost.total when cost components were supplied without an explicit total.
|
||||
if (costProvided && partial.cost?.total === undefined) {
|
||||
|
||||
@@ -1531,8 +1531,15 @@ class CodexStreamProcessor {
|
||||
input_tokens?: number;
|
||||
output_tokens?: number;
|
||||
total_tokens?: number;
|
||||
input_tokens_details?: { cached_tokens?: number };
|
||||
output_tokens_details?: { reasoning_tokens?: number };
|
||||
input_tokens_details?: {
|
||||
cached_tokens?: number;
|
||||
orchestration_input_tokens?: number;
|
||||
orchestration_input_cached_tokens?: number;
|
||||
};
|
||||
output_tokens_details?: {
|
||||
reasoning_tokens?: number;
|
||||
orchestration_output_tokens?: number;
|
||||
};
|
||||
};
|
||||
status?: string;
|
||||
service_tier?: ServiceTier | "default";
|
||||
|
||||
@@ -50,6 +50,7 @@ import {
|
||||
type Tool,
|
||||
type ToolCall,
|
||||
type ToolResultMessage,
|
||||
type Usage,
|
||||
} from "../types";
|
||||
import {
|
||||
getOpenAIResponsesHistoryItems,
|
||||
@@ -341,6 +342,7 @@ export interface OpenAIUsageAccounting {
|
||||
cacheWrite: number;
|
||||
totalTokens: number;
|
||||
reasoningTokens?: number;
|
||||
orchestration?: Usage["orchestration"];
|
||||
}
|
||||
|
||||
export function calculateOpenAIUsageAccounting(accounting: OpenAIUsageAccountingInput): OpenAIUsageAccounting {
|
||||
@@ -2522,19 +2524,42 @@ export function populateResponsesUsageFromResponse(
|
||||
if (!usage) return;
|
||||
const details = usage.input_tokens_details;
|
||||
const outputDetails = usage.output_tokens_details;
|
||||
const reportedInputTokens = usage.input_tokens ?? 0;
|
||||
const reportedOutputTokens = usage.output_tokens ?? 0;
|
||||
const reportedCachedTokens = details?.cached_tokens ?? usage.prompt_cache_hit_tokens ?? 0;
|
||||
const orchestrationInputTokens = details?.orchestration_input_tokens ?? 0;
|
||||
const orchestrationInputCachedTokens = details?.orchestration_input_cached_tokens ?? 0;
|
||||
const orchestrationOutputTokens = outputDetails?.orchestration_output_tokens ?? 0;
|
||||
const reportedTotalTokens = typeof usage.total_tokens === "number" ? usage.total_tokens : undefined;
|
||||
const reportedPrimaryTokens = reportedInputTokens + reportedOutputTokens;
|
||||
const reportedWithSeparateOrchestration =
|
||||
reportedPrimaryTokens + orchestrationInputTokens + orchestrationOutputTokens;
|
||||
const primaryIncludesOrchestration =
|
||||
reportedTotalTokens !== undefined &&
|
||||
orchestrationInputTokens + orchestrationOutputTokens > 0 &&
|
||||
Math.abs(reportedTotalTokens - reportedPrimaryTokens) <=
|
||||
Math.abs(reportedTotalTokens - reportedWithSeparateOrchestration);
|
||||
const orchestrationInputCached = Math.min(orchestrationInputTokens, orchestrationInputCachedTokens);
|
||||
const orchestrationInput = Math.max(0, orchestrationInputTokens - orchestrationInputCached);
|
||||
const accounting = calculateOpenAIUsageAccounting({
|
||||
promptTokens: (usage.input_tokens ?? 0) + orchestrationInputTokens,
|
||||
outputTokens: (usage.output_tokens ?? 0) + orchestrationOutputTokens,
|
||||
cachedTokens: (details?.cached_tokens ?? usage.prompt_cache_hit_tokens ?? 0) + orchestrationInputCachedTokens,
|
||||
promptTokens: Math.max(0, reportedInputTokens - (primaryIncludesOrchestration ? orchestrationInputTokens : 0)),
|
||||
outputTokens: Math.max(0, reportedOutputTokens - (primaryIncludesOrchestration ? orchestrationOutputTokens : 0)),
|
||||
cachedTokens: Math.max(0, reportedCachedTokens - (primaryIncludesOrchestration ? orchestrationInputCached : 0)),
|
||||
reasoningTokens: outputDetails?.reasoning_tokens ?? 0,
|
||||
cacheWriteOpenRouter: details?.cache_write_tokens ?? undefined,
|
||||
cacheWriteDeepSeek: usage.prompt_cache_miss_tokens ?? undefined,
|
||||
hasDeepSeekCacheHitAndMiss:
|
||||
usage.prompt_cache_hit_tokens !== undefined && usage.prompt_cache_miss_tokens !== undefined,
|
||||
});
|
||||
const orchestrationTotal = orchestrationInput + orchestrationInputCached + orchestrationOutputTokens;
|
||||
if (orchestrationTotal > 0) {
|
||||
accounting.orchestration = {
|
||||
...(orchestrationInput > 0 ? { input: orchestrationInput } : {}),
|
||||
...(orchestrationInputCached > 0 ? { cacheRead: orchestrationInputCached } : {}),
|
||||
...(orchestrationOutputTokens > 0 ? { output: orchestrationOutputTokens } : {}),
|
||||
};
|
||||
accounting.totalTokens = reportedTotalTokens ?? accounting.totalTokens + orchestrationTotal;
|
||||
}
|
||||
|
||||
// Wholesale replacement must not drop provider-annotated extras (Copilot
|
||||
// premium-request accounting): the failed/cancelled paths throw right after
|
||||
|
||||
@@ -71,6 +71,38 @@ describe("calculateCost", () => {
|
||||
expect(usage.cost.total).toBeCloseTo(2.18, 8);
|
||||
});
|
||||
|
||||
it("prices provider orchestration tokens without changing visible usage buckets", () => {
|
||||
const model = {
|
||||
...getBundledModel("openai", "gpt-4o-mini"),
|
||||
cost: {
|
||||
input: 1000,
|
||||
output: 2000,
|
||||
cacheRead: 500,
|
||||
cacheWrite: 800,
|
||||
},
|
||||
};
|
||||
const usage: Usage = {
|
||||
input: 100,
|
||||
output: 20,
|
||||
cacheRead: 50,
|
||||
cacheWrite: 10,
|
||||
totalTokens: 250,
|
||||
orchestration: { input: 25, output: 40, cacheRead: 5 },
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
};
|
||||
|
||||
calculateCost(model, usage);
|
||||
|
||||
expect(usage.input).toBe(100);
|
||||
expect(usage.output).toBe(20);
|
||||
expect(usage.cacheRead).toBe(50);
|
||||
expect(usage.cost.input).toBeCloseTo(0.125, 8);
|
||||
expect(usage.cost.output).toBeCloseTo(0.12, 8);
|
||||
expect(usage.cost.cacheRead).toBeCloseTo(0.0275, 8);
|
||||
expect(usage.cost.cacheWrite).toBeCloseTo(0.008, 8);
|
||||
expect(usage.cost.total).toBeCloseTo(0.2805, 8);
|
||||
});
|
||||
|
||||
it("prices OpenAI Codex GPT models from the matching OpenAI catalog entry", () => {
|
||||
const openAIModel = getBundledModel("openai", "gpt-5.4");
|
||||
const codexModel = getBundledModel("openai-codex", "gpt-5.4");
|
||||
|
||||
@@ -860,6 +860,58 @@ describe("openai-codex streaming", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("separates websocket terminal orchestration usage from prompt cache buckets", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
|
||||
class UsageWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(): void {
|
||||
this.sendJson({
|
||||
type: "response.done",
|
||||
response: {
|
||||
id: "resp_usage",
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 185_853,
|
||||
output_tokens: 29,
|
||||
total_tokens: 185_882,
|
||||
input_tokens_details: {
|
||||
cached_tokens: 180_224,
|
||||
orchestration_input_tokens: 5_629,
|
||||
orchestration_input_cached_tokens: 0,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
global.WebSocket = UsageWebSocket as unknown as typeof WebSocket;
|
||||
|
||||
const model = {
|
||||
...createCodexTestModel("https://chatgpt.com/backend-api"),
|
||||
cost: { input: 1000, output: 2000, cacheRead: 500, cacheWrite: 0 },
|
||||
};
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
apiKey: token,
|
||||
sessionId: "ws-orchestration-usage-session",
|
||||
providerSessionState: new Map<string, ProviderSessionState>(),
|
||||
}).result();
|
||||
|
||||
expect(result.usage.input).toBe(0);
|
||||
expect(result.usage.cacheRead).toBe(180_224);
|
||||
expect(result.usage.output).toBe(29);
|
||||
expect(result.usage.orchestration).toEqual({ input: 5_629 });
|
||||
expect(result.usage.totalTokens).toBe(185_882);
|
||||
expect(result.usage.cost.input).toBeCloseTo(5.629, 8);
|
||||
expect(result.usage.cost.cacheRead).toBeCloseTo(90.112, 8);
|
||||
});
|
||||
|
||||
it("omits request-body headers and replaces stale beta headers for websocket handshakes", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
|
||||
@@ -261,7 +261,7 @@ describe("shared OpenAI usage accounting", () => {
|
||||
});
|
||||
|
||||
describe("openai-responses usage attribution", () => {
|
||||
it("folds Fugu Ultra orchestration token details into billable usage", () => {
|
||||
it("separates Responses orchestration tokens from conversation usage", () => {
|
||||
const output: AssistantMessage = {
|
||||
role: "assistant",
|
||||
content: [],
|
||||
@@ -287,11 +287,42 @@ describe("openai-responses usage attribution", () => {
|
||||
},
|
||||
});
|
||||
|
||||
expect(output.usage.input).toBe(135);
|
||||
expect(output.usage.cacheRead).toBe(15);
|
||||
expect(output.usage.output).toBe(120);
|
||||
expect(output.usage.input).toBe(110);
|
||||
expect(output.usage.cacheRead).toBe(10);
|
||||
expect(output.usage.output).toBe(80);
|
||||
expect(output.usage.orchestration).toEqual({ input: 25, cacheRead: 5, output: 40 });
|
||||
expect(output.usage.totalTokens).toBe(270);
|
||||
});
|
||||
|
||||
it("does not label Codex orchestration input as an uncached prompt miss when primary totals include it", () => {
|
||||
const output: AssistantMessage = {
|
||||
role: "assistant",
|
||||
content: [],
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
model: "gpt-5.5",
|
||||
usage: blankUsage(),
|
||||
stopReason: "toolUse",
|
||||
timestamp: 0,
|
||||
};
|
||||
|
||||
populateResponsesUsageFromResponse(output, {
|
||||
input_tokens: 185_853,
|
||||
output_tokens: 29,
|
||||
total_tokens: 185_882,
|
||||
input_tokens_details: {
|
||||
cached_tokens: 180_224,
|
||||
orchestration_input_tokens: 5_629,
|
||||
orchestration_input_cached_tokens: 0,
|
||||
},
|
||||
});
|
||||
|
||||
expect(output.usage.input).toBe(0);
|
||||
expect(output.usage.cacheRead).toBe(180_224);
|
||||
expect(output.usage.output).toBe(29);
|
||||
expect(output.usage.orchestration).toEqual({ input: 5_629 });
|
||||
expect(output.usage.totalTokens).toBe(185_882);
|
||||
});
|
||||
});
|
||||
|
||||
describe("anthropic applyAnthropicUsageExtras", () => {
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed usage cost calculation to include provider orchestration token sidecars without forcing those tokens into normal input/output/cache buckets. ([#4469](https://github.com/can1357/oh-my-pi/issues/4469))
|
||||
|
||||
## [16.3.4] - 2026-07-03
|
||||
|
||||
### Added
|
||||
|
||||
@@ -44,9 +44,10 @@ export function getBundledModels(provider: GeneratedProvider): Model<Api>[] {
|
||||
}
|
||||
|
||||
export function calculateCost<TApi extends Api>(model: Model<TApi>, usage: Usage): Usage["cost"] {
|
||||
usage.cost.input = (model.cost.input / 1000000) * usage.input;
|
||||
usage.cost.output = (model.cost.output / 1000000) * usage.output;
|
||||
usage.cost.cacheRead = (model.cost.cacheRead / 1000000) * usage.cacheRead;
|
||||
const orchestration = usage.orchestration;
|
||||
usage.cost.input = (model.cost.input / 1000000) * (usage.input + (orchestration?.input ?? 0));
|
||||
usage.cost.output = (model.cost.output / 1000000) * (usage.output + (orchestration?.output ?? 0));
|
||||
usage.cost.cacheRead = (model.cost.cacheRead / 1000000) * (usage.cacheRead + (orchestration?.cacheRead ?? 0));
|
||||
usage.cost.cacheWrite = (model.cost.cacheWrite / 1000000) * usage.cacheWrite;
|
||||
usage.cost.total = usage.cost.input + usage.cost.output + usage.cost.cacheRead + usage.cost.cacheWrite;
|
||||
return usage.cost;
|
||||
|
||||
@@ -93,16 +93,25 @@ export type Provider = string;
|
||||
export type ThinkingBudgets = { [key in Effort]?: number };
|
||||
|
||||
export interface Usage {
|
||||
/** Non-cached input tokens (matches the bucket the provider bills as new input). */
|
||||
/** Non-cached conversation input tokens (matches the bucket the provider bills as new input). */
|
||||
input: number;
|
||||
/** Total output tokens for the turn, including thinking, assistant text, and tool-call argument tokens. */
|
||||
/** Total conversation output tokens for the turn, including thinking, assistant text, and tool-call argument tokens. */
|
||||
output: number;
|
||||
/** Tokens read from the prompt cache. */
|
||||
/** Conversation tokens read from the prompt cache. */
|
||||
cacheRead: number;
|
||||
/** Tokens written to the prompt cache (cache creation). */
|
||||
/** Conversation tokens written to the prompt cache (cache creation). */
|
||||
cacheWrite: number;
|
||||
/** Sum of input + output + cacheRead + cacheWrite. */
|
||||
/** Sum of input + output + cacheRead + cacheWrite plus provider-side orchestration tokens when reported. */
|
||||
totalTokens: number;
|
||||
/** Provider-side orchestration tokens, billed but not part of the conversation prompt/cache buckets. */
|
||||
orchestration?: {
|
||||
/** Non-cached orchestration input tokens. */
|
||||
input?: number;
|
||||
/** Orchestration tokens read from provider-side cache. */
|
||||
cacheRead?: number;
|
||||
/** Orchestration output tokens. */
|
||||
output?: number;
|
||||
};
|
||||
/** Copilot premium-request counter, when applicable. */
|
||||
premiumRequests?: number;
|
||||
/**
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed session/status usage totals to preserve provider-reported orchestration tokens separately from ordinary input and cache-hit buckets. ([#4469](https://github.com/can1357/oh-my-pi/issues/4469))
|
||||
|
||||
## [16.3.4] - 2026-07-03
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -23,7 +23,18 @@ function goalState(extra: Partial<GoalModeState["goal"]>): GoalModeState {
|
||||
}
|
||||
|
||||
function usage(output: number): UsageStatistics {
|
||||
return { input: 0, output, cacheRead: 0, cacheWrite: 0, premiumRequests: 0, cost: 0 };
|
||||
return {
|
||||
input: 0,
|
||||
output,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: output,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
};
|
||||
}
|
||||
|
||||
describe("runEvalBudget", () => {
|
||||
|
||||
@@ -1823,6 +1823,10 @@ export class AcpAgent implements Agent {
|
||||
output: usage.output,
|
||||
cacheRead: usage.cacheRead,
|
||||
cacheWrite: usage.cacheWrite,
|
||||
totalTokens: usage.totalTokens,
|
||||
orchestrationInput: usage.orchestrationInput,
|
||||
orchestrationOutput: usage.orchestrationOutput,
|
||||
orchestrationCacheRead: usage.orchestrationCacheRead,
|
||||
premiumRequests: usage.premiumRequests,
|
||||
cost: usage.cost,
|
||||
};
|
||||
@@ -1833,7 +1837,7 @@ export class AcpAgent implements Agent {
|
||||
const outputTokens = Math.max(0, current.output - previous.output);
|
||||
const cachedReadTokens = Math.max(0, current.cacheRead - previous.cacheRead);
|
||||
const cachedWriteTokens = Math.max(0, current.cacheWrite - previous.cacheWrite);
|
||||
const totalTokens = inputTokens + outputTokens + cachedReadTokens + cachedWriteTokens;
|
||||
const totalTokens = Math.max(0, current.totalTokens - previous.totalTokens);
|
||||
|
||||
if (totalTokens === 0) {
|
||||
return undefined;
|
||||
|
||||
@@ -998,6 +998,10 @@ export class StatusLineComponent implements Component {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
};
|
||||
|
||||
@@ -362,10 +362,11 @@ const tokenTotalSegment: StatusLineSegment = {
|
||||
id: "token_total",
|
||||
render(ctx) {
|
||||
// Excludes cacheRead: that field re-reads the full cached context every
|
||||
// turn, making the cumulative sum N×context_size. The dedicated cache_read
|
||||
// segment handles cache monitoring; the cost segment handles billing.
|
||||
const { input, output, cacheWrite } = ctx.usageStats;
|
||||
const total = input + output + cacheWrite;
|
||||
// turn, making the cumulative sum N×context_size. Orchestration cache read
|
||||
// follows the same rule; orchestration input/output remain in the total so
|
||||
// provider-side service work is preserved without labeling it prompt input.
|
||||
const { input, output, cacheWrite, orchestrationInput, orchestrationOutput } = ctx.usageStats;
|
||||
const total = input + output + cacheWrite + orchestrationInput + orchestrationOutput;
|
||||
if (!total) return { content: "", visible: false };
|
||||
|
||||
const content = withIcon(theme.icon.tokens, formatNumber(total));
|
||||
|
||||
@@ -74,6 +74,10 @@ export interface SegmentContext {
|
||||
output: number;
|
||||
cacheRead: number;
|
||||
cacheWrite: number;
|
||||
totalTokens: number;
|
||||
orchestrationInput: number;
|
||||
orchestrationOutput: number;
|
||||
orchestrationCacheRead: number;
|
||||
premiumRequests: number;
|
||||
cost: number;
|
||||
tokensPerSecond: number | null;
|
||||
|
||||
@@ -14869,6 +14869,7 @@ export class AgentSession {
|
||||
let totalCacheRead = 0;
|
||||
let totalReasoning = 0;
|
||||
let totalCacheWrite = 0;
|
||||
let totalTokens = 0;
|
||||
let totalCost = 0;
|
||||
let totalPremiumRequests = 0;
|
||||
|
||||
@@ -14889,6 +14890,7 @@ export class AgentSession {
|
||||
totalReasoning += assistantMsg.usage.reasoningTokens ?? 0;
|
||||
totalCacheRead += assistantMsg.usage.cacheRead;
|
||||
totalCacheWrite += assistantMsg.usage.cacheWrite;
|
||||
totalTokens += assistantMsg.usage.totalTokens;
|
||||
totalPremiumRequests += assistantMsg.usage.premiumRequests ?? 0;
|
||||
totalCost += assistantMsg.usage.cost.total;
|
||||
}
|
||||
@@ -14901,6 +14903,7 @@ export class AgentSession {
|
||||
totalReasoning += usage.reasoningTokens ?? 0;
|
||||
totalCacheRead += usage.cacheRead;
|
||||
totalCacheWrite += usage.cacheWrite;
|
||||
totalTokens += usage.totalTokens;
|
||||
totalPremiumRequests += usage.premiumRequests ?? 0;
|
||||
totalCost += usage.cost.total;
|
||||
}
|
||||
@@ -14921,7 +14924,7 @@ export class AgentSession {
|
||||
reasoning: totalReasoning,
|
||||
cacheRead: totalCacheRead,
|
||||
cacheWrite: totalCacheWrite,
|
||||
total: totalInput + totalOutput + totalCacheRead + totalCacheWrite,
|
||||
total: totalTokens,
|
||||
},
|
||||
cost: totalCost,
|
||||
premiumRequests: totalPremiumRequests,
|
||||
@@ -15567,6 +15570,7 @@ export class AgentSession {
|
||||
let reasoning = 0;
|
||||
let cacheRead = 0;
|
||||
let cacheWrite = 0;
|
||||
let totalTokens = 0;
|
||||
let cost = 0;
|
||||
let user = 0;
|
||||
let assistant = 0;
|
||||
@@ -15580,6 +15584,7 @@ export class AgentSession {
|
||||
reasoning += assistantMsg.usage.reasoningTokens ?? 0;
|
||||
cacheRead += assistantMsg.usage.cacheRead;
|
||||
cacheWrite += assistantMsg.usage.cacheWrite;
|
||||
totalTokens += assistantMsg.usage.totalTokens;
|
||||
cost += assistantMsg.usage.cost.total;
|
||||
}
|
||||
}
|
||||
@@ -15588,7 +15593,7 @@ export class AgentSession {
|
||||
model,
|
||||
contextWindow: model.contextWindow ?? 0,
|
||||
contextTokens,
|
||||
tokens: { input, output, reasoning, cacheRead, cacheWrite, total: input + output + cacheRead + cacheWrite },
|
||||
tokens: { input, output, reasoning, cacheRead, cacheWrite, total: totalTokens },
|
||||
cost,
|
||||
messages: { user, assistant, total: messages.length },
|
||||
};
|
||||
|
||||
@@ -236,6 +236,10 @@ export interface UsageStatistics {
|
||||
output: number;
|
||||
cacheRead: number;
|
||||
cacheWrite: number;
|
||||
totalTokens: number;
|
||||
orchestrationInput: number;
|
||||
orchestrationOutput: number;
|
||||
orchestrationCacheRead: number;
|
||||
premiumRequests: number;
|
||||
cost: number;
|
||||
}
|
||||
|
||||
@@ -114,7 +114,18 @@ function resolveBreadcrumbToInteractiveRoot(sessionFile: string): string {
|
||||
}
|
||||
|
||||
function emptyUsageStatistics(): UsageStatistics {
|
||||
return { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, premiumRequests: 0, cost: 0 };
|
||||
return {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
};
|
||||
}
|
||||
|
||||
function taskUsageFrom(details: unknown): Usage | undefined {
|
||||
@@ -137,6 +148,10 @@ function addUsage(target: UsageStatistics, usage: Usage | undefined): void {
|
||||
target.output += usage.output;
|
||||
target.cacheRead += usage.cacheRead;
|
||||
target.cacheWrite += usage.cacheWrite;
|
||||
target.totalTokens += usage.totalTokens;
|
||||
target.orchestrationInput += usage.orchestration?.input ?? 0;
|
||||
target.orchestrationOutput += usage.orchestration?.output ?? 0;
|
||||
target.orchestrationCacheRead += usage.orchestration?.cacheRead ?? 0;
|
||||
target.premiumRequests += usage.premiumRequests ?? 0;
|
||||
target.cost += usage.cost.total;
|
||||
}
|
||||
|
||||
@@ -150,12 +150,15 @@ export async function buildUsageReportText(runtime: SlashCommandRuntime): Promis
|
||||
}
|
||||
|
||||
const stats = runtime.session.sessionManager.getUsageStatistics();
|
||||
const orchestrationTokens = stats.orchestrationInput + stats.orchestrationOutput + stats.orchestrationCacheRead;
|
||||
return [
|
||||
"Usage",
|
||||
`Input tokens: ${stats.input}`,
|
||||
`Output tokens: ${stats.output}`,
|
||||
`Cache read tokens: ${stats.cacheRead}`,
|
||||
`Cache write tokens: ${stats.cacheWrite}`,
|
||||
`Total tokens: ${stats.totalTokens}`,
|
||||
...(orchestrationTokens > 0 ? [`Orchestration tokens: ${orchestrationTokens}`] : []),
|
||||
`Premium requests: ${stats.premiumRequests}`,
|
||||
`Cost: $${stats.cost.toFixed(6)}`,
|
||||
].join("\n");
|
||||
|
||||
@@ -73,6 +73,10 @@ function makeSession(): ConstructorParameters<typeof StatusLineComponent>[0] {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -89,6 +89,10 @@ describe("executeJs workflow helpers", () => {
|
||||
output: 777,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 787,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -50,6 +50,37 @@ describe("SessionManager usage statistics", () => {
|
||||
expect(usage.premiumRequests).toBe(3);
|
||||
});
|
||||
|
||||
it("keeps orchestration usage out of ordinary input while preserving total tokens", () => {
|
||||
const session = SessionManager.inMemory();
|
||||
|
||||
session.appendMessage({ role: "user", content: "hello", timestamp: 1 });
|
||||
session.appendMessage({
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text: "" }],
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
model: "gpt-5.5",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 29,
|
||||
cacheRead: 180_224,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 185_882,
|
||||
orchestration: { input: 5_629 },
|
||||
cost: { input: 5.629, output: 0, cacheRead: 0, cacheWrite: 0, total: 5.629 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp: 2,
|
||||
});
|
||||
|
||||
const usage = session.getUsageStatistics();
|
||||
expect(usage.input).toBe(0);
|
||||
expect(usage.cacheRead).toBe(180_224);
|
||||
expect(usage.totalTokens).toBe(185_882);
|
||||
expect(usage.orchestrationInput).toBe(5_629);
|
||||
expect(usage.cost).toBeCloseTo(5.629, 8);
|
||||
});
|
||||
|
||||
it("preserves fractional premium request multipliers", () => {
|
||||
const session = SessionManager.inMemory();
|
||||
|
||||
|
||||
@@ -15,6 +15,10 @@ function ctxWith(usage: Partial<SegmentContext["usageStats"]>): SegmentContext {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
tokensPerSecond: null,
|
||||
|
||||
@@ -58,6 +58,10 @@ function makeSession(opts: { messages: unknown[]; contextWindow?: number; usage?
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -66,6 +66,10 @@ function makeSession() {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -29,6 +29,10 @@ function createModelContext(advisorActive: boolean): SegmentContext {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
tokensPerSecond: null,
|
||||
|
||||
@@ -52,6 +52,10 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
tokensPerSecond: null,
|
||||
@@ -87,6 +91,10 @@ function createStatusLineSession(sessionName: string) {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -38,6 +38,10 @@ function createPathContext(): SegmentContext {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
tokensPerSecond: null,
|
||||
|
||||
@@ -69,6 +69,10 @@ function makeSession() {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -57,6 +57,10 @@ function makeSession(sessionName = "Cache Session") {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -48,6 +48,10 @@ function createCtx(activeMs: number): SegmentContext {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
tokensPerSecond: null,
|
||||
@@ -95,6 +99,10 @@ function makeSession(
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -40,6 +40,10 @@ function makeSession() {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -24,6 +24,10 @@ function makeSession(fetchUsageReports: (signal?: AbortSignal) => Promise<unknow
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
@@ -29,6 +29,10 @@ function makeComponent(
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
@@ -192,6 +196,10 @@ describe("usage status-line segment", () => {
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
orchestrationInput: 0,
|
||||
orchestrationOutput: 0,
|
||||
orchestrationCacheRead: 0,
|
||||
premiumRequests: 0,
|
||||
cost: 0,
|
||||
}),
|
||||
|
||||
Reference in New Issue
Block a user