In /vibe mode the director is often idle while workers stream, so the status-line tok/s badge showed a stale/zero rate even while parallel workers were actively generating tokens. The badge now aggregates the main session's live tok/s with every live vibe worker's tok/s, and falls back to the main session's own cached rate when no workers are streaming. - Extract calculateTokensPerSecond to utils/token-rate.ts (neutral location) so vibe/runtime.ts can depend on it without the render layer depending on the heavy vibe/task graph. - Add aggregateVibeWorkerTokensPerSecond to vibe/runtime.ts: sums each live worker's rate via the shared calculator, returns null when no workers are streaming so the main rate shines through. - StatusLineComponent takes the aggregator via an injected setVibeWorkerTokenRateProvider callback (wired in interactive-mode) keeping the render layer off the vibe/task dependency graph. - #getTokensPerSecond splits into #getMainSessionTokensPerSecond (preserves the sticky per-assistant-message cache) plus the worker-aggregation path, so the director-idle case no longer short-circuits to null.
73 lines
2.0 KiB
TypeScript
73 lines
2.0 KiB
TypeScript
/**
|
|
* Token-throughput calculator shared by the status line (main session tok/s
|
|
* badge) and the vibe worker aggregation ({@link aggregateVibeWorkerTokensPerSecond}).
|
|
* Lives in `utils/` so neither the render layer nor the vibe runtime has to
|
|
* depend on the other for a pure arithmetic helper.
|
|
*/
|
|
const MIN_DURATION_MS = 100;
|
|
|
|
type AssistantUsage = {
|
|
output: number;
|
|
};
|
|
|
|
type AssistantLikeMessage = {
|
|
role: "assistant";
|
|
timestamp: number;
|
|
duration?: number;
|
|
usage: AssistantUsage;
|
|
};
|
|
|
|
type MaybeAssistantMessage = {
|
|
role?: string;
|
|
timestamp?: number;
|
|
duration?: number;
|
|
usage?: {
|
|
output?: number;
|
|
};
|
|
};
|
|
|
|
function isAssistantMessage(message: MaybeAssistantMessage | undefined): message is AssistantLikeMessage {
|
|
return (
|
|
message?.role === "assistant" &&
|
|
typeof message.timestamp === "number" &&
|
|
message.usage !== undefined &&
|
|
typeof message.usage.output === "number"
|
|
);
|
|
}
|
|
|
|
function getLastAssistantMessage(messages: ReadonlyArray<MaybeAssistantMessage>): AssistantLikeMessage | null {
|
|
for (let i = messages.length - 1; i >= 0; i--) {
|
|
const message = messages[i];
|
|
if (isAssistantMessage(message)) {
|
|
return message;
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
|
|
export function calculateTokensPerSecond(
|
|
messages: ReadonlyArray<MaybeAssistantMessage>,
|
|
isStreaming: boolean,
|
|
nowMs: number = Date.now(),
|
|
): number | null {
|
|
const assistant = getLastAssistantMessage(messages);
|
|
if (!assistant) return null;
|
|
|
|
const outputTokens = assistant.usage.output;
|
|
if (!Number.isFinite(outputTokens) || outputTokens <= 0) return null;
|
|
|
|
const resolvedDurationMs =
|
|
typeof assistant.duration === "number" && Number.isFinite(assistant.duration) && assistant.duration > 0
|
|
? assistant.duration
|
|
: isStreaming
|
|
? nowMs - assistant.timestamp
|
|
: null;
|
|
|
|
if (resolvedDurationMs === null || resolvedDurationMs < MIN_DURATION_MS) return null;
|
|
|
|
const tokensPerSecond = (outputTokens * 1000) / resolvedDurationMs;
|
|
if (!Number.isFinite(tokensPerSecond) || tokensPerSecond <= 0) return null;
|
|
|
|
return tokensPerSecond;
|
|
}
|