diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ca293dcdc..7b9a4309d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,8 +1,10 @@ # Changelog ## [Unreleased] + ### Added +- Added the `/context` slash command to display an estimated context-usage breakdown panel for the current session - Added `-LidA..LidB` syntax to delete inclusive line ranges in a single atom operation - Added `LidA..LidB=TEXT` range-replace syntax with `\TEXT` and `\` continuation lines for multi-line replacement blocks - Added shorthand cursor+insert operations in atom edits, including `^Lid` (insert before anchor), `^+TEXT`, `$+TEXT`, and `Lid+TEXT`/`@Lid+TEXT` @@ -10,6 +12,7 @@ ### Changed +- Changed token counting to use tokenizer-based estimates instead of a character-per-4 heuristic for context and compaction calculations - Changed hashline anchor auto-rebase tolerance from ±2 lines to ±5 lines for stale Lid recovery - Changed atom input handling so `#`-prefixed lines are treated as comments and ignored - Changed execution when all edits are no-op `Lid=TEXT` replacements to return success with a no-change explanation instead of throwing diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index 428cce871..17dc001e8 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -24,6 +24,7 @@ import { DynamicBorder } from "../../modes/components/dynamic-border"; import { PythonExecutionComponent } from "../../modes/components/python-execution"; import { getMarkdownTheme, getSymbolTheme, theme } from "../../modes/theme/theme"; import type { InteractiveModeContext } from "../../modes/types"; +import { computeContextBreakdown, renderContextUsage } from "../../modes/utils/context-usage"; import { buildHotkeysMarkdown } from "../../modes/utils/hotkeys-markdown"; import { buildToolsMarkdown } from "../../modes/utils/tools-markdown"; import type { AsyncJobSnapshotItem } from "../../session/agent-session"; @@ -529,6 +530,22 @@ export class CommandController { showMarkdownPanel(this.ctx, "Available Tools", tools); } + handleContextCommand(): void { + const breakdown = computeContextBreakdown(this.ctx.session); + if (breakdown.contextWindow <= 0) { + this.ctx.showWarning("Context usage is unavailable: no model is selected for this session."); + return; + } + const output = renderContextUsage(breakdown, theme); + this.ctx.chatContainer.addChild(new Spacer(1)); + this.ctx.chatContainer.addChild(new DynamicBorder()); + this.ctx.chatContainer.addChild(new Text(theme.bold(theme.fg("accent", "Context Usage")), 1, 0)); + this.ctx.chatContainer.addChild(new Spacer(1)); + this.ctx.chatContainer.addChild(new Text(output, 1, 0)); + this.ctx.chatContainer.addChild(new DynamicBorder()); + this.ctx.ui.requestRender(); + } + async handleMemoryCommand(text: string): Promise { const argumentText = text.slice(7).trim(); const action = argumentText.split(/\s+/, 1)[0]?.toLowerCase() || "view"; diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index a64b1a051..2b49f9f61 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -1399,6 +1399,10 @@ export class InteractiveMode implements InteractiveModeContext { this.#commandController.handleToolsCommand(); } + handleContextCommand(): void { + this.#commandController.handleContextCommand(); + } + #prepareSessionSwitch(): void { this.#btwController.dispose(); this.#extensionUiController.clearExtensionTerminalInputListeners(); diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 0e2128809..4f9e2b5f2 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -181,6 +181,7 @@ export interface InteractiveModeContext { handleChangelogCommand(showFull?: boolean): Promise; handleHotkeysCommand(): void; handleToolsCommand(): void; + handleContextCommand(): void; handleDumpCommand(): void; handleDebugTranscriptCommand(): Promise; handleClearCommand(): Promise; diff --git a/packages/coding-agent/src/modes/utils/context-usage.ts b/packages/coding-agent/src/modes/utils/context-usage.ts new file mode 100644 index 000000000..4fce2449f --- /dev/null +++ b/packages/coding-agent/src/modes/utils/context-usage.ts @@ -0,0 +1,294 @@ +import type { Model } from "@oh-my-pi/pi-ai"; +import { countTokens } from "@oh-my-pi/pi-natives"; +import { formatNumber } from "@oh-my-pi/pi-utils"; +import type { Skill } from "../../extensibility/skills"; +import type { AgentSession } from "../../session/agent-session"; +import type { CompactionSettings } from "../../session/compaction"; +import { effectiveReserveTokens, estimateTokens, resolveThresholdTokens } from "../../session/compaction"; +import type { Tool } from "../../tools"; +import type { theme as Theme } from "../theme/theme"; + +const GRID_COLS = 20; +const GRID_ROWS = 10; +const GRID_CELLS = GRID_COLS * GRID_ROWS; +const GRID_GUTTER = " "; + +const CELL_FILLED = "⛁"; +const CELL_FILLED_MESSAGES = "⛃"; +const CELL_FREE = "⛶"; +const CELL_BUFFER = "⛝"; + +type CategoryId = "systemPrompt" | "systemTools" | "skills" | "messages"; + +interface CategoryInfo { + id: CategoryId; + label: string; + tokens: number; + color: "accent" | "warning" | "success" | "userMessageText"; + glyph: string; +} + +export interface ContextBreakdown { + model: Model | undefined; + contextWindow: number; + categories: CategoryInfo[]; + usedTokens: number; + autoCompactBufferTokens: number; + freeTokens: number; +} + +function estimateSkillsTokens(skills: readonly Skill[]): number { + const fragments: string[] = []; + for (const skill of skills) { + // "- name: description\n" wire framing tokenizes ~identically to the + // concatenated form, so encode each piece separately and sum. + fragments.push(skill.name, skill.description); + } + return countTokens(fragments); +} + +function estimateToolSchemaTokens(tools: ReadonlyArray>): number { + const fragments: string[] = []; + for (const tool of tools) { + fragments.push(tool.name, tool.description); + try { + fragments.push(JSON.stringify(tool.parameters ?? {})); + } catch { + // Schema may contain functions or cycles; ignore. + } + } + return countTokens(fragments); +} + +function estimateMessagesTokens(session: AgentSession): number { + let total = 0; + for (const message of session.messages) { + total += estimateTokens(message); + } + return total; +} + +/** + * Compute a breakdown of estimated context usage by category for the active + * session and model. + */ +export function computeContextBreakdown(session: AgentSession): ContextBreakdown { + const model = session.model; + const contextWindow = model?.contextWindow ?? 0; + + const skillsTokens = estimateSkillsTokens(session.skills); + const toolsTokens = estimateToolSchemaTokens(session.agent.state.tools); + const messagesTokens = estimateMessagesTokens(session); + + // The rendered system prompt already contains the skill descriptions and the + // markdown tool descriptions. To present a non-overlapping breakdown: + // System prompt = total system prompt text - skills section (tool descriptions stay) + // Tools = JSON tool schema sent separately on the wire + // Skills = the skill list embedded in the system prompt + // Messages = conversation messages + const systemPromptTextTokens = countTokens(session.systemPrompt); + const systemPromptTokens = Math.max(0, systemPromptTextTokens - skillsTokens); + + const categories: CategoryInfo[] = [ + { id: "systemPrompt", label: "System prompt", tokens: systemPromptTokens, color: "accent", glyph: CELL_FILLED }, + { id: "systemTools", label: "System tools", tokens: toolsTokens, color: "warning", glyph: CELL_FILLED }, + { id: "skills", label: "Skills", tokens: skillsTokens, color: "success", glyph: CELL_FILLED }, + { + id: "messages", + label: "Messages", + tokens: messagesTokens, + color: "userMessageText", + glyph: CELL_FILLED_MESSAGES, + }, + ]; + + const usedTokens = categories.reduce((sum, c) => sum + c.tokens, 0); + + let autoCompactBufferTokens = 0; + if (contextWindow > 0) { + const compactionSettings = session.settings.getGroup("compaction") as CompactionSettings; + if (compactionSettings.enabled && compactionSettings.strategy !== "off") { + const threshold = resolveThresholdTokens(contextWindow, compactionSettings); + autoCompactBufferTokens = Math.max(0, contextWindow - threshold); + } else { + autoCompactBufferTokens = 0; + } + // Even when fully disabled, fall back to a sensible reserve floor for display. + if (autoCompactBufferTokens === 0 && compactionSettings.enabled) { + autoCompactBufferTokens = effectiveReserveTokens(contextWindow, compactionSettings); + } + } + autoCompactBufferTokens = Math.min(autoCompactBufferTokens, Math.max(0, contextWindow - usedTokens)); + + const freeTokens = Math.max(0, contextWindow - usedTokens - autoCompactBufferTokens); + + return { + model, + contextWindow, + categories, + usedTokens, + autoCompactBufferTokens, + freeTokens, + }; +} + +interface CellSpec { + glyph: string; + color: "accent" | "warning" | "success" | "userMessageText" | "muted" | "dim"; +} + +function planCells(breakdown: ContextBreakdown): CellSpec[] { + const cells: CellSpec[] = []; + const window = breakdown.contextWindow; + + if (window <= 0) { + for (let i = 0; i < GRID_CELLS; i++) { + cells.push({ glyph: CELL_FREE, color: "dim" }); + } + return cells; + } + + const tokensPerCell = window / GRID_CELLS; + + const ratioCells = (tokens: number): number => { + if (tokens <= 0) return 0; + return Math.max(1, Math.round(tokens / tokensPerCell)); + }; + + const categoryCounts = breakdown.categories.map(category => ({ + category, + count: ratioCells(category.tokens), + })); + + let bufferCount = ratioCells(breakdown.autoCompactBufferTokens); + + let usedCount = categoryCounts.reduce((sum, c) => sum + c.count, 0); + + // Prevent the visualization from over-running the grid. + const maxUsable = GRID_CELLS - bufferCount; + if (usedCount > maxUsable) { + // Scale categories proportionally down to fit. + let overflow = usedCount - maxUsable; + // Trim from the largest categories first to preserve visibility for small ones. + const order = [...categoryCounts].sort((a, b) => b.count - a.count); + for (const entry of order) { + while (overflow > 0 && entry.count > 1) { + entry.count -= 1; + overflow -= 1; + } + } + usedCount = categoryCounts.reduce((sum, c) => sum + c.count, 0); + if (usedCount + bufferCount > GRID_CELLS) { + bufferCount = Math.max(0, GRID_CELLS - usedCount); + } + } + + for (const { category, count } of categoryCounts) { + for (let i = 0; i < count; i++) { + cells.push({ glyph: category.glyph, color: category.color }); + } + } + + const freeCount = Math.max(0, GRID_CELLS - cells.length - bufferCount); + for (let i = 0; i < freeCount; i++) { + cells.push({ glyph: CELL_FREE, color: "dim" }); + } + for (let i = 0; i < bufferCount; i++) { + cells.push({ glyph: CELL_BUFFER, color: "warning" }); + } + + // Pad to exactly GRID_CELLS in case rounding undershot. + while (cells.length < GRID_CELLS) { + cells.push({ glyph: CELL_FREE, color: "dim" }); + } + return cells.slice(0, GRID_CELLS); +} + +function percentString(part: number, whole: number, fractionDigits = 1): string { + if (whole <= 0) return "0%"; + const pct = (part / whole) * 100; + if (pct > 0 && pct < 0.05) return "<0.1%"; + return `${pct.toFixed(fractionDigits)}%`; +} + +function buildLegendLines(breakdown: ContextBreakdown, theme: typeof Theme): string[] { + const lines: string[] = []; + const { model, contextWindow, categories, usedTokens, autoCompactBufferTokens, freeTokens } = breakdown; + + const modelName = model?.name ?? model?.id ?? "no model"; + const modelId = model?.id ?? "unknown"; + const windowLabel = formatNumber(contextWindow).toLowerCase(); + + lines.push(theme.bold(`${modelName}`) + theme.fg("dim", ` (${windowLabel} context)`)); + lines.push(theme.fg("muted", `${modelId}[${windowLabel}]`)); + lines.push( + `${theme.bold(formatNumber(usedTokens))}${theme.fg("dim", `/${windowLabel} tokens`)}` + + theme.fg("muted", ` (${percentString(usedTokens, contextWindow)})`), + ); + lines.push(""); + lines.push(theme.fg("muted", "Estimated usage by category")); + + for (const category of categories) { + const dot = theme.fg(category.color, category.glyph); + const label = category.label; + const tokens = formatNumber(category.tokens); + const pct = percentString(category.tokens, contextWindow); + lines.push(`${dot} ${label}: ${theme.bold(tokens)} ${theme.fg("dim", `tokens (${pct})`)}`); + } + + const freeDot = theme.fg("dim", CELL_FREE); + lines.push( + `${freeDot} Free space: ${theme.bold(formatNumber(freeTokens))} ${theme.fg("dim", `(${percentString(freeTokens, contextWindow)})`)}`, + ); + + if (autoCompactBufferTokens > 0) { + const bufferDot = theme.fg("warning", CELL_BUFFER); + lines.push( + `${bufferDot} Autocompact buffer: ${theme.bold(formatNumber(autoCompactBufferTokens))} ${theme.fg( + "dim", + `tokens (${percentString(autoCompactBufferTokens, contextWindow)})`, + )}`, + ); + } + + return lines; +} + +/** + * Render a colorful context-usage panel as ANSI text. Output is a series of + * lines pairing the grid (left) with the legend (right). + */ +export function renderContextUsage(breakdown: ContextBreakdown, theme: typeof Theme): string { + if (breakdown.contextWindow <= 0) { + return theme.fg("muted", "Context usage is unavailable: no model is selected for this session."); + } + + const cells = planCells(breakdown); + const legend = buildLegendLines(breakdown, theme); + + const totalLines = Math.max(GRID_ROWS, legend.length); + const lines: string[] = []; + + for (let row = 0; row < totalLines; row++) { + let gridSegment = ""; + if (row < GRID_ROWS) { + const rowCells: string[] = []; + for (let col = 0; col < GRID_COLS; col++) { + const cell = cells[row * GRID_COLS + col]; + rowCells.push(theme.fg(cell.color, cell.glyph)); + } + gridSegment = rowCells.join(" "); + } else { + // Pad with blanks the same visible width as a grid row so legend lines + // past the grid stay aligned with their column. + const blank = " ".repeat(GRID_COLS * 2 - 1); + gridSegment = blank; + } + + const legendSegment = legend[row] ?? ""; + const line = legendSegment.length > 0 ? `${gridSegment}${GRID_GUTTER}${legendSegment}` : gridSegment; + lines.push(line); + } + + return lines.join("\n"); +} diff --git a/packages/coding-agent/src/session/compaction/compaction.ts b/packages/coding-agent/src/session/compaction/compaction.ts index b88d9aaba..58b69be08 100644 --- a/packages/coding-agent/src/session/compaction/compaction.ts +++ b/packages/coding-agent/src/session/compaction/compaction.ts @@ -26,6 +26,7 @@ import { getOpenAIResponsesHistoryPayload, normalizeResponsesToolCallId, } from "@oh-my-pi/pi-ai/utils"; +import { countTokens } from "@oh-my-pi/pi-natives"; import { logger, prompt } from "@oh-my-pi/pi-utils"; import compactionShortSummaryPrompt from "../../prompts/compaction/compaction-short-summary.md" with { type: "text" }; import compactionSummaryPrompt from "../../prompts/compaction/compaction-summary.md" with { type: "text" }; @@ -218,7 +219,7 @@ export function shouldCompact(contextTokens: number, contextWindow: number, sett return contextTokens > thresholdTokens; } -function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number { +export function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number { // Fixed token limit takes priority over percentage const thresholdTokens = settings.thresholdTokens; if (typeof thresholdTokens === "number" && Number.isFinite(thresholdTokens) && thresholdTokens > 0) { @@ -240,67 +241,79 @@ function resolveThresholdTokens(contextWindow: number, settings: CompactionSetti // ============================================================================ /** - * Estimate token count for a message using chars/4 heuristic. - * This is conservative (overestimates tokens). + * Image content has no tokenizer representation; charge a fixed estimate + * matching what providers typically bill for inline images. + */ +const IMAGE_TOKEN_ESTIMATE = 1200; + +/** + * Estimate token count for a message using cl100k_base via the native + * tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't + * publish one) but is within ~5–10% across English/code text. */ export function estimateTokens(message: AgentMessage): number { - let chars = 0; + const fragments: string[] = []; + let extra = 0; switch (message.role) { case "user": { const content = (message as { content: string | Array<{ type: string; text?: string }> }).content; if (typeof content === "string") { - chars = content.length; + fragments.push(content); } else if (Array.isArray(content)) { for (const block of content) { if (block.type === "text" && block.text) { - chars += block.text.length; + fragments.push(block.text); } } } - return Math.ceil(chars / 4); + break; } case "assistant": { const assistant = message as AssistantMessage; for (const block of assistant.content) { if (block.type === "text") { - chars += block.text.length; + fragments.push(block.text); } else if (block.type === "thinking") { - chars += block.thinking.length; + fragments.push(block.thinking); } else if (block.type === "toolCall") { - chars += block.name.length + JSON.stringify(block.arguments).length; + fragments.push(block.name); + fragments.push(JSON.stringify(block.arguments)); } } - return Math.ceil(chars / 4); + break; } case "hookMessage": case "toolResult": { if (typeof message.content === "string") { - chars = message.content.length; + fragments.push(message.content); } else { for (const block of message.content) { if (block.type === "text" && block.text) { - chars += block.text.length; - } - if (block.type === "image") { - chars += 4800; // Estimate images as 4000 chars, or 1200 tokens + fragments.push(block.text); + } else if (block.type === "image") { + extra += IMAGE_TOKEN_ESTIMATE; } } } - return Math.ceil(chars / 4); + break; } case "bashExecution": { - chars = message.command.length + message.output.length; - return Math.ceil(chars / 4); + fragments.push(message.command); + fragments.push(message.output); + break; } case "branchSummary": case "compactionSummary": { - chars = message.summary.length; - return Math.ceil(chars / 4); + fragments.push(message.summary); + break; } + default: + return 0; } - return 0; + if (fragments.length === 0) return extra; + return extra + countTokens(fragments); } function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number { diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index bd6eadc26..f54ffc1dc 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -355,6 +355,14 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ runtime.ctx.editor.setText(""); }, }, + { + name: "context", + description: "Show estimated context usage breakdown", + handle: (_command, runtime) => { + runtime.ctx.handleContextCommand(); + runtime.ctx.editor.setText(""); + }, + }, { name: "extensions", aliases: ["status"],