feat(coding-agent): added /context command flow for interactive dispatch

- Added a `/context` slash command flow from registry to interactive-mode command dispatch.
- Added `handleContextCommand()` to the mode context interface and command-controller wiring.
- Added context usage breakdown utilities, cell allocation, and 20x10 usage rendering for token categories.
- Reworked compaction token estimation to use tokenizer counts, role aggregation, image token estimates, and fallback handling.
- Exported `resolveThresholdTokens()` as a public compaction helper.
This commit is contained in:
can1357
2026-04-30 02:26:04 +02:00
parent fbe051bcbd
commit f8be2ceda5
7 changed files with 362 additions and 22 deletions
+3
View File
@@ -1,8 +1,10 @@
# Changelog
## [Unreleased]
### Added
- Added the `/context` slash command to display an estimated context-usage breakdown panel for the current session
- Added `-LidA..LidB` syntax to delete inclusive line ranges in a single atom operation
- Added `LidA..LidB=TEXT` range-replace syntax with `\TEXT` and `\` continuation lines for multi-line replacement blocks
- Added shorthand cursor+insert operations in atom edits, including `^Lid` (insert before anchor), `^+TEXT`, `$+TEXT`, and `Lid+TEXT`/`@Lid+TEXT`
@@ -10,6 +12,7 @@
### Changed
- Changed token counting to use tokenizer-based estimates instead of a character-per-4 heuristic for context and compaction calculations
- Changed hashline anchor auto-rebase tolerance from ±2 lines to ±5 lines for stale Lid recovery
- Changed atom input handling so `#`-prefixed lines are treated as comments and ignored
- Changed execution when all edits are no-op `Lid=TEXT` replacements to return success with a no-change explanation instead of throwing
@@ -24,6 +24,7 @@ import { DynamicBorder } from "../../modes/components/dynamic-border";
import { PythonExecutionComponent } from "../../modes/components/python-execution";
import { getMarkdownTheme, getSymbolTheme, theme } from "../../modes/theme/theme";
import type { InteractiveModeContext } from "../../modes/types";
import { computeContextBreakdown, renderContextUsage } from "../../modes/utils/context-usage";
import { buildHotkeysMarkdown } from "../../modes/utils/hotkeys-markdown";
import { buildToolsMarkdown } from "../../modes/utils/tools-markdown";
import type { AsyncJobSnapshotItem } from "../../session/agent-session";
@@ -529,6 +530,22 @@ export class CommandController {
showMarkdownPanel(this.ctx, "Available Tools", tools);
}
handleContextCommand(): void {
const breakdown = computeContextBreakdown(this.ctx.session);
if (breakdown.contextWindow <= 0) {
this.ctx.showWarning("Context usage is unavailable: no model is selected for this session.");
return;
}
const output = renderContextUsage(breakdown, theme);
this.ctx.chatContainer.addChild(new Spacer(1));
this.ctx.chatContainer.addChild(new DynamicBorder());
this.ctx.chatContainer.addChild(new Text(theme.bold(theme.fg("accent", "Context Usage")), 1, 0));
this.ctx.chatContainer.addChild(new Spacer(1));
this.ctx.chatContainer.addChild(new Text(output, 1, 0));
this.ctx.chatContainer.addChild(new DynamicBorder());
this.ctx.ui.requestRender();
}
async handleMemoryCommand(text: string): Promise<void> {
const argumentText = text.slice(7).trim();
const action = argumentText.split(/\s+/, 1)[0]?.toLowerCase() || "view";
@@ -1399,6 +1399,10 @@ export class InteractiveMode implements InteractiveModeContext {
this.#commandController.handleToolsCommand();
}
handleContextCommand(): void {
this.#commandController.handleContextCommand();
}
#prepareSessionSwitch(): void {
this.#btwController.dispose();
this.#extensionUiController.clearExtensionTerminalInputListeners();
+1
View File
@@ -181,6 +181,7 @@ export interface InteractiveModeContext {
handleChangelogCommand(showFull?: boolean): Promise<void>;
handleHotkeysCommand(): void;
handleToolsCommand(): void;
handleContextCommand(): void;
handleDumpCommand(): void;
handleDebugTranscriptCommand(): Promise<void>;
handleClearCommand(): Promise<void>;
@@ -0,0 +1,294 @@
import type { Model } from "@oh-my-pi/pi-ai";
import { countTokens } from "@oh-my-pi/pi-natives";
import { formatNumber } from "@oh-my-pi/pi-utils";
import type { Skill } from "../../extensibility/skills";
import type { AgentSession } from "../../session/agent-session";
import type { CompactionSettings } from "../../session/compaction";
import { effectiveReserveTokens, estimateTokens, resolveThresholdTokens } from "../../session/compaction";
import type { Tool } from "../../tools";
import type { theme as Theme } from "../theme/theme";
const GRID_COLS = 20;
const GRID_ROWS = 10;
const GRID_CELLS = GRID_COLS * GRID_ROWS;
const GRID_GUTTER = " ";
const CELL_FILLED = "⛁";
const CELL_FILLED_MESSAGES = "⛃";
const CELL_FREE = "⛶";
const CELL_BUFFER = "⛝";
type CategoryId = "systemPrompt" | "systemTools" | "skills" | "messages";
interface CategoryInfo {
id: CategoryId;
label: string;
tokens: number;
color: "accent" | "warning" | "success" | "userMessageText";
glyph: string;
}
export interface ContextBreakdown {
model: Model | undefined;
contextWindow: number;
categories: CategoryInfo[];
usedTokens: number;
autoCompactBufferTokens: number;
freeTokens: number;
}
function estimateSkillsTokens(skills: readonly Skill[]): number {
const fragments: string[] = [];
for (const skill of skills) {
// "- name: description\n" wire framing tokenizes ~identically to the
// concatenated form, so encode each piece separately and sum.
fragments.push(skill.name, skill.description);
}
return countTokens(fragments);
}
function estimateToolSchemaTokens(tools: ReadonlyArray<Pick<Tool, "name" | "description" | "parameters">>): number {
const fragments: string[] = [];
for (const tool of tools) {
fragments.push(tool.name, tool.description);
try {
fragments.push(JSON.stringify(tool.parameters ?? {}));
} catch {
// Schema may contain functions or cycles; ignore.
}
}
return countTokens(fragments);
}
function estimateMessagesTokens(session: AgentSession): number {
let total = 0;
for (const message of session.messages) {
total += estimateTokens(message);
}
return total;
}
/**
* Compute a breakdown of estimated context usage by category for the active
* session and model.
*/
export function computeContextBreakdown(session: AgentSession): ContextBreakdown {
const model = session.model;
const contextWindow = model?.contextWindow ?? 0;
const skillsTokens = estimateSkillsTokens(session.skills);
const toolsTokens = estimateToolSchemaTokens(session.agent.state.tools);
const messagesTokens = estimateMessagesTokens(session);
// The rendered system prompt already contains the skill descriptions and the
// markdown tool descriptions. To present a non-overlapping breakdown:
// System prompt = total system prompt text - skills section (tool descriptions stay)
// Tools = JSON tool schema sent separately on the wire
// Skills = the skill list embedded in the system prompt
// Messages = conversation messages
const systemPromptTextTokens = countTokens(session.systemPrompt);
const systemPromptTokens = Math.max(0, systemPromptTextTokens - skillsTokens);
const categories: CategoryInfo[] = [
{ id: "systemPrompt", label: "System prompt", tokens: systemPromptTokens, color: "accent", glyph: CELL_FILLED },
{ id: "systemTools", label: "System tools", tokens: toolsTokens, color: "warning", glyph: CELL_FILLED },
{ id: "skills", label: "Skills", tokens: skillsTokens, color: "success", glyph: CELL_FILLED },
{
id: "messages",
label: "Messages",
tokens: messagesTokens,
color: "userMessageText",
glyph: CELL_FILLED_MESSAGES,
},
];
const usedTokens = categories.reduce((sum, c) => sum + c.tokens, 0);
let autoCompactBufferTokens = 0;
if (contextWindow > 0) {
const compactionSettings = session.settings.getGroup("compaction") as CompactionSettings;
if (compactionSettings.enabled && compactionSettings.strategy !== "off") {
const threshold = resolveThresholdTokens(contextWindow, compactionSettings);
autoCompactBufferTokens = Math.max(0, contextWindow - threshold);
} else {
autoCompactBufferTokens = 0;
}
// Even when fully disabled, fall back to a sensible reserve floor for display.
if (autoCompactBufferTokens === 0 && compactionSettings.enabled) {
autoCompactBufferTokens = effectiveReserveTokens(contextWindow, compactionSettings);
}
}
autoCompactBufferTokens = Math.min(autoCompactBufferTokens, Math.max(0, contextWindow - usedTokens));
const freeTokens = Math.max(0, contextWindow - usedTokens - autoCompactBufferTokens);
return {
model,
contextWindow,
categories,
usedTokens,
autoCompactBufferTokens,
freeTokens,
};
}
interface CellSpec {
glyph: string;
color: "accent" | "warning" | "success" | "userMessageText" | "muted" | "dim";
}
function planCells(breakdown: ContextBreakdown): CellSpec[] {
const cells: CellSpec[] = [];
const window = breakdown.contextWindow;
if (window <= 0) {
for (let i = 0; i < GRID_CELLS; i++) {
cells.push({ glyph: CELL_FREE, color: "dim" });
}
return cells;
}
const tokensPerCell = window / GRID_CELLS;
const ratioCells = (tokens: number): number => {
if (tokens <= 0) return 0;
return Math.max(1, Math.round(tokens / tokensPerCell));
};
const categoryCounts = breakdown.categories.map(category => ({
category,
count: ratioCells(category.tokens),
}));
let bufferCount = ratioCells(breakdown.autoCompactBufferTokens);
let usedCount = categoryCounts.reduce((sum, c) => sum + c.count, 0);
// Prevent the visualization from over-running the grid.
const maxUsable = GRID_CELLS - bufferCount;
if (usedCount > maxUsable) {
// Scale categories proportionally down to fit.
let overflow = usedCount - maxUsable;
// Trim from the largest categories first to preserve visibility for small ones.
const order = [...categoryCounts].sort((a, b) => b.count - a.count);
for (const entry of order) {
while (overflow > 0 && entry.count > 1) {
entry.count -= 1;
overflow -= 1;
}
}
usedCount = categoryCounts.reduce((sum, c) => sum + c.count, 0);
if (usedCount + bufferCount > GRID_CELLS) {
bufferCount = Math.max(0, GRID_CELLS - usedCount);
}
}
for (const { category, count } of categoryCounts) {
for (let i = 0; i < count; i++) {
cells.push({ glyph: category.glyph, color: category.color });
}
}
const freeCount = Math.max(0, GRID_CELLS - cells.length - bufferCount);
for (let i = 0; i < freeCount; i++) {
cells.push({ glyph: CELL_FREE, color: "dim" });
}
for (let i = 0; i < bufferCount; i++) {
cells.push({ glyph: CELL_BUFFER, color: "warning" });
}
// Pad to exactly GRID_CELLS in case rounding undershot.
while (cells.length < GRID_CELLS) {
cells.push({ glyph: CELL_FREE, color: "dim" });
}
return cells.slice(0, GRID_CELLS);
}
function percentString(part: number, whole: number, fractionDigits = 1): string {
if (whole <= 0) return "0%";
const pct = (part / whole) * 100;
if (pct > 0 && pct < 0.05) return "<0.1%";
return `${pct.toFixed(fractionDigits)}%`;
}
function buildLegendLines(breakdown: ContextBreakdown, theme: typeof Theme): string[] {
const lines: string[] = [];
const { model, contextWindow, categories, usedTokens, autoCompactBufferTokens, freeTokens } = breakdown;
const modelName = model?.name ?? model?.id ?? "no model";
const modelId = model?.id ?? "unknown";
const windowLabel = formatNumber(contextWindow).toLowerCase();
lines.push(theme.bold(`${modelName}`) + theme.fg("dim", ` (${windowLabel} context)`));
lines.push(theme.fg("muted", `${modelId}[${windowLabel}]`));
lines.push(
`${theme.bold(formatNumber(usedTokens))}${theme.fg("dim", `/${windowLabel} tokens`)}` +
theme.fg("muted", ` (${percentString(usedTokens, contextWindow)})`),
);
lines.push("");
lines.push(theme.fg("muted", "Estimated usage by category"));
for (const category of categories) {
const dot = theme.fg(category.color, category.glyph);
const label = category.label;
const tokens = formatNumber(category.tokens);
const pct = percentString(category.tokens, contextWindow);
lines.push(`${dot} ${label}: ${theme.bold(tokens)} ${theme.fg("dim", `tokens (${pct})`)}`);
}
const freeDot = theme.fg("dim", CELL_FREE);
lines.push(
`${freeDot} Free space: ${theme.bold(formatNumber(freeTokens))} ${theme.fg("dim", `(${percentString(freeTokens, contextWindow)})`)}`,
);
if (autoCompactBufferTokens > 0) {
const bufferDot = theme.fg("warning", CELL_BUFFER);
lines.push(
`${bufferDot} Autocompact buffer: ${theme.bold(formatNumber(autoCompactBufferTokens))} ${theme.fg(
"dim",
`tokens (${percentString(autoCompactBufferTokens, contextWindow)})`,
)}`,
);
}
return lines;
}
/**
* Render a colorful context-usage panel as ANSI text. Output is a series of
* lines pairing the grid (left) with the legend (right).
*/
export function renderContextUsage(breakdown: ContextBreakdown, theme: typeof Theme): string {
if (breakdown.contextWindow <= 0) {
return theme.fg("muted", "Context usage is unavailable: no model is selected for this session.");
}
const cells = planCells(breakdown);
const legend = buildLegendLines(breakdown, theme);
const totalLines = Math.max(GRID_ROWS, legend.length);
const lines: string[] = [];
for (let row = 0; row < totalLines; row++) {
let gridSegment = "";
if (row < GRID_ROWS) {
const rowCells: string[] = [];
for (let col = 0; col < GRID_COLS; col++) {
const cell = cells[row * GRID_COLS + col];
rowCells.push(theme.fg(cell.color, cell.glyph));
}
gridSegment = rowCells.join(" ");
} else {
// Pad with blanks the same visible width as a grid row so legend lines
// past the grid stay aligned with their column.
const blank = " ".repeat(GRID_COLS * 2 - 1);
gridSegment = blank;
}
const legendSegment = legend[row] ?? "";
const line = legendSegment.length > 0 ? `${gridSegment}${GRID_GUTTER}${legendSegment}` : gridSegment;
lines.push(line);
}
return lines.join("\n");
}
@@ -26,6 +26,7 @@ import {
getOpenAIResponsesHistoryPayload,
normalizeResponsesToolCallId,
} from "@oh-my-pi/pi-ai/utils";
import { countTokens } from "@oh-my-pi/pi-natives";
import { logger, prompt } from "@oh-my-pi/pi-utils";
import compactionShortSummaryPrompt from "../../prompts/compaction/compaction-short-summary.md" with { type: "text" };
import compactionSummaryPrompt from "../../prompts/compaction/compaction-summary.md" with { type: "text" };
@@ -218,7 +219,7 @@ export function shouldCompact(contextTokens: number, contextWindow: number, sett
return contextTokens > thresholdTokens;
}
function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number {
export function resolveThresholdTokens(contextWindow: number, settings: CompactionSettings): number {
// Fixed token limit takes priority over percentage
const thresholdTokens = settings.thresholdTokens;
if (typeof thresholdTokens === "number" && Number.isFinite(thresholdTokens) && thresholdTokens > 0) {
@@ -240,67 +241,79 @@ function resolveThresholdTokens(contextWindow: number, settings: CompactionSetti
// ============================================================================
/**
* Estimate token count for a message using chars/4 heuristic.
* This is conservative (overestimates tokens).
* Image content has no tokenizer representation; charge a fixed estimate
* matching what providers typically bill for inline images.
*/
const IMAGE_TOKEN_ESTIMATE = 1200;
/**
* Estimate token count for a message using cl100k_base via the native
* tokenizer. This is not Claude's first-party tokenizer (Anthropic doesn't
* publish one) but is within ~5–10% across English/code text.
*/
export function estimateTokens(message: AgentMessage): number {
let chars = 0;
const fragments: string[] = [];
let extra = 0;
switch (message.role) {
case "user": {
const content = (message as { content: string | Array<{ type: string; text?: string }> }).content;
if (typeof content === "string") {
chars = content.length;
fragments.push(content);
} else if (Array.isArray(content)) {
for (const block of content) {
if (block.type === "text" && block.text) {
chars += block.text.length;
fragments.push(block.text);
}
}
}
return Math.ceil(chars / 4);
break;
}
case "assistant": {
const assistant = message as AssistantMessage;
for (const block of assistant.content) {
if (block.type === "text") {
chars += block.text.length;
fragments.push(block.text);
} else if (block.type === "thinking") {
chars += block.thinking.length;
fragments.push(block.thinking);
} else if (block.type === "toolCall") {
chars += block.name.length + JSON.stringify(block.arguments).length;
fragments.push(block.name);
fragments.push(JSON.stringify(block.arguments));
}
}
return Math.ceil(chars / 4);
break;
}
case "hookMessage":
case "toolResult": {
if (typeof message.content === "string") {
chars = message.content.length;
fragments.push(message.content);
} else {
for (const block of message.content) {
if (block.type === "text" && block.text) {
chars += block.text.length;
}
if (block.type === "image") {
chars += 4800; // Estimate images as 4000 chars, or 1200 tokens
fragments.push(block.text);
} else if (block.type === "image") {
extra += IMAGE_TOKEN_ESTIMATE;
}
}
}
return Math.ceil(chars / 4);
break;
}
case "bashExecution": {
chars = message.command.length + message.output.length;
return Math.ceil(chars / 4);
fragments.push(message.command);
fragments.push(message.output);
break;
}
case "branchSummary":
case "compactionSummary": {
chars = message.summary.length;
return Math.ceil(chars / 4);
fragments.push(message.summary);
break;
}
default:
return 0;
}
return 0;
if (fragments.length === 0) return extra;
return extra + countTokens(fragments);
}
function estimateEntriesTokens(entries: SessionEntry[], startIndex: number, endIndex: number): number {
@@ -355,6 +355,14 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray<BuiltinSlashCommandSpec> = [
runtime.ctx.editor.setText("");
},
},
{
name: "context",
description: "Show estimated context usage breakdown",
handle: (_command, runtime) => {
runtime.ctx.handleContextCommand();
runtime.ctx.editor.setText("");
},
},
{
name: "extensions",
aliases: ["status"],