From 2ddc9c5bc96b113d4df2d8140998ff73ee08864b Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 31 May 2026 08:03:45 +0200 Subject: [PATCH] feat(coding-agent): added turn-budget parsing, multipliers and hard caps - Added +Nk/+Nm turn-budget parsing with whitespace-boundary matching, multipliers, and hard `!` indicator. - Added per-turn budget lifecycle plus APIs (`getTurnBudget`, `recordEvalSubagentUsage`) and hard-cap checks in eval runs. - Added hard budget observability in eval preludes and docs by exposing `budget.hard` and documenting ceiling modes. - Fixed streaming preview stutter with max-row tracking and padding, with tests for preview height and budget parsing. --- packages/coding-agent/CHANGELOG.md | 15 +- .../src/eval/__tests__/budget-bridge.test.ts | 57 +++++--- .../coding-agent/src/eval/agent-bridge.ts | 9 ++ .../coding-agent/src/eval/budget-bridge.ts | 24 +++- .../src/eval/js/shared/prelude.txt | 1 + packages/coding-agent/src/eval/py/prelude.py | 5 + .../src/modes/components/tool-execution.ts | 59 +++++++- .../coding-agent/src/modes/turn-budget.ts | 31 +++++ .../src/prompts/system/workflow-notice.md | 2 +- .../coding-agent/src/prompts/tools/eval.md | 4 +- packages/coding-agent/src/sdk.ts | 2 + .../coding-agent/src/session/agent-session.ts | 3 + .../src/session/session-manager.ts | 32 +++++ packages/coding-agent/src/tools/index.ts | 4 + .../test/core/turn-budget.test.ts | 58 ++++++++ .../test/streaming-preview-height.test.ts | 131 ++++++++++++++++++ 16 files changed, 400 insertions(+), 37 deletions(-) create mode 100644 packages/coding-agent/src/modes/turn-budget.ts create mode 100644 packages/coding-agent/test/core/turn-budget.test.ts create mode 100644 packages/coding-agent/test/streaming-preview-height.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 215abfc2f..dc55f0bd4 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,14 +1,15 @@ # Changelog ## [Unreleased] - ### Added +- Added support for decimal and `k`/`m` suffix turn-budget directives, enabling budgets like `+1.5k` and `+2m` in eval message parsing +- Changed eval budget resolution to honor a user `+Nk` directive over an active Goal Mode limit while falling back to Goal Mode when no per-turn ceiling is set - Added `agent()` eval options `agent_type`/`agentType`, `model`, `context`, and `label`, and returned structured JSON when `schema` is provided in JS and Python eval cells - Added `agent()` to the `eval` runtime so JS and Python cells can spawn one subagent through the existing task executor; JS eval also gained bounded `parallel()` and `pipeline()` helpers for orchestrating subagent calls. - Added a `workflow` magic keyword (mirrors `orchestrate`/`ultrathink`): the standalone word glows amber→green in the editor and appends a hidden notice steering the model to author deterministic multi-subagent fan-outs in `eval` (agent/parallel/pipeline). Matching is word-bounded and case-insensitive; the singular and plural both trigger, but inflections like `workflowed` do not. - Added `parallel()` and `pipeline()` to the Python `eval` runtime (thread-pool over the synchronous `agent()` bridge), mirroring the JS helpers: bounded pool (default 4, max 16), input-order preservation, a barrier between every `pipeline` stage, and contextvar propagation so `agent()` works inside worker threads. -- Added `log()`, `phase()`, and a `budget` object to both `eval` runtimes (Python and JS). `log`/`phase` emit progress/phase status lines; `budget.total`/`budget.spent()`/`budget.remaining()` expose the turn token ceiling and spend (backed by Goal Mode when active, else session output-token usage). +- Added `log()`, `phase()`, and a `budget` object to both `eval` runtimes (Python and JS). `log`/`phase` emit progress/phase status lines; `budget.total`/`budget.spent()`/`budget.remaining()`/`budget.hard` expose a real per-turn output-token budget. A `+Nk` directive in the user's message sets an advisory budget (the model self-limits via `budget.remaining()`); `+Nk!` (or an active Goal Mode budget) makes it a hard ceiling that blocks further eval `agent()` spawns once reached. `budget.spent()` counts output tokens spent this turn across the main loop and all eval-spawned subagents. - Added search support for virtual internal URLs (including `omp://` roots) by resolving and scanning in-memory internal resources as search targets alongside filesystem paths - Added expansion of virtual internal URL search targets so `search` can match multiple internal documents when given `omp://` - Added `/omfg ` slash command that drafts a TTSR rule from a complaint, validates it against the current conversation, saves it to project or `~/.omp/agent/rules`, and registers it live. @@ -17,6 +18,7 @@ ### Changed +- Fixed turn-budget parsing to match `+Nk` directives only at token boundaries, preventing values like `version 1.2.3`, `c++`, and `+500kfoo` from triggering a budget rule - Changed overflowing provider, hook-option, branch-message, agent, extension, and session-tree pickers to support fuzzy type-to-filter search. - Changed Shift+Ctrl+P to cycle role models backward instead of cycling forward without persisting. - Changed empty prompt input so `?` inserts a literal question mark instead of opening `/hotkeys`; use `/hotkeys` explicitly for the shortcut reference. @@ -25,6 +27,10 @@ - Changed `/omfg` to show a live draft panel with generation/validation/saving status and allow canceling an active rule request with `Esc` - Changed keybindings config to use `~/.omp/agent/keybindings.yml`, with automatic migration from legacy `keybindings.json` and continued support for `keybindings.yaml`. +### Removed + +- Removed the `/drop-images` slash command; use `/shake images`, which strips every image from the session through the same `dropImages()` path. + ### Fixed - Fixed `agent()` in eval to enforce plan-mode, spawn allowlist, and disabled-agent checks before launching subagents @@ -36,10 +42,7 @@ - Fixed auto-thinking sessions to persist the concrete resolved effort after classification, so resuming the session restores that level instead of returning to pending `auto`. - Fixed extension-registered CLI flags (e.g. `--spawn-peer `) leaking into the initial prompt: argv is re-parsed once the extension flag set is known so flag values are consumed instead of becoming messages or being misread as `@file` arguments. Registered flags shadow same-named built-ins, so a colliding flag (e.g. plan-mode's `--plan`) is parsed with the extension's semantics rather than being consumed by the built-in branch (which would otherwise eat the following message and corrupt the built-in field). Extension flags and `@file` arguments are now resolved before the session is created, so an unreadable initial `@file` exits without leaving a junk session/terminal breadcrumb behind. ([#1503](https://github.com/can1357/oh-my-pi/pull/1503)) - Fixed footer status-line truncation: the left stats and right model segments now truncate by terminal cell width (via `truncateToWidth`) and strip all VT/ANSI escapes (via `stripVTControlCharacters`) instead of a SGR-only regex plus code-point `substring`, so wide glyphs, OSC hyperlinks, and non-SGR sequences can no longer overflow the line. - -### Removed - -- Removed the `/drop-images` slash command; use `/shake images`, which strips every image from the session through the same `dropImages()` path. +- Fixed the streaming edit diff preview "box grows and shrinks repeatedly" stutter. A whole-file Myers re-diff is recomputed on every streamed chunk and its alignment is not monotonic in payload length — a partial or just-completed line transiently matches a duplicated line further down the file (a brace, a blank line, a repeated token), so the rendered change region gains and loses rows tick to tick. The streaming preview now reserves its high-water rendered height (measured at the real layout width, so soft-wrapped diff lines count exactly), so the box only ever grows mid-stream and collapses once when the edit finalizes. ## [15.7.2] - 2026-05-31 ### Added diff --git a/packages/coding-agent/src/eval/__tests__/budget-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/budget-bridge.test.ts index 185ef4b00..e54345b8f 100644 --- a/packages/coding-agent/src/eval/__tests__/budget-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/budget-bridge.test.ts @@ -4,8 +4,11 @@ import type { UsageStatistics } from "../../session/session-manager"; import type { ToolSession } from "../../tools"; import { runEvalBudget } from "../budget-bridge"; -function makeSession(parts: { goal?: GoalModeState; usage?: UsageStatistics }): ToolSession { +type TurnBudget = { total: number | null; spent: number; hard: boolean }; + +function makeSession(parts: { turn?: TurnBudget; goal?: GoalModeState; usage?: UsageStatistics }): ToolSession { return { + getTurnBudget: parts.turn ? () => parts.turn as TurnBudget : undefined, getGoalModeState: parts.goal ? () => parts.goal : undefined, getUsageStatistics: parts.usage ? () => parts.usage as UsageStatistics : undefined, } as unknown as ToolSession; @@ -15,13 +18,7 @@ function goalState(extra: Partial): GoalModeState { return { enabled: true, mode: "active", - goal: { - id: "g1", - status: "active", - tokensUsed: 0, - timeUsedSeconds: 0, - ...extra, - }, + goal: { id: "g1", status: "active", tokensUsed: 0, timeUsedSeconds: 0, ...extra }, } as GoalModeState; } @@ -30,23 +27,43 @@ function usage(output: number): UsageStatistics { } describe("runEvalBudget", () => { - it("reads tokenBudget/tokensUsed when Goal Mode is enabled", async () => { - const session = makeSession({ goal: goalState({ tokenBudget: 100000, tokensUsed: 4200 }) }); - expect(await runEvalBudget({}, { session })).toEqual({ total: 100000, spent: 4200 }); + it("prefers an active +Nk turn directive over Goal Mode", async () => { + const session = makeSession({ + turn: { total: 200_000, spent: 5_000, hard: true }, + goal: goalState({ tokenBudget: 100_000, tokensUsed: 4_200 }), + }); + expect(await runEvalBudget({}, { session })).toEqual({ total: 200_000, spent: 5_000, hard: true }); }); - it("returns null total when Goal Mode has no tokenBudget", async () => { - const session = makeSession({ goal: goalState({ tokenBudget: undefined, tokensUsed: 1234 }) }); - expect(await runEvalBudget({}, { session })).toEqual({ total: null, spent: 1234 }); + it("reports an advisory turn budget as hard:false", async () => { + const session = makeSession({ turn: { total: 50_000, spent: 1_000, hard: false } }); + expect(await runEvalBudget({}, { session })).toEqual({ total: 50_000, spent: 1_000, hard: false }); }); - it("falls back to session output tokens when Goal Mode is absent", async () => { - const session = makeSession({ usage: usage(777) }); - expect(await runEvalBudget({}, { session })).toEqual({ total: null, spent: 777 }); + it("falls through to Goal Mode when no turn directive set a ceiling", async () => { + const session = makeSession({ + turn: { total: null, spent: 7_777, hard: false }, + goal: goalState({ tokenBudget: 100_000, tokensUsed: 4_200 }), + }); + expect(await runEvalBudget({}, { session })).toEqual({ total: 100_000, spent: 4_200, hard: true }); }); - it("returns zero spent when neither getter is present", async () => { - const session = makeSession({}); - expect(await runEvalBudget({}, { session })).toEqual({ total: null, spent: 0 }); + it("treats a Goal Mode budget as hard, and a budgetless goal as no ceiling", async () => { + const withBudget = makeSession({ goal: goalState({ tokenBudget: 80_000, tokensUsed: 9_000 }) }); + expect(await runEvalBudget({}, { session: withBudget })).toEqual({ total: 80_000, spent: 9_000, hard: true }); + + const noBudget = makeSession({ goal: goalState({ tokenBudget: undefined, tokensUsed: 1_234 }) }); + expect(await runEvalBudget({}, { session: noBudget })).toEqual({ total: null, spent: 1_234, hard: false }); + }); + + it("reports no ceiling but still surfaces spend", async () => { + const fromTurn = makeSession({ turn: { total: null, spent: 333, hard: false } }); + expect(await runEvalBudget({}, { session: fromTurn })).toEqual({ total: null, spent: 333, hard: false }); + + const fromUsage = makeSession({ usage: usage(777) }); + expect(await runEvalBudget({}, { session: fromUsage })).toEqual({ total: null, spent: 777, hard: false }); + + const empty = makeSession({}); + expect(await runEvalBudget({}, { session: empty })).toEqual({ total: null, spent: 0, hard: false }); }); }); diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index c04715b00..0ce2ae070 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -175,6 +175,13 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption assertDepthAllowed(options.session); assertSpawnAllowed(options.session, agentName); + const turnBudget = options.session.getTurnBudget?.(); + if (turnBudget?.hard && turnBudget.total !== null && turnBudget.spent >= turnBudget.total) { + throw new ToolError( + `agent() blocked: turn token budget exhausted (${turnBudget.spent}/${turnBudget.total} output tokens). Raise or drop the +Nk! ceiling to continue.`, + ); + } + const { agents } = await taskDiscovery.discoverAgents(options.session.cwd); const agent = taskDiscovery.getAgent(agents, agentName); if (!agent) { @@ -260,6 +267,8 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption throw new ToolError(failureMessage); } + options.session.recordEvalSubagentUsage?.(result.usage?.output ?? 0); + options.emitStatus?.({ op: "agent", agent: result.agent, diff --git a/packages/coding-agent/src/eval/budget-bridge.ts b/packages/coding-agent/src/eval/budget-bridge.ts index 33924fa2d..9d90c14e1 100644 --- a/packages/coding-agent/src/eval/budget-bridge.ts +++ b/packages/coding-agent/src/eval/budget-bridge.ts @@ -2,9 +2,9 @@ * Host-side handler for the eval `budget` helper. * * Reports the active token ceiling and amount spent so kernel helpers can - * compute remaining budget. When Goal Mode is active the figures come from the - * goal's `tokenBudget`/`tokensUsed`; otherwise there is no ceiling and `spent` - * falls back to cumulative session output tokens. + * compute remaining budget. Precedence: a `+Nk`/`+Nk!` per-turn directive (the + * user's immediate intent) wins; otherwise an active Goal Mode budget; otherwise + * no ceiling, with `spent` still reflecting this turn's output where available. */ import type { ToolSession } from "../tools"; import type { JsStatusEvent } from "./js/shared/types"; @@ -21,18 +21,28 @@ export interface EvalBudgetBridgeOptions { export interface EvalBudgetResult { total: number | null; spent: number; + /** Whether the ceiling is enforced (eval `agent()` throws past it) vs advisory. */ + hard: boolean; } /** * Resolve the current token budget snapshot for an eval cell's `budget` helper. * The returned object is JSON-passed verbatim by the bridge transport; kernel - * helpers read `.total`/`.spent` directly. + * helpers read `.total`/`.spent`/`.hard` directly. */ export async function runEvalBudget(_args: unknown, options: EvalBudgetBridgeOptions): Promise { + const turn = options.session.getTurnBudget?.(); + if (turn && turn.total !== null) { + return { total: turn.total, spent: turn.spent, hard: turn.hard }; + } const goal = options.session.getGoalModeState?.(); if (goal?.enabled && goal.goal) { - return { total: goal.goal.tokenBudget ?? null, spent: goal.goal.tokensUsed ?? 0 }; + return { + total: goal.goal.tokenBudget ?? null, + spent: goal.goal.tokensUsed ?? 0, + hard: goal.goal.tokenBudget != null, + }; } - const usage = options.session.getUsageStatistics?.(); - return { total: null, spent: usage?.output ?? 0 }; + const spent = turn?.spent ?? options.session.getUsageStatistics?.()?.output ?? 0; + return { total: null, spent, hard: false }; } diff --git a/packages/coding-agent/src/eval/js/shared/prelude.txt b/packages/coding-agent/src/eval/js/shared/prelude.txt index be774c8e6..f4dc9b1fe 100644 --- a/packages/coding-agent/src/eval/js/shared/prelude.txt +++ b/packages/coding-agent/src/eval/js/shared/prelude.txt @@ -120,6 +120,7 @@ if (!globalThis.__omp_js_prelude_loaded__) { const s = await __budgetSnap(); return s.total == null ? Infinity : Math.max(0, Number(s.total) - Number(s.spent ?? 0)); }, + hard: async () => Boolean((await __budgetSnap()).hard), }; const display = value => { diff --git a/packages/coding-agent/src/eval/py/prelude.py b/packages/coding-agent/src/eval/py/prelude.py index 510f18f2e..55e5f11bc 100644 --- a/packages/coding-agent/src/eval/py/prelude.py +++ b/packages/coding-agent/src/eval/py/prelude.py @@ -586,6 +586,11 @@ if "__omp_prelude_loaded__" not in globals(): snap = _bridge_call("__budget__", {}) return (snap or {}).get("total") + @property + def hard(self): + snap = _bridge_call("__budget__", {}) + return bool((snap or {}).get("hard")) + def spent(self): snap = _bridge_call("__budget__", {}) return int((snap or {}).get("spent") or 0) diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index a3ebc18ad..bd11aa73c 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -45,6 +45,49 @@ function ensureInvalidate(component: unknown): Component { return c as Component; } +/** + * Wraps a streaming edit preview so its rendered height only ever grows while + * the tool args are still streaming, then collapses once on finalize. + * + * A whole-file line diff is recomputed from scratch on every streamed chunk, + * and the optimal Myers alignment is not monotonic in payload length: a + * partial — or just-completed — line keeps matching a duplicated line further + * down the file (a brace, a blank line, a repeated token), so the visible + * change region gains and loses rows tick to tick. That is the "box grows and + * shrinks repeatedly" stutter. Reserving the high-water row count (padding with + * blank rows the host Box fills with the tool background) holds the box steady + * for the whole stream; the finalized diff renders through a different, + * unwrapped path, so the one allowed collapse happens when args complete. + * + * Rows are measured at the real layout width, so soft-wrapped diff lines are + * counted exactly rather than approximated from newline counts. + */ +class StreamingPreviewHeight implements Component { + #child?: Component; + #maxRows = 0; + + setChild(child: Component): void { + this.#child = child; + } + + render(width: number): string[] { + const child = this.#child; + if (!child) return []; + const lines = child.render(width); + if (lines.length >= this.#maxRows) { + this.#maxRows = lines.length; + return lines; + } + const padded = lines.slice(); + while (padded.length < this.#maxRows) padded.push(""); + return padded; + } + + invalidate(): void { + this.#child?.invalidate(); + } +} + /** * Drop trailing removal/hunk-header lines that appear in a streaming diff * before the matching `+added` lines have arrived. Without this, a partial @@ -172,6 +215,9 @@ export class ToolExecutionComponent extends Container { #editDiffPreview?: PerFileDiffPreview[]; #editDiffAbort?: AbortController; #editDiffLastArgsKey?: string; + // Reserves the streaming edit preview's high-water height so the box never + // shrinks mid-stream; see StreamingPreviewHeight. + #streamPreviewHeight = new StreamingPreviewHeight(); // Cached converted images for Kitty protocol (which requires PNG), keyed by index #convertedImages: Map = new Map(); // Spinner animation for partial task results @@ -651,7 +697,18 @@ export class ToolExecutionComponent extends Container { try { const callComponent = renderer.renderCall(this.#getCallArgsForRender(), this.#renderState, theme); if (callComponent) { - this.#contentBox.addChild(ensureInvalidate(callComponent)); + const child = ensureInvalidate(callComponent); + // While edit args stream, the recomputed diff preview gains and + // loses rows tick to tick (non-monotonic Myers re-alignment), + // stuttering the box larger/smaller. Reserve the high-water + // height so it only grows mid-stream and collapses once the edit + // finalizes (a different, unwrapped render path). + if (isEditLikeToolName(this.#toolName) && !this.#result && !this.#argsComplete) { + this.#streamPreviewHeight.setChild(child); + this.#contentBox.addChild(this.#streamPreviewHeight); + } else { + this.#contentBox.addChild(child); + } } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); diff --git a/packages/coding-agent/src/modes/turn-budget.ts b/packages/coding-agent/src/modes/turn-budget.ts new file mode 100644 index 000000000..49e2b519f --- /dev/null +++ b/packages/coding-agent/src/modes/turn-budget.ts @@ -0,0 +1,31 @@ +/** + * "+Nk" turn token-budget directive. + * + * A standalone `+[k|m]` token in the user's message sets a per-turn + * output-token budget surfaced by the `eval` `budget` helper. By default it is + * ADVISORY — the model self-limits via `budget.remaining()`. Append `!` + * (`+500k!`) to make it a HARD ceiling: eval `agent()` refuses to spawn once the + * turn's spend reaches it. Matching is anchored to token boundaries so it does + * not fire on prices or version strings embedded in prose. + */ + +// Number, optional k/m multiplier, optional `!` hard marker, bounded by whitespace/string edges. +const TURN_BUDGET = /(?:^|\s)\+(\d+(?:\.\d+)?)([km])?(!)?(?=\s|$)/i; + +export interface TurnBudget { + /** Output-token ceiling for the turn. */ + total: number; + /** Whether the ceiling is enforced (eval `agent()` throws past it) vs advisory. */ + hard: boolean; +} + +/** Parse a `+Nk`/`+N`/`+Nm`(`!`) turn-budget directive from `text`, or null when absent. */ +export function parseTurnBudget(text: string): TurnBudget | null { + const match = TURN_BUDGET.exec(text); + if (!match) return null; + const value = Number(match[1]); + if (!Number.isFinite(value) || value <= 0) return null; + const unit = match[2]?.toLowerCase(); + const multiplier = unit === "k" ? 1_000 : unit === "m" ? 1_000_000 : 1; + return { total: Math.round(value * multiplier), hard: match[3] === "!" }; +} diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index dda50cdde..1c620fb85 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -18,7 +18,7 @@ State persists across cells, so scout in one cell and fan out in the next. Every - `pipeline(items, *stages, concurrency=4)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. - `llm(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out. - `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it. -- `budget` — `budget.total` (token ceiling, or `None` when none is set this turn), `budget.spent()`, `budget.remaining()` (`math.inf` when total is `None`). A ceiling exists only under an active turn budget (e.g. Goal Mode); otherwise total is `None` and a budget loop never engages — gate on `budget.total` first. +- `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget. Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across cells and turns for multi-phase work, reading each result before you decide the next phase. diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 8ed242a32..9eff66257 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -56,8 +56,8 @@ log(message) → None Emit a progress line above the status tree. phase(title) → None Start a phase; the status lines that follow group under it. -budget → token budget for this turn - {{#if py}}`budget.total` (ceiling or None), `budget.spent()` (output tokens), `budget.remaining()` (math.inf when no ceiling).{{/if}}{{#if js}}`await budget.total()` (ceiling or null), `await budget.spent()`, `await budget.remaining()` (Infinity when no ceiling).{{/if}} A ceiling exists only when one is set for the turn (e.g. Goal Mode); otherwise total is None/null. +budget → per-turn token budget + {{#if py}}`budget.total` (ceiling or None), `budget.spent()` (output tokens this turn), `budget.remaining()` (math.inf when no ceiling), `budget.hard` (bool).{{/if}}{{#if js}}`await budget.total()` (ceiling or null), `await budget.spent()`, `await budget.remaining()` (Infinity when no ceiling), `await budget.hard()`.{{/if}} A ceiling is set by a `+Nk` message directive (advisory) or `+Nk!`/Goal Mode (hard — `agent()` refuses to spawn past it); otherwise total is None/null and spend is still tracked across the turn (main loop + eval subagents). ``` diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 13be4a306..4faa8dc8c 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1242,6 +1242,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} getGoalModeState: () => session?.getGoalModeState(), getGoalRuntime: () => session?.goalRuntime, getUsageStatistics: () => sessionManager.getUsageStatistics(), + getTurnBudget: () => sessionManager.getTurnBudget(), + recordEvalSubagentUsage: output => sessionManager.recordEvalSubagentOutput(output), getClientBridge: () => session?.clientBridge, getCompactContext: () => session.formatCompactContext(), getTodoPhases: () => session.getTodoPhases(), diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 816986880..e8cb6701f 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -160,6 +160,7 @@ import { resolveMemoryBackend } from "../memory-backend"; import { getMnemosyneSessionState, type MnemosyneSessionState, setMnemosyneSessionState } from "../mnemosyne/state"; import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate"; import { getCurrentThemeName, theme } from "../modes/theme/theme"; +import { parseTurnBudget } from "../modes/turn-budget"; import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink"; import { containsWorkflow, WORKFLOW_NOTICE } from "../modes/workflow"; import type { PlanModeState } from "../plan-mode/state"; @@ -4115,6 +4116,8 @@ export class AgentSession { const keywordNotices: CustomMessage[] = []; if (!options?.synthetic) { const timestamp = Date.now(); + const turnBudget = parseTurnBudget(expandedText); + this.sessionManager.beginTurnBudget(turnBudget?.total ?? null, turnBudget?.hard ?? false); if (containsUltrathink(expandedText)) { keywordNotices.push({ role: "custom", diff --git a/packages/coding-agent/src/session/session-manager.ts b/packages/coding-agent/src/session/session-manager.ts index e5cbc61df..6c19193b6 100644 --- a/packages/coding-agent/src/session/session-manager.ts +++ b/packages/coding-agent/src/session/session-manager.ts @@ -1837,6 +1837,12 @@ export class SessionManager { premiumRequests: 0, cost: 0, } satisfies UsageStatistics; + /** Per-turn output-token budget set by a `+Nk` directive (total null when none this turn). */ + #turnBudget: { total: number | null; hard: boolean } = { total: null, hard: false }; + /** Cumulative `output` snapshot captured when the current turn budget window opened. */ + #turnBaselineOutput = 0; + /** Output tokens consumed by eval-spawned subagents in the current turn window. */ + #turnEvalOutput = 0; #persistWriter: NdjsonFileWriter | undefined; #persistWriterPath: string | undefined; #persistChain: Promise = Promise.resolve(); @@ -2397,6 +2403,32 @@ export class SessionManager { return this.#usageStatistics; } + /** + * Open a new per-turn budget window: snapshot the cumulative output baseline, + * reset the eval-subagent counter, and set the (optional) ceiling. Called once + * per real user message; `total` is null when no `+Nk` directive was present. + */ + beginTurnBudget(total: number | null, hard: boolean): void { + this.#turnBudget = { total, hard }; + this.#turnBaselineOutput = this.#usageStatistics.output; + this.#turnEvalOutput = 0; + } + + /** Record output tokens consumed by an eval-spawned subagent in the current turn. */ + recordEvalSubagentOutput(output: number): void { + if (Number.isFinite(output) && output > 0) this.#turnEvalOutput += output; + } + + /** + * Current turn budget for the eval `budget` helper: the ceiling (null = none), + * output tokens spent this turn (main loop + eval-spawned subagents, no + * double-count), and whether the ceiling is hard. + */ + getTurnBudget(): { total: number | null; spent: number; hard: boolean } { + const mainDelta = Math.max(0, this.#usageStatistics.output - this.#turnBaselineOutput); + return { total: this.#turnBudget.total, spent: mainDelta + this.#turnEvalOutput, hard: this.#turnBudget.hard }; + } + getSessionDir(): string { return this.sessionDir; } diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index ca506c3f7..0496c83c7 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -195,6 +195,10 @@ export interface ToolSession { getGoalRuntime?: () => GoalRuntime | undefined; /** Get cumulative session usage statistics (input/output tokens, cost). */ getUsageStatistics?: () => import("../session/session-manager").UsageStatistics; + /** Current per-turn token budget {total, spent, hard} for the eval `budget` helper. */ + getTurnBudget?: () => { total: number | null; spent: number; hard: boolean }; + /** Record output tokens consumed by an eval-spawned subagent toward the current turn budget. */ + recordEvalSubagentUsage?: (output: number) => void; /** Bridge to the connected client (e.g. ACP editor host). Tools should route fs/terminal/permission requests through this when available. */ getClientBridge?: () => ClientBridge | undefined; /** Get compact conversation context for subagents (excludes tool results, system prompts) */ diff --git a/packages/coding-agent/test/core/turn-budget.test.ts b/packages/coding-agent/test/core/turn-budget.test.ts new file mode 100644 index 000000000..ac3d997de --- /dev/null +++ b/packages/coding-agent/test/core/turn-budget.test.ts @@ -0,0 +1,58 @@ +import { describe, expect, it } from "bun:test"; +import { parseTurnBudget } from "@oh-my-pi/pi-coding-agent/modes/turn-budget"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; + +describe("parseTurnBudget", () => { + it("parses k/m multipliers, plain counts, and decimals", () => { + expect(parseTurnBudget("+500k")).toEqual({ total: 500_000, hard: false }); + expect(parseTurnBudget("+2m")).toEqual({ total: 2_000_000, hard: false }); + expect(parseTurnBudget("+1500")).toEqual({ total: 1_500, hard: false }); + expect(parseTurnBudget("+1.5k")).toEqual({ total: 1_500, hard: false }); + }); + + it("marks the budget hard only with a trailing !", () => { + expect(parseTurnBudget("+500k!")).toEqual({ total: 500_000, hard: true }); + expect(parseTurnBudget("audit this thoroughly +250k!")).toEqual({ total: 250_000, hard: true }); + }); + + it("matches the directive embedded in a sentence", () => { + expect(parseTurnBudget("be exhaustive +500k please")).toEqual({ total: 500_000, hard: false }); + }); + + it("ignores non-directives and junk", () => { + expect(parseTurnBudget("nothing here")).toBeNull(); + expect(parseTurnBudget("version 1.2.3")).toBeNull(); + expect(parseTurnBudget("+0")).toBeNull(); + expect(parseTurnBudget("c++ stuff")).toBeNull(); + // `+` glued to a non-numeric or trailing garbage must not match. + expect(parseTurnBudget("+500kfoo")).toBeNull(); + }); +}); + +describe("SessionManager turn budget accounting", () => { + it("snapshots a window, accrues eval-subagent output, and reports the ceiling + hard flag", () => { + const sm = SessionManager.inMemory(); + + sm.beginTurnBudget(100_000, true); + expect(sm.getTurnBudget()).toEqual({ total: 100_000, spent: 0, hard: true }); + + sm.recordEvalSubagentOutput(3_000); + sm.recordEvalSubagentOutput(1_500); + expect(sm.getTurnBudget()).toEqual({ total: 100_000, spent: 4_500, hard: true }); + + // Non-positive / non-finite deltas are ignored. + sm.recordEvalSubagentOutput(0); + sm.recordEvalSubagentOutput(Number.NaN); + expect(sm.getTurnBudget().spent).toBe(4_500); + }); + + it("resets spend and clears the ceiling when a new window opens with no directive", () => { + const sm = SessionManager.inMemory(); + sm.beginTurnBudget(50_000, false); + sm.recordEvalSubagentOutput(9_000); + expect(sm.getTurnBudget()).toEqual({ total: 50_000, spent: 9_000, hard: false }); + + sm.beginTurnBudget(null, false); + expect(sm.getTurnBudget()).toEqual({ total: null, spent: 0, hard: false }); + }); +}); diff --git a/packages/coding-agent/test/streaming-preview-height.test.ts b/packages/coding-agent/test/streaming-preview-height.test.ts new file mode 100644 index 000000000..764ef3460 --- /dev/null +++ b/packages/coding-agent/test/streaming-preview-height.test.ts @@ -0,0 +1,131 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import type { AgentTool } from "@oh-my-pi/pi-agent-core"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { EDIT_MODE_STRATEGIES } from "@oh-my-pi/pi-coding-agent/edit"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { TUI } from "@oh-my-pi/pi-tui"; +import { ToolExecutionComponent } from "../src/modes/components/tool-execution"; + +// Reproduces the streaming-edit "box grows and shrinks repeatedly" stutter and +// proves the render-level high-water reservation holds the box height steady. +// +// A whole-file Myers re-diff is recomputed on every streamed chunk; its optimal +// alignment is not monotonic in payload length, so the visible change region +// gains and loses rows as a partial/just-completed line transiently matches a +// duplicated line further down the file (here, the downstream `}` braces). +describe("streaming edit preview height (monotonic while streaming)", () => { + const RENDER_WIDTH = 80; + const oldBlock = ["function foo() {", " const x = 1;", " return x;", "}"].join("\n"); + const tail = ["", "function bar() {", " return 2;", "}", "", "function baz() {", " return 3;", "}", ""].join("\n"); + const fileContent = `${oldBlock}\n${tail}`; + const fullNew = [ + "function foo() {", + " const x = 1;", + " const y = 2;", + " const z = 3;", + " return x + y + z;", + "}", + ].join("\n"); + + let tmpDir: string; + let file: string; + let themed = false; + + beforeEach(async () => { + if (!themed) { + await initTheme(); + themed = true; + } + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "stream-height-")); + file = path.join(tmpDir, "mod.ts"); + await fs.writeFile(file, fileContent); + resetSettingsForTest(); + await Settings.init({ inMemory: true, cwd: tmpDir }); + }); + + afterEach(async () => { + resetSettingsForTest(); + await fs.rm(tmpDir, { recursive: true, force: true }); + }); + + // Char-by-char partials of the new function body. + const partials = Array.from({ length: fullNew.length }, (_, i) => fullNew.slice(0, i + 1)); + + function makeComponent(): { component: ToolExecutionComponent; settle: () => Promise } { + let resolveRender: (() => void) | null = null; + const uiStub = { + requestRender() { + const r = resolveRender; + resolveRender = null; + r?.(); + }, + } as unknown as TUI; + const tool = { mode: "replace" } as unknown as AgentTool; + const component = new ToolExecutionComponent( + "edit", + { path: file, edits: [{ old_text: oldBlock, new_text: fullNew.slice(0, 1) }] }, + {}, + tool, + uiStub, + tmpDir, + ); + // Resolve once the next async preview compute lands (or a short cap, so a + // deduped/no-op tick that never re-renders cannot hang the loop). + const settle = () => + Promise.race([new Promise(res => (resolveRender = res)), Bun.sleep(250).then(() => undefined)]); + return { component, settle }; + } + + test("rendered height never shrinks across streamed chunks, then collapses on finalize", async () => { + const { component, settle } = makeComponent(); + await settle(); + + const heights: number[] = []; + for (const newText of partials) { + const next = settle(); + component.updateArgs({ path: file, edits: [{ old_text: oldBlock, new_text: newText }] }); + await next; + heights.push(component.render(RENDER_WIDTH).length); + } + + // A real diff is on screen for the whole stream (not just the title row). + expect(Math.max(...heights)).toBeGreaterThan(5); + + // Core contract: the box only ever grows while args stream. + for (let i = 1; i < heights.length; i++) { + expect(heights[i]).toBeGreaterThanOrEqual(heights[i - 1]); + } + + // Finalize: args complete → unwrapped render path → the one allowed collapse. + component.setArgsComplete(); + await settle(); + const finalHeight = component.render(RENDER_WIDTH).length; + expect(finalHeight).toBeGreaterThan(1); // still shows a real diff + expect(finalHeight).toBeLessThanOrEqual(Math.max(...heights)); + }); + + test("the underlying diff genuinely oscillates (guard against a vacuous test)", async () => { + const ctx = { + cwd: tmpDir, + signal: new AbortController().signal, + snapshots: undefined as never, + allowFuzzy: true, + isStreaming: true, + }; + const rawLineCounts: number[] = []; + for (const newText of partials) { + const previews = await EDIT_MODE_STRATEGIES.replace.computeDiffPreview( + { path: file, edits: [{ old_text: oldBlock, new_text: newText }] }, + ctx, + ); + const first = previews?.[0]; + const diff = first && "diff" in first ? (first.diff ?? "") : ""; + rawLineCounts.push(diff ? diff.split("\n").length : 0); + } + const hasDecrease = rawLineCounts.some((count, i) => i > 0 && count < rawLineCounts[i - 1]); + expect(hasDecrease).toBe(true); + }); +});