From 230514d9377a5add30ebb97b254a59e97b7e5cfd Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 15:34:29 +0000 Subject: [PATCH 01/28] fix(mcp): cap automatic reconnect bursts to prevent fork-bomb MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A stdio MCP server that completes the initialize + tools/list handshake and then exits cleanly will fire `transport.onClose` on every clean exit, and the old `MCPManager.reconnectServer` path spawned again unconditionally. A misconfigured PHP-shebang MCP (e.g. Laravel Boost in a non-Laravel project) hit this loop and forked 66 487 `php84` processes parented directly to the agent's `bun` PID until macOS force-rebooted. Add a per-server sliding-window circuit breaker: at most 5 reconnect attempts per 30 s window. The transport `onClose` callback and the per-tool-call retry in `tool-bridge` are subject to the breaker; `/mcp reconnect` passes `{ manual: true }` to reset the window so users can recover after fixing the underlying misconfiguration. Stale `onClose` is detached when the breaker trips so a late EOF event cannot re-arm the loop. Defended by `mcp-reconnect-storm.test.ts`: a Bun stdio fixture answers the handshake and exits, then asserts the spawn count stays at ≤ 10 (was 127 without the fix). Fixes #1592 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/mcp/manager.ts | 80 ++++++++++++++++++- .../controllers/mcp-command-controller.ts | 2 +- .../test/fixtures/crash-after-init-mcp.ts | 59 ++++++++++++++ .../test/mcp-reconnect-storm.test.ts | 75 +++++++++++++++++ 5 files changed, 215 insertions(+), 5 deletions(-) create mode 100755 packages/coding-agent/test/fixtures/crash-after-init-mcp.ts create mode 100644 packages/coding-agent/test/mcp-reconnect-storm.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9e22dabb5..d83cf2c9f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed unbounded MCP reconnect loop that could fork-bomb the host when a stdio MCP server completes the `initialize`/`tools/list` handshake and then exits. `MCPManager` now enforces a per-server crash circuit breaker (5 reconnects per 30 s window) on the automatic `transport.onClose` path; manual `/mcp reconnect` resets the window so users can recover after fixing the misconfiguration ([#1592](https://github.com/can1357/oh-my-pi/issues/1592)). + ## [15.7.4] - 2026-05-31 ### Removed diff --git a/packages/coding-agent/src/mcp/manager.ts b/packages/coding-agent/src/mcp/manager.ts index 91260afe1..21cdd2ba4 100644 --- a/packages/coding-agent/src/mcp/manager.ts +++ b/packages/coding-agent/src/mcp/manager.ts @@ -59,6 +59,27 @@ type TrackedPromise = { const STARTUP_TIMEOUT_MS = 250; +/** + * Per-server reconnect-storm circuit breaker. + * + * `transport.onClose` (wired in {@link MCPManager.connectServers} and + * {@link MCPManager.#connectAndWireServer}) fires `reconnectServer` on every + * clean process exit, so a stdio MCP server that completes the + * `initialize` + `tools/list` handshake and then exits will pull the agent + * into a fork loop with no rate limit. That pathology shipped in issue #1592 + * (a `php`-shebang MCP fork-bombing macOS, parented directly to the agent's + * `bun` PID via shebang exec). + * + * We keep the sliding window short — older crashes age out so a single + * transient failure stays cheap — but cap the burst tightly enough that the + * agent never spawns more than `RECONNECT_BURST_LIMIT * #doReconnect retries` + * (≤ 25) processes per stuck server per window. Manual `/mcp reconnect` + * resets the window so users can recover after fixing the underlying + * misconfiguration. + */ +const RECONNECT_BURST_WINDOW_MS = 30_000; +const RECONNECT_BURST_LIMIT = 5; + function trackPromise(promise: Promise): TrackedPromise { const tracked: TrackedPromise = { promise, status: "pending" }; promise.then( @@ -166,6 +187,11 @@ export class MCPManager { #pendingReconnections = new Map>(); /** Preserved configs for reconnection after connection loss. */ #serverConfigs = new Map(); + /** + * Timestamps of recent `reconnectServer` invocations per server, used by the + * crash-storm circuit breaker (see {@link RECONNECT_BURST_LIMIT}). + */ + #reconnectHistory = new Map(); /** Monotonic epoch incremented on disconnectAll to invalidate stale reconnections. */ #epoch = 0; @@ -666,6 +692,7 @@ export class MCPManager { this.#sources.delete(name); this.#serverConfigs.delete(name); this.#pendingResourceRefresh.delete(name); + this.#reconnectHistory.delete(name); const connection = this.#connections.get(name); @@ -714,24 +741,69 @@ export class MCPManager { this.#connections.clear(); this.#tools = []; this.#subscribedResources.clear(); + this.#reconnectHistory.clear(); } /** * Reconnect to a server after a connection failure. + * * Tears down the stale connection, re-resolves auth, establishes a new - * connection, reloads tools, and notifies consumers. - * Concurrent calls for the same server share one reconnection attempt. - * Returns the new connection, or null if reconnection failed. + * connection, reloads tools, and notifies consumers. Concurrent calls for + * the same server share one reconnection attempt. Returns the new + * connection, or `null` if reconnection failed or the per-server crash + * burst limit (see {@link RECONNECT_BURST_LIMIT}) is exceeded. + * + * @param options.manual - When `true`, resets the crash-burst window so a + * user-driven retry (e.g. `/mcp reconnect`) is never blocked by an + * earlier storm. Defaults to `false`; the transport `onClose` callback + * and the per-tool-call retry path in `tool-bridge` MUST NOT set it. */ - async reconnectServer(name: string): Promise { + async reconnectServer(name: string, options?: { manual?: boolean }): Promise { + if (options?.manual) { + this.#reconnectHistory.delete(name); + } + const pending = this.#pendingReconnections.get(name); if (pending) return pending; + if (this.#tripReconnectBreaker(name)) { + return null; + } + const attempt = this.#doReconnect(name); this.#pendingReconnections.set(name, attempt); return attempt.finally(() => this.#pendingReconnections.delete(name)); } + /** + * Record a reconnect attempt against the per-server crash window and report + * whether the circuit breaker is now open. Sliding window: entries older + * than {@link RECONNECT_BURST_WINDOW_MS} are pruned before the new + * timestamp is appended, so a single transient failure ages out cheaply + * but repeated rapid crashes accumulate until the limit is hit. + */ + #tripReconnectBreaker(name: string): boolean { + const now = Date.now(); + const previous = this.#reconnectHistory.get(name) ?? []; + const recent = previous.filter(ts => now - ts < RECONNECT_BURST_WINDOW_MS); + recent.push(now); + this.#reconnectHistory.set(name, recent); + + if (recent.length > RECONNECT_BURST_LIMIT) { + logger.error("MCP server crashed too many times; suspending automatic reconnects", { + path: `mcp:${name}`, + crashes: recent.length, + windowMs: RECONNECT_BURST_WINDOW_MS, + }); + // Detach the closed connection's onClose so a late EOF event on a + // process we already gave up on cannot re-trigger this path. + const stale = this.#connections.get(name); + if (stale) stale.transport.onClose = undefined; + return true; + } + return false; + } + async #doReconnect(name: string): Promise { const oldConnection = this.#connections.get(name); const config = oldConnection?.config ?? this.#serverConfigs.get(name); diff --git a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts index 8ad770813..756328b4b 100644 --- a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts @@ -1425,7 +1425,7 @@ export class MCPCommandController { this.#showMessage(["", theme.fg("muted", `Reconnecting to "${name}"...`), ""].join("\n")); try { - const connection = await this.ctx.mcpManager.reconnectServer(name); + const connection = await this.ctx.mcpManager.reconnectServer(name, { manual: true }); if (connection) { // refreshMCPTools re-registers tools and preserves the user's prior // MCP tool selection. No need to call activateDiscoveredMCPTools — diff --git a/packages/coding-agent/test/fixtures/crash-after-init-mcp.ts b/packages/coding-agent/test/fixtures/crash-after-init-mcp.ts new file mode 100755 index 000000000..f1c0b0c19 --- /dev/null +++ b/packages/coding-agent/test/fixtures/crash-after-init-mcp.ts @@ -0,0 +1,59 @@ +#!/usr/bin/env bun +/** + * Test fixture: a minimal stdio MCP server that completes the initialize + + * tools/list handshake and then exits cleanly. Models a misconfigured PHP + * MCP server (e.g. Laravel Boost in a non-Laravel project) that successfully + * advertises tools and then dies on the very next event-loop tick. + * + * Reproduces issue #1592: without a crash circuit breaker, every exit fires + * `transport.onClose`, which triggers an unbounded reconnect storm — the + * spindump in the bug report shows 66 487 PHP processes parented to the + * agent's `bun` PID. + * + * Each invocation atomically appends the PID + timestamp to the path in + * `$OMP_TEST_SPAWN_LOG`, so the test can count spawns without racing. + */ +import * as fs from "node:fs"; +import * as readline from "node:readline"; + +const spawnLog = Bun.env.OMP_TEST_SPAWN_LOG; +if (spawnLog) { + fs.appendFileSync(spawnLog, `${process.pid} ${Date.now()}\n`); +} + +const rl = readline.createInterface({ input: process.stdin }); + +function send(message: Record): void { + process.stdout.write(`${JSON.stringify(message)}\n`); +} + +rl.on("line", line => { + let message: { id?: number | string; method?: string }; + try { + message = JSON.parse(line); + } catch { + return; + } + + if (message.method === "initialize" && message.id !== undefined) { + send({ + jsonrpc: "2.0", + id: message.id, + result: { + protocolVersion: "2025-03-26", + capabilities: { tools: {} }, + serverInfo: { name: "crash-after-init", version: "1.0.0" }, + }, + }); + return; + } + + if (message.method === "tools/list" && message.id !== undefined) { + send({ jsonrpc: "2.0", id: message.id, result: { tools: [] } }); + // Exit on the next tick so the response is fully flushed before EOF. + setImmediate(() => process.exit(0)); + return; + } +}); + +rl.on("close", () => process.exit(0)); diff --git a/packages/coding-agent/test/mcp-reconnect-storm.test.ts b/packages/coding-agent/test/mcp-reconnect-storm.test.ts new file mode 100644 index 000000000..b1ecbd44e --- /dev/null +++ b/packages/coding-agent/test/mcp-reconnect-storm.test.ts @@ -0,0 +1,75 @@ +/** + * Regression test for issue #1592: an MCP stdio server that exits immediately + * after completing initialize + tools/list must not trigger an unbounded + * respawn loop. + * + * The reporter's agent forked 66 487 PHP child processes in ~7 minutes + * (~158 spawns/sec) before macOS force-rebooted. The crashing fixture below + * models that pathology: each spawn answers the handshake and exits cleanly, + * which fires `transport.onClose` → `reconnectServer` with no rate limiter + * in the unpatched build. + * + * The contract this test defends: per-server crash bursts are capped so that + * even a fast-crashing stdio server stays well below the OS process budget. + */ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { MCPManager } from "../src/mcp/manager"; +import type { MCPStdioServerConfig } from "../src/mcp/types"; + +const FIXTURE_PATH = path.join(import.meta.dir, "fixtures", "crash-after-init-mcp.ts"); +const BUN_EXEC = process.execPath; + +describe("MCP reconnect storm (issue #1592)", () => { + let workDir: string; + let spawnLog: string; + + beforeEach(() => { + workDir = fs.mkdtempSync(path.join(os.tmpdir(), "omp-mcp-storm-")); + spawnLog = path.join(workDir, "spawns.log"); + fs.writeFileSync(spawnLog, ""); + }); + + afterEach(() => { + fs.rmSync(workDir, { recursive: true, force: true }); + }); + + function countSpawns(): number { + const text = fs.readFileSync(spawnLog, "utf8"); + return text.split("\n").filter(line => line.trim().length > 0).length; + } + + it("stops respawning after a burst of immediate exits", async () => { + const manager = new MCPManager(workDir); + const config: MCPStdioServerConfig = { + type: "stdio", + command: BUN_EXEC, + args: [FIXTURE_PATH], + env: { OMP_TEST_SPAWN_LOG: spawnLog }, + }; + + try { + await manager.connectServers({ crashy: config }, {}); + // Give the reconnect loop generous time to fire. With the bug this + // produced thousands of processes within a second; with the fix the + // circuit breaker caps the per-server spawn budget. + await Bun.sleep(3000); + } finally { + await manager.disconnectAll(); + } + + const spawns = countSpawns(); + // `RECONNECT_BURST_LIMIT` (5) is the per-server reconnect cap inside + // the burst window. The initial connect from `connectServers` adds one + // more spawn. On the "initialize + tools/list succeed, then exit" path + // the inner retry-with-backoff in `#doReconnect` never fires, so the + // steady-state ceiling is `1 + RECONNECT_BURST_LIMIT + 1` ≈ 7 spawns. + // 10 leaves room for scheduling jitter without weakening the bound. + expect(spawns).toBeLessThanOrEqual(10); + // Sanity check: we did spawn at least once. If the fixture never ran + // the regression target is wrong and the test is meaningless. + expect(spawns).toBeGreaterThan(0); + }, 15_000); +}); From 6881c5cc417d7f9be83b4205ae2204da9bf140e2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 15:38:21 +0000 Subject: [PATCH 02/28] fix(mcp): drop stale connection when reconnect breaker trips Leaving the dead connection in `#connections` made `getConnectionStatus` report `connected` and `waitForConnection` hand a closed transport to callers after the breaker had explicitly suspended the server. Mirror `#doReconnect`'s teardown: detach `onClose`, fire-and-forget `transport.close()`, and drop the entry from `#connections` (plus its in-flight slots in `#pendingConnections`/`#pendingToolLoads`). Tools stay registered in `#tools` so the user can recover with `/mcp reconnect`. Test asserts `getConnectionStatus("crashy") === "disconnected"` after the burst. Refs #1592 --- packages/coding-agent/src/mcp/manager.ts | 17 ++++++++-- .../test/mcp-reconnect-storm.test.ts | 31 ++++++++++++------- 2 files changed, 33 insertions(+), 15 deletions(-) diff --git a/packages/coding-agent/src/mcp/manager.ts b/packages/coding-agent/src/mcp/manager.ts index 21cdd2ba4..8477b02d8 100644 --- a/packages/coding-agent/src/mcp/manager.ts +++ b/packages/coding-agent/src/mcp/manager.ts @@ -795,10 +795,21 @@ export class MCPManager { crashes: recent.length, windowMs: RECONNECT_BURST_WINDOW_MS, }); - // Detach the closed connection's onClose so a late EOF event on a - // process we already gave up on cannot re-trigger this path. + // Tear down the stale connection so `getConnectionStatus()` no + // longer reports it as "connected" and `waitForConnection()` does + // not hand a closed transport to callers. Tools stay registered + // in `#tools` — the user can recover with `/mcp reconnect ` + // once they've fixed the underlying misconfiguration. Mirrors the + // teardown in `#doReconnect`: detach `onClose` first so the + // transport's own `close()` cannot re-arm this path. const stale = this.#connections.get(name); - if (stale) stale.transport.onClose = undefined; + if (stale) { + stale.transport.onClose = undefined; + void stale.transport.close().catch(() => {}); + this.#connections.delete(name); + } + this.#pendingConnections.delete(name); + this.#pendingToolLoads.delete(name); return true; } return false; diff --git a/packages/coding-agent/test/mcp-reconnect-storm.test.ts b/packages/coding-agent/test/mcp-reconnect-storm.test.ts index b1ecbd44e..6f75e8955 100644 --- a/packages/coding-agent/test/mcp-reconnect-storm.test.ts +++ b/packages/coding-agent/test/mcp-reconnect-storm.test.ts @@ -56,20 +56,27 @@ describe("MCP reconnect storm (issue #1592)", () => { // produced thousands of processes within a second; with the fix the // circuit breaker caps the per-server spawn budget. await Bun.sleep(3000); + + const spawns = countSpawns(); + // `RECONNECT_BURST_LIMIT` (5) is the per-server reconnect cap inside + // the burst window. The initial connect from `connectServers` adds + // one more spawn. On the "initialize + tools/list succeed, then + // exit" path the inner retry-with-backoff in `#doReconnect` never + // fires, so the steady-state ceiling is + // `1 + RECONNECT_BURST_LIMIT + 1` ≈ 7 spawns. 10 leaves room for + // scheduling jitter without weakening the bound. + expect(spawns).toBeLessThanOrEqual(10); + // Sanity check: we did spawn at least once. If the fixture never ran + // the regression target is wrong and the test is meaningless. + expect(spawns).toBeGreaterThan(0); + + // Once the breaker trips, the stale connection must be torn down so + // `getConnectionStatus`/`waitForConnection` cannot hand callers a + // dead transport. Tools stay registered in the manager's tool list + // so the user can recover via `/mcp reconnect`. + expect(manager.getConnectionStatus("crashy")).toBe("disconnected"); } finally { await manager.disconnectAll(); } - - const spawns = countSpawns(); - // `RECONNECT_BURST_LIMIT` (5) is the per-server reconnect cap inside - // the burst window. The initial connect from `connectServers` adds one - // more spawn. On the "initialize + tools/list succeed, then exit" path - // the inner retry-with-backoff in `#doReconnect` never fires, so the - // steady-state ceiling is `1 + RECONNECT_BURST_LIMIT + 1` ≈ 7 spawns. - // 10 leaves room for scheduling jitter without weakening the bound. - expect(spawns).toBeLessThanOrEqual(10); - // Sanity check: we did spawn at least once. If the fixture never ran - // the regression target is wrong and the test is meaningless. - expect(spawns).toBeGreaterThan(0); }, 15_000); }); From 0887d28c7ff47d608101953f4ba3f4be3581adaa Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 16:47:24 +0000 Subject: [PATCH 03/28] fix(tui): deferred bottom-anchored shrink on unknown posix viewports MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A shrink across the viewport boundary on POSIX terminals that cannot report scrollback position (kitty, plain xterm) fell through to viewportRepaint. Repainting bottom-anchored newLines at newLength - height left rows newLength - height .. prevLength - height - 1 already committed to native scrollback, so they reappeared at the viewport top — two duplicated rows at the scrollback/viewport boundary in bjin's trace.\n\nMark scrollback dirty and emit deferredShrink instead (padding to the previous row count) so no native rows are re-emitted; the next checkpoint rebuild (e.g. prompt submit -> refreshNativeScrollbackIfDirty) cleans up.\n\nFixes #1566 --- packages/tui/src/tui.ts | 8 +++- packages/tui/test/render-regressions.test.ts | 50 ++++++++++++++++++++ 2 files changed, 57 insertions(+), 1 deletion(-) diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index c0bc01e36..d7e4aca56 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1335,8 +1335,14 @@ export class TUI extends Container { ) { return { kind: "historyRebuild" }; } + // POSIX terminals that cannot report viewport position fall through here + // (`canRebuildNativeScrollbackLive` is false): a viewport-only repaint would + // bottom-anchor `newLines` and re-emit the rows between the new and old + // viewport tops on top of the copies the terminal already kept in native + // scrollback. Pad to the previous row count instead and let the next + // checkpoint rebuild (e.g. prompt submit) clean up. this.#markNativeScrollbackDirty(); - return { kind: "viewportRepaint" }; + return { kind: "deferredShrink", paddedLength: this.#previousLines.length }; } const suppressSuffixScroll = this.#suppressNextSuffixScroll; diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index c0cd54602..9d00fd27e 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -1531,6 +1531,56 @@ describe("TUI terminal-state regressions", () => { tui.stop(); } }); + it("defers bottom-anchored shrink when POSIX viewport state is unknown", async () => { + // Repro for #1566 follow-up (kitty/Linux): a bottom-anchored shrink across the + // viewport boundary used to fall through to `viewportRepaint`, which redrew the + // new transcript at `newLength - height` while leaving rows + // `[newLength - height .. prevLength - height - 1]` already in native + // scrollback — they reappeared at the top of the viewport, duplicating two rows + // at the boundary in the captured trace. + const term = new UnknownViewportTerminal(40, 6); + const tui = new TUI(term); + const body = rows("line-", 12); + const component = new MutableLinesComponent([...body, "spinner-row", "spacer-row", "prompt-row"]); + tui.addChild(component); + + try { + tui.start(); + await settle(term); + + component.setLines([...body, "prompt-row"]); + tui.requestRender(); + await settle(term); + + const scrollback = term.getScrollBuffer(); + for (let i = 0; i < body.length; i++) { + const pattern = new RegExp(`\\bline-${i}\\b`); + expect( + countMatches(scrollback, pattern), + `line-${i} must not duplicate at boundary`, + ).toBeLessThanOrEqual(1); + } + + expect(tui.refreshNativeScrollbackIfDirty({ allowUnknownViewport: true })).toBe(true); + await settle(term); + expect(visible(term).map(line => line.trim())).toEqual([ + "line-7", + "line-8", + "line-9", + "line-10", + "line-11", + "prompt-row", + ]); + const rebuilt = term.getScrollBuffer(); + for (let i = 0; i < body.length; i++) { + const pattern = new RegExp(`\\bline-${i}\\b`); + expect(countMatches(rebuilt, pattern), `line-${i} appears once post-checkpoint`).toBe(1); + } + } finally { + tui.stop(); + } + }); + it("renders streaming row inserts on WSL Windows Terminal even when viewport probe is unavailable", async () => { const originalPlatform = process.platform; Object.defineProperty(process, "platform", { configurable: true, value: "linux" }); From 0b5fcdf2dc53746d98a6cf72b449e296e5dc6e19 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 17:03:01 +0000 Subject: [PATCH 04/28] fix(tui): yielded scrollback when deferred shrink would blank the viewport MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex review on #1599: when a bottom-anchored shrink lands on an unknown POSIX viewport and newLines.length <= scrollbackHighWater, the padded deferredShrink draws the viewport entirely past the end of the new transcript — every viewport row renders as blank, hiding the prompt until the next checkpoint.\n\nFall through to historyRebuild for that case. The yank is the lesser evil vs. a blank viewport that the user cannot interact with to trigger a checkpoint. Small shrinks (where some new content still sits above the scrollback boundary) keep the deferred behavior added in 0887d28c7. --- packages/tui/src/tui.ts | 19 ++++++--- packages/tui/test/render-regressions.test.ts | 45 ++++++++++++++++++++ 2 files changed, 59 insertions(+), 5 deletions(-) diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index d7e4aca56..e3887b664 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1336,11 +1336,20 @@ export class TUI extends Container { return { kind: "historyRebuild" }; } // POSIX terminals that cannot report viewport position fall through here - // (`canRebuildNativeScrollbackLive` is false): a viewport-only repaint would - // bottom-anchor `newLines` and re-emit the rows between the new and old - // viewport tops on top of the copies the terminal already kept in native - // scrollback. Pad to the previous row count instead and let the next - // checkpoint rebuild (e.g. prompt submit) clean up. + // (`canRebuildNativeScrollbackLive` is false). A viewport-only repaint would + // re-emit the rows between the new and old viewport tops on top of the copies + // the terminal already kept in native scrollback. `deferredShrink` pads to the + // previous row count so no committed row is re-emitted, and the next checkpoint + // rebuild (e.g. prompt submit -> `refreshNativeScrollbackIfDirty`) cleans up. + // + // That deferral only carries real content when `newLines.length > scrollbackHighWater` + // — otherwise the padded viewport rows fall entirely past the end of `newLines` + // and render as all blanks, hiding the prompt until the next checkpoint. For + // shrinks that large, yanking a scrolled reader (historyRebuild) is the lesser + // evil; do it unconditionally. + if (newLines.length <= this.#scrollbackHighWater) { + return { kind: "historyRebuild" }; + } this.#markNativeScrollbackDirty(); return { kind: "deferredShrink", paddedLength: this.#previousLines.length }; } diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index 9d00fd27e..c4769e8b8 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -1580,6 +1580,51 @@ describe("TUI terminal-state regressions", () => { tui.stop(); } }); + it("rebuilds history when a shrink leaves no real rows above the scrollback boundary", async () => { + // Reviewer scenario (#1599): a large completion-style collapse (e.g. a 100-row + // streamed transcript shrinking to a 20-row final cell in a 10-row viewport) + // must NOT use the padded `deferredShrink` — the viewport would fall entirely + // past the end of `newLines` and render as all blanks (no prompt visible) until + // the next checkpoint. Yank the scrollback instead so the new tail stays on + // screen. + const term = new UnknownViewportTerminal(40, 10); + const tui = new TUI(term); + const body = rows("line-", 99); + const component = new MutableLinesComponent([...body, "prompt-row"]); + tui.addChild(component); + + try { + tui.start(); + await settle(term); + + const short = rows("short-", 19); + component.setLines([...short, "prompt-row"]); + tui.requestRender(); + await settle(term); + + const viewport = visible(term).map(line => line.trim()); + expect(viewport).toEqual([ + "short-10", + "short-11", + "short-12", + "short-13", + "short-14", + "short-15", + "short-16", + "short-17", + "short-18", + "prompt-row", + ]); + const scrollback = term.getScrollBuffer(); + for (let i = 0; i < short.length; i++) { + const pattern = new RegExp(`\\bshort-${i}\\b`); + expect(countMatches(scrollback, pattern), `short-${i} appears once`).toBe(1); + } + expect(scrollback.join("\n")).not.toContain("line-"); + } finally { + tui.stop(); + } + }); it("renders streaming row inserts on WSL Windows Terminal even when viewport probe is unavailable", async () => { const originalPlatform = process.platform; From a882e9744bd2176d2b82eb7ee394df13af90e01d Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 17:07:41 +0000 Subject: [PATCH 05/28] fix(tui): rebuilt blank deferred shrink from padded top Codex review on #1599: the blank-viewport guard must compare the new transcript length against the viewport top used by the padded repaint, not #scrollbackHighWater. Prior unknown-POSIX viewport repaints can commit a long logical frame without advancing the high-water mark, so the stale high-water mark still lets all-blank deferred shrinks through.\n\nCompute paddedViewportTop from #previousLines.length - height and rebuild when the new tail cannot reach it. Add a regression for an offscreen POSIX viewport repaint that grows the committed frame to 120 rows while high-water remains at the original 20-row overflow, then shrinks to 15 rows. --- packages/tui/src/tui.ts | 17 +++--- packages/tui/test/render-regressions.test.ts | 55 ++++++++++++++++++++ 2 files changed, 66 insertions(+), 6 deletions(-) diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index e3887b664..477673c21 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1342,12 +1342,17 @@ export class TUI extends Container { // previous row count so no committed row is re-emitted, and the next checkpoint // rebuild (e.g. prompt submit -> `refreshNativeScrollbackIfDirty`) cleans up. // - // That deferral only carries real content when `newLines.length > scrollbackHighWater` - // — otherwise the padded viewport rows fall entirely past the end of `newLines` - // and render as all blanks, hiding the prompt until the next checkpoint. For - // shrinks that large, yanking a scrolled reader (historyRebuild) is the lesser - // evil; do it unconditionally. - if (newLines.length <= this.#scrollbackHighWater) { + // That deferral only carries real content when `newLines.length` reaches the + // padded viewport top (`previousLines.length - height`) — otherwise every + // row the padded repaint draws is past the end of `newLines` and renders as + // blank, hiding the prompt until the next checkpoint. This can happen even + // when `scrollbackHighWater` is much lower than `previousLines.length - height`, + // because prior unknown-POSIX viewport repaints commit longer logical frames + // without moving the native scrollback boundary. For shrinks that large, + // yanking a scrolled reader (historyRebuild) is the lesser evil; do it + // unconditionally. + const paddedViewportTop = Math.max(0, this.#previousLines.length - height); + if (newLines.length <= paddedViewportTop) { return { kind: "historyRebuild" }; } this.#markNativeScrollbackDirty(); diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index c4769e8b8..bbd2e635a 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -1625,6 +1625,61 @@ describe("TUI terminal-state regressions", () => { tui.stop(); } }); + it("rebuilds history when prior POSIX repaint left the padded viewport past the new tail", async () => { + const term = new UnknownViewportTerminal(40, 10); + const tui = new TUI(term); + const initial = rows("line-", 19); + const component = new MutableLinesComponent([...initial, "prompt-row"]); + tui.addChild(component); + + try { + tui.start(); + await settle(term); + + // Unknown-POSIX offscreen mutation: repainting the viewport commits the + // 120-row logical frame, but `#emitViewportRepaint` intentionally does not + // advance `#scrollbackHighWater` (it remains at the original 20-row frame's + // 10-row overflow). The later shrink must compare against the padded viewport + // top (`120 - height`) rather than the stale high-water mark. + const expanded = ["edited-line", ...rows("line-", 118), "prompt-row"]; + component.setLines(expanded); + tui.requestRender(); + await settle(term); + expect(visible(term).map(line => line.trim())).toEqual([ + "line-109", + "line-110", + "line-111", + "line-112", + "line-113", + "line-114", + "line-115", + "line-116", + "line-117", + "prompt-row", + ]); + + const short = [...rows("short-", 14), "prompt-row"]; + component.setLines(short); + tui.requestRender(); + await settle(term); + + expect(visible(term).map(line => line.trim())).toEqual([ + "short-5", + "short-6", + "short-7", + "short-8", + "short-9", + "short-10", + "short-11", + "short-12", + "short-13", + "prompt-row", + ]); + expect(term.getScrollBuffer().join("\n")).not.toContain("line-"); + } finally { + tui.stop(); + } + }); it("renders streaming row inserts on WSL Windows Terminal even when viewport probe is unavailable", async () => { const originalPlatform = process.platform; From ff3dd6d9a6e0317a125264d9499920de0a0a8551 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 17:40:56 +0000 Subject: [PATCH 06/28] fix(tool): rendered formal pr reviews Included the reviews field in comments-enabled PR view fetches so pr:// output can show formal review submissions and approvals. Added protocol coverage that emulates gh --json field selection before asserting rendered approval output. Fixes #1600 --- packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/tools/gh.ts | 1 + .../internal-urls/issue-pr-protocol.test.ts | 50 ++++++++++++++++++- 3 files changed, 54 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9e22dabb5..997cf04aa 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `pr://` PR views omitting formal review submissions and approvals when comments are enabled ([#1600](https://github.com/can1357/oh-my-pi/issues/1600)). + ## [15.7.4] - 2026-05-31 ### Removed diff --git a/packages/coding-agent/src/tools/gh.ts b/packages/coding-agent/src/tools/gh.ts index 8f4008899..3e1d84204 100644 --- a/packages/coding-agent/src/tools/gh.ts +++ b/packages/coding-agent/src/tools/gh.ts @@ -75,6 +75,7 @@ const GH_PR_FIELDS = [ "labels", "mergeStateStatus", "number", + "reviews", "reviewDecision", "state", "title", diff --git a/packages/coding-agent/test/internal-urls/issue-pr-protocol.test.ts b/packages/coding-agent/test/internal-urls/issue-pr-protocol.test.ts index 2d14ccdda..7ccccf649 100644 --- a/packages/coding-agent/test/internal-urls/issue-pr-protocol.test.ts +++ b/packages/coding-agent/test/internal-urls/issue-pr-protocol.test.ts @@ -66,7 +66,16 @@ function issuePayload(number: number, body: string, commentBodies: string[] = [] }; } +interface PrPayloadReview { + author: { login: string }; + body: string; + commit: { oid: string }; + state: string; + submittedAt: string; +} + function prPayload(number: number, body: string) { + const reviews: PrPayloadReview[] = []; return { number, title: `PR #${number}`, @@ -81,11 +90,34 @@ function prPayload(number: number, body: string) { url: `https://github.com/owner/example/pull/${number}`, labels: [], files: [], - reviews: [], + reviews, comments: [], }; } +function requestedJsonFields(args: string[]): Set { + const jsonIndex = args.indexOf("--json"); + const fieldsArg = jsonIndex >= 0 ? args[jsonIndex + 1] : undefined; + return new Set((fieldsArg ?? "").split(",").filter(Boolean)); +} + +function prPayloadWithRequestedFields(args: string[], number: number, body: string) { + const payload = prPayload(number, body); + const fields = requestedJsonFields(args); + if (fields.has("reviews")) { + payload.reviews = [ + { + author: { login: "approver" }, + body: "Approved from the formal review flow.", + commit: { oid: "1234567890abcdef1234567890abcdef12345678" }, + state: "APPROVED", + submittedAt: "2026-04-01T12:00:00Z", + }, + ]; + } + return payload; +} + interface DiffFileSpec { name: string; adds?: number; @@ -193,6 +225,22 @@ describe("pr:// protocol handler", () => { expect(spy).toHaveBeenCalledTimes(2); }); + it("requests and renders formal reviews when comments are enabled", async () => { + vi.spyOn(git.github, "json").mockImplementation(async (_cwd, args) => { + if (args.includes("/repos/owner/example/pulls/78/comments")) { + return [] as never; + } + return prPayloadWithRequestedFields(args, 78, "pr body") as never; + }); + + const router = InternalUrlRouter.instance(); + const resource = await router.resolve("pr://owner/example/78"); + + expect(resource.content).toContain("## Reviews (1)"); + expect(resource.content).toContain("### @approver - 2026-04-01T12:00:00Z [APPROVED]"); + expect(resource.content).toContain("Approved from the formal review flow."); + }); + it("rejects invalid pr:// URLs with a friendly message", async () => { const router = InternalUrlRouter.instance(); await expect(router.resolve("pr://owner/example/foo/bar")).rejects.toThrow(/Invalid pr:\/\/ URL/); From a45747f96a48d188e9562a4698cab36470b5cb51 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 31 May 2026 20:10:56 +0200 Subject: [PATCH 07/28] refactor(coding-agent): replaced bracket-style section tags with underlined headers - Converted [SECTION]...[/SECTION] markers to "SECTION\n===" format in system prompt templates. - Updated system conventions doc to reference the new marker style. - Updated tests to match against the new header pattern. --- .../src/prompts/system/project-prompt.md | 5 +++-- .../prompts/system/subagent-system-prompt.md | 20 +++++++++++-------- .../src/prompts/system/system-prompt.md | 14 +++++++------ packages/coding-agent/src/task/executor.ts | 2 +- .../test/system-prompt-templates.test.ts | 8 ++++---- .../task/executor-subagent-reminders.test.ts | 6 +++--- 6 files changed, 31 insertions(+), 24 deletions(-) diff --git a/packages/coding-agent/src/prompts/system/project-prompt.md b/packages/coding-agent/src/prompts/system/project-prompt.md index 9483aeb6c..6ec027800 100644 --- a/packages/coding-agent/src/prompts/system/project-prompt.md +++ b/packages/coding-agent/src/prompts/system/project-prompt.md @@ -1,4 +1,6 @@ -[PROJECT] +PROJECT +=================================== + {{#list environment prefix="- " join="\n"}}{{label}}: {{value}}{{/list}} @@ -47,4 +49,3 @@ Today is {{date}}, and the current working directory is '{{cwd}}'. {{#if appendPrompt}} {{appendPrompt}} {{/if}} -[/PROJECT] diff --git a/packages/coding-agent/src/prompts/system/subagent-system-prompt.md b/packages/coding-agent/src/prompts/system/subagent-system-prompt.md index e4e4a063f..934c37095 100644 --- a/packages/coding-agent/src/prompts/system/subagent-system-prompt.md +++ b/packages/coding-agent/src/prompts/system/subagent-system-prompt.md @@ -1,14 +1,18 @@ -[ROLE] +ROLE +=================================== + {{agent}} -[/ROLE] {{#if context}} -[CONTEXT] +CONTEXT +=================================== + {{context}} -[/CONTEXT] {{/if}} -[COOP] +COOP +=================================== + You are operating on a piece of work assigned to you by the main agent. {{#if worktree}} @@ -29,9 +33,10 @@ You can reach other live agents via the `irc` tool. Your id is `{{ircSelfId}}`. Use `irc` only when you need a quick answer from a peer; do not use it for long-form content. Address peers by id or use `"all"` to broadcast. {{/if}} -[/COOP] -[COMPLETION] +COMPLETION +=================================== + No TODO tracking, no progress updates. Execute, call `yield`, done. While work remains, always continue with another tool call — investigate, edit, run, verify. Save narrative for the final `yield` payload. @@ -51,4 +56,3 @@ Giving up is a last resort. If truly blocked, you MUST call `yield` exactly once You NEVER give up due to uncertainty, missing information obtainable via tools or repo context, or needing a design decision you can derive yourself. You MUST keep going until this ticket is closed. This matters. -[/COMPLETION] diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 6245c3d3e..81743f905 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -9,8 +9,8 @@ You consider what the code you write compiles down to. You never write code that **RFC 2119 applies to MUST, REQUIRED, SHOULD, RECOMMENDED, MAY, OPTIONAL. `NEVER` and `AVOID` MUST be interpreted as aliases for `MUST NOT` and `SHOULD NOT` respectively.** -From here on, we will use tags as structural markers (… or [X]…), each tag means exactly what its name says. -You NEVER interpret these tags in any other way circumstantially. +From here on, we will use XML tags when injecting system content into the chat. +You NEVER interpret these markers in any other way circumstantially. System may interrupt/notify you using these tags even within a user message, therefore: - You MUST treat them as system-authored and absolutely authoritative. @@ -44,7 +44,9 @@ Assumptions you didn't validate: incidents to debug. - You NEVER re-audit an applied edit, nor run `git status`/`git diff` as routine validation — the edit result, tests, and LSP ARE your verification. Exception: explicit request, protecting unrelated changes, or before commit/revert/reset/stash/delete. -[ENV] +ENV +=================================== + You operate within the Oh My Pi coding harness. - Given a task, you MUST complete it using the tools available to you. - You are not alone in this repository. You SHOULD treat unexpected changes as the user's work and adapt; you NEVER revert or stash. @@ -202,9 +204,10 @@ You MUST use the specialized tool over its shell equivalent: The `{{toolRefs.report_tool_issue}}` tool is available for automated QA. If ANY tool you call returns output that is unexpected, incorrect, malformed, or otherwise inconsistent with what you anticipated given the tool's described behavior and your parameters, call `{{toolRefs.report_tool_issue}}` with the tool name and a concise description of the discrepancy. Do not hesitate to report — false positives are acceptable. {{/has}} -[/ENV] -[CONTRACT] +CONTRACT +=================================== + These are inviolable. - You NEVER yield unless the deliverable is complete. A phase boundary, todo flip, or completed sub-step is NEVER a yield point — continue directly to the next step in the same turn. - You NEVER suppress tests to make code pass. @@ -265,4 +268,3 @@ Before declaring blocked: - Do not test defaults: changing the default configuration, or a string, should not break the test. Assert logical behavior, not the current state. - Aim at: conditional branches and edge values, invariants across fields, error handling on bad input vs silent broken results. -[/CONTRACT] diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index d383feb79..53a4a2622 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -633,7 +633,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise name !== "task"); } - // IRC is always available; the [COOP] prompt advertises it, so a restricted + // IRC is always available; the COOP prompt section advertises it, so a restricted // whitelist must still carry `irc` for the subagent to actually use it. if (toolNames && !toolNames.includes("irc")) { toolNames = [...toolNames, "irc"]; diff --git a/packages/coding-agent/test/system-prompt-templates.test.ts b/packages/coding-agent/test/system-prompt-templates.test.ts index 9734c0326..0330bde2a 100644 --- a/packages/coding-agent/test/system-prompt-templates.test.ts +++ b/packages/coding-agent/test/system-prompt-templates.test.ts @@ -181,11 +181,11 @@ describe("system Handlebars prompt templates", () => { assignment: "Do the task.", }); - expect(subagentSystem).toContain("[CONTEXT]\nShared task background\n[/CONTEXT]"); - expect(subagentSystem).toContain("[ROLE]"); + expect(subagentSystem).toMatch(/CONTEXT\n=+\n\nShared task background/); + expect(subagentSystem).toMatch(/ROLE\n=+/); expect(subagentUser).toContain("Complete the assignment below, thoroughly:"); expect(subagentUser).toContain("Do the task."); - expect(subagentUser).not.toContain("[CONTEXT]"); + expect(subagentUser).not.toMatch(/CONTEXT\n=+/); expect(subagentUser).not.toContain("Shared task background"); }); test("system-prompt renders MCP discovery hint when enabled", async () => { @@ -247,7 +247,7 @@ describe("system Handlebars prompt templates", () => { }); expect(systemPrompt).toHaveLength(2); - expect(systemPrompt[0]).toContain("[CONTRACT]"); + expect(systemPrompt[0]).toMatch(/CONTRACT\n=+/); expect(systemPrompt[0]).not.toContain("current working directory"); expect(systemPrompt[1]).toContain(""); expect(systemPrompt[1]).toContain(""); diff --git a/packages/coding-agent/test/task/executor-subagent-reminders.test.ts b/packages/coding-agent/test/task/executor-subagent-reminders.test.ts index 2a0a51409..1fffe79c0 100644 --- a/packages/coding-agent/test/task/executor-subagent-reminders.test.ts +++ b/packages/coding-agent/test/task/executor-subagent-reminders.test.ts @@ -226,10 +226,10 @@ describe("runSubprocess yield reminders", () => { expect(systemPrompt).toHaveLength(4); expect(systemPrompt?.[0]).toBe("system"); expect(systemPrompt?.[1]).toBe("project"); - expect(systemPrompt?.[2]).toContain("[CONTEXT]\nShared task background\n[/CONTEXT]"); - expect(systemPrompt?.[2]).toContain("[ROLE]\ntest\n[/ROLE]"); + expect(systemPrompt?.[2]).toMatch(/CONTEXT\n=+\n\nShared task background/); + expect(systemPrompt?.[2]).toMatch(/ROLE\n=+\n\ntest/); expect(systemPrompt?.[3]).toBe("now"); - expect(userPrompt).not.toContain("[CONTEXT]"); + expect(userPrompt).not.toMatch(/CONTEXT\n=+/); expect(userPrompt).not.toContain("Shared task background"); }); From 7e11bea8f4e3a3a9a35d9f9354aa62ec2dbcdcd9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 31 May 2026 20:17:38 +0200 Subject: [PATCH 08/28] fix(coding-agent): repaired per-field double-encoded JSON in task tool - Added `repairDoubleEncodedJsonString` to unescape fields double-encoded by the model (e.g. literal `\n`, `\"`, `\uXXXX` in `context`/`assignment`/`description`). - Scoped repair to natural-language fields only, leaving code-bearing tools untouched. - Applied repair on both render and execution paths in `TaskTool`. --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/task/index.ts | 5 +- packages/coding-agent/src/task/repair-args.ts | 117 ++++++++++++++++++ .../test/tools/task-repair-args.test.ts | 80 ++++++++++++ 4 files changed, 204 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/src/task/repair-args.ts create mode 100644 packages/coding-agent/test/tools/task-repair-args.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9e22dabb5..9495ccf72 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the `task` tool mangling subagent prompts when a model double-JSON-encodes a string argument: `context` and each task's `assignment`/`description` are now repaired when they arrive uniformly double-escaped (literal `\n`, `\"`, `\uXXXX`), so the subagent receives the intended prose and the call preview renders real newlines. The repair is guarded by a JSON-string round-trip and a double-encode signature, so legitimate backslashes/quotes (Windows paths, regexes, embedded quotes) are left untouched, and it is scoped to these natural-language fields only (never code-bearing tools). + ## [15.7.4] - 2026-05-31 ### Removed diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index d831525bf..eacdf01b7 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -48,6 +48,7 @@ import { runSubprocess } from "./executor"; import { AgentOutputManager } from "./output-manager"; import { mapWithConcurrencyLimit, Semaphore } from "./parallel"; import { renderResult, renderCall as renderTaskCall } from "./render"; +import { repairTaskParams } from "./repair-args"; import { getTaskSimpleModeCapabilities, type TaskSimpleMode } from "./simple-mode"; import { applyNestedPatches, @@ -247,7 +248,7 @@ export class TaskTool implements AgentTool[1], theme: Theme) { - return renderTaskCall(args as TaskParams, options, theme); + return renderTaskCall(repairTaskParams(args as TaskParams), options, theme); } /** Dynamic description that reflects current disabled-agent settings */ @@ -292,7 +293,7 @@ export class TaskTool implements AgentTool, ): Promise> { - const params = rawParams as TaskParams; + const params = repairTaskParams(rawParams as TaskParams); const simpleMode = this.#getTaskSimpleMode(); const validationError = validateTaskModeParams(simpleMode, params); if (validationError) { diff --git a/packages/coding-agent/src/task/repair-args.ts b/packages/coding-agent/src/task/repair-args.ts new file mode 100644 index 000000000..62cc8f7a7 --- /dev/null +++ b/packages/coding-agent/src/task/repair-args.ts @@ -0,0 +1,117 @@ +/** + * Repair double-encoded JSON string arguments for the task tool. + * + * Models occasionally JSON-escape a string value twice when emitting a + * `task` tool call, so a `context`/`assignment` that should read + * + * # Role + * You are a judge … "describe this" … return — + * + * arrives — after the one JSON decode the provider already applied — as the + * literal text + * + * # Role\nYou are a judge … \"describe this\" … return \u2014 + * + * i.e. every newline, quote, and unicode character is still backslash-escaped. + * The subagent then receives that garbled prompt, and the call preview renders + * one long blob with visible `\n` / `\"` / `\uXXXX`. + * + * The *whole-arguments* form of this quirk (the entire `arguments` blob is a + * JSON string) is already auto-corrected by the validator's JSON-string + * coercion. This module handles the *per-field* form, where the object parses + * fine but an individual string value is double-encoded — the validator never + * fires there because a double-encoded string is still a structurally valid + * string. + * + * This is deliberately scoped to the task tool's natural-language fields + * (`context`, `assignment`, `description`). It is NOT applied to code-bearing + * tools (write/edit/bash/search), where a backslash or quote is load-bearing + * and a false-positive unescape would silently corrupt a file or command. + */ +import type { TaskItem, TaskParams } from "./types"; + +/** A backslash that escapes a structural char — `\"`, `\\`, `\/`, or `\uXXXX`. */ +const STRUCTURAL_ESCAPE = /\\(?:["\\/]|u[0-9a-fA-F]{4})/; + +/** + * Whether `value` carries the signature of whole-string double-encoding rather + * than an incidental escape mention. A lone `\n`/`\t` in an instruction (e.g. + * "split lines on \n") is far more likely a literal mention than a + * double-encoded document, so it is left alone; a structural escape (`\"`, + * `\\`, `\uXXXX`) or two-plus escape sequences indicates a re-escaped payload. + */ +function hasDoubleEncodeSignature(value: string): boolean { + if (STRUCTURAL_ESCAPE.test(value)) return true; + let count = 0; + for (let i = 0; i < value.length; i++) { + if (value.charCodeAt(i) === 0x5c /* \ */) { + count += 1; + if (count >= 2) return true; + i += 1; // skip the escaped char so `\\` counts once + } + } + return false; +} + +/** + * Return the once-unescaped string when `value` is uniformly double-encoded + * JSON (a well-formed JSON string body that decodes to a different string); + * otherwise return `value` unchanged. + * + * The `JSON.parse(\`"${value}"\`)` round-trip is the safety net: it only + * succeeds when *every* backslash begins a valid JSON escape and no bare + * double-quote exists — exactly the signature of double-encoding. Genuine + * prose with a Windows path (`C:\Users`), a regex (`\d+`), an embedded quote, + * or a real (already-decoded) newline makes the parse throw, so the value is + * returned untouched. + */ +export function repairDoubleEncodedJsonString(value: string): string { + // Fast path: no backslash → nothing was escaped → the parse can never differ. + if (!value.includes("\\")) return value; + if (!hasDoubleEncodeSignature(value)) return value; + let decoded: unknown; + try { + decoded = JSON.parse(`"${value}"`); + } catch { + return value; + } + return typeof decoded === "string" && decoded !== value ? decoded : value; +} + +/** Repair a single (possibly partial) task item's prose fields. */ +function repairTaskItem(task: TaskItem): TaskItem { + if (task === null || typeof task !== "object") return task; + const assignment = + typeof task.assignment === "string" ? repairDoubleEncodedJsonString(task.assignment) : task.assignment; + const description = + typeof task.description === "string" ? repairDoubleEncodedJsonString(task.description) : task.description; + if (assignment === task.assignment && description === task.description) return task; + return { ...task, assignment, description }; +} + +/** + * Repair double-encoded prose in task-tool params (`context` and each task's + * `assignment`/`description`). Returns the same reference when nothing changed + * so callers can cheaply skip work. Defensive against partially-streamed args + * (missing/undefined fields, partial task arrays) so it is safe on the render + * path as well as on execution. + */ +export function repairTaskParams(params: TaskParams): TaskParams { + if (params === null || typeof params !== "object") return params; + + const context = typeof params.context === "string" ? repairDoubleEncodedJsonString(params.context) : params.context; + + let tasks = params.tasks; + if (Array.isArray(params.tasks)) { + let changed = false; + const repaired = params.tasks.map(task => { + const next = repairTaskItem(task); + if (next !== task) changed = true; + return next; + }); + if (changed) tasks = repaired; + } + + if (context === params.context && tasks === params.tasks) return params; + return { ...params, context, tasks }; +} diff --git a/packages/coding-agent/test/tools/task-repair-args.test.ts b/packages/coding-agent/test/tools/task-repair-args.test.ts new file mode 100644 index 000000000..7979c367c --- /dev/null +++ b/packages/coding-agent/test/tools/task-repair-args.test.ts @@ -0,0 +1,80 @@ +import { describe, expect, it } from "bun:test"; +import { repairDoubleEncodedJsonString, repairTaskParams } from "../../src/task/repair-args"; +import type { TaskParams } from "../../src/task/types"; + +describe("repairDoubleEncodedJsonString", () => { + it("decodes a uniformly double-encoded prose value", () => { + // One JSON decode already applied by the provider; the value still + // carries literal `\n`, `\"`, and `\u2014` because the model escaped twice. + const doubled = '# Role\\nYou are a judge \\"describe this\\" return \\u2014'; + expect(repairDoubleEncodedJsonString(doubled)).toBe('# Role\nYou are a judge "describe this" return —'); + }); + + it("decodes a double-encoded multi-line plain-text value", () => { + expect(repairDoubleEncodedJsonString("line one\\nline two\\nline three")).toBe("line one\nline two\nline three"); + }); + + it("preserves a Windows path (bare backslashes are not valid escapes)", () => { + expect(repairDoubleEncodedJsonString("C:\\Users\\me")).toBe("C:\\Users\\me"); + }); + + it("preserves a regex with a backslash class", () => { + expect(repairDoubleEncodedJsonString("match \\d+ digits")).toBe("match \\d+ digits"); + }); + + it("preserves text containing a bare double quote", () => { + expect(repairDoubleEncodedJsonString('she said "hi" loudly')).toBe('she said "hi" loudly'); + }); + + it("leaves a lone literal \\n mention alone (no double-encode signature)", () => { + expect(repairDoubleEncodedJsonString("split lines on \\n then count")).toBe("split lines on \\n then count"); + }); + + it("is a no-op for plain text without escapes", () => { + const plain = "just some normal instructions"; + expect(repairDoubleEncodedJsonString(plain)).toBe(plain); + }); + + it("leaves a partially-decoded value (real newline mixed with literal escape) untouched", () => { + // A real newline cannot appear inside a JSON string literal unescaped, so + // the round-trip parse throws and the value is preserved as-is. + const mixed = "real\nnewline with \\t tab"; + expect(repairDoubleEncodedJsonString(mixed)).toBe(mixed); + }); +}); + +describe("repairTaskParams", () => { + it("repairs context and each task's assignment/description, leaving ids intact", () => { + const params = { + agent: "task", + context: "# Goal\\nDo the thing \\u2014 carefully", + tasks: [ + { + id: "FirstTask", + description: 'judge \\"sketch\\" accuracy', + assignment: "Score 0-100.\\nUse the full range.\\nNo bunching.", + }, + ], + } as unknown as TaskParams; + + const repaired = repairTaskParams(params); + expect(repaired.context).toBe("# Goal\nDo the thing — carefully"); + expect(repaired.tasks[0].id).toBe("FirstTask"); + expect(repaired.tasks[0].description).toBe('judge "sketch" accuracy'); + expect(repaired.tasks[0].assignment).toBe("Score 0-100.\nUse the full range.\nNo bunching."); + }); + + it("returns the same reference when nothing needs repair", () => { + const params = { + agent: "task", + context: "plain context", + tasks: [{ id: "A", description: "label", assignment: "do work" }], + } as unknown as TaskParams; + expect(repairTaskParams(params)).toBe(params); + }); + + it("tolerates partially-streamed args without throwing", () => { + const partial = { agent: "task", tasks: [{ id: "A" }, undefined] } as unknown as TaskParams; + expect(() => repairTaskParams(partial)).not.toThrow(); + }); +}); From 239bb9858d02b3a81ed2437605605e91f4350674 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 18:53:58 +0000 Subject: [PATCH 09/28] fix(ai): honored openai idle timeout for first events OpenAI-compatible local servers can spend longer than the generic first-event budget processing large prompts before they emit response headers or SSE frames. The OpenAI-specific idle timeout now also acts as the OpenAI-family first-event floor unless an explicit OpenAI first-event timeout is configured. Added regression coverage for OpenAI Responses request setup so a lower generic first-event watchdog no longer undercuts PI_OPENAI_STREAM_IDLE_TIMEOUT_MS. Fixes #1603 --- docs/environment-variables.md | 19 +++++------ packages/ai/CHANGELOG.md | 4 +++ .../src/providers/azure-openai-responses.ts | 7 +++- .../src/providers/openai-codex-responses.ts | 7 +++- .../ai/src/providers/openai-completions.ts | 12 ++++--- packages/ai/src/providers/openai-responses.ts | 7 +++- packages/ai/src/types.ts | 3 ++ packages/ai/src/utils/idle-iterator.ts | 21 ++++++++++++ .../test/openai-first-event-timeout.test.ts | 32 +++++++++++++++++++ .../ai/test/stream-timeout-defaults.test.ts | 27 ++++++++++++++++ 10 files changed, 123 insertions(+), 16 deletions(-) diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 19f54f0ea..b8c4e527b 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -190,15 +190,16 @@ OAuth host chain: `KIMI_CODE_OAUTH_HOST` → `KIMI_OAUTH_HOST` → `https://auth ### OpenAI Codex responses (feature/debug controls) -| Variable | Behavior | -| ------------------------------------ | ---------------------------------------------------- | -| `PI_CODEX_DEBUG` | `1`/`true` enables Codex provider debug logging | -| `PI_CODEX_WEBSOCKET` | `1`/`true` enables websocket transport preference | -| `PI_CODEX_WEBSOCKET_V2` | `1`/`true` enables websocket v2 path | -| `PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS` | Positive integer override (default 300000) | -| `PI_CODEX_WEBSOCKET_RETRY_BUDGET` | Non-negative integer override (default 5) | -| `PI_CODEX_WEBSOCKET_RETRY_DELAY_MS` | Positive integer base backoff override (default 500) | -| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override | +| Variable | Behavior | +| ------------------------------------------ | ---------------------------------------------------- | +| `PI_CODEX_DEBUG` | `1`/`true` enables Codex provider debug logging | +| `PI_CODEX_WEBSOCKET` | `1`/`true` enables websocket transport preference | +| `PI_CODEX_WEBSOCKET_V2` | `1`/`true` enables websocket v2 path | +| `PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS` | Positive integer override (default 300000) | +| `PI_CODEX_WEBSOCKET_RETRY_BUDGET` | Non-negative integer override (default 5) | +| `PI_CODEX_WEBSOCKET_RETRY_DELAY_MS` | Positive integer base backoff override (default 500) | +| `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` | Positive integer OpenAI first-event timeout override | +| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override | ### Cursor provider debug diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f80bf3d04..e9ce3361c 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI-family first-event timeouts so `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` cannot be undercut by a lower generic `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` while local OpenAI-compatible servers are still processing large prompts. `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` is now available for an explicit OpenAI-specific first-event override. ([#1603](https://github.com/can1357/oh-my-pi/issues/1603)) + ## [15.7.4] - 2026-05-31 ### Fixed diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index cb1b4e659..5011a85c8 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -22,6 +22,7 @@ import { createAbortSourceTracker } from "../utils/abort"; import { AssistantMessageEventStream } from "../utils/event-stream"; import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector"; import { + getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, @@ -122,7 +123,11 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses" const params = buildParams(model, context, options, deploymentName, baseUrl); options?.onPayload?.(params); const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); - const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs); + const firstEventTimeoutMs = + options?.streamFirstEventTimeoutMs ?? + (options?.streamIdleTimeoutMs === undefined + ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs) + : getStreamFirstEventTimeoutMs(idleTimeoutMs)); const requestTimeoutMs = firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined; rawRequestDump = { diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index f9769b8ad..f68cd87c9 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -49,6 +49,7 @@ import { import { AssistantMessageEventStream } from "../utils/event-stream"; import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector"; import { + getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, @@ -603,7 +604,11 @@ function createRequestSetup(options: OpenAICodexResponsesOptions | undefined): C : requestAbortController.signal; const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); const websocketIdleTimeoutMs = options?.streamIdleTimeoutMs ?? getCodexWebSocketIdleTimeoutMs(); - const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs); + const firstEventTimeoutMs = + options?.streamFirstEventTimeoutMs ?? + (options?.streamIdleTimeoutMs === undefined + ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs) + : getStreamFirstEventTimeoutMs(idleTimeoutMs)); const websocketFirstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getCodexWebSocketFirstEventTimeoutMs(); const wrapCodexSseStream = ( source: AsyncGenerator>, diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index d80c3c820..fa7bdaf10 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -46,6 +46,7 @@ import { rewriteCopilotError, } from "../utils/http-inspector"; import { + getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, @@ -421,10 +422,13 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( try { const apiKey = options?.apiKey || getEnvApiKey(model.provider) || ""; - const idleTimeoutMs = - options?.streamIdleTimeoutMs ?? - getOpenAIStreamIdleTimeoutMs(getOpenAICompletionsStreamIdleTimeoutFallbackMs(model)); - const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs); + const idleTimeoutFallbackMs = getOpenAICompletionsStreamIdleTimeoutFallbackMs(model); + const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs); + const firstEventTimeoutMs = + options?.streamFirstEventTimeoutMs ?? + (options?.streamIdleTimeoutMs === undefined + ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, idleTimeoutFallbackMs) + : getStreamFirstEventTimeoutMs(idleTimeoutMs)); const requestTimeoutMs = firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined; const { diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 083b5d679..f7f21bd76 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -33,6 +33,7 @@ import { createAbortSourceTracker } from "../utils/abort"; import { AssistantMessageEventStream } from "../utils/event-stream"; import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector"; import { + getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, @@ -228,7 +229,11 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = ( const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState); const { params } = buildParams(model, context, options, providerSessionState, baseUrl); const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); - const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(idleTimeoutMs); + const firstEventTimeoutMs = + options?.streamFirstEventTimeoutMs ?? + (options?.streamIdleTimeoutMs === undefined + ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs) + : getStreamFirstEventTimeoutMs(idleTimeoutMs)); const requestTimeoutMs = firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined; options?.onPayload?.(params); diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 83b754ceb..bad104ab1 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -363,6 +363,9 @@ export interface StreamOptions { * `0` to disable both layers for this request. After the first semantic * event arrives, `streamIdleTimeoutMs` governs inter-event stalls. Falls * back to `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` and then to a 100s default. + * OpenAI-family transports additionally honor + * `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` and use + * `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` as the first-event floor. * * Iterator-level honored by: every built-in provider (via the lazy-stream * forwarder in `register-builtins`). SDK-request honored by: diff --git a/packages/ai/src/utils/idle-iterator.ts b/packages/ai/src/utils/idle-iterator.ts index 0e20ba2bc..58677c8f2 100644 --- a/packages/ai/src/utils/idle-iterator.ts +++ b/packages/ai/src/utils/idle-iterator.ts @@ -58,6 +58,27 @@ export function getStreamFirstEventTimeoutMs( return normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallback); } +/** + * Returns the first-event timeout used for OpenAI-family streaming transports. + * + * `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` is the most specific first-event + * override. When it is unset, `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` also widens + * or disables the first-event watchdog so local OpenAI-compatible servers are + * not undercut by the generic first-event setting during slow prompt processing. + */ +export function getOpenAIStreamFirstEventTimeoutMs( + idleTimeoutMs?: number, + fallbackMs: number = DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS, +): number | undefined { + const fallback = idleTimeoutMs === undefined ? fallbackMs : Math.max(fallbackMs, idleTimeoutMs); + return normalizeIdleTimeoutMs( + $env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS ?? + $env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS ?? + $env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, + fallback, + ); +} + export interface IdleTimeoutIteratorOptions { idleTimeoutMs?: number; firstItemTimeoutMs?: number; diff --git a/packages/ai/test/openai-first-event-timeout.test.ts b/packages/ai/test/openai-first-event-timeout.test.ts index e74afff67..1731079bc 100644 --- a/packages/ai/test/openai-first-event-timeout.test.ts +++ b/packages/ai/test/openai-first-event-timeout.test.ts @@ -320,6 +320,38 @@ describe("OpenAI-family first-event timeouts", () => { ); }); + it("lets PI_OPENAI_STREAM_IDLE_TIMEOUT_MS widen OpenAI responses first-event request setup", async () => { + const previousOpenAIIdleTimeout = Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS; + const previousGenericFirstEventTimeout = Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS; + const timeoutHeaders: string[] = []; + Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "1500"; + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20"; + global.fetch = createDelayedFetch(30, createOpenAIResponsesSuccessResponse, (input, init) => { + timeoutHeaders.push(getRequestHeader(input, init, "X-Stainless-Timeout") ?? ""); + }); + + try { + const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), { + apiKey: "test-key", + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(getFirstTextContent(result)).toMatchObject({ type: "text", text: "Hello delayed" }); + expect(timeoutHeaders).toContain("1"); + } finally { + if (previousOpenAIIdleTimeout === undefined) { + delete Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS; + } else { + Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = previousOpenAIIdleTimeout; + } + if (previousGenericFirstEventTimeout === undefined) { + delete Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS; + } else { + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = previousGenericFirstEventTimeout; + } + } + }); + it("times out OpenAI responses streams that only emit no-progress status events", async () => { global.fetch = ((input: string | URL | Request, init?: RequestInit) => Promise.resolve(createNoProgressOpenAIResponsesStream(getRequestSignal(input, init)))) as typeof fetch; diff --git a/packages/ai/test/stream-timeout-defaults.test.ts b/packages/ai/test/stream-timeout-defaults.test.ts index c01beef75..65a3044ab 100644 --- a/packages/ai/test/stream-timeout-defaults.test.ts +++ b/packages/ai/test/stream-timeout-defaults.test.ts @@ -1,5 +1,6 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import { + getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, @@ -19,6 +20,7 @@ const ENV_KEYS = [ "PI_STREAM_IDLE_TIMEOUT_MS", "PI_OPENAI_STREAM_IDLE_TIMEOUT_MS", "PI_STREAM_FIRST_EVENT_TIMEOUT_MS", + "PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS", ] as const; const originalEnv: Partial> = {}; @@ -102,6 +104,31 @@ describe("getStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => { }); }); +describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => { + it("lets the OpenAI idle env widen a lower generic first-event timeout", () => { + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20"; + Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84"; + expect(getOpenAIStreamFirstEventTimeoutMs(84, 300_000)).toBe(84); + }); + + it("lets the OpenAI first-event env override the OpenAI idle env", () => { + Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42"; + Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84"; + expect(getOpenAIStreamFirstEventTimeoutMs(84, 300_000)).toBe(42); + }); + + it("falls back to the generic first-event env when OpenAI env vars are unset", () => { + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42"; + expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(42); + }); + + it("treats PI_OPENAI_STREAM_IDLE_TIMEOUT_MS=0 as an OpenAI watchdog disable", () => { + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42"; + Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "0"; + expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBeUndefined(); + }); +}); + async function expectRejectsWithMessage(run: () => Promise, message: string): Promise { let caught: unknown; try { From 062ba5e8364ffd3af7606d29d8fb032bee9a76d2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 19:02:53 +0000 Subject: [PATCH 10/28] fix(ai): kept openai first-event env honored with per-call idle Always route OpenAI-family providers through getOpenAIStreamFirstEventTimeoutMs so PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS wins even when callers pass per-call streamIdleTimeoutMs. The OpenAI helper now floors the first-event budget at the caller-resolved idle (which already encompasses per-call streamIdleTimeoutMs or PI_OPENAI_STREAM_IDLE_TIMEOUT_MS upstream), and explicit env disables ("0") on either knob continue to drop the watchdog. Fixes #1603 --- .../src/providers/azure-openai-responses.ts | 6 +--- .../src/providers/openai-codex-responses.ts | 7 +--- .../ai/src/providers/openai-completions.ts | 6 +--- packages/ai/src/providers/openai-responses.ts | 6 +--- packages/ai/src/types.ts | 6 ++-- packages/ai/src/utils/idle-iterator.ts | 28 +++++++++------- .../test/openai-first-event-timeout.test.ts | 33 +++++++++++++++++++ .../ai/test/stream-timeout-defaults.test.ts | 23 +++++++------ 8 files changed, 71 insertions(+), 44 deletions(-) diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 5011a85c8..f9d3a2bed 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -24,7 +24,6 @@ import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-ins import { getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, - getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, } from "../utils/idle-iterator"; import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema"; @@ -124,10 +123,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses" options?.onPayload?.(params); const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); const firstEventTimeoutMs = - options?.streamFirstEventTimeoutMs ?? - (options?.streamIdleTimeoutMs === undefined - ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs) - : getStreamFirstEventTimeoutMs(idleTimeoutMs)); + options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs); const requestTimeoutMs = firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined; rawRequestDump = { diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index f68cd87c9..f8eaedffe 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -51,7 +51,6 @@ import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-ins import { getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, - getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, } from "../utils/idle-iterator"; import { parseStreamingJson, parseStreamingJsonThrottled } from "../utils/json-parse"; @@ -604,11 +603,7 @@ function createRequestSetup(options: OpenAICodexResponsesOptions | undefined): C : requestAbortController.signal; const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); const websocketIdleTimeoutMs = options?.streamIdleTimeoutMs ?? getCodexWebSocketIdleTimeoutMs(); - const firstEventTimeoutMs = - options?.streamFirstEventTimeoutMs ?? - (options?.streamIdleTimeoutMs === undefined - ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs) - : getStreamFirstEventTimeoutMs(idleTimeoutMs)); + const firstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs); const websocketFirstEventTimeoutMs = options?.streamFirstEventTimeoutMs ?? getCodexWebSocketFirstEventTimeoutMs(); const wrapCodexSseStream = ( source: AsyncGenerator>, diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index fa7bdaf10..d93c20e32 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -48,7 +48,6 @@ import { import { getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, - getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, } from "../utils/idle-iterator"; import { parseStreamingJson, parseStreamingJsonThrottled } from "../utils/json-parse"; @@ -425,10 +424,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = ( const idleTimeoutFallbackMs = getOpenAICompletionsStreamIdleTimeoutFallbackMs(model); const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(idleTimeoutFallbackMs); const firstEventTimeoutMs = - options?.streamFirstEventTimeoutMs ?? - (options?.streamIdleTimeoutMs === undefined - ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, idleTimeoutFallbackMs) - : getStreamFirstEventTimeoutMs(idleTimeoutMs)); + options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs); const requestTimeoutMs = firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined; const { diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index f7f21bd76..f9647128d 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -35,7 +35,6 @@ import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } fr import { getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs, - getStreamFirstEventTimeoutMs, iterateWithIdleTimeout, } from "../utils/idle-iterator"; import { parseGitHubCopilotApiKey } from "../utils/oauth/github-copilot"; @@ -230,10 +229,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = ( const { params } = buildParams(model, context, options, providerSessionState, baseUrl); const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(); const firstEventTimeoutMs = - options?.streamFirstEventTimeoutMs ?? - (options?.streamIdleTimeoutMs === undefined - ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs) - : getStreamFirstEventTimeoutMs(idleTimeoutMs)); + options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs); const requestTimeoutMs = firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0 ? firstEventTimeoutMs : undefined; options?.onPayload?.(params); diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index bad104ab1..3efad7ef9 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -364,8 +364,10 @@ export interface StreamOptions { * event arrives, `streamIdleTimeoutMs` governs inter-event stalls. Falls * back to `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` and then to a 100s default. * OpenAI-family transports additionally honor - * `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` and use - * `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` as the first-event floor. + * `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` as the most-specific override and + * floor the first-event budget at the resolved idle (per-call + * `streamIdleTimeoutMs` or `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS`) so slow local + * OpenAI-compatible servers are not undercut during prompt processing. * * Iterator-level honored by: every built-in provider (via the lazy-stream * forwarder in `register-builtins`). SDK-request honored by: diff --git a/packages/ai/src/utils/idle-iterator.ts b/packages/ai/src/utils/idle-iterator.ts index 58677c8f2..d19b5cd57 100644 --- a/packages/ai/src/utils/idle-iterator.ts +++ b/packages/ai/src/utils/idle-iterator.ts @@ -61,22 +61,28 @@ export function getStreamFirstEventTimeoutMs( /** * Returns the first-event timeout used for OpenAI-family streaming transports. * - * `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` is the most specific first-event - * override. When it is unset, `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` also widens - * or disables the first-event watchdog so local OpenAI-compatible servers are - * not undercut by the generic first-event setting during slow prompt processing. + * Precedence: explicit `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` (including a + * `"0"` disable) wins outright. Otherwise the resolved idle (caller-supplied + * `idleTimeoutMs` — which itself already encompasses per-call + * `streamIdleTimeoutMs` or `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` resolved + * upstream) floors the first-event budget so slow local OpenAI-compatible + * servers are not undercut by a shorter `PI_STREAM_FIRST_EVENT_TIMEOUT_MS` + * or the global default during prompt processing. + * + * Returns `undefined` when an explicit env knob disables the watchdog. */ export function getOpenAIStreamFirstEventTimeoutMs( idleTimeoutMs?: number, fallbackMs: number = DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS, ): number | undefined { - const fallback = idleTimeoutMs === undefined ? fallbackMs : Math.max(fallbackMs, idleTimeoutMs); - return normalizeIdleTimeoutMs( - $env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS ?? - $env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS ?? - $env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, - fallback, - ); + const openAIFirstEventRaw = $env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS; + if (openAIFirstEventRaw !== undefined) { + return normalizeIdleTimeoutMs(openAIFirstEventRaw, fallbackMs); + } + const base = normalizeIdleTimeoutMs($env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS, fallbackMs); + if (base === undefined) return undefined; + if (idleTimeoutMs === undefined || idleTimeoutMs <= 0) return base; + return Math.max(base, idleTimeoutMs); } export interface IdleTimeoutIteratorOptions { diff --git a/packages/ai/test/openai-first-event-timeout.test.ts b/packages/ai/test/openai-first-event-timeout.test.ts index 1731079bc..eb9f54e44 100644 --- a/packages/ai/test/openai-first-event-timeout.test.ts +++ b/packages/ai/test/openai-first-event-timeout.test.ts @@ -352,6 +352,39 @@ describe("OpenAI-family first-event timeouts", () => { } }); + it("honors PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS even when caller pins streamIdleTimeoutMs", async () => { + const previousOpenAIFirstEventTimeout = Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS; + const previousGenericFirstEventTimeout = Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS; + const timeoutHeaders: string[] = []; + Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "1500"; + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20"; + global.fetch = createDelayedFetch(30, createOpenAIResponsesSuccessResponse, (input, init) => { + timeoutHeaders.push(getRequestHeader(input, init, "X-Stainless-Timeout") ?? ""); + }); + + try { + const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), { + apiKey: "test-key", + streamIdleTimeoutMs: 5_000, + }).result(); + + expect(result.stopReason).toBe("stop"); + expect(getFirstTextContent(result)).toMatchObject({ type: "text", text: "Hello delayed" }); + expect(timeoutHeaders).toContain("1"); + } finally { + if (previousOpenAIFirstEventTimeout === undefined) { + delete Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS; + } else { + Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = previousOpenAIFirstEventTimeout; + } + if (previousGenericFirstEventTimeout === undefined) { + delete Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS; + } else { + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = previousGenericFirstEventTimeout; + } + } + }); + it("times out OpenAI responses streams that only emit no-progress status events", async () => { global.fetch = ((input: string | URL | Request, init?: RequestInit) => Promise.resolve(createNoProgressOpenAIResponsesStream(getRequestSignal(input, init)))) as typeof fetch; diff --git a/packages/ai/test/stream-timeout-defaults.test.ts b/packages/ai/test/stream-timeout-defaults.test.ts index 65a3044ab..5920ef978 100644 --- a/packages/ai/test/stream-timeout-defaults.test.ts +++ b/packages/ai/test/stream-timeout-defaults.test.ts @@ -105,16 +105,20 @@ describe("getStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => { }); describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => { - it("lets the OpenAI idle env widen a lower generic first-event timeout", () => { + it("floors the first-event budget at the caller-resolved idle when the generic env is lower", () => { Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20"; - Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84"; - expect(getOpenAIStreamFirstEventTimeoutMs(84, 300_000)).toBe(84); + expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBe(1500); }); - it("lets the OpenAI first-event env override the OpenAI idle env", () => { + it("honors PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS even when caller pins per-call idle", () => { Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42"; - Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "84"; - expect(getOpenAIStreamFirstEventTimeoutMs(84, 300_000)).toBe(42); + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "20"; + expect(getOpenAIStreamFirstEventTimeoutMs(5_000, 100_000)).toBe(42); + }); + + it("treats PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 as an explicit watchdog disable", () => { + Bun.env.PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0"; + expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined(); }); it("falls back to the generic first-event env when OpenAI env vars are unset", () => { @@ -122,10 +126,9 @@ describe("getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, fallbackMs)", () => expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBe(42); }); - it("treats PI_OPENAI_STREAM_IDLE_TIMEOUT_MS=0 as an OpenAI watchdog disable", () => { - Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "42"; - Bun.env.PI_OPENAI_STREAM_IDLE_TIMEOUT_MS = "0"; - expect(getOpenAIStreamFirstEventTimeoutMs(undefined, 300_000)).toBeUndefined(); + it("respects PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0 disable when no OpenAI override is set", () => { + Bun.env.PI_STREAM_FIRST_EVENT_TIMEOUT_MS = "0"; + expect(getOpenAIStreamFirstEventTimeoutMs(1500, 100_000)).toBeUndefined(); }); }); From 099eebdb44ad7848ea2099c0dd8662047cc8ab53 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 19:06:15 +0000 Subject: [PATCH 11/28] fix(ai): marked codex sse lazy stream as provider-handled openai-codex-responses already owns its first-event/idle watchdog through getOpenAIStreamFirstEventTimeoutMs, so the outer lazy wrapper must skip the generic PI_STREAM_FIRST_EVENT_TIMEOUT_MS race. Without this the Codex SSE path still aborts at the lower generic budget for the same PI_OPENAI_STREAM_IDLE_TIMEOUT_MS > PI_STREAM_FIRST_EVENT_TIMEOUT_MS combination that motivated this PR. Fixes #1603 --- packages/ai/src/providers/register-builtins.ts | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/packages/ai/src/providers/register-builtins.ts b/packages/ai/src/providers/register-builtins.ts index 813751071..ed33369ca 100644 --- a/packages/ai/src/providers/register-builtins.ts +++ b/packages/ai/src/providers/register-builtins.ts @@ -418,7 +418,10 @@ export const streamGoogleGeminiCli = createLazyStream( GOOGLE_GEMINI_CLI_LAZY_STREAM_LIMITS, ); export const streamGoogleVertex = createLazyStream(loadGoogleVertexProviderModule); -export const streamOpenAICodexResponses = createLazyStream(loadOpenAICodexResponsesProviderModule); +export const streamOpenAICodexResponses = createLazyStream( + loadOpenAICodexResponsesProviderModule, + PROVIDER_HANDLED_STREAM_TIMEOUTS, +); export const streamOpenAICompletions = createLazyStream( loadOpenAICompletionsProviderModule, PROVIDER_HANDLED_STREAM_TIMEOUTS, From 5843a78dbfdac45fe9bed24bd75311778036a06b Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 21:02:25 +0000 Subject: [PATCH 12/28] fix(tiny): isolated tiny model worker in subprocess to skip onnxruntime napi crash Moved the tiny title/memory worker from a Bun Worker thread into a child process spawned via Bun.spawn IPC. The agent CLI gains a hidden --tiny-worker dispatch the parent invokes through process.execPath; the parent SIGKILLs the child on dispose so onnxruntime-node's NAPI finalizer never runs in any address space the agent owns. On Windows that finalizer was segfaulting Bun at shutdown after the tiny title model loaded (issue #1606). Drops the now-dead 'close'/'closed' handshake and the unused parentPort bootstrap, and removes tiny/worker.ts from --compile worker entries in both build scripts plus the regression test that pinned them. Fixes #1606 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/scripts/build-binary.ts | 1 - packages/coding-agent/src/cli.ts | 48 ++++++ .../coding-agent/src/tiny/title-client.ts | 156 +++++++++++++----- .../coding-agent/src/tiny/title-protocol.ts | 15 +- packages/coding-agent/src/tiny/worker.ts | 45 +---- .../test/issue-1150-repro.test.ts | 2 - .../test/issue-1606-repro.test.ts | 41 +++++ scripts/ci-release-build-binaries.ts | 1 - 9 files changed, 217 insertions(+), 96 deletions(-) create mode 100644 packages/coding-agent/test/issue-1606-repro.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9e22dabb5..04649eb11 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `omp` segfaulting on exit on Windows after the tiny title/memory model loaded `onnxruntime-node` (issue [#1606](https://github.com/can1357/oh-my-pi/issues/1606)). The tiny model now runs in a Bun subprocess instead of a Worker thread, so the NAPI finalizer that crashes during shutdown never executes in the agent's address space; the subprocess is `SIGKILL`'d on dispose to skip every native destructor on every platform. + ## [15.7.4] - 2026-05-31 ### Removed diff --git a/packages/coding-agent/scripts/build-binary.ts b/packages/coding-agent/scripts/build-binary.ts index 893fa5530..d0b663f07 100644 --- a/packages/coding-agent/scripts/build-binary.ts +++ b/packages/coding-agent/scripts/build-binary.ts @@ -56,7 +56,6 @@ async function main(): Promise { "../stats/src/sync-worker.ts", "./src/tools/browser/tab-worker-entry.ts", "./src/eval/js/worker-entry.ts", - "./src/tiny/worker.ts", // Legacy pi-* extension compat entrypoints served by // `legacy-pi-compat.ts`. These are reached via computed bunfs paths // (which `--compile`'s static analyzer cannot trace), so each must be diff --git a/packages/coding-agent/src/cli.ts b/packages/coding-agent/src/cli.ts index 8f6f99c1c..1d7f06218 100755 --- a/packages/coding-agent/src/cli.ts +++ b/packages/coding-agent/src/cli.ts @@ -54,12 +54,60 @@ async function runSmokeTest(): Promise { process.stdout.write("smoke-test: ok\n"); } +/** + * Hidden subcommand that boots the tiny-model worker inside this process + * over the parent's IPC channel. The agent's main process spawns the same + * binary with this flag so `onnxruntime-node` (loaded transitively by + * `@huggingface/transformers`) lives in a child address space. The parent + * `SIGKILL`s the child on shutdown so the NAPI finalizer never runs in + * either process — that finalizer segfaults Bun on Windows (issue #1606). + */ +async function runTinyWorker(): Promise { + const { startTinyTitleWorker } = await import("./tiny/worker"); + const { promise: shuttingDown, resolve: shutdown } = Promise.withResolvers(); + const send = (message: unknown): void => { + // `process.send` only exists when spawned with an IPC channel; the + // parent always spawns us that way. If it's missing, the parent + // vanished and there's no one to talk to. + const sender = (process as NodeJS.Process & { send?: (m: unknown) => boolean }).send; + if (!sender) { + shutdown(); + return; + } + try { + sender.call(process, message); + } catch { + shutdown(); + } + }; + startTinyTitleWorker({ + send, + onMessage(handler) { + const wrap = (data: unknown): void => handler(data as never); + process.on("message", wrap); + return () => { + process.off("message", wrap); + }; + }, + }); + // Parent went away (crashed, SIGKILL, etc.) — commit suicide so we don't + // linger as an orphan. SIGKILL via `process.kill` keeps us symmetrical + // with the parent's hard-kill on shutdown: skip every JS/native finalizer. + process.on("disconnect", () => shutdown()); + await shuttingDown; + process.kill(process.pid, "SIGKILL"); +} + /** Run the CLI with the given argv (no `process.argv` prefix). */ export async function runCli(argv: string[]): Promise { if (argv[0] === "--smoke-test") { await runSmokeTest(); return; } + if (argv[0] === "--tiny-worker") { + await runTinyWorker(); + return; + } // --help and --version are handled by run() directly, don't rewrite those. // Everything else that isn't a known subcommand routes to "launch". const first = argv[0]; diff --git a/packages/coding-agent/src/tiny/title-client.ts b/packages/coding-agent/src/tiny/title-client.ts index 1382f40bd..b3a15e995 100644 --- a/packages/coding-agent/src/tiny/title-client.ts +++ b/packages/coding-agent/src/tiny/title-client.ts @@ -1,4 +1,6 @@ +import * as path from "node:path"; import { $env, isCompiledBinary, logger } from "@oh-my-pi/pi-utils"; +import type { Subprocess } from "bun"; import { settings } from "../config/settings"; import { tinyModelDeviceSettingToEnv } from "./device"; import { tinyModelDtypeSettingToEnv } from "./dtype"; @@ -12,6 +14,14 @@ import { } from "./models"; import type { TinyTitleProgressEvent, TinyTitleWorkerInbound, TinyTitleWorkerOutbound } from "./title-protocol"; +/** + * Abstraction over the tiny-model subprocess. Modelled as a worker interface + * so existing callers (titles, memory completions, downloads) compose the + * same way; the runtime implementation is a Bun child process so + * `onnxruntime-node`'s NAPI finalizer never runs inside the main agent + * address space — that destructor segfaults Bun on Windows during shutdown + * (issue #1606). + */ interface WorkerHandle { send(message: TinyTitleWorkerInbound): void; onMessage(handler: (message: TinyTitleWorkerOutbound) => void): () => void; @@ -31,6 +41,12 @@ export interface TinyTitleDownloadOptions { const SMOKE_TEST_TIMEOUT_MS = 5_000; +/** + * Hidden subcommand on the main CLI that boots the tiny-model worker in the + * spawned subprocess. Kept in sync with the dispatch in `cli.ts`. + */ +export const TINY_WORKER_ARG = "--tiny-worker"; + function readTinyModelSetting(path: "providers.tinyModelDevice" | "providers.tinyModelDtype"): string | undefined { try { const value = settings.get(path); @@ -66,49 +82,108 @@ export function tinyWorkerEnvOverlay( } /** - * Env handed to the tiny-model worker. The `PI_TINY_DEVICE` / `PI_TINY_DTYPE` env - * vars win; otherwise the persisted `providers.tinyModelDevice` / - * `providers.tinyModelDtype` settings are mapped onto those vars so the worker's - * env-based resolution picks them up. Resolved once at spawn (pipelines are cached). + * Env handed to the tiny-model subprocess. The `PI_TINY_DEVICE` / `PI_TINY_DTYPE` + * env vars win; otherwise the persisted `providers.tinyModelDevice` / + * `providers.tinyModelDtype` settings are mapped onto those vars so the + * subprocess's env-based resolution picks them up. Resolved once at spawn + * (pipelines are cached for the lifetime of the subprocess). */ -function tinyWorkerEnv(): Record | undefined { +function tinyWorkerEnv(): Record { const overlay = tinyWorkerEnvOverlay( $env, readTinyModelSetting("providers.tinyModelDevice"), readTinyModelSetting("providers.tinyModelDtype"), ); - if (Object.keys(overlay).length === 0) return undefined; - return { ...($env as Record), ...overlay }; + const base = $env as Record; + const merged: Record = {}; + for (const key in base) { + const value = base[key]; + if (typeof value === "string") merged[key] = value; + } + for (const key in overlay) merged[key] = overlay[key]; + return merged; } -export function createTinyTitleWorker(): Worker { - const env = tinyWorkerEnv(); - const options: WorkerOptions = env ? { type: "module", env } : { type: "module" }; - return isCompiledBinary() - ? new Worker("./packages/coding-agent/src/tiny/worker.ts", options) - : new Worker(new URL("./worker.ts", import.meta.url).href, options); +/** + * Resolve the argv used to relaunch the agent CLI into tiny-worker mode. In a + * compiled binary the entry point is the binary itself; in dev/source the + * spawned `bun` needs the absolute path to `cli.ts` so it can resolve module + * imports against the on-disk source tree. + */ +function tinyWorkerSpawnCmd(): string[] { + if (isCompiledBinary()) return [process.execPath, TINY_WORKER_ARG]; + const cliPath = path.resolve(import.meta.dir, "..", "cli.ts"); + return [process.execPath, cliPath, TINY_WORKER_ARG]; } -function wrapBunWorker(worker: Worker): WorkerHandle { - (worker as Worker & { unref?: () => void }).unref?.(); +interface SpawnedSubprocess { + proc: Subprocess<"ignore", "inherit", "inherit">; + inbound: Set<(message: TinyTitleWorkerOutbound) => void>; + errors: Set<(error: Error) => void>; +} + +/** + * Spawn the tiny-model worker as a subprocess. Exported for tests and the + * smoke probe; production callers go through {@link spawnTinyTitleWorker} + * which wraps the result in a {@link WorkerHandle}. + */ +export function createTinyTitleSubprocess(): SpawnedSubprocess { + const inbound = new Set<(message: TinyTitleWorkerOutbound) => void>(); + const errors = new Set<(error: Error) => void>(); + const proc = Bun.spawn({ + cmd: tinyWorkerSpawnCmd(), + env: tinyWorkerEnv(), + stdin: "ignore", + stdout: "inherit", + stderr: "inherit", + serialization: "advanced", + windowsHide: true, + ipc(message) { + for (const handler of inbound) handler(message as TinyTitleWorkerOutbound); + }, + onExit(_proc, exitCode, signalCode) { + if (exitCode === 0 || exitCode === null) return; + const signalSuffix = signalCode ? ` (signal ${signalCode})` : ""; + const err = new Error(`tiny model subprocess exited with code ${exitCode}${signalSuffix}`); + for (const handler of errors) handler(err); + }, + }); + // Don't keep the parent event loop alive on account of an idle worker; the + // agent dispose path calls `terminate()` explicitly when shutting down. + proc.unref(); + return { proc, inbound, errors }; +} + +function wrapSubprocess({ proc, inbound, errors }: SpawnedSubprocess): WorkerHandle { return { send(message) { - worker.postMessage(message); + try { + proc.send(message); + } catch (error) { + logger.debug("tiny-title: send to subprocess failed", { + error: error instanceof Error ? error.message : String(error), + }); + } }, onMessage(handler) { - const wrap = (event: MessageEvent): void => handler(event.data as TinyTitleWorkerOutbound); - worker.addEventListener("message", wrap); - return () => worker.removeEventListener("message", wrap); + inbound.add(handler); + return () => inbound.delete(handler); }, onError(handler) { - const wrap = (event: ErrorEvent): void => { - handler(event.error instanceof Error ? event.error : new Error(event.message || "tiny title worker error")); - }; - worker.addEventListener("error", wrap); - return () => worker.removeEventListener("error", wrap); + errors.add(handler); + return () => errors.delete(handler); }, async terminate() { - worker.terminate(); + // SIGKILL: the whole point of the subprocess isolation is that the + // parent never runs `onnxruntime-node`'s NAPI finalizer. A polite + // SIGTERM lets the subprocess try to clean up, which is exactly the + // codepath that crashes Bun on Windows. Hard-kill instead — the + // model lives in process memory and the OS reclaims everything. + try { + proc.kill("SIGKILL"); + } catch { + // Already gone. + } }, }; } @@ -126,10 +201,6 @@ function spawnInlineUnavailableWorker(error: unknown): WorkerHandle { emit({ type: "pong", id: message.id }); return; } - if (message.type === "close") { - emit({ type: "closed" }); - return; - } emit({ type: "error", id: message.id, error: errorMessage }); }); }, @@ -148,9 +219,9 @@ function spawnInlineUnavailableWorker(error: unknown): WorkerHandle { function spawnTinyTitleWorker(): WorkerHandle { try { - return wrapBunWorker(createTinyTitleWorker()); + return wrapSubprocess(createTinyTitleSubprocess()); } catch (error) { - logger.warn("Tiny title Worker spawn failed; local titles disabled", { + logger.warn("Tiny title worker spawn failed; local titles disabled", { error: error instanceof Error ? error.message : String(error), }); return spawnInlineUnavailableWorker(error); @@ -293,9 +364,9 @@ export class TinyTitleClient { } this.#pending.clear(); try { - worker?.send({ type: "close" }); + await worker?.terminate(); } catch { - // Worker may already be gone. + // Already gone. } } @@ -317,7 +388,6 @@ export class TinyTitleClient { this.#emitProgress(message.event); return; } - if (message.type === "closed") return; if (message.type === "pong") return; const pending = this.#pending.get(message.id); @@ -371,25 +441,25 @@ export async function smokeTestTinyTitleWorker({ }: { timeoutMs?: number; } = {}): Promise { - const worker = createTinyTitleWorker(); + const handle = wrapSubprocess(createTinyTitleSubprocess()); const { promise, resolve, reject } = Promise.withResolvers(); const timer = setTimeout(() => reject(new Error(`tiny title worker did not pong within ${timeoutMs}ms`)), timeoutMs); - worker.onmessage = (event: MessageEvent) => { - const message = event.data; + const unsubscribeMessage = handle.onMessage(message => { if (message.type === "pong") { resolve(); return; } + if (message.type === "log") return; reject(new Error(`tiny title worker: expected pong, got ${JSON.stringify(message)}`)); - }; - worker.onerror = (event: ErrorEvent) => { - reject(event.error instanceof Error ? event.error : new Error(event.message || "tiny title worker error")); - }; + }); + const unsubscribeError = handle.onError(reject); try { - worker.postMessage({ type: "ping", id: "smoke" } satisfies TinyTitleWorkerInbound); + handle.send({ type: "ping", id: "smoke" } satisfies TinyTitleWorkerInbound); await promise; } finally { clearTimeout(timer); - worker.terminate(); + unsubscribeMessage(); + unsubscribeError(); + await handle.terminate(); } } diff --git a/packages/coding-agent/src/tiny/title-protocol.ts b/packages/coding-agent/src/tiny/title-protocol.ts index 9267a0b89..4f1bd67ba 100644 --- a/packages/coding-agent/src/tiny/title-protocol.ts +++ b/packages/coding-agent/src/tiny/title-protocol.ts @@ -31,8 +31,7 @@ export type TinyTitleWorkerInbound = | { type: "ping"; id: string } | { type: "generate"; id: string; modelKey: TinyTitleLocalModelKey; message: string } | { type: "complete"; id: string; modelKey: TinyLocalModelKey; prompt: string; maxTokens?: number } - | { type: "download"; id: string; modelKey: TinyLocalModelKey } - | { type: "close" }; + | { type: "download"; id: string; modelKey: TinyLocalModelKey }; export type TinyTitleWorkerOutbound = | { type: "pong"; id: string } @@ -41,11 +40,17 @@ export type TinyTitleWorkerOutbound = | { type: "downloaded"; id: string } | { type: "error"; id: string; error: string } | { type: "progress"; id: string; event: TinyTitleProgressEvent } - | { type: "log"; level: "debug" | "warn" | "error"; msg: string; meta?: Record } - | { type: "closed" }; + | { type: "log"; level: "debug" | "warn" | "error"; msg: string; meta?: Record }; +/** + * Wire transport between the parent (`TinyTitleClient`) and the tiny-model + * subprocess. The parent owns the subprocess lifecycle (graceful work, hard + * kill on shutdown); the protocol therefore carries no explicit close + * handshake — once the parent decides to terminate, it signals the OS to + * reap the child so `onnxruntime-node`'s NAPI finalizer never runs in any + * shared address space. See `title-client.ts` for the spawn/kill glue. + */ export interface TinyTitleTransport { send(message: TinyTitleWorkerOutbound): void; onMessage(handler: (message: TinyTitleWorkerInbound) => void): () => void; - close(): void; } diff --git a/packages/coding-agent/src/tiny/worker.ts b/packages/coding-agent/src/tiny/worker.ts index 2d7a1fa83..118838a67 100644 --- a/packages/coding-agent/src/tiny/worker.ts +++ b/packages/coding-agent/src/tiny/worker.ts @@ -1,7 +1,6 @@ import * as fs from "node:fs/promises"; import { createRequire } from "node:module"; import * as path from "node:path"; -import { parentPort } from "node:worker_threads"; import type { ProgressInfo, TextGenerationPipeline, @@ -20,12 +19,7 @@ import { type TinyTitleLocalModelSpec, } from "./models"; import { formatTitleUserMessage, normalizeGeneratedTitle } from "./text"; -import type { - TinyTitleProgressEvent, - TinyTitleTransport, - TinyTitleWorkerInbound, - TinyTitleWorkerOutbound, -} from "./title-protocol"; +import type { TinyTitleProgressEvent, TinyTitleTransport, TinyTitleWorkerInbound } from "./title-protocol"; const TITLE_PREFILL = ""; const TITLE_CLOSE = ""; @@ -497,16 +491,6 @@ async function generateCompletion( return generated === "" ? null : generated; } -function releasePipelines(): void { - // Intentionally NOT calling `pipeline.dispose()`. transformers.js disposes the - // underlying onnxruntime InferenceSession, freeing native memory that Bun's - // worker/NAPI teardown then frees a second time — a double-free that aborts the - // process on quit ("malloc: pointer being freed was not allocated" / - // "NAPI FATAL ERROR"). The worker is torn down immediately after `close`, so the - // OS reclaims the model memory regardless; skipping dispose avoids the crash. - pipelines.clear(); -} - function enqueueRequest( transport: TinyTitleTransport, request: Extract, @@ -555,33 +539,6 @@ export function startTinyTitleWorker(transport: TinyTitleTransport): void { transport.send({ type: "pong", id: message.id }); return; } - if (message.type === "close") { - releasePipelines(); - transport.send({ type: "closed" }); - transport.close(); - return; - } enqueueRequest(transport, message); }); } - -if (!parentPort) throw new Error("tiny-title-worker: missing parentPort"); - -const port = parentPort; -const transport: TinyTitleTransport = { - send: (message: TinyTitleWorkerOutbound) => port.postMessage(message), - onMessage: handler => { - const wrap = (data: unknown): void => handler(data as TinyTitleWorkerInbound); - port.on("message", wrap); - return () => port.off("message", wrap); - }, - close: () => { - try { - port.close(); - } catch { - // Already closed. - } - }, -}; - -startTinyTitleWorker(transport); diff --git a/packages/coding-agent/test/issue-1150-repro.test.ts b/packages/coding-agent/test/issue-1150-repro.test.ts index ab00ecd70..777c58869 100644 --- a/packages/coding-agent/test/issue-1150-repro.test.ts +++ b/packages/coding-agent/test/issue-1150-repro.test.ts @@ -35,7 +35,6 @@ describe("issue #1150 — release-build script must list all worker --compile en "./packages/stats/src/sync-worker.ts", "./packages/coding-agent/src/tools/browser/tab-worker-entry.ts", "./packages/coding-agent/src/eval/js/worker-entry.ts", - "./packages/coding-agent/src/tiny/worker.ts", ]; it("scripts/ci-release-build-binaries.ts lists every worker as an explicit --compile entrypoint", async () => { @@ -56,7 +55,6 @@ describe("issue #1150 — release-build script must list all worker --compile en "../stats/src/sync-worker.ts", "./src/tools/browser/tab-worker-entry.ts", "./src/eval/js/worker-entry.ts", - "./src/tiny/worker.ts", ]; const source = await Bun.file(devScriptPath).text(); for (const entry of devEntrypoints) { diff --git a/packages/coding-agent/test/issue-1606-repro.test.ts b/packages/coding-agent/test/issue-1606-repro.test.ts new file mode 100644 index 000000000..1c85c6804 --- /dev/null +++ b/packages/coding-agent/test/issue-1606-repro.test.ts @@ -0,0 +1,41 @@ +/** + * Regression for https://github.com/can1357/oh-my-pi/issues/1606 + * + * On Windows, `onnxruntime-node`'s NAPI finalizer segfaults Bun during + * shutdown after `@huggingface/transformers` has loaded a tiny model in a + * Worker thread. The agent used to host the tiny-model worker as a Worker + * inside its own process; tearing the worker down ran the native destructor + * in the parent's address space and crashed the CLI on exit. + * + * The fix relocates the worker to a child process: `title-client.ts` spawns + * `process.execPath … --tiny-worker`, `cli.ts` dispatches that flag into + * `runTinyWorker`, and the parent `SIGKILL`s the child on dispose so the + * native finalizer never runs in either address space. These tests pin the + * three pieces of that contract so a future refactor cannot quietly land + * the original crash again. + */ +import { describe, expect, it } from "bun:test"; +import { smokeTestTinyTitleWorker, TINY_WORKER_ARG } from "../src/tiny/title-client"; + +describe("issue #1606 — tiny model lives in an isolated subprocess", () => { + it("ping/pongs through the spawned worker subprocess and tears it down cleanly", async () => { + // `smokeTestTinyTitleWorker` is the runtime probe wired into + // `omp --smoke-test`: it spawns the worker subprocess via + // `Bun.spawn`, sends a ping over the IPC channel, awaits the pong, + // then SIGKILLs the child. If anyone reverts the worker to an + // in-process `new Worker(...)` thread or drops the `--tiny-worker` + // CLI dispatch, the spawn either picks up the wrong entrypoint or + // the ping never round-trips, and this test fails. + await expect(smokeTestTinyTitleWorker({ timeoutMs: 15_000 })).resolves.toBeUndefined(); + }, 30_000); + + it("CLI dispatches the flag that `title-client.ts` passes to the spawned child", async () => { + // `tinyWorkerSpawnCmd()` and the cli switch must agree on the exact + // flag, character-for-character — the spawned `bun`/binary sees only + // `argv` and there is no fallback path that "re-routes" the worker + // on misnamed flags. Pin the spelling on both ends. + const cliSource = await Bun.file(new URL("../src/cli.ts", import.meta.url)).text(); + expect(cliSource).toContain(`argv[0] === "${TINY_WORKER_ARG}"`); + expect(cliSource).toContain("runTinyWorker"); + }); +}); diff --git a/scripts/ci-release-build-binaries.ts b/scripts/ci-release-build-binaries.ts index fad13b13a..ab72a8ec1 100644 --- a/scripts/ci-release-build-binaries.ts +++ b/scripts/ci-release-build-binaries.ts @@ -27,7 +27,6 @@ const workerEntrypoints = [ "./packages/stats/src/sync-worker.ts", "./packages/coding-agent/src/tools/browser/tab-worker-entry.ts", "./packages/coding-agent/src/eval/js/worker-entry.ts", - "./packages/coding-agent/src/tiny/worker.ts", ]; const isDryRun = process.argv.includes("--dry-run"); const targets: BinaryTarget[] = [ From e69e8b54f86abd6e2f595b409c556d4b288bbebb Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 21:09:07 +0000 Subject: [PATCH 13/28] fix(tiny): surfaced unexpected subprocess signal exits as worker errors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first cut at the subprocess isolation swallowed every signal exit (`exitCode === null`) on the assumption it was the intentional SIGKILL from `terminate()`. That misclassifies real worker deaths — SIGSEGV from a native crash, SIGKILL from the OOM killer, an operator `kill -9` — so any in-flight title/completion/download promise would await forever while `#worker` still pointed at a dead process. Added an `intentionalExit` flag flipped by `wrapSubprocess.terminate()` right before its SIGKILL. `onExit` swallows only the flagged exit; every other signal exit now fires the `errors` channel with a "signal SIGFOO" message so `TinyTitleClient.#handleWorkerError` clears `#pending` and dumps the dead worker handle. Added two regression tests pinning both branches. Reported by chatgpt-codex-connector on #1607. --- .../coding-agent/src/tiny/title-client.ts | 30 +++++++++-- .../test/issue-1606-repro.test.ts | 50 ++++++++++++++++++- 2 files changed, 74 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/src/tiny/title-client.ts b/packages/coding-agent/src/tiny/title-client.ts index b3a15e995..f575ff342 100644 --- a/packages/coding-agent/src/tiny/title-client.ts +++ b/packages/coding-agent/src/tiny/title-client.ts @@ -120,6 +120,13 @@ interface SpawnedSubprocess { proc: Subprocess<"ignore", "inherit", "inherit">; inbound: Set<(message: TinyTitleWorkerOutbound) => void>; errors: Set<(error: Error) => void>; + /** + * Flipped to `true` by {@link wrapSubprocess}'s `terminate()` right + * before it SIGKILLs the child so `onExit` can distinguish the + * expected hard-kill from a crash/OOM/external signal. Only the + * latter is surfaced as a worker error. + */ + intentionalExit: { value: boolean }; } /** @@ -130,6 +137,7 @@ interface SpawnedSubprocess { export function createTinyTitleSubprocess(): SpawnedSubprocess { const inbound = new Set<(message: TinyTitleWorkerOutbound) => void>(); const errors = new Set<(error: Error) => void>(); + const intentionalExit = { value: false }; const proc = Bun.spawn({ cmd: tinyWorkerSpawnCmd(), env: tinyWorkerEnv(), @@ -142,19 +150,28 @@ export function createTinyTitleSubprocess(): SpawnedSubprocess { for (const handler of inbound) handler(message as TinyTitleWorkerOutbound); }, onExit(_proc, exitCode, signalCode) { - if (exitCode === 0 || exitCode === null) return; - const signalSuffix = signalCode ? ` (signal ${signalCode})` : ""; - const err = new Error(`tiny model subprocess exited with code ${exitCode}${signalSuffix}`); + // Clean exit. The child only exits via SIGKILL in practice, but + // treat code 0 as a no-op for symmetry. + if (exitCode === 0) return; + // `exitCode === null` + non-null `signalCode` covers both the + // expected SIGKILL from `terminate()` AND external kills + // (SIGSEGV from a native crash, SIGKILL from the OOM killer, an + // operator `kill -9`, etc.). Swallow only the expected one; + // every other signal exit is a real worker death that must + // fault every in-flight request so callers don't await forever. + if (exitCode === null && intentionalExit.value) return; + const reason = exitCode !== null ? `code ${exitCode}` : `signal ${signalCode ?? "unknown"}`; + const err = new Error(`tiny model subprocess exited with ${reason}`); for (const handler of errors) handler(err); }, }); // Don't keep the parent event loop alive on account of an idle worker; the // agent dispose path calls `terminate()` explicitly when shutting down. proc.unref(); - return { proc, inbound, errors }; + return { proc, inbound, errors, intentionalExit }; } -function wrapSubprocess({ proc, inbound, errors }: SpawnedSubprocess): WorkerHandle { +function wrapSubprocess({ proc, inbound, errors, intentionalExit }: SpawnedSubprocess): WorkerHandle { return { send(message) { try { @@ -179,6 +196,9 @@ function wrapSubprocess({ proc, inbound, errors }: SpawnedSubprocess): WorkerHan // SIGTERM lets the subprocess try to clean up, which is exactly the // codepath that crashes Bun on Windows. Hard-kill instead — the // model lives in process memory and the OS reclaims everything. + // Flip the intentional-exit flag *before* killing so `onExit` can + // tell this apart from a crash or external SIGKILL. + intentionalExit.value = true; try { proc.kill("SIGKILL"); } catch { diff --git a/packages/coding-agent/test/issue-1606-repro.test.ts b/packages/coding-agent/test/issue-1606-repro.test.ts index 1c85c6804..9b771b336 100644 --- a/packages/coding-agent/test/issue-1606-repro.test.ts +++ b/packages/coding-agent/test/issue-1606-repro.test.ts @@ -15,7 +15,7 @@ * the original crash again. */ import { describe, expect, it } from "bun:test"; -import { smokeTestTinyTitleWorker, TINY_WORKER_ARG } from "../src/tiny/title-client"; +import { createTinyTitleSubprocess, smokeTestTinyTitleWorker, TINY_WORKER_ARG } from "../src/tiny/title-client"; describe("issue #1606 — tiny model lives in an isolated subprocess", () => { it("ping/pongs through the spawned worker subprocess and tears it down cleanly", async () => { @@ -38,4 +38,52 @@ describe("issue #1606 — tiny model lives in an isolated subprocess", () => { expect(cliSource).toContain(`argv[0] === "${TINY_WORKER_ARG}"`); expect(cliSource).toContain("runTinyWorker"); }); + + it("surfaces unexpected signal exits so in-flight callers don't await forever", async () => { + // If the child dies from a signal we did NOT request — SIGSEGV from a + // native crash (the original Windows shutdown bug, now relocated to + // the child), an OOM SIGKILL, or an operator `kill -9` — the + // subprocess wrapper must fault every in-flight request via the + // `errors` channel. The original fix swallowed any `exitCode === null` + // exit unconditionally, which left `TinyTitleClient.#pending` + // promises hanging forever. Pin the new contract: an external + // SIGKILL (no `intentionalExit` flip) MUST surface a worker error. + const sub = createTinyTitleSubprocess(); + try { + const { promise, resolve } = Promise.withResolvers(); + sub.errors.add(resolve); + sub.proc.kill("SIGKILL"); + const err = await promise; + expect(err.message).toMatch(/signal/i); + } finally { + // Ensure the child is reaped even on assertion failure. + try { + sub.proc.kill("SIGKILL"); + } catch {} + await sub.proc.exited; + } + }, 15_000); + + it("does not surface intentional terminate() SIGKILLs as worker errors", async () => { + // Inverse of the previous test: a SIGKILL issued by the wrapper's + // own `terminate()` MUST NOT fault callers — terminate is the + // shutdown path and the worker handle is already torn down by then. + // Regression guard against an over-eager fix that surfaces every + // signal exit indiscriminately. + const sub = createTinyTitleSubprocess(); + let errored = false; + sub.errors.add(() => { + errored = true; + }); + // Simulate what `wrapSubprocess.terminate()` does: flip the flag, + // then SIGKILL. We test the primitive directly rather than going + // through the wrapper to avoid coupling to `WorkerHandle` internals. + sub.intentionalExit.value = true; + sub.proc.kill("SIGKILL"); + await sub.proc.exited; + // Give onExit a microtask to drain — Bun's exited promise resolves + // after onExit fires, but be defensive. + await Bun.sleep(20); + expect(errored).toBe(false); + }, 10_000); }); From 64b48b57b1f92959e5ea4c60a687fcd629f6e2ba Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 1 Jun 2026 02:40:45 +0000 Subject: [PATCH 14/28] fix(tui): rebuilt scrollback during assistant streaming Enable eager native scrollback rebuild mode while assistant text is actively streaming so reflowed Markdown rows do not leave stale duplicated tails in WSL/Windows Terminal scrollback.\n\nFixes #1615 --- packages/coding-agent/CHANGELOG.md | 3 ++ .../src/modes/controllers/event-controller.ts | 46 +++++++++++-------- .../event-controller-tool-render-mode.test.ts | 40 +++++++++++++++- 3 files changed, 70 insertions(+), 19 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9e22dabb5..b81f55cc4 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] +### Fixed + +- Fixed streaming assistant responses leaving duplicated tail rows in WSL/Windows Terminal scrollback by enabling eager native-scrollback rebuilds while assistant text is actively streaming ([#1615](https://github.com/can1357/oh-my-pi/issues/1615)). ## [15.7.4] - 2026-05-31 diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index 603c3ebae..cb25192e7 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -25,12 +25,13 @@ type AgentSessionEventKind = AgentSessionEvent["type"]; const IRC_MESSAGE_VISIBLE_TTL_MS = 10_000; -// Events that change which foreground tools are executing, or that reset a turn. -// The eager native-scrollback rebuild mode is recomputed only on these — other -// events (assistant text streaming, IRC, notices) leave it untouched so plain -// streaming keeps the no-yank deferral. -const TOOL_RENDER_MODE_EVENTS: Record = { +// Events that change foreground streaming state, or that reset a turn. The TUI +// eager native-scrollback rebuild mode is recomputed only on these so unrelated +// IRC/notices/status refreshes do not toggle scrollback replay policy. +const STREAM_RENDER_MODE_EVENTS: Record = { agent_start: true, + message_start: true, + message_end: true, tool_execution_start: true, tool_execution_update: true, tool_execution_end: true, @@ -46,6 +47,7 @@ export class EventController { #renderedCustomMessages = new Set(); #lastIntent: string | undefined = undefined; #backgroundToolCallIds = new Set(); + #assistantMessageStreaming = false; #readToolCallArgs = new Map>(); #readToolCallAssistantComponents = new Map(); #lastAssistantComponent: AssistantMessageComponent | undefined = undefined; @@ -169,24 +171,27 @@ export class EventController { const run = this.#handlers[event.type] as (e: AgentSessionEvent) => Promise; await run(event); - // While a foreground tool is executing, its streaming result re-renders and can - // re-lay-out rows that already scrolled into native scrollback. Let the TUI - // rebuild history on those offscreen edits (a snap to the tail is acceptable - // mid-tool) instead of deferring, which would leave stale/duplicated rows. - // Background-running tools are excluded so their late async updates — and the - // assistant text that streams alongside them — keep the no-yank deferral; - // agent_start resets the mode at every turn boundary. - if (TOOL_RENDER_MODE_EVENTS[event.type]) { + // While assistant text or a foreground tool is streaming, rows above the + // viewport can re-layout after they have already entered native scrollback + // (Markdown fences, wrapping, previews). Let the TUI rebuild history on + // those offscreen edits instead of deferring, which otherwise leaves stale + // tail rows duplicated above the live viewport. + // Background-running tools are excluded so late async updates outside the + // active foreground stream keep the no-yank deferral; agent_start resets + // the mode at every turn boundary. + if (STREAM_RENDER_MODE_EVENTS[event.type]) { this.#refreshToolRenderMode(); } } #refreshToolRenderMode(): void { - let foregroundToolActive = false; - for (const toolCallId of this.ctx.pendingTools.keys()) { - if (!this.#backgroundToolCallIds.has(toolCallId)) { - foregroundToolActive = true; - break; + let foregroundToolActive = this.#assistantMessageStreaming; + if (!foregroundToolActive) { + for (const toolCallId of this.ctx.pendingTools.keys()) { + if (!this.#backgroundToolCallIds.has(toolCallId)) { + foregroundToolActive = true; + break; + } } } this.ctx.ui.setEagerNativeScrollbackRebuild(foregroundToolActive); @@ -196,6 +201,7 @@ export class EventController { this.#lastIntent = undefined; this.#readToolCallArgs.clear(); this.#readToolCallAssistantComponents.clear(); + this.#assistantMessageStreaming = false; this.#lastAssistantComponent = undefined; if (this.ctx.retryEscapeHandler) { this.ctx.editor.onEscape = this.ctx.retryEscapeHandler; @@ -268,6 +274,7 @@ export class EventController { this.ctx.ui.requestRender(); } else if (event.message.role === "assistant") { this.#lastThinkingCount = 0; + this.#assistantMessageStreaming = true; this.#resetReadGroup(); this.ctx.streamingComponent = new AssistantMessageComponent(undefined, this.ctx.hideThinkingBlock, () => this.ctx.ui.requestRender(), @@ -414,6 +421,9 @@ export class EventController { async #handleMessageEnd(event: Extract): Promise { if (event.message.role === "user") return; + if (event.message.role === "assistant") { + this.#assistantMessageStreaming = false; + } if (this.ctx.streamingComponent && event.message.role === "assistant") { this.ctx.streamingMessage = event.message; let errorMessage: string | undefined; diff --git a/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts index 2927e9ec8..4c8ca2e35 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts @@ -1,4 +1,5 @@ -import { afterEach, describe, expect, it, vi } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { EventController } from "@oh-my-pi/pi-coding-agent/modes/controllers/event-controller"; import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; import type { AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -6,11 +7,15 @@ import type { AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent- function createContext() { const setEagerNativeScrollbackRebuild = vi.fn(); const pendingTools = new Map(); + const chatContainer = { addChild: vi.fn() }; const ctx = { isInitialized: true, statusLine: { invalidate: vi.fn() }, updateEditorTopBorder: vi.fn(), pendingTools, + chatContainer, + hideThinkingBlock: false, + session: { isTtsrAbortPending: false, retryAttempt: 0 }, ui: { setEagerNativeScrollbackRebuild, requestRender: vi.fn() }, } as unknown as InteractiveModeContext; return { ctx, pendingTools, setEagerNativeScrollbackRebuild }; @@ -26,7 +31,13 @@ const REFRESH_TRIGGER = { } as unknown as AgentSessionEvent; describe("EventController tool render mode", () => { + beforeEach(async () => { + resetSettingsForTest(); + await Settings.init({ inMemory: true }); + }); + afterEach(() => { + resetSettingsForTest(); vi.restoreAllMocks(); }); @@ -42,4 +53,31 @@ describe("EventController tool render mode", () => { await controller.handleEvent(REFRESH_TRIGGER); expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(false); }); + it("enables eager native scrollback rebuild while assistant text is streaming", async () => { + const { ctx, setEagerNativeScrollbackRebuild } = createContext(); + const controller = new EventController(ctx); + const message = { + role: "assistant", + content: [{ type: "text", text: "" }], + api: "anthropic-messages", + provider: "anthropic", + model: "test-model", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 0, + } as const; + + await controller.handleEvent({ type: "message_start", message } as unknown as AgentSessionEvent); + expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(true); + + await controller.handleEvent({ type: "message_end", message } as unknown as AgentSessionEvent); + expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(false); + }); }); From eaba4031caaa115c991297b48028d61bd42da3dd Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 1 Jun 2026 02:45:37 +0000 Subject: [PATCH 15/28] fix(tui): reset assistant stream rebuild mode Reset eager scrollback rebuild mode on agent_end when provider streams fail before message_end.\n\nFixes #1615 --- .../src/modes/controllers/event-controller.ts | 3 +- .../event-controller-tool-render-mode.test.ts | 69 +++++++++++++------ 2 files changed, 50 insertions(+), 22 deletions(-) diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index cb25192e7..4c24d3c50 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -30,6 +30,7 @@ const IRC_MESSAGE_VISIBLE_TTL_MS = 10_000; // IRC/notices/status refreshes do not toggle scrollback replay policy. const STREAM_RENDER_MODE_EVENTS: Record = { agent_start: true, + agent_end: true, message_start: true, message_end: true, tool_execution_start: true, @@ -606,8 +607,8 @@ export class EventController { } } } - async #handleAgentEnd(_event: Extract): Promise { + this.#assistantMessageStreaming = false; if (this.ctx.loadingAnimation) { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; diff --git a/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts index 4c8ca2e35..f0bb4d15c 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-tool-render-mode.test.ts @@ -7,15 +7,23 @@ import type { AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent- function createContext() { const setEagerNativeScrollbackRebuild = vi.fn(); const pendingTools = new Map(); - const chatContainer = { addChild: vi.fn() }; + const chatContainer = { addChild: vi.fn(), removeChild: vi.fn() }; const ctx = { isInitialized: true, + isBackgrounded: false, statusLine: { invalidate: vi.fn() }, updateEditorTopBorder: vi.fn(), pendingTools, chatContainer, hideThinkingBlock: false, - session: { isTtsrAbortPending: false, retryAttempt: 0 }, + editor: { getText: vi.fn(() => "") }, + flushPendingModelSwitch: vi.fn(), + session: { + agent: { state: { messages: [] } }, + isCompacting: false, + isTtsrAbortPending: false, + retryAttempt: 0, + }, ui: { setEagerNativeScrollbackRebuild, requestRender: vi.fn() }, } as unknown as InteractiveModeContext; return { ctx, pendingTools, setEagerNativeScrollbackRebuild }; @@ -30,6 +38,24 @@ const REFRESH_TRIGGER = { partialResult: { content: [], details: {} }, } as unknown as AgentSessionEvent; +const ASSISTANT_MESSAGE = { + role: "assistant", + content: [{ type: "text", text: "" }], + api: "anthropic-messages", + provider: "anthropic", + model: "test-model", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 0, +} as const; + describe("EventController tool render mode", () => { beforeEach(async () => { resetSettingsForTest(); @@ -53,31 +79,32 @@ describe("EventController tool render mode", () => { await controller.handleEvent(REFRESH_TRIGGER); expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(false); }); + it("enables eager native scrollback rebuild while assistant text is streaming", async () => { const { ctx, setEagerNativeScrollbackRebuild } = createContext(); const controller = new EventController(ctx); - const message = { - role: "assistant", - content: [{ type: "text", text: "" }], - api: "anthropic-messages", - provider: "anthropic", - model: "test-model", - usage: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "stop", - timestamp: 0, - } as const; - await controller.handleEvent({ type: "message_start", message } as unknown as AgentSessionEvent); + await controller.handleEvent({ + type: "message_start", + message: ASSISTANT_MESSAGE, + } as unknown as AgentSessionEvent); expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(true); - await controller.handleEvent({ type: "message_end", message } as unknown as AgentSessionEvent); + await controller.handleEvent({ type: "message_end", message: ASSISTANT_MESSAGE } as unknown as AgentSessionEvent); + expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(false); + }); + + it("resets eager native scrollback rebuild when a stream ends without assistant message_end", async () => { + const { ctx, setEagerNativeScrollbackRebuild } = createContext(); + const controller = new EventController(ctx); + + await controller.handleEvent({ + type: "message_start", + message: ASSISTANT_MESSAGE, + } as unknown as AgentSessionEvent); + expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(true); + + await controller.handleEvent({ type: "agent_end" } as unknown as AgentSessionEvent); expect(setEagerNativeScrollbackRebuild).toHaveBeenLastCalledWith(false); }); }); From d14aa4dbe59f46afe90097583914cf9c4d5e4c82 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 1 Jun 2026 06:20:55 +0000 Subject: [PATCH 16/28] fix(task): demote subagent reminder-loop abort log The catch around the subagent yield-reminder prompt previously logged every exception at ERROR. User cancel (^C) and compaction-driven aborts both surface as ToolAbortError through awaitAbortable, so benign control flow generated 9 spurious 'Subagent prompt failed' errors in 2 days on the reporter's instance. Gate the ERROR branch on '!abortSignal.aborted && !(err instanceof ToolAbortError)' and route the abort path to logger.debug. The outer catch + finally still mark the run aborted, so observable behaviour is unchanged. Fixes #1623 --- packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/task/executor.ts | 16 ++++++-- .../task/executor-subagent-reminders.test.ts | 37 +++++++++++++++++++ 3 files changed, 54 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9e22dabb5..19c56cb30 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed subagent yield-reminder loop logging benign user/compaction aborts as `ERROR`. The catch around `session.prompt`/`waitForIdle` in `task/executor.ts` now demotes `ToolAbortError` and signal-aborted exits to `debug` and keeps `ERROR` for genuine prompt failures only ([#1623](https://github.com/can1357/oh-my-pi/issues/1623)). + ## [15.7.4] - 2026-05-31 ### Removed diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index d383feb79..5e4793a66 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -1446,9 +1446,19 @@ export async function runSubprocess(options: ExecutorOptions): Promise { expect(result.stderr).toMatch(/options\.authStorage.*modelRegistry\.authStorage/); expect(createAgentSessionSpy).not.toHaveBeenCalled(); }); + + it("logs reminder-loop aborts at debug, not error (issue #1623)", async () => { + // Repro: user ^C or compaction aborts pending operations while the + // yield-reminder loop is awaiting session.prompt. awaitAbortable rejects + // with ToolAbortError, which previously surfaced as logger.error and + // polluted operator dashboards. + const abortController = new AbortController(); + const debugSpy = vi.spyOn(logger, "debug").mockImplementation(() => {}); + const errorSpy = vi.spyOn(logger, "error").mockImplementation(() => {}); + + const session = createMockSession(({ promptIndex, emit, state }) => { + if (promptIndex === 1) { + // Initial prompt: stop without yielding so the reminder loop kicks in. + const assistant = createAssistantStopMessage("no yield yet"); + state.messages.push(assistant); + emit({ type: "message_end", message: assistant }); + return; + } + // Reminder prompt: abort the run while it is in flight. The follow-up + // awaitAbortable(session.waitForIdle()) then throws ToolAbortError into + // the catch we are guarding. + abortController.abort(); + }); + + mockCreateAgentSession(session); + + const result = await runSubprocess({ + ...baseOptions, + id: "subagent-abort-during-reminder", + signal: abortController.signal, + }); + + expect(result.aborted).toBe(true); + expect(errorSpy).not.toHaveBeenCalledWith("Subagent prompt failed", expect.anything()); + expect(debugSpy).toHaveBeenCalledWith("Subagent prompt aborted", expect.anything()); + }); }); describe("runSubprocess telemetry propagation", () => { From 8580ba248cff3661351e643519b945ca44189a83 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 1 Jun 2026 06:22:27 +0000 Subject: [PATCH 17/28] fix(tool): handled string paths in find renderer Guarded find renderer path summaries so raw pre-validation string paths render instead of throwing. Added coverage for pending, fallback, empty, and detailed result render paths.\n\nFixes #1622 --- packages/coding-agent/src/tools/find.ts | 19 +++++-- .../test/tools/find-validate-paths.test.ts | 55 ++++++++++++++++++- 2 files changed, 67 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/tools/find.ts b/packages/coding-agent/src/tools/find.ts index 85f6c2484..c8c7e9f6c 100644 --- a/packages/coding-agent/src/tools/find.ts +++ b/packages/coding-agent/src/tools/find.ts @@ -443,10 +443,14 @@ export class FindTool implements AgentTool { // ============================================================================= interface FindRenderArgs { - paths?: string[]; + paths?: string | string[]; limit?: number; } +function formatFindRenderPaths(paths: FindRenderArgs["paths"]): string | undefined { + return Array.isArray(paths) ? paths.join(", ") : paths; +} + const COLLAPSED_LIST_LIMIT = PREVIEW_LIMITS.COLLAPSED_ITEMS; export const findToolRenderer = { @@ -456,7 +460,7 @@ export const findToolRenderer = { if (args.limit !== undefined) meta.push(`limit:${args.limit}`); const text = renderStatusLine( - { icon: "pending", title: "Find", description: args.paths?.join(", ") || "*", meta }, + { icon: "pending", title: "Find", description: formatFindRenderPaths(args.paths) || "*", meta }, uiTheme, ); return new Text(text, 0, 0); @@ -493,7 +497,7 @@ export const findToolRenderer = { { icon: "success", title: "Find", - description: args?.paths?.join(", "), + description: formatFindRenderPaths(args?.paths), meta: [formatCount("file", lines.length)], }, uiTheme, @@ -528,7 +532,7 @@ export const findToolRenderer = { if (fileCount === 0) { const header = renderStatusLine( - { icon: "warning", title: "Find", description: args?.paths?.join(", "), meta: ["0 files"] }, + { icon: "warning", title: "Find", description: formatFindRenderPaths(args?.paths), meta: ["0 files"] }, uiTheme, ); const lines = [header, formatEmptyMessage("No files found", uiTheme)]; @@ -539,7 +543,12 @@ export const findToolRenderer = { if (details?.scopePath) meta.push(`in ${details.scopePath}`); if (truncated) meta.push(uiTheme.fg("warning", "truncated")); const header = renderStatusLine( - { icon: truncated ? "warning" : "success", title: "Find", description: args?.paths?.join(", "), meta }, + { + icon: truncated ? "warning" : "success", + title: "Find", + description: formatFindRenderPaths(args?.paths), + meta, + }, uiTheme, ); diff --git a/packages/coding-agent/test/tools/find-validate-paths.test.ts b/packages/coding-agent/test/tools/find-validate-paths.test.ts index db6d6bd63..ed2167101 100644 --- a/packages/coding-agent/test/tools/find-validate-paths.test.ts +++ b/packages/coding-agent/test/tools/find-validate-paths.test.ts @@ -1,5 +1,25 @@ -import { describe, expect, it } from "bun:test"; -import { validateFindPathInputs } from "../../src/tools/find"; +import { beforeAll, describe, expect, it } from "bun:test"; +import type { Component } from "@oh-my-pi/pi-tui"; +import type { RenderResultOptions } from "../../src/extensibility/custom-tools/types"; +import { getThemeByName, initTheme, type Theme } from "../../src/modes/theme/theme"; +import { findToolRenderer, validateFindPathInputs } from "../../src/tools/find"; + +let uiTheme: Theme; + +beforeAll(async () => { + await initTheme(false, undefined, undefined, "dark", "light"); + const theme = await getThemeByName("dark"); + if (!theme) throw new Error("Missing dark theme"); + uiTheme = theme; +}); +const renderOptions: RenderResultOptions = { + expanded: false, + isPartial: true, +}; + +function renderText(component: Component): string { + return Bun.stripANSI(component.render(160).join("\n")); +} describe("validateFindPathInputs", () => { it("accepts a normal array of glob entries", () => { @@ -40,3 +60,34 @@ describe("validateFindPathInputs", () => { expect(() => validateFindPathInputs(["\\{a,b}"])).toThrow(/paths is an array/); }); }); + +describe("findToolRenderer", () => { + it("accepts a single string paths value before validation", async () => { + const args = { paths: "src/**/*.ts" }; + const renderings = [ + findToolRenderer.renderCall(args, renderOptions, uiTheme), + findToolRenderer.renderResult( + { content: [{ type: "text", text: "src/index.ts\n" }] }, + renderOptions, + uiTheme, + args, + ), + findToolRenderer.renderResult( + { content: [{ type: "text", text: "" }], details: { fileCount: 0, files: [] } }, + renderOptions, + uiTheme, + args, + ), + findToolRenderer.renderResult( + { content: [{ type: "text", text: "src/index.ts" }], details: { fileCount: 1, files: ["src/index.ts"] } }, + renderOptions, + uiTheme, + args, + ), + ]; + + for (const component of renderings) { + expect(renderText(component)).toContain("src/**/*.ts"); + } + }); +}); From c57d0f6aeb1d000199201748b3f017dcb6de0d51 Mon Sep 17 00:00:00 2001 From: ephraimduncan Date: Mon, 1 Jun 2026 10:19:10 +0000 Subject: [PATCH 18/28] fix(ai/anthropic): request current Claude Code OAuth scopes The Anthropic OAuth login requested an outdated scope set (`org:create_api_key user:profile user:inference`). Recent Claude Code versions request `user:profile user:inference user:sessions:claude_code user:mcp_servers user:file_upload`, so omp's consent screen omitted the Claude Code session, connector (MCP), and file-upload grants the current client carries. omp uses the OAuth access token (sk-ant-oat...) directly for inference and never mints an API key, so `org:create_api_key` was unused. Drop it and align with the current Claude Code scope set. --- packages/ai/src/utils/oauth/anthropic.ts | 2 +- packages/ai/test/anthropic-oauth.test.ts | 4 +++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/packages/ai/src/utils/oauth/anthropic.ts b/packages/ai/src/utils/oauth/anthropic.ts index 7174d5d79..8447eb344 100644 --- a/packages/ai/src/utils/oauth/anthropic.ts +++ b/packages/ai/src/utils/oauth/anthropic.ts @@ -11,7 +11,7 @@ const AUTHORIZE_URL = "https://claude.ai/oauth/authorize"; const TOKEN_URL = "https://api.anthropic.com/v1/oauth/token"; const CALLBACK_PORT = 54545; const CALLBACK_PATH = "/callback"; -const SCOPES = "org:create_api_key user:profile user:inference"; +const SCOPES = "user:profile user:inference user:sessions:claude_code user:mcp_servers user:file_upload"; function formatErrorDetails(error: unknown): string { if (error instanceof Error) { diff --git a/packages/ai/test/anthropic-oauth.test.ts b/packages/ai/test/anthropic-oauth.test.ts index 0916eb13f..2af3f501f 100644 --- a/packages/ai/test/anthropic-oauth.test.ts +++ b/packages/ai/test/anthropic-oauth.test.ts @@ -20,7 +20,9 @@ describe("anthropic oauth alignment", () => { const authUrl = new URL(url); expect(authUrl.origin + authUrl.pathname).toBe("https://claude.ai/oauth/authorize"); - expect(authUrl.searchParams.get("scope")).toBe("org:create_api_key user:profile user:inference"); + expect(authUrl.searchParams.get("scope")).toBe( + "user:profile user:inference user:sessions:claude_code user:mcp_servers user:file_upload", + ); expect(authUrl.searchParams.get("state")).toBe(state); expect(authUrl.searchParams.get("redirect_uri")).toBe(redirectUri); expect(authUrl.searchParams.get("code_challenge_method")).toBe("S256"); From 73102d20df60f32cc5bb018e6f9b99d4f2074300 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 31 May 2026 22:57:39 +0000 Subject: [PATCH 19/28] fix(coding-agent): routed local:// reads through the calling session - Extended ResolveContext / WriteContext with localProtocolOptions so the internal-URL router can thread the calling session's local-root mapping through to handlers. - LocalProtocolHandler.resolveOptions now prefers context.localProtocolOptions before consulting the process-global override or the first main-kind session in AgentRegistry, fixing multi-session ACP hosts (cmux) where reads of local://PLAN.md were routing to a sibling session's artifacts dir even though plan-mode writes succeeded against the calling session. - read, find, ast_grep, ast_edit, and search now thread this.session.localProtocolOptions into the router so local://, memory://, agent://, and other handlers see the right caller. - Added regression tests covering the override-vs-context priority and the ENOENT-against-caller-root path. Fixes #1608 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/internal-urls/local-protocol.ts | 34 +++++++---- .../coding-agent/src/internal-urls/types.ts | 15 +++++ packages/coding-agent/src/tools/ast-edit.ts | 3 + packages/coding-agent/src/tools/ast-grep.ts | 3 + packages/coding-agent/src/tools/find.ts | 7 ++- packages/coding-agent/src/tools/path-utils.ts | 15 ++++- packages/coding-agent/src/tools/read.ts | 1 + packages/coding-agent/src/tools/search.ts | 13 +++- .../test/internal-urls/local-protocol.test.ts | 61 +++++++++++++++++++ 10 files changed, 141 insertions(+), 15 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9e22dabb5..d4a233649 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `read local://` resolving to the wrong session's artifacts directory in multi-session ACP hosts (e.g. cmux). `LocalProtocolHandler.resolve` now honors `context.localProtocolOptions` supplied by the calling tool before falling back to the process-wide override or the first `main`-kind session in the global `AgentRegistry`; `read`, `find`, `search`, `ast_grep`, and `ast_edit` thread their session's options through so a `local://PLAN.md` lookup hits the calling session's `local` root instead of a sibling session's ([#1608](https://github.com/can1357/oh-my-pi/issues/1608)). + ## [15.7.4] - 2026-05-31 ### Removed diff --git a/packages/coding-agent/src/internal-urls/local-protocol.ts b/packages/coding-agent/src/internal-urls/local-protocol.ts index f16fe064c..503566cf5 100644 --- a/packages/coding-agent/src/internal-urls/local-protocol.ts +++ b/packages/coding-agent/src/internal-urls/local-protocol.ts @@ -5,7 +5,7 @@ import { isEnoent } from "@oh-my-pi/pi-utils"; import { AgentRegistry } from "../registry/agent-registry"; import { parseInternalUrl } from "./parse"; import { validateRelativePath } from "./skill-protocol"; -import type { InternalResource, InternalUrl, ProtocolHandler, UrlCompletion } from "./types"; +import type { InternalResource, InternalUrl, ProtocolHandler, ResolveContext, UrlCompletion } from "./types"; export interface LocalProtocolOptions { getArtifactsDir?: () => string | null; @@ -164,13 +164,25 @@ export class LocalProtocolHandler implements ProtocolHandler { * Returns the active local-protocol options. * * Resolution order: - * 1. Explicit override installed via {@link setOverride} (used by subagents - * that share their parent's root and by SDK consumers with a custom - * artifacts/session id mapping). - * 2. The main session in `AgentRegistry.global()`. Its `SessionManager` - * supplies both `getArtifactsDir` and `getSessionId`. + * 1. **Caller-supplied** `context.localProtocolOptions` (the actual session + * that initiated the `read`/`find`/`search`/`router.resolve` call). This + * is what keeps `local://` reads pinned to the calling session in + * multi-session hosts (cmux/ACP, embedded SDK consumers) where every + * session registers as `kind: "main"` and "first one wins" would route + * to the wrong artifacts directory. + * 2. Explicit process-global override installed via {@link setOverride} + * (used by SDK consumers with a custom artifacts/session-id mapping and + * by code paths that do not have a calling session, e.g. TUI hyperlink + * resolution). + * 3. The first `main`-kind session in `AgentRegistry.global()`. Its + * `SessionManager` supplies both `getArtifactsDir` and `getSessionId`. + * Last-resort fallback — every caller that has a session reference + * SHOULD thread it through `context` so this branch is never taken in + * multi-session setups. */ - static resolveOptions(): LocalProtocolOptions | undefined { + static resolveOptions(context?: ResolveContext): LocalProtocolOptions | undefined { + const fromContext = context?.localProtocolOptions; + if (fromContext) return fromContext; const override = LocalProtocolHandler.#override; if (override) return override; const main = AgentRegistry.global() @@ -184,8 +196,8 @@ export class LocalProtocolHandler implements ProtocolHandler { }; } - async resolve(url: InternalUrl): Promise { - const opts = LocalProtocolHandler.resolveOptions(); + async resolve(url: InternalUrl, context?: ResolveContext): Promise { + const opts = LocalProtocolHandler.resolveOptions(context); if (!opts) { throw new Error("No session - local:// unavailable"); } @@ -247,8 +259,8 @@ export class LocalProtocolHandler implements ProtocolHandler { }; } - async complete(): Promise { - const opts = LocalProtocolHandler.resolveOptions(); + async complete(_query?: string, context?: ResolveContext): Promise { + const opts = LocalProtocolHandler.resolveOptions(context); if (!opts) return []; const localRoot = path.resolve(resolveLocalRoot(opts)); try { diff --git a/packages/coding-agent/src/internal-urls/types.ts b/packages/coding-agent/src/internal-urls/types.ts index 62015d376..dcbd3174f 100644 --- a/packages/coding-agent/src/internal-urls/types.ts +++ b/packages/coding-agent/src/internal-urls/types.ts @@ -5,6 +5,8 @@ * providing access to agent outputs and server resources without exposing filesystem paths. */ +import type { LocalProtocolOptions } from "./local-protocol"; + /** * Raw resource payload returned by protocol handlers. The `immutable` flag is * applied by the router from {@link ProtocolHandler.immutable}, so handlers do @@ -77,6 +79,17 @@ export interface ResolveContext { settings?: unknown; /** Caller's abort signal. */ signal?: AbortSignal; + /** + * Calling session's `local://` root mapping. When present, the local-protocol + * handler resolves the URL against THIS session's artifacts dir instead of + * picking the first `main`-kind session from the global `AgentRegistry`. + * + * Required for correctness in multi-session hosts (cmux/ACP, embedded SDK + * consumers) where multiple sessions are registered as `main` and the + * "first one wins" lookup picks the wrong artifacts directory — see + * [#1608](https://github.com/can1357/oh-my-pi/issues/1608). + */ + localProtocolOptions?: LocalProtocolOptions; } /** @@ -89,6 +102,8 @@ export interface WriteContext { cwd?: string; /** Caller's abort signal. */ signal?: AbortSignal; + /** Calling session's `local://` root mapping — see {@link ResolveContext.localProtocolOptions}. */ + localProtocolOptions?: LocalProtocolOptions; } /** diff --git a/packages/coding-agent/src/tools/ast-edit.ts b/packages/coding-agent/src/tools/ast-edit.ts index 08c679aa4..4cc1d9e13 100644 --- a/packages/coding-agent/src/tools/ast-edit.ts +++ b/packages/coding-agent/src/tools/ast-edit.ts @@ -230,6 +230,9 @@ export class AstEditTool implements AgentTool { if (hasGlobPathChars(rawPattern)) { throw new ToolError(`Glob patterns are not supported for internal URLs: ${rawPattern}`); } - const resource = await internalRouter.resolve(rawPattern); + const resource = await internalRouter.resolve(rawPattern, { + cwd: this.session.cwd, + settings: this.session.settings, + signal, + localProtocolOptions: this.session.localProtocolOptions, + }); if (!resource.sourcePath) { throw new ToolError(`Cannot find internal URL without a backing file: ${rawPattern}`); } diff --git a/packages/coding-agent/src/tools/path-utils.ts b/packages/coding-agent/src/tools/path-utils.ts index c6e3add2a..749ced72b 100644 --- a/packages/coding-agent/src/tools/path-utils.ts +++ b/packages/coding-agent/src/tools/path-utils.ts @@ -3,7 +3,7 @@ import * as os from "node:os"; import * as path from "node:path"; import * as url from "node:url"; import { isEnoent } from "@oh-my-pi/pi-utils"; -import { InternalUrlRouter } from "../internal-urls"; +import { InternalUrlRouter, type LocalProtocolOptions } from "../internal-urls"; import { ToolError } from "./tool-errors"; const UNICODE_SPACES = /[\u00A0\u2000-\u200A\u202F\u205F\u3000]/g; @@ -740,6 +740,12 @@ export interface ToolScopeOptions { surfaceExactFilePaths?: boolean; /** Extra hint appended to "Path not found" when stat fails and the user supplied multiple paths. */ multipathStatHint?: string; + /** Calling session's settings — forwarded to the internal-URL router so caller-aware handlers (issue://, pr://) honor it. */ + settings?: unknown; + /** Caller's abort signal — forwarded to the internal-URL router. */ + signal?: AbortSignal; + /** Calling session's `local://` root mapping — pins resolutions to the calling session. */ + localProtocolOptions?: LocalProtocolOptions; } export interface ToolScopeResolution { @@ -778,7 +784,12 @@ export async function resolveToolSearchScope(opts: ToolScopeOptions): Promise { cwd: this.session.cwd, settings: this.session.settings, signal, + localProtocolOptions: this.session.localProtocolOptions, }); const details: ReadToolDetails = { resolvedPath: resource.sourcePath, contentType: resource.contentType }; diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index 5679f8031..897902886 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -10,6 +10,7 @@ import { prompt, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; import { recordFileSnapshot } from "../edit/file-snapshot-store"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; +import type { LocalProtocolOptions } from "../internal-urls/local-protocol"; import { InternalUrlRouter } from "../internal-urls/router"; import type { InternalResource, ResolveContext } from "../internal-urls/types"; import type { Theme } from "../modes/theme/theme"; @@ -543,6 +544,7 @@ async function resolveInternalSearchInputs(opts: { settings: unknown; signal?: AbortSignal; archiveDisplayMap: ReadonlyMap; + localProtocolOptions?: LocalProtocolOptions; }): Promise { const internalRouter = InternalUrlRouter.instance(); const paths = opts.resolvedPaths.slice(); @@ -551,7 +553,12 @@ async function resolveInternalSearchInputs(opts: { const virtualInputIndexes = new Set(); const immutableSourcePaths = new Set(); let virtualScopePath: string | undefined; - const context: ResolveContext = { cwd: opts.cwd, settings: opts.settings, signal: opts.signal }; + const context: ResolveContext = { + cwd: opts.cwd, + settings: opts.settings, + signal: opts.signal, + localProtocolOptions: opts.localProtocolOptions, + }; for (let idx = 0; idx < paths.length; idx++) { const rawPath = paths[idx]; @@ -674,6 +681,7 @@ export class SearchTool implements AgentTool { await expect(router.resolve("local://linked/secret.txt")).rejects.toThrow("local:// URL escapes local root"); }); }); + + it("prefers caller-supplied context.localProtocolOptions over the installed override", async () => { + await withTempDir(async tempDir => { + const overrideArtifactsDir = path.join(tempDir, "override-artifacts"); + const callerArtifactsDir = path.join(tempDir, "caller-artifacts"); + await fs.mkdir(path.join(overrideArtifactsDir, "local"), { recursive: true }); + await fs.mkdir(path.join(callerArtifactsDir, "local"), { recursive: true }); + await Bun.write(path.join(overrideArtifactsDir, "local", "PLAN.md"), "# wrong session"); + await Bun.write(path.join(callerArtifactsDir, "local", "PLAN.md"), "# caller session"); + + // Process-global override points at the WRONG session (simulates a + // stale override leaked from a prior subagent, or the multi-`main` + // AgentRegistry case in cmux/ACP where "first one wins" lookup + // picks a sibling session's artifacts dir — issue #1608). + LocalProtocolHandler.setOverride({ + getArtifactsDir: () => overrideArtifactsDir, + getSessionId: () => "stale-session", + }); + + const router = InternalUrlRouter.instance(); + const resource = await router.resolve("local://PLAN.md", { + localProtocolOptions: { + getArtifactsDir: () => callerArtifactsDir, + getSessionId: () => "caller-session", + }, + }); + + const expectedSourcePath = await fs.realpath(path.join(callerArtifactsDir, "local", "PLAN.md")); + + expect(resource.content).toBe("# caller session"); + // `sourcePath` is canonicalized by the handler after symlink escape checks. + // On macOS this may turn `/var/...` into `/private/var/...`. + expect(resource.sourcePath).toBe(expectedSourcePath); + }); + }); + + it("surfaces ENOENT against the caller's local root when the file is missing in that session", async () => { + await withTempDir(async tempDir => { + const overrideArtifactsDir = path.join(tempDir, "override-artifacts"); + const callerArtifactsDir = path.join(tempDir, "caller-artifacts"); + await fs.mkdir(path.join(overrideArtifactsDir, "local"), { recursive: true }); + await fs.mkdir(path.join(callerArtifactsDir, "local"), { recursive: true }); + // PLAN.md exists only in the override-pointed session. + await Bun.write(path.join(overrideArtifactsDir, "local", "PLAN.md"), "# wrong session"); + + LocalProtocolHandler.setOverride({ + getArtifactsDir: () => overrideArtifactsDir, + getSessionId: () => "stale-session", + }); + + const router = InternalUrlRouter.instance(); + await expect( + router.resolve("local://PLAN.md", { + localProtocolOptions: { + getArtifactsDir: () => callerArtifactsDir, + getSessionId: () => "caller-session", + }, + }), + ).rejects.toThrow("Local file not found: local://PLAN.md"); + }); + }); }); From ca7b0b63db4f1e87bd8fcfd6acb87904e699f745 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 1 Jun 2026 11:35:13 +0000 Subject: [PATCH 20/28] fix(tui): suppressed native viewport probe under Windows Terminal When omp runs under Windows Terminal on native Windows, ConPTY routes the console through a pseudo-console whose GetConsoleScreenBufferInfo answer is always pinned to the buffer tail; it does not reflect the user's scroll position in the WT pane. The renderer's planRender used that answer to authorize the shrink-across-viewport historyRebuild intent, emitting the destructive \x1b[2J\x1b[H\x1b[3J on every full redraw and yanking a scrolled-up reader to the top of WT's scrollback. ProcessTerminal.isNativeViewportAtBottom now returns undefined whenever WT_SESSION is set, matching the POSIX fallback. The renderer's existing unknown-viewport deferral keeps streaming-time mutations non-destructive (viewportRepaint / deferredShrink) and reconciles native history at the next prompt-submit checkpoint via refreshNativeScrollbackIfDirty({ allowUnknownViewport: true }), where the user is provably at the bottom. The probe gate is factored into a pure shouldTrustNativeViewportProbe helper so the contract is unit-testable without spying on process state. Fixes #1635 --- packages/tui/CHANGELOG.md | 4 + packages/tui/src/terminal.ts | 32 ++++- packages/tui/test/issue-1635-repro.test.ts | 155 +++++++++++++++++++++ 3 files changed, 190 insertions(+), 1 deletion(-) create mode 100644 packages/tui/test/issue-1635-repro.test.ts diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 1b4e8b7a9..77859e999 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed native Windows + Windows Terminal scrollback being yanked to the top when a streaming response triggered a TUI full redraw. Under ConPTY the `kernel32` `GetConsoleScreenBufferInfo` probe answers about the pseudo-console (always at the buffer tail) and not about WT's host scrollback, so `isNativeViewportAtBottom()` falsely returned `true` while the user was scrolled up and the shrink-across-viewport branch issued a destructive `historyRebuild` (`\x1b[2J\x1b[H\x1b[3J`). The probe now short-circuits to `undefined` whenever `WT_SESSION` is set, letting the existing deferred-rebuild path keep streaming-time mutations non-destructive and reconcile native history at the next prompt-submit checkpoint. ([#1635](https://github.com/can1357/oh-my-pi/issues/1635)) + ## [15.7.3] - 2026-05-31 ### Added diff --git a/packages/tui/src/terminal.ts b/packages/tui/src/terminal.ts index 20a89d382..d1154cf58 100644 --- a/packages/tui/src/terminal.ts +++ b/packages/tui/src/terminal.ts @@ -113,6 +113,27 @@ function isWindowsSubsystemForLinux(): boolean { return process.platform === "linux" && (!!$env.WSL_DISTRO_NAME || !!$env.WSL_INTEROP); } +/** + * Whether the native console viewport-position probe should be consulted. + * + * Returns `true` only on native Windows that is *not* fronted by Windows + * Terminal. The kernel32 `GetConsoleScreenBufferInfo` API answers about the + * ConPTY pseudo-console — which is always pinned to its tail — and not about + * the user-visible scrollback in modern hosts. Treat any such host as + * unreportable so the renderer falls back to the deferred-rebuild path. + * + * Pure helper for unit testing; the runtime call site reads `$env` / + * `process.platform`. See #1635. + */ +export function shouldTrustNativeViewportProbe( + env: { WT_SESSION?: string | undefined } = $env, + platform: NodeJS.Platform = process.platform, +): boolean { + if (platform !== "win32") return false; + if (env.WT_SESSION) return false; + return true; +} + /** * Real terminal using process.stdin/stdout */ @@ -214,9 +235,18 @@ export class ProcessTerminal implements Terminal { /** * Returns true when Windows' active console viewport is at the scrollback tail. * POSIX terminals do not expose native scrollback position through a standard API. + * + * On native Windows running under Windows Terminal (the default modern + * host), the `kernel32` probe answers about the ConPTY pseudo-console — not + * the user-visible WT viewport — so it would always read "at bottom" while + * the user is scrolled up. Return `undefined` there so the renderer falls + * back to the POSIX-style deferred-rebuild path: streaming mutations stay + * non-destructive (no `\x1b[3J`), and the rebuild fires at the next prompt + * checkpoint via {@link TUI.refreshNativeScrollbackIfDirty} where the user + * is already pinned to the bottom by the editor keystroke. See #1635. */ isNativeViewportAtBottom(): boolean | undefined { - if (process.platform !== "win32") return undefined; + if (!shouldTrustNativeViewportProbe()) return undefined; try { const kernel32 = dlopen("kernel32.dll", { GetStdHandle: { args: [FFIType.i32], returns: FFIType.ptr }, diff --git a/packages/tui/test/issue-1635-repro.test.ts b/packages/tui/test/issue-1635-repro.test.ts new file mode 100644 index 000000000..39c21cf2e --- /dev/null +++ b/packages/tui/test/issue-1635-repro.test.ts @@ -0,0 +1,155 @@ +import { describe, expect, it } from "bun:test"; +import { type Component, TUI } from "@oh-my-pi/pi-tui"; +import { shouldTrustNativeViewportProbe } from "@oh-my-pi/pi-tui/terminal"; +import { VirtualTerminal } from "./virtual-terminal"; + +// Regression test for https://github.com/can1357/oh-my-pi/issues/1635 +// +// Native Windows + Windows Terminal (ConPTY) routes `omp` through a +// pseudo-console whose `GetConsoleScreenBufferInfo` answer always reports +// "viewport at bottom" — it cannot see the WT host scrollback. When the user +// scrolled up in WT and the renderer hit a `historyRebuild` intent (the +// shrink-across-viewport branch), the destructive `\x1b[2J\x1b[H\x1b[3J` +// sequence reset the WT viewport to the top of scrollback. +// +// Fix: `shouldTrustNativeViewportProbe` returns false under WT_SESSION so the +// probe falls back to `undefined`, and the renderer's existing +// deferred-rebuild path keeps streaming-time mutations non-destructive. +// +// The renderer assertions below override the VirtualTerminal probe to simulate +// the two relevant post-fix outcomes: +// +// - `undefined`: probe is unreportable (WT-hosted on win32, or any POSIX +// host where the probe never had an answer to begin with). +// - `false`: the host can see scrollback and reports the user scrolled +// up. Both must avoid `\x1b[3J`. +class LineList implements Component { + #lines: string[]; + constructor(lines: string[]) { + this.#lines = [...lines]; + } + invalidate(): void {} + render(width: number): string[] { + return this.#lines.map(l => l.slice(0, width)); + } + setLines(lines: string[]): void { + this.#lines = [...lines]; + } +} + +async function settle(term: VirtualTerminal): Promise { + await new Promise(r => process.nextTick(r)); + await new Promise(r => setTimeout(r, 20)); + await term.flush(); +} + +function capture(term: VirtualTerminal): string[] { + const writes: string[] = []; + const realWrite = term.write.bind(term); + (term as unknown as { write: (s: string) => void }).write = (data: string) => { + writes.push(data); + realWrite(data); + }; + return writes; +} + +function overrideProbe(term: VirtualTerminal, answer: boolean | undefined): void { + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => answer; +} + +const ERASE_SCROLLBACK = /\x1b\[3J/g; + +describe("issue #1635: shouldTrustNativeViewportProbe", () => { + it("returns true on bare native Windows (legacy console)", () => { + expect(shouldTrustNativeViewportProbe({}, "win32")).toBe(true); + }); + + it("returns false when running under Windows Terminal", () => { + expect(shouldTrustNativeViewportProbe({ WT_SESSION: "abcd-efgh" }, "win32")).toBe(false); + }); + + it("returns false on POSIX where the probe has no answer", () => { + expect(shouldTrustNativeViewportProbe({}, "linux")).toBe(false); + expect(shouldTrustNativeViewportProbe({}, "darwin")).toBe(false); + }); + + it("returns false on POSIX even if WT_SESSION leaked through (defense in depth)", () => { + expect(shouldTrustNativeViewportProbe({ WT_SESSION: "x" }, "linux")).toBe(false); + }); +}); + +describe("issue #1635: TUI must not emit \\x1b[3J when probe is unreliable", () => { + it("content shrink with unreportable viewport must not emit \\x1b[3J", async () => { + const term = new VirtualTerminal(100, 24); + overrideProbe(term, undefined); + const tui = new TUI(term); + const component = new LineList(Array.from({ length: 80 }, (_, i) => `init-${i}`)); + tui.addChild(component); + try { + tui.start(); + await settle(term); + const writes = capture(term); + component.setLines(Array.from({ length: 20 }, (_, i) => `shrunk-${i}`)); + tui.requestRender(); + await settle(term); + expect(writes.join("").match(ERASE_SCROLLBACK)).toBeNull(); + } finally { + tui.stop(); + } + }); + + it("content shrink with scrolled-up viewport must not emit \\x1b[3J", async () => { + const term = new VirtualTerminal(100, 24); + overrideProbe(term, false); + const tui = new TUI(term); + const component = new LineList(Array.from({ length: 80 }, (_, i) => `init-${i}`)); + tui.addChild(component); + try { + tui.start(); + await settle(term); + const writes = capture(term); + component.setLines(Array.from({ length: 20 }, (_, i) => `shrunk-${i}`)); + tui.requestRender(); + await settle(term); + expect(writes.join("").match(ERASE_SCROLLBACK)).toBeNull(); + } finally { + tui.stop(); + } + }); + + it("height change with unreportable viewport must not emit \\x1b[3J", async () => { + const term = new VirtualTerminal(100, 24); + overrideProbe(term, undefined); + const tui = new TUI(term); + const component = new LineList(Array.from({ length: 40 }, (_, i) => `init-${i}`)); + tui.addChild(component); + try { + tui.start(); + await settle(term); + const writes = capture(term); + term.resize(100, 25); + await settle(term); + expect(writes.join("").match(ERASE_SCROLLBACK)).toBeNull(); + } finally { + tui.stop(); + } + }); + + it("width change with unreportable viewport must not emit \\x1b[3J", async () => { + const term = new VirtualTerminal(100, 24); + overrideProbe(term, undefined); + const tui = new TUI(term); + const component = new LineList(Array.from({ length: 40 }, (_, i) => `init-${i}`)); + tui.addChild(component); + try { + tui.start(); + await settle(term); + const writes = capture(term); + term.resize(99, 24); + await settle(term); + expect(writes.join("").match(ERASE_SCROLLBACK)).toBeNull(); + } finally { + tui.stop(); + } + }); +}); From 8a41264745531b0bf14bd9b94a9d32dff2f4ced8 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 14:35:43 +0200 Subject: [PATCH 21/28] feat(ai): added Anthropic task budget support via output_config - Added `TokenTaskBudget` type and `taskBudget` option to `StreamOptions`. - Forwarded `taskBudget` as `output_config.task_budget` with the `task-budgets-2026-03-13` beta header. - Fixed `disableThinkingIfToolChoiceForced` to preserve `task_budget` when clearing `effort`. - Accepted `output_config.task_budget` from Anthropic gateway requests. --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/auth-gateway/server.ts | 1 + packages/ai/src/auth-gateway/types.ts | 11 ++- .../anthropic-messages-server-schema.ts | 13 +++ .../providers/anthropic-messages-server.ts | 3 + packages/ai/src/providers/anthropic.ts | 32 ++++++- packages/ai/src/stream.ts | 1 + packages/ai/src/types.ts | 11 +++ packages/ai/test/anthropic-alignment.test.ts | 83 ++++++++++++++++++- .../auth-gateway-anthropic-messages.test.ts | 2 + .../src/prompts/system/system-prompt.md | 2 +- 11 files changed, 156 insertions(+), 7 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f80bf3d04..97b8e7914 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added Anthropic task budget support, forwarding `taskBudget` as `output_config.task_budget` with the required `task-budgets-2026-03-13` beta header and accepting Anthropic gateway requests that send `output_config.task_budget`. + ## [15.7.4] - 2026-05-31 ### Fixed diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index db2ed193d..dad2aa394 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -141,6 +141,7 @@ function buildStreamOptions(parsed: ParsedFormatRequest, api: Api, signal: Abort if (options.reasoning !== undefined) opts.reasoning = options.reasoning; if (options.disableReasoning !== undefined) opts.disableReasoning = options.disableReasoning; if (options.hideThinkingSummary !== undefined) opts.hideThinkingSummary = options.hideThinkingSummary; + if (options.taskBudget !== undefined) opts.taskBudget = options.taskBudget; if (options.serviceTier !== undefined) opts.serviceTier = options.serviceTier; if (options.cacheRetention !== undefined) opts.cacheRetention = options.cacheRetention; // Client-supplied `prompt_cache_key` wins; otherwise derive a stable diff --git a/packages/ai/src/auth-gateway/types.ts b/packages/ai/src/auth-gateway/types.ts index 7d9b6fefe..0390759c0 100644 --- a/packages/ai/src/auth-gateway/types.ts +++ b/packages/ai/src/auth-gateway/types.ts @@ -1,5 +1,12 @@ import type { Effort } from "../model-thinking"; -import type { AssistantMessage, AssistantMessageEventStream, CacheRetention, Context, ServiceTier } from "../types"; +import type { + AssistantMessage, + AssistantMessageEventStream, + CacheRetention, + Context, + ServiceTier, + TokenTaskBudget, +} from "../types"; /** * Wire types for the omp auth-gateway. @@ -61,6 +68,8 @@ export interface AuthGatewayParsedRequestOptions { thinkingBudgets?: Partial>; /** Suppress the provider's reasoning summary stream. */ hideThinkingSummary?: boolean; + /** Anthropic `output_config.task_budget` advisory loop budget. */ + taskBudget?: TokenTaskBudget; // ── Service / routing ───────────────────────────────────────────────── /** OpenAI service tier (auto|default|flex|scale|priority). */ diff --git a/packages/ai/src/providers/anthropic-messages-server-schema.ts b/packages/ai/src/providers/anthropic-messages-server-schema.ts index 09ab75283..1ba88d3b2 100644 --- a/packages/ai/src/providers/anthropic-messages-server-schema.ts +++ b/packages/ai/src/providers/anthropic-messages-server-schema.ts @@ -189,6 +189,18 @@ export const thinkingConfigSchema = z.discriminatedUnion("type", [ }), ]); +const taskBudgetSchema = z.object({ + type: z.literal("tokens"), + total: z.number(), + remaining: z.number().optional(), +}); + +const outputConfigSchema = z.object({ + effort: z.enum(["low", "medium", "high", "xhigh", "max"]).optional(), + task_budget: taskBudgetSchema.optional(), + format: z.unknown().optional(), +}); + // ─── Top-level request ───────────────────────────────────────────────────── export const anthropicMessagesRequestSchema = z.object({ @@ -204,6 +216,7 @@ export const anthropicMessagesRequestSchema = z.object({ stop_sequences: z.array(z.string()).optional(), stream: z.boolean().optional(), thinking: thinkingConfigSchema.optional(), + output_config: outputConfigSchema.optional(), // Anthropic clients commonly send `metadata: { user_id }`; the walker // surfaces it on `options.metadata` for downstream provider forwarding. metadata: z.record(z.string(), z.unknown()).optional(), diff --git a/packages/ai/src/providers/anthropic-messages-server.ts b/packages/ai/src/providers/anthropic-messages-server.ts index e0aeb19be..a84c5a92b 100644 --- a/packages/ai/src/providers/anthropic-messages-server.ts +++ b/packages/ai/src/providers/anthropic-messages-server.ts @@ -344,6 +344,9 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest { break; } } + if (data.output_config?.task_budget) { + options.taskBudget = data.output_config.task_budget; + } const cacheRetention = deriveCacheRetention(data); if (cacheRetention !== undefined) options.cacheRetention = cacheRetention; // Anthropic clients commonly send `metadata: { user_id }`; forward verbatim diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 58adfc9a1..de18d14c7 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -47,6 +47,7 @@ import type { StreamOptions, TextContent, ThinkingContent, + TokenTaskBudget, Tool, ToolCall, ToolResultMessage, @@ -123,6 +124,7 @@ const claudeCodeBetaDefaults = [ const fineGrainedToolStreamingBeta = "fine-grained-tool-streaming-2025-05-14"; const interleavedThinkingBeta = "interleaved-thinking-2025-05-14"; const fastModeBeta = "fast-mode-2026-02-01"; +const taskBudgetBeta = "task-budgets-2026-03-13"; function getHeaderCaseInsensitive(headers: Record | undefined, headerName: string): string | undefined { if (!headers) return undefined; @@ -217,6 +219,16 @@ type AnthropicSamplingParams = MessageCreateParamsStreaming & { top_k?: number; }; +type AnthropicOutputConfig = NonNullable & { + task_budget?: TokenTaskBudget | null; +}; + +function getAnthropicOutputConfig(params: MessageCreateParamsStreaming): AnthropicOutputConfig { + const outputConfig = (params.output_config ?? {}) as AnthropicOutputConfig; + params.output_config = outputConfig as typeof params.output_config; + return outputConfig; +} + const ANTHROPIC_STOP_SEQUENCES_MAX = 4; let warnedStopSequencesTrim = false; @@ -1150,6 +1162,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = ( if (wantsAnthropicPriority && !extraBetas.includes(fastModeBeta)) { extraBetas.push(fastModeBeta); } + if (options?.taskBudget && !extraBetas.includes(taskBudgetBeta)) { + extraBetas.push(taskBudgetBeta); + } const created = createClient(model, { model, @@ -1779,8 +1794,14 @@ function createClient( function disableThinkingIfToolChoiceForced(params: MessageCreateParamsStreaming): void { const toolChoice = params.tool_choice; if (!toolChoice) return; - if (toolChoice.type === "any" || toolChoice.type === "tool") { - delete params.thinking; + if (toolChoice.type !== "any" && toolChoice.type !== "tool") return; + + delete params.thinking; + const outputConfig = params.output_config as AnthropicOutputConfig | undefined; + if (!outputConfig) return; + + delete outputConfig.effort; + if (Object.keys(outputConfig).length === 0) { delete params.output_config; } } @@ -2107,7 +2128,7 @@ function buildParams( if (effort) { // SDK's OutputConfig.effort type is not yet widened to include the new "xhigh" // level introduced with Claude Opus 4.7. Cast until the SDK catches up. - params.output_config = { effort } as typeof params.output_config; + getAnthropicOutputConfig(params).effort = effort; } } else { params.thinking = { @@ -2116,7 +2137,7 @@ function buildParams( display: options.thinkingDisplay ?? "summarized", } as typeof params.thinking; if (mode === "anthropic-budget-effort" && effort) { - params.output_config = { effort } as typeof params.output_config; + getAnthropicOutputConfig(params).effort = effort; } } } else if (options?.thinkingEnabled === false) { @@ -2124,6 +2145,9 @@ function buildParams( } } + if (options?.taskBudget) { + getAnthropicOutputConfig(params).task_budget = options.taskBudget; + } const metadataUserId = resolveAnthropicMetadataUserId(options?.metadata?.user_id, isOAuthToken); if (metadataUserId) { params.metadata = { user_id: metadataUserId }; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 9da81f33f..075808cb6 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -727,6 +727,7 @@ function mapOptionsForApi( initiatorOverride: options?.initiatorOverride, maxRetryDelayMs: options?.maxRetryDelayMs, metadata: options?.metadata, + taskBudget: options?.taskBudget, sessionId: options?.sessionId, promptCacheKey: options?.promptCacheKey, streamFirstEventTimeoutMs: options?.streamFirstEventTimeoutMs, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 83b754ceb..1f6975cc1 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -153,6 +153,12 @@ import type { Effort } from "./model-thinking"; /** Token budgets for each thinking level (token-based providers only) */ export type ThinkingBudgets = { [key in Effort]?: number }; +export interface TokenTaskBudget { + type: "tokens"; + total: number; + remaining?: number; +} + export type MessageAttribution = "user" | "agent"; export type ToolChoice = @@ -319,6 +325,11 @@ export interface StreamOptions { * For example, Anthropic uses `user_id` for abuse tracking and rate limiting. */ metadata?: Record; + /** + * Advisory token budget for a full agentic loop. Anthropic encodes this as + * `output_config.task_budget` with the `task-budgets-2026-03-13` beta header. + */ + taskBudget?: TokenTaskBudget; /** * Optional session identifier for providers that support session-based * routing, request affinity, or transport reuse. Providers may also use this diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index 925d8b72a..329e776be 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -18,7 +18,7 @@ import { stripClaudeToolPrefix, } from "@oh-my-pi/pi-ai/providers/anthropic"; import { getEnvApiKey } from "@oh-my-pi/pi-ai/stream"; -import type { Context, Model, TJsonSchema, Tool } from "@oh-my-pi/pi-ai/types"; +import type { Context, Model, TJsonSchema, TokenTaskBudget, Tool } from "@oh-my-pi/pi-ai/types"; import * as z from "zod/v4"; import { withEnv } from "./helpers"; @@ -57,6 +57,8 @@ type CaptureAnthropicOptions = { temperature?: number; topP?: number; topK?: number; + taskBudget?: TokenTaskBudget; + toolChoice?: "auto" | "any" | "none" | { type: "tool"; name: string }; }; function captureAnthropicPayload( @@ -75,6 +77,8 @@ function captureAnthropicPayload( temperature: options?.temperature, topP: options?.topP, topK: options?.topK, + taskBudget: options?.taskBudget, + toolChoice: options?.toolChoice, onPayload: payload => resolve(payload), }); return promise; @@ -1047,6 +1051,83 @@ describe("Anthropic request fingerprint alignment", () => { expect(payload.output_config).toEqual({ effort: "xhigh" }); }); + it("sends task budgets through Anthropic output_config without dropping adaptive effort", async () => { + const payload = (await captureAnthropicPayload( + { + ...ANTHROPIC_MODEL, + id: "claude-opus-4-7", + name: "Claude Opus 4.7", + thinking: { + mode: "anthropic-adaptive", + minLevel: Effort.Minimal, + maxLevel: Effort.XHigh, + }, + }, + { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Review this repo", timestamp: Date.now() }], + }, + { + thinkingEnabled: true, + reasoning: Effort.High, + taskBudget: { type: "tokens", total: 64_000, remaining: 48_000 }, + }, + )) as { + output_config?: { + effort?: string; + task_budget?: TokenTaskBudget; + }; + }; + + expect(payload.output_config).toEqual({ + effort: "xhigh", + task_budget: { type: "tokens", total: 64_000, remaining: 48_000 }, + }); + }); + + it("preserves task budget when forced tool choice disables thinking", async () => { + const payload = (await captureAnthropicPayload( + { + ...ANTHROPIC_MODEL, + id: "claude-opus-4-7", + name: "Claude Opus 4.7", + thinking: { + mode: "anthropic-adaptive", + minLevel: Effort.Minimal, + maxLevel: Effort.XHigh, + }, + }, + { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Use the tool", timestamp: Date.now() }], + tools: [ + { + name: "lookup", + description: "Lookup a value", + parameters: { type: "object", properties: {}, additionalProperties: false }, + }, + ], + }, + { + thinkingEnabled: true, + reasoning: Effort.High, + taskBudget: { type: "tokens", total: 64_000 }, + toolChoice: "any", + }, + )) as { + thinking?: unknown; + output_config?: { + effort?: string; + task_budget?: TokenTaskBudget; + }; + }; + + expect(payload.thinking).toBeUndefined(); + expect(payload.output_config).toEqual({ + task_budget: { type: "tokens", total: 64_000 }, + }); + }); + it("treats tool prefix helpers as no-ops when prefix is empty", () => { expect(applyClaudeToolPrefix("Read", "")).toBe("Read"); expect(stripClaudeToolPrefix("proxy_Read", "")).toBe("proxy_Read"); diff --git a/packages/ai/test/auth-gateway-anthropic-messages.test.ts b/packages/ai/test/auth-gateway-anthropic-messages.test.ts index deff20cea..98679149f 100644 --- a/packages/ai/test/auth-gateway-anthropic-messages.test.ts +++ b/packages/ai/test/auth-gateway-anthropic-messages.test.ts @@ -62,6 +62,7 @@ describe("anthropic-messages parseRequest", () => { stop_sequences: ["\n\n"], tool_choice: { type: "any" }, thinking: { type: "enabled", budget_tokens: 2048 }, + output_config: { task_budget: { type: "tokens", total: 64_000, remaining: 60_000 } }, system: [ { type: "text", text: "You are X" }, { type: "text", text: "Be brief." }, @@ -119,6 +120,7 @@ describe("anthropic-messages parseRequest", () => { expect(parsed.options.stopSequences).toEqual(["\n\n"]); expect(parsed.options.toolChoice).toBe("required"); expect(parsed.options.explicitThinkingBudgetTokens).toBe(2048); + expect(parsed.options.taskBudget).toEqual({ type: "tokens", total: 64_000, remaining: 60_000 }); expect(parsed.options.extra).toBeUndefined(); expect(parsed.context.tools).toHaveLength(1); diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 81743f905..a6f2375c4 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -9,7 +9,7 @@ You consider what the code you write compiles down to. You never write code that **RFC 2119 applies to MUST, REQUIRED, SHOULD, RECOMMENDED, MAY, OPTIONAL. `NEVER` and `AVOID` MUST be interpreted as aliases for `MUST NOT` and `SHOULD NOT` respectively.** -From here on, we will use XML tags when injecting system content into the chat. +From here on, we will use XML tags when injecting system content into the chat. You NEVER interpret these markers in any other way circumstantially. System may interrupt/notify you using these tags even within a user message, therefore: From e78b66fc6f5ff0a5265a7837dbebe4d3dbfe3ff6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 14:42:58 +0200 Subject: [PATCH 22/28] test(coding-agent): updated fixtures with missing ui and context mocks - Added `setEagerNativeScrollbackRebuild` mock to `ui` objects in test fixtures. - Added `pendingTools` map to context fixtures missing it. --- .../coding-agent/test/event-controller-abort-render.test.ts | 2 +- .../coding-agent/test/input-controller-skill-queue.test.ts | 3 ++- .../modes/controllers/event-controller-idle-compaction.test.ts | 2 +- .../modes/controllers/event-controller-message-start.test.ts | 3 ++- 4 files changed, 6 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/test/event-controller-abort-render.test.ts b/packages/coding-agent/test/event-controller-abort-render.test.ts index 01f9728a1..d73ecc34f 100644 --- a/packages/coding-agent/test/event-controller-abort-render.test.ts +++ b/packages/coding-agent/test/event-controller-abort-render.test.ts @@ -55,7 +55,7 @@ function createFixture(opts: { const ctx = { isInitialized: true, init: vi.fn(async () => {}), - ui: { requestRender }, + ui: { requestRender, setEagerNativeScrollbackRebuild: vi.fn() }, statusLine: { invalidate: vi.fn() }, updateEditorTopBorder: vi.fn(), streamingComponent, diff --git a/packages/coding-agent/test/input-controller-skill-queue.test.ts b/packages/coding-agent/test/input-controller-skill-queue.test.ts index 11d24d13b..1ec65d90b 100644 --- a/packages/coding-agent/test/input-controller-skill-queue.test.ts +++ b/packages/coding-agent/test/input-controller-skill-queue.test.ts @@ -463,11 +463,12 @@ function createEventControllerFixtureForE10() { const ctx = { isInitialized: true, init: vi.fn(async () => {}), - ui: { requestRender }, + ui: { requestRender, setEagerNativeScrollbackRebuild: vi.fn() }, statusLine: { invalidate: vi.fn() }, updateEditorTopBorder: vi.fn(), addMessageToChat, updatePendingMessagesDisplay, + pendingTools: new Map(), session: {}, } as unknown as InteractiveModeContext; diff --git a/packages/coding-agent/test/modes/controllers/event-controller-idle-compaction.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-idle-compaction.test.ts index 57d99a0e3..48f4e51b1 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-idle-compaction.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-idle-compaction.test.ts @@ -54,7 +54,7 @@ describe("EventController idle compaction teardown", () => { streamingMessage: undefined, pendingTools: new Map(), flushPendingModelSwitch: async () => {}, - ui: { requestRender: vi.fn() }, + ui: { requestRender: vi.fn(), setEagerNativeScrollbackRebuild: vi.fn() }, chatContainer: { removeChild: vi.fn() }, statusContainer: { clear: vi.fn() }, statusLine: { invalidate: vi.fn() }, diff --git a/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts index 50953cc61..41f09af0c 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts @@ -39,7 +39,7 @@ function createContext(options: { isInitialized: true, statusLine: { invalidate: vi.fn() }, updateEditorTopBorder: vi.fn(), - ui: { requestRender: vi.fn() }, + ui: { requestRender: vi.fn(), setEagerNativeScrollbackRebuild: vi.fn() }, editor, addMessageToChat, updatePendingMessagesDisplay, @@ -52,6 +52,7 @@ function createContext(options: { .join(""), optimisticUserMessageSignature: options.optimisticSignature, locallySubmittedUserSignatures: new Set(options.locallySubmittedSignatures ?? []), + pendingTools: new Map(), } as unknown as InteractiveModeContext; return { ctx, editor, setText, addMessageToChat, updatePendingMessagesDisplay }; } From ff2de07e7882ad3aaace3ddd86e3f5714237993f Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 14:43:12 +0200 Subject: [PATCH 23/28] test(coding-agent): added workspaceTree fixture to system prompt tests - Introduced helper `createEmptyWorkspaceTree` to reduce duplication. - Populated missing `workspaceTree` field in test contexts for `buildSystemPrompt`. --- .../test/system-prompt-templates.test.ts | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/packages/coding-agent/test/system-prompt-templates.test.ts b/packages/coding-agent/test/system-prompt-templates.test.ts index 0330bde2a..62cea3f61 100644 --- a/packages/coding-agent/test/system-prompt-templates.test.ts +++ b/packages/coding-agent/test/system-prompt-templates.test.ts @@ -104,6 +104,16 @@ async function withTempDir(run: (dir: string) => Promise): Promise { } } +function createEmptyWorkspaceTree(rootPath: string) { + return { + rootPath, + rendered: "", + truncated: false, + totalLines: 0, + agentsMdFiles: [], + }; +} + describe("system Handlebars prompt templates", () => { afterEach(() => { vi.restoreAllMocks(); @@ -212,6 +222,7 @@ describe("system Handlebars prompt templates", () => { skills: [], rules: [], toolNames: ["read"], + workspaceTree: createEmptyWorkspaceTree(os.tmpdir()), }; const enabled = await buildSystemPrompt({ @@ -300,6 +311,7 @@ describe("system Handlebars prompt templates", () => { skills: [], rules: [], toolNames: ["read"], + workspaceTree: createEmptyWorkspaceTree(dir), customPrompt: "Custom prompt body", alwaysApplyRules: [ { name: "no-dynamic-loading", content: duplicateRule, path: "/tmp/no-dynamic-loading.md" }, @@ -325,6 +337,7 @@ describe("system Handlebars prompt templates", () => { skills: [], rules: [], toolNames: ["read"], + workspaceTree: createEmptyWorkspaceTree(os.tmpdir()), customPrompt: ["Custom guidance", "", duplicateRule, "", "More custom guidance"].join("\n"), alwaysApplyRules: [ { name: "small-functions", content: duplicateRule, path: "/tmp/small-functions.md" }, @@ -361,6 +374,7 @@ describe("system Handlebars prompt templates", () => { skills: [], rules: [], toolNames: ["read", "search", "find", "edit", "lsp", "bash", "eval"], + workspaceTree: createEmptyWorkspaceTree(os.tmpdir()), tools: new Map([ ["read", { label: "Read", description: "Reads files" }], ["search", { label: "Search", description: "Searches files" }], @@ -390,6 +404,7 @@ describe("system Handlebars prompt templates", () => { skills: [], rules: [], toolNames: ["read"], + workspaceTree: createEmptyWorkspaceTree(os.tmpdir()), }); const projectPrompt = systemPrompt[1] ?? ""; From a814706740eb17f60b5c4d781457041c7545710e Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 15:03:42 +0200 Subject: [PATCH 24/28] fix(tui): avoided destructive history rebuild when transcript fits viewport - Added an early `viewportRepaint` path when `nativeViewportAtBottom` is unavailable and `newLines.length <= height`. - Marked native scrollback dirty before returning that repaint so cleanup is deferred to the next checkpoint. - Kept the `historyRebuild` fallback for shrinks beyond the padded viewport top to avoid unnecessary repainting. --- packages/tui/src/tui.ts | 42 +++++++++++++++++++++++++++-------------- 1 file changed, 28 insertions(+), 14 deletions(-) diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 477673c21..20e1ed346 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1335,22 +1335,36 @@ export class TUI extends Container { ) { return { kind: "historyRebuild" }; } - // POSIX terminals that cannot report viewport position fall through here - // (`canRebuildNativeScrollbackLive` is false). A viewport-only repaint would - // re-emit the rows between the new and old viewport tops on top of the copies - // the terminal already kept in native scrollback. `deferredShrink` pads to the - // previous row count so no committed row is re-emitted, and the next checkpoint - // rebuild (e.g. prompt submit -> `refreshNativeScrollbackIfDirty`) cleans up. + // POSIX terminals — and Windows Terminal/ConPTY — that cannot report the + // viewport position fall through here (`canRebuildNativeScrollbackLive` is + // false). A destructive rebuild emits `\x1b[3J`, which on modern terminals + // resets the viewport to the top of scrollback and yanks a scrolled-up + // reader (issue #1635), so it is unsafe while the probe is unavailable. + // + // When the shrunk transcript now fits entirely in the viewport there is no + // new native history to preserve during the live frame: repaint the screen + // in place (no `\x1b[3J`) and defer stale-scrollback cleanup to the next + // checkpoint rebuild (e.g. prompt submit -> `refreshNativeScrollbackIfDirty`). + if (nativeViewportAtBottom === undefined && newLines.length <= height) { + this.#markNativeScrollbackDirty(); + return { kind: "viewportRepaint" }; + } + // The shrunk transcript still overflows the viewport. A plain viewport + // repaint would re-emit the rows between the new and old viewport tops on top + // of the copies the terminal already kept in native scrollback; `deferredShrink` + // pads to the previous row count so no committed row is re-emitted, and the + // next checkpoint rebuild cleans up. // // That deferral only carries real content when `newLines.length` reaches the - // padded viewport top (`previousLines.length - height`) — otherwise every - // row the padded repaint draws is past the end of `newLines` and renders as - // blank, hiding the prompt until the next checkpoint. This can happen even - // when `scrollbackHighWater` is much lower than `previousLines.length - height`, - // because prior unknown-POSIX viewport repaints commit longer logical frames - // without moving the native scrollback boundary. For shrinks that large, - // yanking a scrolled reader (historyRebuild) is the lesser evil; do it - // unconditionally. + // padded viewport top (`previousLines.length - height`) — otherwise every row + // the padded repaint draws is past the end of `newLines` and renders blank, + // hiding the prompt until the next checkpoint. This can happen even when + // `scrollbackHighWater` is far below `previousLines.length - height`, because + // prior unknown-POSIX viewport repaints commit longer logical frames without + // moving the native scrollback boundary. For a shrink that large a blank, + // uninteractable viewport is the greater evil, so yank with `historyRebuild`. + // Real win32 unknown probes defer as scrolled above and never reach this; the + // yank only lands on non-win32 hosts whose probe is genuinely unavailable. const paddedViewportTop = Math.max(0, this.#previousLines.length - height); if (newLines.length <= paddedViewportTop) { return { kind: "historyRebuild" }; From da58949320e15333f7be692e2d78127a5be5e9b9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 15:04:12 +0200 Subject: [PATCH 25/28] chore: bump version to 15.7.5 --- Cargo.lock | 12 +++---- Cargo.toml | 2 +- bun.lock | 50 ++++++++++++--------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 18 +++++----- packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 ++ packages/ai/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 ++ packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 ++ packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- 20 files changed, 58 insertions(+), 56 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index a4fb87c92..35e5de7e9 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2331,7 +2331,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "15.7.4" +version = "15.7.5" dependencies = [ "anyhow", "ast-grep-core", @@ -2399,7 +2399,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "15.7.4" +version = "15.7.5" dependencies = [ "async-trait", "libc", @@ -2411,7 +2411,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "15.7.4" +version = "15.7.5" dependencies = [ "anyhow", "arboard", @@ -2457,7 +2457,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "15.7.4" +version = "15.7.5" dependencies = [ "anyhow", "brush-builtins", @@ -3886,9 +3886,9 @@ dependencies = [ [[package]] name = "tree-sitter-swift" -version = "0.7.2" +version = "0.7.3" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3b98fb6bc8e6a6a10023f401aa6a1858115e849dfaf7de57dd8b8ea0f257bd9" +checksum = "fe36052155b9dd69ca82b3b8f1b4ccfb2d867125ac1a4db1dd7331829242668c" dependencies = [ "cc", "tree-sitter-language", diff --git a/Cargo.toml b/Cargo.toml index 86fd4149c..205b38128 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "15.7.4" +version = "15.7.5" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index e44cc81b1..af42e9a10 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "15.7.4", + "version": "15.7.5", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -30,7 +30,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "15.7.4", + "version": "15.7.5", "dependencies": { "@anthropic-ai/sdk": "catalog:", "@bufbuild/protobuf": "catalog:", @@ -45,7 +45,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "15.7.4", + "version": "15.7.5", "bin": { "omp": "src/cli.ts", }, @@ -85,7 +85,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "15.7.4", + "version": "15.7.5", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -96,7 +96,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "15.7.4", + "version": "15.7.5", "bin": { "mnemopi": "src/cli.ts", }, @@ -113,7 +113,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "15.7.4", + "version": "15.7.5", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -121,7 +121,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "15.7.4", + "version": "15.7.5", "bin": { "omp-stats": "./src/index.ts", }, @@ -146,7 +146,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "15.7.4", + "version": "15.7.5", "bin": { "omp-swarm": "src/cli.ts", }, @@ -162,7 +162,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "15.7.4", + "version": "15.7.5", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -203,7 +203,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "15.7.4", + "version": "15.7.5", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "beautiful-mermaid": "catalog:", @@ -244,15 +244,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.7.4", - "@oh-my-pi/omp-stats": "15.7.4", - "@oh-my-pi/pi-agent-core": "15.7.4", - "@oh-my-pi/pi-ai": "15.7.4", - "@oh-my-pi/pi-coding-agent": "15.7.4", - "@oh-my-pi/pi-mnemopi": "15.7.4", - "@oh-my-pi/pi-natives": "15.7.4", - "@oh-my-pi/pi-tui": "15.7.4", - "@oh-my-pi/pi-utils": "15.7.4", + "@oh-my-pi/hashline": "15.7.5", + "@oh-my-pi/omp-stats": "15.7.5", + "@oh-my-pi/pi-agent-core": "15.7.5", + "@oh-my-pi/pi-ai": "15.7.5", + "@oh-my-pi/pi-coding-agent": "15.7.5", + "@oh-my-pi/pi-mnemopi": "15.7.5", + "@oh-my-pi/pi-natives": "15.7.5", + "@oh-my-pi/pi-tui": "15.7.5", + "@oh-my-pi/pi-utils": "15.7.5", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/sdk-trace-base": "^2.7.1", @@ -911,7 +911,7 @@ "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], - "electron-to-chromium": ["electron-to-chromium@1.5.361", "", {}, "sha512-Q6Hts7N9FnJc5LeGRINFvLhCI9xZmNtTDe5ZbcVezQz7cU4a8Aua3GH1b8J2XY8Al9PF+OCwYqhgsOOheMdvkA=="], + "electron-to-chromium": ["electron-to-chromium@1.5.364", "", {}, "sha512-G/dYE3+AYhyHwzTwg8UbnXf7zqMERYh7l2jJ3QujhFsH8agSYwtnGAR2aZ7f0AakIKJXd5En/Hre4igIUrdlYw=="], "elkjs": ["elkjs@0.11.1", "", {}, "sha512-zxxR9k+rx5ktMwT/FwyLdPCrq7xN6e4VGGHH8hA01vVYKjTFik7nHOxBnAYtrgYUB1RpAiLvA1/U2YraWxyKKg=="], @@ -921,7 +921,7 @@ "enabled": ["enabled@2.0.0", "", {}, "sha512-AKrN98kuwOzMIdAizXGI86UFBoo26CL21UM763y1h/GMSJ4/OHU9k2YlsmBpyScFo/wbLzWQJBMCW4+IO3/+OQ=="], - "enhanced-resolve": ["enhanced-resolve@5.22.0", "", { "dependencies": { "graceful-fs": "^4.2.4", "tapable": "^2.3.3" } }, "sha512-xYcDWrpELkFzz9SpZ3PlI6Eu6eD93Yf0WLDRxikGhWJ3MAir2SNZTIVCVZqZ/NUyx8AdMc2gT9C0gPiw18kG+A=="], + "enhanced-resolve": ["enhanced-resolve@5.22.1", "", { "dependencies": { "graceful-fs": "^4.2.4", "tapable": "^2.3.3" } }, "sha512-6QEuw3zoX1SJQc7b87aBXke/no+mG2bTBgw29gWMQonLmpEkWoCAVkl+M49e48AZlWzxiDzDZzYdp6kobcyLww=="], "entities": ["entities@7.0.1", "", {}, "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA=="], @@ -1145,7 +1145,7 @@ "onnxruntime-web": ["onnxruntime-web@1.26.0-dev.20260416-b7804b056c", "", { "dependencies": { "flatbuffers": "^25.1.24", "guid-typescript": "^1.0.9", "long": "^5.2.3", "onnxruntime-common": "1.24.0-dev.20251116-b39e144322", "platform": "^1.3.6", "protobufjs": "^7.2.4" } }, "sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw=="], - "openai": ["openai@6.39.0", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"], "bin": { "openai": "bin/cli" } }, "sha512-O61LIsimY3acVabwvomwFhwrnN36yvHY2quIfy9keEcFytGgWeV35yLHQ6NVMLSBxRpHmcg2yuhCnlu2HT4pLQ=="], + "openai": ["openai@6.39.1", "", { "peerDependencies": { "ws": "^8.18.0", "zod": "^3.25 || ^4.0" }, "optionalPeers": ["ws", "zod"], "bin": { "openai": "bin/cli" } }, "sha512-z3dO9fEWOXBzlXynVb/xZ/tujzUjFWQWn3C0n0mw6Vo0zJTbEkaN4b2cLWjhJ6haJQx8LlREoafHRl+Gu/Hl+A=="], "option": ["option@0.2.4", "", {}, "sha512-pkEqbDyl8ou5cpq+VsnQbe/WlEy5qS7xPzMS1U55OCG9KPvwFD46zDbxQIj3egJSFc3D+XhYOPUzz49zQAVy7A=="], @@ -1247,7 +1247,7 @@ "string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="], - "string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="], + "string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], "strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="], @@ -1403,8 +1403,6 @@ "string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="], - "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], - "wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="], "xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="], @@ -1419,8 +1417,6 @@ "fastembed/onnxruntime-node/tar": ["tar@7.5.15", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-dzGK0boVlC4W5QFuQN1EFSl3bIDYsk7Tj40U6eIBnK2k/8ml7TZ5agbI5j5+qnoVcAA+rNtBml8SEiLxZpNqRQ=="], - "jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], - "log-update/slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], "log-update/wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index ded216d53..6d06613a8 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -68,5 +68,5 @@ use napi_derive::napi; /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV15_7_4")] +#[napi(js_name = "__piNativesV15_7_5")] pub const fn pi_natives_version_sentinel() {} diff --git a/package.json b/package.json index b9596e4c1..53a47408e 100644 --- a/package.json +++ b/package.json @@ -21,15 +21,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.7.4", - "@oh-my-pi/omp-stats": "15.7.4", - "@oh-my-pi/pi-agent-core": "15.7.4", - "@oh-my-pi/pi-ai": "15.7.4", - "@oh-my-pi/pi-coding-agent": "15.7.4", - "@oh-my-pi/pi-mnemopi": "15.7.4", - "@oh-my-pi/pi-natives": "15.7.4", - "@oh-my-pi/pi-tui": "15.7.4", - "@oh-my-pi/pi-utils": "15.7.4", + "@oh-my-pi/hashline": "15.7.5", + "@oh-my-pi/omp-stats": "15.7.5", + "@oh-my-pi/pi-agent-core": "15.7.5", + "@oh-my-pi/pi-ai": "15.7.5", + "@oh-my-pi/pi-coding-agent": "15.7.5", + "@oh-my-pi/pi-mnemopi": "15.7.5", + "@oh-my-pi/pi-natives": "15.7.5", + "@oh-my-pi/pi-tui": "15.7.5", + "@oh-my-pi/pi-utils": "15.7.5", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/sdk-trace-base": "^2.7.1", diff --git a/packages/agent/package.json b/packages/agent/package.json index c340e74d0..fb602e79b 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "15.7.4", + "version": "15.7.5", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 02baafd0c..6f6c1cbda 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.7.5] - 2026-06-01 + ### Added - Added Anthropic task budget support, forwarding `taskBudget` as `output_config.task_budget` with the required `task-budgets-2026-03-13` beta header and accepting Anthropic gateway requests that send `output_config.task_budget`. diff --git a/packages/ai/package.json b/packages/ai/package.json index 67279f0e7..fed9ea14d 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "15.7.4", + "version": "15.7.5", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d5ec8fcf2..032130119 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,8 @@ # Changelog ## [Unreleased] + +## [15.7.5] - 2026-06-01 ### Fixed - Fixed streaming assistant responses leaving duplicated tail rows in WSL/Windows Terminal scrollback by enabling eager native-scrollback rebuilds while assistant text is actively streaming ([#1615](https://github.com/can1357/oh-my-pi/issues/1615)). diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 5b22849b7..c892cd134 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "15.7.4", + "version": "15.7.5", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 3c1d2e56c..484b70e3e 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "15.7.4", + "version": "15.7.5", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index b5d0857bf..8a6b98196 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "15.7.4", + "version": "15.7.5", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index e7ef01a17..abf9d6fee 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -136,7 +136,7 @@ export declare class Shell { * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV15_7_4(): void +export declare function __piNativesV15_7_5(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index 6d504f410..87f6ce97c 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -23,7 +23,7 @@ export const PtySession = nativeBindings.PtySession; export const Shell = nativeBindings.Shell; // functions -export const __piNativesV15_7_4 = nativeBindings.__piNativesV15_7_4; +export const __piNativesV15_7_5 = nativeBindings.__piNativesV15_7_5; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index bd1e330ff..810be34df 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "15.7.4", + "version": "15.7.5", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/stats/package.json b/packages/stats/package.json index 907f64a2a..d0f6796cc 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "15.7.4", + "version": "15.7.5", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 8a70d603d..179f9115e 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "15.7.4", + "version": "15.7.5", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 77859e999..d925bba5f 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.7.5] - 2026-06-01 + ### Fixed - Fixed native Windows + Windows Terminal scrollback being yanked to the top when a streaming response triggered a TUI full redraw. Under ConPTY the `kernel32` `GetConsoleScreenBufferInfo` probe answers about the pseudo-console (always at the buffer tail) and not about WT's host scrollback, so `isNativeViewportAtBottom()` falsely returned `true` while the user was scrolled up and the shrink-across-viewport branch issued a destructive `historyRebuild` (`\x1b[2J\x1b[H\x1b[3J`). The probe now short-circuits to `undefined` whenever `WT_SESSION` is set, letting the existing deferred-rebuild path keep streaming-time mutations non-destructive and reconcile native history at the next prompt-submit checkpoint. ([#1635](https://github.com/can1357/oh-my-pi/issues/1635)) diff --git a/packages/tui/package.json b/packages/tui/package.json index dd74aec8d..efe653b10 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "15.7.4", + "version": "15.7.5", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index bb8ef9d84..4a4457fcb 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "15.7.4", + "version": "15.7.5", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From f9d094de6051a52a02d818dc3e19c95d47225a5b Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 16:07:54 +0200 Subject: [PATCH 26/28] perf(coding-agent): aligned spinner and shimmer animations to the TUI 60fps render interval - Tool execution now renders every 16ms and advances spinner glyphs only every 80ms using a tracked last-advance timestamp. - Output block shimmer ticks were reduced to 16ms so border-frame updates align with the 60fps cadence for smoother animation. --- .../src/modes/components/tool-execution.ts | 24 +++++++++++++++---- packages/coding-agent/src/tui/output-block.ts | 9 +++---- 2 files changed, 25 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index a3ebc18ad..4e7f81587 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -141,6 +141,14 @@ export interface ToolExecutionHandle { setExpanded(expanded: boolean): void; } +/** Drive pending-tool redraws at ~60fps so the animated border sweep is smooth. + * The TUI already throttles at its 16ms `MIN_RENDER_INTERVAL_MS`, so this is the + * natural upper bound and static frames diff to a no-op redraw at ~zero cost. */ +const SPINNER_RENDER_INTERVAL_MS = 16; +/** Advance the spinner glyph at its classic ~12.5fps step, decoupled from the + * 60fps render cadence (mirrors `Loader`). */ +const SPINNER_GLYPH_ADVANCE_MS = 80; + /** * Component that renders a tool call with its result (updateable) */ @@ -177,6 +185,7 @@ export class ToolExecutionComponent extends Container { // Spinner animation for partial task results #spinnerFrame?: number; #spinnerInterval?: NodeJS.Timeout; + #lastSpinnerAdvanceAt = 0; // Todo write completion strikethrough reveal animation #todoStrikeInterval?: NodeJS.Timeout; // Track if args are still being streamed (for edit/write spinner) @@ -404,13 +413,20 @@ export class ToolExecutionComponent extends Container { this.#isPartial && shimmerEnabled() && (this.#toolName === "bash" || this.#toolName === "eval"); const needsSpinner = isStreamingArgs || isPartialTask || isPendingExecBlock; if (needsSpinner && !this.#spinnerInterval) { + this.#lastSpinnerAdvanceAt = performance.now(); this.#spinnerInterval = setInterval(() => { + const now = performance.now(); const frameCount = theme.spinnerFrames.length; - if (frameCount === 0) return; - this.#spinnerFrame = ((this.#spinnerFrame ?? -1) + 1) % frameCount; - this.#renderState.spinnerFrame = this.#spinnerFrame; + // Redraw at ~60fps for a smooth border sweep, but only step the spinner + // glyph at its classic ~12.5fps cadence. The TUI throttles renders at + // 16ms and the differ drops no-op redraws, so the extra ticks are free. + if (frameCount > 0 && now - this.#lastSpinnerAdvanceAt >= SPINNER_GLYPH_ADVANCE_MS) { + this.#spinnerFrame = ((this.#spinnerFrame ?? -1) + 1) % frameCount; + this.#renderState.spinnerFrame = this.#spinnerFrame; + this.#lastSpinnerAdvanceAt = now; + } this.#ui.requestRender(); - }, 80); + }, SPINNER_RENDER_INTERVAL_MS); } else if (!needsSpinner && this.#spinnerInterval) { clearInterval(this.#spinnerInterval); this.#spinnerInterval = undefined; diff --git a/packages/coding-agent/src/tui/output-block.ts b/packages/coding-agent/src/tui/output-block.ts index f6613bb52..1973a5d39 100644 --- a/packages/coding-agent/src/tui/output-block.ts +++ b/packages/coding-agent/src/tui/output-block.ts @@ -19,7 +19,7 @@ export interface OutputBlockOptions { animate?: boolean; } -const BORDER_SHIMMER_TICK_MS = 50; +const BORDER_SHIMMER_TICK_MS = 16; /** Duration of one full left↔right↔left bounce of the bottom-edge segment, in * ms. Position is derived from the wall clock against this fixed cycle so a * resize only nudges the segment proportionally instead of teleporting it. */ @@ -28,9 +28,10 @@ const BORDER_BOUNCE_MS = 3000; const BORDER_SEGMENT_LEN = 8; /** - * Monotonic frame counter for animated borders. Quantized coarse enough to - * coalesce multiple render passes inside one frame, fine enough to advance on - * every spinner interval so cached blocks re-render while the segment travels. + * Monotonic frame counter for animated borders, quantized to the TUI's ~16ms + * render cap so the cache key advances once per ~60fps frame — fine enough for a + * smooth segment sweep, coarse enough to coalesce multiple render passes that + * land inside the same frame. */ export function borderShimmerTick(): number { return Math.floor(Date.now() / BORDER_SHIMMER_TICK_MS); From 9b5677ea7a4a5c671157598add8ae1225b276de4 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 16:17:11 +0200 Subject: [PATCH 27/28] chore: bump models --- packages/ai/src/models.json | 887 ++++++++++++++++++++++++++++++++---- 1 file changed, 803 insertions(+), 84 deletions(-) diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index df3b89965..ea8da1a0d 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -970,8 +970,8 @@ "image" ], "cost": { - "input": 5, - "output": 25, + "input": 5.5, + "output": 27.5, "cacheRead": 0.5, "cacheWrite": 6.25 }, @@ -995,10 +995,10 @@ "image" ], "cost": { - "input": 5, - "output": 25, - "cacheRead": 0.5, - "cacheWrite": 6.25 + "input": 5.5, + "output": 27.5, + "cacheRead": 0.55, + "cacheWrite": 6.875 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -1020,10 +1020,10 @@ "image" ], "cost": { - "input": 5, - "output": 25, - "cacheRead": 0.5, - "cacheWrite": 6.25 + "input": 5.5, + "output": 27.5, + "cacheRead": 0.55, + "cacheWrite": 6.875 }, "contextWindow": 1000000, "maxTokens": 128000, @@ -1095,10 +1095,10 @@ "image" ], "cost": { - "input": 3, - "output": 15, - "cacheRead": 0.3, - "cacheWrite": 3.75 + "input": 3.3, + "output": 16.5, + "cacheRead": 0.33, + "cacheWrite": 4.125 }, "contextWindow": 1000000, "maxTokens": 64000, @@ -3642,6 +3642,31 @@ "maxLevel": "xhigh" } }, + "anthropic/claude-opus-4-8": { + "id": "anthropic/claude-opus-4-8", + "name": "Claude Opus 4.8", + "api": "anthropic-messages", + "provider": "cloudflare-ai-gateway", + "baseUrl": "https://gateway.ai.cloudflare.com/v1///anthropic", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 25, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "anthropic/claude-sonnet-4": { "id": "anthropic/claude-sonnet-4", "name": "Claude Sonnet 4 (latest)", @@ -10366,7 +10391,7 @@ }, "anthropic/claude-opus-4.8-fast": { "id": "anthropic/claude-opus-4.8-fast", - "name": "Anthropic: Claude Opus 4.8 (Fast)", + "name": "Anthropic: Claude Opus 4.8 (Fast) ($$$$)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -13094,6 +13119,25 @@ "maxLevel": "xhigh" } }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "MiniMax: MiniMax M3 (new)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "mistralai/codestral-2508": { "id": "mistralai/codestral-2508", "name": "Mistral: Codestral 2508", @@ -13911,7 +13955,7 @@ }, "nousresearch/hermes-2-pro-llama-3-8b": { "id": "nousresearch/hermes-2-pro-llama-3-8b", - "name": "NousResearch: Hermes 2 Pro - Llama-3 8B", + "name": "NousResearch: Hermes 2 Pro - Llama-3 8B (retires Jun 5)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -14306,7 +14350,7 @@ }, "openai/gpt-4-1106-preview": { "id": "openai/gpt-4-1106-preview", - "name": "OpenAI: GPT-4 Turbo (older v1106)", + "name": "OpenAI: GPT-4 Turbo (older v1106) ($$$$)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -14345,7 +14389,7 @@ }, "openai/gpt-4-turbo-preview": { "id": "openai/gpt-4-turbo-preview", - "name": "OpenAI: GPT-4 Turbo Preview", + "name": "OpenAI: GPT-4 Turbo Preview ($$$$)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -14682,7 +14726,7 @@ }, "openai/gpt-5-image": { "id": "openai/gpt-5-image", - "name": "OpenAI: GPT-5 Image", + "name": "OpenAI: GPT-5 Image ($$$$)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -14758,7 +14802,7 @@ }, "openai/gpt-5-pro": { "id": "openai/gpt-5-pro", - "name": "OpenAI: GPT-5 Pro", + "name": "OpenAI: GPT-5 Pro ($$$$)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -14916,7 +14960,7 @@ }, "openai/gpt-5.2-chat": { "id": "openai/gpt-5.2-chat", - "name": "OpenAI: GPT-5.2 Chat", + "name": "OpenAI: GPT-5.2 Chat (retires Aug 10)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -15403,7 +15447,7 @@ }, "openai/o3-deep-research": { "id": "openai/o3-deep-research", - "name": "OpenAI: o3 Deep Research", + "name": "OpenAI: o3 Deep Research ($$$$)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -16242,7 +16286,7 @@ }, "qwen/qwen3-30b-a3b": { "id": "qwen/qwen3-30b-a3b", - "name": "Qwen: Qwen3 30B A3B", + "name": "Qwen: Qwen3 30B A3B (retires Jun 5)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -17129,7 +17173,7 @@ }, "sao10k/l3-euryale-70b": { "id": "sao10k/l3-euryale-70b", - "name": "Sao10k: Llama 3 Euryale 70B v2.1", + "name": "Sao10k: Llama 3 Euryale 70B v2.1 (retires Jun 5)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -17260,6 +17304,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "stealth/claude-opus-4.8": { + "id": "stealth/claude-opus-4.8", + "name": "Stealth: Claude Opus 4.8 (20% off)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "stealth/claude-sonnet-4.6": { "id": "stealth/claude-sonnet-4.6", "name": "Stealth: Claude Sonnet 4.6 (20% off)", @@ -17279,6 +17342,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "stealth/qwen3.6-plus": { + "id": "stealth/qwen3.6-plus", + "name": "Stealth: Qwen3.6 Plus (50% off)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "stepfun/step-3.5-flash": { "id": "stepfun/step-3.5-flash", "name": "Step 3.5 Flash", @@ -17336,6 +17418,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "stepfun/step-3.7-flash:free": { + "id": "stepfun/step-3.7-flash:free", + "name": "StepFun: Step 3.7 Flash (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "switchpoint/router": { "id": "switchpoint/router", "name": "Switchpoint Router", @@ -19435,6 +19536,44 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "anthropic/claude-opus-4.8": { + "id": "anthropic/claude-opus-4.8", + "name": "anthropic/claude-opus-4.8", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "anthropic/claude-opus-4.8-fast": { + "id": "anthropic/claude-opus-4.8-fast", + "name": "anthropic/claude-opus-4.8-fast", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "anthropic/claude-opus-latest": { "id": "anthropic/claude-opus-latest", "name": "anthropic/claude-opus-latest", @@ -21195,6 +21334,31 @@ "maxLevel": "xhigh" } }, + "claude-opus-4-8": { + "id": "claude-opus-4-8", + "name": "Claude Opus 4.8", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "claude-opus-4-thinking": { "id": "claude-opus-4-thinking", "name": "claude-opus-4-thinking", @@ -22486,6 +22650,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "deepseek/deepseek-v4-flash:discounted": { + "id": "deepseek/deepseek-v4-flash:discounted", + "name": "deepseek/deepseek-v4-flash:discounted", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "deepseek/deepseek-v4-flash:free": { "id": "deepseek/deepseek-v4-flash:free", "name": "deepseek/deepseek-v4-flash:free", @@ -22567,6 +22750,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "deepseek/deepseek-v4-pro:discounted": { + "id": "deepseek/deepseek-v4-pro:discounted", + "name": "deepseek/deepseek-v4-pro:discounted", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "dmind/dmind-1": { "id": "dmind/dmind-1", "name": "dmind/dmind-1", @@ -27900,6 +28102,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "hf:Qwen/Qwen3.6-27B": { + "id": "hf:Qwen/Qwen3.6-27B", + "name": "hf:Qwen/Qwen3.6-27B", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "hf:zai-org/GLM-4.7": { "id": "hf:zai-org/GLM-4.7", "name": "hf:zai-org/GLM-4.7", @@ -32252,6 +32473,25 @@ "maxLevel": "xhigh" } }, + "moonshotai/kimi-k2.6:free": { + "id": "moonshotai/kimi-k2.6:free", + "name": "moonshotai/kimi-k2.6:free", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "moonshotai/kimi-latest": { "id": "moonshotai/kimi-latest", "name": "moonshotai/kimi-latest", @@ -38794,6 +39034,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "stepfun/step-3.7-flash": { + "id": "stepfun/step-3.7-flash", + "name": "stepfun/step-3.7-flash", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "study_gpt-chatgpt-4o-latest": { "id": "study_gpt-chatgpt-4o-latest", "name": "study_gpt-chatgpt-4o-latest", @@ -38832,6 +39091,82 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "syn:large:text": { + "id": "syn:large:text", + "name": "syn:large:text", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "syn:large:vision": { + "id": "syn:large:vision", + "name": "syn:large:vision", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "syn:small:text": { + "id": "syn:small:text", + "name": "syn:small:text", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, + "syn:small:vision": { + "id": "syn:small:vision", + "name": "syn:small:vision", + "api": "openai-completions", + "provider": "litellm", + "baseUrl": "http://localhost:4000/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "TEE/deepseek-r1-0528": { "id": "TEE/deepseek-r1-0528", "name": "TEE/deepseek-r1-0528", @@ -50241,6 +50576,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "minimax/minimax-m3", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "MiniMaxAI/MiniMax-M1-80k": { "id": "MiniMaxAI/MiniMax-M1-80k", "name": "MiniMaxAI/MiniMax-M1-80k", @@ -56081,6 +56435,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "Unbabel/M-Prometheus-14B": { + "id": "Unbabel/M-Prometheus-14B", + "name": "Unbabel/M-Prometheus-14B", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "undi95/remm-slerp-l2-13b": { "id": "undi95/remm-slerp-l2-13b", "name": "undi95/remm-slerp-l2-13b", @@ -59009,6 +59382,31 @@ "maxLevel": "xhigh" } }, + "stepfun-ai/step-3.7-flash": { + "id": "stepfun-ai/step-3.7-flash", + "name": "Step 3.7 Flash", + "api": "openai-completions", + "provider": "nvidia", + "baseUrl": "https://integrate.api.nvidia.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "upstage/solar-10_7b-instruct": { "id": "upstage/solar-10_7b-instruct", "name": "solar-10.7b-instruct", @@ -61119,6 +61517,31 @@ "maxLevel": "xhigh" } }, + "minimax-m3": { + "id": "minimax-m3", + "name": "MiniMax M3", + "api": "anthropic-messages", + "provider": "opencode-go", + "baseUrl": "https://opencode.ai/zen/go", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 2.4, + "cacheRead": 0.12, + "cacheWrite": 0 + }, + "contextWindow": 512000, + "maxTokens": 131072, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "qwen3.5-plus": { "id": "qwen3.5-plus", "name": "Qwen3.5 Plus", @@ -61464,6 +61887,30 @@ "maxLevel": "high" } }, + "deepseek-v4-flash": { + "id": "deepseek-v4-flash", + "name": "DeepSeek V4 Flash", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.28, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 384000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "deepseek-v4-flash-free": { "id": "deepseek-v4-flash-free", "name": "DeepSeek V4 Flash Free", @@ -62370,8 +62817,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 1000000, - "maxTokens": 128000, + "contextWindow": 200000, + "maxTokens": 32000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -62474,6 +62921,31 @@ "maxLevel": "xhigh" } }, + "minimax-m3-free": { + "id": "minimax-m3-free", + "name": "MiniMax M3 Free", + "api": "anthropic-messages", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 200000, + "maxTokens": 32000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "nemotron-3-super-free": { "id": "nemotron-3-super-free", "name": "Nemotron 3 Super Free", @@ -62754,13 +63226,13 @@ "image" ], "cost": { - "input": 0.73, - "output": 3.49, - "cacheRead": 0.25, + "input": 0.684, + "output": 3.42, + "cacheRead": 0.144, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262142, + "maxTokens": 262144, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -63829,8 +64301,8 @@ "text" ], "cost": { - "input": 0.2288, - "output": 0.9144, + "input": 0.20020000000000002, + "output": 0.8000999999999999, "cacheRead": 0.15, "cacheWrite": 0 }, @@ -63992,13 +64464,13 @@ "text" ], "cost": { - "input": 0.252, - "output": 0.378, + "input": 0.2288, + "output": 0.3432, "cacheRead": 0.0252, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 65536, + "maxTokens": 64000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -64040,13 +64512,13 @@ "text" ], "cost": { - "input": 0.09999999999999999, - "output": 0.19999999999999998, - "cacheRead": 0.02, + "input": 0.0983, + "output": 0.1966, + "cacheRead": 0.019700000000000002, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 16384, + "maxTokens": 131072, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -64845,9 +65317,9 @@ "text" ], "cost": { - "input": 0.075, - "output": 0.625, - "cacheRead": 0.015, + "input": 0.3, + "output": 2.5, + "cacheRead": 0.06, "cacheWrite": 0 }, "contextWindow": 262144, @@ -65227,13 +65699,38 @@ "text" ], "cost": { - "input": 0.27899999999999997, + "input": 0.26, "output": 1.2, "cacheRead": 0.059, "cacheWrite": 0 }, "contextWindow": 204800, - "maxTokens": 131072, + "maxTokens": 131070, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "MiniMax: MiniMax M3", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 512000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -65874,13 +66371,13 @@ "image" ], "cost": { - "input": 0.73, - "output": 3.49, - "cacheRead": 0.25, + "input": 0.684, + "output": 3.42, + "cacheRead": 0.144, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262142, + "maxTokens": 262144, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -67301,13 +67798,13 @@ "text" ], "cost": { - "input": 0.03, + "input": 0.029, "output": 0.14, "cacheRead": 0.015, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 131072, + "maxTokens": 65536, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -68062,13 +68559,13 @@ "text" ], "cost": { - "input": 0.14950000000000002, - "output": 1.495, - "cacheRead": 0.055, + "input": 0.09999999999999999, + "output": 0.09999999999999999, + "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 8888, + "maxTokens": 262144, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -68698,13 +69195,13 @@ "image" ], "cost": { - "input": 0.13899999999999998, + "input": 0.14, "output": 1, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 8888, + "maxTokens": 262144, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -69241,13 +69738,13 @@ "text" ], "cost": { - "input": 0.063, - "output": 0.21, - "cacheRead": 0.020999999999999998, + "input": 0.06599999999999999, + "output": 0.26, + "cacheRead": 0.029, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 64000, + "maxTokens": 262144, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -70068,12 +70565,12 @@ ], "cost": { "input": 0.6, - "output": 1.92, + "output": 2.08, "cacheRead": 0.12, "cacheWrite": 0 }, "contextWindow": 202752, - "maxTokens": 128000, + "maxTokens": 16384, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -71118,7 +71615,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 222222, + "contextWindow": 1000000, "maxTokens": 8888, "compat": { "supportsUsageInStreaming": false @@ -72334,6 +72831,34 @@ "supportsUsageInStreaming": false } }, + "minimax-m3": { + "id": "minimax-m3", + "name": "MiniMax M3", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 512000, + "maxTokens": 131072, + "compat": { + "supportsUsageInStreaming": false + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "mistral-31-24b": { "id": "mistral-31-24b", "name": "Venice Medium", @@ -73291,22 +73816,27 @@ }, "alibaba/qwen-3-235b": { "id": "alibaba/qwen-3-235b", - "name": "Qwen3 235B A22b Instruct 2507", + "name": "Qwen3 235B A22B", "api": "anthropic-messages", "baseUrl": "https://ai-gateway.vercel.sh", "provider": "vercel-ai-gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0.6, - "output": 1.2, + "input": 0.22, + "output": 0.88, "cacheRead": 0.6, "cacheWrite": 0 }, - "contextWindow": 131000, - "maxTokens": 40000 + "contextWindow": 262144, + "maxTokens": 16384, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "alibaba/qwen-3-30b": { "id": "alibaba/qwen-3-30b", @@ -73412,7 +73942,7 @@ "api": "anthropic-messages", "baseUrl": "https://ai-gateway.vercel.sh", "provider": "vercel-ai-gateway", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -73423,7 +73953,12 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536 + "maxTokens": 65536, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "alibaba/qwen3-coder-30b-a3b": { "id": "alibaba/qwen3-coder-30b-a3b", @@ -73554,6 +74089,49 @@ "maxLevel": "xhigh" } }, + "alibaba/qwen3-next-80b-a3b-instruct": { + "id": "alibaba/qwen3-next-80b-a3b-instruct", + "name": "Qwen3 Next 80B A3B Instruct", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 1.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768 + }, + "alibaba/qwen3-next-80b-a3b-thinking": { + "id": "alibaba/qwen3-next-80b-a3b-thinking", + "name": "Qwen3 Next 80B A3B Thinking", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 1.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "alibaba/qwen3-vl-thinking": { "id": "alibaba/qwen3-vl-thinking", "name": "Qwen3 VL 235B A22B Thinking", @@ -74175,17 +74753,17 @@ "text" ], "cost": { - "input": 0.77, - "output": 0.77, - "cacheRead": 0, + "input": 0.27, + "output": 1.12, + "cacheRead": 0.135, "cacheWrite": 0 }, "contextWindow": 163840, - "maxTokens": 16384 + "maxTokens": 163840 }, "deepseek/deepseek-v3.1": { "id": "deepseek/deepseek-v3.1", - "name": "DeepSeek-V3.1", + "name": "DeepSeek V3.1", "api": "anthropic-messages", "baseUrl": "https://ai-gateway.vercel.sh", "provider": "vercel-ai-gateway", @@ -74263,7 +74841,8 @@ "provider": "vercel-ai-gateway", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.62, @@ -75036,7 +75615,8 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.6, @@ -75100,6 +75680,31 @@ "maxLevel": "xhigh" } }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "MiniMax M3", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 1000000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "mistral/codestral": { "id": "mistral/codestral", "name": "Mistral Codestral", @@ -75258,6 +75863,25 @@ "maxLevel": "xhigh" } }, + "mistral/mistral-nemo": { + "id": "mistral/mistral-nemo", + "name": "Mistral Nemo 12B", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.02, + "output": 0.04, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 131072 + }, "mistral/mistral-small": { "id": "mistral/mistral-small", "name": "Mistral Small", @@ -75473,6 +76097,30 @@ "maxLevel": "xhigh" } }, + "nvidia/nemotron-3-super-120b-a12b": { + "id": "nvidia/nemotron-3-super-120b-a12b", + "name": "Nemotron 3 Super", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.15, + "output": 0.65, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 32000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "nvidia/nemotron-nano-12b-v2-vl": { "id": "nvidia/nemotron-nano-12b-v2-vl", "name": "Nvidia Nemotron Nano 12B V2 VL", @@ -76235,13 +76883,13 @@ "text" ], "cost": { - "input": 0.09999999999999999, - "output": 0.5, - "cacheRead": 0, + "input": 0.35, + "output": 0.75, + "cacheRead": 0.25, "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 65536, + "maxTokens": 131000, "thinking": { "mode": "budget", "minLevel": "minimal", @@ -76509,6 +77157,50 @@ "maxLevel": "xhigh" } }, + "stepfun/step-3.5-flash": { + "id": "stepfun/step-3.5-flash", + "name": "Step 3.5 Flash", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.09, + "output": 0.3, + "cacheRead": 0, + "cacheWrite": 0.02 + }, + "contextWindow": 262114, + "maxTokens": 262114 + }, + "stepfun/step-3.7-flash": { + "id": "stepfun/step-3.7-flash", + "name": "Step 3.7 Flash", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.19999999999999998, + "output": 1.15, + "cacheRead": 0.04, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 256000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "vercel/v0-1.0-md": { "id": "vercel/v0-1.0-md", "name": "v0-1.0-md", @@ -77346,7 +78038,8 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 1.4, @@ -80498,6 +81191,31 @@ "maxLevel": "xhigh" } }, + "minimax/minimax-m3": { + "id": "minimax/minimax-m3", + "name": "MiniMax: MiniMax M3", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 512000, + "maxTokens": 8888, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "mistralai/mistral-large-2512": { "id": "mistralai/mistral-large-2512", "name": "Mistral: Mistral Large 3", @@ -81888,7 +82606,8 @@ "baseUrl": "https://zenmux.ai/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.2, From e18e4ada717886edd8c109defc9e20cd345d50d9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 1 Jun 2026 16:38:32 +0200 Subject: [PATCH 28/28] fix(eval): keep idle watchdog armed during in-flight agent()/llm() calls The per-cell `timeout` is an inactivity budget that only re-arms on status events, but host-side bridge calls can run long stretches with no intermediate status (a subagent's time-to-first-token on a reasoning model, a long quiet nested tool, or an entire oneshot llm() request). The watchdog mistook that for a stall and aborted working subagents mid-flight. Pump a lightweight heartbeat while a bridge call awaits, re-arming the watchdog through the existing emitStatus -> onStatus channel. The heartbeat is a pure keepalive: forwarded to bump the timer but never stored or rendered, so a genuinely stalled cell is still interrupted once the call settles. - eval/heartbeat.ts: withBridgeHeartbeat() + EVAL_HEARTBEAT_OP - agent-bridge/llm-bridge: wrap runSubprocess / completeSimple - js+py executors: forward heartbeat to onStatus, drop from displayOutputs - tools/eval.ts: bump on heartbeat, skip persist/render --- packages/coding-agent/CHANGELOG.md | 4 + .../src/eval/__tests__/agent-bridge.test.ts | 31 +++++++ .../src/eval/__tests__/heartbeat.test.ts | 66 +++++++++++++++ .../src/eval/__tests__/llm-bridge.test.ts | 23 ++++++ .../coding-agent/src/eval/agent-bridge.ts | 82 ++++++++++--------- packages/coding-agent/src/eval/heartbeat.ts | 66 +++++++++++++++ packages/coding-agent/src/eval/js/executor.ts | 8 +- packages/coding-agent/src/eval/llm-bridge.ts | 34 ++++---- packages/coding-agent/src/eval/py/executor.ts | 8 +- packages/coding-agent/src/tools/eval.ts | 6 ++ 10 files changed, 274 insertions(+), 54 deletions(-) create mode 100644 packages/coding-agent/src/eval/__tests__/heartbeat.test.ts create mode 100644 packages/coding-agent/src/eval/heartbeat.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 032130119..ecaa918fc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the `eval` tool aborting in-flight `agent()`/`parallel()` subagents and `llm()` requests by mistaking them for a stalled cell. The per-cell `timeout` is an *inactivity* budget that only re-arms on status events, but a host-side bridge call can legitimately run long stretches with no intermediate status (a subagent's time-to-first-token on a reasoning model, a long quiet nested tool, or an entire oneshot `llm()` request). Those calls now pump a lightweight heartbeat while they await, re-arming the idle watchdog through the existing status channel; the heartbeat is a pure keepalive and is never persisted or rendered, so a genuinely stalled cell is still interrupted once the call settles. + ## [15.7.5] - 2026-06-01 ### Fixed diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 152ed3e32..92d13f9a3 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -10,6 +10,8 @@ import { AgentOutputManager } from "../../task/output-manager"; import type { AgentDefinition, AgentProgress, SingleResult } from "../../task/types"; import type { ToolSession } from "../../tools"; import { EVAL_AGENT_MAX_DEPTH, runEvalAgent } from "../agent-bridge"; +import { setBridgeHeartbeatIntervalMs } from "../heartbeat"; +import { IdleTimeout } from "../idle-timeout"; import { disposeAllVmContexts } from "../js/context-manager"; import { executeJs } from "../js/executor"; import { disposeAllKernelSessions, executePython } from "../py/executor"; @@ -232,6 +234,7 @@ describe("runEvalAgent", () => { describe("agent() through eval runtimes", () => { afterEach(() => { vi.restoreAllMocks(); + setBridgeHeartbeatIntervalMs(); }); afterAll(async () => { @@ -430,4 +433,32 @@ describe("agent() through eval runtimes", () => { ); expect(displayAgentEvents.length).toBe(2); }); + + it("keeps the idle watchdog armed while a quiet agent() runs past the budget", async () => { + using tempDir = TempDir.createSync("@omp-eval-agent-heartbeat-"); + const { session } = makeEvalSession(tempDir, "js-agent-heartbeat"); + mockAgents(); + // Heartbeat cadence well under the idle budget so a working-but-silent + // subagent re-arms the watchdog several times before it could expire. + setBridgeHeartbeatIntervalMs(15); + + // runSubprocess runs far past the budget and emits NO progress of its own + // — the only thing standing between the subagent and a spurious idle abort + // is the heartbeat keepalive the bridge pumps while it awaits. + vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => { + await Bun.sleep(200); + return singleResult(options, { output: "done" }); + }); + + // Mirror the eval tool's wiring: an IdleTimeout drives cancellation and + // every status event re-arms it. + using idle = new IdleTimeout(60); + const result = await runEvalAgent( + { prompt: "investigate" }, + { session, signal: idle.signal, emitStatus: () => idle.bump() }, + ); + + expect(idle.signal.aborted).toBe(false); + expect(result.text).toBe("done"); + }); }); diff --git a/packages/coding-agent/src/eval/__tests__/heartbeat.test.ts b/packages/coding-agent/src/eval/__tests__/heartbeat.test.ts new file mode 100644 index 000000000..a9391d9bf --- /dev/null +++ b/packages/coding-agent/src/eval/__tests__/heartbeat.test.ts @@ -0,0 +1,66 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { EVAL_HEARTBEAT_OP, setBridgeHeartbeatIntervalMs, withBridgeHeartbeat } from "../heartbeat"; +import type { JsStatusEvent } from "../js/shared/types"; + +describe("withBridgeHeartbeat", () => { + afterEach(() => { + setBridgeHeartbeatIntervalMs(); + }); + + it("pumps heartbeat events on cadence while the operation is pending, then stops", async () => { + setBridgeHeartbeatIntervalMs(20); + const events: JsStatusEvent[] = []; + + const value = await withBridgeHeartbeat( + event => events.push(event), + async () => { + await Bun.sleep(130); + return "done"; + }, + ); + + expect(value).toBe("done"); + // ~6 ticks fit in 130ms at a 20ms cadence; assert it ticked repeatedly + // without pinning the exact count (scheduler jitter). + expect(events.length).toBeGreaterThanOrEqual(3); + expect(events.every(event => event.op === EVAL_HEARTBEAT_OP)).toBe(true); + + // The interval is cleared once the operation settles: no further ticks. + const settledCount = events.length; + await Bun.sleep(80); + expect(events.length).toBe(settledCount); + }); + + it("runs the operation without emitting when no status sink is wired", async () => { + setBridgeHeartbeatIntervalMs(5); + let ran = 0; + + const value = await withBridgeHeartbeat(undefined, async () => { + ran++; + await Bun.sleep(40); + return 42; + }); + + expect(value).toBe(42); + expect(ran).toBe(1); + }); + + it("clears the heartbeat even when the operation throws", async () => { + setBridgeHeartbeatIntervalMs(15); + const events: JsStatusEvent[] = []; + + await expect( + withBridgeHeartbeat( + event => events.push(event), + async () => { + await Bun.sleep(60); + throw new Error("boom"); + }, + ), + ).rejects.toThrow("boom"); + + const afterThrow = events.length; + await Bun.sleep(60); + expect(events.length).toBe(afterThrow); + }); +}); diff --git a/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts index c5d2ce6be..9dde7d5cb 100644 --- a/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/llm-bridge.test.ts @@ -8,6 +8,8 @@ import type { ModelRegistry } from "../../config/model-registry"; import { Settings } from "../../config/settings"; import type { ToolSession } from "../../tools"; import { ToolError } from "../../tools/tool-errors"; +import { setBridgeHeartbeatIntervalMs } from "../heartbeat"; +import { IdleTimeout } from "../idle-timeout"; import { disposeAllVmContexts } from "../js/context-manager"; import { executeJs } from "../js/executor"; import { runEvalLlm } from "../llm-bridge"; @@ -97,6 +99,7 @@ function assistant(opts: { describe("runEvalLlm", () => { afterEach(() => { vi.restoreAllMocks(); + setBridgeHeartbeatIntervalMs(); }); it("resolves each tier to its expected model", async () => { @@ -213,6 +216,26 @@ describe("runEvalLlm", () => { ToolError, ); }); + + it("keeps the idle watchdog armed while a slow llm() request is in flight", async () => { + // A oneshot completion emits no status until it returns; a slow request + // must not look like a stalled cell. The bridge pumps a heartbeat while it + // awaits, re-arming the watchdog through emitStatus. + setBridgeHeartbeatIntervalMs(15); + vi.spyOn(ai, "completeSimple").mockImplementation(async () => { + await Bun.sleep(200); + return assistant({ text: "the answer" }); + }); + + using idle = new IdleTimeout(60); + const result = await runEvalLlm( + { prompt: "q", model: "smol" }, + { session: makeSession(), signal: idle.signal, emitStatus: () => idle.bump() }, + ); + + expect(idle.signal.aborted).toBe(false); + expect(result.text).toBe("the answer"); + }); }); describe("llm() through eval runtimes", () => { diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index 479e017af..54dcecad1 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -16,6 +16,7 @@ import { AgentOutputManager } from "../task/output-manager"; import type { AgentDefinition, AgentProgress } from "../task/types"; import type { ToolSession } from "../tools"; import { ToolError } from "../tools/tool-errors"; +import { withBridgeHeartbeat } from "./heartbeat"; import type { JsStatusEvent } from "./js/shared/types"; // Import review tools for side effects (registers subagent tool handlers). import "../tools/review"; @@ -231,44 +232,49 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption const id = await outputManager.allocate(outputIdBase(parsed.label, agentName)); const assignment = parsed.prompt.trim(); const context = trimToUndefined(parsed.context); - const result = await taskExecutor.runSubprocess({ - cwd: options.session.cwd, - agent: effectiveAgent, - task: renderSubagentPrompt(assignment), - assignment, - context, - description: trimToUndefined(parsed.label), - index: 0, - id, - taskDepth: options.session.taskDepth ?? 0, - modelOverride, - parentActiveModelPattern, - thinkingLevel: effectiveAgent.thinkingLevel, - outputSchema: structured ? parsed.schema : undefined, - sessionFile, - persistArtifacts: Boolean(sessionFile), - artifactsDir, - contextFile, - enableLsp: (options.session.enableLsp ?? true) && options.session.settings.get("task.enableLsp"), - signal: options.signal, - eventBus: options.session.eventBus, - onProgress: progress => emitProgressStatus(options.emitStatus, progress), - authStorage: options.session.authStorage, - modelRegistry: options.session.modelRegistry, - settings: options.session.settings, - mcpManager, - contextFiles, - skills: availableSkills, - autoloadSkills: resolvedAutoloadSkills, - workspaceTree: options.session.workspaceTree, - promptTemplates: options.session.promptTemplates, - localProtocolOptions, - parentArtifactManager, - parentHindsightSessionState: options.session.getHindsightSessionState?.(), - parentMnemopiSessionState: options.session.getMnemopiSessionState?.(), - parentTelemetry: options.session.getTelemetry?.(), - parentEvalSessionId, - }); + // Pump a heartbeat while the subagent runs so the eval idle watchdog stays + // armed across quiet stretches (time-to-first-token, long nested tools) + // where `onProgress` would otherwise emit no status to re-arm it. + const result = await withBridgeHeartbeat(options.emitStatus, () => + taskExecutor.runSubprocess({ + cwd: options.session.cwd, + agent: effectiveAgent, + task: renderSubagentPrompt(assignment), + assignment, + context, + description: trimToUndefined(parsed.label), + index: 0, + id, + taskDepth: options.session.taskDepth ?? 0, + modelOverride, + parentActiveModelPattern, + thinkingLevel: effectiveAgent.thinkingLevel, + outputSchema: structured ? parsed.schema : undefined, + sessionFile, + persistArtifacts: Boolean(sessionFile), + artifactsDir, + contextFile, + enableLsp: (options.session.enableLsp ?? true) && options.session.settings.get("task.enableLsp"), + signal: options.signal, + eventBus: options.session.eventBus, + onProgress: progress => emitProgressStatus(options.emitStatus, progress), + authStorage: options.session.authStorage, + modelRegistry: options.session.modelRegistry, + settings: options.session.settings, + mcpManager, + contextFiles, + skills: availableSkills, + autoloadSkills: resolvedAutoloadSkills, + workspaceTree: options.session.workspaceTree, + promptTemplates: options.session.promptTemplates, + localProtocolOptions, + parentArtifactManager, + parentHindsightSessionState: options.session.getHindsightSessionState?.(), + parentMnemopiSessionState: options.session.getMnemopiSessionState?.(), + parentTelemetry: options.session.getTelemetry?.(), + parentEvalSessionId, + }), + ); if (result.exitCode !== 0 || result.error) { const failureMessage = diff --git a/packages/coding-agent/src/eval/heartbeat.ts b/packages/coding-agent/src/eval/heartbeat.ts new file mode 100644 index 000000000..ceabcc03f --- /dev/null +++ b/packages/coding-agent/src/eval/heartbeat.ts @@ -0,0 +1,66 @@ +/** + * Keepalive for in-flight host-side eval bridge calls. + * + * The eval idle watchdog ({@link ../tools/eval IdleTimeout}) treats a cell's + * `timeout` as an *inactivity* budget and only re-arms when a status event + * reaches it. Host-side bridge helpers — `agent()`/`parallel()` (via + * `runSubprocess`) and `llm()` (a single completion) — can legitimately run for + * long stretches with **no** intermediate status: a subagent's time-to-first + * token on a reasoning model, a long quiet nested tool, or the entire body of a + * oneshot `llm()` call. Without a keepalive the watchdog mistakes that work for + * a stall and aborts the cell mid-flight, killing the subagent. + * + * {@link withBridgeHeartbeat} fixes that by pumping a synthetic + * {@link EVAL_HEARTBEAT_OP} status event on a fixed cadence while the wrapped + * operation is pending. The event rides the same `emitStatus → onStatus` channel + * both runtimes already forward, so it re-arms the watchdog without any new + * plumbing. Consumers MUST treat the heartbeat as a pure keepalive: bump the + * watchdog and drop it (never persist or render it) — see the executor display + * sinks and the eval tool's `onStatus` handler. + */ +import type { JsStatusEvent } from "./js/shared/types"; + +/** + * Synthetic status op emitted purely to keep the eval idle watchdog alive while + * a host-side bridge call is in flight. Carries no payload. + */ +export const EVAL_HEARTBEAT_OP = "heartbeat"; + +/** + * Heartbeat cadence. Comfortably below the default 30s idle budget (and the + * larger budgets long fanouts run under), so a working bridge call always bumps + * the watchdog before it expires, while a genuine stall is still bounded once + * the call settles and the heartbeat stops. + */ +const HEARTBEAT_INTERVAL_MS = 5_000; + +let heartbeatIntervalMs = HEARTBEAT_INTERVAL_MS; + +/** + * Test seam: override the heartbeat cadence so integration tests can exercise + * the keepalive within a sub-second idle budget. Pass no value to restore the + * production default. + */ +export function setBridgeHeartbeatIntervalMs(ms?: number): void { + heartbeatIntervalMs = ms === undefined ? HEARTBEAT_INTERVAL_MS : Math.max(1, Math.floor(ms)); +} + +/** + * Run {@link operation}, pumping {@link EVAL_HEARTBEAT_OP} status events through + * {@link emitStatus} on a fixed cadence until it settles. A no-op wrapper when + * no `emitStatus` sink is wired (the heartbeat would reach nobody). + */ +export async function withBridgeHeartbeat( + emitStatus: ((event: JsStatusEvent) => void) | undefined, + operation: () => Promise, +): Promise { + if (!emitStatus) return operation(); + const timer = setInterval(() => emitStatus({ op: EVAL_HEARTBEAT_OP }), heartbeatIntervalMs); + // Never keep the event loop alive for the heartbeat alone. + timer.unref?.(); + try { + return await operation(); + } finally { + clearInterval(timer); + } +} diff --git a/packages/coding-agent/src/eval/js/executor.ts b/packages/coding-agent/src/eval/js/executor.ts index 8c73b0005..f490bb9ee 100644 --- a/packages/coding-agent/src/eval/js/executor.ts +++ b/packages/coding-agent/src/eval/js/executor.ts @@ -1,6 +1,7 @@ import { DEFAULT_MAX_BYTES, OutputSink } from "../../session/streaming-output"; import type { ToolSession } from "../../tools"; import { resolveOutputMaxColumns, resolveOutputSinkHeadBytes } from "../../tools/output-meta"; +import { EVAL_HEARTBEAT_OP } from "../heartbeat"; import { executeInVmContext, type JsDisplayOutput } from "./context-manager"; import type { JsStatusEvent } from "./shared/types"; @@ -105,8 +106,13 @@ export async function executeJs(code: string, options: JsExecutorOptions): Promi signal, onText: chunk => outputSink.push(chunk), onDisplay: output => { + if (output.type === "status") { + // Heartbeats are pure idle-watchdog keepalives: forward them so + // the eval tool re-arms its timer, but never store or render them. + options.onStatus?.(output.event); + if (output.event.op === EVAL_HEARTBEAT_OP) return; + } displayOutputs.push(output); - if (output.type === "status") options.onStatus?.(output.event); }, }, }); diff --git a/packages/coding-agent/src/eval/llm-bridge.ts b/packages/coding-agent/src/eval/llm-bridge.ts index 301d553bf..39fd1168f 100644 --- a/packages/coding-agent/src/eval/llm-bridge.ts +++ b/packages/coding-agent/src/eval/llm-bridge.ts @@ -18,6 +18,7 @@ import { extractTextContent, extractToolCall, parseJsonPayload } from "../commit import { expandRoleAlias, formatModelString, resolveModelFromString } from "../config/model-resolver"; import type { ToolSession } from "../tools"; import { ToolError } from "../tools/tool-errors"; +import { withBridgeHeartbeat } from "./heartbeat"; import type { JsStatusEvent } from "./js/shared/types"; /** Synthetic bridge name reserved for the `llm()` helper across both runtimes. */ @@ -131,20 +132,25 @@ export async function runEvalLlm(args: unknown, options: EvalLlmBridgeOptions): const telemetry = resolveTelemetry(options.session.getTelemetry?.(), options.session.getSessionId?.() ?? undefined); - const response = await instrumentedCompleteSimple( - model, - { - systemPrompt: system ? [system] : undefined, - messages: [{ role: "user", content: [{ type: "text", text: prompt }], timestamp: Date.now() }], - tools, - }, - { - apiKey, - signal: options.signal, - reasoning: reasoningForTier(tier, model), - toolChoice: schema ? { type: "tool", name: STRUCTURED_TOOL_NAME } : undefined, - }, - { telemetry, oneshotKind: "eval_llm" }, + // A oneshot completion emits no status until it returns, so pump a heartbeat + // while it runs to keep the eval idle watchdog armed across a slow (e.g. + // reasoning-tier) request that would otherwise look like a stalled cell. + const response = await withBridgeHeartbeat(options.emitStatus, () => + instrumentedCompleteSimple( + model, + { + systemPrompt: system ? [system] : undefined, + messages: [{ role: "user", content: [{ type: "text", text: prompt }], timestamp: Date.now() }], + tools, + }, + { + apiKey, + signal: options.signal, + reasoning: reasoningForTier(tier, model), + toolChoice: schema ? { type: "tool", name: STRUCTURED_TOOL_NAME } : undefined, + }, + { telemetry, oneshotKind: "eval_llm" }, + ), ); if (response.stopReason === "error") { diff --git a/packages/coding-agent/src/eval/py/executor.ts b/packages/coding-agent/src/eval/py/executor.ts index ea492bb6d..01549d77c 100644 --- a/packages/coding-agent/src/eval/py/executor.ts +++ b/packages/coding-agent/src/eval/py/executor.ts @@ -5,6 +5,7 @@ import { Settings } from "../../config/settings"; import { OutputSink } from "../../session/streaming-output"; import type { ToolSession } from "../../tools"; import { resolveOutputMaxColumns, resolveOutputSinkHeadBytes } from "../../tools/output-meta"; +import { EVAL_HEARTBEAT_OP } from "../heartbeat"; import type { JsStatusEvent } from "../js/shared/types"; import { checkPythonKernelAvailability, @@ -496,8 +497,13 @@ async function executeWithKernel( // Collect every display output and, for status events, stream them live so // long-running bridge helpers (e.g. `agent()`) surface progress mid-cell. const collectDisplay = (output: KernelDisplayOutput) => { + if (output.type === "status") { + // Heartbeats are pure idle-watchdog keepalives: forward them so the + // eval tool re-arms its timer, but never store or render them. + options?.onStatus?.(output.event); + if (output.event.op === EVAL_HEARTBEAT_OP) return; + } displayOutputs.push(output); - if (output.type === "status") options?.onStatus?.(output.event); }; const emitStatus = options?.emitStatus ?? ((event: JsStatusEvent) => collectDisplay({ type: "status", event })); const runId = `py-${crypto.randomUUID()}`; diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index e2129907c..f4b6afbe4 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -7,6 +7,7 @@ import * as z from "zod/v4"; import { settings } from "../config/settings"; import { jsBackend, pythonBackend } from "../eval"; import type { ExecutorBackend, ExecutorBackendResult } from "../eval/backend"; +import { EVAL_HEARTBEAT_OP } from "../eval/heartbeat"; import { IdleTimeout } from "../eval/idle-timeout"; import { defaultEvalSessionId } from "../eval/session-id"; import type { EvalCellResult, EvalDisplayOutput, EvalLanguage, EvalStatusEvent, EvalToolDetails } from "../eval/types"; @@ -388,7 +389,12 @@ export class EvalTool implements AgentTool { outputSink!.push(chunk); }, onStatus: event => { + // Every status event re-arms the inactivity watchdog. A + // heartbeat is a pure keepalive emitted while a host-side + // bridge call (agent()/llm()) runs: it bumps the timer but + // carries no payload, so don't persist or render it. idle.bump(); + if (event.op === EVAL_HEARTBEAT_OP) return; cellResult.statusEvents ??= []; upsertStatusEvent(cellResult.statusEvents, event); pushUpdate();