From 3827c9ce2bafa949848985b1114b8a86b17d21c8 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 18:25:02 +0000 Subject: [PATCH 01/48] fix(tui): gated kitty unicode placeholders to terminals that honor U=1 The 15.9 placeholder rollout enabled the U=1/U+10EEEE grid by default for every terminal advertising the Kitty image protocol. Only kitty and ghostty actually ship a working implementation; WezTerm advertises Kitty graphics but has not implemented placeholder support (upstream wezterm/wezterm#986 has it unchecked), and the tmux/screen fallback can force Kitty mode on any outer terminal (Terminal.app, etc.). On those paths every placeholder cell rendered as a literal PUA box glyph - the screenshot in #1877 - and each repaint re-emitted the full columns x rows grid (~4.8KB per frame for a typical 80x20 image vs ~22 bytes for the direct a=p APC), which produced the laggy scrolling. Add detectKittyUnicodePlaceholdersSupport(terminalId, env) and seed the feature flag from terminal-capabilities after TERMINAL_ID resolves. Defaults on only for kitty and ghostty; everything else falls back to direct a=p,i=...,p=... placement, which those paths already render correctly. PI_NO_KITTY_PLACEHOLDERS=1 remains a hard opt-out, and a new PI_KITTY_PLACEHOLDERS=1 opts in on otherwise-unsupported terminals (for e.g. wezterm nightlies that have merged placeholder support). Fixes #1877 --- packages/coding-agent/CHANGELOG.md | 4 +++ packages/tui/CHANGELOG.md | 4 +++ packages/tui/src/kitty-graphics.ts | 35 ++++++++++++++++++-- packages/tui/src/terminal-capabilities.ts | 8 +++++ packages/tui/test/kitty-graphics.test.ts | 40 +++++++++++++++++++++++ 5 files changed, 89 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..3b8a1931c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed inline images rendering as a wall of empty PUA box glyphs with laggy scrolling on Kitty-protocol terminals that do not honor Unicode placeholders (most notably WezTerm and tmux/screen passthrough to a non-Kitty outer terminal). The 15.9 placeholder rollout enabled the `U=1`/U+10EEEE grid for every Kitty-protocol path; it now defaults on only for `kitty` and `ghostty`, with `PI_NO_KITTY_PLACEHOLDERS=1` as a hard opt-out and `PI_KITTY_PLACEHOLDERS=1` as opt-in for terminals (e.g. wezterm nightlies) that have since added support ([#1877](https://github.com/can1357/oh-my-pi/issues/1877)). + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5fb546c22..a29c1d083 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed inline images (added in 15.9) rendering as a wall of empty PUA box glyphs and producing laggy scrolling on Kitty-protocol terminals that do not implement Unicode placeholders — most notably WezTerm (per upstream wezterm/wezterm#986, placeholder support is still unchecked) and the tmux/screen `getFallbackImageProtocol` path that forces Kitty mode even on non-supporting outer terminals (Terminal.app, etc.). `unicodePlaceholders` now defaults on only for `kitty` and `ghostty`; everything else falls back to direct `a=p,i=…,p=…` placement, which those paths already render correctly. `PI_NO_KITTY_PLACEHOLDERS=1` is still honored as a hard opt-out, and a new `PI_KITTY_PLACEHOLDERS=1` opts in on otherwise-unsupported terminals (e.g. a wezterm nightly that has merged placeholder support) ([#1877](https://github.com/can1357/oh-my-pi/issues/1877)). + ## [15.9.1] - 2026-06-04 ### Fixed diff --git a/packages/tui/src/kitty-graphics.ts b/packages/tui/src/kitty-graphics.ts index b531be732..ca0aef62a 100644 --- a/packages/tui/src/kitty-graphics.ts +++ b/packages/tui/src/kitty-graphics.ts @@ -17,7 +17,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { $env, $flag, logger } from "@oh-my-pi/pi-utils"; +import { $env, logger } from "@oh-my-pi/pi-utils"; /** Kitty Unicode placeholder base character (U+10EEEE, Plane 16 PUA). */ export const KITTY_PLACEHOLDER = "\u{10eeee}"; @@ -76,8 +76,39 @@ function transmissionOverride(): KittyTransmissionMedium | "auto" { return "auto"; } +/** + * Whether the detected terminal renders Kitty Unicode placeholders (`U=1` + + * U+10EEEE with row/column diacritics). + * + * Only `kitty` (the protocol's origin) and `ghostty` ship a working + * implementation; WezTerm advertises Kitty graphics but treats placeholder + * cells as literal PUA glyphs (see wezterm/wezterm#986, "placeholder support" + * still unchecked), and the tmux/screen fallback can land on any outer + * terminal. Enabling placeholders on those paths emits a `columns × rows` + * grid of U+10EEEE per image per frame; the cells render as boxed fallback + * glyphs and re-emit on every repaint, which is exactly the + * "stuck/laggy scrolling + ASCII artifact" symptom reported in #1877. + * + * `PI_NO_KITTY_PLACEHOLDERS=1` forces off (e.g. for tmux passthrough to a + * non-supporting outer terminal); `PI_KITTY_PLACEHOLDERS=1` forces on (e.g. + * for a wezterm nightly that has merged placeholder support). + */ +export function detectKittyUnicodePlaceholdersSupport( + terminalId: string, + env: NodeJS.ProcessEnv = Bun.env, +): boolean { + const offRaw = env.PI_NO_KITTY_PLACEHOLDERS?.trim().toLowerCase(); + if (offRaw === "1" || offRaw === "true" || offRaw === "on" || offRaw === "yes" || offRaw === "y") return false; + const force = env.PI_KITTY_PLACEHOLDERS?.trim().toLowerCase(); + if (force === "1" || force === "true" || force === "on" || force === "yes" || force === "y") return true; + if (force === "0" || force === "false" || force === "off" || force === "no" || force === "n") return false; + return terminalId === "kitty" || terminalId === "ghostty"; +} + let features: KittyGraphicsFeatures = { - unicodePlaceholders: !$flag("PI_NO_KITTY_PLACEHOLDERS"), + // Off until `terminal-capabilities` seeds it from the detected terminal id — + // the default-on path corrupts wezterm and tmux-passthrough sessions. + unicodePlaceholders: false, // Start direct; a successful probe (or explicit `temp-file` override) promotes. transmissionMedium: transmissionOverride() === "temp-file" ? "temp-file" : "direct", }; diff --git a/packages/tui/src/terminal-capabilities.ts b/packages/tui/src/terminal-capabilities.ts index bc7ef4d19..31fce69e8 100644 --- a/packages/tui/src/terminal-capabilities.ts +++ b/packages/tui/src/terminal-capabilities.ts @@ -1,12 +1,14 @@ import { encodeSixel } from "@oh-my-pi/pi-natives"; import { $env, isBunTestRuntime } from "@oh-my-pi/pi-utils"; import { + detectKittyUnicodePlaceholdersSupport, encodeKittyTempFileTransmit, getKittyGraphics, isPngBase64, KITTY_PLACEHOLDER, kittyPlaceholdersFit, renderKittyPlaceholderLines, + setKittyGraphics, } from "./kitty-graphics"; export enum ImageProtocol { @@ -321,6 +323,12 @@ export const TERMINAL = (() => { return resolved; })(); +// Seed Kitty Unicode placeholder support from the resolved terminal id. Only +// kitty/ghostty are known to honor `U=1` placement; other Kitty-protocol paths +// (wezterm, tmux/screen fallback) treat the placeholder cells as literal PUA +// glyphs, which is the "ASCII artifact + laggy scrolling" reported in #1877. +setKittyGraphics({ unicodePlaceholders: detectKittyUnicodePlaceholdersSupport(TERMINAL.id, Bun.env) }); + type MutableTerminalInfo = { imageProtocol: ImageProtocol | null; deccara: boolean; diff --git a/packages/tui/test/kitty-graphics.test.ts b/packages/tui/test/kitty-graphics.test.ts index 03870907c..b2e81fb6c 100644 --- a/packages/tui/test/kitty-graphics.test.ts +++ b/packages/tui/test/kitty-graphics.test.ts @@ -2,6 +2,7 @@ import { afterEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import { visibleWidth } from "@oh-my-pi/pi-natives"; import { + detectKittyUnicodePlaceholdersSupport, encodeKittyPlaceholderGrid, encodeKittyTempFileProbe, encodeKittyTempFileTransmit, @@ -156,3 +157,42 @@ describe("kitty graphics feature state", () => { } }); }); + +describe("detectKittyUnicodePlaceholdersSupport", () => { + function env(extra: Record = {}): NodeJS.ProcessEnv { + return extra as NodeJS.ProcessEnv; + } + + it("enables for kitty and ghostty by default (the only terminals that render U=1 placement)", () => { + expect(detectKittyUnicodePlaceholdersSupport("kitty", env())).toBe(true); + expect(detectKittyUnicodePlaceholdersSupport("ghostty", env())).toBe(true); + }); + + it("disables for wezterm and other Kitty-protocol paths that treat placeholders as literal PUA glyphs (#1877)", () => { + expect(detectKittyUnicodePlaceholdersSupport("wezterm", env())).toBe(false); + // Tmux/screen fallback: base terminal id with Kitty protocol forced on by + // `getFallbackImageProtocol`. The outer terminal need not understand U=1. + expect(detectKittyUnicodePlaceholdersSupport("base", env())).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("iterm2", env())).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("alacritty", env())).toBe(false); + }); + + it("honors PI_NO_KITTY_PLACEHOLDERS=1 as a hard off override on supporting terminals", () => { + expect(detectKittyUnicodePlaceholdersSupport("kitty", env({ PI_NO_KITTY_PLACEHOLDERS: "1" }))).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("ghostty", env({ PI_NO_KITTY_PLACEHOLDERS: "true" }))).toBe(false); + }); + + it("honors PI_KITTY_PLACEHOLDERS=1 as opt-in on otherwise-unsupported terminals", () => { + expect(detectKittyUnicodePlaceholdersSupport("wezterm", env({ PI_KITTY_PLACEHOLDERS: "1" }))).toBe(true); + }); + + it("PI_NO_KITTY_PLACEHOLDERS beats PI_KITTY_PLACEHOLDERS when both are set", () => { + const both = env({ PI_NO_KITTY_PLACEHOLDERS: "1", PI_KITTY_PLACEHOLDERS: "1" }); + expect(detectKittyUnicodePlaceholdersSupport("kitty", both)).toBe(false); + }); + + it("PI_KITTY_PLACEHOLDERS=0 forces off on a default-on terminal", () => { + expect(detectKittyUnicodePlaceholdersSupport("kitty", env({ PI_KITTY_PLACEHOLDERS: "0" }))).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("ghostty", env({ PI_KITTY_PLACEHOLDERS: "off" }))).toBe(false); + }); +}); From ced288676f3650b98eebfef0f762b3822b7b5696 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 18:25:08 +0000 Subject: [PATCH 02/48] style: bun run fix --- packages/tui/src/kitty-graphics.ts | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/packages/tui/src/kitty-graphics.ts b/packages/tui/src/kitty-graphics.ts index ca0aef62a..8b65bf5d8 100644 --- a/packages/tui/src/kitty-graphics.ts +++ b/packages/tui/src/kitty-graphics.ts @@ -93,10 +93,7 @@ function transmissionOverride(): KittyTransmissionMedium | "auto" { * non-supporting outer terminal); `PI_KITTY_PLACEHOLDERS=1` forces on (e.g. * for a wezterm nightly that has merged placeholder support). */ -export function detectKittyUnicodePlaceholdersSupport( - terminalId: string, - env: NodeJS.ProcessEnv = Bun.env, -): boolean { +export function detectKittyUnicodePlaceholdersSupport(terminalId: string, env: NodeJS.ProcessEnv = Bun.env): boolean { const offRaw = env.PI_NO_KITTY_PLACEHOLDERS?.trim().toLowerCase(); if (offRaw === "1" || offRaw === "true" || offRaw === "on" || offRaw === "yes" || offRaw === "y") return false; const force = env.PI_KITTY_PLACEHOLDERS?.trim().toLowerCase(); From cc64804859079782408d1a07699ac23340aeafe1 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 19:13:37 +0000 Subject: [PATCH 03/48] fix(ai/openai-responses): route streaming events by item_id for parallel tool calls processResponsesStream kept a singleton currentItem/currentBlock and ignored event.item_id / event.output_index. OpenAI's own API serialises function_call items so the bug stayed latent there, but llama.cpp (and other local Responses-compat hosts) emits parallel function_calls interleaved: two output_item.added events fire before any deltas, then the deltas for fc_a get concatenated into fc_b's partialJson, and parseStreamingJson returns {} when fc_a is finalised. Every call but the last dispatched with empty arguments. Track every open item in registries keyed by output_index and item_id. Each per-item event (delta/done for function_call/custom_tool_call, content_part/ text/refusal/reasoning) now looks up the right block via the event identifier and falls back to the most recently added item only for tests / minimal mock providers that omit identifiers. contentIndex on every emitted stream event is computed from output.content.indexOf(block) so deltas for an earlier parallel item no longer report the index of the latest item. Regression covers: interleaved deltas with item_id routing, done-only finalisation when args ship as a single chunk on each item, and contentIndex correctness for both delta and toolcall_end events. Fixes #1880 --- packages/ai/CHANGELOG.md | 4 + .../src/providers/openai-responses-shared.ts | 252 +++++++++++------- ...enai-responses-parallel-tool-calls.test.ts | 205 ++++++++++++++ 3 files changed, 360 insertions(+), 101 deletions(-) create mode 100644 packages/ai/test/openai-responses-parallel-tool-calls.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2789f64c8..3f2811b20 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed parallel `function_call` items on the OpenAI Responses API losing arguments on every call except the last when the upstream server interleaves their stream events (observed against llama.cpp and other local Responses-compat hosts). `processResponsesStream` no longer routes `function_call_arguments.{delta,done}`, `output_item.done`, content_part/text/refusal/reasoning events through a singleton `currentItem`/`currentBlock` reference; it now tracks every open item in registries keyed by `output_index` and `item_id` so each event is folded into the matching block and the emitted `toolcall_end` carries the correct `contentIndex`. ([#1880](https://github.com/can1357/oh-my-pi/issues/1880)) + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 2125ce726..97fdfc68e 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -395,19 +395,54 @@ export async function processResponsesStream( model: Model, options?: ProcessResponsesStreamOptions, ): Promise { - let currentItem: - | ResponseReasoningItem - | ResponseOutputMessage - | ResponseFunctionToolCall - | ResponseCustomToolCall - | null = null; - let currentBlock: - | ThinkingContent - | TextContent - | (ToolCall & { partialJson: string; lastParseLen?: number }) - | null = null; - const blocks = output.content; - const blockIndex = () => blocks.length - 1; + type StreamingToolCallBlock = ToolCall & { partialJson: string; lastParseLen?: number }; + interface StreamingItem { + item: ResponseReasoningItem | ResponseOutputMessage | ResponseFunctionToolCall | ResponseCustomToolCall; + block: ThinkingContent | TextContent | StreamingToolCallBlock; + } + + // Multiple items (parallel function_calls in particular) can be open at the same + // time. OpenAI's spec routes every per-item event by `output_index`/`item_id`; + // see https://github.com/can1357/oh-my-pi/issues/1880 — llama.cpp emits parallel + // function_call deltas interleaved, and a singleton `current` reference would + // fold them into the wrong block and drop arguments on every call but the last. + const openItemsByOutputIndex = new Map(); + const openItemsByItemId = new Map(); + let lastOpenItem: StreamingItem | null = null; + + const registerOpenItem = ( + outputIndex: number | undefined, + itemId: string | undefined, + entry: StreamingItem, + ): void => { + if (typeof outputIndex === "number") openItemsByOutputIndex.set(outputIndex, entry); + if (itemId) openItemsByItemId.set(itemId, entry); + lastOpenItem = entry; + }; + const lookupOpenItem = (event: { output_index?: number; item_id?: string }): StreamingItem | undefined => { + if (typeof event.output_index === "number") { + const found = openItemsByOutputIndex.get(event.output_index); + if (found) return found; + } + if (event.item_id) { + const found = openItemsByItemId.get(event.item_id); + if (found) return found; + } + // Fallback for tests / mock providers that omit identifiers on stream events. + return lastOpenItem ?? undefined; + }; + const closeOpenItem = ( + outputIndex: number | undefined, + itemId: string | undefined, + entry: StreamingItem | undefined, + ): void => { + if (typeof outputIndex === "number") openItemsByOutputIndex.delete(outputIndex); + if (itemId) openItemsByItemId.delete(itemId); + if (entry && lastOpenItem === entry) lastOpenItem = null; + }; + const contentIndexOf = (block: ThinkingContent | TextContent | StreamingToolCallBlock): number => + output.content.indexOf(block); + let sawFirstToken = false; for await (const event of openaiStream) { @@ -420,29 +455,28 @@ export async function processResponsesStream( } const item = event.item; if (item.type === "reasoning") { - currentItem = item; - currentBlock = { type: "thinking", thinking: "", itemId: item.id }; - output.content.push(currentBlock); - stream.push({ type: "thinking_start", contentIndex: blockIndex(), partial: output }); + const block: ThinkingContent = { type: "thinking", thinking: "", itemId: item.id }; + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "thinking_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "message") { - currentItem = item; - currentBlock = { type: "text", text: "" }; - output.content.push(currentBlock); - stream.push({ type: "text_start", contentIndex: blockIndex(), partial: output }); + const block: TextContent = { type: "text", text: "" }; + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "text_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "function_call") { - currentItem = item; - currentBlock = { + const block: StreamingToolCallBlock = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), name: item.name, arguments: {}, partialJson: item.arguments || "", }; - output.content.push(currentBlock); - stream.push({ type: "toolcall_start", contentIndex: blockIndex(), partial: output }); + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "custom_tool_call") { - currentItem = item; - currentBlock = { + const block: StreamingToolCallBlock = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), // Preserve the raw wire name (e.g. `apply_patch`). The agent-loop @@ -456,39 +490,43 @@ export async function processResponsesStream( // accumulation buffer so later code that inspects the field still works. partialJson: item.input ?? "", }; - output.content.push(currentBlock); - stream.push({ type: "toolcall_start", contentIndex: blockIndex(), partial: output }); + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } } else if (event.type === "response.reasoning_summary_part.added") { - if (currentItem?.type === "reasoning") { - currentItem.summary = currentItem.summary || []; - currentItem.summary.push(event.part); + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning") { + entry.item.summary = entry.item.summary || []; + entry.item.summary.push(event.part); } } else if (event.type === "response.reasoning_summary_text.delta") { - if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") { - currentItem.summary = currentItem.summary || []; - const lastPart = currentItem.summary[currentItem.summary.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning" && entry.block.type === "thinking") { + entry.item.summary = entry.item.summary || []; + const lastPart = entry.item.summary[entry.item.summary.length - 1]; if (lastPart) { - currentBlock.thinking += event.delta; + entry.block.thinking += event.delta; lastPart.text += event.delta; stream.push({ type: "thinking_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } } else if (event.type === "response.reasoning_summary_part.done") { - if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") { - currentItem.summary = currentItem.summary || []; - const lastPart = currentItem.summary[currentItem.summary.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning" && entry.block.type === "thinking") { + entry.item.summary = entry.item.summary || []; + const lastPart = entry.item.summary[entry.item.summary.length - 1]; if (lastPart) { - currentBlock.thinking += "\n\n"; + entry.block.thinking += "\n\n"; lastPart.text += "\n\n"; stream.push({ type: "thinking_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: "\n\n", partial: output, }); @@ -497,91 +535,103 @@ export async function processResponsesStream( } else if (event.type === "response.reasoning_text.delta") { // Raw reasoning text delta from local providers that stream thinking // directly rather than via the OpenAI summary tracking protocol. - if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") { - currentBlock.thinking += event.delta; + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning" && entry.block.type === "thinking") { + entry.block.thinking += event.delta; stream.push({ type: "thinking_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } else if (event.type === "response.content_part.added") { - if (currentItem?.type === "message") { - currentItem.content = currentItem.content || []; + const entry = lookupOpenItem(event); + if (entry?.item.type === "message") { + entry.item.content = entry.item.content || []; if (event.part.type === "output_text" || event.part.type === "refusal") { - currentItem.content.push(event.part); + entry.item.content.push(event.part); } } } else if (event.type === "response.output_text.delta") { - if (currentItem?.type === "message" && currentBlock?.type === "text") { - const lastPart = currentItem.content?.[currentItem.content.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "message" && entry.block.type === "text") { + const lastPart = entry.item.content?.[entry.item.content.length - 1]; if (lastPart?.type === "output_text") { - currentBlock.text += event.delta; + entry.block.text += event.delta; lastPart.text += event.delta; stream.push({ type: "text_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } } else if (event.type === "response.refusal.delta") { - if (currentItem?.type === "message" && currentBlock?.type === "text") { - const lastPart = currentItem.content?.[currentItem.content.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "message" && entry.block.type === "text") { + const lastPart = entry.item.content?.[entry.item.content.length - 1]; if (lastPart?.type === "refusal") { - currentBlock.text += event.delta; + entry.block.text += event.delta; lastPart.refusal += event.delta; stream.push({ type: "text_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } } else if (event.type === "response.function_call_arguments.delta") { - if (currentItem?.type === "function_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson += event.delta; - const throttled = parseStreamingJsonThrottled(currentBlock.partialJson, currentBlock.lastParseLen ?? 0); + const entry = lookupOpenItem(event); + if (entry?.item.type === "function_call" && entry.block.type === "toolCall") { + const block = entry.block; + block.partialJson += event.delta; + const throttled = parseStreamingJsonThrottled(block.partialJson, block.lastParseLen ?? 0); if (throttled) { - currentBlock.arguments = throttled.value; - currentBlock.lastParseLen = throttled.parsedLen; + block.arguments = throttled.value; + block.lastParseLen = throttled.parsedLen; } stream.push({ type: "toolcall_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(block), delta: event.delta, partial: output, }); } } else if (event.type === "response.function_call_arguments.done") { - if (currentItem?.type === "function_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson = event.arguments; - currentBlock.arguments = parseStreamingJson(currentBlock.partialJson); - delete (currentBlock as { partialJson?: string }).partialJson; - delete (currentBlock as { lastParseLen?: number }).lastParseLen; + const entry = lookupOpenItem(event); + if (entry?.item.type === "function_call" && entry.block.type === "toolCall") { + const block = entry.block; + block.partialJson = event.arguments; + block.arguments = parseStreamingJson(block.partialJson); + delete (block as { partialJson?: string }).partialJson; + delete (block as { lastParseLen?: number }).lastParseLen; } } else if (event.type === "response.custom_tool_call_input.delta") { - if (currentItem?.type === "custom_tool_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson += event.delta; - currentBlock.arguments = { input: currentBlock.partialJson }; + const entry = lookupOpenItem(event); + if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") { + const block = entry.block; + block.partialJson += event.delta; + block.arguments = { input: block.partialJson }; stream.push({ type: "toolcall_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(block), delta: event.delta, partial: output, }); } } else if (event.type === "response.custom_tool_call_input.done") { - if (currentItem?.type === "custom_tool_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson = event.input; - currentBlock.arguments = { input: event.input }; + const entry = lookupOpenItem(event); + if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") { + entry.block.partialJson = event.input; + entry.block.arguments = { input: event.input }; } } else if (event.type === "response.output_item.done") { const item = structuredCloneJSON(event.item); options?.onOutputItemDone?.(item); + const entry = lookupOpenItem({ output_index: event.output_index, item_id: item.id }); if (item.type === "reasoning") { const thinking = item.summary?.length > 0 @@ -595,54 +645,53 @@ export async function processResponsesStream( if (reasoningBlock) { reasoningBlock.thinking = thinking; reasoningBlock.thinkingSignature = JSON.stringify(item); - const reasoningBlockIndex = output.content.indexOf(reasoningBlock); stream.push({ type: "thinking_end", - contentIndex: reasoningBlockIndex, + contentIndex: contentIndexOf(reasoningBlock), content: thinking, partial: output, }); } - if ((currentBlock as ThinkingContent | null)?.itemId === item.id) currentBlock = null; - } else if (item.type === "message" && currentBlock?.type === "text") { - currentBlock.text = item.content + closeOpenItem(event.output_index, item.id, entry); + } else if (item.type === "message" && entry?.block.type === "text") { + const block = entry.block; + block.text = item.content .map(part => (part.type === "output_text" ? (part.text ?? "") : (part.refusal ?? ""))) .join(""); - currentBlock.textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined); + block.textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined); stream.push({ type: "text_end", - contentIndex: blockIndex(), - content: currentBlock.text, + contentIndex: contentIndexOf(block), + content: block.text, partial: output, }); - currentBlock = null; + closeOpenItem(event.output_index, item.id, entry); } else if (item.type === "function_call") { - const args = - currentBlock?.type === "toolCall" && currentBlock.partialJson - ? parseStreamingJson(currentBlock.partialJson) - : parseStreamingJson(item.arguments || "{}"); + const block = entry?.block.type === "toolCall" ? entry.block : undefined; + const args = block?.partialJson + ? parseStreamingJson(block.partialJson) + : parseStreamingJson(item.arguments || "{}"); const toolCall: ToolCall = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), name: item.name, arguments: args, }; - if (currentBlock?.type === "toolCall") { + if (block) { // Persist the authoritative final args on the stored block. The // throttled delta parser may have skipped the last partial parse, - // leaving currentBlock.arguments stale (often `{}`); the emitted - // toolCall and the persisted block must agree. - currentBlock.arguments = args; - delete (currentBlock as { partialJson?: string }).partialJson; - delete (currentBlock as { lastParseLen?: number }).lastParseLen; + // leaving block.arguments stale (often `{}`); the emitted toolCall + // and the persisted block must agree. + block.arguments = args; + delete (block as { partialJson?: string }).partialJson; + delete (block as { lastParseLen?: number }).lastParseLen; } - currentBlock = null; - stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output }); + const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; + closeOpenItem(event.output_index, item.id, entry); + stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } else if (item.type === "custom_tool_call") { - const rawInput = - currentBlock?.type === "toolCall" && currentBlock.partialJson - ? currentBlock.partialJson - : (item.input ?? ""); + const block = entry?.block.type === "toolCall" ? entry.block : undefined; + const rawInput = block?.partialJson ? block.partialJson : (item.input ?? ""); const toolCall: ToolCall = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), @@ -650,8 +699,9 @@ export async function processResponsesStream( arguments: { input: rawInput }, customWireName: item.name, }; - currentBlock = null; - stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output }); + const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; + closeOpenItem(event.output_index, item.id, entry); + stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } } else if (event.type === "response.completed") { const response = event.response; diff --git a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts new file mode 100644 index 000000000..56e076e6d --- /dev/null +++ b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts @@ -0,0 +1,205 @@ +// Regression for https://github.com/can1357/oh-my-pi/issues/1880. +// +// llama.cpp (and any OpenAI-Responses-compatible host that interleaves +// multiple function_call items) emits `output_item.added` for every parallel +// call before the deltas arrive, then routes deltas via `item_id`/`output_index` +// instead of relying on a single in-flight item. `processResponsesStream` +// previously kept a singleton `currentBlock` reference and ignored those +// identifiers, so deltas for the first call were folded into the buffer of the +// most-recently-added block. The dispatcher then received empty `{}` arguments +// for every call except the last one. +// +// These tests pin the contract: each `function_call_arguments.{delta,done}` and +// `output_item.done` event must be routed by `output_index`/`item_id`, not by +// arrival order. +import { describe, expect, test } from "bun:test"; +import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; +import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types"; +import type { ResponseStreamEvent } from "openai/resources/responses/responses"; + +function makeModel(): Model<"openai-responses"> { + return { + api: "openai-responses", + name: "Llama", + id: "llama-3", + provider: "llama.cpp", + baseUrl: "http://127.0.0.1:8080/v1", + contextWindow: 8192, + maxTokens: 2048, + input: ["text"], + reasoning: false, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + }; +} + +function makeOutput(): AssistantMessage { + return { + role: "assistant", + content: [], + timestamp: Date.now(), + provider: "llama.cpp", + model: "llama-3", + api: "openai-responses", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + }; +} + +async function* makeStream(events: unknown[]): AsyncIterable { + for (const e of events) yield e as ResponseStreamEvent; +} + +type EmittedEvent = { type?: string } & Record; + +describe("processResponsesStream: parallel function_call items", () => { + test("routes deltas to the correct block when both items are added before any delta", async () => { + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never; + + const argsA = JSON.stringify({ _i: "Reading test", path: "test.txt" }); + const argsB = JSON.stringify({ _i: "Reading test", path: "test.md" }); + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: "" }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: "" }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_a", + delta: argsA, + }, + { + type: "response.function_call_arguments.delta", + output_index: 1, + item_id: "fc_b", + delta: argsB, + }, + { + type: "response.function_call_arguments.done", + output_index: 0, + item_id: "fc_a", + arguments: argsA, + }, + { + type: "response.function_call_arguments.done", + output_index: 1, + item_id: "fc_b", + arguments: argsB, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.done", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: argsB }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.content).toHaveLength(2); + const [blockA, blockB] = output.content; + expect(blockA?.type).toBe("toolCall"); + expect(blockB?.type).toBe("toolCall"); + if (blockA?.type !== "toolCall" || blockB?.type !== "toolCall") throw new Error("expected toolCalls"); + expect(blockA.arguments).toEqual({ _i: "Reading test", path: "test.txt" }); + expect(blockB.arguments).toEqual({ _i: "Reading test", path: "test.md" }); + + const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{ + toolCall: { id: string; arguments: Record }; + contentIndex: number; + }>; + expect(ends).toHaveLength(2); + const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e])); + expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ _i: "Reading test", path: "test.txt" }); + expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ _i: "Reading test", path: "test.md" }); + expect(byCallId.get("call_a")?.contentIndex).toBe(0); + expect(byCallId.get("call_b")?.contentIndex).toBe(1); + + // Delta events must also carry the per-block contentIndex — otherwise the + // streaming UI updates the wrong block while args are still arriving. + const deltas = emitted.filter(e => e.type === "toolcall_delta") as Array<{ + delta: string; + contentIndex: number; + }>; + expect(deltas).toHaveLength(2); + const deltaForA = deltas.find(d => d.delta === argsA); + const deltaForB = deltas.find(d => d.delta === argsB); + expect(deltaForA?.contentIndex).toBe(0); + expect(deltaForB?.contentIndex).toBe(1); + }); + + test("routes done-only finalization to the correct block when arguments stream as a single chunk on each item", async () => { + // Some local Responses-compat hosts skip the per-delta protocol entirely + // and stash the full arguments string on `output_item.added`/`done`. + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never; + + const argsA = JSON.stringify({ path: "test.txt" }); + const argsB = JSON.stringify({ path: "test.md" }); + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: argsB }, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.done", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: argsB }, + }, + ]), + output, + stream, + makeModel(), + ); + + const [blockA, blockB] = output.content; + if (blockA?.type !== "toolCall" || blockB?.type !== "toolCall") throw new Error("expected toolCalls"); + expect(blockA.arguments).toEqual({ path: "test.txt" }); + expect(blockB.arguments).toEqual({ path: "test.md" }); + + const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{ + toolCall: { id: string; arguments: Record }; + }>; + expect(ends).toHaveLength(2); + const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e])); + expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ path: "test.txt" }); + expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ path: "test.md" }); + }); +}); From efc046a83828d971560bc0b52a72c636d9da86f1 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 19:18:00 +0000 Subject: [PATCH 04/48] fix(ai): allow per-model opt-out of max_output_tokens on the wire Add `Model.omitMaxOutputTokens` (`models.yml` model definitions and `modelOverrides` accept the same field). When set, the openai-responses and openai-completions providers stop emitting `max_output_tokens` / `max_tokens` / `max_completion_tokens` so the upstream API applies its own default cap. Catalog `maxTokens` is still honoured for local budgeting (compaction, context promotion); only the wire field is suppressed. Restores the pre-v15.8.3 behaviour for Ollama proxies fronting cloud catalogs (GLM, Kimi, DeepSeek): OMP cannot discover their true output limit, so users previously set `maxTokens` to the Ollama context window to unlock the model. v15.8.3 began sending that value as `max_output_tokens`, triggering HTTP 400 from the upstream provider. `applyCommonResponsesSamplingParams` now takes the model object instead of a bare provider string so the wire-suppression flag is available to the sampling-params builder. Fixes #1881 --- packages/ai/CHANGELOG.md | 4 + .../src/providers/azure-openai-responses.ts | 2 +- .../ai/src/providers/openai-completions.ts | 2 +- .../src/providers/openai-responses-shared.ts | 13 +++- packages/ai/src/providers/openai-responses.ts | 2 +- packages/ai/src/types.ts | 12 +++ ...i-responses-omit-max-output-tokens.test.ts | 73 +++++++++++++++++++ packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/config/model-registry.ts | 6 ++ .../src/config/models-config-schema.ts | 2 + .../coding-agent/test/model-registry.test.ts | 45 ++++++++++++ 11 files changed, 158 insertions(+), 7 deletions(-) create mode 100644 packages/ai/test/openai-responses-omit-max-output-tokens.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2789f64c8..2729e3433 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `Model.omitMaxOutputTokens` so providers (notably Ollama proxies fronting cloud catalogs) can suppress `max_output_tokens` (Responses) and `max_tokens`/`max_completion_tokens` (Completions) on the wire while still using the catalog `maxTokens` for local budgeting. Without it, `applyCommonResponsesSamplingParams` unconditionally sent the catalog cap and HTTP-400'd against upstream APIs whose true output limit was unknown to OMP. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881)) + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index f9d3a2bed..04027d02a 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -296,7 +296,7 @@ function buildParams( prompt_cache_key: normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), }; - applyCommonResponsesSamplingParams(params, options, model.provider); + applyCommonResponsesSamplingParams(params, options, model); if (context.tools) { params.tools = convertTools(context.tools); diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 93fbf86df..b8b5e591c 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -1180,7 +1180,7 @@ function buildParams( params.store = false; } - if (effectiveMaxTokens) { + if (effectiveMaxTokens && !model.omitMaxOutputTokens) { if (compat.maxTokensField === "max_tokens") { params.max_tokens = effectiveMaxTokens; } else { diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 2125ce726..8780db8cd 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -752,21 +752,26 @@ type CommonSamplingOptions = Pick< /** * Apply the common `StreamOptions` → Responses sampling-parameter mapping (max output tokens, * temperature, top-p/k, min-p, presence/repetition penalties, service tier). Mutates `params`. + * + * `max_output_tokens` is suppressed when {@link Model.omitMaxOutputTokens} is `true`, so + * proxies (notably Ollama) that forward to upstream APIs with an unknown output-token cap + * can let the upstream apply its own default instead of 400-ing on `maxTokens` values that + * reflect the model's context window rather than the upstream output limit. */ export function applyCommonResponsesSamplingParams

( params: P, options: CommonSamplingOptions | undefined, - provider: string, + model: Pick, ): void { - if (options?.maxTokens) params.max_output_tokens = options.maxTokens; + if (options?.maxTokens && !model.omitMaxOutputTokens) params.max_output_tokens = options.maxTokens; if (options?.temperature !== undefined) params.temperature = options.temperature; if (options?.topP !== undefined) params.top_p = options.topP; if (options?.topK !== undefined) params.top_k = options.topK; if (options?.minP !== undefined) params.min_p = options.minP; if (options?.presencePenalty !== undefined) params.presence_penalty = options.presencePenalty; if (options?.repetitionPenalty !== undefined) params.repetition_penalty = options.repetitionPenalty; - if (shouldSendServiceTier(options?.serviceTier, provider)) { - const resolved = resolveServiceTier(options?.serviceTier, provider); + if (shouldSendServiceTier(options?.serviceTier, model.provider)) { + const resolved = resolveServiceTier(options?.serviceTier, model.provider); if (resolved === "flex" || resolved === "scale" || resolved === "priority") { params.service_tier = resolved; } diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index f9647128d..ac1684b43 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -463,7 +463,7 @@ function buildParams( stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined, }; - applyCommonResponsesSamplingParams(params, options, model.provider); + applyCommonResponsesSamplingParams(params, options, model); // TODO: openai responses has no top-level `stop`/`stop_sequences`; surface via reasoning.stop? // `StreamOptions.stopSequences` is intentionally dropped for this provider. // TODO: openai responses has no top-level `frequency_penalty` field as of the current SDK; diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 95bbd72e6..9b03d99d8 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -905,6 +905,18 @@ export interface Model { premiumMultiplier?: number; contextWindow: number; maxTokens: number; + /** + * When `true`, providers MUST omit `max_output_tokens` (Responses) / + * `max_tokens` / `max_completion_tokens` (Completions) from the outbound + * request and let the upstream API decide the per-response cap. `maxTokens` + * is still used locally for budgeting (compaction, context promotion); only + * the wire field is suppressed. + * + * Use this for proxies (notably Ollama) that forward to a backend whose true + * output limit OMP cannot discover — sending the wrong value triggers 400s + * from the upstream provider. + */ + omitMaxOutputTokens?: boolean; headers?: Record; /** * Streaming transport override. When `"pi-native"`, `streamSimple` routes diff --git a/packages/ai/test/openai-responses-omit-max-output-tokens.test.ts b/packages/ai/test/openai-responses-omit-max-output-tokens.test.ts new file mode 100644 index 000000000..a5dcb1537 --- /dev/null +++ b/packages/ai/test/openai-responses-omit-max-output-tokens.test.ts @@ -0,0 +1,73 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { getBundledModel } from "../src/models"; +import { streamSimple } from "../src/stream"; +import type { Context, Model } from "../src/types"; + +const originalFetch = global.fetch; + +const baseModel = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-responses">; + +function mockSseFetch(): Record { + const captured: Record = {}; + const fetchMock = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => { + const body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record) : {}; + Object.assign(captured, body); + const event = { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 1, + output_tokens: 1, + total_tokens: 2, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }; + return new Response(`data: ${JSON.stringify(event)}\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); + }); + global.fetch = Object.assign(fetchMock, { preconnect: originalFetch.preconnect }) as typeof fetch; + return captured; +} + +const ctx: Context = { + systemPrompt: ["hi"], + messages: [{ role: "user", content: "ping", timestamp: Date.now() }], +}; + +async function drain(model: Model<"openai-responses">): Promise> { + const captured = mockSseFetch(); + const stream = streamSimple(model, ctx, { apiKey: "k" }); + for await (const event of stream) { + if (event.type === "done" || event.type === "error") break; + } + return captured; +} + +beforeEach(() => { + expect(baseModel.maxTokens).toBeGreaterThan(0); +}); + +afterEach(() => { + global.fetch = originalFetch; + vi.restoreAllMocks(); +}); + +describe("openai-responses max_output_tokens opt-out", () => { + it("sends max_output_tokens = model.maxTokens by default", async () => { + const body = await drain(baseModel); + expect(body.max_output_tokens).toBe(baseModel.maxTokens); + }); + + it("omits max_output_tokens when model.omitMaxOutputTokens is true", async () => { + const model: Model<"openai-responses"> = { ...baseModel, omitMaxOutputTokens: true }; + const body = await drain(model); + expect(body).not.toHaveProperty("max_output_tokens"); + // maxTokens is still populated locally for budgeting, even though we + // don't put it on the wire. + expect(model.maxTokens).toBe(baseModel.maxTokens); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..1dbfa50be 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `omitMaxOutputTokens` to `models.yml` model definitions and `modelOverrides`, so users can opt a model out of the on-the-wire `max_output_tokens` / `max_tokens` cap while keeping the catalog `maxTokens` for local budgeting. Intended for Ollama-style proxies whose upstream output limit OMP cannot discover. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881)) + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 20f723336..3f0008b43 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -547,6 +547,7 @@ function applyModelOverride(model: Model, override: ModelOverride): Model; compat?: Model["compat"]; contextPromotionTarget?: string; @@ -597,6 +599,7 @@ type CustomModelOverlay = { cost?: { input: number; output: number; cacheRead: number; cacheWrite: number }; contextWindow?: number; maxTokens?: number; + omitMaxOutputTokens?: boolean; headers?: Record; compat?: Model["compat"]; contextPromotionTarget?: string; @@ -667,6 +670,7 @@ function buildCustomModelOverlay( cost: modelDef.cost, contextWindow: modelDef.contextWindow, maxTokens: modelDef.maxTokens, + omitMaxOutputTokens: modelDef.omitMaxOutputTokens, headers: mergeCustomModelHeaders(providerHeaders, modelDef.headers, authHeader, providerApiKey), compat: mergeCompat(providerCompat, modelDef.compat), contextPromotionTarget: modelDef.contextPromotionTarget, @@ -823,6 +827,7 @@ function finalizeCustomModel(model: CustomModelOverlay, options: CustomModelBuil resolvedModel.contextWindow ?? reference?.contextWindow ?? (options.useDefaults ? 128000 : undefined), maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : undefined), headers: resolvedModel.headers, + omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens, compat: mergeCompat(reference?.compat, resolvedModel.compat), contextPromotionTarget: resolvedModel.contextPromotionTarget, premiumMultiplier: resolvedModel.premiumMultiplier, @@ -1124,6 +1129,7 @@ export class ModelRegistry { cost: customModel.cost ?? existingModel.cost, contextWindow: customModel.contextWindow ?? existingModel.contextWindow, maxTokens: customModel.maxTokens ?? existingModel.maxTokens, + omitMaxOutputTokens: customModel.omitMaxOutputTokens ?? existingModel.omitMaxOutputTokens, // Same-id custom definitions replace bundled transport behavior. Provider-level // headers/compat were already folded into customModel during parsing; do not // re-merge bundled transport metadata here. diff --git a/packages/coding-agent/src/config/models-config-schema.ts b/packages/coding-agent/src/config/models-config-schema.ts index 965d85d33..1911651bb 100644 --- a/packages/coding-agent/src/config/models-config-schema.ts +++ b/packages/coding-agent/src/config/models-config-schema.ts @@ -93,6 +93,7 @@ const ModelDefinitionSchema = z.object({ premiumMultiplier: z.number().optional(), contextWindow: z.number().optional(), maxTokens: z.number().optional(), + omitMaxOutputTokens: z.boolean().optional(), headers: z.record(z.string(), z.string()).optional(), compat: OpenAICompatSchema.optional(), contextPromotionTarget: z.string().min(1).optional(), @@ -114,6 +115,7 @@ export const ModelOverrideSchema = z.object({ premiumMultiplier: z.number().optional(), contextWindow: z.number().optional(), maxTokens: z.number().optional(), + omitMaxOutputTokens: z.boolean().optional(), headers: z.record(z.string(), z.string()).optional(), compat: OpenAICompatSchema.optional(), contextPromotionTarget: z.string().min(1).optional(), diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index a6a9e5858..c7a1817d8 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -1405,6 +1405,51 @@ describe("ModelRegistry", () => { )?.name; expect(restoredName).not.toBe("Custom Name"); }); + + test("modelOverrides can set omitMaxOutputTokens on a built-in model", () => { + writeRawModelsJson({ + openai: { + modelOverrides: { + "gpt-5.4": { + omitMaxOutputTokens: true, + }, + }, + }, + }); + + const registry = new ModelRegistry(authStorage, modelsJsonPath); + const model = registry.find("openai", "gpt-5.4"); + expect(model?.omitMaxOutputTokens).toBe(true); + // maxTokens is still populated locally — only the wire emission is suppressed. + expect(model?.maxTokens).toBeGreaterThan(0); + }); + + test("custom model definitions accept omitMaxOutputTokens", () => { + writeRawModelsJson({ + ollama: { + baseUrl: "http://localhost:11434/v1", + api: "openai-responses", + auth: "none", + models: [ + { + id: "glm-5.1:cloud", + name: "GLM 5.1 Cloud (Ollama)", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 202752, + maxTokens: 202752, + omitMaxOutputTokens: true, + }, + ], + }, + }); + + const registry = new ModelRegistry(authStorage, modelsJsonPath); + const model = registry.find("ollama", "glm-5.1:cloud"); + expect(model?.omitMaxOutputTokens).toBe(true); + expect(model?.maxTokens).toBe(202752); + }); }); describe("github-copilot oauth endpoint alignment", () => { From b2c6b7becf25d45aad5e216e631e54b2235c8b80 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 21:42:52 +0000 Subject: [PATCH 05/48] fix(plugins): loaded package import extensions Resolved plugin-local package import aliases while loading extension source graphs so legacy Pi plugins with TypeScript source imports like #src/* register correctly under OMP. Added a regression covering legacy scope rewrites through a package import module.\n\nFixes #1889 --- .../extensibility/plugins/legacy-pi-compat.ts | 249 ++++++++++++++++-- .../test/plugin-extensions-discovery.test.ts | 58 ++++ 2 files changed, 284 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index 90def20a2..27f169b99 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -45,6 +45,10 @@ const LEGACY_PI_IMPORT_SPECIFIER_REGEX = new RegExp( "g", ); const resolvedSpecifierFallbacks = new Map(); +const SOURCE_MODULE_EXTENSIONS = [".ts", ".tsx", ".mts", ".cts", ".js", ".jsx", ".mjs", ".cjs"] as const; +const PACKAGE_IMPORT_CONDITIONS = ["bun", "import", "default", "types"] as const; +const packageRootCache = new Map(); +const packageImportsCache = new Map | null>(); // Extensions that imported `@sinclair/typebox` directly used to resolve against a // real `@sinclair/typebox` install. The runtime dep was replaced with the Zod-backed @@ -224,30 +228,226 @@ function rewriteLegacyPiImports(source: string): string { const TYPEBOX_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s*\(\s*)["'])(@sinclair\/typebox)(["'])/g; /** - * Rewrite the legacy specifiers a Pi extension may import — `@(scope)/pi-*` and - * the bare `@sinclair/typebox` root — to absolute `file://` URLs pointing at the - * bundled package or compat shim. Every other specifier (relative siblings, the - * extension's own bare dependencies) is left untouched so Bun resolves it + * Rewrite the extension-owned specifiers OMP must host-resolve — legacy + * `@(scope)/pi-*`, bare `@sinclair/typebox`, and package `imports` aliases like + * `#src/*` — to absolute `file://` URLs. Every other specifier (relative + * siblings and third-party dependencies) is left untouched so Bun resolves it * natively from the extension's real on-disk location. */ -function rewriteLegacyExtensionSource(source: string): string { +async function rewriteLegacyExtensionSource(source: string, importerPath: string): Promise { const withPi = rewriteLegacyPiImports(source); - return withPi.replace( + const withTypeBox = withPi.replace( TYPEBOX_IMPORT_SPECIFIER_REGEX, (_match, prefix: string, _specifier: string, suffix: string) => { return `${prefix}${toImportSpecifier(TYPEBOX_SHIM_PATH)}${suffix}`; }, ); + return rewriteExtensionPackageImports(withTypeBox, importerPath); +} + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +async function pathExists(p: string): Promise { + try { + await fs.stat(p); + return true; + } catch { + return false; + } +} + +async function resolveSourceModuleFile(basePath: string): Promise { + try { + const stats = await fs.stat(basePath); + if (stats.isFile()) { + return realpathOrSelf(basePath); + } + if (stats.isDirectory()) { + for (const extension of SOURCE_MODULE_EXTENSIONS) { + const resolved = await resolveSourceModuleFile(path.join(basePath, `index${extension}`)); + if (resolved) return resolved; + } + } + } catch { + // Fall through to extension candidates below. + } + + if (path.extname(basePath)) { + return null; + } + + for (const extension of SOURCE_MODULE_EXTENSIONS) { + const resolved = await resolveSourceModuleFile(`${basePath}${extension}`); + if (resolved) return resolved; + } + return null; +} + +async function findPackageRoot(importerPath: string): Promise { + let dir = path.dirname(importerPath); + while (true) { + const cached = packageRootCache.get(dir); + if (cached !== undefined) { + return cached; + } + + if (await pathExists(path.join(dir, "package.json"))) { + packageRootCache.set(path.dirname(importerPath), dir); + return dir; + } + + const parent = path.dirname(dir); + if (parent === dir) { + packageRootCache.set(path.dirname(importerPath), null); + return null; + } + dir = parent; + } +} + +async function readPackageImports(packageRoot: string): Promise | null> { + const cached = packageImportsCache.get(packageRoot); + if (cached !== undefined) { + return cached; + } + + let imports: Record | null = null; + try { + const pkg = await Bun.file(path.join(packageRoot, "package.json")).json(); + if (isRecord(pkg) && isRecord(pkg.imports)) { + imports = pkg.imports; + } + } catch { + imports = null; + } + packageImportsCache.set(packageRoot, imports); + return imports; +} + +function selectPackageImportTarget(entry: unknown): string | null { + if (typeof entry === "string") { + return entry; + } + if (Array.isArray(entry)) { + for (const item of entry) { + const target = selectPackageImportTarget(item); + if (target) return target; + } + return null; + } + if (!isRecord(entry)) { + return null; + } + for (const condition of PACKAGE_IMPORT_CONDITIONS) { + const target = selectPackageImportTarget(entry[condition]); + if (target) return target; + } + for (const value of Object.values(entry)) { + const target = selectPackageImportTarget(value); + if (target) return target; + } + return null; +} + +async function resolvePackageImportTarget( + packageRoot: string, + target: string, + wildcard: string | null, +): Promise { + if (!target.startsWith("./")) { + return null; + } + const substituted = wildcard === null ? target : target.replaceAll("*", wildcard); + return resolveSourceModuleFile(path.resolve(packageRoot, substituted)); +} + +async function resolvePackageImportSpecifier(specifier: string, importerPath: string): Promise { + if (!specifier.startsWith("#")) { + return null; + } + + const packageRoot = await findPackageRoot(importerPath); + if (!packageRoot) { + return null; + } + + const imports = await readPackageImports(packageRoot); + if (!imports) { + return null; + } + + const exactTarget = selectPackageImportTarget(imports[specifier]); + if (exactTarget) { + return resolvePackageImportTarget(packageRoot, exactTarget, null); + } + + let bestMatch: { keyLength: number; target: string; wildcard: string } | null = null; + for (const [key, entry] of Object.entries(imports)) { + const starIndex = key.indexOf("*"); + if (starIndex === -1) continue; + + const prefix = key.slice(0, starIndex); + const suffix = key.slice(starIndex + 1); + if (!specifier.startsWith(prefix) || !specifier.endsWith(suffix)) { + continue; + } + + const target = selectPackageImportTarget(entry); + if (!target) { + continue; + } + + if (!bestMatch || key.length > bestMatch.keyLength) { + bestMatch = { + keyLength: key.length, + target, + wildcard: specifier.slice(prefix.length, specifier.length - suffix.length), + }; + } + } + + if (!bestMatch) { + return null; + } + return resolvePackageImportTarget(packageRoot, bestMatch.target, bestMatch.wildcard); +} + +const PACKAGE_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s*\(\s*)["'])(#[^"'()\s]+)(["'])/g; + +async function rewriteExtensionPackageImports(source: string, importerPath: string): Promise { + let rewritten = ""; + let lastIndex = 0; + for (const match of source.matchAll(PACKAGE_IMPORT_SPECIFIER_REGEX)) { + const matchIndex = match.index; + if (matchIndex === undefined) continue; + + const [fullMatch, prefix, specifier, suffix] = match; + if (!prefix || !specifier || !suffix) continue; + + const resolved = await resolvePackageImportSpecifier(specifier, importerPath); + if (!resolved) continue; + + rewritten += source.slice(lastIndex, matchIndex); + rewritten += `${prefix}${toImportSpecifier(resolved)}${suffix}`; + lastIndex = matchIndex + fullMatch.length; + } + + if (lastIndex === 0) { + return source; + } + return `${rewritten}${source.slice(lastIndex)}`; } function escapeRegExp(value: string): string { return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); } -// Match relative import specifiers (static `from "./…"` and dynamic -// `import("./…")`). Used to walk an extension's own module graph; bare and -// absolute specifiers are deliberately excluded. -const RELATIVE_IMPORT_SPECIFIER_REGEX = /(?:from\s+|import\s*\(\s*)["'](\.\.?\/[^"']+)["']/g; +// Match source modules in an extension graph (relative imports and package +// `imports` aliases such as `#src/*`). Bare third-party dependencies remain +// native Bun resolutions. +const EXTENSION_GRAPH_SPECIFIER_REGEX = /(?:from\s+|import\s*\(\s*)["']((?:\.\.?\/|#)[^"']+)["']/g; // Extension entry realpaths that already have a load-time rewrite hook // installed. Each `Bun.plugin()` registration is process-global and permanent, @@ -287,10 +487,14 @@ async function collectExtensionModules(entryRealPath: string): Promise { if (hookedExtensionEntries.has(entryRealPath)) { @@ -322,9 +527,8 @@ async function ensureExtensionGraphHook(entryRealPath: string): Promise { name: `omp:legacy-pi-ext:${Bun.hash(entryRealPath).toString(36)}`, setup(build) { build.onLoad({ filter, namespace: "file" }, async args => { - // Re-read on every load so a `?mtime` reload picks up edited source. const raw = await Bun.file(args.path).text(); - return { contents: rewriteLegacyExtensionSource(raw), loader: getLoader(args.path) }; + return { contents: await rewriteLegacyExtensionSource(raw, args.path), loader: getLoader(args.path) }; }); }, }); @@ -337,9 +541,8 @@ async function ensureExtensionGraphHook(entryRealPath: string): Promise { * and `__dirname`-relative `readFileSync` asset loads (HTML/CSS bundled next to * the entry) resolve exactly as they do under the original Pi runtime — no * temp-directory mirroring and no asset copying. An `onLoad` hook scoped to the - * entry's relative-import graph rewrites only the legacy `@(scope)/pi-*` and - * `@sinclair/typebox` imports in the extension's own source; everything else - * resolves natively. + * entry's source graph rewrites only host-resolved compatibility imports in the + * extension's own source; everything else resolves natively. */ export async function loadLegacyPiModule(resolvedPath: string): Promise { // Bun reports the realpath of a loaded module to `onLoad` and exposes it as diff --git a/packages/coding-agent/test/plugin-extensions-discovery.test.ts b/packages/coding-agent/test/plugin-extensions-discovery.test.ts index ab34ec339..8d2e88444 100644 --- a/packages/coding-agent/test/plugin-extensions-discovery.test.ts +++ b/packages/coding-agent/test/plugin-extensions-discovery.test.ts @@ -139,6 +139,64 @@ describe("plugin extension discovery", () => { expect(extension?.tools.has("legacy-pi-ext")).toBe(true); }); + it("loads installed legacy Pi plugin extensions that use package imports", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "package-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src", "feature"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "package-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "package-import-plugin", + version: "1.0.0", + imports: { + "#src/*": "./src/*", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#src/feature/command";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "src", "feature", "command.ts"), + [ + 'import { isToolCallEventType as legacyExtensions } from "@earendil-works/pi-coding-agent/extensibility/extensions";', + `import { isToolCallEventType as modernExtensions } from ${JSON.stringify(currentPiExtensionsPath)};`, + "", + 'if (legacyExtensions !== modernExtensions) throw new Error("legacy extension import did not remap");', + 'export const commandName = "package-import-ext";', + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("package-import-ext")).toBe(true); + }); + it("loads installed plugin extensions whose manifest entry points at a directory with index.ts", async () => { const pluginsDir = getPluginsDir(); const pluginDir = path.join(pluginsDir, "node_modules", "dir-entry-plugin"); From 6e47ae81505fc1c87abc5f8ecd2ed8bf0f7ffb43 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 21:49:06 +0000 Subject: [PATCH 06/48] fix(plugins): matched side-effect imports in compat rewrite Extended the legacy Pi, TypeBox, package-import alias, and extension graph regexes to recognise the bare \"import \\"specifier\\";\" shape so side-effect-only loads such as \"import \\"#src/register\\";\" walk into the source graph and get their legacy @(scope)/pi-* imports rewritten. Added a regression that loads a plugin with a side-effect alias import whose target contains a legacy scope import.\n\nFixes #1889 --- .../extensibility/plugins/legacy-pi-compat.ts | 8 +- .../test/plugin-extensions-discovery.test.ts | 79 +++++++++++++++++++ 2 files changed, 83 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index 27f169b99..fddae4262 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -41,7 +41,7 @@ const PI_SUBPATH_REMAPS: ReadonlyMap = new Map([ const LEGACY_PI_SPECIFIER_FILTER = new RegExp(`^@(?:${PI_SCOPE_ALTERNATION})/(?:${PI_PACKAGE_ALTERNATION})(?:/.*)?$`); const LEGACY_PI_IMPORT_SPECIFIER_REGEX = new RegExp( - `((?:from\\s+|import\\s*\\(\\s*)["'])(@(?:${PI_SCOPE_ALTERNATION})/(?:${PI_PACKAGE_ALTERNATION})(?:/[^"'()\\s]+)?)(["'])`, + `((?:from\\s+|import\\s+|import\\s*\\(\\s*)["'])(@(?:${PI_SCOPE_ALTERNATION})/(?:${PI_PACKAGE_ALTERNATION})(?:/[^"'()\\s]+)?)(["'])`, "g", ); const resolvedSpecifierFallbacks = new Map(); @@ -225,7 +225,7 @@ function rewriteLegacyPiImports(source: string): string { // Match the bare `@sinclair/typebox` import specifier (static + dynamic). // Subpath imports like `@sinclair/typebox/compiler` are intentionally excluded — // they expose TypeBox-only APIs the Zod-backed shim does not provide. -const TYPEBOX_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s*\(\s*)["'])(@sinclair\/typebox)(["'])/g; +const TYPEBOX_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s+|import\s*\(\s*)["'])(@sinclair\/typebox)(["'])/g; /** * Rewrite the extension-owned specifiers OMP must host-resolve — legacy @@ -414,7 +414,7 @@ async function resolvePackageImportSpecifier(specifier: string, importerPath: st return resolvePackageImportTarget(packageRoot, bestMatch.target, bestMatch.wildcard); } -const PACKAGE_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s*\(\s*)["'])(#[^"'()\s]+)(["'])/g; +const PACKAGE_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s+|import\s*\(\s*)["'])(#[^"'()\s]+)(["'])/g; async function rewriteExtensionPackageImports(source: string, importerPath: string): Promise { let rewritten = ""; @@ -447,7 +447,7 @@ function escapeRegExp(value: string): string { // Match source modules in an extension graph (relative imports and package // `imports` aliases such as `#src/*`). Bare third-party dependencies remain // native Bun resolutions. -const EXTENSION_GRAPH_SPECIFIER_REGEX = /(?:from\s+|import\s*\(\s*)["']((?:\.\.?\/|#)[^"']+)["']/g; +const EXTENSION_GRAPH_SPECIFIER_REGEX = /(?:from\s+|import\s+|import\s*\(\s*)["']((?:\.\.?\/|#)[^"']+)["']/g; // Extension entry realpaths that already have a load-time rewrite hook // installed. Each `Bun.plugin()` registration is process-global and permanent, diff --git a/packages/coding-agent/test/plugin-extensions-discovery.test.ts b/packages/coding-agent/test/plugin-extensions-discovery.test.ts index 8d2e88444..285d61d81 100644 --- a/packages/coding-agent/test/plugin-extensions-discovery.test.ts +++ b/packages/coding-agent/test/plugin-extensions-discovery.test.ts @@ -197,6 +197,85 @@ describe("plugin extension discovery", () => { expect(extension?.commands.has("package-import-ext")).toBe(true); }); + it("rewrites side-effect imports of package-import aliases and legacy Pi scopes", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "side-effect-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "side-effect-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "side-effect-plugin", + version: "1.0.0", + imports: { + "#src/*": "./src/*", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync( + extensionPath, + [ + // Side-effect imports — no `from`, no dynamic `import()`. The + // regex matchers must walk and rewrite both shapes so the legacy + // `@earendil-works` import inside `register.ts` resolves to the + // host `@oh-my-pi` package. + 'import "#src/register";', + 'import "./marker";', + "", + "declare global { var __sideEffectMarker: { ok: boolean; runs: number } | undefined; }", + "", + "export default function(pi) {", + '\tif (!globalThis.__sideEffectMarker?.ok) throw new Error("register side-effect did not run");', + '\tpi.registerCommand("side-effect-ext", { handler: async () => {} });', + "}", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "src", "register.ts"), + [ + 'import { isToolCallEventType as legacyExtensions } from "@earendil-works/pi-coding-agent/extensibility/extensions";', + `import { isToolCallEventType as modernExtensions } from ${JSON.stringify(currentPiExtensionsPath)};`, + "", + 'if (legacyExtensions !== modernExtensions) throw new Error("legacy side-effect import did not remap");', + "(globalThis as { __sideEffectMarker?: { ok: boolean; runs: number } }).__sideEffectMarker = { ok: true, runs: 1 };", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "src", "marker.ts"), + [ + "const slot = (globalThis as { __sideEffectMarker?: { ok: boolean; runs: number } }).__sideEffectMarker;", + 'if (!slot) throw new Error("relative side-effect import did not run before sibling");', + "slot.runs += 1;", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("side-effect-ext")).toBe(true); + expect((globalThis as { __sideEffectMarker?: { ok: boolean; runs: number } }).__sideEffectMarker).toEqual({ + ok: true, + runs: 2, + }); + delete (globalThis as { __sideEffectMarker?: unknown }).__sideEffectMarker; + }); + it("loads installed plugin extensions whose manifest entry points at a directory with index.ts", async () => { const pluginsDir = getPluginsDir(); const pluginDir = path.join(pluginsDir, "node_modules", "dir-entry-plugin"); From 251d9152fcb99dbed286d5de1f4a62963a507724 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 21:51:26 +0000 Subject: [PATCH 07/48] fix(plugins): honored package import condition order Selected conditional package import targets by package.json object order for supported Bun runtime conditions, including node, instead of probing a fixed precedence list. Added a regression where node precedes import and must be selected for a plugin-local #src/* import.\n\nFixes #1889 --- .../extensibility/plugins/legacy-pi-compat.ts | 11 ++-- .../test/plugin-extensions-discovery.test.ts | 62 +++++++++++++++++++ 2 files changed, 67 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index fddae4262..d21cd5f71 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -46,7 +46,7 @@ const LEGACY_PI_IMPORT_SPECIFIER_REGEX = new RegExp( ); const resolvedSpecifierFallbacks = new Map(); const SOURCE_MODULE_EXTENSIONS = [".ts", ".tsx", ".mts", ".cts", ".js", ".jsx", ".mjs", ".cjs"] as const; -const PACKAGE_IMPORT_CONDITIONS = ["bun", "import", "default", "types"] as const; +const SUPPORTED_PACKAGE_IMPORT_CONDITIONS = new Set(["bun", "node", "import", "default"]); const packageRootCache = new Map(); const packageImportsCache = new Map | null>(); @@ -340,11 +340,10 @@ function selectPackageImportTarget(entry: unknown): string | null { if (!isRecord(entry)) { return null; } - for (const condition of PACKAGE_IMPORT_CONDITIONS) { - const target = selectPackageImportTarget(entry[condition]); - if (target) return target; - } - for (const value of Object.values(entry)) { + for (const [condition, value] of Object.entries(entry)) { + if (!SUPPORTED_PACKAGE_IMPORT_CONDITIONS.has(condition)) { + continue; + } const target = selectPackageImportTarget(value); if (target) return target; } diff --git a/packages/coding-agent/test/plugin-extensions-discovery.test.ts b/packages/coding-agent/test/plugin-extensions-discovery.test.ts index 285d61d81..1be82cb64 100644 --- a/packages/coding-agent/test/plugin-extensions-discovery.test.ts +++ b/packages/coding-agent/test/plugin-extensions-discovery.test.ts @@ -197,6 +197,68 @@ describe("plugin extension discovery", () => { expect(extension?.commands.has("package-import-ext")).toBe(true); }); + it("honors package import conditional object order", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "conditional-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "node"), { recursive: true }); + fs.mkdirSync(path.join(pluginDir, "import"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "conditional-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "conditional-import-plugin", + version: "1.0.0", + imports: { + "#src/*": { + node: "./node/*", + import: "./import/*", + }, + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.mkdirSync(path.dirname(extensionPath), { recursive: true }); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#src/command";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "node", "command.ts"), + 'export const commandName = "node-conditional-ext";', + ); + fs.writeFileSync( + path.join(pluginDir, "import", "command.ts"), + 'export const commandName = "import-conditional-ext";', + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("node-conditional-ext")).toBe(true); + expect(extension?.commands.has("import-conditional-ext")).toBe(false); + }); + it("rewrites side-effect imports of package-import aliases and legacy Pi scopes", async () => { const pluginsDir = getPluginsDir(); const pluginDir = path.join(pluginsDir, "node_modules", "side-effect-plugin"); From 994ae513f59e6a84e359f82dd2aa83ac010b768b Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 21:54:33 +0000 Subject: [PATCH 08/48] fix(plugins): skipped non-source package import targets Returned null from resolveSourceModuleFile when a package imports alias points at a JSON, WASM, or other non-code asset so the on-load rewrite hook no longer claims it and forces it through the JS loader. Bun resolves these targets natively. Added a regression where #schema maps to a .json file imported with a JSON type assertion.\n\nFixes #1889 --- .../extensibility/plugins/legacy-pi-compat.ts | 10 +++- .../test/plugin-extensions-discovery.test.ts | 49 +++++++++++++++++++ 2 files changed, 58 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index d21cd5f71..8ecf04a6f 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -258,11 +258,19 @@ async function pathExists(p: string): Promise { } } +function hasSourceModuleExtension(p: string): boolean { + const ext = path.extname(p).toLowerCase(); + return (SOURCE_MODULE_EXTENSIONS as readonly string[]).includes(ext); +} + async function resolveSourceModuleFile(basePath: string): Promise { try { const stats = await fs.stat(basePath); if (stats.isFile()) { - return realpathOrSelf(basePath); + // Non-source files (JSON, WASM, text assets, etc.) bypass the on-load + // rewrite hook so Bun's native loaders handle them; our hook would + // otherwise pass them through `getLoader()` which falls back to `js`. + return hasSourceModuleExtension(basePath) ? realpathOrSelf(basePath) : null; } if (stats.isDirectory()) { for (const extension of SOURCE_MODULE_EXTENSIONS) { diff --git a/packages/coding-agent/test/plugin-extensions-discovery.test.ts b/packages/coding-agent/test/plugin-extensions-discovery.test.ts index 1be82cb64..7858b3365 100644 --- a/packages/coding-agent/test/plugin-extensions-discovery.test.ts +++ b/packages/coding-agent/test/plugin-extensions-discovery.test.ts @@ -259,6 +259,55 @@ describe("plugin extension discovery", () => { expect(extension?.commands.has("import-conditional-ext")).toBe(false); }); + it("leaves package import aliases that point at non-source files for Bun's native loaders", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "json-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "json-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "json-import-plugin", + version: "1.0.0", + imports: { + "#schema": "./src/schema.json", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync(path.join(pluginDir, "src", "schema.json"), JSON.stringify({ commandName: "json-schema-ext" })); + fs.writeFileSync( + extensionPath, + [ + 'import schema from "#schema" with { type: "json" };', + "", + "export default function(pi) {", + "\tpi.registerCommand(schema.commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("json-schema-ext")).toBe(true); + }); + it("rewrites side-effect imports of package-import aliases and legacy Pi scopes", async () => { const pluginsDir = getPluginsDir(); const pluginDir = path.join(pluginsDir, "node_modules", "side-effect-plugin"); From 356db2b845ee0a36cac6f6981bb17021129e5988 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 4 Jun 2026 22:27:23 +0000 Subject: [PATCH 09/48] fix(coding-agent): surfaced title generation failures Logged structured session-title generation skip and failure outcomes with session/model context, and prevented credential lookup errors from escaping into the caller's swallowed promise path. Added regression coverage for missing and failing title credentials. Fixes #1892 --- packages/coding-agent/CHANGELOG.md | 2 + .../src/modes/controllers/input-controller.ts | 10 +- .../coding-agent/src/utils/title-generator.ts | 95 +++++++++++-------- .../coding-agent/test/title-generator.test.ts | 60 ++++++++++++ 4 files changed, 128 insertions(+), 39 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..1c206941f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -19,6 +19,8 @@ ### Fixed +- Fixed auto session-title generation failures being swallowed without an actionable diagnostic. Title generation now logs structured start, missing-model/API-key, provider-error, empty-result, and exception outcomes with the session id and resolved title model; the interactive auto-title caller also logs uncaught persistence/generation errors instead of dropping them. ([#1892](https://github.com/can1357/oh-my-pi/issues/1892)) + - Fixed a streamed assistant message freezing at a partial prefix (e.g. only "Nat" of "Natives built, now…") on ED3-risk terminals (Ghostty/kitty/iTerm2/Alacritty), with the final text appearing only after a resize. `TranscriptContainer` freezes each non-live block by replaying its last live render, but render coalescing can finalize a block's content and append the next block within the same throttled frame — so the block was sealed at its stale mid-stream snapshot and never repainted until the next `thaw`. The block that was live on the previous render is now recomputed once on the live→frozen transition, sealing it at its final content. - Fixed ACP/RPC stdio startup so protocol frames are no longer consumed as one-shot piped prompt input before the JSON-RPC transport starts. diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index e13ef9abf..8cf4086b8 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -1,6 +1,6 @@ import * as fs from "node:fs/promises"; import type { AutocompleteProvider, SlashCommand } from "@oh-my-pi/pi-tui"; -import { $env, sanitizeText } from "@oh-my-pi/pi-utils"; +import { $env, logger, sanitizeText } from "@oh-my-pi/pi-utils"; import { getRoleInfo } from "../../config/model-registry"; import { isSettingsInitialized, settings } from "../../config/settings"; import { renderSegmentTrack } from "../../modes/components/segment-track"; @@ -406,7 +406,13 @@ export class InputController { } } }) - .catch(() => {}); + .catch(err => { + logger.warn("title-generator: uncaught auto-title error", { + sessionId: this.ctx.session.sessionId, + reason: "uncaught-auto-title-error", + error: err instanceof Error ? err.message : String(err), + }); + }); } if (this.ctx.onInputCallback) { diff --git a/packages/coding-agent/src/utils/title-generator.ts b/packages/coding-agent/src/utils/title-generator.ts index 6416205fc..84f4eab1e 100644 --- a/packages/coding-agent/src/utils/title-generator.ts +++ b/packages/coding-agent/src/utils/title-generator.ts @@ -149,7 +149,10 @@ export async function generateSessionTitle( // tiny title model can't reliably decline trivial input, so this happens // deterministically before any model is invoked; the caller retries on the // next user message while the session stays unnamed. - if (isLowSignalTitleInput(firstMessage)) return null; + if (isLowSignalTitleInput(firstMessage)) { + logger.debug("title-generator: skipped low-signal input", { sessionId, reason: "low-signal" }); + return null; + } const tinyModel = settings.get("providers.tinyModel"); if (tinyModel === ONLINE_TINY_TITLE_MODEL_KEY) { @@ -159,7 +162,14 @@ export async function generateSessionTitle( const onlineAbortController = new AbortController(); const localTitle = tinyTitleClient.generate(tinyModel, firstMessage).then( title => title || null, - () => null, + err => { + logger.warn("title-generator: local model error", { + sessionId, + model: tinyModel, + error: err instanceof Error ? err.message : String(err), + }); + return null; + }, ); const startOnline = (): Promise => generateTitleOnline( @@ -188,49 +198,48 @@ export async function generateTitleOnline( ): Promise { const model = getTitleModel(registry, settings, currentModel); if (!model) { - logger.debug("title-generator: no title model found"); + logger.warn("title-generator: no title model found", { sessionId, reason: "no-title-model" }); return null; } const userMessage = formatTitleUserMessage(firstMessage); - - const apiKey = await registry.getApiKey(model, sessionId); - if (!apiKey) { - logger.debug("title-generator: no API key for smol model", { - provider: model.provider, - id: model.id, - }); - return null; - } - // Resolve metadata after getApiKey so the session-sticky credential for this - // request is already recorded; metadataResolver can then return the correct - // account_uuid rather than the snapshot-at-call-site value. - const metadata = metadataResolver?.(model.provider); - - // Title generation is a 3-6 word task, but some reasoning backends ignore - // disableReasoning. Keep the normal cheap budget for non-reasoning models - // while reserving enough output room for reasoning models to still emit - // the forced tool call after any unavoidable thinking tokens. - const maxTokens = model.reasoning ? Math.max(TITLE_MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : TITLE_MAX_TOKENS; - const request = { - model: `${model.provider}/${model.id}`, - systemPrompt: TITLE_SYSTEM_PROMPT, - userMessage, - maxTokens, + const modelName = `${model.provider}/${model.id}`; + const modelContext = { + sessionId, + provider: model.provider, + id: model.id, + model: modelName, }; - logger.debug("title-generator: request", request); + logger.debug("title-generator: start", modelContext); try { + const apiKey = await registry.getApiKey(model, sessionId); + if (!apiKey) { + logger.warn("title-generator: no API key", { ...modelContext, reason: "missing-api-key" }); + return null; + } + // Resolve metadata after getApiKey so the session-sticky credential for this + // request is already recorded; metadataResolver can then return the correct + // account_uuid rather than the snapshot-at-call-site value. + const metadata = metadataResolver?.(model.provider); + + // Title generation is a 3-6 word task, but some reasoning backends ignore + // disableReasoning. Keep the normal cheap budget for non-reasoning models + // while reserving enough output room for reasoning models to still emit + // the forced tool call after any unavoidable thinking tokens. + const maxTokens = model.reasoning ? Math.max(TITLE_MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : TITLE_MAX_TOKENS; + logger.debug("title-generator: request", { ...modelContext, maxTokens }); + const response = await completeSimple( model, { - systemPrompt: [request.systemPrompt], - messages: [{ role: "user", content: request.userMessage, timestamp: Date.now() }], + systemPrompt: [TITLE_SYSTEM_PROMPT], + messages: [{ role: "user", content: userMessage, timestamp: Date.now() }], tools: [setTitleTool], }, { apiKey, - maxTokens: request.maxTokens, + maxTokens, disableReasoning: true, toolChoice: { type: "tool", name: SET_TITLE_TOOL_NAME }, metadata, @@ -239,8 +248,9 @@ export async function generateTitleOnline( ); if (response.stopReason === "error") { - logger.debug("title-generator: response error", { - model: request.model, + logger.warn("title-generator: response error", { + ...modelContext, + reason: "provider-response-error", stopReason: response.stopReason, errorMessage: response.errorMessage, }); @@ -249,8 +259,18 @@ export async function generateTitleOnline( const title = normalizeGeneratedTitle(extractGeneratedTitle(response.content)); - logger.debug("title-generator: response", { - model: request.model, + if (!title) { + logger.debug("title-generator: no title returned", { + ...modelContext, + reason: "model-returned-none", + usage: response.usage, + stopReason: response.stopReason, + }); + return null; + } + + logger.debug("title-generator: success", { + ...modelContext, title, usage: response.usage, stopReason: response.stopReason, @@ -258,8 +278,9 @@ export async function generateTitleOnline( return title; } catch (err) { - logger.debug("title-generator: error", { - model: request.model, + logger.warn("title-generator: error", { + ...modelContext, + reason: "exception", error: err instanceof Error ? err.message : String(err), }); return null; diff --git a/packages/coding-agent/test/title-generator.test.ts b/packages/coding-agent/test/title-generator.test.ts index 3447b3f03..dc1a6b4a4 100644 --- a/packages/coding-agent/test/title-generator.test.ts +++ b/packages/coding-agent/test/title-generator.test.ts @@ -1,6 +1,7 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import * as ai from "@oh-my-pi/pi-ai"; import { type Api, getBundledModel, type Model } from "@oh-my-pi/pi-ai"; +import { logger } from "@oh-my-pi/pi-utils"; import { generateSessionTitle } from "../src/utils/title-generator"; function getModelOrThrow(id: string): Model { @@ -116,6 +117,65 @@ describe("title generator", () => { expect(completeSimpleMock).toHaveBeenCalledTimes(1); }); + it("logs and returns null when title credentials are missing", async () => { + const model = getModelOrThrow("claude-sonnet-4-5"); + const completeSimpleMock = vi.spyOn(ai, "completeSimple"); + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + const title = await generateSessionTitle( + "Investigate the resolver", + { + getAvailable: () => [model], + getApiKey: async () => undefined, + } as never, + createSettings(model), + "session-1", + ); + + expect(title).toBeNull(); + expect(completeSimpleMock).not.toHaveBeenCalled(); + expect(warnSpy).toHaveBeenCalledWith( + "title-generator: no API key", + expect.objectContaining({ + sessionId: "session-1", + provider: model.provider, + id: model.id, + reason: "missing-api-key", + }), + ); + }); + + it("logs and returns null when title credential lookup throws", async () => { + const model = getModelOrThrow("claude-sonnet-4-5"); + const completeSimpleMock = vi.spyOn(ai, "completeSimple"); + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + const title = await generateSessionTitle( + "Investigate the resolver", + { + getAvailable: () => [model], + getApiKey: async () => { + throw new Error("credential lookup failed"); + }, + } as never, + createSettings(model), + "session-2", + ); + + expect(title).toBeNull(); + expect(completeSimpleMock).not.toHaveBeenCalled(); + expect(warnSpy).toHaveBeenCalledWith( + "title-generator: error", + expect.objectContaining({ + sessionId: "session-2", + provider: model.provider, + id: model.id, + reason: "exception", + error: "credential lookup failed", + }), + ); + }); + it("uses a reasoning-safe output budget for reasoning models", async () => { const model = getModelOrThrow("claude-sonnet-4-5"); const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({ From 7447086186baac629f1442529c12212e27751c45 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 00:53:39 +0200 Subject: [PATCH 10/48] fix(tui): pinned live region to stop scrollback dup on ED3 streams - Added NativeScrollbackLiveRegion seam reporting where a component's transient suffix begins. - Pinned the live block and chrome below it out of native history during foreground streams on ED3-risk terminals. - Consumed pending forced scrollback wipes while pinned to avoid yanking a scrolled-up reader (#1682). --- packages/coding-agent/CHANGELOG.md | 4 + .../modes/components/transcript-container.ts | 14 +- .../components/transcript-container.test.ts | 10 ++ packages/tui/CHANGELOG.md | 9 + packages/tui/src/tui.ts | 156 ++++++++++++++++-- .../test/reasoning-stream-dup-repro.test.ts | 105 ++++++++++++ packages/tui/test/render-stress-harness.ts | 57 ++++--- 7 files changed, 323 insertions(+), 32 deletions(-) create mode 100644 packages/tui/test/reasoning-stream-dup-repro.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..cb7fd45d6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Changed + +- Changed `TranscriptContainer` to report its live-region boundary to the renderer (`NativeScrollbackLiveRegion`): on ED3-risk terminals it now exposes the line offset where the bottom-most live block begins, so the TUI pins that block and the chrome below it out of native scrollback during streaming instead of committing its overflow and then leaving stale duplicates when the block re-lays-out or collapses. + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/modes/components/transcript-container.ts b/packages/coding-agent/src/modes/components/transcript-container.ts index 64af9ce42..c4fea9f7f 100644 --- a/packages/coding-agent/src/modes/components/transcript-container.ts +++ b/packages/coding-agent/src/modes/components/transcript-container.ts @@ -1,4 +1,4 @@ -import { type Component, Container, TERMINAL } from "@oh-my-pi/pi-tui"; +import { type Component, Container, type NativeScrollbackLiveRegion, TERMINAL } from "@oh-my-pi/pi-tui"; const kSnapshot = Symbol("transcript.frozenRender"); @@ -34,10 +34,14 @@ interface SnapshotCarrier { * and any drift reconciles safely. On terminals that can rebuild history this * freezing is unnecessary, so it renders every block live for full fidelity. */ -export class TranscriptContainer extends Container { +export class TranscriptContainer extends Container implements NativeScrollbackLiveRegion { // Bumped to invalidate every block's snapshot at once; a snapshot is only // honored when its stored generation still matches. #generation = 0; + // Local line index where the current bottom-most block begins in the most + // recent render. TUI extends the native-scrollback pinned region from this + // point through the live block and the root chrome rendered below it. + #nativeScrollbackLiveRegionStart: number | undefined; // The block that was bottom-most (live) on the previous render. When the live // position moves past it, its snapshot was last refreshed mid-stream and may // predate content that finalized in the same coalesced frame that appended the @@ -56,6 +60,10 @@ export class TranscriptContainer extends Container { super.clear(); } + getNativeScrollbackLiveRegionStart(): number | undefined { + return this.#nativeScrollbackLiveRegionStart; + } + /** * Retire all frozen snapshots so the next render reflects each block's current * state. Call at reconciliation checkpoints (prompt submit) where the whole @@ -68,6 +76,7 @@ export class TranscriptContainer extends Container { override render(width: number): string[] { width = Math.max(1, width); + this.#nativeScrollbackLiveRegionStart = undefined; if (!TERMINAL.eagerEraseScrollbackRisk) return super.render(width); const lines: string[] = []; @@ -76,6 +85,7 @@ export class TranscriptContainer extends Container { const prevLiveChild = this.#prevLiveChild; this.#prevLiveChild = liveChild; for (let i = 0; i < this.children.length; i++) { + if (i === liveIndex) this.#nativeScrollbackLiveRegionStart = lines.length; const child = this.children[i]! as Component & SnapshotCarrier; if (child !== liveChild) { const snapshot = child[kSnapshot]; diff --git a/packages/coding-agent/test/modes/components/transcript-container.test.ts b/packages/coding-agent/test/modes/components/transcript-container.test.ts index 5ff34729a..448c9b3a4 100644 --- a/packages/coding-agent/test/modes/components/transcript-container.test.ts +++ b/packages/coding-agent/test/modes/components/transcript-container.test.ts @@ -53,6 +53,16 @@ describe("TranscriptContainer", () => { expect(container.render(40)).toEqual(["a2", "b2"]); }); + it("reports the live-region boundary at the bottom-most block on ED3-risk terminals", () => { + riskFlag.eagerEraseScrollbackRisk = true; + const container = new TranscriptContainer(); + container.addChild(new MutableBlock(["a1", "a2"])); + container.addChild(new MutableBlock(["b1"])); + + expect(container.render(40)).toEqual(["a1", "a2", "b1"]); + expect(container.getNativeScrollbackLiveRegionStart()).toBe(2); + }); + it("seals the prior block at its final content when finalize+append coalesce (ED3-risk)", () => { riskFlag.eagerEraseScrollbackRisk = true; const container = new TranscriptContainer(); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5fb546c22..af8aa2e50 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,15 @@ ## [Unreleased] +### Added + +- Added a `NativeScrollbackLiveRegion` component seam (`getNativeScrollbackLiveRegionStart()`): a component reports the local line index where its live/transient suffix begins, and `TUI` treats that suffix — plus every root child rendered below it — as not yet safe to commit to native scrollback on ED3-risk terminals whose viewport position is unobservable (Ghostty/kitty/Alacritty/VTE/iTerm2 on POSIX). + +### Fixed + +- Fixed persistent line duplication in native scrollback while a foreground turn streamed on ED3-risk terminals with an unobservable viewport. The bottom-most live block scrolled its overflow into native history during growth; when that same block later re-laid-out, shrank, or collapsed (a tool preview collapsing to its compact result, a Markdown list/table reflow, a reasoning re-wrap) the viewport repaint correctly showed the new tail, but the stale overflow rows stayed in committed history and re-appeared above the live region — and they cannot be removed without ED3 (`CSI 3 J`), which is forbidden mid-turn because it yanks a reader scrolled into history (#1682). The renderer now pins the reported live region: during eager streaming it repaints that suffix bottom-anchored without ever committing transient rows to native scrollback (only sealed rows below the live boundary are appended), so the live block never pollutes history. Completed/sealed blocks stay scrollable mid-turn and the prompt-submit checkpoint still reconciles a clean, duplicate-free transcript. +- Fixed a viewport yank on ED3-risk hosts when a pending forced scrollback wipe (`requestRender(true, { clearScrollback: true })`, an image-budget demotion, or a checkpoint) was requested while the live-region pin was active: the pin suppressed the wipe but never consumed the flag, so a later frame where the pin disengaged (e.g. content collapsing to empty) emitted the deferred `CSI 3 J` and snapped a scrolled-up reader to the tail. The pin now consumes the flag while keeping native scrollback dirty, so the wipe is deferred to the post-stream checkpoint instead. + ## [15.9.1] - 2026-06-04 ### Fixed diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index b94625eda..b9f2a7955 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -117,6 +117,21 @@ export interface Component { invalidate(): void; } +/** + * Optional component seam for native-scrollback pinning. A component that + * renders a stable prefix followed by a live/transient suffix reports the local + * line index where that suffix begins after each render. TUI treats that suffix + * — and every root child rendered below it — as not yet safe to commit to native + * scrollback on ED3-risk terminals whose viewport position is unobservable. + */ +export interface NativeScrollbackLiveRegion { + getNativeScrollbackLiveRegionStart(): number | undefined; +} + +function getNativeScrollbackLiveRegionStart(component: Component): number | undefined { + return (component as Component & Partial).getNativeScrollbackLiveRegionStart?.(); +} + /** * Interface for components that can receive focus and display a cursor. * When focused, the component should emit CURSOR_MARKER at the cursor position @@ -317,6 +332,9 @@ export class Container implements Component { * - `historyRebuild`: a geometry change (terminal resize) left native history * wrapped at the old size — clear viewport and scrollback so it rewraps at the * new geometry. Also flushes deferred content-only rewrites. + * - `liveRegionPinned`: ED3-risk/unknown foreground stream with a reported live + * suffix — optionally append newly sealed rows, then repaint the live tail + * without letting transient rows enter native history. * - `viewportRepaint`: rewrite the visible viewport in place. If `appendFrom` * is set, emit those tail rows as scrollback growth first so streaming * output reaches terminal history before the corrected viewport is drawn. @@ -335,6 +353,7 @@ type RenderIntent = | { kind: "historyRebuild" } | { kind: "overlayRebuild" } | { kind: "viewportRepaint"; appendFrom?: number } + | { kind: "liveRegionPinned"; appendFrom: number; appendTo: number } | { kind: "deferredShrink"; paddedLength: number } | { kind: "deferredMutation" } | { kind: "shrink" } @@ -383,6 +402,7 @@ export class TUI extends Container { // Set after a clear+full replay so the next insert-above-suffix frame does // not scroll replayed live chrome (status/editor) into fresh history. #suppressNextSuffixScroll = false; + #nativeScrollbackLiveRegionStart: number | undefined; #nativeScrollbackDirty = false; #fullRedrawCount = 0; // Caps how many inline images render as live graphics; older ones fall back @@ -423,6 +443,25 @@ export class TUI extends Container { this.#showHardwareCursor = showHardwareCursor === undefined ? this.#showHardwareCursor : showHardwareCursor; } + override render(width: number): string[] { + width = Math.max(1, width); + this.#nativeScrollbackLiveRegionStart = undefined; + const lines: string[] = []; + for (const child of this.children) { + const offset = lines.length; + const childLines = child.render(width); + const liveRegionStart = getNativeScrollbackLiveRegionStart(child); + if (liveRegionStart !== undefined) { + const boundedStart = Number.isFinite(liveRegionStart) + ? Math.max(0, Math.min(childLines.length, Math.trunc(liveRegionStart))) + : childLines.length; + this.#nativeScrollbackLiveRegionStart = offset + boundedStart; + } + lines.push(...childLines); + } + return lines; + } + #syncTerminalCursorMode(component: Component | null): void { if (isFocusable(component)) { component.setUseTerminalCursor?.(this.#showHardwareCursor); @@ -1396,6 +1435,7 @@ export class TUI extends Container { visibleOverlayComponents.length > 0, overlayVisibilityReduced, allowUnknownViewportMutation, + this.#nativeScrollbackLiveRegionStart, ); if (this.#eagerNativeScrollbackRebuildDisablePending) { this.#eagerNativeScrollbackRebuildDisablePending = false; @@ -1448,6 +1488,14 @@ export class TUI extends Container { }); this.#emitViewportRepaint(lines, width, height, cursorPos); return; + case "liveRegionPinned": + // Consume any pending forced scrollback wipe: honoring it now would emit + // ED3 mid-stream and yank a reader scrolled into history. The pin keeps + // native scrollback dirty, so the post-stream checkpoint still rebuilds. + this.#clearScrollbackOnNextRender = false; + this.#emitLiveRegionPinnedRepaint(lines, width, height, cursorPos, intent.appendFrom, intent.appendTo); + this.#hasEverRendered = true; + return; case "viewportRepaint": if (intent.appendFrom !== undefined) { this.#emitAppendTail(lines, intent.appendFrom, height, width, prevViewportTop, prevHardwareCursorRow); @@ -1500,17 +1548,28 @@ export class TUI extends Container { hasVisibleOverlay: boolean, overlayVisibilityReduced: boolean, allowUnknownViewportMutation: boolean, + liveRegionStart: number | undefined, ): RenderIntent { - // Initial paint after start(): scrollback must keep its prior shell - // content, but the viewport must be cleared so stale rows do not bleed - // into the new UI. - if (!this.#hasEverRendered) return { kind: "initial" }; - - // Caller opted into a scrollback wipe via requestRender(true, { clearScrollback: true }). - if (this.#clearScrollbackOnNextRender) return { kind: "sessionReplace" }; - const forceViewportRepaint = this.#forceViewportRepaintOnNextRender; const eagerEraseScrollbackRisk = process.platform !== "win32" && TERMINAL.eagerEraseScrollbackRisk; + const liveRegionPinnedIntent = this.#planLiveRegionPinnedRender( + newLines, + height, + liveRegionStart, + eagerEraseScrollbackRisk, + ); + + // Caller opted into a scrollback wipe via requestRender(true, { clearScrollback: true }). + if (this.#clearScrollbackOnNextRender) return liveRegionPinnedIntent ?? { kind: "sessionReplace" }; + + // Initial paint after start(): scrollback must keep its prior shell + // content, but the viewport must be cleared so stale rows do not bleed + // into the new UI. If a foreground stream was already active before the + // first paint, honor its live-region pin and avoid committing transient + // rows to native history. + if (!this.#hasEverRendered) return liveRegionPinnedIntent ?? { kind: "initial" }; + if (liveRegionPinnedIntent) return liveRegionPinnedIntent; + if (overlayVisibilityReduced && !isMultiplexerSession()) { return hasVisibleOverlay ? { kind: "overlayRebuild" } : { kind: "historyRebuild" }; } @@ -2036,6 +2095,32 @@ export class TUI extends Container { ); } + #planLiveRegionPinnedRender( + newLines: string[], + height: number, + liveRegionStart: number | undefined, + eagerEraseScrollbackRisk: boolean, + ): RenderIntent | undefined { + if ( + liveRegionStart === undefined || + liveRegionStart >= newLines.length || + !this.#eagerNativeScrollbackRebuild || + !eagerEraseScrollbackRisk || + isMultiplexerSession() + ) { + return undefined; + } + if (newLines.length <= height && this.#scrollbackHighWater === 0) return undefined; + if (this.#readNativeViewportAtBottom() !== undefined) return undefined; + + this.#markNativeScrollbackDirty(); + const viewportTop = Math.max(0, newLines.length - height); + const sealedEnd = Math.max(0, Math.min(liveRegionStart, newLines.length)); + const appendTo = Math.min(sealedEnd, viewportTop); + const appendFrom = Math.min(this.#scrollbackHighWater, appendTo); + return { kind: "liveRegionPinned", appendFrom, appendTo }; + } + #padDeferredShrinkLines(lines: string[], paddedLength: number): string[] { if (lines.length >= paddedLength) return lines; return [...lines, ...new Array(paddedLength - lines.length).fill("")]; @@ -2208,6 +2293,53 @@ export class TUI extends Container { this.#commit(lines, width, height, viewportTop, toRow); } + /** + * Foreground-stream live-region paint for ED3-risk terminals with an + * unobservable viewport. Newly sealed rows are appended to native history by + * writing only that sealed chunk plus the final viewport after an ED2 viewport + * clear; transient live rows are written only into the active grid, never into + * saved lines. + */ + #emitLiveRegionPinnedRepaint( + lines: string[], + width: number, + height: number, + cursorPos: { row: number; col: number } | null, + appendFrom: number, + appendTo: number, + ): void { + this.#fullRedrawCount += 1; + const viewportTop = Math.max(0, lines.length - height); + const boundedAppendTo = Math.max(0, Math.min(appendTo, viewportTop, lines.length)); + const boundedAppendFrom = Math.max(0, Math.min(appendFrom, boundedAppendTo)); + let buffer = `${this.#paintBeginSequence}\x1b[H\x1b[2J`; + let wroteLine = false; + for (let i = boundedAppendFrom; i < boundedAppendTo; i++) { + if (wroteLine) buffer += "\r\n"; + buffer += this.#fitLineToWidth(lines[i] ?? "", width); + wroteLine = true; + } + for (let screenRow = 0; screenRow < height; screenRow++) { + if (wroteLine) buffer += "\r\n"; + buffer += this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width); + wroteLine = true; + } + const viewportBottomRow = viewportTop + height - 1; + const contentBottomRow = Math.min(viewportBottomRow, Math.max(viewportTop, lines.length - 1)); + const parkUp = viewportBottomRow - contentBottomRow; + if (parkUp > 0) buffer += `\x1b[${parkUp}A`; + const { seq, toRow } = this.#cursorControlSequence(cursorPos, lines.length, contentBottomRow); + buffer += seq; + buffer += this.#paintEndSequence; + this.terminal.write(buffer); + + this.#maxLinesRendered = lines.length; + if (boundedAppendTo > this.#scrollbackHighWater) { + this.#scrollbackHighWater = boundedAppendTo; + } + this.#commit(lines, width, height, viewportTop, toRow); + } + /** * Push the appended tail into terminal scrollback by `\r\n`-ing past the * previous viewport bottom. Used as a prefix to {@link #emitViewportRepaint} @@ -2443,9 +2575,11 @@ export class TUI extends Container { const detail = intent.kind === "diff" ? `${intent.kind}(first=${intent.firstChanged}, last=${intent.lastChanged}, appended=${intent.appendedLines})` - : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined - ? `${intent.kind}(appendFrom=${intent.appendFrom})` - : intent.kind; + : intent.kind === "liveRegionPinned" + ? `${intent.kind}(append=${intent.appendFrom}..${intent.appendTo})` + : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined + ? `${intent.kind}(appendFrom=${intent.appendFrom})` + : intent.kind; const msg = `[${new Date().toISOString()}] render: ${detail} (prev=${this.#previousLines.length}, new=${newLength}, height=${height})\n`; fs.appendFileSync(getDebugLogPath(), msg); } diff --git a/packages/tui/test/reasoning-stream-dup-repro.test.ts b/packages/tui/test/reasoning-stream-dup-repro.test.ts new file mode 100644 index 000000000..4fa3d8b4c --- /dev/null +++ b/packages/tui/test/reasoning-stream-dup-repro.test.ts @@ -0,0 +1,105 @@ +import { describe, expect, it } from "bun:test"; +import { getTerminalInfo, TERMINAL } from "../src/terminal-capabilities"; +import { type Component, type NativeScrollbackLiveRegion, TUI } from "../src/tui"; +import { VirtualTerminal } from "./virtual-terminal"; + +class LineList implements Component, NativeScrollbackLiveRegion { + #lines: string[]; + constructor(lines: string[]) { + this.#lines = lines; + } + setLines(lines: string[]): void { + this.#lines = lines; + } + invalidate(): void {} + getNativeScrollbackLiveRegionStart(): number | undefined { + return 0; + } + render(_width: number): string[] { + return this.#lines; + } +} + +async function settle(term: VirtualTerminal): Promise { + await term.waitForRender(); +} + +function overrideProbe(term: VirtualTerminal, answer: boolean | undefined): void { + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => answer; +} + +type MutableTerminalInfo = { eagerEraseScrollbackRisk: boolean }; +const MUX_KEYS = ["TMUX", "STY", "ZELLIJ"] as const; + +async function withGhostty(run: () => Promise): Promise { + const mut = TERMINAL as unknown as MutableTerminalInfo; + const prev = mut.eagerEraseScrollbackRisk; + const prevEnv: Record = {}; + for (const key of MUX_KEYS) { + prevEnv[key] = Bun.env[key]; + delete (Bun.env as Record)[key]; + } + mut.eagerEraseScrollbackRisk = getTerminalInfo("ghostty").eagerEraseScrollbackRisk; + try { + await run(); + } finally { + mut.eagerEraseScrollbackRisk = prev; + for (const key of MUX_KEYS) { + if (prevEnv[key] === undefined) delete (Bun.env as Record)[key]; + else (Bun.env as Record)[key] = prevEnv[key]; + } + } +} + +function dupNonblank(lines: string[]): string[] { + const seen = new Set(); + const dups: string[] = []; + for (const line of lines.map(l => l.trimEnd())) { + if (line.length === 0) continue; + if (seen.has(line)) dups.push(line); + seen.add(line); + } + return dups; +} + +describe("foreground-stream scrollback duplication on ED3-risk ghostty", () => { + it("does not duplicate history rows when overflowing content then shrinks", async () => { + await withGhostty(async () => { + const term = new VirtualTerminal(20, 4); + overrideProbe(term, undefined); + const tui = new TUI(term); + const list = new LineList([]); + tui.addChild(list); + try { + tui.start(); + await settle(term); + tui.setEagerNativeScrollbackRebuild(true); // foreground stream turn + + // Grow past the viewport so rows scroll into native history. + const grown = Array.from({ length: 10 }, (_v, i) => `row-${i}`); + list.setLines(grown); + tui.requestRender(); + await settle(term); + + // Re-layout shrink (e.g. preview/reasoning collapses), still overflowing. + const shrunk = Array.from({ length: 7 }, (_v, i) => `row-${i}`); + list.setLines(shrunk); + tui.requestRender(); + await settle(term); + const streamingBuffer = term.getScrollBuffer(); + expect(dupNonblank(streamingBuffer)).toEqual([]); + + tui.setEagerNativeScrollbackRebuild(false); + tui.requestRender(); + await settle(term); + expect(tui.refreshNativeScrollbackIfDirty({ allowUnknownViewport: true })).toBe(true); + await settle(term); + const checkpointBuffer = term.getScrollBuffer(); + expect(checkpointBuffer.map(line => line.trimEnd())).toEqual(shrunk); + expect(dupNonblank(checkpointBuffer)).toEqual([]); + } finally { + tui.stop(); + } + }); + }); +}); diff --git a/packages/tui/test/render-stress-harness.ts b/packages/tui/test/render-stress-harness.ts index b80d93277..8d584077f 100644 --- a/packages/tui/test/render-stress-harness.ts +++ b/packages/tui/test/render-stress-harness.ts @@ -8,6 +8,7 @@ import { type Component, CURSOR_MARKER, type Focusable, + type NativeScrollbackLiveRegion, type OverlayAnchor, type OverlayHandle, type OverlayOptions, @@ -980,18 +981,24 @@ function reflowToWidth(lines: readonly string[], width: number): string[] { return out; } -class StressComponent implements Component, Focusable { +class StressComponent implements Component, Focusable, NativeScrollbackLiveRegion { focused = false; #model: StressModel; #reflow: boolean; + #pinAsLiveRegion: boolean; - constructor(model: StressModel, reflow = false) { + constructor(model: StressModel, reflow = false, pinAsLiveRegion = false) { this.#model = model; this.#reflow = reflow; + this.#pinAsLiveRegion = pinAsLiveRegion; } invalidate(): void {} + getNativeScrollbackLiveRegionStart(): number | undefined { + return this.#pinAsLiveRegion ? 0 : undefined; + } + render(width: number): string[] { const lines = this.#model.renderedLines(width, this.focused); return this.#reflow ? reflowToWidth(lines, width) : lines; @@ -1141,7 +1148,7 @@ class StressDriver { this.#scheduler = new StressRenderScheduler(); const maxHeight = maxOf(scenario.heightChoices); this.#model = new StressModel(this.#streams.content, maxHeight + 12, scenario.uniqueContent, "root-"); - this.#component = new StressComponent(this.#model, scenario.reflow); + this.#component = new StressComponent(this.#model, scenario.reflow, this.#traits.foregroundStreaming); this.#children = [0, 1].map(id => { const model = new StressModel( this.#streams.children, @@ -1166,15 +1173,15 @@ class StressDriver { async run(): Promise { // Foreground-tool streaming faithfully: pin the ED3-risk trait (independent // of whatever real terminal hosts the worker) and keep the turn-long eager - // rebuild opt-in enabled. On an ED3-risk terminal that opt-in is gated off, - // so content frames flow through `viewportRepaint`/`diff` rather than a - // forced history rebuild — see `#renderContentFrame`. + // rebuild opt-in enabled before the initial paint. On an ED3-risk terminal + // that opt-in is gated off, so content frames flow through the live-region + // pin rather than a forced history rebuild — see `#renderContentFrame`. const terminalInfo = TERMINAL as unknown as { eagerEraseScrollbackRisk: boolean }; const savedRisk = terminalInfo.eagerEraseScrollbackRisk; if (this.#traits.foregroundStreaming) terminalInfo.eagerEraseScrollbackRisk = this.#traits.ed3ScrollbackEraseRisk; try { - this.#tui.start(); if (this.#traits.foregroundStreaming) this.#tui.setEagerNativeScrollbackRebuild(true); + this.#tui.start(); await this.#settle(); this.#assertOracles( { @@ -2034,7 +2041,7 @@ class StressDriver { forcedRender: true, mutatesViewport: true, checkpoint: true, - reconcilesNativeScrollback: true, + reconcilesNativeScrollback: !this.#traits.foregroundStreaming, }, before, after, @@ -2560,6 +2567,13 @@ class StressDriver { } #assertNoStaleOverlaySentinels(op: AppliedOperation, before: Snapshot, after: Snapshot, index: number): void { + if (this.#traits.foregroundStreaming && op.geometryChanged) { + this.#nativeScrollbackAuditBlocked = true; + return; + } + if (this.#traits.foregroundStreaming && this.#nativeScrollbackAuditBlocked && !op.reconcilesNativeScrollback) { + return; + } if (this.#hiddenOverlaySentinels.size === 0) return; const visibleSentinels = new Set( this.#overlays @@ -2580,7 +2594,6 @@ class StressDriver { } } } - #assertUniqueContentNoUnexpectedDuplicates( op: AppliedOperation, before: Snapshot, @@ -2588,6 +2601,14 @@ class StressDriver { index: number, ): void { if (!this.#scenario.uniqueContent) return; + if (this.#traits.foregroundStreaming && op.geometryChanged) { + this.#nativeScrollbackAuditBlocked = true; + return; + } + if (this.#traits.foregroundStreaming && this.#nativeScrollbackAuditBlocked) { + if (!op.reconcilesNativeScrollback) return; + this.#nativeScrollbackAuditBlocked = false; + } // Accumulate even when the check below is skipped (scrolled/overlay): the // frame's legitimate duplicates commit to scrollback regardless of where // the viewport is parked. @@ -3589,16 +3610,11 @@ function coreTemplates(): ScenarioTemplate[] { // Foreground tool actively streaming on an ED3-risk terminal whose // viewport position is unobservable (ghostty/kitty/alacritty/VTE/iTerm2; // see `detectTerminalEagerEraseScrollbackRisk`). The agent requests an - // eager native-scrollback rebuild for the streaming turn, but that opt-in - // is gated off on ED3-risk terminals, so `allowUnknownViewportMutation` - // stays false and content frames flow through `viewportRepaint`/`diff` - // instead of a forced history rebuild. An offscreen-edit growth then - // repaints in place — advancing the rendered line count without committing - // the overflow to native history — and the next shrink must still - // re-anchor the bottom of the viewport from that lagging high-water mark. - // The default content-frame path forces `allowUnknownViewportMutation` and - // never reaches this state (a notification chip rendering over the active - // tool render: the original report). + // eager native-scrollback rebuild for the streaming turn, but on ED3-risk + // terminals the live-region seam pins all foreground-stream rows to the + // active viewport instead of letting them enter native history. The + // duplicate oracle is enabled here (`uniqueContent`) so any later shrink + // or reflow that re-exposes a committed live row fails loudly. name: "darwin-unknown-ghostty-stream-small", platform: "darwin", terminalMode: "unknown", @@ -3610,6 +3626,7 @@ function coreTemplates(): ScenarioTemplate[] { heightChoices: [3, 4, 6], scrollbackRows: 10_000, foregroundStream: true, + uniqueContent: true, }, { name: "linux-unknown-ghostty-stream-large", @@ -3623,6 +3640,7 @@ function coreTemplates(): ScenarioTemplate[] { heightChoices: [8, 12, 24], scrollbackRows: 10_000, foregroundStream: true, + uniqueContent: true, }, { // Width-reflowing content (wrapped/markdown-style) uses the same grapheme @@ -3657,6 +3675,7 @@ function coreTemplates(): ScenarioTemplate[] { scrollbackRows: 10_000, reflow: true, foregroundStream: true, + uniqueContent: true, }, ]; } From 52c58d9ad06f7cae30c0a5ef5b30326dde29fee6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 01:20:56 +0200 Subject: [PATCH 11/48] test(tui): added regressions for pinned live region scrollback - Asserted no full-screen erase (ED2/ED3) while pinning live region. - Verified a scrolled reader's viewport stays fixed during streaming. --- .../test/reasoning-stream-dup-repro.test.ts | 94 +++++++++++++++++++ 1 file changed, 94 insertions(+) diff --git a/packages/tui/test/reasoning-stream-dup-repro.test.ts b/packages/tui/test/reasoning-stream-dup-repro.test.ts index 4fa3d8b4c..03db59619 100644 --- a/packages/tui/test/reasoning-stream-dup-repro.test.ts +++ b/packages/tui/test/reasoning-stream-dup-repro.test.ts @@ -20,6 +20,22 @@ class LineList implements Component, NativeScrollbackLiveRegion { } } +// A non-live (sealed) block: it does NOT report a live region, so the TUI treats +// it as committable native scrollback below the live boundary. +class LineListSealed implements Component { + #lines: string[]; + constructor(lines: string[]) { + this.#lines = lines; + } + setLines(lines: string[]): void { + this.#lines = lines; + } + invalidate(): void {} + render(_width: number): string[] { + return this.#lines; + } +} + async function settle(term: VirtualTerminal): Promise { await term.waitForRender(); } @@ -102,4 +118,82 @@ describe("foreground-stream scrollback duplication on ED3-risk ghostty", () => { } }); }); + + it("never emits a full-screen erase (ED2/ED3) while pinning the live region", async () => { + await withGhostty(async () => { + const term = new VirtualTerminal(20, 4); + overrideProbe(term, undefined); + let captured = ""; + const realWrite = term.write.bind(term); + (term as unknown as { write: (s: string) => void }).write = (s: string) => { + captured += s; + realWrite(s); + }; + const tui = new TUI(term); + const list = new LineList([]); + tui.addChild(list); + try { + tui.start(); + await settle(term); + tui.setEagerNativeScrollbackRebuild(true); + captured = ""; // focus on the streaming frames + + // Grow the live block well past the viewport, one line at a time, then + // re-layout shrink — the exact pattern that snapped the view before. + const acc: string[] = []; + for (let n = 1; n <= 24; n++) { + acc.push(`line-${n}`); + list.setLines([...acc]); + tui.requestRender(); + await settle(term); + } + list.setLines(acc.slice(0, 18)); + tui.requestRender(); + await settle(term); + + // The pin must repaint incrementally. A full-screen erase (ED2) or a + // scrollback wipe (ED3) snaps a scrolled-up Ghostty reader to the tail. + expect(captured.includes("\x1b[2J")).toBe(false); + expect(captured.includes("\x1b[3J")).toBe(false); + // Sanity: the pin actually ran (otherwise the assertions are vacuous). + expect(tui.fullRedraws).toBeGreaterThan(0); + } finally { + tui.stop(); + } + }); + }); + + it("keeps a scrolled reader's viewport fixed while the live region streams", async () => { + await withGhostty(async () => { + const term = new VirtualTerminal(20, 4); + overrideProbe(term, undefined); + const tui = new TUI(term); + const sealed = new LineListSealed(Array.from({ length: 12 }, (_v, i) => `prior-${i}`)); + const live = new LineList([]); + tui.addChild(sealed); + tui.addChild(live); + try { + tui.start(); + await settle(term); + // Commit the prior conversation to native history (checkpoint), then + // begin a streaming turn and scroll up into that history. + expect(tui.refreshNativeScrollbackIfDirty({ allowUnknownViewport: true })).toBe(false); + tui.setEagerNativeScrollbackRebuild(true); + live.setLines(Array.from({ length: 6 }, (_v, i) => `think-${i}`)); + tui.requestRender(); + await settle(term); + term.scrollLines(-3); // user scrolls up into history + const before = term.getBufferPosition().viewportY; + for (let n = 7; n <= 20; n++) { + live.setLines(Array.from({ length: n }, (_v, i) => `think-${i}`)); + tui.requestRender(); + await settle(term); + } + // Streaming output must not drag the scrolled viewport down. + expect(term.getBufferPosition().viewportY).toBe(before); + } finally { + tui.stop(); + } + }); + }); }); From f863d98f581e203f9a728cf445f60fcdaf1fb86d Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 01:21:04 +0200 Subject: [PATCH 12/48] fix(tui): pinned live region repaint without screen-wide erase - Replaced full-screen clear (ED2) and absolute home with relative cursor moves and per-line `\x1b[2K`. - Threaded prior viewport top and hardware cursor row to compute the relative move. - Stopped Ghostty from snapping a scrolled-back reader to the bottom on every live-tail frame. --- packages/tui/src/tui.ts | 46 ++++++++++++++++++++++++++++++++++------- 1 file changed, 38 insertions(+), 8 deletions(-) diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index b9f2a7955..bcb565ae4 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1493,7 +1493,16 @@ export class TUI extends Container { // ED3 mid-stream and yank a reader scrolled into history. The pin keeps // native scrollback dirty, so the post-stream checkpoint still rebuilds. this.#clearScrollbackOnNextRender = false; - this.#emitLiveRegionPinnedRepaint(lines, width, height, cursorPos, intent.appendFrom, intent.appendTo); + this.#emitLiveRegionPinnedRepaint( + lines, + width, + height, + cursorPos, + intent.appendFrom, + intent.appendTo, + prevViewportTop, + prevHardwareCursorRow, + ); this.#hasEverRendered = true; return; case "viewportRepaint": @@ -2295,10 +2304,15 @@ export class TUI extends Container { /** * Foreground-stream live-region paint for ED3-risk terminals with an - * unobservable viewport. Newly sealed rows are appended to native history by - * writing only that sealed chunk plus the final viewport after an ED2 viewport - * clear; transient live rows are written only into the active grid, never into - * saved lines. + * unobservable viewport. Commits the newly-sealed chunk to native scrollback + * (so finished blocks stay scrollable) and repaints the live tail in place, + * leaving the transient live region out of saved lines. + * + * Uses only the no-scroll-snap vocabulary of {@link #emitDiff}: relative + * cursor moves, per-line `\x1b[2K`, and `\r\n` to push the sealed chunk into + * history. It deliberately avoids a full-screen erase (`\x1b[2J`) and absolute + * cursor home (`\x1b[H`): on Ghostty those snap a reader scrolled into history + * back to the bottom on every frame (the live tail repaints every token). */ #emitLiveRegionPinnedRepaint( lines: string[], @@ -2307,23 +2321,39 @@ export class TUI extends Container { cursorPos: { row: number; col: number } | null, appendFrom: number, appendTo: number, + prevViewportTop: number, + prevHardwareCursorRow: number, ): void { this.#fullRedrawCount += 1; const viewportTop = Math.max(0, lines.length - height); const boundedAppendTo = Math.max(0, Math.min(appendTo, viewportTop, lines.length)); const boundedAppendFrom = Math.max(0, Math.min(appendFrom, boundedAppendTo)); - let buffer = `${this.#paintBeginSequence}\x1b[H\x1b[2J`; + + // Position at the top visible row with a relative move. Terminals clamp the + // hardware cursor to the viewport on resize, so clamp our tracking to match + // before computing the delta (mirrors #emitDiff). + const clampedCursor = Math.min(prevHardwareCursorRow, prevViewportTop + height - 1); + const currentScreenRow = Math.max(0, Math.min(height - 1, clampedCursor - prevViewportTop)); + let buffer = this.#paintBeginSequence; + if (currentScreenRow > 0) buffer += `\x1b[${currentScreenRow}A`; + buffer += "\r"; + + // Write the sealed chunk followed by the full viewport from the top row. + // The first (boundedAppendTo - boundedAppendFrom) rows scroll into native + // history; the trailing `height` rows fill the viewport. Each row clears + // itself with `\x1b[2K` instead of relying on a screen-wide erase. let wroteLine = false; for (let i = boundedAppendFrom; i < boundedAppendTo; i++) { if (wroteLine) buffer += "\r\n"; - buffer += this.#fitLineToWidth(lines[i] ?? "", width); + buffer += `\x1b[2K${this.#fitLineToWidth(lines[i] ?? "", width)}`; wroteLine = true; } for (let screenRow = 0; screenRow < height; screenRow++) { if (wroteLine) buffer += "\r\n"; - buffer += this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width); + buffer += `\x1b[2K${this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width)}`; wroteLine = true; } + const viewportBottomRow = viewportTop + height - 1; const contentBottomRow = Math.min(viewportBottomRow, Math.max(viewportTop, lines.length - 1)); const parkUp = viewportBottomRow - contentBottomRow; From a1da2a8f011b9ecf89f75890528254c7686c9d8a Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 02:35:33 +0200 Subject: [PATCH 13/48] fix(tui): skipped live-region pin on resize to avoid stale rows - Threaded a geometry-changed flag into the pinned emitter. - Deferred width/height changes to the reflow rebuild path, since relative cursor moves from pre-resize geometry landed on wrong rows. --- packages/tui/CHANGELOG.md | 3 ++- packages/tui/src/tui.ts | 7 +++++++ 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index af8aa2e50..f20635dba 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -8,8 +8,9 @@ ### Fixed -- Fixed persistent line duplication in native scrollback while a foreground turn streamed on ED3-risk terminals with an unobservable viewport. The bottom-most live block scrolled its overflow into native history during growth; when that same block later re-laid-out, shrank, or collapsed (a tool preview collapsing to its compact result, a Markdown list/table reflow, a reasoning re-wrap) the viewport repaint correctly showed the new tail, but the stale overflow rows stayed in committed history and re-appeared above the live region — and they cannot be removed without ED3 (`CSI 3 J`), which is forbidden mid-turn because it yanks a reader scrolled into history (#1682). The renderer now pins the reported live region: during eager streaming it repaints that suffix bottom-anchored without ever committing transient rows to native scrollback (only sealed rows below the live boundary are appended), so the live block never pollutes history. Completed/sealed blocks stay scrollable mid-turn and the prompt-submit checkpoint still reconciles a clean, duplicate-free transcript. +- Fixed persistent line duplication in native scrollback while a foreground turn streamed on ED3-risk terminals with an unobservable viewport. The bottom-most live block scrolled its overflow into native history during growth; when that same block later re-laid-out, shrank, or collapsed (a tool preview collapsing to its compact result, a Markdown list/table reflow, a reasoning re-wrap) the viewport repaint correctly showed the new tail, but the stale overflow rows stayed in committed history and re-appeared above the live region — and they cannot be removed without ED3 (`CSI 3 J`), which is forbidden mid-turn because it yanks a reader scrolled into history (#1682). The renderer now pins the reported live region: during eager streaming it repaints that suffix in place — incrementally, using relative cursor moves and per-line erases (`CSI 2 K`), never a full-screen erase (`CSI 2 J`) or absolute cursor home — without ever committing transient rows to native scrollback (only sealed rows below the live boundary are appended). Completed/sealed blocks stay scrollable mid-turn and the prompt-submit checkpoint still reconciles a clean, duplicate-free transcript. - Fixed a viewport yank on ED3-risk hosts when a pending forced scrollback wipe (`requestRender(true, { clearScrollback: true })`, an image-budget demotion, or a checkpoint) was requested while the live-region pin was active: the pin suppressed the wipe but never consumed the flag, so a later frame where the pin disengaged (e.g. content collapsing to empty) emitted the deferred `CSI 3 J` and snapped a scrolled-up reader to the tail. The pin now consumes the flag while keeping native scrollback dirty, so the wipe is deferred to the post-stream checkpoint instead. +- Fixed a terminal resize not re-rendering (stale, overlapping rows) during a foreground stream on ED3-risk hosts. The live-region pin fired on width/height changes too, repainting with relative cursor moves computed from the pre-resize geometry — after the terminal reflowed its grid those landed on the wrong rows. Resizes now skip the pin and take the geometry reflow path, which repaints the viewport at the new dimensions. ## [15.9.1] - 2026-06-04 ### Fixed diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index bcb565ae4..7d9867463 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1566,6 +1566,7 @@ export class TUI extends Container { height, liveRegionStart, eagerEraseScrollbackRisk, + widthChanged || heightChanged, ); // Caller opted into a scrollback wipe via requestRender(true, { clearScrollback: true }). @@ -2109,12 +2110,18 @@ export class TUI extends Container { height: number, liveRegionStart: number | undefined, eagerEraseScrollbackRisk: boolean, + geometryChanged: boolean, ): RenderIntent | undefined { + // A width/height change reflows the whole terminal: the relative cursor + // positioning this emitter relies on is computed from the pre-resize + // geometry and would land on the wrong rows. Defer to the geometry branch + // (a full reflow rebuild), which is the established behavior for resizes. if ( liveRegionStart === undefined || liveRegionStart >= newLines.length || !this.#eagerNativeScrollbackRebuild || !eagerEraseScrollbackRisk || + geometryChanged || isMultiplexerSession() ) { return undefined; From 124e5593e930964813186fbcaef57c710a5bbae9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 03:41:51 +0000 Subject: [PATCH 14/48] fix(coding-agent): resolved omp docs urls Accepted omp://docs as the embedded documentation root and mapped docs-prefixed paths to the generated documentation index. Added regression coverage for omp://docs and omp://docs/tools/read.md resolution. Fixes #1898 --- packages/coding-agent/CHANGELOG.md | 2 ++ .../src/internal-urls/omp-protocol.ts | 10 ++++++++-- .../test/internal-urls/omp-protocol.test.ts | 20 +++++++++++++++++++ 3 files changed, 30 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/test/internal-urls/omp-protocol.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..0e27e5fef 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -19,6 +19,8 @@ ### Fixed +- Fixed `omp://docs` and `omp://docs/...` internal documentation URLs in the distributed package to resolve through the embedded documentation index instead of failing with `Documentation file not found` ([#1898](https://github.com/can1357/oh-my-pi/issues/1898)). + - Fixed a streamed assistant message freezing at a partial prefix (e.g. only "Nat" of "Natives built, now…") on ED3-risk terminals (Ghostty/kitty/iTerm2/Alacritty), with the final text appearing only after a resize. `TranscriptContainer` freezes each non-live block by replaying its last live render, but render coalescing can finalize a block's content and append the next block within the same throttled frame — so the block was sealed at its stale mid-stream snapshot and never repainted until the next `thaw`. The block that was live on the previous render is now recomputed once on the live→frozen transition, sealing it at its final content. - Fixed ACP/RPC stdio startup so protocol frames are no longer consumed as one-shot piped prompt input before the JSON-RPC transport starts. diff --git a/packages/coding-agent/src/internal-urls/omp-protocol.ts b/packages/coding-agent/src/internal-urls/omp-protocol.ts index a36c29edf..ee5e13534 100644 --- a/packages/coding-agent/src/internal-urls/omp-protocol.ts +++ b/packages/coding-agent/src/internal-urls/omp-protocol.ts @@ -64,9 +64,15 @@ export class OmpProtocolHandler implements ProtocolHandler { throw new Error("Path traversal (..) is not allowed in omp:// URLs"); } - const content = EMBEDDED_DOCS[normalized]; + const docPath = + normalized === "docs" ? "" : normalized.startsWith("docs/") ? normalized.slice("docs/".length) : normalized; + if (!docPath) { + return this.#listDocs(url); + } + + const content = EMBEDDED_DOCS[docPath]; if (content === undefined) { - const lookup = normalized.replace(/\.md$/, ""); + const lookup = docPath.replace(/\.md$/, ""); const suggestions = EMBEDDED_DOC_FILENAMES.filter( f => f.includes(lookup) || lookup.includes(f.replace(/\.md$/, "")), ).slice(0, 5); diff --git a/packages/coding-agent/test/internal-urls/omp-protocol.test.ts b/packages/coding-agent/test/internal-urls/omp-protocol.test.ts new file mode 100644 index 000000000..d779bf393 --- /dev/null +++ b/packages/coding-agent/test/internal-urls/omp-protocol.test.ts @@ -0,0 +1,20 @@ +import { describe, expect, it } from "bun:test"; +import { InternalUrlRouter } from "@oh-my-pi/pi-coding-agent/internal-urls"; + +describe("OmpProtocolHandler", () => { + it("treats omp://docs as the documentation root", async () => { + const resource = await InternalUrlRouter.instance().resolve("omp://docs"); + + expect(resource.content).toContain("# Documentation"); + expect(resource.content).toContain("tools/read.md"); + }); + + it("resolves docs-prefixed documentation paths", async () => { + const router = InternalUrlRouter.instance(); + const direct = await router.resolve("omp://tools/read.md"); + const prefixed = await router.resolve("omp://docs/tools/read.md"); + + expect(prefixed.content).toBe(direct.content); + expect(prefixed.content).toContain("# read"); + }); +}); From 61c26af62aeb109857b11c72b74d27ac240ed6e9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 03:44:50 +0000 Subject: [PATCH 15/48] fix(coding-agent): expanded omp docs search alias Handled omp://docs as an embedded documentation search root alongside omp://. Added regression coverage for searching embedded docs through the docs alias. Fixes #1898 --- packages/coding-agent/src/tools/search.ts | 2 +- .../test/tools/search-internal-urls.test.ts | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index d36f04fcc..221b10797 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -259,7 +259,7 @@ interface IndexedContentLines { } const INTERNAL_URL_DISPLAY_RE = /^[a-z][a-z0-9+.-]*:\/\//i; -const OMP_ROOT_URL_RE = /^omp:\/\/\/?$/i; +const OMP_ROOT_URL_RE = /^omp:\/\/(?:\/?|docs\/?)$/i; function normalizeSearchLine(line: string): string { return line.endsWith("\r") ? line.slice(0, -1) : line; diff --git a/packages/coding-agent/test/tools/search-internal-urls.test.ts b/packages/coding-agent/test/tools/search-internal-urls.test.ts index a3383faee..52edda3b5 100644 --- a/packages/coding-agent/test/tools/search-internal-urls.test.ts +++ b/packages/coding-agent/test/tools/search-internal-urls.test.ts @@ -183,6 +183,20 @@ describe("SearchTool internal URL resolution", () => { expect(text).toContain("Search file contents with a regex across files"); }); + it("expands omp://docs to grep embedded documentation files", async () => { + const session = createSession(); + const tool = new SearchTool(session); + + const result = await tool.execute("test-call", { + pattern: "Read files, directories, archives", + paths: ["omp://docs"], + }); + + const text = getResultText(result); + expect(text).toContain("# omp://tools/read.md"); + expect(text).toContain("Read files, directories, archives"); + }); + it("throws when internal URL has no sourcePath", async () => { const session = createSession(); const tool = new SearchTool(session); From 42146e3f1494f3d127d9b1d6f16c627332a3da47 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 03:47:46 +0000 Subject: [PATCH 16/48] docs(coding-agent): moved omp docs changelog entry to unreleased Followed the project changelog convention so the next release inherits the fix note instead of mutating the immutable 15.9.1 section. Fixes #1898 --- packages/coding-agent/CHANGELOG.md | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 0e27e5fef..7a32a9df4 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `omp://docs` and `omp://docs/...` internal documentation URLs in the distributed package to resolve through the embedded documentation index instead of failing with `Documentation file not found` ([#1898](https://github.com/can1357/oh-my-pi/issues/1898)). + ## [15.9.1] - 2026-06-04 ### Added @@ -19,8 +23,6 @@ ### Fixed -- Fixed `omp://docs` and `omp://docs/...` internal documentation URLs in the distributed package to resolve through the embedded documentation index instead of failing with `Documentation file not found` ([#1898](https://github.com/can1357/oh-my-pi/issues/1898)). - - Fixed a streamed assistant message freezing at a partial prefix (e.g. only "Nat" of "Natives built, now…") on ED3-risk terminals (Ghostty/kitty/iTerm2/Alacritty), with the final text appearing only after a resize. `TranscriptContainer` freezes each non-live block by replaying its last live render, but render coalescing can finalize a block's content and append the next block within the same throttled frame — so the block was sealed at its stale mid-stream snapshot and never repainted until the next `thaw`. The block that was live on the previous render is now recomputed once on the live→frozen transition, sealing it at its final content. - Fixed ACP/RPC stdio startup so protocol frames are no longer consumed as one-shot piped prompt input before the JSON-RPC transport starts. From 4e54836f944d492005ff348f6222ce93adcbd031 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 06:42:05 +0200 Subject: [PATCH 17/48] chore(tui): adopt #1895 streaming-scrollback-defer; drop live-region pin --- packages/coding-agent/CHANGELOG.md | 4 - .../modes/components/transcript-container.ts | 14 +- .../components/transcript-container.test.ts | 10 - packages/tui/CHANGELOG.md | 11 +- packages/tui/src/tui.ts | 263 ++++++------------ .../test/reasoning-stream-dup-repro.test.ts | 199 ------------- packages/tui/test/render-stress-harness.ts | 57 ++-- .../test/streaming-scrollback-defer.test.ts | 157 +++++++++++ 8 files changed, 261 insertions(+), 454 deletions(-) delete mode 100644 packages/tui/test/reasoning-stream-dup-repro.test.ts create mode 100644 packages/tui/test/streaming-scrollback-defer.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index cb7fd45d6..8846e99e6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,10 +2,6 @@ ## [Unreleased] -### Changed - -- Changed `TranscriptContainer` to report its live-region boundary to the renderer (`NativeScrollbackLiveRegion`): on ED3-risk terminals it now exposes the line offset where the bottom-most live block begins, so the TUI pins that block and the chrome below it out of native scrollback during streaming instead of committing its overflow and then leaving stale duplicates when the block re-lays-out or collapses. - ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/modes/components/transcript-container.ts b/packages/coding-agent/src/modes/components/transcript-container.ts index c4fea9f7f..64af9ce42 100644 --- a/packages/coding-agent/src/modes/components/transcript-container.ts +++ b/packages/coding-agent/src/modes/components/transcript-container.ts @@ -1,4 +1,4 @@ -import { type Component, Container, type NativeScrollbackLiveRegion, TERMINAL } from "@oh-my-pi/pi-tui"; +import { type Component, Container, TERMINAL } from "@oh-my-pi/pi-tui"; const kSnapshot = Symbol("transcript.frozenRender"); @@ -34,14 +34,10 @@ interface SnapshotCarrier { * and any drift reconciles safely. On terminals that can rebuild history this * freezing is unnecessary, so it renders every block live for full fidelity. */ -export class TranscriptContainer extends Container implements NativeScrollbackLiveRegion { +export class TranscriptContainer extends Container { // Bumped to invalidate every block's snapshot at once; a snapshot is only // honored when its stored generation still matches. #generation = 0; - // Local line index where the current bottom-most block begins in the most - // recent render. TUI extends the native-scrollback pinned region from this - // point through the live block and the root chrome rendered below it. - #nativeScrollbackLiveRegionStart: number | undefined; // The block that was bottom-most (live) on the previous render. When the live // position moves past it, its snapshot was last refreshed mid-stream and may // predate content that finalized in the same coalesced frame that appended the @@ -60,10 +56,6 @@ export class TranscriptContainer extends Container implements NativeScrollbackLi super.clear(); } - getNativeScrollbackLiveRegionStart(): number | undefined { - return this.#nativeScrollbackLiveRegionStart; - } - /** * Retire all frozen snapshots so the next render reflects each block's current * state. Call at reconciliation checkpoints (prompt submit) where the whole @@ -76,7 +68,6 @@ export class TranscriptContainer extends Container implements NativeScrollbackLi override render(width: number): string[] { width = Math.max(1, width); - this.#nativeScrollbackLiveRegionStart = undefined; if (!TERMINAL.eagerEraseScrollbackRisk) return super.render(width); const lines: string[] = []; @@ -85,7 +76,6 @@ export class TranscriptContainer extends Container implements NativeScrollbackLi const prevLiveChild = this.#prevLiveChild; this.#prevLiveChild = liveChild; for (let i = 0; i < this.children.length; i++) { - if (i === liveIndex) this.#nativeScrollbackLiveRegionStart = lines.length; const child = this.children[i]! as Component & SnapshotCarrier; if (child !== liveChild) { const snapshot = child[kSnapshot]; diff --git a/packages/coding-agent/test/modes/components/transcript-container.test.ts b/packages/coding-agent/test/modes/components/transcript-container.test.ts index 448c9b3a4..5ff34729a 100644 --- a/packages/coding-agent/test/modes/components/transcript-container.test.ts +++ b/packages/coding-agent/test/modes/components/transcript-container.test.ts @@ -53,16 +53,6 @@ describe("TranscriptContainer", () => { expect(container.render(40)).toEqual(["a2", "b2"]); }); - it("reports the live-region boundary at the bottom-most block on ED3-risk terminals", () => { - riskFlag.eagerEraseScrollbackRisk = true; - const container = new TranscriptContainer(); - container.addChild(new MutableBlock(["a1", "a2"])); - container.addChild(new MutableBlock(["b1"])); - - expect(container.render(40)).toEqual(["a1", "a2", "b1"]); - expect(container.getNativeScrollbackLiveRegionStart()).toBe(2); - }); - it("seals the prior block at its final content when finalize+append coalesce (ED3-risk)", () => { riskFlag.eagerEraseScrollbackRisk = true; const container = new TranscriptContainer(); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index f20635dba..7f6a7f9fb 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -1,16 +1,9 @@ # Changelog ## [Unreleased] +### Changed -### Added - -- Added a `NativeScrollbackLiveRegion` component seam (`getNativeScrollbackLiveRegionStart()`): a component reports the local line index where its live/transient suffix begins, and `TUI` treats that suffix — plus every root child rendered below it — as not yet safe to commit to native scrollback on ED3-risk terminals whose viewport position is unobservable (Ghostty/kitty/Alacritty/VTE/iTerm2 on POSIX). - -### Fixed - -- Fixed persistent line duplication in native scrollback while a foreground turn streamed on ED3-risk terminals with an unobservable viewport. The bottom-most live block scrolled its overflow into native history during growth; when that same block later re-laid-out, shrank, or collapsed (a tool preview collapsing to its compact result, a Markdown list/table reflow, a reasoning re-wrap) the viewport repaint correctly showed the new tail, but the stale overflow rows stayed in committed history and re-appeared above the live region — and they cannot be removed without ED3 (`CSI 3 J`), which is forbidden mid-turn because it yanks a reader scrolled into history (#1682). The renderer now pins the reported live region: during eager streaming it repaints that suffix in place — incrementally, using relative cursor moves and per-line erases (`CSI 2 K`), never a full-screen erase (`CSI 2 J`) or absolute cursor home — without ever committing transient rows to native scrollback (only sealed rows below the live boundary are appended). Completed/sealed blocks stay scrollable mid-turn and the prompt-submit checkpoint still reconciles a clean, duplicate-free transcript. -- Fixed a viewport yank on ED3-risk hosts when a pending forced scrollback wipe (`requestRender(true, { clearScrollback: true })`, an image-budget demotion, or a checkpoint) was requested while the live-region pin was active: the pin suppressed the wipe but never consumed the flag, so a later frame where the pin disengaged (e.g. content collapsing to empty) emitted the deferred `CSI 3 J` and snapped a scrolled-up reader to the tail. The pin now consumes the flag while keeping native scrollback dirty, so the wipe is deferred to the post-stream checkpoint instead. -- Fixed a terminal resize not re-rendering (stale, overlapping rows) during a foreground stream on ED3-risk hosts. The live-region pin fired on width/height changes too, repainting with relative cursor moves computed from the pre-resize geometry — after the terminal reflowed its grid those landed on the wrong rows. Resizes now skip the pin and take the geometry reflow path, which repaints the viewport at the new dimensions. +- Changed streaming frames to cap rendered content to viewport height and suppress `\r\n` scrolling, so intermediate tool output never enters terminal native scrollback. Rows are committed once via `historyRebuild` when the tool finishes (or via the stable-prefix commit path for already-stabilized content). ([#NNNN](https://github.com/can1357/oh-my-pi/pull/NNNN)) ## [15.9.1] - 2026-06-04 ### Fixed diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 7d9867463..6059cfa0c 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -117,21 +117,6 @@ export interface Component { invalidate(): void; } -/** - * Optional component seam for native-scrollback pinning. A component that - * renders a stable prefix followed by a live/transient suffix reports the local - * line index where that suffix begins after each render. TUI treats that suffix - * — and every root child rendered below it — as not yet safe to commit to native - * scrollback on ED3-risk terminals whose viewport position is unobservable. - */ -export interface NativeScrollbackLiveRegion { - getNativeScrollbackLiveRegionStart(): number | undefined; -} - -function getNativeScrollbackLiveRegionStart(component: Component): number | undefined { - return (component as Component & Partial).getNativeScrollbackLiveRegionStart?.(); -} - /** * Interface for components that can receive focus and display a cursor. * When focused, the component should emit CURSOR_MARKER at the cursor position @@ -332,9 +317,6 @@ export class Container implements Component { * - `historyRebuild`: a geometry change (terminal resize) left native history * wrapped at the old size — clear viewport and scrollback so it rewraps at the * new geometry. Also flushes deferred content-only rewrites. - * - `liveRegionPinned`: ED3-risk/unknown foreground stream with a reported live - * suffix — optionally append newly sealed rows, then repaint the live tail - * without letting transient rows enter native history. * - `viewportRepaint`: rewrite the visible viewport in place. If `appendFrom` * is set, emit those tail rows as scrollback growth first so streaming * output reaches terminal history before the corrected viewport is drawn. @@ -353,7 +335,6 @@ type RenderIntent = | { kind: "historyRebuild" } | { kind: "overlayRebuild" } | { kind: "viewportRepaint"; appendFrom?: number } - | { kind: "liveRegionPinned"; appendFrom: number; appendTo: number } | { kind: "deferredShrink"; paddedLength: number } | { kind: "deferredMutation" } | { kind: "shrink" } @@ -402,8 +383,17 @@ export class TUI extends Container { // Set after a clear+full replay so the next insert-above-suffix frame does // not scroll replayed live chrome (status/editor) into fresh history. #suppressNextSuffixScroll = false; - #nativeScrollbackLiveRegionStart: number | undefined; #nativeScrollbackDirty = false; + // Highest `#maxLinesRendered` reached during a foreground tool turn while + // intermediate frames were prevented from committing to terminal scrollback. + // Used after the tool finishes to push the settled content into scrollback + // via a non-destructive full paint (no ED 3). Reset to 0 once rows are + // committed (via any `#emitFullPaint`, `#emitDiff`, or `#emitAppendTail` + // path). + #streamingHighWater = 0; + // Tracks whether the previous frame was inside a foreground tool streaming + // turn. Used to reset `#streamingHighWater` on fresh streaming starts. + #previousStreamingActive = false; #fullRedrawCount = 0; // Caps how many inline images render as live graphics; older ones fall back // to text via a purge + full redraw. Cap is configured by the host app. @@ -443,25 +433,6 @@ export class TUI extends Container { this.#showHardwareCursor = showHardwareCursor === undefined ? this.#showHardwareCursor : showHardwareCursor; } - override render(width: number): string[] { - width = Math.max(1, width); - this.#nativeScrollbackLiveRegionStart = undefined; - const lines: string[] = []; - for (const child of this.children) { - const offset = lines.length; - const childLines = child.render(width); - const liveRegionStart = getNativeScrollbackLiveRegionStart(child); - if (liveRegionStart !== undefined) { - const boundedStart = Number.isFinite(liveRegionStart) - ? Math.max(0, Math.min(childLines.length, Math.trunc(liveRegionStart))) - : childLines.length; - this.#nativeScrollbackLiveRegionStart = offset + boundedStart; - } - lines.push(...childLines); - } - return lines; - } - #syncTerminalCursorMode(component: Component | null): void { if (isFocusable(component)) { component.setUseTerminalCursor?.(this.#showHardwareCursor); @@ -1426,7 +1397,7 @@ export class TUI extends Container { this.#allowUnknownViewportMutationOnNextRender = false; // 3. Classify intent. - const intent = this.#planRender( + let intent = this.#planRender( lines, widthChanged, heightChanged, @@ -1435,8 +1406,54 @@ export class TUI extends Container { visibleOverlayComponents.length > 0, overlayVisibilityReduced, allowUnknownViewportMutation, - this.#nativeScrollbackLiveRegionStart, ); + // 3b. During foreground tool streaming, suppress any destructive scrollback + // commit. Intermediate frames repaint the viewport in place so transient + // streaming output never enters terminal native scrollback. Track the peak + // row count so the settled content can be pushed into scrollback later. + const streamingWasActive = this.#eagerNativeScrollbackRebuild; + if (streamingWasActive && !this.#previousStreamingActive) { + this.#streamingHighWater = 0; + } + this.#previousStreamingActive = streamingWasActive; + if (streamingWasActive) { + const streamingActive = + this.#eagerNativeScrollbackRebuild && !this.#eagerNativeScrollbackRebuildDisablePending; + const streamingJustEnded = !streamingActive && streamingWasActive; + if (streamingJustEnded) { + // Streaming just ended. Commit the full content to scrollback. + // On safe terminals use ED 3 + re-emit; on ED3-risk terminals + // (VTE, Windows) use a non-destructive viewport repaint — + // the shrink-across-high-water paths handle cleanup later. + this.#streamingHighWater = 0; + this.#clearScrollbackOnNextRender = false; + this.#clearNativeScrollbackDirty(); + if (!eagerEraseScrollbackRisk) { + intent = { kind: "historyRebuild" }; + } + // On ED3-risk the intent stays as planRender returned (e.g. noop + // if content unchanged) — the capped state persists until shrink. + } else if (streamingActive) { + if ( + intent.kind === "sessionReplace" || + intent.kind === "historyRebuild" || + intent.kind === "overlayRebuild" || + (intent.kind === "diff" && intent.appendedLines) + ) { + this.#clearScrollbackOnNextRender = false; + this.#clearNativeScrollbackDirty(); + // Cap lines to viewport height during streaming so frames + // never grow #previousLines past the visible area. + this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); + this.#scrollbackHighWater = 0; + lines = lines.slice(-height); + intent = { kind: "viewportRepaint" }; + } else { + // Frame wasn't suppressed (e.g. noop), but still track peak. + this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); + } + } + } if (this.#eagerNativeScrollbackRebuildDisablePending) { this.#eagerNativeScrollbackRebuildDisablePending = false; this.#eagerNativeScrollbackRebuild = false; @@ -1488,23 +1505,6 @@ export class TUI extends Container { }); this.#emitViewportRepaint(lines, width, height, cursorPos); return; - case "liveRegionPinned": - // Consume any pending forced scrollback wipe: honoring it now would emit - // ED3 mid-stream and yank a reader scrolled into history. The pin keeps - // native scrollback dirty, so the post-stream checkpoint still rebuilds. - this.#clearScrollbackOnNextRender = false; - this.#emitLiveRegionPinnedRepaint( - lines, - width, - height, - cursorPos, - intent.appendFrom, - intent.appendTo, - prevViewportTop, - prevHardwareCursorRow, - ); - this.#hasEverRendered = true; - return; case "viewportRepaint": if (intent.appendFrom !== undefined) { this.#emitAppendTail(lines, intent.appendFrom, height, width, prevViewportTop, prevHardwareCursorRow); @@ -1557,29 +1557,17 @@ export class TUI extends Container { hasVisibleOverlay: boolean, overlayVisibilityReduced: boolean, allowUnknownViewportMutation: boolean, - liveRegionStart: number | undefined, ): RenderIntent { - const forceViewportRepaint = this.#forceViewportRepaintOnNextRender; - const eagerEraseScrollbackRisk = process.platform !== "win32" && TERMINAL.eagerEraseScrollbackRisk; - const liveRegionPinnedIntent = this.#planLiveRegionPinnedRender( - newLines, - height, - liveRegionStart, - eagerEraseScrollbackRisk, - widthChanged || heightChanged, - ); - - // Caller opted into a scrollback wipe via requestRender(true, { clearScrollback: true }). - if (this.#clearScrollbackOnNextRender) return liveRegionPinnedIntent ?? { kind: "sessionReplace" }; - // Initial paint after start(): scrollback must keep its prior shell // content, but the viewport must be cleared so stale rows do not bleed - // into the new UI. If a foreground stream was already active before the - // first paint, honor its live-region pin and avoid committing transient - // rows to native history. - if (!this.#hasEverRendered) return liveRegionPinnedIntent ?? { kind: "initial" }; - if (liveRegionPinnedIntent) return liveRegionPinnedIntent; + // into the new UI. + if (!this.#hasEverRendered) return { kind: "initial" }; + // Caller opted into a scrollback wipe via requestRender(true, { clearScrollback: true }). + if (this.#clearScrollbackOnNextRender) return { kind: "sessionReplace" }; + + const forceViewportRepaint = this.#forceViewportRepaintOnNextRender; + const eagerEraseScrollbackRisk = process.platform !== "win32" && TERMINAL.eagerEraseScrollbackRisk; if (overlayVisibilityReduced && !isMultiplexerSession()) { return hasVisibleOverlay ? { kind: "overlayRebuild" } : { kind: "historyRebuild" }; } @@ -1601,6 +1589,19 @@ export class TUI extends Container { return { kind: "viewportRepaint" }; } + // After foreground tool streaming: when content finally shrinks from the + // streaming peak, rebuild with ED 3 to commit the settled state cleanly. + // The check uses `#streamingHighWater` (the real peak) rather than + // `#previousLines.length` because streaming capped lines to viewport + // height, so `#previousLines` never reflects the true transcript size. + if (this.#streamingHighWater > height && newLines.length < this.#streamingHighWater && newLines.length > height) { + this.#streamingHighWater = 0; + return { kind: "historyRebuild" }; + } + if (this.#streamingHighWater > 0 && newLines.length <= height) { + this.#streamingHighWater = 0; + } + if ( this.#nativeScrollbackDirty && !isMultiplexerSession() && @@ -2105,38 +2106,6 @@ export class TUI extends Container { ); } - #planLiveRegionPinnedRender( - newLines: string[], - height: number, - liveRegionStart: number | undefined, - eagerEraseScrollbackRisk: boolean, - geometryChanged: boolean, - ): RenderIntent | undefined { - // A width/height change reflows the whole terminal: the relative cursor - // positioning this emitter relies on is computed from the pre-resize - // geometry and would land on the wrong rows. Defer to the geometry branch - // (a full reflow rebuild), which is the established behavior for resizes. - if ( - liveRegionStart === undefined || - liveRegionStart >= newLines.length || - !this.#eagerNativeScrollbackRebuild || - !eagerEraseScrollbackRisk || - geometryChanged || - isMultiplexerSession() - ) { - return undefined; - } - if (newLines.length <= height && this.#scrollbackHighWater === 0) return undefined; - if (this.#readNativeViewportAtBottom() !== undefined) return undefined; - - this.#markNativeScrollbackDirty(); - const viewportTop = Math.max(0, newLines.length - height); - const sealedEnd = Math.max(0, Math.min(liveRegionStart, newLines.length)); - const appendTo = Math.min(sealedEnd, viewportTop); - const appendFrom = Math.min(this.#scrollbackHighWater, appendTo); - return { kind: "liveRegionPinned", appendFrom, appendTo }; - } - #padDeferredShrinkLines(lines: string[], paddedLength: number): string[] { if (lines.length >= paddedLength) return lines; return [...lines, ...new Array(paddedLength - lines.length).fill("")]; @@ -2309,74 +2278,6 @@ export class TUI extends Container { this.#commit(lines, width, height, viewportTop, toRow); } - /** - * Foreground-stream live-region paint for ED3-risk terminals with an - * unobservable viewport. Commits the newly-sealed chunk to native scrollback - * (so finished blocks stay scrollable) and repaints the live tail in place, - * leaving the transient live region out of saved lines. - * - * Uses only the no-scroll-snap vocabulary of {@link #emitDiff}: relative - * cursor moves, per-line `\x1b[2K`, and `\r\n` to push the sealed chunk into - * history. It deliberately avoids a full-screen erase (`\x1b[2J`) and absolute - * cursor home (`\x1b[H`): on Ghostty those snap a reader scrolled into history - * back to the bottom on every frame (the live tail repaints every token). - */ - #emitLiveRegionPinnedRepaint( - lines: string[], - width: number, - height: number, - cursorPos: { row: number; col: number } | null, - appendFrom: number, - appendTo: number, - prevViewportTop: number, - prevHardwareCursorRow: number, - ): void { - this.#fullRedrawCount += 1; - const viewportTop = Math.max(0, lines.length - height); - const boundedAppendTo = Math.max(0, Math.min(appendTo, viewportTop, lines.length)); - const boundedAppendFrom = Math.max(0, Math.min(appendFrom, boundedAppendTo)); - - // Position at the top visible row with a relative move. Terminals clamp the - // hardware cursor to the viewport on resize, so clamp our tracking to match - // before computing the delta (mirrors #emitDiff). - const clampedCursor = Math.min(prevHardwareCursorRow, prevViewportTop + height - 1); - const currentScreenRow = Math.max(0, Math.min(height - 1, clampedCursor - prevViewportTop)); - let buffer = this.#paintBeginSequence; - if (currentScreenRow > 0) buffer += `\x1b[${currentScreenRow}A`; - buffer += "\r"; - - // Write the sealed chunk followed by the full viewport from the top row. - // The first (boundedAppendTo - boundedAppendFrom) rows scroll into native - // history; the trailing `height` rows fill the viewport. Each row clears - // itself with `\x1b[2K` instead of relying on a screen-wide erase. - let wroteLine = false; - for (let i = boundedAppendFrom; i < boundedAppendTo; i++) { - if (wroteLine) buffer += "\r\n"; - buffer += `\x1b[2K${this.#fitLineToWidth(lines[i] ?? "", width)}`; - wroteLine = true; - } - for (let screenRow = 0; screenRow < height; screenRow++) { - if (wroteLine) buffer += "\r\n"; - buffer += `\x1b[2K${this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width)}`; - wroteLine = true; - } - - const viewportBottomRow = viewportTop + height - 1; - const contentBottomRow = Math.min(viewportBottomRow, Math.max(viewportTop, lines.length - 1)); - const parkUp = viewportBottomRow - contentBottomRow; - if (parkUp > 0) buffer += `\x1b[${parkUp}A`; - const { seq, toRow } = this.#cursorControlSequence(cursorPos, lines.length, contentBottomRow); - buffer += seq; - buffer += this.#paintEndSequence; - this.terminal.write(buffer); - - this.#maxLinesRendered = lines.length; - if (boundedAppendTo > this.#scrollbackHighWater) { - this.#scrollbackHighWater = boundedAppendTo; - } - this.#commit(lines, width, height, viewportTop, toRow); - } - /** * Push the appended tail into terminal scrollback by `\r\n`-ing past the * previous viewport bottom. Used as a prefix to {@link #emitViewportRepaint} @@ -2612,11 +2513,9 @@ export class TUI extends Container { const detail = intent.kind === "diff" ? `${intent.kind}(first=${intent.firstChanged}, last=${intent.lastChanged}, appended=${intent.appendedLines})` - : intent.kind === "liveRegionPinned" - ? `${intent.kind}(append=${intent.appendFrom}..${intent.appendTo})` - : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined - ? `${intent.kind}(appendFrom=${intent.appendFrom})` - : intent.kind; + : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined + ? `${intent.kind}(appendFrom=${intent.appendFrom})` + : intent.kind; const msg = `[${new Date().toISOString()}] render: ${detail} (prev=${this.#previousLines.length}, new=${newLength}, height=${height})\n`; fs.appendFileSync(getDebugLogPath(), msg); } diff --git a/packages/tui/test/reasoning-stream-dup-repro.test.ts b/packages/tui/test/reasoning-stream-dup-repro.test.ts deleted file mode 100644 index 03db59619..000000000 --- a/packages/tui/test/reasoning-stream-dup-repro.test.ts +++ /dev/null @@ -1,199 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { getTerminalInfo, TERMINAL } from "../src/terminal-capabilities"; -import { type Component, type NativeScrollbackLiveRegion, TUI } from "../src/tui"; -import { VirtualTerminal } from "./virtual-terminal"; - -class LineList implements Component, NativeScrollbackLiveRegion { - #lines: string[]; - constructor(lines: string[]) { - this.#lines = lines; - } - setLines(lines: string[]): void { - this.#lines = lines; - } - invalidate(): void {} - getNativeScrollbackLiveRegionStart(): number | undefined { - return 0; - } - render(_width: number): string[] { - return this.#lines; - } -} - -// A non-live (sealed) block: it does NOT report a live region, so the TUI treats -// it as committable native scrollback below the live boundary. -class LineListSealed implements Component { - #lines: string[]; - constructor(lines: string[]) { - this.#lines = lines; - } - setLines(lines: string[]): void { - this.#lines = lines; - } - invalidate(): void {} - render(_width: number): string[] { - return this.#lines; - } -} - -async function settle(term: VirtualTerminal): Promise { - await term.waitForRender(); -} - -function overrideProbe(term: VirtualTerminal, answer: boolean | undefined): void { - (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => answer; -} - -type MutableTerminalInfo = { eagerEraseScrollbackRisk: boolean }; -const MUX_KEYS = ["TMUX", "STY", "ZELLIJ"] as const; - -async function withGhostty(run: () => Promise): Promise { - const mut = TERMINAL as unknown as MutableTerminalInfo; - const prev = mut.eagerEraseScrollbackRisk; - const prevEnv: Record = {}; - for (const key of MUX_KEYS) { - prevEnv[key] = Bun.env[key]; - delete (Bun.env as Record)[key]; - } - mut.eagerEraseScrollbackRisk = getTerminalInfo("ghostty").eagerEraseScrollbackRisk; - try { - await run(); - } finally { - mut.eagerEraseScrollbackRisk = prev; - for (const key of MUX_KEYS) { - if (prevEnv[key] === undefined) delete (Bun.env as Record)[key]; - else (Bun.env as Record)[key] = prevEnv[key]; - } - } -} - -function dupNonblank(lines: string[]): string[] { - const seen = new Set(); - const dups: string[] = []; - for (const line of lines.map(l => l.trimEnd())) { - if (line.length === 0) continue; - if (seen.has(line)) dups.push(line); - seen.add(line); - } - return dups; -} - -describe("foreground-stream scrollback duplication on ED3-risk ghostty", () => { - it("does not duplicate history rows when overflowing content then shrinks", async () => { - await withGhostty(async () => { - const term = new VirtualTerminal(20, 4); - overrideProbe(term, undefined); - const tui = new TUI(term); - const list = new LineList([]); - tui.addChild(list); - try { - tui.start(); - await settle(term); - tui.setEagerNativeScrollbackRebuild(true); // foreground stream turn - - // Grow past the viewport so rows scroll into native history. - const grown = Array.from({ length: 10 }, (_v, i) => `row-${i}`); - list.setLines(grown); - tui.requestRender(); - await settle(term); - - // Re-layout shrink (e.g. preview/reasoning collapses), still overflowing. - const shrunk = Array.from({ length: 7 }, (_v, i) => `row-${i}`); - list.setLines(shrunk); - tui.requestRender(); - await settle(term); - const streamingBuffer = term.getScrollBuffer(); - expect(dupNonblank(streamingBuffer)).toEqual([]); - - tui.setEagerNativeScrollbackRebuild(false); - tui.requestRender(); - await settle(term); - expect(tui.refreshNativeScrollbackIfDirty({ allowUnknownViewport: true })).toBe(true); - await settle(term); - const checkpointBuffer = term.getScrollBuffer(); - expect(checkpointBuffer.map(line => line.trimEnd())).toEqual(shrunk); - expect(dupNonblank(checkpointBuffer)).toEqual([]); - } finally { - tui.stop(); - } - }); - }); - - it("never emits a full-screen erase (ED2/ED3) while pinning the live region", async () => { - await withGhostty(async () => { - const term = new VirtualTerminal(20, 4); - overrideProbe(term, undefined); - let captured = ""; - const realWrite = term.write.bind(term); - (term as unknown as { write: (s: string) => void }).write = (s: string) => { - captured += s; - realWrite(s); - }; - const tui = new TUI(term); - const list = new LineList([]); - tui.addChild(list); - try { - tui.start(); - await settle(term); - tui.setEagerNativeScrollbackRebuild(true); - captured = ""; // focus on the streaming frames - - // Grow the live block well past the viewport, one line at a time, then - // re-layout shrink — the exact pattern that snapped the view before. - const acc: string[] = []; - for (let n = 1; n <= 24; n++) { - acc.push(`line-${n}`); - list.setLines([...acc]); - tui.requestRender(); - await settle(term); - } - list.setLines(acc.slice(0, 18)); - tui.requestRender(); - await settle(term); - - // The pin must repaint incrementally. A full-screen erase (ED2) or a - // scrollback wipe (ED3) snaps a scrolled-up Ghostty reader to the tail. - expect(captured.includes("\x1b[2J")).toBe(false); - expect(captured.includes("\x1b[3J")).toBe(false); - // Sanity: the pin actually ran (otherwise the assertions are vacuous). - expect(tui.fullRedraws).toBeGreaterThan(0); - } finally { - tui.stop(); - } - }); - }); - - it("keeps a scrolled reader's viewport fixed while the live region streams", async () => { - await withGhostty(async () => { - const term = new VirtualTerminal(20, 4); - overrideProbe(term, undefined); - const tui = new TUI(term); - const sealed = new LineListSealed(Array.from({ length: 12 }, (_v, i) => `prior-${i}`)); - const live = new LineList([]); - tui.addChild(sealed); - tui.addChild(live); - try { - tui.start(); - await settle(term); - // Commit the prior conversation to native history (checkpoint), then - // begin a streaming turn and scroll up into that history. - expect(tui.refreshNativeScrollbackIfDirty({ allowUnknownViewport: true })).toBe(false); - tui.setEagerNativeScrollbackRebuild(true); - live.setLines(Array.from({ length: 6 }, (_v, i) => `think-${i}`)); - tui.requestRender(); - await settle(term); - term.scrollLines(-3); // user scrolls up into history - const before = term.getBufferPosition().viewportY; - for (let n = 7; n <= 20; n++) { - live.setLines(Array.from({ length: n }, (_v, i) => `think-${i}`)); - tui.requestRender(); - await settle(term); - } - // Streaming output must not drag the scrolled viewport down. - expect(term.getBufferPosition().viewportY).toBe(before); - } finally { - tui.stop(); - } - }); - }); -}); diff --git a/packages/tui/test/render-stress-harness.ts b/packages/tui/test/render-stress-harness.ts index 8d584077f..b80d93277 100644 --- a/packages/tui/test/render-stress-harness.ts +++ b/packages/tui/test/render-stress-harness.ts @@ -8,7 +8,6 @@ import { type Component, CURSOR_MARKER, type Focusable, - type NativeScrollbackLiveRegion, type OverlayAnchor, type OverlayHandle, type OverlayOptions, @@ -981,24 +980,18 @@ function reflowToWidth(lines: readonly string[], width: number): string[] { return out; } -class StressComponent implements Component, Focusable, NativeScrollbackLiveRegion { +class StressComponent implements Component, Focusable { focused = false; #model: StressModel; #reflow: boolean; - #pinAsLiveRegion: boolean; - constructor(model: StressModel, reflow = false, pinAsLiveRegion = false) { + constructor(model: StressModel, reflow = false) { this.#model = model; this.#reflow = reflow; - this.#pinAsLiveRegion = pinAsLiveRegion; } invalidate(): void {} - getNativeScrollbackLiveRegionStart(): number | undefined { - return this.#pinAsLiveRegion ? 0 : undefined; - } - render(width: number): string[] { const lines = this.#model.renderedLines(width, this.focused); return this.#reflow ? reflowToWidth(lines, width) : lines; @@ -1148,7 +1141,7 @@ class StressDriver { this.#scheduler = new StressRenderScheduler(); const maxHeight = maxOf(scenario.heightChoices); this.#model = new StressModel(this.#streams.content, maxHeight + 12, scenario.uniqueContent, "root-"); - this.#component = new StressComponent(this.#model, scenario.reflow, this.#traits.foregroundStreaming); + this.#component = new StressComponent(this.#model, scenario.reflow); this.#children = [0, 1].map(id => { const model = new StressModel( this.#streams.children, @@ -1173,15 +1166,15 @@ class StressDriver { async run(): Promise { // Foreground-tool streaming faithfully: pin the ED3-risk trait (independent // of whatever real terminal hosts the worker) and keep the turn-long eager - // rebuild opt-in enabled before the initial paint. On an ED3-risk terminal - // that opt-in is gated off, so content frames flow through the live-region - // pin rather than a forced history rebuild — see `#renderContentFrame`. + // rebuild opt-in enabled. On an ED3-risk terminal that opt-in is gated off, + // so content frames flow through `viewportRepaint`/`diff` rather than a + // forced history rebuild — see `#renderContentFrame`. const terminalInfo = TERMINAL as unknown as { eagerEraseScrollbackRisk: boolean }; const savedRisk = terminalInfo.eagerEraseScrollbackRisk; if (this.#traits.foregroundStreaming) terminalInfo.eagerEraseScrollbackRisk = this.#traits.ed3ScrollbackEraseRisk; try { - if (this.#traits.foregroundStreaming) this.#tui.setEagerNativeScrollbackRebuild(true); this.#tui.start(); + if (this.#traits.foregroundStreaming) this.#tui.setEagerNativeScrollbackRebuild(true); await this.#settle(); this.#assertOracles( { @@ -2041,7 +2034,7 @@ class StressDriver { forcedRender: true, mutatesViewport: true, checkpoint: true, - reconcilesNativeScrollback: !this.#traits.foregroundStreaming, + reconcilesNativeScrollback: true, }, before, after, @@ -2567,13 +2560,6 @@ class StressDriver { } #assertNoStaleOverlaySentinels(op: AppliedOperation, before: Snapshot, after: Snapshot, index: number): void { - if (this.#traits.foregroundStreaming && op.geometryChanged) { - this.#nativeScrollbackAuditBlocked = true; - return; - } - if (this.#traits.foregroundStreaming && this.#nativeScrollbackAuditBlocked && !op.reconcilesNativeScrollback) { - return; - } if (this.#hiddenOverlaySentinels.size === 0) return; const visibleSentinels = new Set( this.#overlays @@ -2594,6 +2580,7 @@ class StressDriver { } } } + #assertUniqueContentNoUnexpectedDuplicates( op: AppliedOperation, before: Snapshot, @@ -2601,14 +2588,6 @@ class StressDriver { index: number, ): void { if (!this.#scenario.uniqueContent) return; - if (this.#traits.foregroundStreaming && op.geometryChanged) { - this.#nativeScrollbackAuditBlocked = true; - return; - } - if (this.#traits.foregroundStreaming && this.#nativeScrollbackAuditBlocked) { - if (!op.reconcilesNativeScrollback) return; - this.#nativeScrollbackAuditBlocked = false; - } // Accumulate even when the check below is skipped (scrolled/overlay): the // frame's legitimate duplicates commit to scrollback regardless of where // the viewport is parked. @@ -3610,11 +3589,16 @@ function coreTemplates(): ScenarioTemplate[] { // Foreground tool actively streaming on an ED3-risk terminal whose // viewport position is unobservable (ghostty/kitty/alacritty/VTE/iTerm2; // see `detectTerminalEagerEraseScrollbackRisk`). The agent requests an - // eager native-scrollback rebuild for the streaming turn, but on ED3-risk - // terminals the live-region seam pins all foreground-stream rows to the - // active viewport instead of letting them enter native history. The - // duplicate oracle is enabled here (`uniqueContent`) so any later shrink - // or reflow that re-exposes a committed live row fails loudly. + // eager native-scrollback rebuild for the streaming turn, but that opt-in + // is gated off on ED3-risk terminals, so `allowUnknownViewportMutation` + // stays false and content frames flow through `viewportRepaint`/`diff` + // instead of a forced history rebuild. An offscreen-edit growth then + // repaints in place — advancing the rendered line count without committing + // the overflow to native history — and the next shrink must still + // re-anchor the bottom of the viewport from that lagging high-water mark. + // The default content-frame path forces `allowUnknownViewportMutation` and + // never reaches this state (a notification chip rendering over the active + // tool render: the original report). name: "darwin-unknown-ghostty-stream-small", platform: "darwin", terminalMode: "unknown", @@ -3626,7 +3610,6 @@ function coreTemplates(): ScenarioTemplate[] { heightChoices: [3, 4, 6], scrollbackRows: 10_000, foregroundStream: true, - uniqueContent: true, }, { name: "linux-unknown-ghostty-stream-large", @@ -3640,7 +3623,6 @@ function coreTemplates(): ScenarioTemplate[] { heightChoices: [8, 12, 24], scrollbackRows: 10_000, foregroundStream: true, - uniqueContent: true, }, { // Width-reflowing content (wrapped/markdown-style) uses the same grapheme @@ -3675,7 +3657,6 @@ function coreTemplates(): ScenarioTemplate[] { scrollbackRows: 10_000, reflow: true, foregroundStream: true, - uniqueContent: true, }, ]; } diff --git a/packages/tui/test/streaming-scrollback-defer.test.ts b/packages/tui/test/streaming-scrollback-defer.test.ts new file mode 100644 index 000000000..737f102ff --- /dev/null +++ b/packages/tui/test/streaming-scrollback-defer.test.ts @@ -0,0 +1,157 @@ +import { describe, expect, it } from "bun:test"; +import { type Component, TERMINAL, TUI } from "@oh-my-pi/pi-tui"; +import { VirtualTerminal } from "./virtual-terminal"; + +class LineList implements Component { + #lines: string[]; + + constructor(lines: string[]) { + this.#lines = [...lines]; + } + + invalidate(): void {} + + render(width: number): string[] { + return this.#lines.map(line => line.slice(0, width)); + } + + setLines(lines: string[]): void { + this.#lines = [...lines]; + } +} + +async function settle(term: VirtualTerminal): Promise { + await Bun.sleep(20); + await term.flush(); +} + +function capture(term: VirtualTerminal): string[] { + const writes: string[] = []; + const realWrite = term.write.bind(term); + (term as unknown as { write: (s: string) => void }).write = (data: string) => { + writes.push(data); + realWrite(data); + }; + return writes; +} + +function overrideProbe(term: VirtualTerminal, answer: boolean | undefined): void { + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = + () => answer; +} + +type MutableTerminalInfo = { + eagerEraseScrollbackRisk: boolean; +}; + +const mutableTerminalInfo = TERMINAL as unknown as MutableTerminalInfo; + +async function withTerminalRisk(risk: boolean, run: () => T | Promise): Promise { + const saved = TERMINAL.eagerEraseScrollbackRisk; + mutableTerminalInfo.eagerEraseScrollbackRisk = risk; + try { + return await run(); + } finally { + mutableTerminalInfo.eagerEraseScrollbackRisk = saved; + } +} + +const ERASE_SCROLLBACK = /\x1b\[3J/g; + +function eraseScrollbackCount(writes: string[]): number { + return writes.join("").match(ERASE_SCROLLBACK)?.length ?? 0; +} + +function rows(prefix: string, count: number): string[] { + return Array.from({ length: count }, (_, i) => `${prefix}${i}`); +} + +describe("streaming scrollback defer", () => { + it("suppresses scrollback growth during eager streaming and commits on disable", async () => { + await withTerminalRisk(false, async () => { + const term = new VirtualTerminal(40, 10); + overrideProbe(term, undefined); + const tui = new TUI(term); + const component = new LineList([...rows("init-", 10), "prompt"]); + + try { + tui.addChild(component); + tui.start(); + await settle(term); + + const writes = capture(term); + const scrollbackBefore = term.getScrollBuffer().length; + + tui.setEagerNativeScrollbackRebuild(true); + + // Grow content past the viewport — should be capped, no + // rows enter native scrollback during streaming. + component.setLines([...rows("stream-", 10), ...rows("more-", 30), "prompt"]); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect(term.getScrollBuffer().length).toBe(scrollbackBefore); + expect(term.getViewport().map(line => line.trim()).at(-1)).toBe("prompt"); + + // Grow even more + component.setLines([...rows("stream-", 10), ...rows("more-", 50), "prompt"]); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect(term.getScrollBuffer().length).toBe(scrollbackBefore); + + // Disable eager mode — should fire single ED3 + re-emit + tui.setEagerNativeScrollbackRebuild(false); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(1); + + const scrollbackAfter = term.getScrollBuffer(); + expect(scrollbackAfter.length).toBeGreaterThan(scrollbackBefore); + expect(scrollbackAfter.join("\n")).toContain("stream-"); + expect(scrollbackAfter.join("\n")).toContain("more-"); + } finally { + tui.stop(); + } + }); + }); + + it("does not emit ED3 during streaming on ED3-risk terminals", async () => { + if (process.platform === "win32") return; + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(40, 10); + overrideProbe(term, undefined); + const tui = new TUI(term); + const component = new LineList([...rows("init-", 10), "prompt"]); + + try { + tui.addChild(component); + tui.start(); + await settle(term); + + const writes = capture(term); + + tui.setEagerNativeScrollbackRebuild(true); + + component.setLines([...rows("grow-", 30), "prompt"]); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + + // Disable on ED3-risk — no historyRebuild + tui.setEagerNativeScrollbackRebuild(false); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect(term.getViewport().map(line => line.trim()).at(-1)).toBe("prompt"); + } finally { + tui.stop(); + } + }); + }); +}); From d719e07915341e1fc625971c6add0630574b9b9e Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 06:58:58 +0200 Subject: [PATCH 18/48] fix(tui): scope #1895 streaming defer to ED3-risk; keep scrollback dirty for checkpoint --- packages/tui/CHANGELOG.md | 2 +- packages/tui/src/tui.ts | 70 +++++++++---------- .../test/streaming-scrollback-defer.test.ts | 35 ++++++---- 3 files changed, 58 insertions(+), 49 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 7f6a7f9fb..f3eceb00a 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -3,7 +3,7 @@ ## [Unreleased] ### Changed -- Changed streaming frames to cap rendered content to viewport height and suppress `\r\n` scrolling, so intermediate tool output never enters terminal native scrollback. Rows are committed once via `historyRebuild` when the tool finishes (or via the stable-prefix commit path for already-stabilized content). ([#NNNN](https://github.com/can1357/oh-my-pi/pull/NNNN)) +- Changed foreground-stream rendering on ED3-risk terminals (Ghostty/kitty/Alacritty/VTE/iTerm2 on POSIX) to defer native-scrollback commits: while a turn streams, each frame caps rendered content to the viewport and suppresses `\r\n` scroll growth, so transient output (spinner ticks, partial lines, status rows) never pollutes terminal history. Native scrollback stays marked dirty and is reconciled in a single ED3 (`CSI 3 J`) + re-emit at the next checkpoint (prompt submit) or on an explicit user-input/IME opt-in; an erase is never emitted mid-stream under a possibly-scrolled reader. Non-ED3-risk terminals keep their eager live rebuild. ([#1895](https://github.com/can1357/oh-my-pi/pull/1895)) ## [15.9.1] - 2026-06-04 ### Fixed diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 6059cfa0c..f857d9f71 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -1393,7 +1393,8 @@ export class TUI extends Container { (resizeEventOccurred && this.#previousHeight > 0); const eagerEraseScrollbackRisk = process.platform !== "win32" && TERMINAL.eagerEraseScrollbackRisk; const eagerRebuildAllowed = this.#eagerNativeScrollbackRebuild && !eagerEraseScrollbackRisk; - const allowUnknownViewportMutation = this.#allowUnknownViewportMutationOnNextRender || eagerRebuildAllowed; + const explicitViewportMutation = this.#allowUnknownViewportMutationOnNextRender; + const allowUnknownViewportMutation = explicitViewportMutation || eagerRebuildAllowed; this.#allowUnknownViewportMutationOnNextRender = false; // 3. Classify intent. @@ -1407,51 +1408,48 @@ export class TUI extends Container { overlayVisibilityReduced, allowUnknownViewportMutation, ); - // 3b. During foreground tool streaming, suppress any destructive scrollback - // commit. Intermediate frames repaint the viewport in place so transient - // streaming output never enters terminal native scrollback. Track the peak - // row count so the settled content can be pushed into scrollback later. + // 3b. Defer scrollback commits during foreground streaming, but only on + // ED3-risk terminals whose committed scrollback cannot be rewritten without + // yanking a scrolled reader. There the eager rebuild is gated off and the + // diff emitter would otherwise `\r\n`-scroll every transient frame (spinner + // ticks, partial output) into native history. Non-ED3-risk terminals keep + // their eager live rebuild, which already commits cleanly. Explicit + // reconciles — the prompt-submit checkpoint (`clearScrollbackOnNextRender`) + // and user-input/IME opt-ins (`explicitViewportMutation`) — are never + // deferred: ED3 is safe there because the keystroke pins the host to the + // bottom. const streamingWasActive = this.#eagerNativeScrollbackRebuild; if (streamingWasActive && !this.#previousStreamingActive) { this.#streamingHighWater = 0; } this.#previousStreamingActive = streamingWasActive; - if (streamingWasActive) { + if (streamingWasActive && eagerEraseScrollbackRisk) { const streamingActive = this.#eagerNativeScrollbackRebuild && !this.#eagerNativeScrollbackRebuildDisablePending; - const streamingJustEnded = !streamingActive && streamingWasActive; - if (streamingJustEnded) { - // Streaming just ended. Commit the full content to scrollback. - // On safe terminals use ED 3 + re-emit; on ED3-risk terminals - // (VTE, Windows) use a non-destructive viewport repaint — - // the shrink-across-high-water paths handle cleanup later. + const explicitReconcile = explicitViewportMutation || this.#clearScrollbackOnNextRender; + if (!streamingActive) { + // Streaming just ended. Keep native scrollback dirty so the next + // checkpoint reconciles the settled transcript; never erase here. this.#streamingHighWater = 0; - this.#clearScrollbackOnNextRender = false; - this.#clearNativeScrollbackDirty(); - if (!eagerEraseScrollbackRisk) { - intent = { kind: "historyRebuild" }; - } - // On ED3-risk the intent stays as planRender returned (e.g. noop - // if content unchanged) — the capped state persists until shrink. - } else if (streamingActive) { - if ( - intent.kind === "sessionReplace" || + this.#markNativeScrollbackDirty(); + } else if ( + !explicitReconcile && + (intent.kind === "sessionReplace" || intent.kind === "historyRebuild" || intent.kind === "overlayRebuild" || - (intent.kind === "diff" && intent.appendedLines) - ) { - this.#clearScrollbackOnNextRender = false; - this.#clearNativeScrollbackDirty(); - // Cap lines to viewport height during streaming so frames - // never grow #previousLines past the visible area. - this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); - this.#scrollbackHighWater = 0; - lines = lines.slice(-height); - intent = { kind: "viewportRepaint" }; - } else { - // Frame wasn't suppressed (e.g. noop), but still track peak. - this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); - } + (intent.kind === "diff" && intent.appendedLines)) + ) { + // Cap the frame to the viewport and keep scrollback dirty: transient + // rows never enter history, and the checkpoint reconciles later. + this.#markNativeScrollbackDirty(); + this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); + this.#scrollbackHighWater = 0; + lines = lines.slice(-height); + intent = { kind: "viewportRepaint" }; + } else { + // Explicit reconcile or a non-committing frame (noop): let the + // planned intent stand, but keep tracking the streaming peak. + this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); } } if (this.#eagerNativeScrollbackRebuildDisablePending) { diff --git a/packages/tui/test/streaming-scrollback-defer.test.ts b/packages/tui/test/streaming-scrollback-defer.test.ts index 737f102ff..638310566 100644 --- a/packages/tui/test/streaming-scrollback-defer.test.ts +++ b/packages/tui/test/streaming-scrollback-defer.test.ts @@ -36,8 +36,7 @@ function capture(term: VirtualTerminal): string[] { } function overrideProbe(term: VirtualTerminal, answer: boolean | undefined): void { - (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = - () => answer; + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => answer; } type MutableTerminalInfo = { @@ -67,8 +66,9 @@ function rows(prefix: string, count: number): string[] { } describe("streaming scrollback defer", () => { - it("suppresses scrollback growth during eager streaming and commits on disable", async () => { - await withTerminalRisk(false, async () => { + it("defers scrollback growth during eager streaming on ED3-risk and reconciles at the checkpoint", async () => { + if (process.platform === "win32") return; + await withTerminalRisk(true, async () => { const term = new VirtualTerminal(40, 10); overrideProbe(term, undefined); const tui = new TUI(term); @@ -84,17 +84,22 @@ describe("streaming scrollback defer", () => { tui.setEagerNativeScrollbackRebuild(true); - // Grow content past the viewport — should be capped, no - // rows enter native scrollback during streaming. + // Grow content past the viewport — capped, no rows enter native + // scrollback during streaming, and no ED3 erase fires. component.setLines([...rows("stream-", 10), ...rows("more-", 30), "prompt"]); tui.requestRender(); await settle(term); expect(eraseScrollbackCount(writes)).toBe(0); expect(term.getScrollBuffer().length).toBe(scrollbackBefore); - expect(term.getViewport().map(line => line.trim()).at(-1)).toBe("prompt"); + expect( + term + .getViewport() + .map(line => line.trim()) + .at(-1), + ).toBe("prompt"); - // Grow even more + // Grow even more — still capped, still no ED3. component.setLines([...rows("stream-", 10), ...rows("more-", 50), "prompt"]); tui.requestRender(); await settle(term); @@ -102,9 +107,10 @@ describe("streaming scrollback defer", () => { expect(eraseScrollbackCount(writes)).toBe(0); expect(term.getScrollBuffer().length).toBe(scrollbackBefore); - // Disable eager mode — should fire single ED3 + re-emit - tui.setEagerNativeScrollbackRebuild(false); - tui.requestRender(); + // The prompt-submit checkpoint reconciles the deferred transcript with + // a single ED3 + re-emit — even while eager is still active, because an + // explicit reconcile is never deferred. + expect(tui.refreshNativeScrollbackIfDirty({ allowUnknownViewport: true })).toBe(true); await settle(term); expect(eraseScrollbackCount(writes)).toBe(1); @@ -148,7 +154,12 @@ describe("streaming scrollback defer", () => { await settle(term); expect(eraseScrollbackCount(writes)).toBe(0); - expect(term.getViewport().map(line => line.trim()).at(-1)).toBe("prompt"); + expect( + term + .getViewport() + .map(line => line.trim()) + .at(-1), + ).toBe("prompt"); } finally { tui.stop(); } From 34005e66564d3a482bc6900fa64ab882e3e4afb2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 05:25:45 +0000 Subject: [PATCH 19/48] fix(coding-agent/hindsight): reacted to live bank scope changes and flushed before dispose Mid-session edits to hindsight.bankId / bankIdPrefix / scoping kept the active HindsightSessionState pinned to the bank selected at session start, so retain/recall/reflect calls landed in the stale bank. Settings hooks now fire onHindsightScopeChanged; the backend rebuilds the primary state against the recomputed scope, disposing the previous one after flushing its queue so queued tool-initiated retains still land in the bank they were enqueued for. Also: - Renamed ensureBankMission to ensureBankExists. The old version skipped creation entirely when bankMission was blank, so the first mental-model POST (auto-seed) could land against a never-PUT bank. Bank creation is now idempotent and unconditional, and runs before mental-model bootstrap. - Fixed AgentSession.dispose to flush the retain queue BEFORE clearing the session state pointer. Reversed, HindsightRetainQueue.#doFlush's identity guard would see the cleared pointer and drop the spliced batch with a 'session vanished' warning. - Snapshotted hindsightScopeCallbacks before iterating because each rebuild subscribes a fresh callback inside the same fire; iterating the live Set would spin. Fixes #1902 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/config/settings.ts | 38 +++ .../coding-agent/src/hindsight/backend.ts | 168 +++++++++--- packages/coding-agent/src/hindsight/bank.ts | 54 ++-- .../src/hindsight/mental-models.ts | 2 +- packages/coding-agent/src/hindsight/state.ts | 28 +- .../coding-agent/src/session/agent-session.ts | 7 +- .../coding-agent/src/tools/memory-reflect.ts | 4 +- .../test/hindsight-backend.test.ts | 254 +++++++++++++++++- .../coding-agent/test/hindsight-bank.test.ts | 33 ++- .../coding-agent/test/memory-tools.test.ts | 2 +- 11 files changed, 511 insertions(+), 83 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..882d38356 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Hindsight retain/recall/reflect calls staying pinned to the bank that was selected when the session started after the operator edited `hindsight.bankId`, `hindsight.bankIdPrefix`, or `hindsight.scoping` mid-session. The backend now subscribes to those settings via `onHindsightScopeChanged` and rebuilds the active `HindsightSessionState` against the recomputed scope, disposing the old state after flushing its queue so in-flight tool-initiated retains still land in the bank they were enqueued for. Also renamed `ensureBankMission` to `ensureBankExists` so a blank `bankMission` no longer skips bank creation entirely, and called it before mental-model bootstrap so `createMentalModel` is never the first POST against a missing bank. `AgentSession.dispose` now flushes the retain queue before clearing `#hindsightSessionState`, since the queue's identity guard would otherwise drop the spliced batch ([#1902](https://github.com/can1357/oh-my-pi/issues/1902)). + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index c735aa770..f5dc33299 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -907,6 +907,9 @@ const SETTING_HOOKS: Partial>> = { for (const cb of appendOnlyModeCallbacks) cb(value); } }, + "hindsight.bankId": () => fireHindsightScopeChanged(), + "hindsight.bankIdPrefix": () => fireHindsightScopeChanged(), + "hindsight.scoping": () => fireHindsightScopeChanged(), }; /** Callbacks invoked when `provider.appendOnlyContext` changes at runtime. */ const appendOnlyModeCallbacks = new Set<(value: string) => void>(); @@ -923,6 +926,41 @@ export function onAppendOnlyModeChanged(cb: (value: string) => void): () => void }; } +/** Callbacks fired when any `hindsight.bankId` / `bankIdPrefix` / `scoping` value changes. */ +const hindsightScopeCallbacks = new Set<() => void>(); + +function fireHindsightScopeChanged(): void { + // Snapshot the callback set before invoking — a callback's body is allowed + // to subscribe a NEW callback (the Hindsight backend re-registers the + // fresh state's listener on every rebuild). Iterating the live Set would + // re-invoke those just-added callbacks within the same fire, which spins + // in place: subscribe → invoke → subscribe → invoke → … + for (const cb of [...hindsightScopeCallbacks]) { + try { + cb(); + } catch (err) { + logger.warn("Settings: hindsight scope hook failed", { error: String(err) }); + } + } +} + +/** + * Subscribe to changes in the Hindsight bank-scoping settings. Lets the + * Hindsight backend rebuild the active `HindsightSessionState` when the + * operator switches `hindsight.bankId`, `hindsight.bankIdPrefix`, or + * `hindsight.scoping` mid-session so subsequent retain/recall calls land in + * the new bank instead of the one selected at session start. + * + * Returns an unsubscribe function. The callback receives no arguments — the + * caller is expected to re-read the relevant settings via `Settings.get`. + */ +export function onHindsightScopeChanged(cb: () => void): () => void { + hindsightScopeCallbacks.add(cb); + return () => { + hindsightScopeCallbacks.delete(cb); + }; +} + // ═══════════════════════════════════════════════════════════════════════════ // Global Singleton // ═══════════════════════════════════════════════════════════════════════════ diff --git a/packages/coding-agent/src/hindsight/backend.ts b/packages/coding-agent/src/hindsight/backend.ts index 2d00dee27..79ff8599f 100644 --- a/packages/coding-agent/src/hindsight/backend.ts +++ b/packages/coding-agent/src/hindsight/backend.ts @@ -9,10 +9,10 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import { logger } from "@oh-my-pi/pi-utils"; -import type { Settings } from "../config/settings"; +import { onHindsightScopeChanged, type Settings } from "../config/settings"; import type { MemoryBackend, MemoryBackendStartOptions } from "../memory-backend/types"; import type { AgentSession } from "../session/agent-session"; -import { computeBankScope } from "./bank"; +import { type BankScope, computeBankScope } from "./bank"; import { createHindsightClient } from "./client"; import { isHindsightConfigured, loadHindsightConfig } from "./config"; import type { HindsightMessage } from "./content"; @@ -60,12 +60,16 @@ export const hindsightBackend: MemoryBackend = { recallTagsMatch: parent.recallTagsMatch, config: parent.config, session, - missionsSet: parent.missionsSet, + banksSet: parent.banksSet, lastRetainedTurn: 0, hasRecalledForFirstTurn: true, aliasOf: parent, }), ); + // Aliases don't run auto-recall/auto-retain, so any pending retain + // queue belongs to the previous alias and is safe to drop after a + // best-effort flush (`flushRetainQueue` is no-op when empty). + await previous?.flushRetainQueue(); previous?.dispose(); return; } @@ -76,38 +80,7 @@ export const hindsightBackend: MemoryBackend = { return; } - const client = createHindsightClient(config); - const scope = computeBankScope(config, session.sessionManager.getCwd()); - - const state = new HindsightSessionState({ - sessionId, - client, - bankId: scope.bankId, - retainTags: scope.retainTags, - recallTags: scope.recallTags, - recallTagsMatch: scope.recallTagsMatch, - config, - session, - missionsSet: new Set(), - lastRetainedTurn: 0, - hasRecalledForFirstTurn: false, - }); - - // Cleanup any stale state for this session (defensive — prevents leaks - // when a session is reused without going through dispose). - const previous = session.setHindsightSessionState(state); - previous?.dispose(); - state.attachSessionListeners(); - - // Kick off mental-model bootstrap. Resolves asynchronously; the first - // turn races and is covered in `beforeAgentStartPrompt` via - // `mentalModelsLoadPromise`. Subsequent turns see the populated cache - // because `runMentalModelLoad` calls `refreshBaseSystemPrompt`. - if (config.mentalModelsEnabled) { - state.mentalModelsLoadPromise = state.runMentalModelLoad(scope).catch(err => { - logger.debug("Hindsight: mental-model bootstrap failed", { bankId: state.bankId, error: String(err) }); - }); - } + await installPrimaryState(session, settings, new Set()); }, async buildDeveloperInstructions(_agentDir, settings, session): Promise { @@ -173,6 +146,131 @@ export const hindsightBackend: MemoryBackend = { return await state.recallForCompaction(flat); }, }; +/** + * Build (or rebuild) the primary `HindsightSessionState` for `session` from + * the current settings and install it. Disposes any previous primary state + * after flushing its retain queue so in-flight tool-initiated retains land in + * the bank that was selected when they were enqueued, not in the new bank. + * + * The created state takes ownership of the `onHindsightScopeChanged` + * subscription so subsequent `hindsight.bankId` / `bankIdPrefix` / `scoping` + * edits trigger another rebuild from the same wiring. + */ +async function installPrimaryState( + session: AgentSession, + settings: Settings, + banksSet: Set, +): Promise { + const sessionId = session.sessionId; + if (!sessionId) return undefined; + + const config = loadHindsightConfig(settings); + if (!isHindsightConfigured(config)) return undefined; + + const client = createHindsightClient(config); + const scope = computeBankScope(config, session.sessionManager.getCwd()); + + const state = new HindsightSessionState({ + sessionId, + client, + bankId: scope.bankId, + retainTags: scope.retainTags, + recallTags: scope.recallTags, + recallTagsMatch: scope.recallTagsMatch, + config, + session, + banksSet, + lastRetainedTurn: 0, + hasRecalledForFirstTurn: false, + }); + + // Subscribe BEFORE installing: if the operator manages to flip another + // setting between install and subscribe, we'd miss the edge. + state.unsubscribeScope = onHindsightScopeChanged(() => { + void rebuildPrimaryStateOnScopeChange(session); + }); + + // Cleanup any stale state for this session (defensive — prevents leaks + // when a session is reused without going through dispose). Flush the + // previous state's retain queue BEFORE clearing it, otherwise + // `HindsightRetainQueue.#doFlush` sees `session.getHindsightSessionState() + // !== state` and drops the batch. + const previous = session.getHindsightSessionState(); + if (previous && previous !== state) { + await previous.flushRetainQueue(); + } + session.setHindsightSessionState(state); + previous?.dispose(); + state.attachSessionListeners(); + + // Kick off mental-model bootstrap. Resolves asynchronously; the first + // turn races and is covered in `beforeAgentStartPrompt` via + // `mentalModelsLoadPromise`. Subsequent turns see the populated cache + // because `runMentalModelLoad` calls `refreshBaseSystemPrompt`. + if (config.mentalModelsEnabled) { + state.mentalModelsLoadPromise = state.runMentalModelLoad(scope).catch(err => { + logger.debug("Hindsight: mental-model bootstrap failed", { bankId: state.bankId, error: String(err) }); + }); + } + + return state; +} + +/** + * `onHindsightScopeChanged` handler: re-evaluate the bank scope from current + * settings and rebuild the primary state when it has actually drifted. No-op + * when the scope is unchanged or the session is no longer hosting a primary + * state (e.g. it was wiped to `undefined`, or this is a subagent alias). + */ +async function rebuildPrimaryStateOnScopeChange(session: AgentSession): Promise { + const current = session.getHindsightSessionState(); + if (!current || current.aliasOf) return; + + const settings = session.settings; + const config = loadHindsightConfig(settings); + if (!isHindsightConfigured(config)) { + // Hindsight effectively unwired mid-session. Flush before clearing so + // queued retains don't get dropped by `HindsightRetainQueue.#doFlush`. + await current.flushRetainQueue(); + const previous = session.setHindsightSessionState(undefined); + previous?.dispose(); + return; + } + + const next = computeBankScope(config, session.sessionManager.getCwd()); + if (bankScopesEqual(next, current)) return; + + // Preserve the banksSet so we don't re-PUT banks we've already confirmed. + await installPrimaryState(session, settings, current.banksSet); +} + +/** Tag-array equality: order matters because we never reorder on the way in. */ +function stringArraysEqual(a: string[] | undefined, b: string[] | undefined): boolean { + if (a === b) return true; + if (!a || !b) return false; + if (a.length !== b.length) return false; + for (let i = 0; i < a.length; i++) { + if (a[i] !== b[i]) return false; + } + return true; +} + +/** + * Structural compare of a freshly resolved `BankScope` against a live state's + * bank routing. Used by the scope-change handler to skip rebuilds that don't + * actually move the bank or its tag filters. + */ +function bankScopesEqual( + scope: BankScope, + state: Pick, +): boolean { + return ( + scope.bankId === state.bankId && + stringArraysEqual(scope.retainTags, state.retainTags) && + stringArraysEqual(scope.recallTags, state.recallTags) && + scope.recallTagsMatch === state.recallTagsMatch + ); +} /** Reduce arbitrary AgentMessages into the Hindsight flat-text shape. */ function flattenMessagesForRecall(messages: AgentMessage[]): HindsightMessage[] { diff --git a/packages/coding-agent/src/hindsight/bank.ts b/packages/coding-agent/src/hindsight/bank.ts index a6f9cc149..32bc28ce4 100644 --- a/packages/coding-agent/src/hindsight/bank.ts +++ b/packages/coding-agent/src/hindsight/bank.ts @@ -1,5 +1,5 @@ /** - * Bank ID derivation, project-tag scoping, and first-use mission setup. + * Bank ID derivation, project-tag scoping, and first-use bank setup. * * Three scoping modes (`HindsightConfig.scoping`): * - `global` — single shared bank, no per-project filter. @@ -11,10 +11,13 @@ * The base bank id is `bankIdPrefix-bankId` (default `omp`). Per-project mode * appends `-`; tagged mode leaves the bank untouched and uses tags. * - * Mission setup is idempotent at module level — a missionsSet keeps track of - * banks we've already POSTed to so each session boundary doesn't fire a fresh - * `createBank` call. Failures are swallowed: missions are an optimisation, not - * a precondition for retain/recall. + * Bank existence is idempotent at module level — a banksSet keeps track of + * banks we've already PUT so each session boundary doesn't fire a fresh + * `createBank` call. The PUT is idempotent server-side, so re-firing on a hot + * path would only burn round-trips. Failures are swallowed: missing the + * mission patch is an optimisation, but the bank ITSELF must exist before + * mental-model bootstrap or the first retain, otherwise the very first POST + * lands against a missing bank. */ import * as path from "node:path"; @@ -93,39 +96,46 @@ export function deriveBankId(config: HindsightConfig, directory: string): string } /** - * Ensure a bank's reflect/retain mission is set, exactly once per process. + * Ensure a bank exists, and patch its reflect/retain mission on first use. * - * Tracked via the supplied set; on overflow we drop the oldest half so the set - * cannot grow unboundedly across long-lived processes. + * Idempotent: skips the PUT when the bank id is already in the supplied set. + * The mission body is optional — when `bankMission` is blank we still PUT to + * make sure the bank itself is created, so mental-model bootstrap and the + * first retain don't land against a non-existent bank. + * + * The set is capped; on overflow we drop the oldest half so it cannot grow + * unboundedly across long-lived processes. */ -export async function ensureBankMission( +export async function ensureBankExists( client: HindsightApi, bankId: string, config: HindsightConfig, - missionsSet: Set, + banksSet: Set, ): Promise { + if (banksSet.has(bankId)) return; + const mission = config.bankMission?.trim(); - if (!mission) return; - if (missionsSet.has(bankId)) return; + const retainMission = config.retainMission?.trim(); try { await client.createBank(bankId, { - reflectMission: mission, - retainMission: config.retainMission?.trim() || undefined, + reflectMission: mission || undefined, + retainMission: retainMission || undefined, }); - missionsSet.add(bankId); - if (missionsSet.size > MISSION_SET_CAP) { - const keys = [...missionsSet].sort(); + banksSet.add(bankId); + if (banksSet.size > MISSION_SET_CAP) { + const keys = [...banksSet].sort(); for (const key of keys.slice(0, keys.length >> 1)) { - missionsSet.delete(key); + banksSet.delete(key); } } if (config.debug) { - logger.debug("Hindsight: set mission for bank", { bankId }); + logger.debug("Hindsight: ensured bank", { bankId, mission: Boolean(mission) }); } } catch (err) { - // Mission set is best-effort; the bank may not exist yet, or the API may - // reject the call. Either way, retain/recall still work, so swallow. - logger.debug("Hindsight: ensureBankMission failed", { bankId, error: String(err) }); + // Bank creation is best-effort; the server may already have it, or the + // API may reject the call. Either way, downstream retain/recall calls + // will surface a clearer error if the bank really is missing. + logger.debug("Hindsight: ensureBankExists failed", { bankId, error: String(err) }); } } diff --git a/packages/coding-agent/src/hindsight/mental-models.ts b/packages/coding-agent/src/hindsight/mental-models.ts index 294fb5fab..210137193 100644 --- a/packages/coding-agent/src/hindsight/mental-models.ts +++ b/packages/coding-agent/src/hindsight/mental-models.ts @@ -112,7 +112,7 @@ function dedupe(items: T[]): T[] { * Idempotently create any seed mental models that don't already exist on the * bank. Best-effort: a list/create failure does not throw — mental models are * an optimization, not a precondition for retain/recall, and we mirror the - * swallow-on-failure pattern used by `ensureBankMission`. + * swallow-on-failure pattern used by `ensureBankExists`. * * Existing models are NEVER modified. See module docstring. */ diff --git a/packages/coding-agent/src/hindsight/state.ts b/packages/coding-agent/src/hindsight/state.ts index 883e5f714..26f3e7d58 100644 --- a/packages/coding-agent/src/hindsight/state.ts +++ b/packages/coding-agent/src/hindsight/state.ts @@ -1,6 +1,6 @@ import { logger } from "@oh-my-pi/pi-utils"; import type { AgentSession } from "../session/agent-session"; -import { type BankScope, ensureBankMission } from "./bank"; +import { type BankScope, ensureBankExists } from "./bank"; import type { HindsightApi, MemoryItemInput } from "./client"; import type { HindsightConfig } from "./config"; import { @@ -45,12 +45,12 @@ export interface HindsightSessionStateOptions { recallTagsMatch?: "any" | "all" | "any_strict" | "all_strict"; config: HindsightConfig; session: AgentSession; - missionsSet: Set; + banksSet: Set; lastRetainedTurn?: number; hasRecalledForFirstTurn?: boolean; /** * When set, this entry is a subagent alias that reuses the parent's bank, - * scope, config, client, and missionsSet. Aliases skip auto-recall and + * scope, config, client, and banksSet. Aliases skip auto-recall and * auto-retain — those run on the parent only — but the recall/retain/reflect * tools resolve via the alias so they persist to the same bank as the parent. */ @@ -148,7 +148,7 @@ export class HindsightRetainQueue { } try { - await ensureBankMission(state.client, state.bankId, state.config, state.missionsSet); + await ensureBankExists(state.client, state.bankId, state.config, state.banksSet); const batch: MemoryItemInput[] = items.map(item => ({ content: item.content, context: item.context ?? state.config.retainContext, @@ -198,7 +198,7 @@ export class HindsightSessionState { recallTagsMatch?: "any" | "all" | "any_strict" | "all_strict"; config: HindsightConfig; session: AgentSession; - missionsSet: Set; + banksSet: Set; lastRetainedTurn: number; hasRecalledForFirstTurn: boolean; lastRecallSnippet?: string; @@ -213,6 +213,12 @@ export class HindsightSessionState { */ mentalModelsLoadPromise?: Promise; unsubscribe?: () => void; + /** + * Releases the `onHindsightScopeChanged` subscription that drives live + * rebuilds when `hindsight.bankId` / `bankIdPrefix` / `scoping` change. + * Only set on primary states; aliases inherit the parent's subscription. + */ + unsubscribeScope?: () => void; /** Alias states delegate persistence config to a primary parent state. */ aliasOf?: HindsightSessionState; readonly retainQueue: HindsightRetainQueue; @@ -226,7 +232,7 @@ export class HindsightSessionState { this.recallTagsMatch = options.recallTagsMatch; this.config = options.config; this.session = options.session; - this.missionsSet = options.missionsSet; + this.banksSet = options.banksSet; this.lastRetainedTurn = options.lastRetainedTurn ?? 0; this.hasRecalledForFirstTurn = options.hasRecalledForFirstTurn ?? false; this.aliasOf = options.aliasOf; @@ -291,7 +297,7 @@ export class HindsightSessionState { const { transcript } = prepareRetentionTranscript(target, true); if (!transcript) return; - await ensureBankMission(this.client, this.bankId, this.config, this.missionsSet); + await ensureBankExists(this.client, this.bankId, this.config, this.banksSet); await this.client.retain(this.bankId, transcript, { documentId, context: this.config.retainContext, @@ -398,6 +404,12 @@ export class HindsightSessionState { async runMentalModelLoad(scope: BankScope): Promise { if (!this.config.mentalModelsEnabled) return; + // Create/ensure the bank BEFORE the first mental-model POST so we don't + // land `createMentalModel` against a bank the server has never seen — + // that surfaces as a FK / 404 on Hindsight's side. `ensureBankExists` + // is idempotent (PUT) and skips after the first call via `banksSet`. + await ensureBankExists(this.client, this.bankId, this.config, this.banksSet); + // Seeding is opt-in (`hindsight.mentalModelAutoSeed`). Default behaviour is // read-only: we surface whatever models the operator has curated on the // bank, but we do NOT POST to create new ones unless they explicitly @@ -456,6 +468,8 @@ export class HindsightSessionState { dispose(): void { this.unsubscribe?.(); this.unsubscribe = undefined; + this.unsubscribeScope?.(); + this.unsubscribeScope = undefined; this.retainQueue.dispose(); } diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index b7760bb5c..d0ea75ac8 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2990,8 +2990,13 @@ export class AgentSession { this.#releasePowerAssertion(); await this.sessionManager.close(); this.#closeAllProviderSessions("dispose"); - const hindsightState = this.setHindsightSessionState(undefined); + // Flush the retain queue BEFORE clearing the session's pointer so + // `HindsightRetainQueue.#doFlush` still sees `session.getHindsightSessionState() === state`. + // Reversed, the spliced batch survives just long enough to fail the + // identity check and get dropped with a `session vanished` warning. + const hindsightState = this.getHindsightSessionState(); await hindsightState?.flushRetainQueue(); + this.setHindsightSessionState(undefined); hindsightState?.dispose(); const mnemopiState = setMnemopiSessionState(this, undefined); mnemopiState?.dispose(); diff --git a/packages/coding-agent/src/tools/memory-reflect.ts b/packages/coding-agent/src/tools/memory-reflect.ts index 19e5fd2d3..8227a3328 100644 --- a/packages/coding-agent/src/tools/memory-reflect.ts +++ b/packages/coding-agent/src/tools/memory-reflect.ts @@ -1,7 +1,7 @@ import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; import { logger, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { ensureBankMission } from "../hindsight/bank"; +import { ensureBankExists } from "../hindsight/bank"; import reflectDescription from "../prompts/tools/reflect.md" with { type: "text" }; import type { ToolSession } from "."; @@ -67,7 +67,7 @@ export class MemoryReflectTool implements AgentTool } try { - await ensureBankMission(state.client, state.bankId, state.config, state.missionsSet); + await ensureBankExists(state.client, state.bankId, state.config, state.banksSet); const response = await state.client.reflect(state.bankId, params.query, { context: params.context, budget: state.config.recallBudget, diff --git a/packages/coding-agent/test/hindsight-backend.test.ts b/packages/coding-agent/test/hindsight-backend.test.ts index 550306447..fbe17d9fe 100644 --- a/packages/coding-agent/test/hindsight-backend.test.ts +++ b/packages/coding-agent/test/hindsight-backend.test.ts @@ -19,6 +19,7 @@ interface FakeSessionDeps { sessionId: string | null; cwd?: string; entries?: Array<{ role: "user" | "assistant"; text: string }>; + settings?: Settings; } function makeFakeSession(deps: FakeSessionDeps) { @@ -27,7 +28,7 @@ function makeFakeSession(deps: FakeSessionDeps) { let hindsightState: HindsightSessionState | undefined; const session = { sessionId: deps.sessionId, - settings: Settings.isolated(), + settings: deps.settings ?? Settings.isolated(), sessionManager: { getEntries: () => entries.map((e, i) => ({ @@ -207,7 +208,7 @@ describe("hindsightBackend.start", () => { expect(subState?.aliasOf).toBe(parentState); expect(subState?.bankId).toBe(parentState?.bankId); expect(subState?.client).toBe(parentState?.client); - expect(subState?.missionsSet).toBe(parentState?.missionsSet); + expect(subState?.banksSet).toBe(parentState?.banksSet); // Aliases must not subscribe to session events — the parent owns auto-recall/auto-retain. expect(subState?.unsubscribe).toBeUndefined(); // hasRecalledForFirstTurn=true suppresses beforeAgentStartPrompt auto-recall on the sub. @@ -532,3 +533,252 @@ describe("hindsightBackend.clear", () => { expect(deleteSpy).not.toHaveBeenCalled(); }); }); + +describe("hindsightBackend live bank routing", () => { + beforeEach(() => { + resetSettingsForTest(); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + // Regression for issue #1902: changing `hindsight.bankId` during a live + // session used to leave the active `HindsightSessionState` pinned to the + // bank that was selected when the session started, so subsequent retains + // kept landing in the stale bank ("omp") instead of the new one + // ("Minigames"). The bank-routing settings must re-resolve on `set`. + it("rebuilds the primary state when hindsight.bankId changes mid-session", async () => { + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.scoping": "global", + }); + // Seed bankId via `set` (not `isolated` overrides), otherwise the + // follow-up `set` writes to `#global` while `get` keeps returning the + // `#overrides` value — exactly the precedence the live settings UI + // does NOT have, since real config writes land in `#global`. + settings.set("hindsight.bankId", "omp"); + const session = makeFakeSession({ sessionId: "s-rebuild", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + + const initial = session.getHindsightSessionState(); + expect(initial?.bankId).toBe("omp"); + + settings.set("hindsight.bankId", "Minigames"); + // Hook is sync but the rebuild is async; yield once so the handler runs. + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next).toBeDefined(); + expect(next?.bankId).toBe("Minigames"); + // Must be a brand-new state — the old one was disposed. + expect(next).not.toBe(initial); + }); + + // Same regression, exercising the `hindsight.scoping` axis: switching + // scope mode also reshapes the bank id / tag filters and must rebuild. + it("rebuilds the primary state when hindsight.scoping changes mid-session", async () => { + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.scoping", "global"); + const session = makeFakeSession({ sessionId: "s-scoping", cwd: "/work/proj", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + + const initial = session.getHindsightSessionState(); + expect(initial?.bankId).toBe("omp"); + expect(initial?.retainTags).toBeUndefined(); + + settings.set("hindsight.scoping", "per-project"); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next).toBeDefined(); + expect(next?.bankId).toBe("omp-proj"); + expect(next).not.toBe(initial); + }); + + // Same setting written with the same value MUST NOT rebuild — a rebuild + // would reset `lastRetainedTurn` / `hasRecalledForFirstTurn` and force a + // fresh mental-model bootstrap for no observable reason. + it("does not rebuild when the bank-routing setting is rewritten with the same value", async () => { + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.bankId", "omp"); + const session = makeFakeSession({ sessionId: "s-noop", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + + const initial = session.getHindsightSessionState(); + settings.set("hindsight.bankId", "omp"); // unchanged + await Bun.sleep(0); + + expect(session.getHindsightSessionState()).toBe(initial); + }); + + // Regression for issue #1902 fix #2: mental-model auto-seed used to POST + // `createMentalModel` against a bank the server never saw, because the + // old `ensureBankMission` skipped creation entirely when `bankMission` + // was blank. The bank MUST be PUT (created) before any mental-model POST. + it("creates the bank before mental-model bootstrap even when bankMission is blank", async () => { + const callOrder: string[] = []; + const createBankSpy = vi.spyOn(HindsightApi.prototype, "createBank").mockImplementation(async () => { + callOrder.push("createBank"); + return {} as never; + }); + const listMentalSpy = vi.spyOn(HindsightApi.prototype, "listMentalModels").mockImplementation(async () => { + callOrder.push("listMentalModels"); + return { items: [] } as never; + }); + const createMentalModelSpy = vi + .spyOn(HindsightApi.prototype, "createMentalModel") + .mockImplementation(async () => { + callOrder.push("createMentalModel"); + return {} as never; + }); + + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.mentalModelsEnabled": true, + "hindsight.mentalModelAutoSeed": true, + "hindsight.bankMission": "", + }); + const session = makeFakeSession({ sessionId: "s-bank-first", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + await session.getHindsightSessionState()?.mentalModelsLoadPromise; + + expect(createBankSpy).toHaveBeenCalled(); + // First call must be `createBank`. Otherwise the mental-model POST + // lands against a never-created bank and the server FK-fails it. + expect(callOrder[0]).toBe("createBank"); + // Mental-model POSTs are allowed but they MUST come after the bank + // is on the server. + if (createMentalModelSpy.mock.calls.length > 0) { + const bankIdx = callOrder.indexOf("createBank"); + const mmIdx = callOrder.indexOf("createMentalModel"); + expect(bankIdx).toBeLessThan(mmIdx); + } + void listMentalSpy; + }); +}); + +describe("hindsightBackend retain queue flush on session teardown", () => { + beforeEach(() => { + resetSettingsForTest(); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + // Regression for issue #1902 fix #3: `AgentSession.dispose` used to clear + // `#hindsightSessionState` BEFORE flushing the retain queue, so + // `HindsightRetainQueue.#doFlush` saw `session.getHindsightSessionState() + // !== state` and dropped the spliced batch. The fix flips the order: + // flush MUST complete while the session pointer still references the same + // state. We defend the contract end-to-end by enqueuing a tool-initiated + // retain, then calling `flushRetainQueue` in the order + // `AgentSession.dispose` uses (flush → clear → state.dispose). + it("flushes the retain queue to the server before the session pointer clears", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.bankId": "omp", + }); + const session = makeFakeSession({ sessionId: "s-dispose-flush", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + const state = session.getHindsightSessionState(); + expect(state).toBeDefined(); + + state!.enqueueRetain("durable fact", "test context"); + + // AgentSession.dispose order: flush first, THEN clear, THEN dispose. + // Reversed (clear → flush), `#doFlush`'s identity check would fail and + // the batch would be dropped with a `session vanished` warning. + await state!.flushRetainQueue(); + session.setHindsightSessionState(undefined); + state!.dispose(); + + expect(retainBatchSpy).toHaveBeenCalledTimes(1); + const [bankId, items] = retainBatchSpy.mock.calls[0]; + expect(bankId).toBe("omp"); + expect(items).toHaveLength(1); + expect(items[0].content).toBe("durable fact"); + }); + + // Companion contract test: documents the failure mode the dispose-order + // fix prevents. If the session pointer is cleared first, the queue's + // identity guard drops the spliced batch (and `retainBatch` is NOT + // called). This is exactly what was happening before the fix. + it("drops the spliced batch when the session pointer is cleared before flush (anti-regression)", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + const session = makeFakeSession({ sessionId: "s-buggy-order", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + const state = session.getHindsightSessionState(); + state!.enqueueRetain("dropped fact"); + + // Buggy order — clear THEN flush. The queue's identity check fails. + session.setHindsightSessionState(undefined); + await state!.flushRetainQueue(); + + expect(retainBatchSpy).not.toHaveBeenCalled(); + }); +}); diff --git a/packages/coding-agent/test/hindsight-bank.test.ts b/packages/coding-agent/test/hindsight-bank.test.ts index bbda61602..0ce81b792 100644 --- a/packages/coding-agent/test/hindsight-bank.test.ts +++ b/packages/coding-agent/test/hindsight-bank.test.ts @@ -1,5 +1,5 @@ import { afterEach, beforeEach, describe, expect, it, type Mock, vi } from "bun:test"; -import { computeBankScope, deriveBankId, ensureBankMission } from "@oh-my-pi/pi-coding-agent/hindsight/bank"; +import { computeBankScope, deriveBankId, ensureBankExists } from "@oh-my-pi/pi-coding-agent/hindsight/bank"; import { HindsightApi } from "@oh-my-pi/pi-coding-agent/hindsight/client"; import type { HindsightConfig } from "@oh-my-pi/pi-coding-agent/hindsight/config"; @@ -117,7 +117,7 @@ describe("deriveBankId (legacy wrapper)", () => { }); }); -describe("ensureBankMission", () => { +describe("ensureBankExists", () => { let client: HindsightApi; let createSpy: Mock | undefined; @@ -129,14 +129,14 @@ describe("ensureBankMission", () => { createSpy?.mockRestore(); }); - it("calls createBank exactly once per bank id", async () => { + it("calls createBank exactly once per bank id and forwards the mission body", async () => { createSpy = vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); const seen = new Set(); const config = baseConfig({ bankMission: "remember everything", retainMission: "extract facts" }); - await ensureBankMission(client, "bank-a", config, seen); - await ensureBankMission(client, "bank-a", config, seen); - await ensureBankMission(client, "bank-b", config, seen); + await ensureBankExists(client, "bank-a", config, seen); + await ensureBankExists(client, "bank-a", config, seen); + await ensureBankExists(client, "bank-b", config, seen); expect(createSpy).toHaveBeenCalledTimes(2); expect(createSpy).toHaveBeenCalledWith( @@ -148,13 +148,22 @@ describe("ensureBankMission", () => { expect(seen.has("bank-b")).toBe(true); }); - it("is a no-op when no mission is configured", async () => { + // Regression: mental-model auto-seed used to POST `createMentalModel` against + // a never-created bank when `bankMission` was blank, because the old + // `ensureBankMission` skipped creation entirely without a mission. + it("still PUTs the bank when no mission is configured (so the bank gets created)", async () => { createSpy = vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); const seen = new Set(); - await ensureBankMission(client, "bank", baseConfig({ bankMission: "" }), seen); - await ensureBankMission(client, "bank", baseConfig({ bankMission: " " }), seen); - expect(createSpy).not.toHaveBeenCalled(); - expect(seen.size).toBe(0); + + await ensureBankExists(client, "bank", baseConfig({ bankMission: "" }), seen); + await ensureBankExists(client, "bank", baseConfig({ bankMission: " " }), seen); + + expect(createSpy).toHaveBeenCalledTimes(1); + expect(createSpy).toHaveBeenCalledWith( + "bank", + expect.objectContaining({ reflectMission: undefined, retainMission: undefined }), + ); + expect(seen.has("bank")).toBe(true); }); it("swallows API failures and does not mark the bank as initialised", async () => { @@ -162,7 +171,7 @@ describe("ensureBankMission", () => { const seen = new Set(); const config = baseConfig({ bankMission: "do the thing" }); - await expect(ensureBankMission(client, "bank-x", config, seen)).resolves.toBeUndefined(); + await expect(ensureBankExists(client, "bank-x", config, seen)).resolves.toBeUndefined(); expect(seen.has("bank-x")).toBe(false); }); }); diff --git a/packages/coding-agent/test/memory-tools.test.ts b/packages/coding-agent/test/memory-tools.test.ts index bd9da3a22..9875bf69d 100644 --- a/packages/coding-agent/test/memory-tools.test.ts +++ b/packages/coding-agent/test/memory-tools.test.ts @@ -100,7 +100,7 @@ function registerState(client: HindsightApi, settings?: Settings, opts: Register getHindsightSessionState: () => registeredState, ...opts.sessionOverrides, } as never, - missionsSet: new Set(), + banksSet: new Set(), lastRetainedTurn: 0, hasRecalledForFirstTurn: false, }); From a8b3aeaeaa6fb53a67ad557076ca31530de571d6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 05:35:09 +0000 Subject: [PATCH 20/48] fix(coding-agent/hindsight): serialized live scope rebuilds Coalesced synchronous Hindsight routing setting hooks into one serialized session rebuild so cwd reloads and multi-setting updates cannot have multiple continuations capture the same old state and leak fresh session listeners. Also re-read the current state after awaiting the previous queue flush before installing the replacement state, so an unexpected concurrent owner cannot leave the actual current state undisposed. Fixes #1902 --- .../coding-agent/src/hindsight/backend.ts | 71 ++++++++++++++++--- .../test/hindsight-backend.test.ts | 45 ++++++++++++ 2 files changed, 106 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/src/hindsight/backend.ts b/packages/coding-agent/src/hindsight/backend.ts index 79ff8599f..f01c92002 100644 --- a/packages/coding-agent/src/hindsight/backend.ts +++ b/packages/coding-agent/src/hindsight/backend.ts @@ -146,6 +146,45 @@ export const hindsightBackend: MemoryBackend = { return await state.recallForCompaction(flat); }, }; +interface PrimaryRebuildTask { + pending: boolean; +} + +const primaryRebuildTasks = new WeakMap(); + +/** + * Coalesce and serialize live scope rebuilds for one session. Cwd reloads fire + * all settings hooks synchronously; running every callback immediately would + * let multiple rebuilds capture the same old state and leak the fresh states + * installed by earlier continuations. + */ +function schedulePrimaryStateRebuild(session: AgentSession): void { + const task = primaryRebuildTasks.get(session); + if (task) { + task.pending = true; + return; + } + + const nextTask: PrimaryRebuildTask = { pending: true }; + primaryRebuildTasks.set(session, nextTask); + void Promise.resolve() + .then(async () => { + while (nextTask.pending) { + nextTask.pending = false; + try { + await rebuildPrimaryStateOnScopeChange(session); + } catch (err) { + logger.warn("Hindsight: scope rebuild failed", { error: String(err) }); + } + } + }) + .finally(() => { + if (primaryRebuildTasks.get(session) === nextTask) { + primaryRebuildTasks.delete(session); + } + }); +} + /** * Build (or rebuild) the primary `HindsightSessionState` for `session` from * the current settings and install it. Disposes any previous primary state @@ -170,6 +209,23 @@ async function installPrimaryState( const client = createHindsightClient(config); const scope = computeBankScope(config, session.sessionManager.getCwd()); + // Cleanup any stale state for this session (defensive — prevents leaks + // when a session is reused without going through dispose). Flush the + // previous state's retain queue BEFORE clearing it, otherwise + // `HindsightRetainQueue.#doFlush` sees `session.getHindsightSessionState() + // !== state` and drops the batch. Re-read after the await so a concurrent + // owner cannot leave the actual current state undisposed. + let previous = session.getHindsightSessionState(); + if (previous) { + await previous.flushRetainQueue(); + } + const latest = session.getHindsightSessionState(); + if (latest && latest !== previous) { + previous?.dispose(); + previous = latest; + await previous.flushRetainQueue(); + } + const state = new HindsightSessionState({ sessionId, client, @@ -187,19 +243,14 @@ async function installPrimaryState( // Subscribe BEFORE installing: if the operator manages to flip another // setting between install and subscribe, we'd miss the edge. state.unsubscribeScope = onHindsightScopeChanged(() => { - void rebuildPrimaryStateOnScopeChange(session); + schedulePrimaryStateRebuild(session); }); - // Cleanup any stale state for this session (defensive — prevents leaks - // when a session is reused without going through dispose). Flush the - // previous state's retain queue BEFORE clearing it, otherwise - // `HindsightRetainQueue.#doFlush` sees `session.getHindsightSessionState() - // !== state` and drops the batch. - const previous = session.getHindsightSessionState(); - if (previous && previous !== state) { - await previous.flushRetainQueue(); + const displaced = session.setHindsightSessionState(state); + if (displaced && displaced !== previous) { + await displaced.flushRetainQueue(); + displaced.dispose(); } - session.setHindsightSessionState(state); previous?.dispose(); state.attachSessionListeners(); diff --git a/packages/coding-agent/test/hindsight-backend.test.ts b/packages/coding-agent/test/hindsight-backend.test.ts index fbe17d9fe..0ac38be24 100644 --- a/packages/coding-agent/test/hindsight-backend.test.ts +++ b/packages/coding-agent/test/hindsight-backend.test.ts @@ -71,6 +71,7 @@ function makeFakeSession(deps: FakeSessionDeps) { emit(event: Parameters[0]) { for (const l of [...listeners]) l(event); }, + listenerCount: () => listeners.size, }; return session; } @@ -643,6 +644,50 @@ describe("hindsightBackend live bank routing", () => { expect(session.getHindsightSessionState()).toBe(initial); }); + it("coalesces synchronous routing hooks so rebuilt states do not leak agent listeners", async () => { + const retainSpy = vi.spyOn(HindsightApi.prototype, "retain").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.retainEveryNTurns": 1, + }); + settings.set("hindsight.bankId", "omp"); + settings.set("hindsight.scoping", "global"); + const entries = [ + { role: "user" as const, text: "remember this routing coalesce fact" }, + { role: "assistant" as const, text: "acknowledged routing coalesce fact" }, + ]; + const session = makeFakeSession({ sessionId: "s-coalesce", cwd: "/work/proj", entries, settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + expect(session.listenerCount()).toBe(1); + + // Mirrors `Settings.#fireAllHooks()` during cwd reload: all three + // Hindsight routing hooks can fire synchronously before the first async + // queue flush continuation resumes. They must collapse into one rebuild. + settings.set("hindsight.bankIdPrefix", "live"); + settings.set("hindsight.bankId", "Minigames"); + settings.set("hindsight.scoping", "per-project"); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next?.bankId).toBe("live-Minigames-proj"); + expect(session.listenerCount()).toBe(1); + + session.emit({ type: "agent_end", messages: [] }); + await Bun.sleep(0); + + expect(retainSpy).toHaveBeenCalledTimes(1); + expect(retainSpy.mock.calls[0][0]).toBe("live-Minigames-proj"); + }); + // Regression for issue #1902 fix #2: mental-model auto-seed used to POST // `createMentalModel` against a bank the server never saw, because the // old `ensureBankMission` skipped creation entirely when `bankMission` From a4be51f08c7e69bfcee81e54945140f75a510403 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 05:36:05 +0000 Subject: [PATCH 21/48] fix(discovery): scan .github/skills via the github provider MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The github provider registered context-files (.github/copilot-instructions.md) and instructions (.github/instructions/*.instructions.md), but no skills capability — so .github/skills//SKILL.md, the layout GitHub documents for Copilot Agent Skills, was silently never discovered. The skill:// URL resolved to 'Available: none' and nothing surfaced in the system prompt. Register a skill capability on the github provider (priority 30, project-only) pointing at .github/skills/ and reuse scanSkillsFromDir with requireDescription: true to match the Agent Skills spec and the sibling native/omp-plugins providers. Pin the wiring with a discovery test that loads the skills capability scoped to the github provider against a temp cwd containing a SKILL.md, and a negative case that drops a skill missing a description. Fixes #1906 --- docs/skills.md | 1 + packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/discovery/github.ts | 38 +++++++++- .../test/discovery/github-skills.test.ts | 70 +++++++++++++++++++ 4 files changed, 112 insertions(+), 1 deletion(-) create mode 100644 packages/coding-agent/test/discovery/github-skills.test.ts diff --git a/docs/skills.md b/docs/skills.md index 845671d16..beca954bd 100644 --- a/docs/skills.md +++ b/docs/skills.md @@ -89,6 +89,7 @@ Current registered skill providers: - `agents` - `codex` 5. `opencode` (priority 55) +6. `github` (priority 30) — `.github/skills//SKILL.md` (GitHub Agent Skills layout, project-only) Dedup key is skill name. First item with a given name wins. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..fa185ba42 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the `github` discovery provider silently ignoring `.github/skills//SKILL.md`, GitHub's documented Agent Skills layout. The provider now registers a `skills` capability (priority 30, project-only) that scans `.github/skills/` non-recursively via `scanSkillsFromDir` with `requireDescription: true`, matching the Agent Skills spec and the sibling `native`/`omp-plugins` providers ([#1906](https://github.com/can1357/oh-my-pi/issues/1906)). + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/discovery/github.ts b/packages/coding-agent/src/discovery/github.ts index 0a92df7ff..bb2a200c5 100644 --- a/packages/coding-agent/src/discovery/github.ts +++ b/packages/coding-agent/src/discovery/github.ts @@ -10,6 +10,7 @@ * Capabilities: * - context-files: copilot-instructions.md in .github/ * - instructions: *.instructions.md in .github/instructions/ with applyTo frontmatter + * - skills: /SKILL.md in .github/skills/ (GitHub Agent Skills layout) */ import * as path from "node:path"; import { parseFrontmatter } from "@oh-my-pi/pi-utils"; @@ -17,9 +18,10 @@ import { registerProvider } from "../capability"; import { type ContextFile, contextFileCapability } from "../capability/context-file"; import { readFile } from "../capability/fs"; import { type Instruction, instructionCapability } from "../capability/instruction"; +import { type Skill, skillCapability } from "../capability/skill"; import type { LoadContext, LoadResult, SourceMeta } from "../capability/types"; -import { calculateDepth, createSourceMeta, getProjectPath, loadFilesFromDir } from "./helpers"; +import { calculateDepth, createSourceMeta, getProjectPath, loadFilesFromDir, scanSkillsFromDir } from "./helpers"; const PROVIDER_ID = "github"; const DISPLAY_NAME = "GitHub Copilot"; @@ -97,6 +99,32 @@ function transformInstruction(name: string, content: string, filePath: string, s }; } +// ============================================================================= +// Skills +// ============================================================================= + +/** + * Load skills from `.github/skills//SKILL.md`. + * + * GitHub documents this layout for Copilot Agent Skills and matches the + * non-recursive shape `scanSkillsFromDir` already expects. `requireDescription` + * is on to match the Agent Skills spec (name + description are mandatory) and + * the sibling `native`/`omp-plugins` providers. + * + * @see https://docs.github.com/en/copilot/how-tos/copilot-on-github/customize-copilot/customize-cloud-agent/add-skills + */ +async function loadSkills(ctx: LoadContext): Promise> { + const skillsDir = getProjectPath(ctx, "github", "skills"); + if (!skillsDir) return { items: [], warnings: [] }; + + return scanSkillsFromDir(ctx, { + dir: skillsDir, + providerId: PROVIDER_ID, + level: "project", + requireDescription: true, + }); +} + // ============================================================================= // Provider Registration // ============================================================================= @@ -116,3 +144,11 @@ registerProvider(instructionCapability.id, { priority: PRIORITY, load: loadInstructions, }); + +registerProvider(skillCapability.id, { + id: PROVIDER_ID, + displayName: DISPLAY_NAME, + description: "Load skills from .github/skills/*/SKILL.md", + priority: PRIORITY, + load: loadSkills, +}); diff --git a/packages/coding-agent/test/discovery/github-skills.test.ts b/packages/coding-agent/test/discovery/github-skills.test.ts new file mode 100644 index 000000000..019d16129 --- /dev/null +++ b/packages/coding-agent/test/discovery/github-skills.test.ts @@ -0,0 +1,70 @@ +/** + * Regression for https://github.com/can1357/oh-my-pi/issues/1906 + * + * The `github` discovery provider previously registered only context-files and + * instructions, leaving `.github/skills//SKILL.md` — the layout GitHub + * documents for agent skills — silently unscanned. This test pins the wiring: + * loading the `skills` capability with the github provider scoped to a cwd + * containing `.github/skills//SKILL.md` must surface the skill. + * + * @see https://docs.github.com/en/copilot/how-tos/copilot-on-github/customize-copilot/customize-cloud-agent/add-skills + */ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { loadCapability } from "@oh-my-pi/pi-coding-agent/capability"; +import { clearCache } from "@oh-my-pi/pi-coding-agent/capability/fs"; +import type { Skill } from "@oh-my-pi/pi-coding-agent/capability/skill"; +import "@oh-my-pi/pi-coding-agent/capability/skill"; +import "@oh-my-pi/pi-coding-agent/discovery/github"; + +function writeSkill(root: string, name: string, description: string | null): void { + const skillDir = path.join(root, name); + fs.mkdirSync(skillDir, { recursive: true }); + const frontmatter = + description === null ? `---\nname: ${name}\n---\n` : `---\nname: ${name}\ndescription: ${description}\n---\n`; + fs.writeFileSync(path.join(skillDir, "SKILL.md"), `${frontmatter}\n# ${name}\n\nSkill body.\n`); +} + +describe("github discovery — skills", () => { + let tempDir!: string; + + beforeEach(() => { + clearCache(); + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-github-skills-")); + }); + + afterEach(() => { + clearCache(); + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + + test("discovers .github/skills//SKILL.md via the github provider", async () => { + writeSkill(path.join(tempDir, ".github", "skills"), "demo-skill", "Demo skill for Copilot"); + + const result = await loadCapability("skills", { cwd: tempDir, providers: ["github"] }); + + const found = result.all.find(skill => skill.name === "demo-skill"); + expect(found).toBeDefined(); + expect(found?.path).toBe(path.join(tempDir, ".github", "skills", "demo-skill", "SKILL.md")); + expect(found?.level).toBe("project"); + expect(found?._source.provider).toBe("github"); + expect(result.warnings).toEqual([]); + }); + + test("skips skills missing a description (matches GitHub agent-skills standard)", async () => { + writeSkill(path.join(tempDir, ".github", "skills"), "no-desc", null); + + const result = await loadCapability("skills", { cwd: tempDir, providers: ["github"] }); + + expect(result.all.find(skill => skill.name === "no-desc")).toBeUndefined(); + }); + + test("returns no skills when .github/skills/ is absent", async () => { + const result = await loadCapability("skills", { cwd: tempDir, providers: ["github"] }); + + expect(result.all).toEqual([]); + expect(result.warnings).toEqual([]); + }); +}); From 95f64b61423924a913f5da1604aaf0cb3bfb94dc Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 05:47:09 +0000 Subject: [PATCH 22/48] fix(mcp): clear stale OAuth credential on definitive refresh failure When an HTTP MCP server returns invalid_grant (or invalid_token / revoked / plain 401 from the token endpoint) during OAuth refresh, MCPManager previously logged "MCP OAuth refresh failed, using existing token" and re-attached the stale access token as Authorization: Bearer on every subsequent request. The next tool-load 401'd with invalid_token, future sessions repeated the loop, and the only recovery was to hand-clear the credential row in agent.db. Reported with Logfire as the trigger; any remote HTTP MCP that rotates / revokes refresh tokens is affected. #resolveAuthConfig now reuses pi-ai's isDefinitiveOAuthFailure classifier (same one auth-broker and AuthStorage use for first-party providers): on a definitive failure it calls AuthStorage.remove(credentialId), drops the Bearer entirely, and the next request surfaces a clean auth error so the user can /mcp reauth (or /mcp unauth) to recover. Transient failures (network/fetch failed/ECONNREFUSED) still fall back to the existing token to ride out blips. Verified with new mcp-manager-oauth-refresh.test.ts (invalid_grant, 401, transient fallback, happy-path rotation). The full mcp-* test set (45 tests across 5 files) still passes. Fixes #1908 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/mcp/manager.ts | 61 ++++++--- .../test/mcp-manager-oauth-refresh.test.ts | 128 ++++++++++++++++++ 3 files changed, 172 insertions(+), 21 deletions(-) create mode 100644 packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..35d112dbe 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed remote MCP OAuth refresh failures leaving stale credentials in `agent.db`: when the token endpoint returns a definitive failure (`invalid_grant`, `invalid_token`, `revoked`, plain 401/403 not classified as transient), `MCPManager#resolveAuthConfig` now drops the credential via `AuthStorage.remove(credentialId)` and skips re-attaching the dead `Authorization: Bearer …` header. Previously a revoked refresh token kept producing `401 invalid_token` on every MCP request and survived restarts, so users had to hand-clear the credential row to recover; the next connect now surfaces a clean auth error and `/mcp reauth ` (or `/mcp unauth`) recovers without restarting. Transient refresh failures (network/`fetch failed`/`ECONNREFUSED`) still fall back to the existing access token ([#1908](https://github.com/can1357/oh-my-pi/issues/1908)). + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/mcp/manager.ts b/packages/coding-agent/src/mcp/manager.ts index 8477b02d8..250b43a27 100644 --- a/packages/coding-agent/src/mcp/manager.ts +++ b/packages/coding-agent/src/mcp/manager.ts @@ -6,7 +6,7 @@ */ import * as path from "node:path"; import * as url from "node:url"; -import type { TSchema } from "@oh-my-pi/pi-ai"; +import { isDefinitiveOAuthFailure, type TSchema } from "@oh-my-pi/pi-ai"; import { logger } from "@oh-my-pi/pi-utils"; import type { SourceMeta } from "../capability/types"; import { resolveConfigValue } from "../config/resolve-config-value"; @@ -1184,29 +1184,48 @@ export class MCPManager { await this.#authStorage.set(credentialId, refreshedCredential); credential = refreshedCredential; } catch (refreshError) { - logger.warn("MCP OAuth refresh failed, using existing token", { - credentialId, - error: refreshError, - }); + const errorMsg = refreshError instanceof Error ? refreshError.message : String(refreshError); + if (isDefinitiveOAuthFailure(errorMsg)) { + // `invalid_grant` / `invalid_token` / 401 from the token endpoint means + // the server has retired this credential — keeping the stale access + // token would just re-fail with 401 on every MCP request and leave a + // poisoned row in agent.db that survives restarts. Drop it now so the + // next connect attempt surfaces a clean "needs reauth" failure and + // the user can recover with `/mcp reauth ` (or `/mcp unauth` + // to forget the server entirely). + logger.warn("MCP OAuth refresh failed definitively; cleared credential", { + credentialId, + error: errorMsg, + }); + await this.#authStorage.remove(credentialId); + credential = undefined; + } else { + logger.warn("MCP OAuth refresh failed, using existing token", { + credentialId, + error: refreshError, + }); + } } } - if (resolved.type === "http" || resolved.type === "sse") { - resolved = { - ...resolved, - headers: { - ...resolved.headers, - Authorization: `Bearer ${credential.access}`, - }, - }; - } else { - resolved = { - ...resolved, - env: { - ...resolved.env, - OAUTH_ACCESS_TOKEN: credential.access, - }, - }; + if (credential?.type === "oauth") { + if (resolved.type === "http" || resolved.type === "sse") { + resolved = { + ...resolved, + headers: { + ...resolved.headers, + Authorization: `Bearer ${credential.access}`, + }, + }; + } else { + resolved = { + ...resolved, + env: { + ...resolved.env, + OAUTH_ACCESS_TOKEN: credential.access, + }, + }; + } } } } catch (error) { diff --git a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts new file mode 100644 index 000000000..a1455aa16 --- /dev/null +++ b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts @@ -0,0 +1,128 @@ +/** + * Regression tests for MCP OAuth refresh failure handling (issue #1908). + * + * Before the fix, a refresh that came back with `invalid_grant` was logged and + * the stale access token was re-attached as `Authorization: Bearer …` on every + * subsequent MCP request — producing a permanent 401 / reauth loop until the + * user hand-cleared the row in `agent.db`. The fix routes definitive failures + * (`invalid_grant`, `invalid_token`, `revoked`, plain 401/403 not classified as + * transient) through `AuthStorage.remove(credentialId)` and suppresses the + * Bearer injection, so the next request surfaces a clean auth error instead. + */ +import { Database } from "bun:sqlite"; +import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; +import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai"; +import { MCPManager } from "../src/mcp/manager"; +import * as oauthFlow from "../src/mcp/oauth-flow"; +import type { MCPServerConfig } from "../src/mcp/types"; + +const CREDENTIAL_ID = "mcp_oauth_test_1908"; +const TOKEN_URL = "https://example.com/oauth/token"; +const STALE_ACCESS = "stale-access-token"; +const STALE_REFRESH = "stale-refresh-token"; + +/** Build a `Headers` snapshot from a prepared MCP config. */ +function getAuthorizationHeader(config: MCPServerConfig): string | undefined { + if (config.type !== "http" && config.type !== "sse") return undefined; + return config.headers?.Authorization; +} + +describe("MCPManager OAuth refresh failure", () => { + let manager: MCPManager; + let authStorage: AuthStorage; + let serverConfig: MCPServerConfig; + + beforeEach(async () => { + const store = new SqliteAuthCredentialStore(new Database(":memory:")); + authStorage = new AuthStorage(store); + await authStorage.reload(); + + // Seed an expired credential so `#resolveAuthConfig` decides to refresh + // (a non-expired credential takes the no-refresh branch and never reaches + // the bug). + await authStorage.set(CREDENTIAL_ID, { + type: "oauth", + access: STALE_ACCESS, + refresh: STALE_REFRESH, + expires: Date.now() - 60_000, + }); + + manager = new MCPManager(process.cwd()); + manager.setAuthStorage(authStorage); + + serverConfig = { + type: "http", + url: "https://logfire.example.com/mcp", + auth: { + type: "oauth", + credentialId: CREDENTIAL_ID, + tokenUrl: TOKEN_URL, + }, + }; + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + test("clears the credential and skips Bearer injection on invalid_grant", async () => { + const refreshSpy = vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockRejectedValue( + new Error( + 'MCP OAuth refresh failed: 400 {"error":"invalid_grant","error_description":"Refresh token has been revoked"}', + ), + ); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(refreshSpy).toHaveBeenCalledTimes(1); + // The poisoned Bearer must not be re-injected — that is the loop the user + // reported (#1908). + expect(getAuthorizationHeader(prepared)).toBeUndefined(); + // The credential row is gone so neither this nor a future session keeps + // shipping the dead refresh token. + expect(authStorage.get(CREDENTIAL_ID)).toBeUndefined(); + }); + + test("clears the credential when the token endpoint replies HTTP 401", async () => { + vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockRejectedValue( + new Error("MCP OAuth refresh failed: 401 Unauthorized"), + ); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(getAuthorizationHeader(prepared)).toBeUndefined(); + expect(authStorage.get(CREDENTIAL_ID)).toBeUndefined(); + }); + + test("keeps the credential and falls back to the existing token on transient failure", async () => { + // Network blip during refresh — the access token may still be live, so + // we preserve the prior behavior of one best-effort attempt with what we + // already have rather than tearing down the credential. + vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockRejectedValue( + new Error("MCP OAuth refresh failed: fetch failed ECONNREFUSED 127.0.0.1:443"), + ); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(getAuthorizationHeader(prepared)).toBe(`Bearer ${STALE_ACCESS}`); + const remaining = authStorage.get(CREDENTIAL_ID); + expect(remaining?.type).toBe("oauth"); + }); + + test("persists rotated credential on successful refresh", async () => { + // Sanity: the happy path still rotates the row and attaches the fresh + // Bearer. Guards against accidentally short-circuiting refresh while + // fixing the failure path. + vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockResolvedValue({ + access: "fresh-access", + refresh: "fresh-refresh", + expires: Date.now() + 3_600_000, + }); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(getAuthorizationHeader(prepared)).toBe("Bearer fresh-access"); + const remaining = authStorage.get(CREDENTIAL_ID); + expect(remaining).toMatchObject({ type: "oauth", access: "fresh-access", refresh: "fresh-refresh" }); + }); +}); From 93f25ebf66137590d9768004f9aa38526773f3de Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 05:47:33 +0000 Subject: [PATCH 23/48] style: bun run fix --- .../test/mcp-manager-oauth-refresh.test.ts | 12 +++++++----- 1 file changed, 7 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts index a1455aa16..f137d9801 100644 --- a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts +++ b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts @@ -66,11 +66,13 @@ describe("MCPManager OAuth refresh failure", () => { }); test("clears the credential and skips Bearer injection on invalid_grant", async () => { - const refreshSpy = vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockRejectedValue( - new Error( - 'MCP OAuth refresh failed: 400 {"error":"invalid_grant","error_description":"Refresh token has been revoked"}', - ), - ); + const refreshSpy = vi + .spyOn(oauthFlow, "refreshMCPOAuthToken") + .mockRejectedValue( + new Error( + 'MCP OAuth refresh failed: 400 {"error":"invalid_grant","error_description":"Refresh token has been revoked"}', + ), + ); const prepared = await manager.prepareConfig(serverConfig); From ccc3533c45979aed1a1ec29cf2a719aa07a83658 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 05:57:47 +0000 Subject: [PATCH 24/48] fix(tui): differentiate /tree empty-state for fresh sessions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The session-tree selector collapsed to "No entries found" whenever the default filter rejected every entry. On a fresh session that is the normal case: `sdk.ts` writes `model_change` + `thinking_level_change` at startup so model + thinking state survive resumes, and the default filter treats both as bookkeeping. `selector-controller.showTreeSelector` guards on `tree.length === 0` so the selector still opens, and the user sees an unexplained empty panel with `(0/0)` next to a "Recent sessions" list that does contain data. Split the empty-state branch into three shapes: - `flatNodes.length === 0` → unchanged "No entries found". - `searchQuery` non-empty → "No entries match search \"…\"" plus a Backspace hint, with the real total in `(0/N)`. - otherwise → "N entries hidden by the current filter [mode]" plus "Press Alt+A to show all, Alt+D for default", with the real total in `(0/N)`. So the fresh-session case now explains why the panel is empty and how to widen it instead of reading as "/tree is broken". Fixes #1909 --- packages/coding-agent/CHANGELOG.md | 4 + .../src/modes/components/tree-selector.ts | 33 ++++++- .../tree-selector-empty-state-1909.test.ts | 89 +++++++++++++++++++ 3 files changed, 124 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..0c5e4de5c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `/tree` rendering a bare "No entries found" line on a fresh session where the only persisted entries are the `model_change` + `thinking_level_change` written by `sdk.ts` at startup — both are hidden by the tree-selector's default filter, so `#filteredNodes.length === 0` while `tree.length === 2` and the controller's `tree.length === 0` short-circuit never fired. The selector now splits the empty-state into three distinct shapes — truly empty tree, search query with no matches, and filter mode rejecting every entry — surfacing the cause and the recovery key (`Alt+A` to show all, `Backspace` to clear a stale search) so users on a fresh session can see immediately that the panel isn't broken ([#1909](https://github.com/can1357/oh-my-pi/issues/1909)). + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/modes/components/tree-selector.ts b/packages/coding-agent/src/modes/components/tree-selector.ts index c2df593aa..48a4e7880 100644 --- a/packages/coding-agent/src/modes/components/tree-selector.ts +++ b/packages/coding-agent/src/modes/components/tree-selector.ts @@ -441,8 +441,37 @@ class TreeList implements Component { const lines: string[] = []; if (this.#filteredNodes.length === 0) { - lines.push(truncateToWidth(theme.fg("muted", " No entries found"), width)); - lines.push(truncateToWidth(theme.fg("muted", ` (0/0)${this.#getFilterLabel()}`), width)); + // Three empty-state shapes: + // - flatNodes empty → no entries at all (truly fresh session). + // - search query rejects everything → tell the user the search is the cause. + // - filter mode rejects everything → tell the user the filter is the cause and + // how to widen it. Otherwise fresh sessions whose only persisted entries are + // `model_change` + `thinking_level_change` (both hidden by the default filter) + // read as "broken /tree" — see #1909. + if (this.#flatNodes.length === 0) { + lines.push(truncateToWidth(theme.fg("muted", " No entries found"), width)); + lines.push(truncateToWidth(theme.fg("muted", ` (0/0)${this.#getFilterLabel()}`), width)); + } else if (this.#searchQuery.length > 0) { + lines.push( + truncateToWidth(theme.fg("muted", ` No entries match search "${this.#searchQuery}"`), width), + ); + lines.push(truncateToWidth(theme.fg("muted", " Press Backspace to clear the search"), width)); + lines.push( + truncateToWidth(theme.fg("muted", ` (0/${this.#flatNodes.length})${this.#getFilterLabel()}`), width), + ); + } else { + const filterLabel = this.#getFilterLabel().trim() || "[default]"; + lines.push( + truncateToWidth( + theme.fg("muted", ` ${this.#flatNodes.length} entries hidden by the current filter ${filterLabel}`), + width, + ), + ); + lines.push(truncateToWidth(theme.fg("muted", " Press Alt+A to show all, Alt+D for default"), width)); + lines.push( + truncateToWidth(theme.fg("muted", ` (0/${this.#flatNodes.length})${this.#getFilterLabel()}`), width), + ); + } return lines; } diff --git a/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts b/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts new file mode 100644 index 000000000..3b6365186 --- /dev/null +++ b/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts @@ -0,0 +1,89 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import { TreeSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tree-selector"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { SessionEntry, SessionTreeNode } from "@oh-my-pi/pi-coding-agent/session/session-manager"; + +beforeAll(async () => { + await initTheme(false, undefined, undefined, "dark", "light"); +}); + +function freshSessionTree(): SessionTreeNode[] { + // Mirror what `sdk.ts` writes on session start: a `model_change` plus a + // `thinking_level_change`. Both are settings/bookkeeping entries that the + // tree selector's default filter hides — a fresh session contains nothing + // else, so the selector sees `flatNodes.length === 2`, `filteredNodes.length === 0`. + const modelChange: SessionEntry = { + type: "model_change", + id: "e1", + parentId: null, + timestamp: new Date().toISOString(), + model: "anthropic/claude-sonnet-4-20250514", + }; + const thinkingChange: SessionEntry = { + type: "thinking_level_change", + id: "e2", + parentId: "e1", + timestamp: new Date().toISOString(), + thinkingLevel: "medium", + }; + return [ + { + entry: modelChange, + children: [{ entry: thinkingChange, children: [] }], + }, + ]; +} + +function userMessageTree(): SessionTreeNode[] { + const entry: SessionEntry = { + type: "message", + id: "e1", + parentId: null, + timestamp: new Date().toISOString(), + message: { role: "user", content: "hello there", timestamp: 1 }, + }; + return [{ entry, children: [] }]; +} + +function renderSelector(selector: TreeSelectorComponent): string { + const lines = (selector as unknown as { render: (w: number) => string[] }).render(120); + return Bun.stripANSI(lines.join("\n")); +} + +describe("issue #1909: tree-selector empty-state messaging", () => { + it("explains that the filter — not missing data — is hiding entries on a fresh session", () => { + const selector = new TreeSelectorComponent(freshSessionTree(), "e2", 60, () => {}, () => {}); + const text = renderSelector(selector); + + // Filter-hiding hint and recovery key must both be present so the user knows + // the panel isn't broken and can widen the view without leaving the screen. + expect(text).toContain("hidden by the current filter"); + expect(text).toContain("[default]"); + expect(text.toLowerCase()).toContain("alt+a"); + // Total count must reflect the real flatNodes count, not 0/0 (otherwise the + // "filter hides things" framing is unconvincing). + expect(text).toContain("(0/2)"); + }); + + it("explains a zero-result search as a search problem, not a filter problem", () => { + const selector = new TreeSelectorComponent(userMessageTree(), "e1", 60, () => {}, () => {}); + // Type a character that won't match anything in the tree. + selector.handleInput("z"); + const text = renderSelector(selector); + + expect(text).toContain('No entries match search "z"'); + expect(text.toLowerCase()).toContain("backspace"); + // Must NOT misattribute the empty result to the filter mode. + expect(text).not.toContain("hidden by the current filter"); + }); + + it("falls back to the bare 'No entries found' line when the tree is genuinely empty", () => { + const selector = new TreeSelectorComponent([], null, 60, () => {}, () => {}); + const text = renderSelector(selector); + + expect(text).toContain("No entries found"); + expect(text).toContain("(0/0)"); + // Don't tell the user to widen the filter when there's nothing to widen to. + expect(text).not.toContain("hidden by the current filter"); + }); +}); From 01677a0c1fcadcf56b904a0f9348f0b5ab898f92 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 05:58:04 +0000 Subject: [PATCH 25/48] style: bun run fix --- .../src/modes/components/tree-selector.ts | 4 +--- .../tree-selector-empty-state-1909.test.ts | 24 ++++++++++++++++--- 2 files changed, 22 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/src/modes/components/tree-selector.ts b/packages/coding-agent/src/modes/components/tree-selector.ts index 48a4e7880..c015f8ca4 100644 --- a/packages/coding-agent/src/modes/components/tree-selector.ts +++ b/packages/coding-agent/src/modes/components/tree-selector.ts @@ -452,9 +452,7 @@ class TreeList implements Component { lines.push(truncateToWidth(theme.fg("muted", " No entries found"), width)); lines.push(truncateToWidth(theme.fg("muted", ` (0/0)${this.#getFilterLabel()}`), width)); } else if (this.#searchQuery.length > 0) { - lines.push( - truncateToWidth(theme.fg("muted", ` No entries match search "${this.#searchQuery}"`), width), - ); + lines.push(truncateToWidth(theme.fg("muted", ` No entries match search "${this.#searchQuery}"`), width)); lines.push(truncateToWidth(theme.fg("muted", " Press Backspace to clear the search"), width)); lines.push( truncateToWidth(theme.fg("muted", ` (0/${this.#flatNodes.length})${this.#getFilterLabel()}`), width), diff --git a/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts b/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts index 3b6365186..273306856 100644 --- a/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts +++ b/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts @@ -52,7 +52,13 @@ function renderSelector(selector: TreeSelectorComponent): string { describe("issue #1909: tree-selector empty-state messaging", () => { it("explains that the filter — not missing data — is hiding entries on a fresh session", () => { - const selector = new TreeSelectorComponent(freshSessionTree(), "e2", 60, () => {}, () => {}); + const selector = new TreeSelectorComponent( + freshSessionTree(), + "e2", + 60, + () => {}, + () => {}, + ); const text = renderSelector(selector); // Filter-hiding hint and recovery key must both be present so the user knows @@ -66,7 +72,13 @@ describe("issue #1909: tree-selector empty-state messaging", () => { }); it("explains a zero-result search as a search problem, not a filter problem", () => { - const selector = new TreeSelectorComponent(userMessageTree(), "e1", 60, () => {}, () => {}); + const selector = new TreeSelectorComponent( + userMessageTree(), + "e1", + 60, + () => {}, + () => {}, + ); // Type a character that won't match anything in the tree. selector.handleInput("z"); const text = renderSelector(selector); @@ -78,7 +90,13 @@ describe("issue #1909: tree-selector empty-state messaging", () => { }); it("falls back to the bare 'No entries found' line when the tree is genuinely empty", () => { - const selector = new TreeSelectorComponent([], null, 60, () => {}, () => {}); + const selector = new TreeSelectorComponent( + [], + null, + 60, + () => {}, + () => {}, + ); const text = renderSelector(selector); expect(text).toContain("No entries found"); From 510a2b8596ea115ca6bbb55144dc117b2f1f555c Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 07:54:37 +0000 Subject: [PATCH 26/48] fix(coding-agent): preserved setup login urls Wrapped setup sign-in OAuth status lines instead of truncating them, and added a full OSC8 login link so narrow terminals still expose the complete URL. Fixes #1919 --- .../src/modes/setup-wizard/scenes/sign-in.ts | 13 +++- .../test/setup-wizard-sign-in.test.ts | 66 +++++++++++++++++++ 2 files changed, 76 insertions(+), 3 deletions(-) create mode 100644 packages/coding-agent/test/setup-wizard-sign-in.test.ts diff --git a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts index b854f85cd..cc841b942 100644 --- a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts +++ b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts @@ -1,6 +1,6 @@ import type { AuthStorage } from "@oh-my-pi/pi-ai"; import type { OAuthProvider } from "@oh-my-pi/pi-ai/utils/oauth/types"; -import { Input, matchesKey, truncateToWidth } from "@oh-my-pi/pi-tui"; +import { Input, matchesKey, wrapTextWithAnsi } from "@oh-my-pi/pi-tui"; import { getAgentDbPath } from "@oh-my-pi/pi-utils"; import { OAuthSelectorComponent } from "../../components/oauth-selector"; import { theme } from "../../theme/theme"; @@ -15,6 +15,10 @@ const CALLBACK_SERVER_PROVIDERS: Partial> = { "google-antigravity": true, }; +function loginUrlLink(url: string): string { + return `\x1b]8;;${url}\x07Open login URL\x1b]8;;\x07`; +} + interface PromptState { message: string; placeholder?: string; @@ -79,7 +83,7 @@ export class SignInTab implements SetupTab { lines.push(...this.#selector.render(width)); } if (this.#statusLines.length > 0) { - lines.push("", ...this.#statusLines.map(line => truncateToWidth(line, width))); + lines.push("", ...this.#statusLines.flatMap(line => wrapTextWithAnsi(line, width))); } if (this.#prompt) { lines.push("", theme.fg("warning", this.#prompt.message)); @@ -116,7 +120,9 @@ export class SignInTab implements SetupTab { await this.#authStorage.login(providerId as OAuthProvider, { signal: this.#loginAbort.signal, onAuth: info => { - this.#statusLines.push(theme.fg("accent", `Open this URL: ${info.url}`)); + this.#statusLines.push(theme.fg("accent", "Open this URL in your browser:")); + this.#statusLines.push(theme.fg("accent", info.url)); + this.#statusLines.push(theme.fg("dim", loginUrlLink(info.url))); if (info.instructions) { this.#statusLines.push(theme.fg("warning", info.instructions)); } @@ -174,6 +180,7 @@ export class SignInTab implements SetupTab { this.#resolvePrompt(value); }; input.onEscape = () => { + this.#loginAbort?.abort(); this.#resolvePrompt(""); }; this.host.setFocus(input); diff --git a/packages/coding-agent/test/setup-wizard-sign-in.test.ts b/packages/coding-agent/test/setup-wizard-sign-in.test.ts new file mode 100644 index 000000000..6c18d803e --- /dev/null +++ b/packages/coding-agent/test/setup-wizard-sign-in.test.ts @@ -0,0 +1,66 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import type { OAuthLoginCallbacks, OAuthProviderId } from "@oh-my-pi/pi-ai/utils/oauth/types"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { SignInTab } from "../src/modes/setup-wizard/scenes/sign-in"; +import type { SetupSceneHost } from "../src/modes/setup-wizard/scenes/types"; + +beforeAll(async () => { + await initTheme(); +}); + +describe("SignInTab", () => { + it("wraps the full OAuth URL and keeps a full OSC8 link at narrow widths", async () => { + const url = `https://example.com/oauth/authorize?client_id=omp&redirect_uri=http%3A%2F%2Flocalhost%3A45454%2Fcallback&state=${"a".repeat(96)}`; + const loginGate = Promise.withResolvers(); + const openedUrls: string[] = []; + + const authStorage = { + has: (_providerId: string) => false, + hasAuth: (_providerId: string) => false, + async login(_provider: OAuthProviderId, ctrl: OAuthLoginCallbacks): Promise { + ctrl.onAuth({ url }); + await loginGate.promise; + }, + } as unknown as AuthStorage; + + const host = { + ctx: { + openInBrowser(openedUrl: string): void { + openedUrls.push(openedUrl); + }, + session: { + modelRegistry: { + authStorage, + async refresh(): Promise {}, + }, + }, + }, + requestRender(): void {}, + finish(): void {}, + setFocus(): void {}, + restoreFocus(): void {}, + } as unknown as SetupSceneHost; + + const tab = new SignInTab(host); + try { + for (const char of "anthropic") { + tab.handleInput(char); + } + tab.handleInput("\n"); + + const rendered = tab.render(36); + const compact = rendered.map(line => Bun.stripANSI(line).trim()).join(""); + const urlStart = compact.indexOf(url.slice(0, 24)); + expect(urlStart).toBeGreaterThanOrEqual(0); + expect(compact.slice(urlStart)).toContain(url); + expect(compact.slice(urlStart, urlStart + url.length + 8)).not.toContain("…"); + expect(rendered.join("\n")).toContain(`\x1b]8;;${url}\x07Open login URL\x1b]8;;\x07`); + expect(openedUrls).toEqual([url]); + } finally { + tab.dispose(); + loginGate.resolve(); + await loginGate.promise; + } + }); +}); From c4fa93e460f5eddda5a48da0b996dc5e46f25e53 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 07:59:28 +0000 Subject: [PATCH 27/48] fix(plugins): preserved null package import exclusions Kept explicit null package imports as exclusions so exact entries and active conditions do not fall through to wildcard or fallback targets.\n\nFixes #1889 --- .../extensibility/plugins/legacy-pi-compat.ts | 24 +++-- .../test/plugin-extensions-discovery.test.ts | 102 ++++++++++++++++++ 2 files changed, 119 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index 8ecf04a6f..70817b916 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -49,6 +49,7 @@ const SOURCE_MODULE_EXTENSIONS = [".ts", ".tsx", ".mts", ".cts", ".js", ".jsx", const SUPPORTED_PACKAGE_IMPORT_CONDITIONS = new Set(["bun", "node", "import", "default"]); const packageRootCache = new Map(); const packageImportsCache = new Map | null>(); +const PACKAGE_IMPORT_EXCLUDED = Symbol("packageImportExcluded"); // Extensions that imported `@sinclair/typebox` directly used to resolve against a // real `@sinclair/typebox` install. The runtime dep was replaced with the Zod-backed @@ -334,14 +335,20 @@ async function readPackageImports(packageRoot: string): Promise { expect(extension?.commands.has("json-schema-ext")).toBe(true); }); + it("preserves exact null package import exclusions ahead of wildcard fallbacks", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "null-exact-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "null-exact-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "null-exact-import-plugin", + version: "1.0.0", + imports: { + "#src/internal": null, + "#src/*": "./src/*", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync(path.join(pluginDir, "src", "internal.ts"), 'export const commandName = "null-exact-ext";'); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#src/internal";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + const pluginError = result.errors.find(err => err.path === extensionPath); + + expect(pluginError?.error).toContain("#src/internal"); + expect(extension).toBeUndefined(); + }); + + it("preserves active null conditional package import exclusions", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "null-conditional-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "null-conditional-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "null-conditional-import-plugin", + version: "1.0.0", + imports: { + "#blocked": { + node: null, + default: "./src/blocked.ts", + }, + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync(path.join(pluginDir, "src", "blocked.ts"), 'export const commandName = "null-conditional-ext";'); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#blocked";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + const pluginError = result.errors.find(err => err.path === extensionPath); + + expect(pluginError?.error).toContain("#blocked"); + expect(extension).toBeUndefined(); + }); + it("rewrites side-effect imports of package-import aliases and legacy Pi scopes", async () => { const pluginsDir = getPluginsDir(); const pluginDir = path.join(pluginsDir, "node_modules", "side-effect-plugin"); From 9333fccc0377bdcb3d6b1b468d2a888628a5aaff Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 08:00:16 +0000 Subject: [PATCH 28/48] fix(coding-agent): kept setup login prompts visible Rendered manual-code prompts before wrapped OAuth status lines so wizard body clipping cannot hide the focused input behind a long login URL. Fixes #1919 --- .../coding-agent/src/modes/setup-wizard/scenes/sign-in.ts | 6 +++--- packages/coding-agent/test/setup-wizard-sign-in.test.ts | 6 ++++++ 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts index cc841b942..67240e134 100644 --- a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts +++ b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts @@ -82,9 +82,6 @@ export class SignInTab implements SetupTab { } else { lines.push(...this.#selector.render(width)); } - if (this.#statusLines.length > 0) { - lines.push("", ...this.#statusLines.flatMap(line => wrapTextWithAnsi(line, width))); - } if (this.#prompt) { lines.push("", theme.fg("warning", this.#prompt.message)); if (this.#prompt.placeholder) { @@ -92,6 +89,9 @@ export class SignInTab implements SetupTab { } lines.push(this.#prompt.input.render(width)[0] ?? ""); } + if (this.#statusLines.length > 0) { + lines.push("", ...this.#statusLines.flatMap(line => wrapTextWithAnsi(line, width))); + } return lines; } diff --git a/packages/coding-agent/test/setup-wizard-sign-in.test.ts b/packages/coding-agent/test/setup-wizard-sign-in.test.ts index 6c18d803e..43c55bb3e 100644 --- a/packages/coding-agent/test/setup-wizard-sign-in.test.ts +++ b/packages/coding-agent/test/setup-wizard-sign-in.test.ts @@ -20,7 +20,9 @@ describe("SignInTab", () => { hasAuth: (_providerId: string) => false, async login(_provider: OAuthProviderId, ctrl: OAuthLoginCallbacks): Promise { ctrl.onAuth({ url }); + const prompt = ctrl.onManualCodeInput?.(); await loginGate.promise; + await prompt; }, } as unknown as AuthStorage; @@ -57,6 +59,10 @@ describe("SignInTab", () => { expect(compact.slice(urlStart, urlStart + url.length + 8)).not.toContain("…"); expect(rendered.join("\n")).toContain(`\x1b]8;;${url}\x07Open login URL\x1b]8;;\x07`); expect(openedUrls).toEqual([url]); + + const clippedBody = rendered.slice(0, 7).map(line => Bun.stripANSI(line).trim()); + expect(clippedBody).toContain("Paste the authorization code (or full redirect URL):"); + expect(clippedBody.some(line => line.startsWith(">"))).toBe(true); } finally { tab.dispose(); loginGate.resolve(); From b896007e750478fbf527d0f63ec402a2753e76d7 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 08:07:43 +0000 Subject: [PATCH 29/48] fix(coding-agent): pinned setup login link above prompt Hoisted an always-visible OSC8 'Browser login' row in the setup sign-in tab so wizard body clipping never hides both the clickable link and the focused manual-code prompt. The wrappable URL text still renders below for terminals without OSC 8 support. Fixes #1919 --- .../src/modes/setup-wizard/scenes/sign-in.ts | 22 +++++++++++++++---- .../test/setup-wizard-sign-in.test.ts | 16 +++++++++----- 2 files changed, 28 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts index 67240e134..632455e52 100644 --- a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts +++ b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts @@ -37,6 +37,7 @@ export class SignInTab implements SetupTab { #authStorage: AuthStorage; #selector: OAuthSelectorComponent; #statusLines: string[] = []; + #authUrl: string | undefined; #prompt: PromptState | undefined; #promptResolve: ((value: string) => void) | undefined; #loginAbort: AbortController | undefined; @@ -78,10 +79,15 @@ export class SignInTab implements SetupTab { render(width: number): string[] { const lines = [theme.fg("muted", "Pick a provider to sign in — you can connect more than one."), ""]; if (this.#loggingInProvider) { - lines.push(theme.bold(`Signing in to ${this.#loggingInProvider}`), ""); + lines.push(theme.bold(`Signing in to ${this.#loggingInProvider}`)); } else { lines.push(...this.#selector.render(width)); } + if (this.#authUrl) { + // Always-visible actionable row — wizard body clipping cannot push it + // off-screen even when the wrapped URL below would not fit. + lines.push("", theme.fg("accent", `Browser login: ${loginUrlLink(this.#authUrl)}`)); + } if (this.#prompt) { lines.push("", theme.fg("warning", this.#prompt.message)); if (this.#prompt.placeholder) { @@ -89,6 +95,11 @@ export class SignInTab implements SetupTab { } lines.push(this.#prompt.input.render(width)[0] ?? ""); } + if (this.#authUrl) { + // Plain-text URL for terminals without OSC 8 support. May clip on tiny + // terminals; the OSC 8 row above remains the always-visible fallback. + lines.push("", ...wrapTextWithAnsi(theme.fg("dim", this.#authUrl), width)); + } if (this.#statusLines.length > 0) { lines.push("", ...this.#statusLines.flatMap(line => wrapTextWithAnsi(line, width))); } @@ -113,6 +124,7 @@ export class SignInTab implements SetupTab { this.#selector.stopValidation(); this.#loggingInProvider = providerId; this.#statusLines = [theme.fg("dim", "Starting OAuth flow…")]; + this.#authUrl = undefined; this.#loginAbort = new AbortController(); this.host.restoreFocus(); this.host.requestRender(); @@ -120,9 +132,8 @@ export class SignInTab implements SetupTab { await this.#authStorage.login(providerId as OAuthProvider, { signal: this.#loginAbort.signal, onAuth: info => { - this.#statusLines.push(theme.fg("accent", "Open this URL in your browser:")); - this.#statusLines.push(theme.fg("accent", info.url)); - this.#statusLines.push(theme.fg("dim", loginUrlLink(info.url))); + this.#authUrl = info.url; + this.#statusLines = []; if (info.instructions) { this.#statusLines.push(theme.fg("warning", info.instructions)); } @@ -146,6 +157,7 @@ export class SignInTab implements SetupTab { theme.fg("success", `${theme.status.success} Signed in to ${providerId}`), theme.fg("dim", `Credentials saved to ${getAgentDbPath()}`), ]; + this.#authUrl = undefined; this.#loggingInProvider = undefined; this.#loginAbort = undefined; this.#selector.stopValidation(); @@ -156,12 +168,14 @@ export class SignInTab implements SetupTab { if (this.#disposed) return; if (this.#loginAbort?.signal.aborted) { this.#statusLines = [theme.fg("dim", "Login cancelled.")]; + this.#authUrl = undefined; } else { const message = error instanceof Error ? error.message : String(error); this.#statusLines = [ theme.fg("error", `Login failed: ${message}`), theme.fg("dim", "Choose another provider or press Esc to continue."), ]; + this.#authUrl = undefined; } this.#loggingInProvider = undefined; this.#loginAbort = undefined; diff --git a/packages/coding-agent/test/setup-wizard-sign-in.test.ts b/packages/coding-agent/test/setup-wizard-sign-in.test.ts index 43c55bb3e..dbc24c783 100644 --- a/packages/coding-agent/test/setup-wizard-sign-in.test.ts +++ b/packages/coding-agent/test/setup-wizard-sign-in.test.ts @@ -10,7 +10,7 @@ beforeAll(async () => { }); describe("SignInTab", () => { - it("wraps the full OAuth URL and keeps a full OSC8 link at narrow widths", async () => { + it("keeps the OSC8 login link and manual-code prompt above clipped wizard rows", async () => { const url = `https://example.com/oauth/authorize?client_id=omp&redirect_uri=http%3A%2F%2Flocalhost%3A45454%2Fcallback&state=${"a".repeat(96)}`; const loginGate = Promise.withResolvers(); const openedUrls: string[] = []; @@ -53,16 +53,20 @@ describe("SignInTab", () => { const rendered = tab.render(36); const compact = rendered.map(line => Bun.stripANSI(line).trim()).join(""); - const urlStart = compact.indexOf(url.slice(0, 24)); - expect(urlStart).toBeGreaterThanOrEqual(0); - expect(compact.slice(urlStart)).toContain(url); - expect(compact.slice(urlStart, urlStart + url.length + 8)).not.toContain("…"); + expect(compact).toContain(url); + expect(compact).not.toContain("…"); expect(rendered.join("\n")).toContain(`\x1b]8;;${url}\x07Open login URL\x1b]8;;\x07`); expect(openedUrls).toEqual([url]); - const clippedBody = rendered.slice(0, 7).map(line => Bun.stripANSI(line).trim()); + // On a ~24-row terminal the wizard body ends up ~8 rows; both the + // OSC8 link row and the focused input must survive that clip. + const clippedBody = rendered.slice(0, 8).map(line => Bun.stripANSI(line).trim()); + expect(clippedBody.some(line => line === "Browser login: Open login URL")).toBe(true); expect(clippedBody).toContain("Paste the authorization code (or full redirect URL):"); expect(clippedBody.some(line => line.startsWith(">"))).toBe(true); + expect(clippedBody.indexOf("Browser login: Open login URL")).toBeLessThan( + clippedBody.findIndex(line => line.startsWith(">")), + ); } finally { tab.dispose(); loginGate.resolve(); From 2ef758f5e32876ee58cbd1bad808cfa0420d8819 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 08:12:43 +0000 Subject: [PATCH 30/48] fix(coding-agent): kept setup login url row visible Reserved the first wrapped plain-text login URL rows above the manual-code prompt, while rendering the complete wrapped URL again below the prompt for terminals without OSC 8 support. Fixes #1919 --- .../src/modes/setup-wizard/scenes/sign-in.ts | 19 +++++++++---------- .../test/setup-wizard-sign-in.test.ts | 13 +++++++------ 2 files changed, 16 insertions(+), 16 deletions(-) diff --git a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts index 632455e52..7ddc176f4 100644 --- a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts +++ b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts @@ -77,31 +77,30 @@ export class SignInTab implements SetupTab { } render(width: number): string[] { - const lines = [theme.fg("muted", "Pick a provider to sign in — you can connect more than one."), ""]; + const lines: string[] = []; if (this.#loggingInProvider) { lines.push(theme.bold(`Signing in to ${this.#loggingInProvider}`)); } else { + lines.push(theme.fg("muted", "Pick a provider to sign in — you can connect more than one."), ""); lines.push(...this.#selector.render(width)); } + + const urlLines = this.#authUrl ? wrapTextWithAnsi(theme.fg("dim", this.#authUrl), width) : []; if (this.#authUrl) { - // Always-visible actionable row — wizard body clipping cannot push it - // off-screen even when the wrapped URL below would not fit. - lines.push("", theme.fg("accent", `Browser login: ${loginUrlLink(this.#authUrl)}`)); + lines.push(theme.fg("accent", `Browser login: ${loginUrlLink(this.#authUrl)}`), ...urlLines.slice(0, 2)); } if (this.#prompt) { - lines.push("", theme.fg("warning", this.#prompt.message)); + lines.push(theme.fg("warning", this.#prompt.message)); if (this.#prompt.placeholder) { lines.push(theme.fg("dim", this.#prompt.placeholder)); } lines.push(this.#prompt.input.render(width)[0] ?? ""); } - if (this.#authUrl) { - // Plain-text URL for terminals without OSC 8 support. May clip on tiny - // terminals; the OSC 8 row above remains the always-visible fallback. - lines.push("", ...wrapTextWithAnsi(theme.fg("dim", this.#authUrl), width)); + if (urlLines.length > 2) { + lines.push(...urlLines); } if (this.#statusLines.length > 0) { - lines.push("", ...this.#statusLines.flatMap(line => wrapTextWithAnsi(line, width))); + lines.push(...this.#statusLines.flatMap(line => wrapTextWithAnsi(line, width))); } return lines; } diff --git a/packages/coding-agent/test/setup-wizard-sign-in.test.ts b/packages/coding-agent/test/setup-wizard-sign-in.test.ts index dbc24c783..f4c0c4864 100644 --- a/packages/coding-agent/test/setup-wizard-sign-in.test.ts +++ b/packages/coding-agent/test/setup-wizard-sign-in.test.ts @@ -58,15 +58,16 @@ describe("SignInTab", () => { expect(rendered.join("\n")).toContain(`\x1b]8;;${url}\x07Open login URL\x1b]8;;\x07`); expect(openedUrls).toEqual([url]); - // On a ~24-row terminal the wizard body ends up ~8 rows; both the - // OSC8 link row and the focused input must survive that clip. + // On a ~24-row terminal the wizard body ends up ~8 rows; the OSC8 + // link, a plain URL row, and the focused input must survive that clip. const clippedBody = rendered.slice(0, 8).map(line => Bun.stripANSI(line).trim()); + const plainUrlIndex = clippedBody.findIndex(line => line.startsWith("https://example.com/oauth/authorize?")); + const inputIndex = clippedBody.findIndex(line => line.startsWith(">")); expect(clippedBody.some(line => line === "Browser login: Open login URL")).toBe(true); + expect(plainUrlIndex).toBeGreaterThanOrEqual(0); expect(clippedBody).toContain("Paste the authorization code (or full redirect URL):"); - expect(clippedBody.some(line => line.startsWith(">"))).toBe(true); - expect(clippedBody.indexOf("Browser login: Open login URL")).toBeLessThan( - clippedBody.findIndex(line => line.startsWith(">")), - ); + expect(inputIndex).toBeGreaterThanOrEqual(0); + expect(plainUrlIndex).toBeLessThan(inputIndex); } finally { tab.dispose(); loginGate.resolve(); From 0db52a72dcbbeb9ac4cbf2a2d835b31454d0791d Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 08:17:46 +0000 Subject: [PATCH 31/48] test(coding-agent/hindsight): defended bankId reset routing in both scope modes Added two regression tests covering the bidirectional reporter scenario: clearing hindsight.bankId after a non-empty value rebuilds the live state and routes subsequent retainBatch calls to the recomputed bank under both per-project and global scoping. The PR's existing fix already handles this direction; these lock the contract in place explicitly. Fixes #1902 --- .../test/hindsight-backend.test.ts | 82 +++++++++++++++++++ 1 file changed, 82 insertions(+) diff --git a/packages/coding-agent/test/hindsight-backend.test.ts b/packages/coding-agent/test/hindsight-backend.test.ts index 0ac38be24..e479a39a1 100644 --- a/packages/coding-agent/test/hindsight-backend.test.ts +++ b/packages/coding-agent/test/hindsight-backend.test.ts @@ -644,6 +644,88 @@ describe("hindsightBackend live bank routing", () => { expect(session.getHindsightSessionState()).toBe(initial); }); + // Same regression flipped: resetting `hindsight.bankId` back to blank / + // default after a non-empty value MUST also rebuild and route subsequent + // retains to the default bank. The Codex-flagged follow-up was that the + // fix had to be bidirectional — set→value AND value→reset. We defend the + // reset direction end-to-end by enqueuing a retain after the reset and + // asserting the batch call hits the recomputed bank, not the previous one. + it("rebuilds when hindsight.bankId is reset to blank after a non-empty value", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.scoping", "per-project"); + settings.set("hindsight.bankId", "Minigames"); + const session = makeFakeSession({ sessionId: "s-reset", cwd: "/work/_NEW_XenGameKit", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + const initial = session.getHindsightSessionState(); + expect(initial?.bankId).toBe("Minigames-_NEW_XenGameKit"); + + // Operator clears the bankId via the TUI — `settings.set(path, "")` is + // the same call shape `#setSettingValue` uses for an empty text input. + settings.set("hindsight.bankId", ""); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next).toBeDefined(); + expect(next).not.toBe(initial); + // With scoping=per-project the base falls back to the default ("omp"), + // so the reset bank id picks up the project suffix from cwd. + expect(next?.bankId).toBe("omp-_NEW_XenGameKit"); + + next!.enqueueRetain("post-reset fact", "reset routing"); + await next!.flushRetainQueue(); + + expect(retainBatchSpy).toHaveBeenCalledTimes(1); + expect(retainBatchSpy.mock.calls[0][0]).toBe("omp-_NEW_XenGameKit"); + }); + + // Companion case: when `hindsight.scoping` is `global`, clearing the + // non-empty bankId should restore the bare `omp` default — the operator's + // stated expectation in the live repro from #1902. + it("routes future retains to the bare omp bank when bankId is cleared in global scoping", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.scoping", "global"); + settings.set("hindsight.bankId", "Minigames-_NEW_XenGameKit"); + const session = makeFakeSession({ sessionId: "s-reset-global", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + expect(session.getHindsightSessionState()?.bankId).toBe("Minigames-_NEW_XenGameKit"); + + settings.set("hindsight.bankId", ""); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next?.bankId).toBe("omp"); + + next!.enqueueRetain("post-reset global fact"); + await next!.flushRetainQueue(); + + expect(retainBatchSpy).toHaveBeenCalledTimes(1); + expect(retainBatchSpy.mock.calls[0][0]).toBe("omp"); + }); + it("coalesces synchronous routing hooks so rebuilt states do not leak agent listeners", async () => { const retainSpy = vi.spyOn(HindsightApi.prototype, "retain").mockResolvedValue({} as never); vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); From 0f83efdf78b9dd669265de36e4ea76927c4ec833 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 10:53:01 +0200 Subject: [PATCH 32/48] fix(coding-agent): relativized rule path in TTSR injections - Stopped leaking absolute home directory to the model in ttsr-interrupt and ttsr-tool-reminder blocks. - Rendered rule paths as cwd-relative in-project, `~`-relative under home, else raw. --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/session/agent-session.ts | 31 ++++++++- .../test/agent-session-concurrent.test.ts | 64 +++++++++++++++++++ 3 files changed, 97 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..16905fcec 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed TTSR rule-violation injections leaking the absolute home directory to the model: the `ttsr-interrupt` / `ttsr-tool-reminder` blocks rendered the matched rule's `path` as its absolute on-disk path (e.g. `/Users/me/Projects/app/.omp/rules/no-any.md`). The path is now relativized to the session cwd when the rule lives in the project (`.omp/rules/no-any.md`), or `~`-relative when it lives under home, so no absolute path is fed into the agent's context outside the system prompt. + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index b7760bb5c..dee1e25a9 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -15,6 +15,7 @@ import * as crypto from "node:crypto"; import * as fs from "node:fs"; +import * as os from "node:os"; import * as path from "node:path"; import { scheduler } from "node:timers/promises"; import { isPromise } from "node:util/types"; @@ -94,6 +95,7 @@ import { isUnexpectedSocketCloseMessage, logger, prompt, + relativePathWithinRoot, Snowflake, } from "@oh-my-pi/pi-utils"; import { type AsyncJob, type AsyncJobDeliveryState, AsyncJobManager } from "../async"; @@ -2128,12 +2130,31 @@ export class AgentSession { if (this.#pendingTtsrInjections.length === 0) return undefined; const rules = this.#pendingTtsrInjections; const content = rules - .map(r => prompt.render(ttsrInterruptTemplate, { name: r.name, path: r.path, content: r.content })) + .map(r => + prompt.render(ttsrInterruptTemplate, { + name: r.name, + path: this.#displayRulePath(r.path), + content: r.content, + }), + ) .join("\n\n"); this.#pendingTtsrInjections = []; return { content, rules }; } + /** + * Render a rule's file path for model-facing TTSR injections without leaking + * the absolute home directory: cwd-relative when the rule lives in the + * project, `~`-relative when it lives under home, else the raw path. + */ + #displayRulePath(rulePath: string): string { + const cwdRel = relativePathWithinRoot(this.sessionManager.getCwd(), rulePath); + if (cwdRel) return cwdRel; + const homeRel = relativePathWithinRoot(os.homedir(), rulePath); + if (homeRel) return `~/${homeRel}`; + return rulePath; + } + #addPendingTtsrInjections(rules: Rule[]): void { const seen = new Set(this.#pendingTtsrInjections.map(rule => rule.name)); for (const rule of rules) { @@ -2186,7 +2207,13 @@ export class AgentSession { if (!rules || rules.length === 0) return undefined; this.#perToolTtsrInjections.delete(ctx.toolCall.id); const reminder = rules - .map(r => prompt.render(ttsrToolReminderTemplate, { name: r.name, path: r.path, content: r.content })) + .map(r => + prompt.render(ttsrToolReminderTemplate, { + name: r.name, + path: this.#displayRulePath(r.path), + content: r.content, + }), + ) .join("\n\n"); // The TTSR manager was already claimed at bucket time; only persistence remains. const ruleNames = rules.map(r => r.name.trim()).filter(n => n.length > 0); diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index d7bb707e5..38ad9e7d0 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -732,6 +732,70 @@ describe("AgentSession TTSR resume gate", () => { expect(session.isStreaming).toBe(false); }); + it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; + let streamCallCount = 0; + + const sessionManager = SessionManager.inMemory(); + const cwd = sessionManager.getCwd(); + const ruleAbsPath = path.join(cwd, ".omp", "rules", "no-unwrap.md"); + const expectedRel = path.relative(cwd, ruleAbsPath); + const rule: Rule = { + name: "no-unwrap", + path: ruleAbsPath, + content: "Do not use .unwrap()", + condition: ["\\.unwrap\\("], + _source: { provider: "test", providerName: "test", path: ruleAbsPath, level: "project" }, + }; + + const ttsrManager = new TtsrManager({ + enabled: true, + contextMode: "discard", + interruptMode: "always", + repeatMode: "once", + repeatGap: 10, + }); + ttsrManager.addRule(rule); + + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model, systemPrompt: ["Test"], tools: [] }, + streamFn: (_model, _context, options) => { + streamCallCount++; + const stream = new AssistantMessageEventStream(); + if (streamCallCount === 1) { + pushAbortableTtsrStream(stream, options?.signal); + } else { + pushContinuationStream(stream, () => {}); + } + return stream; + }, + }); + + const settings = Settings.isolated(); + const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-rel.db")); + authStorages.push(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + + session = new AgentSession({ agent, sessionManager, settings, modelRegistry, ttsrManager }); + + await session.prompt("Write some Rust code"); + + const injection = sessionManager + .getEntries() + .find(e => e.type === "custom_message" && e.customType === "ttsr-injection"); + expect(injection?.type).toBe("custom_message"); + const content = injection?.type === "custom_message" ? injection.content : undefined; + expect(typeof content).toBe("string"); + const text = content as string; + // The rendered interrupt the model receives references the rule by a + // project-relative path, never the absolute home path. + expect(text).toContain('reason="rule_violation"'); + expect(text).toContain(`path="${expectedRel}"`); + expect(text).not.toContain(ruleAbsPath); + }); + it("prompt() blocks until TTSR deferred continuation completes", async () => { const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let streamCallCount = 0; From de21a28e438234b7c532ead07053aad7d49adc22 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 08:54:49 +0000 Subject: [PATCH 33/48] fix(task): fallback to sync when AsyncJobManager is unavailable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When `async.enabled` is true but `AsyncJobManager.instance()` returns `undefined` (orphaned-session state, host that never wired one up, etc.), the `task` tool was returning a hard error and was unusable for the rest of the session — even though the existing sync codepath (`#executeSync`, which still parallelizes via `mapWithConcurrencyLimit`) was right there. Fall back to `#executeSync` instead and emit a `logger.warn` so the missing-manager state stays diagnosable. Background/job-poll semantics are lost in this degraded mode, but the tool keeps working. Fixes #1922 --- packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/task/index.ts | 12 ++-- .../test/tools/task-async-fallback.test.ts | 64 +++++++++++++++++++ 3 files changed, 75 insertions(+), 5 deletions(-) create mode 100644 packages/coding-agent/test/tools/task-async-fallback.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..7ed391094 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the `task` tool returning a hard `Async execution is enabled but no async job manager is available.` error when `async.enabled` was true but `AsyncJobManager.instance()` returned `undefined`, leaving `task` non-functional for the rest of the session. The tool now falls back to the existing synchronous execution path (which still runs subagents concurrently via `mapWithConcurrencyLimit`), and logs a warning so the missing-manager state stays diagnosable ([#1922](https://github.com/can1357/oh-my-pi/issues/1922)). + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index 466d966d1..6d9c15932 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -17,7 +17,7 @@ import * as os from "node:os"; import path from "node:path"; import type { AgentTool, AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core"; import type { Usage } from "@oh-my-pi/pi-ai"; -import { $env, prompt, Snowflake } from "@oh-my-pi/pi-utils"; +import { $env, logger, prompt, Snowflake } from "@oh-my-pi/pi-utils"; import type { ToolSession } from ".."; import { AsyncJobManager } from "../async"; import { resolveAgentModelPatterns } from "../config/model-resolver"; @@ -345,10 +345,12 @@ export class TaskTool implements AgentTool> = {}): ToolSession { + return { + cwd: "/tmp", + hasUI: false, + settings: Settings.isolated(overrides), + getSessionFile: () => null, + getSessionSpawns: () => "*", + } as unknown as ToolSession; +} + +function getFirstText(result: { content: Array<{ type: string; text?: string }> }): string { + const content = result.content.find(part => part.type === "text"); + return content?.type === "text" ? (content.text ?? "") : ""; +} + +describe("task.async-fallback", () => { + afterEach(() => { + vi.restoreAllMocks(); + AsyncJobManager.resetForTests(); + }); + + it("falls back to sync execution when async is enabled but no manager is registered", async () => { + // Two-stage spy: the initial discovery during `TaskTool.create` advertises + // `task` so the tool builds; the executor's later call (inside + // `#executeSync`) advertises *nothing*, forcing the unique "Unknown agent" + // message — which is only reachable from the sync codepath. Hitting it + // proves we fell back instead of returning the old hard error. + const discoverSpy = vi.spyOn(discoveryModule, "discoverAgents"); + discoverSpy.mockResolvedValueOnce({ + agents: [ + { + name: "task", + description: "General-purpose task agent", + systemPrompt: "You are a task agent.", + source: "bundled", + }, + ], + projectAgentsDir: null, + }); + discoverSpy.mockResolvedValue({ agents: [], projectAgentsDir: null }); + + AsyncJobManager.resetForTests(); + expect(AsyncJobManager.instance()).toBeUndefined(); + + const tool = await TaskTool.create(createSession({ "async.enabled": true })); + + const result = await tool.execute("tool-1", { + agent: "task", + tasks: [{ id: "One", description: "label", assignment: "Do the thing." }], + } as TaskParams); + + const text = getFirstText(result); + expect(text).toContain('Unknown agent "task"'); + expect(text).not.toContain("no async job manager is available"); + }); +}); From a0ff234500b5b8c028c56240b87f7f509be0073e Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 10:59:38 +0200 Subject: [PATCH 34/48] feat(coding-agent/cli): added dry-balance CLI dry-run check for OAuth account balancing - Added `omp dry-balance` command with model, count, concurrency, and JSON flags. - Implemented random session-id sampling with bounded concurrency for OAuth access dry-run checks. - Added success/failure summary generation with account and reason stats and optional JSON output. - Set CLI exit status to 1 when any dry-balance attempt fails. - Added the new dry-balance capability to the unreleased changelog notes. --- packages/coding-agent/CHANGELOG.md | 6 +- packages/coding-agent/src/cli-commands.ts | 1 + .../coding-agent/src/cli/dry-balance-cli.ts | 331 ++++++++++++++++++ .../coding-agent/src/commands/dry-balance.ts | 40 +++ 4 files changed, 377 insertions(+), 1 deletion(-) create mode 100644 packages/coding-agent/src/cli/dry-balance-cli.ts create mode 100644 packages/coding-agent/src/commands/dry-balance.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 16905fcec..18f2b7ddc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,10 @@ # Changelog ## [Unreleased] +### Added + +- Added `dry-balance` CLI command to perform a dry-run OAuth account balancing check across configurable random session IDs, with sample and concurrency options, JSON output, and success/failure summary reporting +- Added `--json` output mode and machine-readable result format to `omp dry-balance` for automated use ### Fixed @@ -9337,4 +9341,4 @@ Initial public release. - Git branch display in footer - Message queueing during streaming responses - OAuth integration for Gmail and Google Calendar access -- HTML export with syntax highlighting and collapsible sections +- HTML export with syntax highlighting and collapsible sections \ No newline at end of file diff --git a/packages/coding-agent/src/cli-commands.ts b/packages/coding-agent/src/cli-commands.ts index 8fa568001..c9ac00741 100644 --- a/packages/coding-agent/src/cli-commands.ts +++ b/packages/coding-agent/src/cli-commands.ts @@ -20,6 +20,7 @@ export const commands: CommandEntry[] = [ { name: "completions", load: () => import("./commands/completions").then(m => m.default) }, { name: "__complete", load: () => import("./commands/complete").then(m => m.default) }, { name: "config", load: () => import("./commands/config").then(m => m.default) }, + { name: "dry-balance", load: () => import("./commands/dry-balance").then(m => m.default) }, { name: "grep", load: () => import("./commands/grep").then(m => m.default) }, { name: "grievances", load: () => import("./commands/grievances").then(m => m.default) }, { name: "install", load: () => import("./commands/install").then(m => m.default) }, diff --git a/packages/coding-agent/src/cli/dry-balance-cli.ts b/packages/coding-agent/src/cli/dry-balance-cli.ts new file mode 100644 index 000000000..03249c705 --- /dev/null +++ b/packages/coding-agent/src/cli/dry-balance-cli.ts @@ -0,0 +1,331 @@ +import type { Api, Model, OAuthAccess } from "@oh-my-pi/pi-ai"; +import { getProjectDir } from "@oh-my-pi/pi-utils"; +import chalk from "chalk"; +import { ModelRegistry } from "../config/model-registry"; +import { + formatModelString, + resolveAllowedModels, + resolveCliModel, + resolveModelRoleValue, + type ModelMatchPreferences, +} from "../config/model-resolver"; +import { Settings } from "../config/settings"; +import { discoverAuthStorage } from "../sdk"; + +const DEFAULT_SAMPLE_COUNT = 100; +const DEFAULT_CONCURRENCY = 32; + +export interface DryBalanceCommandArgs { + model?: string; + flags: { + model?: string; + count?: number; + concurrency?: number; + json?: boolean; + }; +} + +export interface DryBalanceAuthOptions { + baseUrl?: string; + modelId?: string; + signal?: AbortSignal; +} + +export interface DryBalanceAuthStorage { + getOAuthAccess(provider: string, sessionId?: string, options?: DryBalanceAuthOptions): Promise; +} + +export interface DryBalanceModelRegistry { + authStorage: DryBalanceAuthStorage; + getAll(): Model[]; + getAvailable(): Model[]; + getApiKey(model: Model, sessionId?: string): Promise; + getCanonicalVariants(model: Model): Model[]; + resolveCanonicalModel?(model: Model): Model | undefined; + getCanonicalId?(model: Model): string | undefined; +} + +export interface DryBalanceRuntime { + modelRegistry: DryBalanceModelRegistry; + settings?: Settings; + close?: () => void; +} + +export interface DryBalanceAccountStat { + account: string; + count: number; + percent: number; +} + +export interface DryBalanceFailureStat { + reason: string; + count: number; + percent: number; +} + +export interface DryBalanceSummary { + model: string; + provider: string; + samples: number; + concurrency: number; + success: { + total: number; + accounts: DryBalanceAccountStat[]; + }; + failure: { + total: number; + reasons: DryBalanceFailureStat[]; + }; +} + +interface DryBalanceDependencies { + createRuntime?: () => Promise; + randomSessionId?: () => string; + writeStdout?: (text: string) => void; + writeStderr?: (text: string) => void; + setExitCode?: (code: number) => void; +} + +type DryBalanceAttemptResult = + | { + ok: true; + account: string; + } + | { + ok: false; + reason: string; + }; + +function normalizePositiveInteger(name: string, value: number | undefined, fallback: number): number { + const resolved = value ?? fallback; + if (!Number.isInteger(resolved) || resolved <= 0) { + throw new Error(`--${name} must be a positive integer`); + } + return resolved; +} + +async function createDefaultRuntime(): Promise { + const authStorage = await discoverAuthStorage(); + try { + const settings = await Settings.init({ cwd: getProjectDir() }); + const modelRegistry = new ModelRegistry(authStorage); + return { + modelRegistry, + settings, + close: () => authStorage.close(), + }; + } catch (error) { + authStorage.close(); + throw error; + } +} + +async function resolveDryBalanceModel( + modelSelector: string | undefined, + modelRegistry: DryBalanceModelRegistry, + settings: Settings | undefined, + randomSessionId: () => string, +): Promise<{ model: Model; warning?: string }> { + const preferences: ModelMatchPreferences = { + usageOrder: settings?.getStorage()?.getModelUsageOrder(), + }; + if (modelSelector) { + const resolved = resolveCliModel({ + cliModel: modelSelector, + modelRegistry, + preferences, + }); + if (resolved.error) throw new Error(resolved.error); + if (!resolved.model) throw new Error(`Model "${modelSelector}" not found`); + return { model: resolved.model, warning: resolved.warning }; + } + + const allowedModels = await resolveAllowedModels(modelRegistry, settings, preferences); + if (allowedModels.length === 0) { + throw new Error("No models available. Use --model to select a model or configure enabledModels/default model settings."); + } + + const defaultRoleSpec = resolveModelRoleValue(settings?.getModelRole("default"), allowedModels, { + settings, + matchPreferences: preferences, + modelRegistry, + }); + if (defaultRoleSpec.model) { + return { model: defaultRoleSpec.model, warning: defaultRoleSpec.warning }; + } + + for (const candidate of allowedModels) { + const apiKey = await modelRegistry.getApiKey(candidate, randomSessionId()); + if (apiKey) return { model: candidate }; + } + + return { + model: allowedModels[0], + warning: "No allowed model had usable credentials during default resolution; dry-balance will report OAuth failures for the first allowed model.", + }; +} + + + +async function runOneAttempt( + model: Model, + modelRegistry: DryBalanceModelRegistry, + sessionId: string, +): Promise { + try { + // AuthStorage.getOAuthAccess shares the OAuth credential ranking, refresh, + // usage-limit, broker, and session-sticky path used by getApiKey(), while + // returning the selected account metadata instead of bearer bytes. + const access = await modelRegistry.authStorage.getOAuthAccess(model.provider, sessionId, { + baseUrl: model.baseUrl, + modelId: model.id, + }); + if (!access) return { ok: false, reason: "no OAuth access resolved" }; + const account = access.email ?? access.accountId ?? access.projectId ?? access.enterpriseUrl ?? "(unknown oauth account)"; + return { ok: true, account }; + } catch (error) { + return { ok: false, reason: error instanceof Error ? error.message : String(error) }; + } +} + +async function mapConcurrent(items: T[], concurrency: number, fn: (item: T) => Promise): Promise { + const results = new Array(items.length); + let nextIndex = 0; + const workerCount = Math.min(concurrency, items.length); + await Promise.all( + Array.from({ length: workerCount }, async () => { + while (true) { + const index = nextIndex; + nextIndex += 1; + if (index >= items.length) return; + results[index] = await fn(items[index]); + } + }), + ); + return results; +} + + +function sortedStats(map: Map, samples: number): Array<{ label: string; count: number; percent: number }> { + return [...map.entries()] + .map(([label, count]) => ({ label, count, percent: (count / samples) * 100 })) + .sort((left, right) => right.count - left.count || left.label.localeCompare(right.label)); +} + +function summarizeResults( + model: Model, + samples: number, + concurrency: number, + results: DryBalanceAttemptResult[], +): DryBalanceSummary { + const accounts = new Map(); + const reasons = new Map(); + for (const result of results) { + if (result.ok) { + accounts.set(result.account, (accounts.get(result.account) ?? 0) + 1); + } else { + reasons.set(result.reason, (reasons.get(result.reason) ?? 0) + 1); + } + } + const accountStats: DryBalanceAccountStat[] = sortedStats(accounts, samples).map(stat => ({ + account: stat.label, + count: stat.count, + percent: stat.percent, + })); + const failureStats: DryBalanceFailureStat[] = sortedStats(reasons, samples).map(stat => ({ + reason: stat.label, + count: stat.count, + percent: stat.percent, + })); + return { + model: formatModelString(model), + provider: model.provider, + samples, + concurrency, + success: { + total: results.filter(result => result.ok).length, + accounts: accountStats, + }, + failure: { + total: results.filter(result => !result.ok).length, + reasons: failureStats, + }, + }; +} + + +function formatRows(rows: Array<{ count: number; percent: number; label: string }>): string[] { + if (rows.length === 0) return [` ${chalk.dim("(none)")}`]; + const maxCountWidth = Math.max(...rows.map(row => row.count.toString().length)); + return rows.map(row => { + const count = row.count.toString().padStart(maxCountWidth); + const percent = `${row.percent.toFixed(1)}%`.padStart(6); + return ` ${count} ${percent} ${row.label}`; + }); +} + +export function formatDryBalanceText(summary: DryBalanceSummary): string { + const accountRows = summary.success.accounts.map(row => ({ + count: row.count, + percent: row.percent, + label: row.account, + })); + const failureRows = summary.failure.reasons.map(row => ({ + count: row.count, + percent: row.percent, + label: row.reason, + })); + const lines = [ + chalk.bold("dry-balance"), + `model: ${summary.model}`, + `provider: ${summary.provider}`, + `samples: ${summary.samples}`, + `concurrency: ${summary.concurrency}`, + "", + `${chalk.green("success")} ${summary.success.total}`, + ...formatRows(accountRows), + "", + `${summary.failure.total > 0 ? chalk.red("failure") : chalk.dim("failure")} ${summary.failure.total}`, + ...formatRows(failureRows), + ]; + return `${lines.join("\n")}\n`; +} + +export async function runDryBalanceCommand( + command: DryBalanceCommandArgs, + deps: DryBalanceDependencies = {}, +): Promise { + const samples = normalizePositiveInteger("count", command.flags.count, DEFAULT_SAMPLE_COUNT); + const concurrency = Math.min(samples, normalizePositiveInteger("concurrency", command.flags.concurrency, DEFAULT_CONCURRENCY)); + const randomSessionId = deps.randomSessionId ?? (() => Bun.randomUUIDv7()); + const writeStdout = deps.writeStdout ?? ((text: string) => process.stdout.write(text)); + const writeStderr = deps.writeStderr ?? ((text: string) => process.stderr.write(text)); + const setExitCode = deps.setExitCode ?? ((code: number) => { + process.exitCode = code; + }); + const runtime = await (deps.createRuntime ?? createDefaultRuntime)(); + try { + const modelSelector = command.flags.model ?? command.model; + const { model, warning } = await resolveDryBalanceModel( + modelSelector, + runtime.modelRegistry, + runtime.settings, + randomSessionId, + ); + if (warning) writeStderr(`${chalk.yellow(`Warning: ${warning}`)}\n`); + const sessionIds = Array.from({ length: samples }, () => randomSessionId()); + const results = await mapConcurrent(sessionIds, concurrency, sessionId => + runOneAttempt(model, runtime.modelRegistry, sessionId), + ); + const summary = summarizeResults(model, samples, concurrency, results); + if (command.flags.json) { + writeStdout(`${JSON.stringify(summary, null, 2)}\n`); + } else { + writeStdout(formatDryBalanceText(summary)); + } + if (summary.failure.total > 0) setExitCode(1); + return summary; + } finally { + runtime.close?.(); + } +} diff --git a/packages/coding-agent/src/commands/dry-balance.ts b/packages/coding-agent/src/commands/dry-balance.ts new file mode 100644 index 000000000..2e2c2631e --- /dev/null +++ b/packages/coding-agent/src/commands/dry-balance.ts @@ -0,0 +1,40 @@ +import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli"; +import { runDryBalanceCommand } from "../cli/dry-balance-cli"; + +export default class DryBalance extends Command { + static description = "Dry-run OAuth account balancing across random session ids"; + + static args = { + model: Args.string({ + description: "Model selector (provider/model or fuzzy id). Defaults to the configured default model.", + required: false, + }), + }; + + static flags = { + model: Flags.string({ description: "Model selector (same syntax as --model on omp)" }), + count: Flags.integer({ description: "Number of random session ids to try", default: 100 }), + concurrency: Flags.integer({ description: "Maximum concurrent credential resolutions", default: 32 }), + json: Flags.boolean({ description: "Output JSON" }), + }; + + static examples = [ + "# Dry-run the configured default model with 100 random session ids\n omp dry-balance", + "# Dry-run a specific model\n omp dry-balance anthropic/claude-sonnet-4-5", + "# Larger run with bounded concurrency\n omp dry-balance --model openai-codex/gpt-5-codex --count 1000 --concurrency 64", + "# Machine-readable output\n omp dry-balance --json", + ]; + + async run(): Promise { + const { args, flags } = await this.parse(DryBalance); + await runDryBalanceCommand({ + model: args.model, + flags: { + model: flags.model, + count: flags.count, + concurrency: flags.concurrency, + json: flags.json, + }, + }); + } +} From 36db2530a98fa2b78b5c85fde49671bccea067b5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 09:06:53 +0000 Subject: [PATCH 35/48] fix(sdk): keep primary AsyncJobManager when secondary top-level session disposes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Any in-process secondary createAgentSession() (e.g. the Agent Control Center's create flow in agent-dashboard.ts) was constructing its own AsyncJobManager, overwriting the process-global singleton, and then clearing it on its own dispose. The primary session still held its #ownedAsyncJobManager reference, but AsyncJobManager.instance() was undefined for the rest of the process — the task async path hard-failed with "Async execution is enabled but no async job manager is available" and only a full restart cleared it. - sdk.ts: skip constructing/installing a second AsyncJobManager when a singleton is already live, so secondary top-level sessions share the owning session's manager instead of clobbering it.\n- agent-session.ts: scope #cancelOwnAsyncJobs so a secondary session inheriting the singleton with the default MAIN_AGENT_ID can no longer cancel the primary session's running bash/task jobs at dispose time. Subagents still reach the inherited singleton via their unique agent ids; the owning session still cancels its own jobs through #ownedAsyncJobManager. Fixes #1923 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/sdk.ts | 14 ++- .../coding-agent/src/session/agent-session.ts | 15 ++- .../sdk-async-job-manager-singleton.test.ts | 105 ++++++++++++++++++ 4 files changed, 131 insertions(+), 7 deletions(-) create mode 100644 packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8846e99e6..d4af64036 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. The secondary now shares the live singleton instead of clobbering it, and its `cancelOwnAsyncJobs` dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 0e02b1f9d..cf969e940 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1168,12 +1168,16 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} return preview; }; - // Only top-level sessions own an AsyncJobManager. Subagents reach the - // parent's manager via `AsyncJobManager.instance()` (set below), so creating - // a second instance here just to leave it orphaned wastes a constructor and - // risks accidental disposal of the parent's manager on subagent teardown. + // Only the first top-level session in a process owns an AsyncJobManager. + // Subagents inherit the parent's manager via `AsyncJobManager.instance()` + // (set below), and any additional top-level session spun up in-process + // (e.g. the agent-creation architect in `agent-dashboard.ts`) must share + // the live singleton — otherwise its dispose path would clobber the + // owning session's manager and break the `task`/`bash` async paths + // (issue #1923). The `instance()` guard means later sessions also skip + // constructing an orphaned manager that nothing would ever route to. const asyncJobManager = - backgroundJobsEnabled && !options.parentTaskPrefix + backgroundJobsEnabled && !options.parentTaskPrefix && !AsyncJobManager.instance() ? new AsyncJobManager({ maxRunningJobs: asyncMaxJobs, onJobComplete: async (jobId, result, job) => { diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index b7760bb5c..d2c19fa93 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1398,11 +1398,22 @@ export class AgentSession { * Cancel async jobs registered by *this* agent only. Used by lifecycle * transitions (newSession, switchSession, handoff, dispose) so a subagent * cleans up its own background work without touching its parent's jobs. - * No-op when no manager is installed or this session has no agent id. + * + * Cancellation runs against the manager THIS session owns. Subagents have + * unique agent ids and may still reach the inherited singleton (which is + * the parent's manager) to clean up their own scoped jobs. A secondary + * in-process top-level session — which inherits the singleton without + * owning it AND defaults to `MAIN_AGENT_ID` — must NOT cancel via the + * inherited singleton, or it would tear down the owning primary session's + * bash/task jobs at dispose time (issue #1923). + * + * No-op when no manager is reachable or this session has no agent id. */ #cancelOwnAsyncJobs(): void { if (!this.#agentId) return; - AsyncJobManager.instance()?.cancelAll({ ownerId: this.#agentId }); + const manager = + this.#ownedAsyncJobManager ?? (this.#agentId === MAIN_AGENT_ID ? undefined : AsyncJobManager.instance()); + manager?.cancelAll({ ownerId: this.#agentId }); } // ========================================================================= diff --git a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts new file mode 100644 index 000000000..4c0f14b5b --- /dev/null +++ b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts @@ -0,0 +1,105 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async/job-manager"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; +import { Snowflake } from "@oh-my-pi/pi-utils"; + +describe("AsyncJobManager singleton across concurrent top-level sessions", () => { + const tempDirs: string[] = []; + + afterEach(async () => { + for (const tempDir of tempDirs.splice(0)) { + fs.rmSync(tempDir, { recursive: true, force: true }); + } + AsyncJobManager.resetForTests(); + }); + + async function spawnTopLevelSession() { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-sdk-async-singleton-${Snowflake.next()}-`)); + tempDirs.push(tempDir); + const cwd = path.join(tempDir, `project-${Snowflake.next()}`); + const agentDir = path.join(tempDir, "agent"); + fs.mkdirSync(cwd, { recursive: true }); + const { session } = await createAgentSession({ + cwd, + agentDir, + settings: Settings.isolated({ "bash.autoBackground.enabled": true }), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + }); + return session; + } + + it("keeps the primary session's manager installed after a secondary session disposes", async () => { + const primary = await spawnTopLevelSession(); + try { + const primaryManager = AsyncJobManager.instance(); + expect(primaryManager).toBeDefined(); + + const secondary = await spawnTopLevelSession(); + try { + // While the secondary is alive the global instance MUST still point at + // the primary's manager so background tools keep delivering completions + // to the primary session that owns them. + expect(AsyncJobManager.instance()).toBe(primaryManager); + } finally { + await secondary.dispose(); + } + + // After the secondary disposes, the primary's manager MUST still be the + // reachable singleton — otherwise the `task` async path errors with + // "Async execution is enabled but no async job manager is available". + expect(AsyncJobManager.instance()).toBe(primaryManager); + } finally { + await primary.dispose(); + } + + // Once the owning primary session disposes the singleton clears, matching + // the documented single-owner invariant. + expect(AsyncJobManager.instance()).toBeUndefined(); + }); + + it("does not cancel the primary session's running jobs when a secondary session disposes", async () => { + const primary = await spawnTopLevelSession(); + try { + const primaryManager = AsyncJobManager.instance(); + expect(primaryManager).toBeDefined(); + + // Register a long-running job on the primary's manager under the + // MAIN_AGENT_ID owner — the same owner the secondary would inherit by + // default. The secondary's dispose-time `cancelOwnAsyncJobs` must NOT + // cancel this job (issue #1923). + const release = Promise.withResolvers(); + const jobId = primaryManager!.register( + "bash", + "sleep", + async ({ signal }) => { + const aborted = Promise.withResolvers(); + signal.addEventListener("abort", () => aborted.resolve(), { once: true }); + await Promise.race([release.promise, aborted.promise]); + return signal.aborted ? "aborted" : "completed"; + }, + { ownerId: "Main" }, + ); + + const secondary = await spawnTopLevelSession(); + await secondary.dispose(); + + const job = primaryManager!.getJob(jobId); + expect(job?.status).toBe("running"); + + release.resolve("done"); + await primaryManager!.waitForAll(); + } finally { + await primary.dispose(); + } + }); +}); From 5004ba057fd8e16a2547f29c02a2d68a59ce8646 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:08:23 +0200 Subject: [PATCH 36/48] feat(auth-broker): added encrypted local snapshot cache - Added AES-GCM cache for at-rest broker snapshots keyed on token, with URL as additional data. - Added `onSnapshot` hook to RemoteAuthCredentialStore for persisting applied snapshots. - Exposed cache read/write and TTL defaults through the coding-agent re-exports. - Added `getAuthBrokerSnapshotCachePath` with `OMP_AUTH_BROKER_SNAPSHOT_CACHE` override. --- packages/ai/src/auth-broker/index.ts | 1 + packages/ai/src/auth-broker/remote-store.ts | 14 ++ packages/ai/src/auth-broker/snapshot-cache.ts | 174 +++++++++++++++++ packages/ai/src/auth-broker/types.ts | 3 + packages/ai/src/auth-storage.ts | 19 ++ .../ai/test/auth-broker-remote-store.test.ts | 24 +++ .../test/auth-broker-snapshot-cache.test.ts | 184 ++++++++++++++++++ packages/coding-agent/src/sdk.ts | 59 +++++- .../coding-agent/src/session/auth-storage.ts | 4 + packages/utils/src/dirs.ts | 11 ++ 10 files changed, 489 insertions(+), 4 deletions(-) create mode 100644 packages/ai/src/auth-broker/snapshot-cache.ts create mode 100644 packages/ai/test/auth-broker-snapshot-cache.test.ts diff --git a/packages/ai/src/auth-broker/index.ts b/packages/ai/src/auth-broker/index.ts index 4858fbfdf..a2f31df85 100644 --- a/packages/ai/src/auth-broker/index.ts +++ b/packages/ai/src/auth-broker/index.ts @@ -1,5 +1,6 @@ export * from "./client"; export * from "./refresher"; export * from "./remote-store"; +export * from "./snapshot-cache"; export * from "./server"; export * from "./types"; diff --git a/packages/ai/src/auth-broker/remote-store.ts b/packages/ai/src/auth-broker/remote-store.ts index 27388cf66..c0f9f17a3 100644 --- a/packages/ai/src/auth-broker/remote-store.ts +++ b/packages/ai/src/auth-broker/remote-store.ts @@ -73,11 +73,17 @@ export interface RemoteAuthCredentialStoreOptions { * to long-poll permanently when the broker returns 404. Default `true`. */ streamSnapshots?: boolean; + /** + * Called after broker-sourced full snapshots are applied. The constructor's + * initial snapshot intentionally does not trigger this hook. + */ + onSnapshot?: (snapshot: SnapshotResponse, generation: number) => void; } export class RemoteAuthCredentialStore implements AuthCredentialStore { readonly #client: AuthBrokerClient; readonly #streamSnapshots: boolean; + readonly #onSnapshot?: (snapshot: SnapshotResponse, generation: number) => void; #snapshot: SnapshotResponse = emptySnapshot(); #snapshotReceivedAt = Date.now(); #generation = 0; @@ -100,6 +106,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { this.#client = opts.client; this.#streamSnapshots = opts.streamSnapshots ?? true; this.#applySnapshot(opts.initialSnapshot ?? emptySnapshot(), opts.initialSnapshot?.generation ?? 0); + this.#onSnapshot = opts.onSnapshot; void this.#runBackground(); } @@ -115,6 +122,13 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { this.#snapshot = snapshot; this.#generation = generation; this.#snapshotReceivedAt = Date.now(); + const onSnapshot = this.#onSnapshot; + if (!onSnapshot) return; + try { + onSnapshot(snapshot, generation); + } catch (error) { + logger.debug("auth-broker snapshot callback failed", { error: String(error) }); + } } async #runBackground(): Promise { diff --git a/packages/ai/src/auth-broker/snapshot-cache.ts b/packages/ai/src/auth-broker/snapshot-cache.ts new file mode 100644 index 000000000..db806e185 --- /dev/null +++ b/packages/ai/src/auth-broker/snapshot-cache.ts @@ -0,0 +1,174 @@ +/** + * AES-GCM encrypted local cache for auth-broker snapshots. + * + * The cache is defense-in-depth for at-rest snapshots: a copied cache file is + * useless without the matching broker bearer token and URL. The token itself is + * still the trust boundary; a process that can read both the token and this file + * can decrypt the snapshot. + */ +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { isEnoent, logger } from "@oh-my-pi/pi-utils"; +import type { SnapshotResponse } from "./types"; +import { snapshotResponseSchema } from "./wire-schemas"; + +const MAGIC = new Uint8Array([0x4f, 0x4d, 0x50, 0x53]); // "OMPS" +const VERSION = 1; +const VERSION_OFFSET = MAGIC.byteLength; +const IV_OFFSET = VERSION_OFFSET + 1; +const IV_LENGTH = 12; +const HEADER_LENGTH = IV_OFFSET + IV_LENGTH; +const AES_ALGORITHM = "AES-GCM"; +const TEXT_ENCODER = new TextEncoder(); +const TEXT_DECODER = new TextDecoder(); +const HEX = "0123456789abcdef"; + +export interface ReadAuthBrokerSnapshotCacheOptions { + path: string; + token: string; + url: string; + ttlMs: number; + /** Override clock for deterministic tests. */ + now?: () => number; +} + +export interface WriteAuthBrokerSnapshotCacheOptions { + path: string; + token: string; + url: string; + snapshot: SnapshotResponse; +} + +export async function readAuthBrokerSnapshotCache( + opts: ReadAuthBrokerSnapshotCacheOptions, +): Promise { + if (opts.ttlMs <= 0) return null; + let data: Uint8Array; + try { + data = await fs.readFile(opts.path); + } catch (error) { + if (isEnoent(error)) return null; + throw error; + } + + try { + const plaintext = await decryptCachePayload(data, opts.token, opts.url); + if (!plaintext) return null; + const parsed: unknown = JSON.parse(TEXT_DECODER.decode(plaintext)); + const result = snapshotResponseSchema.safeParse(parsed); + if (!result.success) { + logger.debug("auth-broker snapshot cache schema invalid", { path: opts.path }); + return null; + } + const snapshot = result.data; + const now = opts.now?.() ?? Date.now(); + if (now - snapshot.generatedAt > opts.ttlMs) return null; + return snapshot; + } catch (error) { + logger.debug("auth-broker snapshot cache read failed", { path: opts.path, error: String(error) }); + return null; + } +} + +export async function writeAuthBrokerSnapshotCache(opts: WriteAuthBrokerSnapshotCacheOptions): Promise { + const payload = await encryptCachePayload(opts.snapshot, opts.token, opts.url); + await fs.mkdir(path.dirname(opts.path), { recursive: true }); + const tmpPath = `${opts.path}.${process.pid}.${randomHex(8)}.tmp`; + let removeTemp = false; + try { + const handle = await fs.open(tmpPath, "wx", 0o600); + removeTemp = true; + try { + await handle.writeFile(payload); + } finally { + await handle.close(); + } + await fs.chmod(tmpPath, 0o600); + await fs.rename(tmpPath, opts.path); + removeTemp = false; + } finally { + if (removeTemp) await fs.rm(tmpPath, { force: true }).catch(() => {}); + } +} + +async function encryptCachePayload(snapshot: SnapshotResponse, token: string, url: string): Promise { + const key = await deriveAesKey(token, ["encrypt"]); + const iv = new Uint8Array(IV_LENGTH); + globalThis.crypto.getRandomValues(iv); + const plaintext = TEXT_ENCODER.encode(JSON.stringify(snapshot)); + const ciphertext = new Uint8Array( + await globalThis.crypto.subtle.encrypt( + { + name: AES_ALGORITHM, + iv, + additionalData: TEXT_ENCODER.encode(url), + }, + key, + plaintext, + ), + ); + const payload = new Uint8Array(HEADER_LENGTH + ciphertext.byteLength); + payload.set(MAGIC, 0); + payload[VERSION_OFFSET] = VERSION; + payload.set(iv, IV_OFFSET); + payload.set(ciphertext, HEADER_LENGTH); + return payload; +} + +async function decryptCachePayload(data: Uint8Array, token: string, url: string): Promise { + if (data.byteLength <= HEADER_LENGTH) { + logger.debug("auth-broker snapshot cache file too short"); + return null; + } + for (let i = 0; i < MAGIC.byteLength; i++) { + if (data[i] !== MAGIC[i]) { + logger.debug("auth-broker snapshot cache magic mismatch"); + return null; + } + } + if (data[VERSION_OFFSET] !== VERSION) { + logger.debug("auth-broker snapshot cache version mismatch", { version: data[VERSION_OFFSET] }); + return null; + } + const key = await deriveAesKey(token, ["decrypt"]); + const iv = asStrict(data.subarray(IV_OFFSET, HEADER_LENGTH)); + const ciphertext = asStrict(data.subarray(HEADER_LENGTH)); + try { + return new Uint8Array( + await globalThis.crypto.subtle.decrypt( + { + name: AES_ALGORITHM, + iv, + additionalData: TEXT_ENCODER.encode(url), + }, + key, + ciphertext, + ), + ); + } catch (error) { + logger.debug("auth-broker snapshot cache decrypt failed", { error: String(error) }); + return null; + } +} + +async function deriveAesKey(token: string, usages: Array<"encrypt" | "decrypt">): Promise { + const digest = await globalThis.crypto.subtle.digest("SHA-256", TEXT_ENCODER.encode(token)); + return globalThis.crypto.subtle.importKey("raw", digest, AES_ALGORITHM, false, usages); +} + +function asStrict(bytes: Uint8Array): Uint8Array { + if (bytes.buffer instanceof ArrayBuffer && bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength) { + return bytes as Uint8Array; + } + const copy = new Uint8Array(bytes.byteLength); + copy.set(bytes); + return copy; +} + +function randomHex(byteLength: number): string { + const bytes = new Uint8Array(byteLength); + globalThis.crypto.getRandomValues(bytes); + let out = ""; + for (const byte of bytes) out += HEX[byte >> 4] + HEX[byte & 15]; + return out; +} diff --git a/packages/ai/src/auth-broker/types.ts b/packages/ai/src/auth-broker/types.ts index 3cfd1bbc8..1387b6473 100644 --- a/packages/ai/src/auth-broker/types.ts +++ b/packages/ai/src/auth-broker/types.ts @@ -117,6 +117,9 @@ export const DEFAULT_REFRESH_SKEW_MS = 5 * 60_000; /** Default broker refresh-loop cadence. */ export const DEFAULT_REFRESH_INTERVAL_MS = 60_000; +/** Default freshness window for the encrypted local broker snapshot cache. */ +export const DEFAULT_SNAPSHOT_CACHE_TTL_MS = 60 * 60_000; + /** Keepalive cadence for `GET /v1/snapshot/stream` SSE comments. */ export const DEFAULT_STREAM_KEEPALIVE_MS = 20_000; diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 907c7b7f0..39e4c14ef 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -725,6 +725,25 @@ class AuthStorageUsageCache implements UsageCache { // ───────────────────────────────────────────────────────────────────────────── type StoredCredential = { id: number; credential: AuthCredential }; +type OAuthSelection = { credential: OAuthCredential; index: number }; + +type OAuthCandidate = { + selection: OAuthSelection; + usage: UsageReport | null; + usageChecked: boolean; +}; + +type RankedOAuthCandidate = OAuthCandidate & { + blocked: boolean; + blockedUntil?: number; + hasPriorityBoost: boolean; + planPriority: number; + secondaryUsed: number; + secondaryDrainRate: number; + primaryUsed: number; + primaryDrainRate: number; + orderPos: number; +}; // ───────────────────────────────────────────────────────────────────────────── // AuthStorage Class diff --git a/packages/ai/test/auth-broker-remote-store.test.ts b/packages/ai/test/auth-broker-remote-store.test.ts index 206d0fc71..ee288a0bb 100644 --- a/packages/ai/test/auth-broker-remote-store.test.ts +++ b/packages/ai/test/auth-broker-remote-store.test.ts @@ -8,6 +8,7 @@ import { AuthStorage, REMOTE_REFRESH_SENTINEL, RemoteAuthCredentialStore, + type SnapshotResponse, SqliteAuthCredentialStore, startAuthBroker, } from "../src"; @@ -109,4 +110,27 @@ describe("RemoteAuthCredentialStore SSE integration", () => { await waitUntil(() => remote!.snapshot.credentials.length === 1); expect(remote!.snapshot.credentials[0].id).not.toBe(bId); }); + + test("calls onSnapshot for broker snapshots but not the constructor snapshot", async () => { + const client = new AuthBrokerClient({ url: handle!.url, token }); + const initialResult = await client.fetchSnapshot(); + if (initialResult.status !== 200) throw new Error("expected initial snapshot"); + const callbacks: Array<{ snapshot: SnapshotResponse; generation: number }> = []; + remote = new RemoteAuthCredentialStore({ + client, + initialSnapshot: initialResult.snapshot, + streamSnapshots: false, + onSnapshot: (snapshot, generation) => { + callbacks.push({ snapshot, generation }); + }, + }); + expect(callbacks).toHaveLength(0); + + storage!.upsertCredential("anthropic", mintOAuthCredential("callback", Date.now() + 120_000)); + const refreshed = await remote.refreshSnapshot(); + + expect(callbacks).toHaveLength(1); + expect(callbacks[0].generation).toBe(refreshed.generation); + expect(callbacks[0].snapshot).toEqual(refreshed); + }); }); diff --git a/packages/ai/test/auth-broker-snapshot-cache.test.ts b/packages/ai/test/auth-broker-snapshot-cache.test.ts new file mode 100644 index 000000000..58aa20b72 --- /dev/null +++ b/packages/ai/test/auth-broker-snapshot-cache.test.ts @@ -0,0 +1,184 @@ +import { describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { + readAuthBrokerSnapshotCache, + type SnapshotResponse, + writeAuthBrokerSnapshotCache, +} from "../src"; + +const TOKEN = "broker-cache-token"; +const URL = "http://127.0.0.1:8765"; + +function makeSnapshot(generatedAt: number): SnapshotResponse { + return { + generation: 7, + generatedAt, + serverNowMs: generatedAt, + refresher: { + enabled: true, + intervalMs: 60_000, + skewMs: 300_000, + nextSweepInMs: 10_000, + }, + credentials: [ + { + id: 1, + provider: "anthropic", + credential: { type: "api_key", key: "secret-api-key" }, + identityKey: null, + rotatesInMs: null, + }, + ], + }; +} + +async function withCachePath(run: (cachePath: string) => Promise): Promise { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "auth-broker-snapshot-cache-")); + try { + await run(path.join(tempDir, "snapshot.enc")); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } +} + +describe("auth-broker snapshot cache", () => { + test("round-trips an encrypted snapshot and writes mode 0600", async () => { + await withCachePath(async cachePath => { + const snapshot = makeSnapshot(1_000_000); + await writeAuthBrokerSnapshotCache({ path: cachePath, token: TOKEN, url: URL, snapshot }); + + const stat = await fs.stat(cachePath); + expect(stat.mode & 0o777).toBe(0o600); + const payload = await fs.readFile(cachePath); + expect(new TextDecoder().decode(payload)).not.toContain("secret-api-key"); + + const decoded = await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }); + expect(decoded).toEqual(snapshot); + }); + }); + + test("returns null when token, url binding, or ciphertext integrity do not match", async () => { + await withCachePath(async cachePath => { + const snapshot = makeSnapshot(1_000_000); + await writeAuthBrokerSnapshotCache({ path: cachePath, token: TOKEN, url: URL, snapshot }); + + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: "wrong-token", + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: "http://127.0.0.1:9999", + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + const tampered = await fs.readFile(cachePath); + tampered[tampered.byteLength - 1] ^= 0xff; + await fs.writeFile(cachePath, tampered); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + }); + }); + + test("enforces generatedAt-based TTL", async () => { + await withCachePath(async cachePath => { + const snapshot = makeSnapshot(10_000); + await writeAuthBrokerSnapshotCache({ path: cachePath, token: TOKEN, url: URL, snapshot }); + + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 100, + now: () => 10_100, + }), + ).toEqual(snapshot); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 100, + now: () => 10_101, + }), + ).toBeNull(); + }); + }); + + test("returns null for missing, short, unencrypted, and schema-invalid files", async () => { + await withCachePath(async cachePath => { + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + await fs.writeFile(cachePath, new Uint8Array([0x4f, 0x4d])); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + await fs.writeFile(cachePath, JSON.stringify(makeSnapshot(1_000_000))); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + await writeAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + snapshot: { generation: 1 } as unknown as SnapshotResponse, + }); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + }); + }); +}); diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 0e02b1f9d..9ae14fd17 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -27,6 +27,7 @@ import { extractRetryHint, getAgentDbPath, getAgentDir, + getAuthBrokerSnapshotCachePath, getProjectDir, logger, postmortem, @@ -101,7 +102,15 @@ import { } from "./secrets"; import { AgentSession } from "./session/agent-session"; import { resolveAuthBrokerConfig } from "./session/auth-broker-config"; -import { AuthBrokerClient, AuthStorage, RemoteAuthCredentialStore } from "./session/auth-storage"; +import { + AuthBrokerClient, + AuthStorage, + DEFAULT_SNAPSHOT_CACHE_TTL_MS, + readAuthBrokerSnapshotCache, + RemoteAuthCredentialStore, + type SnapshotResponse, + writeAuthBrokerSnapshotCache, +} from "./session/auth-storage"; import { type CustomMessage, convertToLlm, wrapSteeringForModel } from "./session/messages"; import { getRestorableSessionModels, SessionManager } from "./session/session-manager"; import { closeAllConnections } from "./ssh/connection-manager"; @@ -418,6 +427,15 @@ function getDefaultAgentDir(): string { return getAgentDir(); } +function resolveSnapshotTtlMs(): number { + const raw = process.env.OMP_AUTH_BROKER_SNAPSHOT_TTL_MS; + if (!raw) return DEFAULT_SNAPSHOT_CACHE_TTL_MS; + const ttlMs = Number(raw); + if (Number.isFinite(ttlMs) && ttlMs >= 0) return ttlMs; + logger.warn("Invalid OMP_AUTH_BROKER_SNAPSHOT_TTL_MS; using default", { value: raw }); + return DEFAULT_SNAPSHOT_CACHE_TTL_MS; +} + // Discovery Functions /** @@ -435,9 +453,42 @@ export async function discoverAuthStorage(agentDir: string = getDefaultAgentDir( const brokerConfig = await resolveAuthBrokerConfig(); if (brokerConfig) { const client = new AuthBrokerClient({ url: brokerConfig.url, token: brokerConfig.token }); - const initialResult = await client.fetchSnapshot(); - if (initialResult.status !== 200) throw new Error("Auth broker returned no initial snapshot"); - const store = new RemoteAuthCredentialStore({ client, initialSnapshot: initialResult.snapshot }); + const ttlMs = resolveSnapshotTtlMs(); + const cachePath = getAuthBrokerSnapshotCachePath(); + const persist = + ttlMs > 0 + ? (snapshot: SnapshotResponse): void => { + void writeAuthBrokerSnapshotCache({ + path: cachePath, + token: brokerConfig.token, + url: brokerConfig.url, + snapshot, + }).catch(error => { + logger.debug("auth-broker snapshot cache write failed", { error: String(error) }); + }); + } + : undefined; + + let initialSnapshot: SnapshotResponse | undefined; + if (ttlMs > 0) { + initialSnapshot = + (await readAuthBrokerSnapshotCache({ + path: cachePath, + token: brokerConfig.token, + url: brokerConfig.url, + ttlMs, + }).catch(error => { + logger.debug("auth-broker snapshot cache read failed", { error: String(error) }); + return null; + })) ?? undefined; + } + if (!initialSnapshot) { + const initialResult = await client.fetchSnapshot(); + if (initialResult.status !== 200) throw new Error("Auth broker returned no initial snapshot"); + initialSnapshot = initialResult.snapshot; + persist?.(initialSnapshot); + } + const store = new RemoteAuthCredentialStore({ client, initialSnapshot, onSnapshot: persist }); // Refresh + usage hooks live on RemoteAuthCredentialStore; AuthStorage // discovers them automatically when no explicit option overrides them. const storage = new AuthStorage(store, { diff --git a/packages/coding-agent/src/session/auth-storage.ts b/packages/coding-agent/src/session/auth-storage.ts index 33f0d1607..8a9b14fad 100644 --- a/packages/coding-agent/src/session/auth-storage.ts +++ b/packages/coding-agent/src/session/auth-storage.ts @@ -12,12 +12,16 @@ export type { AuthStorageOptions, OAuthCredential, SerializedAuthStorage, + SnapshotResponse, StoredAuthCredential, } from "@oh-my-pi/pi-ai"; export { AuthBrokerClient, AuthStorage, + DEFAULT_SNAPSHOT_CACHE_TTL_MS, + readAuthBrokerSnapshotCache, REMOTE_REFRESH_SENTINEL, RemoteAuthCredentialStore, SqliteAuthCredentialStore, + writeAuthBrokerSnapshotCache, } from "@oh-my-pi/pi-ai"; diff --git a/packages/utils/src/dirs.ts b/packages/utils/src/dirs.ts index 3aa0f4192..507bda3a5 100644 --- a/packages/utils/src/dirs.ts +++ b/packages/utils/src/dirs.ts @@ -343,6 +343,17 @@ export function getGithubCacheDbPath(): string { return dirs.rootSubdir(path.join("cache", "github-cache.db"), "cache"); } +/** + * Get the encrypted auth-broker snapshot cache path (~/.omp/cache/auth-broker-snapshot.enc). + * Honors the `OMP_AUTH_BROKER_SNAPSHOT_CACHE` env var when set so tests and + * operators can isolate or relocate the cache file. + */ +export function getAuthBrokerSnapshotCachePath(): string { + const override = process.env.OMP_AUTH_BROKER_SNAPSHOT_CACHE; + if (override) return override; + return dirs.rootSubdir(path.join("cache", "auth-broker-snapshot.enc"), "cache"); +} + /** Get the local FastEmbed model cache directory (~/.omp/cache/fastembed). */ export function getFastembedCacheDir(): string { return dirs.rootSubdir(path.join("cache", "fastembed"), "cache"); From 363d88ca5cfc2e3a573d9665a7886161dc1f0d26 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:09:57 +0200 Subject: [PATCH 37/48] feat(auth): added session-weighted OAuth credential selection - Distributed sessions across unblocked credentials via priority-weighted hashing. - Bucketed candidates by ranking priority for proportional fallback weighting. - Extracted candidate comparison and ordering into dedicated helpers. --- packages/ai/src/auth-storage.ts | 172 +++++++++++++----- .../test/auth-storage-codex-selection.test.ts | 27 +++ 2 files changed, 151 insertions(+), 48 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 39e4c14ef..330227af9 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -2752,35 +2752,126 @@ export class AuthStorage { return usedFraction / elapsedHours; } + #compareRankedOAuthCandidatePriority( + left: RankedOAuthCandidate, + right: RankedOAuthCandidate, + provider: string, + modelId: string | undefined, + ): number { + if (left.blocked !== right.blocked) return left.blocked ? 1 : -1; + if (left.blocked && right.blocked) { + const leftBlockedUntil = left.blockedUntil ?? Number.POSITIVE_INFINITY; + const rightBlockedUntil = right.blockedUntil ?? Number.POSITIVE_INFINITY; + if (leftBlockedUntil !== rightBlockedUntil) return leftBlockedUntil - rightBlockedUntil; + return 0; + } + if (requiresOpenAICodexProModel(provider, modelId) && left.planPriority !== right.planPriority) { + return left.planPriority - right.planPriority; + } + if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; + if (left.secondaryDrainRate !== right.secondaryDrainRate) { + return left.secondaryDrainRate - right.secondaryDrainRate; + } + if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed; + if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate; + if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed; + return 0; + } + + #compareRankedOAuthCandidates( + left: RankedOAuthCandidate, + right: RankedOAuthCandidate, + provider: string, + modelId: string | undefined, + ): number { + const priority = this.#compareRankedOAuthCandidatePriority(left, right, provider, modelId); + return priority !== 0 ? priority : left.orderPos - right.orderPos; + } + + #orderRankedOAuthCandidates( + candidates: RankedOAuthCandidate[], + sessionId: string | undefined, + provider: string, + modelId: string | undefined, + ): OAuthCandidate[] { + candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, provider, modelId)); + if (!sessionId) { + return candidates.map(candidate => ({ + selection: candidate.selection, + usage: candidate.usage, + usageChecked: candidate.usageChecked, + })); + } + + const unblocked = candidates.filter(candidate => !candidate.blocked); + if (unblocked.length <= 1) { + return candidates.map(candidate => ({ + selection: candidate.selection, + usage: candidate.usage, + usageChecked: candidate.usageChecked, + })); + } + + const priorityByCandidate = new Map(); + let bucketIndex = 0; + let previous = unblocked[0]; + const bucketByCandidate = new Map(); + for (const candidate of unblocked) { + if ( + candidate !== previous && + this.#compareRankedOAuthCandidatePriority(previous, candidate, provider, modelId) !== 0 + ) { + bucketIndex += 1; + } + bucketByCandidate.set(candidate, bucketIndex); + previous = candidate; + } + const maxBucket = bucketIndex; + for (const candidate of unblocked) { + const bucket = bucketByCandidate.get(candidate) ?? 0; + priorityByCandidate.set(candidate, maxBucket === 0 ? 0 : 1 - bucket / maxBucket); + } + + let totalWeight = 0; + for (const candidate of unblocked) { + totalWeight += 1 + (priorityByCandidate.get(candidate) ?? 0); + } + + const hit = ((Bun.hash.xxHash32(sessionId) >>> 0) / 2 ** 32) * totalWeight; + let cursor = 0; + let selected = unblocked[unblocked.length - 1]; + for (const candidate of unblocked) { + cursor += 1 + (priorityByCandidate.get(candidate) ?? 0); + if (hit < cursor) { + selected = candidate; + break; + } + } + + const ordered = [ + selected, + ...unblocked.filter(candidate => candidate !== selected), + ...candidates.filter(candidate => candidate.blocked), + ]; + return ordered.map(candidate => ({ + selection: candidate.selection, + usage: candidate.usage, + usageChecked: candidate.usageChecked, + })); + } + async #rankOAuthSelections(args: { providerKey: string; provider: string; order: number[]; - credentials: Array<{ credential: OAuthCredential; index: number }>; + credentials: OAuthSelection[]; options?: AuthApiKeyOptions; + sessionId?: string; strategy: CredentialRankingStrategy; - }): Promise< - Array<{ - selection: { credential: OAuthCredential; index: number }; - usage: UsageReport | null; - usageChecked: boolean; - }> - > { + }): Promise { const nowMs = Date.now(); const { strategy } = args; - const ranked: Array<{ - selection: { credential: OAuthCredential; index: number }; - usage: UsageReport | null; - usageChecked: boolean; - blocked: boolean; - blockedUntil?: number; - hasPriorityBoost: boolean; - secondaryUsed: number; - secondaryDrainRate: number; - primaryUsed: number; - primaryDrainRate: number; - orderPos: number; - }> = []; + const ranked: RankedOAuthCandidate[] = []; // Pre-fetch usage reports in parallel for non-blocked credentials. // Wrap with a timeout so slow/429'd fetches don't indefinitely block // credential selection — better to pick a credential without usage data @@ -2840,6 +2931,7 @@ export class AuthStorage { blocked, blockedUntil, hasPriorityBoost: strategy.hasPriorityBoost?.(primary) ?? false, + planPriority: getOpenAICodexPlanPriority(usage), secondaryUsed: this.#normalizeUsageFraction(secondaryTarget), secondaryDrainRate: this.#computeWindowDrainRate( secondaryTarget, @@ -2851,32 +2943,7 @@ export class AuthStorage { orderPos, }); } - ranked.sort((left, right) => { - if (left.blocked !== right.blocked) return left.blocked ? 1 : -1; - if (left.blocked && right.blocked) { - const leftBlockedUntil = left.blockedUntil ?? Number.POSITIVE_INFINITY; - const rightBlockedUntil = right.blockedUntil ?? Number.POSITIVE_INFINITY; - if (leftBlockedUntil !== rightBlockedUntil) return leftBlockedUntil - rightBlockedUntil; - return left.orderPos - right.orderPos; - } - if (requiresOpenAICodexProModel(args.provider, args.options?.modelId)) { - const leftPlanPriority = getOpenAICodexPlanPriority(left.usage); - const rightPlanPriority = getOpenAICodexPlanPriority(right.usage); - if (leftPlanPriority !== rightPlanPriority) return leftPlanPriority - rightPlanPriority; - } - if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; - if (left.secondaryDrainRate !== right.secondaryDrainRate) - return left.secondaryDrainRate - right.secondaryDrainRate; - if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed; - if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate; - if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed; - return left.orderPos - right.orderPos; - }); - return ranked.map(candidate => ({ - selection: candidate.selection, - usage: candidate.usage, - usageChecked: candidate.usageChecked, - })); + return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.provider, args.options?.modelId); } /** @@ -2913,8 +2980,17 @@ export class AuthStorage { const sessionPreferredIsAvailable = sessionPreferredIndex !== undefined && !this.#isCredentialBlocked(providerKey, sessionPreferredIndex); const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel); + const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order; const candidates = shouldRank - ? await this.#rankOAuthSelections({ providerKey, provider, order, credentials, options, strategy: strategy! }) + ? await this.#rankOAuthSelections({ + providerKey, + provider, + order: rankingOrder, + credentials, + options, + sessionId, + strategy: strategy!, + }) : order .map(idx => credentials[idx]) .filter((selection): selection is { credential: OAuthCredential; index: number } => Boolean(selection)) diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 6402e39c6..b1eb5b63b 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -103,6 +103,33 @@ function createCredential(accountId: string, email: string): OAuthCredentials { }; } +async function countApiKeySelections( + authStorage: AuthStorage, + provider: string, + sessionPrefix: string, + samples = 150, +): Promise> { + const counts = new Map(); + for (let index = 0; index < samples; index += 1) { + const apiKey = await authStorage.getApiKey(provider, `${sessionPrefix}-${index}`); + if (!apiKey) continue; + counts.set(apiKey, (counts.get(apiKey) ?? 0) + 1); + } + return counts; +} + +function countFor(counts: Map, apiKey: string): number { + return counts.get(apiKey) ?? 0; +} + +function expectWeightedPreference(counts: Map, preferred: string, fallback: string): void { + const preferredCount = countFor(counts, preferred); + const fallbackCount = countFor(counts, fallback); + expect(preferredCount).toBeGreaterThan(fallbackCount); + expect(preferredCount / fallbackCount).toBeGreaterThan(1.4); + expect(preferredCount / fallbackCount).toBeLessThan(2.4); +} + describe("AuthStorage codex oauth ranking", () => { let tempDir = ""; let store: AuthCredentialStore | null = null; From b8a602ac2a770b3d3a145efa81485a7046329a13 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:13:44 +0200 Subject: [PATCH 38/48] test(auth): switched OAuth ranking tests to weighted selection - Replaced top-rank assertions with weighted-preference distribution checks. - Added cases for equal-priority balancing and 2x best-bucket cap. - Added coding-agent snapshot-cache boot and seed tests. --- docs/auth-broker-gateway.md | 10 ++ docs/environment-variables.md | 2 + packages/ai/CHANGELOG.md | 8 + packages/ai/src/auth-broker/index.ts | 2 +- .../ai/test/auth-broker-remote-store.test.ts | 1 - .../test/auth-broker-snapshot-cache.test.ts | 6 +- .../test/auth-storage-codex-selection.test.ts | 95 +++++++++--- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/cli/dry-balance-cli.ts | 40 +++-- packages/coding-agent/src/sdk.ts | 2 +- .../coding-agent/src/session/auth-storage.ts | 2 +- .../test/auth-broker-snapshot-cache.test.ts | 140 ++++++++++++++++++ packages/utils/CHANGELOG.md | 4 + 13 files changed, 273 insertions(+), 40 deletions(-) create mode 100644 packages/coding-agent/test/auth-broker-snapshot-cache.test.ts diff --git a/docs/auth-broker-gateway.md b/docs/auth-broker-gateway.md index ec7948d1b..1fccca164 100644 --- a/docs/auth-broker-gateway.md +++ b/docs/auth-broker-gateway.md @@ -140,6 +140,14 @@ When the gateway (or any other broker client) calls `fetchUsageReports()` / `get The 15 s client window deliberately sits below the broker’s 5 min server cache, so almost every client poll is served from the broker’s already-cached value; the client cache exists to absorb the parallel fan-out generated by `AuthStorage.#rankOAuthSelections` into a single broker round-trip. +## Client snapshot cache + +`discoverAuthStorage()` persists the broker snapshot to `~/.omp/cache/auth-broker-snapshot.enc` after the initial `/v1/snapshot` fetch and after later broker-sourced full snapshots. The file is AES-256-GCM encrypted with `SHA-256(OMP_AUTH_BROKER_TOKEN)` and authenticated with the broker URL as additional data, so changing either the token or URL makes the cache unreadable. The file is written atomically with mode `0600`. + +Freshness is anchored to the broker-stamped `snapshot.generatedAt`, not local write time. Default TTL is 1 h (`OMP_AUTH_BROKER_SNAPSHOT_TTL_MS`); `0` disables the cache and restores the old always-fetch boot path. When the cached snapshot is still fresh, `omp` boots from it and skips the blocking `/v1/snapshot` query. `RemoteAuthCredentialStore` still starts its normal SSE / long-poll background sync immediately, so deleted or rotated credentials reconcile after startup, and expired OAuth access tokens still refresh through `POST /v1/credential/:id/refresh`. + +If the broker is down at boot and a fresh cache exists, startup now succeeds from the cached snapshot. If the cache is missing, expired, corrupt, written for a different URL, or encrypted with a different token, startup falls back to the live fetch and fails the same way it did before if the broker is unreachable. + ## Operator opt-in The broker is **off** unless `OMP_AUTH_BROKER_URL` (or `auth.broker.url` in `config.yml`) is set. When set, `discoverAuthStorage` in `packages/coding-agent/src/sdk.ts` swaps the local SQLite credential store for `RemoteAuthCredentialStore` and every API call resolves credentials through the broker. @@ -150,6 +158,8 @@ The broker is **off** unless `OMP_AUTH_BROKER_URL` (or `auth.broker.url` in `con | ----------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | | `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`). Selecting this puts the client in broker mode — local SQLite is bypassed. | Any time the omp client should resolve credentials through a broker (and required by `omp auth-gateway serve`). | | `OMP_AUTH_BROKER_TOKEN` | Bearer token used for every broker endpoint except `/v1/healthz`. | When `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `/auth-broker.token`. | +| `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` | Freshness window for the encrypted local snapshot cache. Default `3600000` (1 h); `0` disables cache reads and writes. | Optional in broker mode. | +| `OMP_AUTH_BROKER_SNAPSHOT_CACHE` | Path override for the encrypted local snapshot cache. Default `~/.omp/cache/auth-broker-snapshot.enc` (or XDG cache equivalent). | Optional in broker mode. | Resolution order in `resolveAuthBrokerConfig()`: diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 131d9170c..e556301e8 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -97,6 +97,8 @@ When the broker is enabled, the local SQLite credential store is bypassed and al | ----------------------- | -------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | | `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`); selects broker mode | Resolving credentials through a broker; also required by `omp auth-gateway serve` (the gateway is itself a broker client) | Wins over `auth.broker.url` in `config.yml`. When set with no resolvable token, `resolveAuthBrokerConfig()` hard-errors instead of falling back to local SQLite. | | `OMP_AUTH_BROKER_TOKEN` | Bearer token sent on every broker endpoint except `/v1/healthz` | `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `/auth-broker.token` | Resolution: this env → `auth.broker.token` (`$ENV_NAME` indirection supported) → `/auth-broker.token` (mode `0600`). `` is `~/.omp/` (respecting `PI_CONFIG_DIR`). | +| `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` | Freshness window for the encrypted local broker snapshot cache | Optional in broker mode | Default `3600000` (1 h). Freshness is based on broker `snapshot.generatedAt`; `0` disables cache reads/writes and forces the old blocking fetch every startup. | +| `OMP_AUTH_BROKER_SNAPSHOT_CACHE` | Path to the encrypted local broker snapshot cache | Optional in broker mode | Defaults to `~/.omp/cache/auth-broker-snapshot.enc` (or XDG cache equivalent). Useful for tests, ephemeral hosts, or relocating the `0600` cache file. | The gateway has no dedicated env vars — it inherits `OMP_AUTH_BROKER_*`. Its own inbound bearer token lives at `/auth-gateway.token` and is managed via `omp auth-gateway token`. diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2789f64c8..59f2033dd 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +### Added + +- Added an AES-256-GCM auth-broker snapshot cache module and `RemoteAuthCredentialStoreOptions.onSnapshot` so broker clients can persist broker-sourced full snapshots without blocking startup on every run. + +### Changed + +- Changed usage-ranked OAuth credential selection to pick deterministic session-sticky weighted buckets instead of always choosing the top-ranked account, capping the best account at 2x the baseline session likelihood while keeping equal-priority accounts evenly balanced. + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/ai/src/auth-broker/index.ts b/packages/ai/src/auth-broker/index.ts index a2f31df85..189d377c5 100644 --- a/packages/ai/src/auth-broker/index.ts +++ b/packages/ai/src/auth-broker/index.ts @@ -1,6 +1,6 @@ export * from "./client"; export * from "./refresher"; export * from "./remote-store"; -export * from "./snapshot-cache"; export * from "./server"; +export * from "./snapshot-cache"; export * from "./types"; diff --git a/packages/ai/test/auth-broker-remote-store.test.ts b/packages/ai/test/auth-broker-remote-store.test.ts index ee288a0bb..50be18938 100644 --- a/packages/ai/test/auth-broker-remote-store.test.ts +++ b/packages/ai/test/auth-broker-remote-store.test.ts @@ -126,7 +126,6 @@ describe("RemoteAuthCredentialStore SSE integration", () => { }); expect(callbacks).toHaveLength(0); - storage!.upsertCredential("anthropic", mintOAuthCredential("callback", Date.now() + 120_000)); const refreshed = await remote.refreshSnapshot(); expect(callbacks).toHaveLength(1); diff --git a/packages/ai/test/auth-broker-snapshot-cache.test.ts b/packages/ai/test/auth-broker-snapshot-cache.test.ts index 58aa20b72..9480472c6 100644 --- a/packages/ai/test/auth-broker-snapshot-cache.test.ts +++ b/packages/ai/test/auth-broker-snapshot-cache.test.ts @@ -2,11 +2,7 @@ import { describe, expect, test } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; -import { - readAuthBrokerSnapshotCache, - type SnapshotResponse, - writeAuthBrokerSnapshotCache, -} from "../src"; +import { readAuthBrokerSnapshotCache, type SnapshotResponse, writeAuthBrokerSnapshotCache } from "../src"; const TOKEN = "broker-cache-token"; const URL = "http://127.0.0.1:8765"; diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index b1eb5b63b..7e23abae9 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -173,7 +173,7 @@ describe("AuthStorage codex oauth ranking", () => { } }); - test("prefers near-reset weekly account over lower-used far-reset account", async () => { + test("weights near-reset weekly account over lower-used far-reset account", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("openai-codex", [ @@ -198,11 +198,11 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-weekly-reset"); - expect(apiKey).toBe("api-acct-near"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-near"); + expectWeightedPreference(counts, "api-acct-near", "api-acct-far"); }); - test("prioritizes fresh 5h ticker account at 0% usage", async () => { + test("weights fresh 5h ticker account at 0% usage", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("openai-codex", [ @@ -235,8 +235,8 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-five-hour-start"); - expect(apiKey).toBe("api-acct-zero"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-zero"); + expectWeightedPreference(counts, "api-acct-zero", "api-acct-progress"); }); test("skips exhausted weekly account even when reset is near", async () => { if (!authStorage) throw new Error("test setup failed"); @@ -426,7 +426,7 @@ describe("AuthStorage codex oauth ranking", () => { expect(elapsedMs).toBeLessThan(1_000); }); - test("sorts 3 accounts by weekly drain rate", async () => { + test("weights 3 accounts by weekly drain rate", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("openai-codex", [ @@ -460,8 +460,9 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-three-accounts"); - expect(apiKey).toBe("api-acct-slow"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-three"); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-medium")); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-fast")); }); test("handles usage fetch failure gracefully (null report)", async () => { @@ -482,8 +483,8 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-null-usage"); - expect(apiKey).toBe("api-acct-known"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-known", 300); + expectWeightedPreference(counts, "api-acct-known", "api-acct-null"); }); test("refreshes expired oauth candidates in parallel before selection", async () => { if (!authStorage) throw new Error("test setup failed"); @@ -650,7 +651,7 @@ describe("AuthStorage claude oauth ranking", () => { } }); - test("prefers lower secondary drain rate account", async () => { + test("weights lower secondary drain rate account", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("anthropic", [ @@ -675,8 +676,67 @@ describe("AuthStorage claude oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("anthropic", "session-claude-drain"); - expect(apiKey).toBe("api-acct-near"); + const counts = await countApiKeySelections(authStorage, "anthropic", "weighted-claude-near"); + expectWeightedPreference(counts, "api-acct-near", "api-acct-far"); + }); + + test("balances equal-priority accounts evenly", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("anthropic", [ + { type: "oauth", ...createCredential("acct-a", "a@example.com") }, + { type: "oauth", ...createCredential("acct-b", "b@example.com") }, + ]); + + for (const accountId of ["acct-a", "acct-b"]) { + usageByAccount.set( + accountId, + createClaudeUsageReport({ + accountId, + primary: { usedFraction: 0.25, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.25, resetInMs: 4 * 24 * HOUR_MS }, + }), + ); + } + + const counts = await countApiKeySelections(authStorage, "anthropic", "weighted-claude-equal", 200); + expect(Math.abs(countFor(counts, "api-acct-a") - countFor(counts, "api-acct-b"))).toBeLessThanOrEqual(25); + }); + + test("caps the strongest priority bucket at about 2x baseline weight", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("anthropic", [ + { type: "oauth", ...createCredential("acct-best", "best@example.com") }, + { type: "oauth", ...createCredential("acct-base-a", "base-a@example.com") }, + { type: "oauth", ...createCredential("acct-base-b", "base-b@example.com") }, + ]); + + usageByAccount.set( + "acct-best", + createClaudeUsageReport({ + accountId: "acct-best", + primary: { usedFraction: 0.05, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.05, resetInMs: 6 * 24 * HOUR_MS }, + }), + ); + for (const accountId of ["acct-base-a", "acct-base-b"]) { + usageByAccount.set( + accountId, + createClaudeUsageReport({ + accountId, + primary: { usedFraction: 0.7, resetInMs: 2 * HOUR_MS }, + secondary: { usedFraction: 0.7, resetInMs: 2 * 24 * HOUR_MS }, + }), + ); + } + + const counts = await countApiKeySelections(authStorage, "anthropic", "claude-cap", 300); + expectWeightedPreference(counts, "api-acct-best", "api-acct-base-a"); + expectWeightedPreference(counts, "api-acct-best", "api-acct-base-b"); + expect(Math.abs(countFor(counts, "api-acct-base-a") - countFor(counts, "api-acct-base-b"))).toBeLessThanOrEqual( + 15, + ); }); test("skips exhausted account and picks healthy", async () => { @@ -737,7 +797,7 @@ describe("AuthStorage claude oauth ranking", () => { expect(apiKey).toBe("api-acct-soon"); }); - test("sorts 3 accounts by secondary drain rate", async () => { + test("weights 3 accounts by secondary drain rate", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("anthropic", [ @@ -771,8 +831,9 @@ describe("AuthStorage claude oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("anthropic", "session-claude-three"); - expect(apiKey).toBe("api-acct-slow"); + const counts = await countApiKeySelections(authStorage, "anthropic", "weighted-claude-three"); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-medium")); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-fast")); }); test("single credential works without ranking", async () => { diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 18f2b7ddc..792cb455e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -3,6 +3,7 @@ ## [Unreleased] ### Added +- Added an encrypted local auth-broker snapshot cache for `discoverAuthStorage`, with `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` and `OMP_AUTH_BROKER_SNAPSHOT_CACHE`, so fresh cached broker credentials can boot without a blocking `/v1/snapshot` fetch and survive broker-down startup windows. - Added `dry-balance` CLI command to perform a dry-run OAuth account balancing check across configurable random session IDs, with sample and concurrency options, JSON output, and success/failure summary reporting - Added `--json` output mode and machine-readable result format to `omp dry-balance` for automated use diff --git a/packages/coding-agent/src/cli/dry-balance-cli.ts b/packages/coding-agent/src/cli/dry-balance-cli.ts index 03249c705..1e16d5237 100644 --- a/packages/coding-agent/src/cli/dry-balance-cli.ts +++ b/packages/coding-agent/src/cli/dry-balance-cli.ts @@ -4,10 +4,10 @@ import chalk from "chalk"; import { ModelRegistry } from "../config/model-registry"; import { formatModelString, + type ModelMatchPreferences, resolveAllowedModels, resolveCliModel, resolveModelRoleValue, - type ModelMatchPreferences, } from "../config/model-resolver"; import { Settings } from "../config/settings"; import { discoverAuthStorage } from "../sdk"; @@ -32,7 +32,11 @@ export interface DryBalanceAuthOptions { } export interface DryBalanceAuthStorage { - getOAuthAccess(provider: string, sessionId?: string, options?: DryBalanceAuthOptions): Promise; + getOAuthAccess( + provider: string, + sessionId?: string, + options?: DryBalanceAuthOptions, + ): Promise; } export interface DryBalanceModelRegistry { @@ -142,7 +146,9 @@ async function resolveDryBalanceModel( const allowedModels = await resolveAllowedModels(modelRegistry, settings, preferences); if (allowedModels.length === 0) { - throw new Error("No models available. Use --model to select a model or configure enabledModels/default model settings."); + throw new Error( + "No models available. Use --model to select a model or configure enabledModels/default model settings.", + ); } const defaultRoleSpec = resolveModelRoleValue(settings?.getModelRole("default"), allowedModels, { @@ -161,12 +167,11 @@ async function resolveDryBalanceModel( return { model: allowedModels[0], - warning: "No allowed model had usable credentials during default resolution; dry-balance will report OAuth failures for the first allowed model.", + warning: + "No allowed model had usable credentials during default resolution; dry-balance will report OAuth failures for the first allowed model.", }; } - - async function runOneAttempt( model: Model, modelRegistry: DryBalanceModelRegistry, @@ -181,7 +186,8 @@ async function runOneAttempt( modelId: model.id, }); if (!access) return { ok: false, reason: "no OAuth access resolved" }; - const account = access.email ?? access.accountId ?? access.projectId ?? access.enterpriseUrl ?? "(unknown oauth account)"; + const account = + access.email ?? access.accountId ?? access.projectId ?? access.enterpriseUrl ?? "(unknown oauth account)"; return { ok: true, account }; } catch (error) { return { ok: false, reason: error instanceof Error ? error.message : String(error) }; @@ -205,8 +211,10 @@ async function mapConcurrent(items: T[], concurrency: number, fn: (item: T return results; } - -function sortedStats(map: Map, samples: number): Array<{ label: string; count: number; percent: number }> { +function sortedStats( + map: Map, + samples: number, +): Array<{ label: string; count: number; percent: number }> { return [...map.entries()] .map(([label, count]) => ({ label, count, percent: (count / samples) * 100 })) .sort((left, right) => right.count - left.count || left.label.localeCompare(right.label)); @@ -253,7 +261,6 @@ function summarizeResults( }; } - function formatRows(rows: Array<{ count: number; percent: number; label: string }>): string[] { if (rows.length === 0) return [` ${chalk.dim("(none)")}`]; const maxCountWidth = Math.max(...rows.map(row => row.count.toString().length)); @@ -296,13 +303,18 @@ export async function runDryBalanceCommand( deps: DryBalanceDependencies = {}, ): Promise { const samples = normalizePositiveInteger("count", command.flags.count, DEFAULT_SAMPLE_COUNT); - const concurrency = Math.min(samples, normalizePositiveInteger("concurrency", command.flags.concurrency, DEFAULT_CONCURRENCY)); + const concurrency = Math.min( + samples, + normalizePositiveInteger("concurrency", command.flags.concurrency, DEFAULT_CONCURRENCY), + ); const randomSessionId = deps.randomSessionId ?? (() => Bun.randomUUIDv7()); const writeStdout = deps.writeStdout ?? ((text: string) => process.stdout.write(text)); const writeStderr = deps.writeStderr ?? ((text: string) => process.stderr.write(text)); - const setExitCode = deps.setExitCode ?? ((code: number) => { - process.exitCode = code; - }); + const setExitCode = + deps.setExitCode ?? + ((code: number) => { + process.exitCode = code; + }); const runtime = await (deps.createRuntime ?? createDefaultRuntime)(); try { const modelSelector = command.flags.model ?? command.model; diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 9ae14fd17..63a82447f 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -106,8 +106,8 @@ import { AuthBrokerClient, AuthStorage, DEFAULT_SNAPSHOT_CACHE_TTL_MS, - readAuthBrokerSnapshotCache, RemoteAuthCredentialStore, + readAuthBrokerSnapshotCache, type SnapshotResponse, writeAuthBrokerSnapshotCache, } from "./session/auth-storage"; diff --git a/packages/coding-agent/src/session/auth-storage.ts b/packages/coding-agent/src/session/auth-storage.ts index 8a9b14fad..d3486d7e2 100644 --- a/packages/coding-agent/src/session/auth-storage.ts +++ b/packages/coding-agent/src/session/auth-storage.ts @@ -19,9 +19,9 @@ export { AuthBrokerClient, AuthStorage, DEFAULT_SNAPSHOT_CACHE_TTL_MS, - readAuthBrokerSnapshotCache, REMOTE_REFRESH_SENTINEL, RemoteAuthCredentialStore, + readAuthBrokerSnapshotCache, SqliteAuthCredentialStore, writeAuthBrokerSnapshotCache, } from "@oh-my-pi/pi-ai"; diff --git a/packages/coding-agent/test/auth-broker-snapshot-cache.test.ts b/packages/coding-agent/test/auth-broker-snapshot-cache.test.ts new file mode 100644 index 000000000..7dba4027c --- /dev/null +++ b/packages/coding-agent/test/auth-broker-snapshot-cache.test.ts @@ -0,0 +1,140 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { type AuthBrokerServerHandle, AuthStorage, SqliteAuthCredentialStore, startAuthBroker } from "@oh-my-pi/pi-ai"; +import { discoverAuthStorage } from "../src/sdk"; +import { + readAuthBrokerSnapshotCache, + type SnapshotResponse, + writeAuthBrokerSnapshotCache, +} from "../src/session/auth-storage"; + +const ENV_KEYS = [ + "OMP_AUTH_BROKER_URL", + "OMP_AUTH_BROKER_TOKEN", + "OMP_AUTH_BROKER_SNAPSHOT_CACHE", + "OMP_AUTH_BROKER_SNAPSHOT_TTL_MS", +] as const; +const PROVIDER = "unit-auth-broker-cache"; +const TOKEN = "coding-agent-cache-token"; + +const savedEnv: Partial> = {}; + +function makeSnapshot(urlTime: number): SnapshotResponse { + return { + generation: 11, + generatedAt: urlTime, + serverNowMs: urlTime, + refresher: { + enabled: false, + intervalMs: 60_000, + skewMs: 300_000, + nextSweepInMs: Number.MAX_SAFE_INTEGER, + }, + credentials: [ + { + id: 1, + provider: PROVIDER, + credential: { type: "api_key", key: "cached-api-key" }, + identityKey: null, + rotatesInMs: null, + }, + ], + }; +} + +async function waitUntil(predicate: () => boolean | Promise, timeoutMs = 2_000): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + if (await predicate()) return; + await Bun.sleep(10); + } + if (!(await predicate())) throw new Error("waitUntil timeout"); +} + +describe("discoverAuthStorage auth-broker snapshot cache", () => { + let tempDir = ""; + + beforeEach(async () => { + for (const key of ENV_KEYS) savedEnv[key] = process.env[key]; + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "coding-agent-auth-broker-cache-")); + }); + + afterEach(async () => { + for (const key of ENV_KEYS) { + if (savedEnv[key] === undefined) delete process.env[key]; + else process.env[key] = savedEnv[key]; + } + await fs.rm(tempDir, { recursive: true, force: true }); + }); + + test("boots from a fresh encrypted cache when the broker is down", async () => { + const cachePath = path.join(tempDir, "snapshot.enc"); + const downUrl = "http://127.0.0.1:1"; + process.env.OMP_AUTH_BROKER_URL = downUrl; + process.env.OMP_AUTH_BROKER_TOKEN = TOKEN; + process.env.OMP_AUTH_BROKER_SNAPSHOT_CACHE = cachePath; + process.env.OMP_AUTH_BROKER_SNAPSHOT_TTL_MS = "3600000"; + await writeAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: downUrl, + snapshot: makeSnapshot(Date.now()), + }); + + const storage = await discoverAuthStorage(tempDir); + try { + expect(await storage.getApiKey(PROVIDER)).toBe("cached-api-key"); + } finally { + storage.close(); + } + }); + + test("seeds the encrypted cache after an initial broker fetch", async () => { + const cachePath = path.join(tempDir, "snapshot.enc"); + const brokerStore = await SqliteAuthCredentialStore.open(path.join(tempDir, "broker.db")); + brokerStore.saveApiKey(PROVIDER, "broker-api-key"); + const brokerStorage = new AuthStorage(brokerStore); + await brokerStorage.reload(); + let handle: AuthBrokerServerHandle | undefined; + let storage: AuthStorage | undefined; + try { + handle = startAuthBroker({ + storage: brokerStorage, + bind: "127.0.0.1:0", + bearerTokens: [TOKEN], + disableRefresher: true, + }); + process.env.OMP_AUTH_BROKER_URL = handle.url; + process.env.OMP_AUTH_BROKER_TOKEN = TOKEN; + process.env.OMP_AUTH_BROKER_SNAPSHOT_CACHE = cachePath; + process.env.OMP_AUTH_BROKER_SNAPSHOT_TTL_MS = "3600000"; + + storage = await discoverAuthStorage(tempDir); + expect(await storage.getApiKey(PROVIDER)).toBe("broker-api-key"); + await waitUntil(async () => { + const cached = await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: handle!.url, + ttlMs: 3_600_000, + }); + return cached?.credentials.some(entry => entry.provider === PROVIDER) ?? false; + }); + const cached = await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: handle.url, + ttlMs: 3_600_000, + }); + const entry = cached?.credentials.find(candidate => candidate.provider === PROVIDER); + expect(entry?.credential).toEqual({ type: "api_key", key: "broker-api-key" }); + } finally { + storage?.close(); + await handle?.close(); + brokerStorage.close(); + brokerStore.close(); + } + }); +}); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index a33a5cd87..dc16a8eeb 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `getAuthBrokerSnapshotCachePath()` with `OMP_AUTH_BROKER_SNAPSHOT_CACHE` override support for isolating the encrypted broker snapshot cache. + ## [15.9.1] - 2026-06-04 ### Fixed From eda5eebe7113fbda92cd00e812490a887695912a Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 09:22:18 +0000 Subject: [PATCH 39/48] fix(sdk): route bash/task/job through ToolSession.asyncJobManager to keep secondary sessions isolated MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per PR review on #1926: a secondary in-process top-level createAgentSession() that exposes bash/task/job tools would still call AsyncJobManager.instance() at execute time, register on the primary's manager, and have the primary's onJobComplete enqueue results into the primary's yieldQueue — corrupting the owning session's conversation. ToolSession now carries an asyncJobManager reference scoped to its session: the constructed manager for top-level sessions, the inherited singleton for subagents (so their bash/task completions still flow into the spawning conversation as before), and undefined for secondary in-process top-level sessions that found a singleton already installed. bash, task, and job tools resolve the manager through ToolSession instead of the process-global singleton, so a secondary session whose tools attempt async work fails fast with the standard "Async job manager unavailable" error instead of contaminating the primary. --- packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/sdk.ts | 7 +++++ packages/coding-agent/src/task/index.ts | 3 +- packages/coding-agent/src/tools/bash.ts | 7 ++--- packages/coding-agent/src/tools/index.ts | 15 ++++++++++ packages/coding-agent/src/tools/job.ts | 4 +-- .../test/async-yield-queue.test.ts | 5 ++-- .../sdk-async-job-manager-singleton.test.ts | 30 +++++++++++++++++-- 8 files changed, 60 insertions(+), 13 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d4af64036..992490086 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. The secondary now shares the live singleton instead of clobbering it, and its `cancelOwnAsyncJobs` dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). +- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. The secondary now shares the live singleton instead of clobbering it, and its `cancelOwnAsyncJobs` dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools now resolve the manager through their `ToolSession` rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager (which would route completions into the wrong `yieldQueue`); subagents still inherit the parent's manager via their `ToolSession.asyncJobManager` ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). ## [15.9.1] - 2026-06-04 diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index cf969e940..1cf415733 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1297,6 +1297,13 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} authStorage, modelRegistry, getTelemetry: () => agent?.telemetry, + // Subagents inherit the singleton (the parent's manager) so their bash/task + // completions still flow into the spawning conversation's yieldQueue. + // Secondary in-process top-level sessions (no parentTaskPrefix, no + // constructed manager because the singleton was already installed) leave + // this undefined so tools refuse async work instead of silently routing it + // into the owning session (issue #1923). + asyncJobManager: asyncJobManager ?? (options.parentTaskPrefix ? AsyncJobManager.instance() : undefined), }; // Wire process-wide internal URL singletons owned by their real classes. diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index 466d966d1..8f3e33b70 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -19,7 +19,6 @@ import type { AgentTool, AgentToolResult, AgentToolUpdateCallback } from "@oh-my import type { Usage } from "@oh-my-pi/pi-ai"; import { $env, prompt, Snowflake } from "@oh-my-pi/pi-utils"; import type { ToolSession } from ".."; -import { AsyncJobManager } from "../async"; import { resolveAgentModelPatterns } from "../config/model-resolver"; import { MCPManager } from "../mcp/manager"; import type { Theme } from "../modes/theme/theme"; @@ -343,7 +342,7 @@ export class TaskTool implements AgentTool { onUpdate?: AgentToolUpdateCallback; startBackgrounded: boolean; }): ManagedBashJobHandle { - const manager = AsyncJobManager.instance(); + const manager = this.session.asyncJobManager; if (!manager) { throw new ToolError("Background job manager unavailable for this session."); } @@ -716,7 +715,7 @@ export class BashTool implements AgentTool { if (timeoutClampNotice) pendingNotices.push(timeoutClampNotice); if (asyncRequested) { - if (!AsyncJobManager.instance()) { + if (!this.session.asyncJobManager) { throw new ToolError("Async job manager unavailable for this session."); } const job = this.#startManagedBashJob({ @@ -737,7 +736,7 @@ export class BashTool implements AgentTool { }); } - const autoBgManager = AsyncJobManager.instance(); + const autoBgManager = this.session.asyncJobManager; if (this.#autoBackgroundEnabled && !pty && autoBgManager) { const autoBackgroundWaitMs = this.#resolveAutoBackgroundWaitMs(timeoutMs); const startBackgrounded = autoBackgroundWaitMs === 0; diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index fba695757..0043fba0b 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -183,6 +183,21 @@ export interface ToolSession { modelRegistry?: import("../config/model-registry").ModelRegistry; /** Agent output manager for unique agent:// IDs across task invocations */ agentOutputManager?: AgentOutputManager; + /** + * Async job manager scoped to this session. + * + * - Top-level session that constructed one: its own manager. + * - Subagent (`parentTaskPrefix` set): the parent's manager, so background + * bash/task work and `onJobComplete` deliveries flow into the conversation + * that spawned it. + * - Secondary in-process top-level session that found a singleton already + * installed (issue #1923): `undefined`. Tools refuse async work rather + * than silently route completions into the owning session's `yieldQueue`. + * + * Tools MUST use this instead of `AsyncJobManager.instance()` so a secondary + * session never borrows the owning session's manager by accident. + */ + asyncJobManager?: import("../async/job-manager").AsyncJobManager; /** MCP manager visible to subagents without relying on the process-global singleton. */ mcpManager?: MCPManager; /** Local protocol root to propagate to nested subagents and eval-created agents. */ diff --git a/packages/coding-agent/src/tools/job.ts b/packages/coding-agent/src/tools/job.ts index ba2daeb25..ae5833930 100644 --- a/packages/coding-agent/src/tools/job.ts +++ b/packages/coding-agent/src/tools/job.ts @@ -90,7 +90,7 @@ export class JobTool implements AgentTool { onUpdate?: AgentToolUpdateCallback, _context?: AgentToolContext, ): Promise> { - const manager = AsyncJobManager.instance(); + const manager = this.session.asyncJobManager; if (!manager) { return { content: [{ type: "text", text: "Async execution is disabled; no background jobs are available." }], @@ -254,7 +254,7 @@ export class JobTool implements AgentTool { ): JobSnapshot[] { const now = Date.now(); return jobs.map(j => { - const current = AsyncJobManager.instance()?.getJob(j.id); + const current = this.session.asyncJobManager?.getJob(j.id); const latest = current ?? j; return { id: latest.id, diff --git a/packages/coding-agent/test/async-yield-queue.test.ts b/packages/coding-agent/test/async-yield-queue.test.ts index 07984a619..e4ad7a146 100644 --- a/packages/coding-agent/test/async-yield-queue.test.ts +++ b/packages/coding-agent/test/async-yield-queue.test.ts @@ -47,7 +47,7 @@ function asyncDetails(message: AgentMessage): AsyncDetails { return (message as CustomMessage).details ?? { jobs: [] }; } -function createToolSession(): ToolSession { +function createToolSession(asyncJobManager?: AsyncJobManager): ToolSession { return { cwd: process.cwd(), hasUI: false, @@ -57,6 +57,7 @@ function createToolSession(): ToolSession { getSessionFile: () => null, getSessionSpawns: () => null, getAgentId: () => null, + asyncJobManager, } as unknown as ToolSession; } @@ -130,7 +131,7 @@ describe("async result yield queue delivery", () => { await harness.manager.waitForAll(); await waitUntil(() => harness.queue.has("async-result"), "Timed out waiting for staged async result"); - const tool = new JobTool(createToolSession()); + const tool = new JobTool(createToolSession(harness.manager)); const result = await tool.execute("tool-call", { poll: [jobId] }); expect(result.details?.jobs.find(job => job.id === jobId)?.status).toBe("completed"); diff --git a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts index 4c0f14b5b..5ece9eff8 100644 --- a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts +++ b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts @@ -17,7 +17,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () => AsyncJobManager.resetForTests(); }); - async function spawnTopLevelSession() { + async function spawnTopLevelSession(extraSettings?: Record) { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-sdk-async-singleton-${Snowflake.next()}-`)); tempDirs.push(tempDir); const cwd = path.join(tempDir, `project-${Snowflake.next()}`); @@ -26,7 +26,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () => const { session } = await createAgentSession({ cwd, agentDir, - settings: Settings.isolated({ "bash.autoBackground.enabled": true }), + settings: Settings.isolated({ "bash.autoBackground.enabled": true, ...(extraSettings ?? {}) }), disableExtensionDiscovery: true, skills: [], contextFiles: [], @@ -102,4 +102,30 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () => await primary.dispose(); } }); + + it("refuses async bash from a secondary session instead of routing it to the primary's manager", async () => { + const primary = await spawnTopLevelSession({ "async.enabled": true }); + try { + const primaryManager = AsyncJobManager.instance(); + expect(primaryManager).toBeDefined(); + const primaryJobCountBefore = primaryManager!.getAllJobs().length; + + const secondary = await spawnTopLevelSession({ "async.enabled": true }); + try { + const bashTool = secondary.getToolByName("bash"); + expect(bashTool).toBeDefined(); + await expect(bashTool!.execute("call-1", { command: "echo hi", async: true })).rejects.toThrow( + /Async job manager unavailable/, + ); + } finally { + await secondary.dispose(); + } + + // The secondary's failed async attempt must not have leaked a job into + // the primary's manager. + expect(primaryManager!.getAllJobs().length).toBe(primaryJobCountBefore); + } finally { + await primary.dispose(); + } + }); }); From 5d4bba80c2eaa583de2f498cea1e57b27e7e6277 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 09:22:23 +0000 Subject: [PATCH 40/48] style: bun run fix --- packages/coding-agent/src/tools/job.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/coding-agent/src/tools/job.ts b/packages/coding-agent/src/tools/job.ts index ae5833930..18defb3f7 100644 --- a/packages/coding-agent/src/tools/job.ts +++ b/packages/coding-agent/src/tools/job.ts @@ -3,7 +3,7 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { type AsyncJob, AsyncJobManager, isBackgroundJobSupportEnabled } from "../async"; +import { type AsyncJob, type AsyncJobManager, isBackgroundJobSupportEnabled } from "../async"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import type { Theme } from "../modes/theme/theme"; import jobDescription from "../prompts/tools/job.md" with { type: "text" }; From 1e3a8d5cdf58d726d6c6e61eb14245294fe731ef Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:29:18 +0200 Subject: [PATCH 41/48] fix(tui): pinned native scrollback to commit sealed live-region rows - Added `NativeScrollbackLiveRegion` seam so components report the live suffix start. - Stopped ED3-risk streaming from dropping sealed transcript rows above the live block. - Appended newly sealed rows once while keeping the active tail deferred to checkpoint. --- .../modes/components/transcript-container.ts | 17 +- .../components/transcript-container.test.ts | 16 ++ packages/tui/CHANGELOG.md | 6 +- packages/tui/src/tui.ts | 177 +++++++++++++++++- .../test/streaming-scrollback-defer.test.ts | 50 ++++- 5 files changed, 256 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/src/modes/components/transcript-container.ts b/packages/coding-agent/src/modes/components/transcript-container.ts index 64af9ce42..3925a6f27 100644 --- a/packages/coding-agent/src/modes/components/transcript-container.ts +++ b/packages/coding-agent/src/modes/components/transcript-container.ts @@ -1,4 +1,4 @@ -import { type Component, Container, TERMINAL } from "@oh-my-pi/pi-tui"; +import { type Component, Container, type NativeScrollbackLiveRegion, TERMINAL } from "@oh-my-pi/pi-tui"; const kSnapshot = Symbol("transcript.frozenRender"); @@ -34,7 +34,7 @@ interface SnapshotCarrier { * and any drift reconciles safely. On terminals that can rebuild history this * freezing is unnecessary, so it renders every block live for full fidelity. */ -export class TranscriptContainer extends Container { +export class TranscriptContainer extends Container implements NativeScrollbackLiveRegion { // Bumped to invalidate every block's snapshot at once; a snapshot is only // honored when its stored generation still matches. #generation = 0; @@ -43,6 +43,10 @@ export class TranscriptContainer extends Container { // predate content that finalized in the same coalesced frame that appended the // block now below it — so it must recompute once on the live→frozen transition. #prevLiveChild: Component | undefined; + // Local line index where the current bottom-most block begins in the most + // recent render. TUI extends the native-scrollback pinned region from this + // point through the live block and the root chrome rendered below it. + #nativeScrollbackLiveRegionStart: number | undefined; override invalidate(): void { // A theme/global invalidation forces a full recompute on the rebuild that @@ -56,6 +60,10 @@ export class TranscriptContainer extends Container { super.clear(); } + getNativeScrollbackLiveRegionStart(): number | undefined { + return this.#nativeScrollbackLiveRegionStart; + } + /** * Retire all frozen snapshots so the next render reflects each block's current * state. Call at reconciliation checkpoints (prompt submit) where the whole @@ -68,6 +76,7 @@ export class TranscriptContainer extends Container { override render(width: number): string[] { width = Math.max(1, width); + this.#nativeScrollbackLiveRegionStart = undefined; if (!TERMINAL.eagerEraseScrollbackRisk) return super.render(width); const lines: string[] = []; @@ -77,7 +86,9 @@ export class TranscriptContainer extends Container { this.#prevLiveChild = liveChild; for (let i = 0; i < this.children.length; i++) { const child = this.children[i]! as Component & SnapshotCarrier; - if (child !== liveChild) { + if (child === liveChild) { + this.#nativeScrollbackLiveRegionStart = lines.length; + } else { const snapshot = child[kSnapshot]; // Replay the block's last render from while it was live. A stale // generation (post-thaw) or width mismatch (resize in flight, an diff --git a/packages/coding-agent/test/modes/components/transcript-container.test.ts b/packages/coding-agent/test/modes/components/transcript-container.test.ts index 5ff34729a..eb7c4435c 100644 --- a/packages/coding-agent/test/modes/components/transcript-container.test.ts +++ b/packages/coding-agent/test/modes/components/transcript-container.test.ts @@ -53,6 +53,22 @@ describe("TranscriptContainer", () => { expect(container.render(40)).toEqual(["a2", "b2"]); }); + it("reports the live block start for native scrollback pinning (ED3-risk)", () => { + riskFlag.eagerEraseScrollbackRisk = true; + const container = new TranscriptContainer(); + const a = new MutableBlock(["a1", "a2"]); + const b = new MutableBlock(["b1"]); + container.addChild(a); + container.addChild(b); + + expect(container.render(40)).toEqual(["a1", "a2", "b1"]); + expect(container.getNativeScrollbackLiveRegionStart()).toBe(2); + + b.set(["b1", "b2"]); + expect(container.render(40)).toEqual(["a1", "a2", "b1", "b2"]); + expect(container.getNativeScrollbackLiveRegionStart()).toBe(2); + }); + it("seals the prior block at its final content when finalize+append coalesce (ED3-risk)", () => { riskFlag.eagerEraseScrollbackRisk = true; const container = new TranscriptContainer(); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index f3eceb00a..af11c62ab 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -3,7 +3,11 @@ ## [Unreleased] ### Changed -- Changed foreground-stream rendering on ED3-risk terminals (Ghostty/kitty/Alacritty/VTE/iTerm2 on POSIX) to defer native-scrollback commits: while a turn streams, each frame caps rendered content to the viewport and suppresses `\r\n` scroll growth, so transient output (spinner ticks, partial lines, status rows) never pollutes terminal history. Native scrollback stays marked dirty and is reconciled in a single ED3 (`CSI 3 J`) + re-emit at the next checkpoint (prompt submit) or on an explicit user-input/IME opt-in; an erase is never emitted mid-stream under a possibly-scrolled reader. Non-ED3-risk terminals keep their eager live rebuild. ([#1895](https://github.com/can1357/oh-my-pi/pull/1895)) +- Changed foreground-stream rendering on ED3-risk terminals (Ghostty/kitty/Alacritty/VTE/iTerm2 on POSIX) to defer native-scrollback commits for unpinned transient frames: while a turn streams, generic frames repaint only the viewport and suppress `\r\n` scroll growth, so transient output (spinner ticks, partial lines, status rows) never pollutes terminal history. Components that report a `NativeScrollbackLiveRegion` still commit newly sealed prefix rows while keeping the active suffix dirty for checkpoint replay. Native scrollback is reconciled in a single ED3 (`CSI 3 J`) + re-emit at the next checkpoint (prompt submit) or on an explicit user-input/IME opt-in; an erase is never emitted mid-stream under a possibly-scrolled reader. Non-ED3-risk terminals keep their eager live rebuild. ([#1895](https://github.com/can1357/oh-my-pi/pull/1895)) + +### Fixed + +- Fixed ED3-risk foreground streaming dropping sealed transcript rows above the live block until the next prompt-submit checkpoint, which made scrollback beyond the viewport appear duplicated or out of order. The renderer restores native-scrollback live-region pinning so newly sealed rows are appended once while active live rows remain deferred. ## [15.9.1] - 2026-06-04 ### Fixed diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index f857d9f71..100ab0a0b 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -117,6 +117,21 @@ export interface Component { invalidate(): void; } +/** + * Optional component seam for native-scrollback pinning. A component that + * renders a stable prefix followed by a live/transient suffix reports the local + * line index where that suffix begins after each render. TUI treats that suffix + * — and every root child rendered below it — as not yet safe to commit to native + * scrollback on ED3-risk terminals whose viewport position is unobservable. + */ +export interface NativeScrollbackLiveRegion { + getNativeScrollbackLiveRegionStart(): number | undefined; +} + +function getNativeScrollbackLiveRegionStart(component: Component): number | undefined { + return (component as Component & Partial).getNativeScrollbackLiveRegionStart?.(); +} + /** * Interface for components that can receive focus and display a cursor. * When focused, the component should emit CURSOR_MARKER at the cursor position @@ -317,6 +332,9 @@ export class Container implements Component { * - `historyRebuild`: a geometry change (terminal resize) left native history * wrapped at the old size — clear viewport and scrollback so it rewraps at the * new geometry. Also flushes deferred content-only rewrites. + * - `liveRegionPinned`: ED3-risk/unknown foreground stream with a reported live + * suffix — optionally append newly sealed rows, then repaint the live tail + * without letting transient rows enter native history. * - `viewportRepaint`: rewrite the visible viewport in place. If `appendFrom` * is set, emit those tail rows as scrollback growth first so streaming * output reaches terminal history before the corrected viewport is drawn. @@ -334,6 +352,7 @@ type RenderIntent = | { kind: "sessionReplace" } | { kind: "historyRebuild" } | { kind: "overlayRebuild" } + | { kind: "liveRegionPinned"; appendFrom: number; appendTo: number } | { kind: "viewportRepaint"; appendFrom?: number } | { kind: "deferredShrink"; paddedLength: number } | { kind: "deferredMutation" } @@ -383,6 +402,7 @@ export class TUI extends Container { // Set after a clear+full replay so the next insert-above-suffix frame does // not scroll replayed live chrome (status/editor) into fresh history. #suppressNextSuffixScroll = false; + #nativeScrollbackLiveRegionStart: number | undefined; #nativeScrollbackDirty = false; // Highest `#maxLinesRendered` reached during a foreground tool turn while // intermediate frames were prevented from committing to terminal scrollback. @@ -433,6 +453,25 @@ export class TUI extends Container { this.#showHardwareCursor = showHardwareCursor === undefined ? this.#showHardwareCursor : showHardwareCursor; } + override render(width: number): string[] { + width = Math.max(1, width); + this.#nativeScrollbackLiveRegionStart = undefined; + const lines: string[] = []; + for (const child of this.children) { + const offset = lines.length; + const childLines = child.render(width); + const liveRegionStart = getNativeScrollbackLiveRegionStart(child); + if (liveRegionStart !== undefined) { + const boundedStart = Number.isFinite(liveRegionStart) + ? Math.max(0, Math.min(childLines.length, Math.trunc(liveRegionStart))) + : childLines.length; + this.#nativeScrollbackLiveRegionStart = offset + boundedStart; + } + lines.push(...childLines); + } + return lines; + } + #syncTerminalCursorMode(component: Component | null): void { if (isFocusable(component)) { component.setUseTerminalCursor?.(this.#showHardwareCursor); @@ -1407,6 +1446,7 @@ export class TUI extends Container { visibleOverlayComponents.length > 0, overlayVisibilityReduced, allowUnknownViewportMutation, + this.#nativeScrollbackLiveRegionStart, ); // 3b. Defer scrollback commits during foreground streaming, but only on // ED3-risk terminals whose committed scrollback cannot be rewritten without @@ -1503,6 +1543,18 @@ export class TUI extends Container { }); this.#emitViewportRepaint(lines, width, height, cursorPos); return; + case "liveRegionPinned": + this.#emitLiveRegionPinnedRepaint( + lines, + width, + height, + cursorPos, + intent.appendFrom, + intent.appendTo, + prevViewportTop, + prevHardwareCursorRow, + ); + return; case "viewportRepaint": if (intent.appendFrom !== undefined) { this.#emitAppendTail(lines, intent.appendFrom, height, width, prevViewportTop, prevHardwareCursorRow); @@ -1555,6 +1607,7 @@ export class TUI extends Container { hasVisibleOverlay: boolean, overlayVisibilityReduced: boolean, allowUnknownViewportMutation: boolean, + liveRegionStart: number | undefined, ): RenderIntent { // Initial paint after start(): scrollback must keep its prior shell // content, but the viewport must be cleared so stale rows do not bleed @@ -1587,11 +1640,21 @@ export class TUI extends Container { return { kind: "viewportRepaint" }; } + const liveRegionPinnedIntent = this.#planLiveRegionPinnedRender( + newLines, + height, + liveRegionStart, + eagerEraseScrollbackRisk, + allowUnknownViewportMutation, + widthChanged || heightChanged, + ); + if (liveRegionPinnedIntent) return liveRegionPinnedIntent; + // After foreground tool streaming: when content finally shrinks from the // streaming peak, rebuild with ED 3 to commit the settled state cleanly. // The check uses `#streamingHighWater` (the real peak) rather than - // `#previousLines.length` because streaming capped lines to viewport - // height, so `#previousLines` never reflects the true transcript size. + // `#previousLines.length` because unpinned ED3-risk streaming frames may + // commit only a viewport slice while native history is deferred. if (this.#streamingHighWater > height && newLines.length < this.#streamingHighWater && newLines.length > height) { this.#streamingHighWater = 0; return { kind: "historyRebuild" }; @@ -2104,6 +2167,40 @@ export class TUI extends Container { ); } + #planLiveRegionPinnedRender( + newLines: string[], + height: number, + liveRegionStart: number | undefined, + eagerEraseScrollbackRisk: boolean, + allowUnknownViewportMutation: boolean, + geometryChanged: boolean, + ): RenderIntent | undefined { + // A width/height change reflows the whole terminal: the relative cursor + // positioning this emitter relies on is computed from the pre-resize + // geometry and would land on the wrong rows. Defer to the geometry branch + // (a full reflow rebuild), which is the established behavior for resizes. + if ( + liveRegionStart === undefined || + liveRegionStart >= newLines.length || + !this.#eagerNativeScrollbackRebuild || + !eagerEraseScrollbackRisk || + allowUnknownViewportMutation || + geometryChanged || + isMultiplexerSession() + ) { + return undefined; + } + if (newLines.length <= height && this.#scrollbackHighWater === 0) return undefined; + if (this.#readNativeViewportAtBottom() !== undefined) return undefined; + + this.#markNativeScrollbackDirty(); + const viewportTop = Math.max(0, newLines.length - height); + const sealedEnd = Math.max(0, Math.min(liveRegionStart, newLines.length)); + const appendTo = Math.min(sealedEnd, viewportTop); + const appendFrom = Math.min(this.#scrollbackHighWater, appendTo); + return { kind: "liveRegionPinned", appendFrom, appendTo }; + } + #padDeferredShrinkLines(lines: string[], paddedLength: number): string[] { if (lines.length >= paddedLength) return lines; return [...lines, ...new Array(paddedLength - lines.length).fill("")]; @@ -2276,6 +2373,74 @@ export class TUI extends Container { this.#commit(lines, width, height, viewportTop, toRow); } + /** + * Foreground-stream live-region paint for ED3-risk terminals with an + * unobservable viewport. Commits the newly-sealed chunk to native scrollback + * (so finished blocks stay scrollable) and repaints the live tail in place, + * leaving the transient live region out of saved lines. + * + * Uses only the no-scroll-snap vocabulary of {@link #emitDiff}: relative + * cursor moves, per-line `\x1b[2K`, and `\r\n` to push the sealed chunk into + * history. It deliberately avoids a full-screen erase (`\x1b[2J`) and absolute + * cursor home (`\x1b[H`): on Ghostty those snap a reader scrolled into history + * back to the bottom on every frame. + */ + #emitLiveRegionPinnedRepaint( + lines: string[], + width: number, + height: number, + cursorPos: { row: number; col: number } | null, + appendFrom: number, + appendTo: number, + prevViewportTop: number, + prevHardwareCursorRow: number, + ): void { + this.#fullRedrawCount += 1; + const viewportTop = Math.max(0, lines.length - height); + const boundedAppendTo = Math.max(0, Math.min(appendTo, viewportTop, lines.length)); + const boundedAppendFrom = Math.max(0, Math.min(appendFrom, boundedAppendTo)); + + // Position at the top visible row with a relative move. Terminals clamp the + // hardware cursor to the viewport on resize, so clamp our tracking to match + // before computing the delta (mirrors #emitDiff). + const clampedCursor = Math.min(prevHardwareCursorRow, prevViewportTop + height - 1); + const currentScreenRow = Math.max(0, Math.min(height - 1, clampedCursor - prevViewportTop)); + let buffer = this.#paintBeginSequence; + if (currentScreenRow > 0) buffer += `\x1b[${currentScreenRow}A`; + buffer += "\r"; + + // Write the sealed chunk followed by the full viewport from the top row. + // The first (boundedAppendTo - boundedAppendFrom) rows scroll into native + // history; the trailing `height` rows fill the viewport. Each row clears + // itself with `\x1b[2K` instead of relying on a screen-wide erase. + let wroteLine = false; + for (let i = boundedAppendFrom; i < boundedAppendTo; i++) { + if (wroteLine) buffer += "\r\n"; + buffer += `\x1b[2K${this.#fitLineToWidth(lines[i] ?? "", width)}`; + wroteLine = true; + } + for (let screenRow = 0; screenRow < height; screenRow++) { + if (wroteLine) buffer += "\r\n"; + buffer += `\x1b[2K${this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width)}`; + wroteLine = true; + } + + const viewportBottomRow = viewportTop + height - 1; + const contentBottomRow = Math.min(viewportBottomRow, Math.max(viewportTop, lines.length - 1)); + const parkUp = viewportBottomRow - contentBottomRow; + if (parkUp > 0) buffer += `\x1b[${parkUp}A`; + const { seq, toRow } = this.#cursorControlSequence(cursorPos, lines.length, contentBottomRow); + buffer += seq; + buffer += this.#paintEndSequence; + this.terminal.write(buffer); + + this.#maxLinesRendered = lines.length; + if (boundedAppendTo > this.#scrollbackHighWater) { + this.#scrollbackHighWater = boundedAppendTo; + } + this.#commit(lines, width, height, viewportTop, toRow); + } + /** * Push the appended tail into terminal scrollback by `\r\n`-ing past the * previous viewport bottom. Used as a prefix to {@link #emitViewportRepaint} @@ -2511,9 +2676,11 @@ export class TUI extends Container { const detail = intent.kind === "diff" ? `${intent.kind}(first=${intent.firstChanged}, last=${intent.lastChanged}, appended=${intent.appendedLines})` - : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined - ? `${intent.kind}(appendFrom=${intent.appendFrom})` - : intent.kind; + : intent.kind === "liveRegionPinned" + ? `${intent.kind}(append=${intent.appendFrom}..${intent.appendTo})` + : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined + ? `${intent.kind}(appendFrom=${intent.appendFrom})` + : intent.kind; const msg = `[${new Date().toISOString()}] render: ${detail} (prev=${this.#previousLines.length}, new=${newLength}, height=${height})\n`; fs.appendFileSync(getDebugLogPath(), msg); } diff --git a/packages/tui/test/streaming-scrollback-defer.test.ts b/packages/tui/test/streaming-scrollback-defer.test.ts index 638310566..3e6233e4f 100644 --- a/packages/tui/test/streaming-scrollback-defer.test.ts +++ b/packages/tui/test/streaming-scrollback-defer.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { type Component, TERMINAL, TUI } from "@oh-my-pi/pi-tui"; +import { type Component, type NativeScrollbackLiveRegion, TERMINAL, TUI } from "@oh-my-pi/pi-tui"; import { VirtualTerminal } from "./virtual-terminal"; class LineList implements Component { @@ -20,6 +20,12 @@ class LineList implements Component { } } +class LiveLineList extends LineList implements NativeScrollbackLiveRegion { + getNativeScrollbackLiveRegionStart(): number | undefined { + return 0; + } +} + async function settle(term: VirtualTerminal): Promise { await Bun.sleep(20); await term.flush(); @@ -66,6 +72,48 @@ function rows(prefix: string, count: number): string[] { } describe("streaming scrollback defer", () => { + it("keeps sealed prefix scrollable while deferring live-region rows on ED3-risk terminals", async () => { + if (process.platform === "win32") return; + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(20, 4); + overrideProbe(term, undefined); + const tui = new TUI(term); + const sealed = new LineList(rows("prior-", 12)); + const live = new LiveLineList([]); + + try { + tui.addChild(sealed); + tui.addChild(live); + tui.start(); + await settle(term); + + const writes = capture(term); + tui.setEagerNativeScrollbackRebuild(true); + + live.setLines(rows("think-", 6)); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect(term.getScrollBuffer().map(line => line.trimEnd())).toEqual([ + ...rows("prior-", 12), + ...rows("think-", 6).slice(-4), + ]); + + live.setLines(rows("think-", 8)); + tui.requestRender(); + await settle(term); + + const buffer = term.getScrollBuffer().map(line => line.trimEnd()); + expect(eraseScrollbackCount(writes)).toBe(0); + expect(buffer.filter(line => line.startsWith("prior-"))).toEqual(rows("prior-", 12)); + expect(buffer.slice(-4)).toEqual(rows("think-", 8).slice(-4)); + } finally { + tui.stop(); + } + }); + }); + it("defers scrollback growth during eager streaming on ED3-risk and reconciles at the checkpoint", async () => { if (process.platform === "win32") return; await withTerminalRisk(true, async () => { From 2dab082a68e9d43326f68bf00558b6b5a6db6cc1 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:29:27 +0200 Subject: [PATCH 42/48] fix(tui): restored live block boundary reporting to TUI - Appended newly sealed transcript blocks to native scrollback once. - Deferred only the active live block during ED3-risk streaming. - Hardened snapshot TTL parsing to handle whitespace-only env values. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/sdk.ts | 6 ++++-- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 792cb455e..967b9b7f6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -29,6 +29,7 @@ ### Fixed - Fixed a streamed assistant message freezing at a partial prefix (e.g. only "Nat" of "Natives built, now…") on ED3-risk terminals (Ghostty/kitty/iTerm2/Alacritty), with the final text appearing only after a resize. `TranscriptContainer` freezes each non-live block by replaying its last live render, but render coalescing can finalize a block's content and append the next block within the same throttled frame — so the block was sealed at its stale mid-stream snapshot and never repainted until the next `thaw`. The block that was live on the previous render is now recomputed once on the live→frozen transition, sealing it at its final content. +- Fixed `TranscriptContainer` reporting the live block boundary to the TUI again, so ED3-risk foreground streaming can append newly sealed transcript blocks to native scrollback once while deferring only the active live block. - Fixed ACP/RPC stdio startup so protocol frames are no longer consumed as one-shot piped prompt input before the JSON-RPC transport starts. - Fixed `omp completions` to await the completion script write before exiting. diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 63a82447f..61361e883 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -429,8 +429,10 @@ function getDefaultAgentDir(): string { function resolveSnapshotTtlMs(): number { const raw = process.env.OMP_AUTH_BROKER_SNAPSHOT_TTL_MS; - if (!raw) return DEFAULT_SNAPSHOT_CACHE_TTL_MS; - const ttlMs = Number(raw); + if (raw === undefined) return DEFAULT_SNAPSHOT_CACHE_TTL_MS; + const value = raw.trim(); + if (value === "") return DEFAULT_SNAPSHOT_CACHE_TTL_MS; + const ttlMs = Number(value); if (Number.isFinite(ttlMs) && ttlMs >= 0) return ttlMs; logger.warn("Invalid OMP_AUTH_BROKER_SNAPSHOT_TTL_MS; using default", { value: raw }); return DEFAULT_SNAPSHOT_CACHE_TTL_MS; From ac304eacdcbd61656a161a7bf4adffd85a729756 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 09:29:30 +0000 Subject: [PATCH 43/48] fix(sdk): scoped async job snapshots to sessions AgentSession now stores the same scoped AsyncJobManager reference that tools receive: owning top-level sessions use their constructed manager, subagents inherit the parent's manager, and secondary in-process top-level sessions get no manager when a singleton is already live. getAsyncJobSnapshot and ACP delivery drains now use that scoped manager instead of AsyncJobManager.instance(), so secondary sessions cannot report or drain the primary session's background jobs. The regression test covers a secondary session created while the primary has a Main-owned running job. --- packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/sdk.ts | 9 +++-- .../coding-agent/src/session/agent-session.ts | 38 +++++++++++++------ packages/coding-agent/src/tools/index.ts | 3 +- .../test/agent-session-concurrent.test.ts | 1 + .../sdk-async-job-manager-singleton.test.ts | 7 +++- 6 files changed, 43 insertions(+), 17 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 992490086..948b44080 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. The secondary now shares the live singleton instead of clobbering it, and its `cancelOwnAsyncJobs` dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools now resolve the manager through their `ToolSession` rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager (which would route completions into the wrong `yieldQueue`); subagents still inherit the parent's manager via their `ToolSession.asyncJobManager` ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). +- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. Secondary sessions now leave the live singleton untouched, and their dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools and session job snapshots now resolve the manager through session-scoped async manager wiring rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager or report the owning session's jobs; subagents still inherit the parent's manager via their scoped async manager ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). ## [15.9.1] - 2026-06-04 diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 1cf415733..0f5d73e9b 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1196,6 +1196,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }) : undefined; + const scopedAsyncJobManager = asyncJobManager ?? (options.parentTaskPrefix ? AsyncJobManager.instance() : undefined); + const agentRegistry = options.agentRegistry ?? AgentRegistry.global(); const resolvedAgentId = options.agentId ?? options.parentTaskPrefix ?? MAIN_AGENT_ID; const resolvedAgentDisplayName = @@ -1301,9 +1303,9 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // completions still flow into the spawning conversation's yieldQueue. // Secondary in-process top-level sessions (no parentTaskPrefix, no // constructed manager because the singleton was already installed) leave - // this undefined so tools refuse async work instead of silently routing it - // into the owning session (issue #1923). - asyncJobManager: asyncJobManager ?? (options.parentTaskPrefix ? AsyncJobManager.instance() : undefined), + // this undefined so tools and session job snapshots refuse async work + // instead of silently routing into the owning session (issue #1923). + asyncJobManager: scopedAsyncJobManager, }; // Wire process-wide internal URL singletons owned by their real classes. @@ -2060,6 +2062,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // AsyncJobManager on teardown; subagents inherit the parent's and // **MUST NOT** tear it down. ownedAsyncJobManager: asyncJobManager, + asyncJobManager: scopedAsyncJobManager, scopedModels: options.scopedModels, promptTemplates, slashCommands, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index d2c19fa93..f5e125cc3 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -366,6 +366,15 @@ export interface AgentSessionConfig { * **MUST NOT** dispose it on their own teardown. */ ownedAsyncJobManager?: AsyncJobManager; + /** + * AsyncJobManager reachable by this session for scoped job actions. + * + * Top-level owners receive their own manager, subagents receive the inherited + * parent manager, and secondary in-process top-level sessions receive + * `undefined` so job snapshots and ACP drains cannot observe the primary's + * state. + */ + asyncJobManager?: AsyncJobManager; /** Agent identity (registry id like "Main" or "Alice") used for IRC routing. */ agentId?: string; /** Shared agent registry (for forwarding IRC observations to the main session UI). */ @@ -890,6 +899,14 @@ export class AgentSession { * this undefined and **MUST NOT** dispose the global instance on teardown. */ readonly #ownedAsyncJobManager: AsyncJobManager | undefined; + /** + * AsyncJobManager scoped to this session for introspection/cancellation. + * + * This differs from `#ownedAsyncJobManager`: subagents can inherit a parent + * manager for their own owner id, while secondary top-level sessions are left + * undefined to avoid reading the primary's jobs. + */ + readonly #asyncJobManager: AsyncJobManager | undefined; #pendingPythonMessages: PythonExecutionMessage[] = []; #activeEvalExecutions = new Set>(); #evalExecutionDisposing = false; @@ -1080,6 +1097,7 @@ export class AgentSession { this.#evalKernelOwnerId = config.evalKernelOwnerId ?? `agent-session:${Snowflake.next()}`; this.#parentEvalSessionId = config.parentEvalSessionId; this.#ownedAsyncJobManager = config.ownedAsyncJobManager; + this.#asyncJobManager = config.asyncJobManager ?? config.ownedAsyncJobManager; this.#scopedModels = config.scopedModels ?? []; if (config.thinkingLevel === AUTO_THINKING) { // `auto` is session-level: keep the flag and show a provisional concrete @@ -1373,7 +1391,7 @@ export class AgentSession { } getAsyncJobSnapshot(options?: { recentLimit?: number }): AsyncJobSnapshot | null { - const manager = AsyncJobManager.instance(); + const manager = this.#asyncJobManager; if (!manager) return null; const ownerFilter = this.#agentId ? { ownerId: this.#agentId } : undefined; const running = manager.getRunningJobs(ownerFilter).map(job => ({ @@ -1399,20 +1417,18 @@ export class AgentSession { * transitions (newSession, switchSession, handoff, dispose) so a subagent * cleans up its own background work without touching its parent's jobs. * - * Cancellation runs against the manager THIS session owns. Subagents have - * unique agent ids and may still reach the inherited singleton (which is - * the parent's manager) to clean up their own scoped jobs. A secondary - * in-process top-level session — which inherits the singleton without - * owning it AND defaults to `MAIN_AGENT_ID` — must NOT cancel via the - * inherited singleton, or it would tear down the owning primary session's - * bash/task jobs at dispose time (issue #1923). + * Cancellation runs against this session's scoped manager. Subagents have + * unique agent ids and inherit the parent's manager to clean up their own + * jobs. A secondary in-process top-level session gets no scoped manager, + * because it defaults to `MAIN_AGENT_ID`; reaching through the global + * singleton would tear down the owning primary session's bash/task jobs at + * dispose time (issue #1923). * * No-op when no manager is reachable or this session has no agent id. */ #cancelOwnAsyncJobs(): void { if (!this.#agentId) return; - const manager = - this.#ownedAsyncJobManager ?? (this.#agentId === MAIN_AGENT_ID ? undefined : AsyncJobManager.instance()); + const manager = this.#asyncJobManager; manager?.cancelAll({ ownerId: this.#agentId }); } @@ -3080,7 +3096,7 @@ export class AgentSession { } async drainAsyncJobDeliveriesForAcp(options?: { timeoutMs?: number }): Promise { - const manager = AsyncJobManager.instance(); + const manager = this.#asyncJobManager; if (!manager) return false; const ownerFilter = this.#agentId ? { ownerId: this.#agentId } : undefined; const before = manager.getDeliveryState(ownerFilter); diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 0043fba0b..baf336e4c 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -2,6 +2,7 @@ import type { InMemorySnapshotStore } from "@oh-my-pi/hashline"; import type { AgentTelemetryConfig, AgentTool } from "@oh-my-pi/pi-agent-core"; import type { ToolChoice } from "@oh-my-pi/pi-ai"; import { logger } from "@oh-my-pi/pi-utils"; +import type { AsyncJobManager } from "../async/job-manager"; import type { PromptTemplate } from "../config/prompt-templates"; import type { Settings } from "../config/settings"; import { EditTool } from "../edit"; @@ -197,7 +198,7 @@ export interface ToolSession { * Tools MUST use this instead of `AsyncJobManager.instance()` so a secondary * session never borrows the owning session's manager by accident. */ - asyncJobManager?: import("../async/job-manager").AsyncJobManager; + asyncJobManager?: AsyncJobManager; /** MCP manager visible to subagents without relying on the process-global singleton. */ mcpManager?: MCPManager; /** Local protocol root to propagate to nested subagents and eval-created agents. */ diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index d7bb707e5..300cc4272 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -548,6 +548,7 @@ describe("AgentSession concurrent prompt guard", () => { settings, modelRegistry, agentId: "acp-session-b", + asyncJobManager, }); session = new AgentSession({ agent: agentA, diff --git a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts index 5ece9eff8..420a7be1e 100644 --- a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts +++ b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts @@ -89,9 +89,14 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () => }, { ownerId: "Main" }, ); + expect(primary.getAsyncJobSnapshot()?.running.some(job => job.id === jobId)).toBe(true); const secondary = await spawnTopLevelSession(); - await secondary.dispose(); + try { + expect(secondary.getAsyncJobSnapshot()).toBeNull(); + } finally { + await secondary.dispose(); + } const job = primaryManager!.getJob(jobId); expect(job?.status).toBe("running"); From be9e5218edc60378ac339116a3a09d3f6b542267 Mon Sep 17 00:00:00 2001 From: roboomp Date: Fri, 5 Jun 2026 09:34:10 +0000 Subject: [PATCH 44/48] fix(sdk): cleared async singleton after startup failures If createAgentSession failed after installing a newly created AsyncJobManager but before AgentSession took ownership, the process-global singleton stayed installed. The new singleton guard then caused the next top-level session to skip constructing a scoped manager, disabling async bash/task support. The startup-error cleanup now clears the singleton only when it still points at the newly created manager, disposes that manager, and then continues the existing registry/kernel cleanup. The regression test forces a startup failure after singleton installation and verifies the next top-level session can create and use its own async manager. --- packages/coding-agent/CHANGELOG.md | 2 +- packages/coding-agent/src/sdk.ts | 6 ++++ .../sdk-async-job-manager-singleton.test.ts | 36 +++++++++++++++++++ 3 files changed, 43 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 948b44080..5c674773d 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. Secondary sessions now leave the live singleton untouched, and their dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools and session job snapshots now resolve the manager through session-scoped async manager wiring rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager or report the owning session's jobs; subagents still inherit the parent's manager via their scoped async manager ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). +- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. Secondary sessions now leave the live singleton untouched, and their dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools and session job snapshots now resolve the manager through session-scoped async manager wiring rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager or report the owning session's jobs; subagents still inherit the parent's manager via their scoped async manager. Startup failures after a top-level session installs its manager now clear and dispose that manager before the next session decides whether it can create its own ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). ## [15.9.1] - 2026-06-04 diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 0f5d73e9b..7a6e69802 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -2276,6 +2276,12 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} await session.dispose(); } else { if (hasRegistered) agentRegistry.unregister(resolvedAgentId); + if (asyncJobManager) { + if (AsyncJobManager.instance() === asyncJobManager) { + AsyncJobManager.setInstance(undefined); + } + await asyncJobManager.dispose({ timeoutMs: 3_000 }); + } await disposeKernelSessionsByOwner(evalKernelOwnerId); } } catch (cleanupError) { diff --git a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts index 420a7be1e..0d4a8029d 100644 --- a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts +++ b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts @@ -133,4 +133,40 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () => await primary.dispose(); } }); + + it("clears a manager installed before a top-level session startup failure takes ownership", async () => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-sdk-async-startup-failure-${Snowflake.next()}-`)); + tempDirs.push(tempDir); + const cwd = path.join(tempDir, `project-${Snowflake.next()}`); + const agentDir = path.join(tempDir, "agent"); + fs.mkdirSync(cwd, { recursive: true }); + + await expect( + createAgentSession({ + cwd, + agentDir, + settings: Settings.isolated({ "bash.autoBackground.enabled": true }), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + systemPrompt: () => { + throw new Error("forced startup failure"); + }, + }), + ).rejects.toThrow("forced startup failure"); + + expect(AsyncJobManager.instance()).toBeUndefined(); + + const replacement = await spawnTopLevelSession(); + try { + expect(AsyncJobManager.instance()).toBeDefined(); + expect(replacement.getAsyncJobSnapshot()).not.toBeNull(); + } finally { + await replacement.dispose(); + } + }); }); From eb8e4f765720f4b2d860c4b5be55af6cc5a6e217 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:36:13 +0200 Subject: [PATCH 45/48] feat(task): added read-summarize override for subagents - Parsed `read-summarize` frontmatter into `readSummarize` field. - Applied `read.summarize.enabled: false` override on isolated subagent settings. - Disabled summarization for `explore` and `librarian` agents. --- docs/task-agent-discovery.md | 3 ++- .../coding-agent/src/discovery/helpers.ts | 4 +++- .../src/prompts/agents/explore.md | 1 + .../src/prompts/agents/librarian.md | 1 + packages/coding-agent/src/task/executor.ts | 11 +++++++-- packages/coding-agent/src/task/types.ts | 2 ++ .../test/discovery/agent-fields.test.ts | 23 +++++++++++++++++++ 7 files changed, 41 insertions(+), 4 deletions(-) diff --git a/docs/task-agent-discovery.md b/docs/task-agent-discovery.md index 502bb482e..82156f310 100644 --- a/docs/task-agent-discovery.md +++ b/docs/task-agent-discovery.md @@ -24,7 +24,7 @@ It covers runtime behavior as implemented today, including precedence, invalid-d Task agents normalize into `AgentDefinition` (`src/task/types.ts`): - `name`, `description`, `systemPrompt` (required for a valid loaded agent) -- optional `tools`, `spawns`, `model`, `thinkingLevel`, `output`, `blocking` +- optional `tools`, `spawns`, `model`, `thinkingLevel`, `output`, `blocking`, `autoloadSkills`, `readSummarize` - `source`: `"bundled" | "user" | "project"` - optional `filePath` @@ -35,6 +35,7 @@ Parsing comes from frontmatter via `parseAgentFields()` (`src/discovery/helpers. - `spawns` accepts `*`, CSV, or array - backward-compat behavior: if `spawns` missing but `tools` includes `task`, `spawns` becomes `*` - `output` is passed through as opaque schema data +- `read-summarize: false` (parsed as `readSummarize`) forces the subagent's `read` tool to return verbatim file content instead of structural summaries — `runSubprocess` applies it as a `read.summarize.enabled: false` override on the subagent's isolated settings (`src/task/executor.ts`). `explore` and `librarian` ship with it disabled. Defaults to enabled when the field is absent. ## Bundled agents diff --git a/packages/coding-agent/src/discovery/helpers.ts b/packages/coding-agent/src/discovery/helpers.ts index 72e200c67..ac6e92ad0 100644 --- a/packages/coding-agent/src/discovery/helpers.ts +++ b/packages/coding-agent/src/discovery/helpers.ts @@ -212,6 +212,7 @@ export interface ParsedAgentFields { output?: unknown; thinkingLevel?: ThinkingLevel; autoloadSkills?: string[]; + readSummarize?: boolean; blocking?: boolean; } @@ -265,10 +266,11 @@ export function parseAgentFields(frontmatter: Record): ParsedAg const thinkingLevel = parseThinkingLevel(rawThinkingLevel); const model = parseModelList(frontmatter.model); const blocking = parseBoolean(frontmatter.blocking); + const readSummarize = parseBoolean(frontmatter.readSummarize); const autoloadSkills = parseArrayOrCSV(frontmatter.autoloadSkills) ?.map(s => s.trim()) .filter(Boolean); - return { name, description, tools, spawns, model, output, thinkingLevel, blocking, autoloadSkills }; + return { name, description, tools, spawns, model, output, thinkingLevel, blocking, autoloadSkills, readSummarize }; } async function globIf( diff --git a/packages/coding-agent/src/prompts/agents/explore.md b/packages/coding-agent/src/prompts/agents/explore.md index 6ba32f97d..d7ceb117e 100644 --- a/packages/coding-agent/src/prompts/agents/explore.md +++ b/packages/coding-agent/src/prompts/agents/explore.md @@ -4,6 +4,7 @@ description: Fast read-only codebase scout returning compressed context for hand tools: read, search, find, web_search model: pi/smol thinking-level: med +read-summarize: false output: properties: summary: diff --git a/packages/coding-agent/src/prompts/agents/librarian.md b/packages/coding-agent/src/prompts/agents/librarian.md index a805c886c..766aaecfa 100644 --- a/packages/coding-agent/src/prompts/agents/librarian.md +++ b/packages/coding-agent/src/prompts/agents/librarian.md @@ -4,6 +4,7 @@ description: Researches external libraries and APIs by reading source code. Retu tools: read, search, find, bash, lsp, web_search, ast_grep model: pi/smol thinking-level: minimal +read-summarize: false output: properties: answer: diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index da5bada9d..c884bb468 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -531,7 +531,10 @@ function createMCPProxyTools(mcpManager: MCPManager): CustomTool[] { }); } -function createSubagentSettings(baseSettings: Settings): Settings { +function createSubagentSettings( + baseSettings: Settings, + overrides?: Partial>, +): Settings { const snapshot: Partial> = {}; for (const key of Object.keys(SETTINGS_SCHEMA) as SettingPath[]) { snapshot[key] = baseSettings.get(key); @@ -545,6 +548,7 @@ function createSubagentSettings(baseSettings: Settings): Settings { // the parent task approval is the authorization boundary. Use yolo mode // to preserve unattended subagent execution. User `tools.approval` policies still apply. "tools.approvalMode": "yolo", + ...overrides, }); } @@ -619,7 +623,10 @@ export async function runSubprocess(options: ExecutorOptions): Promise { expect(fields).toBeDefined(); expect(fields?.autoloadSkills).toBeUndefined(); }); + + test("parses readSummarize from boolean frontmatter", () => { + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: false })?.readSummarize).toBe( + false, + ); + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: true })?.readSummarize).toBe( + true, + ); + }); + + test("parses readSummarize from string frontmatter", () => { + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: "false" })?.readSummarize).toBe( + false, + ); + }); + + test("ignores invalid readSummarize values", () => { + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: "nope" })?.readSummarize).toBeUndefined(); + }); + + test("returns undefined readSummarize when field absent", () => { + expect(parseAgentFields({ name: "explore", description: "desc" })?.readSummarize).toBeUndefined(); + }); }); From 88703a4ebba410243ae6fde6ef925c8ab6b11ad0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:41:07 +0200 Subject: [PATCH 46/48] feat(dry-balance): added live bench mode for OAuth accounts - Added `getOAuthAccesses` to resolve each stored credential once. - Sent one live request per account, reporting TTFT and TPS. - Streamed per-account progress with interactive status lines. --- packages/ai/src/auth-storage.ts | 83 +++ .../coding-agent/src/cli/dry-balance-cli.ts | 503 +++++++++++++++++- .../coding-agent/src/commands/dry-balance.ts | 3 + .../src/prompts/dry-balance-bench.md | 8 + .../tools/task-agent-capabilities.test.ts | 10 + 5 files changed, 591 insertions(+), 16 deletions(-) create mode 100644 packages/coding-agent/src/prompts/dry-balance-bench.md diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 330227af9..980323b74 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -557,11 +557,25 @@ type OAuthResolutionResult = { apiKey: string; credential: OAuthCredential }; */ export interface OAuthAccess { accessToken: string; + credentialId?: number; accountId?: string; email?: string; projectId?: string; enterpriseUrl?: string; } + +export interface OAuthAccessFailure { + credentialId?: number; + accountId?: string; + email?: string; + projectId?: string; + enterpriseUrl?: string; + error: string; +} + +export type OAuthAccessResolution = + | ({ ok: true } & OAuthAccess) + | ({ ok: false } & OAuthAccessFailure); export interface InvalidateCredentialMatchingOptions { signal?: AbortSignal; sessionId?: string; @@ -3494,6 +3508,75 @@ export class AuthStorage { }; } + /** + * Resolve every stored OAuth credential for `provider` independently. + * + * Refreshes credentials through the same broker/local path as + * {@link AuthStorage.getOAuthAccess}, but does not rank, round-robin, or + * stop after the first usable account. Intended for diagnostics that must + * exercise each stored account exactly once. + */ + async getOAuthAccesses(provider: string, options?: AuthApiKeyOptions): Promise { + if (this.#runtimeOverrides.has(provider) || this.#configOverrides.has(provider)) { + return []; + } + const providerKey = this.#getProviderTypeKey(provider, "oauth"); + const selections = this.#getStoredCredentials(provider) + .map((entry, index) => ({ credentialId: entry.id, credential: entry.credential, index })) + .filter( + (entry): entry is { credentialId: number; credential: OAuthCredential; index: number } => + entry.credential.type === "oauth", + ); + return Promise.all( + selections.map(async (selection): Promise => { + try { + const resolved = await this.#tryOAuthCredential( + provider, + { credential: selection.credential, index: selection.index }, + providerKey, + undefined, + options, + { + checkUsage: false, + allowBlocked: true, + }, + ); + if (!resolved) { + return { + ok: false, + credentialId: selection.credentialId, + accountId: selection.credential.accountId, + email: selection.credential.email, + projectId: selection.credential.projectId, + enterpriseUrl: selection.credential.enterpriseUrl, + error: "OAuth access unavailable", + }; + } + const { credential } = resolved; + return { + ok: true, + credentialId: selection.credentialId, + accessToken: credential.access, + accountId: credential.accountId, + email: credential.email, + projectId: credential.projectId, + enterpriseUrl: credential.enterpriseUrl, + }; + } catch (error) { + return { + ok: false, + credentialId: selection.credentialId, + accountId: selection.credential.accountId, + email: selection.credential.email, + projectId: selection.credential.projectId, + enterpriseUrl: selection.credential.enterpriseUrl, + error: error instanceof Error ? error.message : String(error), + }; + } + }), + ); + } + #extractStructuredApiKeyToken(apiKey: string): string | undefined { if (!apiKey.startsWith("{")) return undefined; try { diff --git a/packages/coding-agent/src/cli/dry-balance-cli.ts b/packages/coding-agent/src/cli/dry-balance-cli.ts index 1e16d5237..3fa3b95fd 100644 --- a/packages/coding-agent/src/cli/dry-balance-cli.ts +++ b/packages/coding-agent/src/cli/dry-balance-cli.ts @@ -1,5 +1,17 @@ -import type { Api, Model, OAuthAccess } from "@oh-my-pi/pi-ai"; -import { getProjectDir } from "@oh-my-pi/pi-utils"; +import type { + Api, + AssistantMessage, + AssistantMessageEvent, + AssistantMessageEventStream, + Context, + Model, + OAuthAccess, + OAuthAccessResolution, + SimpleStreamOptions, +} from "@oh-my-pi/pi-ai"; +import { streamSimple } from "@oh-my-pi/pi-ai"; +import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils"; +import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui"; import chalk from "chalk"; import { ModelRegistry } from "../config/model-registry"; import { @@ -11,9 +23,16 @@ import { } from "../config/model-resolver"; import { Settings } from "../config/settings"; import { discoverAuthStorage } from "../sdk"; +import dryBalanceBenchPrompt from "../prompts/dry-balance-bench.md" with { type: "text" }; const DEFAULT_SAMPLE_COUNT = 100; const DEFAULT_CONCURRENCY = 32; +const BENCH_MAX_TOKENS = 512; +const BENCH_RENDER_INTERVAL_MS = 80; +const BENCH_ACCOUNT_WIDTH = 60; +const BENCH_ERROR_WIDTH = 110; +const BENCH_SPINNER_FRAMES = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"] as const; +const DRY_BALANCE_BENCH_PROMPT = dryBalanceBenchPrompt.trim(); export interface DryBalanceCommandArgs { model?: string; @@ -22,6 +41,7 @@ export interface DryBalanceCommandArgs { count?: number; concurrency?: number; json?: boolean; + bench?: boolean; }; } @@ -37,6 +57,7 @@ export interface DryBalanceAuthStorage { sessionId?: string, options?: DryBalanceAuthOptions, ): Promise; + getOAuthAccesses?(provider: string, options?: DryBalanceAuthOptions): Promise; } export interface DryBalanceModelRegistry { @@ -67,6 +88,37 @@ export interface DryBalanceFailureStat { percent: number; } +export interface DryBalanceBenchSuccessResult { + ok: true; + account: string; + ttftMs: number; + durationMs: number; + outputTokens: number; + tokensPerSecond: number; +} + +export interface DryBalanceBenchFailureResult { + ok: false; + account?: string; + error: string; +} + +export type DryBalanceBenchResult = DryBalanceBenchSuccessResult | DryBalanceBenchFailureResult; + +export interface DryBalanceBenchSummary { + total: number; + success: { + total: number; + averageTtftMs: number | null; + averageTokensPerSecond: number | null; + }; + failure: { + total: number; + reasons: DryBalanceFailureStat[]; + }; + results: DryBalanceBenchResult[]; +} + export interface DryBalanceSummary { model: string; provider: string; @@ -80,14 +132,25 @@ export interface DryBalanceSummary { total: number; reasons: DryBalanceFailureStat[]; }; + bench?: DryBalanceBenchSummary; } -interface DryBalanceDependencies { +type DryBalanceStreamSimple = ( + model: Model, + context: Context, + options?: SimpleStreamOptions, +) => AssistantMessageEventStream; + +export interface DryBalanceDependencies { createRuntime?: () => Promise; randomSessionId?: () => string; writeStdout?: (text: string) => void; writeStderr?: (text: string) => void; setExitCode?: (code: number) => void; + streamSimple?: DryBalanceStreamSimple; + now?: () => number; + stdoutIsTTY?: boolean; + stderrIsTTY?: boolean; } type DryBalanceAttemptResult = @@ -100,6 +163,30 @@ type DryBalanceAttemptResult = reason: string; }; +type DryBalanceBenchProgressStatus = + | { state: "waiting" } + | { state: "running"; account: string } + | { state: "success"; result: DryBalanceBenchSuccessResult } + | { state: "failure"; result: DryBalanceBenchFailureResult }; + +interface DryBalanceBenchProgressSink { + markRunning(index: number, account: string): void; + complete(index: number, result: DryBalanceBenchResult): void; + close(): void; +} + +type DryBalanceBenchTarget = + | { + ok: true; + account: string; + accessToken: string; + } + | { + ok: false; + account: string; + error: string; + }; + function normalizePositiveInteger(name: string, value: number | undefined, fallback: number): number { const resolved = value ?? fallback; if (!Number.isInteger(resolved) || resolved <= 0) { @@ -108,6 +195,297 @@ function normalizePositiveInteger(name: string, value: number | undefined, fallb return resolved; } +function getErrorMessage(error: unknown): string { + if (error instanceof Error && error.message) return error.message; + const message = String(error); + return message ? message : "Unknown error"; +} + +function extractAccount(access: { + email?: string; + accountId?: string; + projectId?: string; + enterpriseUrl?: string; +}): string { + return access.email ?? access.accountId ?? access.projectId ?? access.enterpriseUrl ?? "(unknown oauth account)"; +} + +function getBenchTargetKey(access: { + credentialId?: number; + email?: string; + accountId?: string; + projectId?: string; + enterpriseUrl?: string; + accessToken?: string; +}): string { + return ( + access.email ?? + access.accountId ?? + access.projectId ?? + access.enterpriseUrl ?? + (access.credentialId === undefined ? access.accessToken : `credential:${access.credentialId}`) ?? + "(unknown oauth account)" + ); +} + +function sanitizeBenchText(text: string, width: number): string { + return truncateToWidth(replaceTabs(text).replace(/\r?\n/g, " "), width); +} + +function formatBenchIndex(index: number, total: number): string { + return `#${String(index + 1).padStart(String(total).length, "0")}`; +} + +function formatBenchAccount(account: string | undefined): string { + return account ? sanitizeBenchText(account, BENCH_ACCOUNT_WIDTH) : chalk.dim("(no account)"); +} + +function formatBenchDuration(ms: number): string { + return formatDuration(Math.max(0, Math.round(ms))); +} + +function formatBenchTps(tokensPerSecond: number): string { + return `${tokensPerSecond.toFixed(1)}/s`; +} + +function isBenchSuccess(result: DryBalanceBenchResult): result is DryBalanceBenchSuccessResult { + return result.ok; +} + +function isBenchFirstTokenEvent(event: AssistantMessageEvent): boolean { + switch (event.type) { + case "text_delta": + case "thinking_delta": + case "toolcall_delta": + return event.delta.length > 0; + case "text_end": + case "thinking_end": + return event.content.length > 0; + default: + return false; + } +} + +function resolveBenchMaxTokens(model: Model): number { + return Number.isFinite(model.maxTokens) && model.maxTokens > 0 + ? Math.min(BENCH_MAX_TOKENS, model.maxTokens) + : BENCH_MAX_TOKENS; +} + +function normalizeBenchMs(value: number): number { + return Number.isFinite(value) && value > 0 ? value : 0; +} + +function renderBenchResultLine(index: number, total: number, result: DryBalanceBenchResult): string { + const prefix = formatBenchIndex(index, total); + if (result.ok) { + return `${chalk.green("✓")} ${prefix} ${formatBenchAccount(result.account)} ${chalk.dim("TTFT")} ${formatBenchDuration( + result.ttftMs, + )} ${chalk.dim("TPS")} ${formatBenchTps(result.tokensPerSecond)}`; + } + return `${chalk.red("✗")} ${prefix} ${formatBenchAccount(result.account)} ${chalk.red( + sanitizeBenchText(result.error, BENCH_ERROR_WIDTH), + )}`; +} + +function renderBenchStatusLine(status: DryBalanceBenchProgressStatus, index: number, total: number, frame: number): string { + const prefix = formatBenchIndex(index, total); + switch (status.state) { + case "waiting": + return `${chalk.dim("○")} ${prefix} ${chalk.dim("waiting")}`; + case "running": { + const spinner = BENCH_SPINNER_FRAMES[frame % BENCH_SPINNER_FRAMES.length] ?? "*"; + return `${chalk.yellow(spinner)} ${prefix} ${formatBenchAccount(status.account)} ${chalk.dim("sending request")}`; + } + case "success": + return renderBenchResultLine(index, total, status.result); + case "failure": + return renderBenchResultLine(index, total, status.result); + } +} + +function createBenchProgressSink( + total: number, + write: (text: string) => void, + interactive: boolean, +): DryBalanceBenchProgressSink { + const statuses: DryBalanceBenchProgressStatus[] = Array.from({ length: total }, () => ({ state: "waiting" })); + if (!interactive) { + return { + markRunning(index, account) { + statuses[index] = { state: "running", account }; + write(`${renderBenchStatusLine(statuses[index], index, total, 0)}\n`); + }, + complete(index, result) { + statuses[index] = result.ok ? { state: "success", result } : { state: "failure", result }; + write(`${renderBenchResultLine(index, total, result)}\n`); + }, + close() {}, + }; + } + + let frame = 0; + let lineCount = 0; + let timer: NodeJS.Timeout | undefined; + const render = (): void => { + const lines = [ + chalk.bold("bench requests"), + ...statuses.map((status, index) => renderBenchStatusLine(status, index, total, frame)), + ]; + if (lineCount > 0) write(`\x1b[${lineCount}A`); + write(`${lines.map(line => `\x1b[2K${line}`).join("\n")}\n`); + lineCount = lines.length; + }; + render(); + timer = setInterval(() => { + frame += 1; + render(); + }, BENCH_RENDER_INTERVAL_MS); + timer.unref?.(); + return { + markRunning(index, account) { + statuses[index] = { state: "running", account }; + render(); + }, + complete(index, result) { + statuses[index] = result.ok ? { state: "success", result } : { state: "failure", result }; + render(); + }, + close() { + if (timer) { + clearInterval(timer); + timer = undefined; + } + render(); + }, + }; +} + +async function runBenchRequest( + model: Model, + sessionId: string, + account: string, + accessToken: string, + streamFn: DryBalanceStreamSimple, + now: () => number, +): Promise { + const startedAt = now(); + let firstTokenAt: number | undefined; + try { + const context: Context = { + messages: [ + { + role: "user", + content: DRY_BALANCE_BENCH_PROMPT, + timestamp: Date.now(), + attribution: "user", + }, + ], + }; + const stream = streamFn(model, context, { + apiKey: accessToken, + sessionId, + maxTokens: resolveBenchMaxTokens(model), + temperature: 0.2, + disableReasoning: true, + hideThinkingSummary: true, + }); + let message: AssistantMessage | undefined; + for await (const event of stream) { + if (firstTokenAt === undefined && isBenchFirstTokenEvent(event)) { + firstTokenAt = now(); + } + if (event.type === "error") { + return { ok: false, account, error: event.error.errorMessage ?? "request failed" }; + } + if (event.type === "done") { + message = event.message; + } + } + message ??= await stream.result(); + if (message.stopReason === "error" || message.errorMessage) { + return { ok: false, account, error: message.errorMessage ?? "request failed" }; + } + const durationMs = normalizeBenchMs(message.duration ?? now() - startedAt); + const ttftMs = normalizeBenchMs(message.ttft ?? (firstTokenAt === undefined ? durationMs : firstTokenAt - startedAt)); + const outputTokens = Number.isFinite(message.usage.output) && message.usage.output > 0 ? message.usage.output : 0; + const tokensPerSecond = durationMs > 0 ? (outputTokens * 1000) / durationMs : 0; + return { + ok: true, + account, + ttftMs, + durationMs, + outputTokens, + tokensPerSecond, + }; + } catch (error) { + return { ok: false, account, error: getErrorMessage(error) }; + } +} + +async function resolveBenchTargets( + model: Model, + authStorage: DryBalanceAuthStorage, +): Promise { + const resolved = authStorage.getOAuthAccesses + ? await authStorage.getOAuthAccesses(model.provider, { + baseUrl: model.baseUrl, + modelId: model.id, + }) + : await authStorage.getOAuthAccess(model.provider, undefined, { + baseUrl: model.baseUrl, + modelId: model.id, + }).then(access => (access ? [{ ok: true as const, ...access }] : [])); + const targets: DryBalanceBenchTarget[] = []; + const seen = new Set(); + for (const entry of resolved) { + const key = getBenchTargetKey(entry); + if (seen.has(key)) continue; + seen.add(key); + const account = extractAccount(entry); + if (entry.ok) { + targets.push({ ok: true, account, accessToken: entry.accessToken }); + } else { + targets.push({ ok: false, account, error: entry.error }); + } + } + return targets; +} + +async function runBenchTargets( + model: Model, + targets: DryBalanceBenchTarget[], + randomSessionId: () => string, + progress: DryBalanceBenchProgressSink | undefined, + streamFn: DryBalanceStreamSimple, + now: () => number, +): Promise { + return Promise.all( + targets.map(async (target, index) => { + if (!target.ok) { + const result: DryBalanceBenchFailureResult = { + ok: false, + account: target.account, + error: target.error, + }; + progress?.complete(index, result); + return result; + } + progress?.markRunning(index, target.account); + const result = await runBenchRequest( + model, + randomSessionId(), + target.account, + target.accessToken, + streamFn, + now, + ); + progress?.complete(index, result); + return result; + }), + ); +} + async function createDefaultRuntime(): Promise { const authStorage = await discoverAuthStorage(); try { @@ -186,15 +564,17 @@ async function runOneAttempt( modelId: model.id, }); if (!access) return { ok: false, reason: "no OAuth access resolved" }; - const account = - access.email ?? access.accountId ?? access.projectId ?? access.enterpriseUrl ?? "(unknown oauth account)"; - return { ok: true, account }; + return { ok: true, account: extractAccount(access) }; } catch (error) { - return { ok: false, reason: error instanceof Error ? error.message : String(error) }; + return { ok: false, reason: getErrorMessage(error) }; } } -async function mapConcurrent(items: T[], concurrency: number, fn: (item: T) => Promise): Promise { +async function mapConcurrent( + items: T[], + concurrency: number, + fn: (item: T, index: number) => Promise, +): Promise { const results = new Array(items.length); let nextIndex = 0; const workerCount = Math.min(concurrency, items.length); @@ -204,7 +584,7 @@ async function mapConcurrent(items: T[], concurrency: number, fn: (item: T const index = nextIndex; nextIndex += 1; if (index >= items.length) return; - results[index] = await fn(items[index]); + results[index] = await fn(items[index], index); } }), ); @@ -220,6 +600,36 @@ function sortedStats( .sort((left, right) => right.count - left.count || left.label.localeCompare(right.label)); } +function summarizeBenchResults(results: DryBalanceBenchResult[]): DryBalanceBenchSummary | undefined { + if (results.length === 0) return undefined; + const successes = results.filter(isBenchSuccess); + const failureReasons = new Map(); + for (const result of results) { + if (!result.ok) { + failureReasons.set(result.error, (failureReasons.get(result.error) ?? 0) + 1); + } + } + const average = (values: number[]): number | null => + values.length === 0 ? null : values.reduce((sum, value) => sum + value, 0) / values.length; + return { + total: results.length, + success: { + total: successes.length, + averageTtftMs: average(successes.map(result => result.ttftMs)), + averageTokensPerSecond: average(successes.map(result => result.tokensPerSecond)), + }, + failure: { + total: results.length - successes.length, + reasons: sortedStats(failureReasons, results.length).map(stat => ({ + reason: stat.label, + count: stat.count, + percent: stat.percent, + })), + }, + results, + }; +} + function summarizeResults( model: Model, samples: number, @@ -245,7 +655,7 @@ function summarizeResults( count: stat.count, percent: stat.percent, })); - return { + const summary: DryBalanceSummary = { model: formatModelString(model), provider: model.provider, samples, @@ -259,6 +669,7 @@ function summarizeResults( reasons: failureStats, }, }; + return summary; } function formatRows(rows: Array<{ count: number; percent: number; label: string }>): string[] { @@ -295,6 +706,32 @@ export function formatDryBalanceText(summary: DryBalanceSummary): string { `${summary.failure.total > 0 ? chalk.red("failure") : chalk.dim("failure")} ${summary.failure.total}`, ...formatRows(failureRows), ]; + if (summary.bench) { + const avgTtft = + summary.bench.success.averageTtftMs === null + ? "-" + : formatBenchDuration(summary.bench.success.averageTtftMs); + const avgTps = + summary.bench.success.averageTokensPerSecond === null + ? "-" + : formatBenchTps(summary.bench.success.averageTokensPerSecond); + const benchFailureRows = summary.bench.failure.reasons.map(row => ({ + count: row.count, + percent: row.percent, + label: row.reason, + })); + lines.push( + "", + chalk.bold("bench"), + `requests: ${summary.bench.total}`, + `${chalk.green("success")} ${summary.bench.success.total}`, + `avg TTFT: ${avgTtft}`, + `avg TPS: ${avgTps}`, + "", + `${summary.bench.failure.total > 0 ? chalk.red("failure") : chalk.dim("failure")} ${summary.bench.failure.total}`, + ...formatRows(benchFailureRows), + ); + } return `${lines.join("\n")}\n`; } @@ -315,7 +752,16 @@ export async function runDryBalanceCommand( ((code: number) => { process.exitCode = code; }); + const streamFn = deps.streamSimple ?? streamSimple; + const now = deps.now ?? (() => performance.now()); const runtime = await (deps.createRuntime ?? createDefaultRuntime)(); + let progress: DryBalanceBenchProgressSink | undefined; + let progressClosed = false; + const closeProgress = (): void => { + if (progressClosed) return; + progressClosed = true; + progress?.close(); + }; try { const modelSelector = command.flags.model ?? command.model; const { model, warning } = await resolveDryBalanceModel( @@ -325,19 +771,44 @@ export async function runDryBalanceCommand( randomSessionId, ); if (warning) writeStderr(`${chalk.yellow(`Warning: ${warning}`)}\n`); - const sessionIds = Array.from({ length: samples }, () => randomSessionId()); - const results = await mapConcurrent(sessionIds, concurrency, sessionId => - runOneAttempt(model, runtime.modelRegistry, sessionId), - ); - const summary = summarizeResults(model, samples, concurrency, results); + let results: DryBalanceAttemptResult[]; + let benchResults: DryBalanceBenchResult[] | undefined; + let summarySamples = samples; + let summaryConcurrency = concurrency; + if (command.flags.bench) { + const targets = await resolveBenchTargets(model, runtime.modelRegistry.authStorage); + summarySamples = targets.length; + summaryConcurrency = targets.length; + const progressWrite = command.flags.json ? writeStderr : writeStdout; + const progressInteractive = command.flags.json + ? (deps.stderrIsTTY ?? process.stderr.isTTY === true) + : (deps.stdoutIsTTY ?? process.stdout.isTTY === true); + progress = createBenchProgressSink(targets.length, progressWrite, progressInteractive); + benchResults = await runBenchTargets(model, targets, randomSessionId, progress, streamFn, now); + results = targets.map(target => + target.ok ? { ok: true, account: target.account } : { ok: false, reason: target.error }, + ); + } else { + const sessionIds = Array.from({ length: samples }, () => randomSessionId()); + results = await mapConcurrent(sessionIds, concurrency, sessionId => + runOneAttempt(model, runtime.modelRegistry, sessionId), + ); + } + closeProgress(); + const summary = summarizeResults(model, summarySamples, summaryConcurrency, results); + if (benchResults) { + const benchSummary = summarizeBenchResults(benchResults); + if (benchSummary) summary.bench = benchSummary; + } if (command.flags.json) { writeStdout(`${JSON.stringify(summary, null, 2)}\n`); } else { writeStdout(formatDryBalanceText(summary)); } - if (summary.failure.total > 0) setExitCode(1); + if (summary.failure.total > 0 || (summary.bench?.failure.total ?? 0) > 0) setExitCode(1); return summary; } finally { + closeProgress(); runtime.close?.(); } } diff --git a/packages/coding-agent/src/commands/dry-balance.ts b/packages/coding-agent/src/commands/dry-balance.ts index 2e2c2631e..876ed18f4 100644 --- a/packages/coding-agent/src/commands/dry-balance.ts +++ b/packages/coding-agent/src/commands/dry-balance.ts @@ -16,12 +16,14 @@ export default class DryBalance extends Command { count: Flags.integer({ description: "Number of random session ids to try", default: 100 }), concurrency: Flags.integer({ description: "Maximum concurrent credential resolutions", default: 32 }), json: Flags.boolean({ description: "Output JSON" }), + bench: Flags.boolean({ description: "Send one live benchmark request per sampled session id" }), }; static examples = [ "# Dry-run the configured default model with 100 random session ids\n omp dry-balance", "# Dry-run a specific model\n omp dry-balance anthropic/claude-sonnet-4-5", "# Larger run with bounded concurrency\n omp dry-balance --model openai-codex/gpt-5-codex --count 1000 --concurrency 64", + "# Benchmark resolved accounts with live request status\n omp dry-balance --bench --count 8 --concurrency 4", "# Machine-readable output\n omp dry-balance --json", ]; @@ -34,6 +36,7 @@ export default class DryBalance extends Command { count: flags.count, concurrency: flags.concurrency, json: flags.json, + bench: flags.bench, }, }); } diff --git a/packages/coding-agent/src/prompts/dry-balance-bench.md b/packages/coding-agent/src/prompts/dry-balance-bench.md new file mode 100644 index 000000000..c05f5a177 --- /dev/null +++ b/packages/coding-agent/src/prompts/dry-balance-bench.md @@ -0,0 +1,8 @@ +Write a 20-line poem about balancing OAuth accounts across many providers. + +Form: +- Exactly 20 lines, no title, no stanza breaks. +- Each line is terse and image-driven, in the spirit of haiku: 7 words or fewer, no end punctuation. +- Let the imagery carry the theme — tokens, scopes, refresh cycles, expiry, consent, revocation — rather than naming them literally. + +Output only the 20 lines. No preamble, no commentary, no code fences. diff --git a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts index 71fed5634..d1de88092 100644 --- a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts +++ b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts @@ -36,6 +36,16 @@ describe("task agent capability descriptions", () => { } }); + it("disables read summarization for explore and librarian, leaves other agents summarizing", () => { + const agents = loadBundledAgents(); + + expect(agentByName(agents, "explore").readSummarize).toBe(false); + expect(agentByName(agents, "librarian").readSummarize).toBe(false); + for (const name of ["task", "quick_task", "plan", "reviewer", "oracle", "designer"]) { + expect(agentByName(agents, name).readSummarize).toBeUndefined(); + } + }); + it("marks read-only agents in the task description and keeps full agents unmarked", async () => { vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ agents: [ From c39350d598da627b6c392b5d94381c143a672886 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 11:42:44 +0200 Subject: [PATCH 47/48] fix(dry-balance): decoupled bench mode from sampling flags - Skipped count/concurrency normalization when --bench is set. - Errored when no OAuth accounts resolve for the provider. - Updated flag docs to run one request per OAuth account. --- packages/ai/src/auth-storage.ts | 4 +- .../coding-agent/src/cli/dry-balance-cli.ts | 49 +++++++++++-------- .../coding-agent/src/commands/dry-balance.ts | 4 +- packages/coding-agent/src/task/executor.ts | 5 +- .../test/discovery/agent-fields.test.ts | 8 +-- 5 files changed, 37 insertions(+), 33 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 980323b74..f03f7b810 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -573,9 +573,7 @@ export interface OAuthAccessFailure { error: string; } -export type OAuthAccessResolution = - | ({ ok: true } & OAuthAccess) - | ({ ok: false } & OAuthAccessFailure); +export type OAuthAccessResolution = ({ ok: true } & OAuthAccess) | ({ ok: false } & OAuthAccessFailure); export interface InvalidateCredentialMatchingOptions { signal?: AbortSignal; sessionId?: string; diff --git a/packages/coding-agent/src/cli/dry-balance-cli.ts b/packages/coding-agent/src/cli/dry-balance-cli.ts index 3fa3b95fd..dc35d254f 100644 --- a/packages/coding-agent/src/cli/dry-balance-cli.ts +++ b/packages/coding-agent/src/cli/dry-balance-cli.ts @@ -10,10 +10,11 @@ import type { SimpleStreamOptions, } from "@oh-my-pi/pi-ai"; import { streamSimple } from "@oh-my-pi/pi-ai"; -import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils"; import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui"; +import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils"; import chalk from "chalk"; -import { ModelRegistry } from "../config/model-registry"; +import type { CanonicalModelVariant } from "../config/model-equivalence"; +import { type CanonicalModelQueryOptions, ModelRegistry } from "../config/model-registry"; import { formatModelString, type ModelMatchPreferences, @@ -22,8 +23,8 @@ import { resolveModelRoleValue, } from "../config/model-resolver"; import { Settings } from "../config/settings"; -import { discoverAuthStorage } from "../sdk"; import dryBalanceBenchPrompt from "../prompts/dry-balance-bench.md" with { type: "text" }; +import { discoverAuthStorage } from "../sdk"; const DEFAULT_SAMPLE_COUNT = 100; const DEFAULT_CONCURRENCY = 32; @@ -65,8 +66,8 @@ export interface DryBalanceModelRegistry { getAll(): Model[]; getAvailable(): Model[]; getApiKey(model: Model, sessionId?: string): Promise; - getCanonicalVariants(model: Model): Model[]; - resolveCanonicalModel?(model: Model): Model | undefined; + getCanonicalVariants(canonicalId: string, options?: CanonicalModelQueryOptions): CanonicalModelVariant[]; + resolveCanonicalModel?(canonicalId: string, options?: CanonicalModelQueryOptions): Model | undefined; getCanonicalId?(model: Model): string | undefined; } @@ -288,7 +289,12 @@ function renderBenchResultLine(index: number, total: number, result: DryBalanceB )}`; } -function renderBenchStatusLine(status: DryBalanceBenchProgressStatus, index: number, total: number, frame: number): string { +function renderBenchStatusLine( + status: DryBalanceBenchProgressStatus, + index: number, + total: number, + frame: number, +): string { const prefix = formatBenchIndex(index, total); switch (status.state) { case "waiting": @@ -407,7 +413,9 @@ async function runBenchRequest( return { ok: false, account, error: message.errorMessage ?? "request failed" }; } const durationMs = normalizeBenchMs(message.duration ?? now() - startedAt); - const ttftMs = normalizeBenchMs(message.ttft ?? (firstTokenAt === undefined ? durationMs : firstTokenAt - startedAt)); + const ttftMs = normalizeBenchMs( + message.ttft ?? (firstTokenAt === undefined ? durationMs : firstTokenAt - startedAt), + ); const outputTokens = Number.isFinite(message.usage.output) && message.usage.output > 0 ? message.usage.output : 0; const tokensPerSecond = durationMs > 0 ? (outputTokens * 1000) / durationMs : 0; return { @@ -432,10 +440,12 @@ async function resolveBenchTargets( baseUrl: model.baseUrl, modelId: model.id, }) - : await authStorage.getOAuthAccess(model.provider, undefined, { - baseUrl: model.baseUrl, - modelId: model.id, - }).then(access => (access ? [{ ok: true as const, ...access }] : [])); + : await authStorage + .getOAuthAccess(model.provider, undefined, { + baseUrl: model.baseUrl, + modelId: model.id, + }) + .then(access => (access ? [{ ok: true as const, ...access }] : [])); const targets: DryBalanceBenchTarget[] = []; const seen = new Set(); for (const entry of resolved) { @@ -708,9 +718,7 @@ export function formatDryBalanceText(summary: DryBalanceSummary): string { ]; if (summary.bench) { const avgTtft = - summary.bench.success.averageTtftMs === null - ? "-" - : formatBenchDuration(summary.bench.success.averageTtftMs); + summary.bench.success.averageTtftMs === null ? "-" : formatBenchDuration(summary.bench.success.averageTtftMs); const avgTps = summary.bench.success.averageTokensPerSecond === null ? "-" @@ -739,11 +747,11 @@ export async function runDryBalanceCommand( command: DryBalanceCommandArgs, deps: DryBalanceDependencies = {}, ): Promise { - const samples = normalizePositiveInteger("count", command.flags.count, DEFAULT_SAMPLE_COUNT); - const concurrency = Math.min( - samples, - normalizePositiveInteger("concurrency", command.flags.concurrency, DEFAULT_CONCURRENCY), - ); + const isBench = command.flags.bench === true; + const samples = isBench ? 0 : normalizePositiveInteger("count", command.flags.count, DEFAULT_SAMPLE_COUNT); + const concurrency = isBench + ? 0 + : Math.min(samples, normalizePositiveInteger("concurrency", command.flags.concurrency, DEFAULT_CONCURRENCY)); const randomSessionId = deps.randomSessionId ?? (() => Bun.randomUUIDv7()); const writeStdout = deps.writeStdout ?? ((text: string) => process.stdout.write(text)); const writeStderr = deps.writeStderr ?? ((text: string) => process.stderr.write(text)); @@ -775,8 +783,9 @@ export async function runDryBalanceCommand( let benchResults: DryBalanceBenchResult[] | undefined; let summarySamples = samples; let summaryConcurrency = concurrency; - if (command.flags.bench) { + if (isBench) { const targets = await resolveBenchTargets(model, runtime.modelRegistry.authStorage); + if (targets.length === 0) throw new Error(`No OAuth accounts resolved for provider ${model.provider}`); summarySamples = targets.length; summaryConcurrency = targets.length; const progressWrite = command.flags.json ? writeStderr : writeStdout; diff --git a/packages/coding-agent/src/commands/dry-balance.ts b/packages/coding-agent/src/commands/dry-balance.ts index 876ed18f4..e27763014 100644 --- a/packages/coding-agent/src/commands/dry-balance.ts +++ b/packages/coding-agent/src/commands/dry-balance.ts @@ -16,14 +16,14 @@ export default class DryBalance extends Command { count: Flags.integer({ description: "Number of random session ids to try", default: 100 }), concurrency: Flags.integer({ description: "Maximum concurrent credential resolutions", default: 32 }), json: Flags.boolean({ description: "Output JSON" }), - bench: Flags.boolean({ description: "Send one live benchmark request per sampled session id" }), + bench: Flags.boolean({ description: "Send one live benchmark request per OAuth account" }), }; static examples = [ "# Dry-run the configured default model with 100 random session ids\n omp dry-balance", "# Dry-run a specific model\n omp dry-balance anthropic/claude-sonnet-4-5", "# Larger run with bounded concurrency\n omp dry-balance --model openai-codex/gpt-5-codex --count 1000 --concurrency 64", - "# Benchmark resolved accounts with live request status\n omp dry-balance --bench --count 8 --concurrency 4", + "# Benchmark every OAuth account in parallel\n omp dry-balance --bench", "# Machine-readable output\n omp dry-balance --json", ]; diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index c884bb468..94fb9a63b 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -531,10 +531,7 @@ function createMCPProxyTools(mcpManager: MCPManager): CustomTool[] { }); } -function createSubagentSettings( - baseSettings: Settings, - overrides?: Partial>, -): Settings { +function createSubagentSettings(baseSettings: Settings, overrides?: Partial>): Settings { const snapshot: Partial> = {}; for (const key of Object.keys(SETTINGS_SCHEMA) as SettingPath[]) { snapshot[key] = baseSettings.get(key); diff --git a/packages/coding-agent/test/discovery/agent-fields.test.ts b/packages/coding-agent/test/discovery/agent-fields.test.ts index 2ae792de2..56a3e8974 100644 --- a/packages/coding-agent/test/discovery/agent-fields.test.ts +++ b/packages/coding-agent/test/discovery/agent-fields.test.ts @@ -114,9 +114,7 @@ describe("parseAgentFields", () => { expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: false })?.readSummarize).toBe( false, ); - expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: true })?.readSummarize).toBe( - true, - ); + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: true })?.readSummarize).toBe(true); }); test("parses readSummarize from string frontmatter", () => { @@ -126,7 +124,9 @@ describe("parseAgentFields", () => { }); test("ignores invalid readSummarize values", () => { - expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: "nope" })?.readSummarize).toBeUndefined(); + expect( + parseAgentFields({ name: "explore", description: "desc", readSummarize: "nope" })?.readSummarize, + ).toBeUndefined(); }); test("returns undefined readSummarize when field absent", () => { From 241a42350c0c67edc6e79cffecdadb5cc8d0455b Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 5 Jun 2026 13:27:40 +0200 Subject: [PATCH 48/48] chore: bump version to 15.9.2 --- Cargo.lock | 20 +++++----- Cargo.toml | 2 +- bun.lock | 48 +++++++++++++----------- crates/pi-natives/src/lib.rs | 2 +- package.json | 18 ++++----- packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 7 ++-- packages/ai/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 54 +++++++-------------------- packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 7 ++-- packages/tui/package.json | 2 +- packages/utils/CHANGELOG.md | 2 + packages/utils/package.json | 2 +- 21 files changed, 82 insertions(+), 102 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 79930c842..d04f2f533 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -549,9 +549,9 @@ dependencies = [ [[package]] name = "chrono" -version = "0.4.44" +version = "0.4.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c673075a2e0e5f4a1dde27ce9dee1ea4558c7ffe648f576438a20ca1d2acc4b0" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" dependencies = [ "iana-time-zone", "js-sys", @@ -1527,9 +1527,9 @@ checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" [[package]] name = "ignore" -version = "0.4.25" +version = "0.4.26" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3d782a365a015e0f5c04902246139249abf769125006fbe7649e2ee88169b4a" +checksum = "b915661dd01db3f05050265b2477bcc6527b3792388e2749b41623cc592be67d" dependencies = [ "crossbeam-deque", "globset", @@ -1748,9 +1748,9 @@ dependencies = [ [[package]] name = "log" -version = "0.4.31" +version = "0.4.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "113b30b4cd05f7c06868fdb2854f66a7b9fece9a48425351cd532e810d74024f" +checksum = "953f07c43838f8e6f9758cab68bf5bed85465e7587ebe0b823f1bcd81978ad3a" [[package]] name = "lru" @@ -2331,7 +2331,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "15.9.1" +version = "15.9.2" dependencies = [ "anyhow", "ast-grep-core", @@ -2399,7 +2399,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "15.9.1" +version = "15.9.2" dependencies = [ "async-trait", "libc", @@ -2411,7 +2411,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "15.9.1" +version = "15.9.2" dependencies = [ "anyhow", "arboard", @@ -2457,7 +2457,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "15.9.1" +version = "15.9.2" dependencies = [ "anyhow", "brush-builtins", diff --git a/Cargo.toml b/Cargo.toml index 02f333bbf..68aa16696 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "15.9.1" +version = "15.9.2" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index ebffae31e..538d79fd5 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -30,7 +30,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -44,7 +44,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.1", + "version": "15.9.2", "bin": { "omp": "src/cli.ts", }, @@ -90,7 +90,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -101,7 +101,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.1", + "version": "15.9.2", "bin": { "mnemopi": "src/cli.ts", }, @@ -118,7 +118,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "15.9.1", + "version": "15.9.2", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -126,7 +126,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "15.9.1", + "version": "15.9.2", "bin": { "omp-stats": "./src/index.ts", }, @@ -151,7 +151,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "15.9.1", + "version": "15.9.2", "bin": { "omp-swarm": "src/cli.ts", }, @@ -167,7 +167,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -208,7 +208,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "beautiful-mermaid": "catalog:", @@ -248,15 +248,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.1", - "@oh-my-pi/omp-stats": "15.9.1", - "@oh-my-pi/pi-agent-core": "15.9.1", - "@oh-my-pi/pi-ai": "15.9.1", - "@oh-my-pi/pi-coding-agent": "15.9.1", - "@oh-my-pi/pi-mnemopi": "15.9.1", - "@oh-my-pi/pi-natives": "15.9.1", - "@oh-my-pi/pi-tui": "15.9.1", - "@oh-my-pi/pi-utils": "15.9.1", + "@oh-my-pi/hashline": "15.9.2", + "@oh-my-pi/omp-stats": "15.9.2", + "@oh-my-pi/pi-agent-core": "15.9.2", + "@oh-my-pi/pi-ai": "15.9.2", + "@oh-my-pi/pi-coding-agent": "15.9.2", + "@oh-my-pi/pi-mnemopi": "15.9.2", + "@oh-my-pi/pi-natives": "15.9.2", + "@oh-my-pi/pi-tui": "15.9.2", + "@oh-my-pi/pi-utils": "15.9.2", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -787,7 +787,7 @@ "@types/node": ["@types/node@25.9.1", "", { "dependencies": { "undici-types": ">=7.24.0 <7.24.7" } }, "sha512-xfrlY7UD5rMJk3ZVJP8BNzS28J36YJg+xp+LPXV1TdWxr8uMH5A860QNxYDGQe/ylDSgjxE52Q9VnO7p75tJxg=="], - "@types/react": ["@types/react@19.2.15", "", { "dependencies": { "csstype": "^3.2.2" } }, "sha512-eRwcGNHve+E8qtEQSSRl6urh+rFop4v8gm6O8rGv25CodbvFdLjA1vVQ1KkiFE0w0UPOnb8tDiFKL5lp0rtY5Q=="], + "@types/react": ["@types/react@19.2.16", "", { "dependencies": { "csstype": "^3.2.2" } }, "sha512-esJiCAnl0kfpNdE69f3So4WJUXy95dLZydX0KwK46riIHDzHM7O9Vtf9xCHW0PXIqvgqNrswl522kA/5yx+F4w=="], "@types/react-dom": ["@types/react-dom@19.2.3", "", { "peerDependencies": { "@types/react": "^19.2.0" } }, "sha512-jp2L/eY6fn+KgVVQAOqYItbF0VY/YApe5Mz2F0aykSO8gx31bYCZyvSeYxCHKvzHG5eZjc+zyaS5BrBWya2+kQ=="], @@ -1139,7 +1139,7 @@ "neo-async": ["neo-async@2.6.2", "", {}, "sha512-Yd3UES5mWCSqR+qNT93S3UoYUkqAZ9lLg8a7g9rimsWmYGK8cVToA4/sF3RrshdyV3sAGMXVUmpMYOw+dLpOuw=="], - "node-releases": ["node-releases@2.0.46", "", {}, "sha512-GYVXHE2KnrzAfsAjl4uP++evGFCrAU1jta4ubEjIG7YWt/64Gqv66a30yKwWczVjA6j3bM4nBwH7Pk1JmDHaxQ=="], + "node-releases": ["node-releases@2.0.47", "", {}, "sha512-Uzmd6LXpouKo8EUK68IjH4+E01w/hXyV3R3g/geCJo+rXLNfh1xucB+LOzYEOQPSiUK3h/xZf0cQGcSsmyL2Og=="], "nth-check": ["nth-check@2.1.1", "", { "dependencies": { "boolbase": "^1.0.0" } }, "sha512-lqjrjmaOoAnWfMmBPL+XNnynZh2+swxiX3WUE0s4yEHI6m+AwrK2UZOimIRl3X/4QctVqS8AiZjFqyOGrMXb/w=="], @@ -1259,7 +1259,7 @@ "string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="], - "string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], + "string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="], "strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="], @@ -1413,6 +1413,8 @@ "string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="], + "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], + "wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="], "xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="], @@ -1427,6 +1429,8 @@ "fastembed/onnxruntime-node/tar": ["tar@7.5.16", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w=="], + "jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], + "log-update/slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], "log-update/wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 4ed19386d..231322545 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -68,5 +68,5 @@ use napi_derive::napi; /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV15_9_1")] +#[napi(js_name = "__piNativesV15_9_2")] pub const fn pi_natives_version_sentinel() {} diff --git a/package.json b/package.json index 4cce29fa6..a34d1e667 100644 --- a/package.json +++ b/package.json @@ -20,15 +20,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.1", - "@oh-my-pi/omp-stats": "15.9.1", - "@oh-my-pi/pi-agent-core": "15.9.1", - "@oh-my-pi/pi-ai": "15.9.1", - "@oh-my-pi/pi-coding-agent": "15.9.1", - "@oh-my-pi/pi-mnemopi": "15.9.1", - "@oh-my-pi/pi-natives": "15.9.1", - "@oh-my-pi/pi-tui": "15.9.1", - "@oh-my-pi/pi-utils": "15.9.1", + "@oh-my-pi/hashline": "15.9.2", + "@oh-my-pi/omp-stats": "15.9.2", + "@oh-my-pi/pi-agent-core": "15.9.2", + "@oh-my-pi/pi-ai": "15.9.2", + "@oh-my-pi/pi-coding-agent": "15.9.2", + "@oh-my-pi/pi-mnemopi": "15.9.2", + "@oh-my-pi/pi-natives": "15.9.2", + "@oh-my-pi/pi-tui": "15.9.2", + "@oh-my-pi/pi-utils": "15.9.2", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index d58b4a8fb..9eb31590f 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.1", + "version": "15.9.2", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 79c330433..83afdf41e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,18 +2,17 @@ ## [Unreleased] +## [15.9.2] - 2026-06-05 + ### Added - Added an AES-256-GCM auth-broker snapshot cache module and `RemoteAuthCredentialStoreOptions.onSnapshot` so broker clients can persist broker-sourced full snapshots without blocking startup on every run. +- Added `Model.omitMaxOutputTokens` so providers (notably Ollama proxies fronting cloud catalogs) can suppress `max_output_tokens` (Responses) and `max_tokens`/`max_completion_tokens` (Completions) on the wire while still using the catalog `maxTokens` for local budgeting. Without it, `applyCommonResponsesSamplingParams` unconditionally sent the catalog cap and HTTP-400'd against upstream APIs whose true output limit was unknown to OMP. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881)) ### Changed - Changed usage-ranked OAuth credential selection to pick deterministic session-sticky weighted buckets instead of always choosing the top-ranked account, capping the best account at 2x the baseline session likelihood while keeping equal-priority accounts evenly balanced. -### Added - -- Added `Model.omitMaxOutputTokens` so providers (notably Ollama proxies fronting cloud catalogs) can suppress `max_output_tokens` (Responses) and `max_tokens`/`max_completion_tokens` (Completions) on the wire while still using the catalog `maxTokens` for local budgeting. Without it, `applyCommonResponsesSamplingParams` unconditionally sent the catalog cap and HTTP-400'd against upstream APIs whose true output limit was unknown to OMP. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881)) - ### Fixed - Fixed parallel `function_call` items on the OpenAI Responses API losing arguments on every call except the last when the upstream server interleaves their stream events (observed against llama.cpp and other local Responses-compat hosts). `processResponsesStream` no longer routes `function_call_arguments.{delta,done}`, `output_item.done`, content_part/text/refusal/reasoning events through a singleton `currentItem`/`currentBlock` reference; it now tracks every open item in registries keyed by `output_index` and `item_id` so each event is folded into the matching block and the emitted `toolcall_end` carries the correct `contentIndex`. ([#1880](https://github.com/can1357/oh-my-pi/issues/1880)) diff --git a/packages/ai/package.json b/packages/ai/package.json index f80274b8b..a68ebf4c0 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "15.9.1", + "version": "15.9.2", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 067d3abd0..0644055c1 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,51 +1,29 @@ # Changelog ## [Unreleased] + +## [15.9.2] - 2026-06-05 + ### Added - Added an encrypted local auth-broker snapshot cache for `discoverAuthStorage`, with `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` and `OMP_AUTH_BROKER_SNAPSHOT_CACHE`, so fresh cached broker credentials can boot without a blocking `/v1/snapshot` fetch and survive broker-down startup windows. - Added `dry-balance` CLI command to perform a dry-run OAuth account balancing check across configurable random session IDs, with sample and concurrency options, JSON output, and success/failure summary reporting - Added `--json` output mode and machine-readable result format to `omp dry-balance` for automated use - -### Fixed - -- Fixed TTSR rule-violation injections leaking the absolute home directory to the model: the `ttsr-interrupt` / `ttsr-tool-reminder` blocks rendered the matched rule's `path` as its absolute on-disk path (e.g. `/Users/me/Projects/app/.omp/rules/no-any.md`). The path is now relativized to the session cwd when the rule lives in the project (`.omp/rules/no-any.md`), or `~`-relative when it lives under home, so no absolute path is fed into the agent's context outside the system prompt. - -### Fixed - -- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. Secondary sessions now leave the live singleton untouched, and their dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools and session job snapshots now resolve the manager through session-scoped async manager wiring rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager or report the owning session's jobs; subagents still inherit the parent's manager via their scoped async manager. Startup failures after a top-level session installs its manager now clear and dispose that manager before the next session decides whether it can create its own ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). - -### Fixed - -- Fixed the `task` tool returning a hard `Async execution is enabled but no async job manager is available.` error when `async.enabled` was true but `AsyncJobManager.instance()` returned `undefined`, leaving `task` non-functional for the rest of the session. The tool now falls back to the existing synchronous execution path (which still runs subagents concurrently via `mapWithConcurrencyLimit`), and logs a warning so the missing-manager state stays diagnosable ([#1922](https://github.com/can1357/oh-my-pi/issues/1922)). - -### Fixed - -- Fixed Hindsight retain/recall/reflect calls staying pinned to the bank that was selected when the session started after the operator edited `hindsight.bankId`, `hindsight.bankIdPrefix`, or `hindsight.scoping` mid-session. The backend now subscribes to those settings via `onHindsightScopeChanged` and rebuilds the active `HindsightSessionState` against the recomputed scope, disposing the old state after flushing its queue so in-flight tool-initiated retains still land in the bank they were enqueued for. Also renamed `ensureBankMission` to `ensureBankExists` so a blank `bankMission` no longer skips bank creation entirely, and called it before mental-model bootstrap so `createMentalModel` is never the first POST against a missing bank. `AgentSession.dispose` now flushes the retain queue before clearing `#hindsightSessionState`, since the queue's identity guard would otherwise drop the spliced batch ([#1902](https://github.com/can1357/oh-my-pi/issues/1902)). - -### Fixed - -- Fixed `/tree` rendering a bare "No entries found" line on a fresh session where the only persisted entries are the `model_change` + `thinking_level_change` written by `sdk.ts` at startup — both are hidden by the tree-selector's default filter, so `#filteredNodes.length === 0` while `tree.length === 2` and the controller's `tree.length === 0` short-circuit never fired. The selector now splits the empty-state into three distinct shapes — truly empty tree, search query with no matches, and filter mode rejecting every entry — surfacing the cause and the recovery key (`Alt+A` to show all, `Backspace` to clear a stale search) so users on a fresh session can see immediately that the panel isn't broken ([#1909](https://github.com/can1357/oh-my-pi/issues/1909)). - -### Fixed - -- Fixed remote MCP OAuth refresh failures leaving stale credentials in `agent.db`: when the token endpoint returns a definitive failure (`invalid_grant`, `invalid_token`, `revoked`, plain 401/403 not classified as transient), `MCPManager#resolveAuthConfig` now drops the credential via `AuthStorage.remove(credentialId)` and skips re-attaching the dead `Authorization: Bearer …` header. Previously a revoked refresh token kept producing `401 invalid_token` on every MCP request and survived restarts, so users had to hand-clear the credential row to recover; the next connect now surfaces a clean auth error and `/mcp reauth ` (or `/mcp unauth`) recovers without restarting. Transient refresh failures (network/`fetch failed`/`ECONNREFUSED`) still fall back to the existing access token ([#1908](https://github.com/can1357/oh-my-pi/issues/1908)). - -### Fixed - -- Fixed `omp://docs` and `omp://docs/...` internal documentation URLs in the distributed package to resolve through the embedded documentation index instead of failing with `Documentation file not found` ([#1898](https://github.com/can1357/oh-my-pi/issues/1898)). - -### Fixed - -- Fixed the `github` discovery provider silently ignoring `.github/skills//SKILL.md`, GitHub's documented Agent Skills layout. The provider now registers a `skills` capability (priority 30, project-only) that scans `.github/skills/` non-recursively via `scanSkillsFromDir` with `requireDescription: true`, matching the Agent Skills spec and the sibling `native`/`omp-plugins` providers ([#1906](https://github.com/can1357/oh-my-pi/issues/1906)). - -### Added - - Added `omitMaxOutputTokens` to `models.yml` model definitions and `modelOverrides`, so users can opt a model out of the on-the-wire `max_output_tokens` / `max_tokens` cap while keeping the catalog `maxTokens` for local budgeting. Intended for Ollama-style proxies whose upstream output limit OMP cannot discover. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881)) ### Fixed +- Fixed TTSR rule-violation injections leaking the absolute home directory to the model: the `ttsr-interrupt` / `ttsr-tool-reminder` blocks rendered the matched rule's `path` as its absolute on-disk path (e.g. `/Users/me/Projects/app/.omp/rules/no-any.md`). The path is now relativized to the session cwd when the rule lives in the project (`.omp/rules/no-any.md`), or `~`-relative when it lives under home, so no absolute path is fed into the agent's context outside the system prompt. +- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. Secondary sessions now leave the live singleton untouched, and their dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools and session job snapshots now resolve the manager through session-scoped async manager wiring rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager or report the owning session's jobs; subagents still inherit the parent's manager via their scoped async manager. Startup failures after a top-level session installs its manager now clear and dispose that manager before the next session decides whether it can create its own ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). +- Fixed the `task` tool returning a hard `Async execution is enabled but no async job manager is available.` error when `async.enabled` was true but `AsyncJobManager.instance()` returned `undefined`, leaving `task` non-functional for the rest of the session. The tool now falls back to the existing synchronous execution path (which still runs subagents concurrently via `mapWithConcurrencyLimit`), and logs a warning so the missing-manager state stays diagnosable ([#1922](https://github.com/can1357/oh-my-pi/issues/1922)). +- Fixed Hindsight retain/recall/reflect calls staying pinned to the bank that was selected when the session started after the operator edited `hindsight.bankId`, `hindsight.bankIdPrefix`, or `hindsight.scoping` mid-session. The backend now subscribes to those settings via `onHindsightScopeChanged` and rebuilds the active `HindsightSessionState` against the recomputed scope, disposing the old state after flushing its queue so in-flight tool-initiated retains still land in the bank they were enqueued for. Also renamed `ensureBankMission` to `ensureBankExists` so a blank `bankMission` no longer skips bank creation entirely, and called it before mental-model bootstrap so `createMentalModel` is never the first POST against a missing bank. `AgentSession.dispose` now flushes the retain queue before clearing `#hindsightSessionState`, since the queue's identity guard would otherwise drop the spliced batch ([#1902](https://github.com/can1357/oh-my-pi/issues/1902)). +- Fixed `/tree` rendering a bare "No entries found" line on a fresh session where the only persisted entries are the `model_change` + `thinking_level_change` written by `sdk.ts` at startup — both are hidden by the tree-selector's default filter, so `#filteredNodes.length === 0` while `tree.length === 2` and the controller's `tree.length === 0` short-circuit never fired. The selector now splits the empty-state into three distinct shapes — truly empty tree, search query with no matches, and filter mode rejecting every entry — surfacing the cause and the recovery key (`Alt+A` to show all, `Backspace` to clear a stale search) so users on a fresh session can see immediately that the panel isn't broken ([#1909](https://github.com/can1357/oh-my-pi/issues/1909)). +- Fixed remote MCP OAuth refresh failures leaving stale credentials in `agent.db`: when the token endpoint returns a definitive failure (`invalid_grant`, `invalid_token`, `revoked`, plain 401/403 not classified as transient), `MCPManager#resolveAuthConfig` now drops the credential via `AuthStorage.remove(credentialId)` and skips re-attaching the dead `Authorization: Bearer …` header. Previously a revoked refresh token kept producing `401 invalid_token` on every MCP request and survived restarts, so users had to hand-clear the credential row to recover; the next connect now surfaces a clean auth error and `/mcp reauth ` (or `/mcp unauth`) recovers without restarting. Transient refresh failures (network/`fetch failed`/`ECONNREFUSED`) still fall back to the existing access token ([#1908](https://github.com/can1357/oh-my-pi/issues/1908)). +- Fixed `omp://docs` and `omp://docs/...` internal documentation URLs in the distributed package to resolve through the embedded documentation index instead of failing with `Documentation file not found` ([#1898](https://github.com/can1357/oh-my-pi/issues/1898)). +- Fixed the `github` discovery provider silently ignoring `.github/skills//SKILL.md`, GitHub's documented Agent Skills layout. The provider now registers a `skills` capability (priority 30, project-only) that scans `.github/skills/` non-recursively via `scanSkillsFromDir` with `requireDescription: true`, matching the Agent Skills spec and the sibling `native`/`omp-plugins` providers ([#1906](https://github.com/can1357/oh-my-pi/issues/1906)). - Fixed inline images rendering as a wall of empty PUA box glyphs with laggy scrolling on Kitty-protocol terminals that do not honor Unicode placeholders (most notably WezTerm and tmux/screen passthrough to a non-Kitty outer terminal). The 15.9 placeholder rollout enabled the `U=1`/U+10EEEE grid for every Kitty-protocol path; it now defaults on only for `kitty` and `ghostty`, with `PI_NO_KITTY_PLACEHOLDERS=1` as a hard opt-out and `PI_KITTY_PLACEHOLDERS=1` as opt-in for terminals (e.g. wezterm nightlies) that have since added support ([#1877](https://github.com/can1357/oh-my-pi/issues/1877)). +- Fixed auto session-title generation failures being swallowed without an actionable diagnostic. Title generation now logs structured start, missing-model/API-key, provider-error, empty-result, and exception outcomes with the session id and resolved title model; the interactive auto-title caller also logs uncaught persistence/generation errors instead of dropping them. ([#1892](https://github.com/can1357/oh-my-pi/issues/1892)) +- Fixed `TranscriptContainer` reporting the live block boundary to the TUI again, so ED3-risk foreground streaming can append newly sealed transcript blocks to native scrollback once while deferring only the active live block. ## [15.9.1] - 2026-06-04 @@ -64,11 +42,7 @@ ### Fixed -- Fixed auto session-title generation failures being swallowed without an actionable diagnostic. Title generation now logs structured start, missing-model/API-key, provider-error, empty-result, and exception outcomes with the session id and resolved title model; the interactive auto-title caller also logs uncaught persistence/generation errors instead of dropping them. ([#1892](https://github.com/can1357/oh-my-pi/issues/1892)) - - Fixed a streamed assistant message freezing at a partial prefix (e.g. only "Nat" of "Natives built, now…") on ED3-risk terminals (Ghostty/kitty/iTerm2/Alacritty), with the final text appearing only after a resize. `TranscriptContainer` freezes each non-live block by replaying its last live render, but render coalescing can finalize a block's content and append the next block within the same throttled frame — so the block was sealed at its stale mid-stream snapshot and never repainted until the next `thaw`. The block that was live on the previous render is now recomputed once on the live→frozen transition, sealing it at its final content. -- Fixed `TranscriptContainer` reporting the live block boundary to the TUI again, so ED3-risk foreground streaming can append newly sealed transcript blocks to native scrollback once while deferring only the active live block. - - Fixed ACP/RPC stdio startup so protocol frames are no longer consumed as one-shot piped prompt input before the JSON-RPC transport starts. - Fixed `omp completions` to await the completion script write before exiting. - Fixed `AssistantMessageComponent` exposing its stable-prefix completion API again so streamed assistant messages remain unstable until explicitly completed. @@ -9381,4 +9355,4 @@ Initial public release. - Git branch display in footer - Message queueing during streaming responses - OAuth integration for Gmail and Google Calendar access -- HTML export with syntax highlighting and collapsible sections \ No newline at end of file +- HTML export with syntax highlighting and collapsible sections diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index b87565e10..d983390e5 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.1", + "version": "15.9.2", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 31dcc4e9d..dde5a1987 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "15.9.1", + "version": "15.9.2", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 4de037377..af532fc40 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.1", + "version": "15.9.2", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 3890932c1..060e22860 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -136,7 +136,7 @@ export declare class Shell { * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV15_9_1(): void +export declare function __piNativesV15_9_2(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index fdc0d51b6..4bfa3a826 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -23,7 +23,7 @@ export const PtySession = nativeBindings.PtySession; export const Shell = nativeBindings.Shell; // functions -export const __piNativesV15_9_1 = nativeBindings.__piNativesV15_9_1; +export const __piNativesV15_9_2 = nativeBindings.__piNativesV15_9_2; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 33d473f79..b331763c8 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "15.9.1", + "version": "15.9.2", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/stats/package.json b/packages/stats/package.json index 0547ef087..e6e296803 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "15.9.1", + "version": "15.9.2", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index fa42f9758..25868ff20 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "15.9.1", + "version": "15.9.2", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 03d98b78f..44f7af3cd 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] + +## [15.9.2] - 2026-06-05 + ### Changed - Changed foreground-stream rendering on ED3-risk terminals (Ghostty/kitty/Alacritty/VTE/iTerm2 on POSIX) to defer native-scrollback commits for unpinned transient frames: while a turn streams, generic frames repaint only the viewport and suppress `\r\n` scroll growth, so transient output (spinner ticks, partial lines, status rows) never pollutes terminal history. Components that report a `NativeScrollbackLiveRegion` still commit newly sealed prefix rows while keeping the active suffix dirty for checkpoint replay. Native scrollback is reconciled in a single ED3 (`CSI 3 J`) + re-emit at the next checkpoint (prompt submit) or on an explicit user-input/IME opt-in; an erase is never emitted mid-stream under a possibly-scrolled reader. Non-ED3-risk terminals keep their eager live rebuild. ([#1895](https://github.com/can1357/oh-my-pi/pull/1895)) @@ -8,12 +11,10 @@ ### Fixed - Fixed ED3-risk foreground streaming dropping sealed transcript rows above the live block until the next prompt-submit checkpoint, which made scrollback beyond the viewport appear duplicated or out of order. The renderer restores native-scrollback live-region pinning so newly sealed rows are appended once while active live rows remain deferred. - -### Fixed - - Fixed inline images (added in 15.9) rendering as a wall of empty PUA box glyphs and producing laggy scrolling on Kitty-protocol terminals that do not implement Unicode placeholders — most notably WezTerm (per upstream wezterm/wezterm#986, placeholder support is still unchecked) and the tmux/screen `getFallbackImageProtocol` path that forces Kitty mode even on non-supporting outer terminals (Terminal.app, etc.). `unicodePlaceholders` now defaults on only for `kitty` and `ghostty`; everything else falls back to direct `a=p,i=…,p=…` placement, which those paths already render correctly. `PI_NO_KITTY_PLACEHOLDERS=1` is still honored as a hard opt-out, and a new `PI_KITTY_PLACEHOLDERS=1` opts in on otherwise-unsupported terminals (e.g. a wezterm nightly that has merged placeholder support) ([#1877](https://github.com/can1357/oh-my-pi/issues/1877)). ## [15.9.1] - 2026-06-04 + ### Fixed - Fixed the OSC 11 appearance poll re-querying every 2s forever on terminals that support Mode 2031 but never change theme, whose repeated OSC 11/DA1 writes cleared the user's active text selection (breaking copy every 2 seconds). The poll now stops as soon as DECRQM confirms Mode 2031 support, since push notifications make polling redundant. diff --git a/packages/tui/package.json b/packages/tui/package.json index bcbb60469..a28a4c50b 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "15.9.1", + "version": "15.9.2", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index dc16a8eeb..23eee711b 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.9.2] - 2026-06-05 + ### Added - Added `getAuthBrokerSnapshotCachePath()` with `OMP_AUTH_BROKER_SNAPSHOT_CACHE` override support for isolating the encrypted broker snapshot cache. diff --git a/packages/utils/package.json b/packages/utils/package.json index 231cd020e..5fd061372 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "15.9.1", + "version": "15.9.2", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk",