From 7cdecd824a871efce161fe68aa9e230ca8a8788b Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 15:56:49 +0000 Subject: [PATCH 001/293] fix(catalog): widened GLM coding-plan idle timeout to opencode gateways GLM-5.x coding-plan SKUs idle for minutes mid-reasoning, so they get a 600s stream idle-timeout floor instead of the 120s default. That floor was gated to the native Z.AI/Zhipu hosts only, so GLM-5.2 served through the OpenCode Go/Zen gateways fell back to the 120s watchdog and stalled with "OpenAI completions stream stalled while waiting for the next event" during the slow /plan writing phase. Fixes #4758 --- packages/catalog/CHANGELOG.md | 4 +++ packages/catalog/src/compat/openai.ts | 2 +- packages/catalog/test/zhipu-compat.test.ts | 33 ++++++++++++++++++++++ 3 files changed, 38 insertions(+), 1 deletion(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 42e39218a..ee14f3ef7 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed GLM-5.x coding-plan streams via the OpenCode Go/Zen gateways (`opencode.ai/zen/…`) timing out with `OpenAI completions stream stalled while waiting for the next event` during slow plan-writing/reasoning phases. The 600s idle-timeout floor for GLM coding-plan SKUs was gated to the native Z.AI/Zhipu hosts only, so OpenCode-fronted GLM fell back to the 120s default watchdog. ([#4758](https://github.com/can1357/oh-my-pi/issues/4758)) + ## [16.3.11] - 2026-07-06 ### Added diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index d5627ccd5..6f1dd0ed8 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -377,7 +377,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv // for minutes while reasoning or cold-loading weights; widen the idle // timeout so warm-ups stop aborting and retrying. const streamIdleTimeoutMs = - GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu) + GLM_CODING_PLAN_MODEL_PATTERN.test(spec.id) && (isZai || isZhipu || isOpenCodeHost) ? GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS : provider === "alibaba-coding-plan" ? ALIBABA_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS diff --git a/packages/catalog/test/zhipu-compat.test.ts b/packages/catalog/test/zhipu-compat.test.ts index 82ff798be..b69b9afb7 100644 --- a/packages/catalog/test/zhipu-compat.test.ts +++ b/packages/catalog/test/zhipu-compat.test.ts @@ -120,6 +120,39 @@ describe("openai-completions compat — zhipu-coding-plan branch", () => { }); }); +describe("openai-completions compat — GLM coding-plan stream idle timeout", () => { + function glm52(provider: string, baseUrl: string): ModelSpec<"openai-completions"> { + return { ...baseModel, id: "glm-5.2", name: "GLM-5.2", provider, baseUrl }; + } + + // GLM coding-plan SKUs idle for minutes mid-reasoning; the 600s watchdog + // floor must apply on every gateway that fronts them, not just the native + // Z.AI/Zhipu hosts (issue #4758: GLM-5.2 via opencode-go stalled with + // "OpenAI completions stream stalled while waiting for the next event"). + it("widens the idle timeout to 600s for GLM-5.x on Z.AI, Zhipu, and OpenCode gateways", () => { + expect(buildOpenAICompat(glm52("zai", "https://api.z.ai/api/coding/paas/v4")).streamIdleTimeoutMs).toBe(600_000); + expect( + buildOpenAICompat(glm52("zhipu-coding-plan", "https://open.bigmodel.cn/api/coding/paas/v4")) + .streamIdleTimeoutMs, + ).toBe(600_000); + expect(buildOpenAICompat(glm52("opencode-go", "https://opencode.ai/zen/go/v1")).streamIdleTimeoutMs).toBe( + 600_000, + ); + expect(buildOpenAICompat(glm52("opencode-zen", "https://opencode.ai/zen/v1")).streamIdleTimeoutMs).toBe(600_000); + }); + + it("does not widen non-GLM models on the OpenCode gateway via the GLM floor", () => { + const kimi = buildOpenAICompat({ + ...baseModel, + id: "kimi-k2.5", + name: "Kimi K2.5", + provider: "opencode-go", + baseUrl: "https://opencode.ai/zen/go/v1", + }); + expect(kimi.streamIdleTimeoutMs).toBeUndefined(); + }); +}); + describe("zhipu-coding-plan model discovery", () => { it("uses the dedicated Coding Plan endpoint by default", async () => { let requestedUrl = ""; From e9dc1616b7f51b2196f01ba000397f02ee250dca Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:01:09 +0000 Subject: [PATCH 002/293] fix(tui): anchored IME cursors in interactive inputs Preserved the editor cursor marker under autocomplete and forwarded dialog focus into Other-response editors. Fixes #4760 --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../src/modes/components/hook-editor.ts | 19 +++++++++++++++++-- .../coding-agent/test/hook-editor.test.ts | 14 +++++++++++++- packages/tui/CHANGELOG.md | 4 ++++ packages/tui/src/components/editor.ts | 5 +++-- packages/tui/test/editor.test.ts | 9 ++++++++- 6 files changed, 49 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 459ec14a8..7857bf095 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `Other` response editors leaving Windows Terminal IME candidate windows at the terminal edge by forwarding dialog focus to the nested editor ([#4760](https://github.com/can1357/oh-my-pi/issues/4760)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/modes/components/hook-editor.ts b/packages/coding-agent/src/modes/components/hook-editor.ts index 19fe74c18..1abc5ce73 100644 --- a/packages/coding-agent/src/modes/components/hook-editor.ts +++ b/packages/coding-agent/src/modes/components/hook-editor.ts @@ -7,7 +7,7 @@ * (Ctrl+Q / Ctrl+Enter) submits, bordered popup * - Prompt-style (ask): Enter submits, Shift+Enter inserts newline, legacy ask chrome */ -import { Container, Editor, matchesKey, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; +import { Container, Editor, type Focusable, matchesKey, Spacer, Text, type TUI } from "@oh-my-pi/pi-tui"; import { getEditorTheme, theme } from "../../modes/theme/theme"; import { matchesAppExternalEditor, @@ -22,12 +22,15 @@ export interface HookEditorOptions { promptStyle?: boolean; } -export class HookEditorComponent extends Container { +/** Interactive multiline dialog used by hooks and the ask tool's Other response. */ +export class HookEditorComponent extends Container implements Focusable { #editor: Editor; #onSubmitCallback: (value: string) => void; #onCancelCallback: () => void; #tui: TUI; #promptStyle: boolean; + /** Focus state mirrored to the nested editor during rendering. */ + focused = false; constructor( tui: TUI, @@ -75,6 +78,18 @@ export class HookEditorComponent extends Container { this.addChild(new DynamicBorder()); } + /** Keep the nested editor's software/hardware cursor mode aligned with the dialog focus target. */ + setUseTerminalCursor(useTerminalCursor: boolean): void { + if (this.#editor.getUseTerminalCursor() === useTerminalCursor) return; + this.#editor.setUseTerminalCursor(useTerminalCursor); + } + + /** Render the dialog after forwarding its focus state to the nested editor. */ + override render(width: number): readonly string[] { + this.#editor.focused = this.focused; + return super.render(width); + } + handleInput(keyData: string): void { if (this.#promptStyle) { this.#handlePromptStyleInput(keyData); diff --git a/packages/coding-agent/test/hook-editor.test.ts b/packages/coding-agent/test/hook-editor.test.ts index a42ace9fc..ede312a0c 100644 --- a/packages/coding-agent/test/hook-editor.test.ts +++ b/packages/coding-agent/test/hook-editor.test.ts @@ -4,7 +4,7 @@ import { HookEditorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/ import { ExtensionUiController } from "@oh-my-pi/pi-coding-agent/modes/controllers/extension-ui-controller"; import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; -import { setKeybindings, type TUI } from "@oh-my-pi/pi-tui"; +import { CURSOR_MARKER, isFocusable, setKeybindings, type TUI } from "@oh-my-pi/pi-tui"; beforeAll(async () => { const theme = await getThemeByName("dark"); @@ -364,6 +364,18 @@ describe("HookEditorComponent prompt-style mode", () => { expect(rendered).toContain("ctrl+g external editor"); }); + it("anchors the hardware cursor while entering an Other response", () => { + const component = new HookEditorComponent(createTui(), "Prompt", undefined, vi.fn(), vi.fn(), { + promptStyle: true, + }); + if (!isFocusable(component)) throw new Error("Hook editor must forward focus to its inner editor"); + + component.focused = true; + component.setUseTerminalCursor?.(true); + + expect(component.render(120).some(line => line.includes(CURSOR_MARKER))).toBe(true); + }); + it("keeps the prompt gutter visible after typing in prompt-style mode", () => { const component = new HookEditorComponent(createTui(), "Prompt", undefined, vi.fn(), vi.fn(), { promptStyle: true, diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index b6d4864c9..5e9d45a45 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed autocomplete popups moving Windows Terminal IME candidate windows away from the prompt by keeping the terminal cursor anchored at the text insertion point ([#4760](https://github.com/can1357/oh-my-pi/issues/4760)). + ## [16.3.10] - 2026-07-06 ### Fixed diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 8a631d790..f433aa037 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -852,8 +852,9 @@ export class Editor implements Component, Focusable { } // Render each layout line - // Emit hardware cursor marker only when focused and not showing autocomplete - const emitCursorMarker = this.focused && !this.#autocompleteState; + // Keep the hardware cursor at the text insertion point while autocomplete + // rows render below it; terminals use that position to anchor IME candidates. + const emitCursorMarker = this.focused; const lineContentWidth = contentAreaWidth; // Compute inline hint text (dim ghost text after cursor) diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index d451c758e..2bb6ad654 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -289,9 +289,12 @@ describe("Editor component", () => { }); describe("autocomplete triggers", () => { - it("triggers slash-command autocomplete when typing slash", async () => { + it("triggers slash-command autocomplete without losing the hardware cursor anchor", async () => { const editor = new Editor(defaultEditorTheme); + editor.focused = true; + editor.setUseTerminalCursor(true); const { promise, resolve } = Promise.withResolvers(); + const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers(); editor.setAutocompleteProvider({ async getSuggestions(lines, cursorLine, cursorCol) { @@ -303,10 +306,14 @@ describe("Editor component", () => { return { lines, cursorLine, cursorCol }; }, }); + editor.onAutocompleteUpdate = resolveAutocompleteUpdated; editor.handleInput("/"); await expect(promise).resolves.toBe("/"); + await autocompleteUpdated; + expect(editor.isShowingAutocomplete()).toBe(true); + expect(editor.render(80).some(line => line.includes(CURSOR_MARKER))).toBe(true); }); it("triggers file-reference autocomplete when typing at-sign", async () => { From a15817d8689d581a6f2059a370dd7c9dfdd74cc5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:03:22 +0000 Subject: [PATCH 003/293] fix(tui): preserved drafts on skill autocomplete - Kept Enter acceptance non-submitting for mid-prompt skill completions. - Added regression coverage for preserving the surrounding composer draft. Fixes #4773 --- packages/tui/CHANGELOG.md | 4 +++ packages/tui/src/autocomplete.ts | 2 +- packages/tui/src/components/editor.ts | 7 +++-- .../test/editor-autocomplete-actions.test.ts | 29 +++++++++++++++++++ 4 files changed, 38 insertions(+), 4 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index b6d4864c9..7cf35cce0 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Enter accepting a mid-prompt `/skill:` autocomplete from submitting and clearing the draft; acceptance now inserts the skill token and leaves the prompt open ([#4773](https://github.com/can1357/oh-my-pi/issues/4773)). + ## [16.3.10] - 2026-07-06 ### Fixed diff --git a/packages/tui/src/autocomplete.ts b/packages/tui/src/autocomplete.ts index 99c111aed..01e1b93b5 100644 --- a/packages/tui/src/autocomplete.ts +++ b/packages/tui/src/autocomplete.ts @@ -421,7 +421,7 @@ export class CombinedAutocompleteProvider implements AutocompleteProvider { // Preserve the full text-before-cursor for submitted slash // commands so the editor's Enter-staleness check still applies // completion for ` /sk`. Mid-prompt skill lookup keeps only - // the slash token because accepting it replaces the whole draft. + // the slash token because acceptance replaces only that token. prefix: isMidPromptSkillLookup ? commandText : textBeforeCursor, }; } diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 8a631d790..84cc1b144 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -1159,11 +1159,12 @@ export class Editor implements Component, Focusable { return; } - // If Enter was pressed on a slash command (not an absolute-path - // completion sharing the leading-slash prefix), apply and submit + // If Enter was pressed on a submitted slash command (not an absolute-path + // completion sharing the leading-slash prefix), apply and submit. if ( (kb.matches(data, "tui.input.submit") || data === "\n") && findLeadingSlashCommandStart(this.#autocompletePrefix) !== null && + this.#isInSubmittedSlashCommandContext() && !this.#selectedCompletionIsPath() ) { // Check for stale autocomplete state due to debounce @@ -1192,7 +1193,7 @@ export class Editor implements Component, Focusable { } // Don't return - fall through to submission logic } - // If Enter was pressed on a file path, apply completion + // Otherwise, apply the completion without submitting the surrounding draft. else if (kb.matches(data, "tui.input.submit") || data === "\n") { // Check for stale autocomplete state due to buffer edits since last refresh. const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; diff --git a/packages/tui/test/editor-autocomplete-actions.test.ts b/packages/tui/test/editor-autocomplete-actions.test.ts index 626d82acf..3d39ed516 100644 --- a/packages/tui/test/editor-autocomplete-actions.test.ts +++ b/packages/tui/test/editor-autocomplete-actions.test.ts @@ -232,6 +232,35 @@ describe("Editor Enter handler sync slash completion", () => { expect(editor.getText()).toBe("explain this\n/skill:security-scan "); }); + it("inserts a mid-prompt skill token without submitting on Enter", async () => { + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider( + [ + { name: "skill:security-scan", description: "Security scan" }, + { name: "model", description: "Switch model" }, + ], + "/tmp", + ), + ); + let submitted: string | undefined; + editor.onSubmit = text => { + submitted = text; + }; + + editor.setText("explain this\n"); + editor.handleInput("/"); + await Promise.resolve(); + + expect(editor.isShowingAutocomplete()).toBe(true); + + editor.handleInput("security"); + editor.handleInput("\r"); + + expect(editor.getText()).toBe("explain this\n/skill:security-scan "); + expect(submitted).toBeUndefined(); + }); + it("preserves Tab file completion for an absolute path token after prose", async () => { let forceFileCalls = 0; const editor = new Editor(defaultEditorTheme); From 43a89d20bdca206f78c471773be31cc562a7e198 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:03:44 +0000 Subject: [PATCH 004/293] fix(editor): accepted upstream-pi editor constructor in CustomEditor Plugins subclass CustomEditor/Editor and forward the upstream-pi super(tui, theme, keybindings) constructor, which is the arg order setEditorComponent's factory contract advertises. omp's base constructor is constructor(theme), so the TUI landed in the theme slot and every render threw "undefined is not an object (evaluating 'this.#theme.symbols.boxRound')". CustomEditor now resolves the real EditorTheme by shape rather than position and captures a leading TUI so plugin overrides calling this.tui.requestRender() keep working. Fixes #4766 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../custom-editor-plugin-ctor.test.ts | 36 +++++++++++ .../src/modes/components/custom-editor.ts | 63 ++++++++++++++++++- 3 files changed, 102 insertions(+), 1 deletion(-) create mode 100644 packages/coding-agent/src/modes/components/custom-editor-plugin-ctor.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 459ec14a8..9d7660d88 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed omp crashing at startup (`TypeError: undefined is not an object (evaluating 'this.#theme.symbols.boxRound')`) after installing a plugin whose custom editor subclasses `CustomEditor`/`Editor` and forwards the upstream-pi `super(tui, theme, keybindings)` constructor — the arg order that `setEditorComponent`'s factory contract advertises. `CustomEditor` now resolves the real `EditorTheme` by shape rather than position and captures a leading `TUI` for plugin overrides ([#4766](https://github.com/can1357/oh-my-pi/issues/4766)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/modes/components/custom-editor-plugin-ctor.test.ts b/packages/coding-agent/src/modes/components/custom-editor-plugin-ctor.test.ts new file mode 100644 index 000000000..6b92ef628 --- /dev/null +++ b/packages/coding-agent/src/modes/components/custom-editor-plugin-ctor.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from "bun:test"; +import { ProcessTerminal, TUI } from "@oh-my-pi/pi-tui"; +import { getEditorTheme, initTheme } from "../theme/theme"; +import { CustomEditor } from "./custom-editor"; + +/** + * Regression for issue #4766: plugins written against upstream pi subclass + * `CustomEditor`/`Editor` and forward `super(tui, theme, keybindings)`. omp's + * `setEditorComponent` factory contract advertises exactly that arg order, so + * the base constructor must resolve the real theme by shape (not position) or + * every render throws `undefined is not an object (evaluating + * 'this.#theme.symbols.boxRound')`. + */ +describe("CustomEditor upstream-pi constructor compatibility (#4766)", () => { + it("renders when constructed as (tui, theme, keybindings)", async () => { + await initTheme(); + const tui = new TUI(new ProcessTerminal()); + const editor = new CustomEditor(tui, getEditorTheme(), {}); + editor.setText("run this workflow"); + expect(() => editor.render(80)).not.toThrow(); + // The rounded border glyphs from the resolved theme must reach the frame. + const frame = editor.render(80).join("\n"); + expect(frame).toContain(getEditorTheme().symbols.boxRound.horizontal); + // The leading TUI is captured so plugin overrides calling + // `this.tui.requestRender()` keep working. + expect(editor.tui).toBe(tui); + }); + + it("still accepts omp's own (theme) constructor", async () => { + await initTheme(); + const editor = new CustomEditor(getEditorTheme()); + editor.setText("hello"); + expect(() => editor.render(80)).not.toThrow(); + expect(editor.tui).toBeUndefined(); + }); +}); diff --git a/packages/coding-agent/src/modes/components/custom-editor.ts b/packages/coding-agent/src/modes/components/custom-editor.ts index bf144ff11..6af27dfcf 100644 --- a/packages/coding-agent/src/modes/components/custom-editor.ts +++ b/packages/coding-agent/src/modes/components/custom-editor.ts @@ -1,6 +1,15 @@ import { fileURLToPath } from "node:url"; import type { ImageContent } from "@oh-my-pi/pi-ai"; -import { addKeyAliases, canonicalKeyId, Editor, type KeyId, parseKey, parseKittySequence } from "@oh-my-pi/pi-tui"; +import { + addKeyAliases, + canonicalKeyId, + Editor, + type EditorTheme, + type KeyId, + parseKey, + parseKittySequence, + TUI, +} from "@oh-my-pi/pi-tui"; import { BracketedPasteHandler } from "@oh-my-pi/pi-tui/bracketed-paste"; import type { AppKeybinding } from "../../config/keybindings"; import { isSettingsInitialized, settings } from "../../config/settings"; @@ -283,6 +292,31 @@ export function extractImagePathFromText(text: string): string | undefined { return undefined; } +/** + * Resolve the {@link EditorTheme} from a `CustomEditor`/`Editor` constructor + * argument list, tolerating both the omp `(theme)` and upstream-pi + * `(tui, theme, keybindings)` conventions (see {@link CustomEditor}'s + * constructor). A real `EditorTheme` is identified structurally — it exposes a + * `borderColor` function and a `symbols` object — so a `TUI` passed in the first + * slot is skipped rather than mistaken for the theme. + */ +function pickEditorTheme(args: readonly unknown[]): EditorTheme { + for (const arg of args) { + if (isEditorTheme(arg)) return arg; + } + // Fall back to the first argument so a caller passing a bare theme that + // somehow fails the shape probe still reaches the base constructor. + return args[0] as EditorTheme; +} + +function isEditorTheme(value: unknown): value is EditorTheme { + if (typeof value !== "object" || value === null) return false; + const candidate = value as Partial; + return ( + typeof candidate.borderColor === "function" && typeof candidate.symbols === "object" && candidate.symbols !== null + ); +} + /** * Custom editor that handles configurable app-level shortcuts for coding-agent. */ @@ -296,6 +330,33 @@ export class CustomEditor extends Editor { * `undefined` entries are images without a backing reference yet. */ pendingImageLinks: (string | undefined)[] = []; + /** + * The host {@link TUI}, captured when a plugin constructs this editor through + * the upstream-pi `(tui, theme, keybindings)` convention. Undefined for omp's + * own `new CustomEditor(theme)` callers (they drive repaints through the + * interactive-mode wiring instead). Plugins that call `this.tui.requestRender()` + * in their overrides read it here (issue #4766). + */ + tui?: TUI; + + /** + * Accept both the omp constructor convention — `new CustomEditor(theme)` — + * and the upstream-pi `Editor` convention — `new Editor(tui, theme, keybindings)` + * — that {@link ExtensionUIContext.setEditorComponent}'s factory contract + * advertises `(tui, theme, keybindings)`. Plugins written against upstream pi + * subclass `CustomEditor`/`Editor` and forward `super(tui, theme, keybindings)`; + * without this shim the `TUI` lands in the `theme` slot and every render throws + * `undefined is not an object (evaluating 'this.#theme.symbols.boxRound')` + * (issue #4766). We locate the real {@link EditorTheme} among the args by shape + * (it carries `symbols`/`borderColor`) rather than by position, and capture a + * leading {@link TUI} so plugin overrides calling `this.tui.requestRender()` + * keep working. + */ + constructor(...args: readonly unknown[]) { + super(pickEditorTheme(args)); + if (args[0] instanceof TUI) this.tui = args[0]; + } + /** Clear the composer draft: optionally commit `historyText` to history, then * reset the editor text and all pending draft-image state. The shared tail of * every "message submitted" path; pass no argument for a plain discard. */ From b6b947bdbe9b65990401120d6ad7d823236c3b0e Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:15:53 +0000 Subject: [PATCH 005/293] fix(openai): rendered native response images - Normalized completed image_generation_call results into assistant image blocks. - Persisted image bytes through the session blob store and rendered them in live, replay, ACP, proxy, telemetry, and HTML paths. - Added response normalization, persistence, and TUI rendering regressions. Fixes #4768 --- packages/agent/src/agent-loop.ts | 3 + packages/agent/src/proxy.ts | 11 ++++ packages/agent/src/telemetry.ts | 3 + packages/ai/src/dialect/owned-stream.ts | 11 ++++ packages/ai/src/providers/openai-shared.ts | 14 +++++ packages/ai/src/types.ts | 10 +++- .../ai/src/utils/empty-completion-retry.ts | 9 ++- .../ai/src/utils/leaked-thinking-stream.ts | 20 ++++++- .../openai-responses-stream-terminal.test.ts | 35 ++++++++++++ packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/cli/bench-cli.ts | 5 +- .../coding-agent/src/export/html/template.js | 2 + .../coding-agent/src/modes/acp/acp-agent.ts | 17 ++++++ .../src/modes/acp/acp-event-mapper.ts | 8 +++ .../src/modes/components/assistant-message.ts | 36 ++++++++---- .../components/chat-transcript-builder.ts | 4 +- .../utils/interactive-context-helpers.ts | 7 ++- .../modes/utils/transcript-render-helpers.ts | 5 +- .../coding-agent/src/session/agent-session.ts | 15 +++-- .../src/session/session-listing.ts | 11 ++-- .../src/session/session-loader.ts | 9 +++ .../src/session/session-persistence.ts | 11 ++++ .../agent-session-eager-compaction.test.ts | 9 +-- .../test/agent-session-eager-task.test.ts | 9 +-- .../test/agent-session-eager-todo.test.ts | 9 +-- ...nt-session-openai-responses-replay.test.ts | 5 +- ...gent-session-plan-mode-convergence.test.ts | 9 +-- ...-session-plan-reference-compaction.test.ts | 9 +-- .../test/agent-session-skill-keywords.test.ts | 10 ++-- .../assistant-message-mermaid.test.ts | 14 ++++- .../test/session-persistence-images.test.ts | 57 +++++++++++++++++++ 31 files changed, 323 insertions(+), 58 deletions(-) diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 4c8f7b64f..683db07b3 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -152,6 +152,7 @@ type AssistantToolCallBlock = Extract( } closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id)); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); + } else if (item.type === "image_generation_call" && item.status === "completed" && item.result) { + const image: ImageContent = { + type: "image", + data: item.result, + mimeType: parseImageMetadata(Buffer.from(item.result, "base64"))?.mimeType ?? "image/png", + }; + output.content.push(image); + stream.push({ + type: "image_end", + contentIndex: output.content.length - 1, + content: image, + partial: output, + }); } } else if (event.type === "response.completed" || event.type === "response.incomplete") { const response = event.response; diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index a34df07c8..e096d8983 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -696,7 +696,14 @@ export interface ContextSnapshot { export interface AssistantMessage { role: "assistant"; - content: (TextContent | ThinkingContent | RedactedThinkingContent | AnthropicFallbackContent | ToolCall)[]; + content: ( + | TextContent + | ThinkingContent + | RedactedThinkingContent + | AnthropicFallbackContent + | ImageContent + | ToolCall + )[]; api: Api; provider: Provider; model: string; @@ -881,6 +888,7 @@ export type AssistantMessageEvent = | { type: "thinking_start"; contentIndex: number; partial: AssistantMessage } | { type: "thinking_delta"; contentIndex: number; delta: string; partial: AssistantMessage } | { type: "thinking_end"; contentIndex: number; content: string; partial: AssistantMessage } + | { type: "image_end"; contentIndex: number; content: ImageContent; partial: AssistantMessage } | { type: "toolcall_start"; contentIndex: number; partial: AssistantMessage } | { type: "toolcall_delta"; contentIndex: number; delta: string; partial: AssistantMessage } | { type: "toolcall_end"; contentIndex: number; toolCall: ToolCall; partial: AssistantMessage } diff --git a/packages/ai/src/utils/empty-completion-retry.ts b/packages/ai/src/utils/empty-completion-retry.ts index 6ace44c1f..3af9337e5 100644 --- a/packages/ai/src/utils/empty-completion-retry.ts +++ b/packages/ai/src/utils/empty-completion-retry.ts @@ -27,12 +27,13 @@ export const EMPTY_COMPLETION_BASE_DELAY_MS = 500; const NON_WHITESPACE_RE = /\S/; /** - * Whether a completed assistant message carries content worth delivering: a tool - * call or any non-whitespace text. An empty/whitespace-only message — or one - * that only ever produced thinking — is the "empty response" failure. + * Whether a completed assistant message carries content worth delivering: an + * image, tool call, or any non-whitespace text. An empty/whitespace-only message + * — or one that only ever produced thinking — is the "empty response" failure. */ export function hasVisibleAssistantContent(message: AssistantMessage): boolean { for (const block of message.content) { + if (block.type === "image") return true; if (block.type === "toolCall") return true; if (block.type === "text" && NON_WHITESPACE_RE.test(block.text)) return true; } @@ -49,6 +50,8 @@ function isMeaningfulCompletionEvent(event: AssistantMessageEvent): boolean { case "text_end": case "thinking_end": return event.content.length > 0; + case "image_end": + return true; case "toolcall_start": case "toolcall_end": return true; diff --git a/packages/ai/src/utils/leaked-thinking-stream.ts b/packages/ai/src/utils/leaked-thinking-stream.ts index 074a4c507..6acd5e961 100644 --- a/packages/ai/src/utils/leaked-thinking-stream.ts +++ b/packages/ai/src/utils/leaked-thinking-stream.ts @@ -25,7 +25,7 @@ * events are forwarded verbatim. */ -import type { AssistantMessage, TextContent, ThinkingContent, ToolCall } from "../types"; +import type { AssistantMessage, ImageContent, TextContent, ThinkingContent, ToolCall } from "../types"; import { clearStreamingPartialJson, getStreamingPartialJson, @@ -78,6 +78,10 @@ export function wrapLeakedThinkingStream(inner: AssistantMessageEventStream): As projector.thinking(event.delta, block?.type === "thinking" ? block.thinkingSignature : undefined); break; } + case "image_end": + projector ??= new LeakedThinkingProjector(out, event.partial); + projector.image(event.content); + break; case "toolcall_start": { projector ??= new LeakedThinkingProjector(out, event.partial); const block = event.partial.content[event.contentIndex]; @@ -163,6 +167,20 @@ class LeakedThinkingProjector { this.#out.push({ type: "thinking_delta", contentIndex: index, delta, partial: this.#partial }); } + /** Forward a completed native image after releasing held text. */ + image(content: ImageContent): void { + this.#apply(this.#healer.flushEvents(), this.#lastTextSignature); + this.#closeText(); + this.#closeThinking(); + this.#partial.content.push(content); + this.#out.push({ + type: "image_end", + contentIndex: this.#partial.content.length - 1, + content, + partial: this.#partial, + }); + } + /** Forward a native tool call's start, releasing any held-back text first. */ toolStart(srcIndex: number, source: StreamingToolCall | undefined): void { if (!source) return; diff --git a/packages/ai/test/openai-responses-stream-terminal.test.ts b/packages/ai/test/openai-responses-stream-terminal.test.ts index 95c501834..a3bc28afc 100644 --- a/packages/ai/test/openai-responses-stream-terminal.test.ts +++ b/packages/ai/test/openai-responses-stream-terminal.test.ts @@ -277,6 +277,41 @@ describe("processResponsesStream: lost output_item.added recovery", () => { expect(end?.content).toBe("Recovered text"); }); + test("normalizes a completed native image generation call into visible assistant content", async () => { + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (event: unknown) => emitted.push(event as EmittedEvent), end: () => {} } as never; + const data = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+A8AAQUBAScY42YAAAAASUVORK5CYII="; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.done", + output_index: 0, + item: { + type: "image_generation_call", + id: "ig_1", + status: "completed", + result: data, + }, + }, + { type: "response.completed", response: { id: "resp_image", status: "completed" } }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.content).toEqual([{ type: "image", data, mimeType: "image/png" }]); + const end = emitted.find(event => event.type === "image_end"); + expect(end).toEqual({ + type: "image_end", + contentIndex: 0, + content: { type: "image", data, mimeType: "image/png" }, + partial: output, + }); + }); + test("routes reasoning finalization by output_index when item ids are absent", async () => { const output = makeOutput(); const stream = { push: () => {}, end: () => {} } as never; diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 459ec14a8..dde4b82bb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Rendered and persisted native OpenAI Responses `image_generation_call` results as session images ([#4768](https://github.com/can1357/oh-my-pi/issues/4768)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/cli/bench-cli.ts b/packages/coding-agent/src/cli/bench-cli.ts index fd635e43f..ad342c421 100644 --- a/packages/coding-agent/src/cli/bench-cli.ts +++ b/packages/coding-agent/src/cli/bench-cli.ts @@ -159,12 +159,14 @@ function isFirstTokenEvent(event: AssistantMessageEvent): boolean { case "text_end": case "thinking_end": return event.content.length > 0; + case "image_end": + return true; default: return false; } } -/** Final message carries visible output — non-empty text/thinking or a tool call. */ +/** Final message carries visible output — non-empty text/thinking, an image, or a tool call. */ function hasVisibleFinalContent(message: AssistantMessage): boolean { return message.content.some(block => { switch (block.type) { @@ -172,6 +174,7 @@ function hasVisibleFinalContent(message: AssistantMessage): boolean { return block.text.length > 0; case "thinking": return block.thinking.length > 0; + case "image": case "redactedThinking": case "toolCall": return true; diff --git a/packages/coding-agent/src/export/html/template.js b/packages/coding-agent/src/export/html/template.js index 68fa096ab..d971daa66 100644 --- a/packages/coding-agent/src/export/html/template.js +++ b/packages/coding-agent/src/export/html/template.js @@ -1082,6 +1082,8 @@
${escapeHtml(thinking)}
Thinking ...
`; + } else if (block.type === 'image') { + html += `
`; } } for (const block of msg.content) { diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index 8d4083c6c..3ab9b35e6 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -1982,6 +1982,23 @@ export class AcpAgent implements Agent { }); continue; } + if ( + item.type === "image" && + "data" in item && + typeof item.data === "string" && + "mimeType" in item && + typeof item.mimeType === "string" + ) { + notifications.push({ + sessionId, + update: { + sessionUpdate: "agent_message_chunk", + content: { type: "image", data: item.data, mimeType: item.mimeType }, + messageId, + }, + }); + continue; + } if (item.type === "thinking" && "thinking" in item && typeof item.thinking === "string") { const thinking = canonicalizeMessage(item.thinking); if (thinking.length === 0) continue; diff --git a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts index bde460d11..82a6c9773 100644 --- a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts +++ b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts @@ -254,6 +254,14 @@ function mapAssistantMessageUpdate( let text: string; const progress = options.getMessageProgress?.(event.message); switch (event.assistantMessageEvent.type) { + case "image_end": + return [ + toSessionNotification(sessionId, { + sessionUpdate: "agent_message_chunk", + content: event.assistantMessageEvent.content, + messageId: options.getMessageId?.(event.message), + }), + ]; case "text_delta": sessionUpdate = "agent_message_chunk"; text = event.assistantMessageEvent.delta; diff --git a/packages/coding-agent/src/modes/components/assistant-message.ts b/packages/coding-agent/src/modes/components/assistant-message.ts index e09fc8c15..75391dceb 100644 --- a/packages/coding-agent/src/modes/components/assistant-message.ts +++ b/packages/coding-agent/src/modes/components/assistant-message.ts @@ -173,6 +173,7 @@ export class AssistantMessageComponent extends Container { #lastMessage?: AssistantMessage; #toolImagesByCallId = new Map(); #convertedKittyImages = new Map(); + #showImages = true; #kittyConversionsInFlight = new Set(); #transcriptBlockFinalized: boolean; /** @@ -497,6 +498,15 @@ export class AssistantMessageComponent extends Container { } } + /** Toggle rendering for assistant-native and tool-result images. */ + setImagesVisible(visible: boolean): void { + if (this.#showImages === visible) return; + this.#showImages = visible; + if (this.#lastMessage) { + this.updateContent(this.#lastMessage, { transient: this.#lastUpdateTransient }); + } + } + setToolResultImages(toolCallId: string, images: ImageContent[]): void { if (!toolCallId) return; const validImages = images.filter(img => img.type === "image" && img.data && img.mimeType); @@ -514,19 +524,17 @@ export class AssistantMessageComponent extends Container { this.#toolImagesByCallId.delete(toolCallId); } else { this.#toolImagesByCallId.set(toolCallId, validImages); - this.#convertToolImagesForKitty(toolCallId, validImages); + this.#convertImagesForKitty(validImages.map((image, index) => ({ image, key: `${toolCallId}:${index}` }))); } if (this.#lastMessage) { this.updateContent(this.#lastMessage, { transient: this.#lastUpdateTransient }); } } - #convertToolImagesForKitty(toolCallId: string, images: ImageContent[]): void { + #convertImagesForKitty(entries: Array<{ image: ImageContent; key: string }>): void { if (TERMINAL.imageProtocol !== ImageProtocol.Kitty) return; - for (let index = 0; index < images.length; index++) { - const image = images[index]; - if (!image || image.mimeType === "image/png") continue; - const key = `${toolCallId}:${index}`; + for (const { image, key } of entries) { + if (image.mimeType === "image/png") continue; if (this.#convertedKittyImages.has(key) || this.#kittyConversionsInFlight.has(key)) continue; this.#kittyConversionsInFlight.add(key); new Bun.Image(Buffer.from(image.data, "base64")) @@ -550,11 +558,19 @@ export class AssistantMessageComponent extends Container { } } - #renderToolImages(): void { - const imageEntries = Array.from(this.#toolImagesByCallId.entries()).flatMap(([toolCallId, images]) => + #renderImages(message: AssistantMessage): void { + if (!this.#showImages) return; + const nativeEntries = message.content.flatMap((content, index) => + content.type === "image" && content.data && content.mimeType + ? [{ image: content, key: `native:${index}` }] + : [], + ); + const toolEntries = Array.from(this.#toolImagesByCallId.entries()).flatMap(([toolCallId, images]) => images.map((image, index) => ({ image, key: `${toolCallId}:${index}` })), ); + const imageEntries = [...nativeEntries, ...toolEntries]; if (imageEntries.length === 0) return; + this.#convertImagesForKitty(imageEntries); this.#contentContainer.addChild(new Spacer(1)); for (const { image, key } of imageEntries) { @@ -620,7 +636,7 @@ export class AssistantMessageComponent extends Container { #canFastPath(message: AssistantMessage): boolean { for (const content of message.content) { - if (content.type === "toolCall") return false; + if (content.type === "toolCall" || content.type === "image") return false; } if (this.#toolImagesByCallId.size > 0) return false; const errorPresentation = resolveAssistantErrorPresentation(message); @@ -826,7 +842,7 @@ export class AssistantMessageComponent extends Container { this.#stopThinkingAnimation(); } - this.#renderToolImages(); + this.#renderImages(message); const errorPresentation = resolveAssistantErrorPresentation(message); const hasToolCalls = message.content.some(c => c.type === "toolCall"); if (errorPresentation.kind === "compact-recovered") { diff --git a/packages/coding-agent/src/modes/components/chat-transcript-builder.ts b/packages/coding-agent/src/modes/components/chat-transcript-builder.ts index 3d6c810cc..8e8bd02bb 100644 --- a/packages/coding-agent/src/modes/components/chat-transcript-builder.ts +++ b/packages/coding-agent/src/modes/components/chat-transcript-builder.ts @@ -274,13 +274,15 @@ export class ChatTranscriptBuilder { const hideThinkingBlock = this.deps.hideThinkingBlock?.() ?? false; const proseOnlyThinking = this.deps.proseOnlyThinking ? this.deps.proseOnlyThinking() : true; const assistantComponent = new AssistantMessageComponent( - message, + undefined, hideThinkingBlock, () => this.deps.requestRender(), this.deps.getMessageRenderer ? undefined : [], // placeholder for thinkingRenderers undefined, // placeholder for imageBudget proseOnlyThinking, ); + assistantComponent.setImagesVisible(settings.get("terminal.showImages")); + assistantComponent.updateContent(message); this.container.addChild(assistantComponent); if (settings.get("display.cacheMissMarker")) { diff --git a/packages/coding-agent/src/modes/utils/interactive-context-helpers.ts b/packages/coding-agent/src/modes/utils/interactive-context-helpers.ts index e6c3f4ae2..525daa8ca 100644 --- a/packages/coding-agent/src/modes/utils/interactive-context-helpers.ts +++ b/packages/coding-agent/src/modes/utils/interactive-context-helpers.ts @@ -16,12 +16,15 @@ export function createAssistantMessageComponent( ctx: InteractiveModeContext, message?: AssistantMessage, ): AssistantMessageComponent { - return new AssistantMessageComponent( - message, + const component = new AssistantMessageComponent( + undefined, ctx.effectiveHideThinkingBlock, () => ctx.ui.requestRender(), ctx.viewSession.extensionRunner?.getAssistantThinkingRenderers(), ctx.ui.imageBudget, ctx.proseOnlyThinking, ); + component.setImagesVisible(ctx.settings.get("terminal.showImages")); + if (message) component.updateContent(message); + return component; } diff --git a/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts b/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts index 68b9be1d2..5ee5a9c52 100644 --- a/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts +++ b/packages/coding-agent/src/modes/utils/transcript-render-helpers.ts @@ -125,12 +125,13 @@ export function buildFileMentionBlock(files: FileMentionMessage["files"], indent } /** - * Whether an assistant turn has visible text or thinking content (after - * canonicalization) — i.e. content that closes the current read-tool run. + * Whether an assistant turn has visible text, thinking, or image content — i.e. + * content that closes the current read-tool run. */ export function assistantHasVisibleContent(message: AssistantAgentMessage): boolean { return message.content.some( content => + content.type === "image" || (content.type === "text" && canonicalizeMessage(content.text)) || (content.type === "thinking" && canonicalizeMessage(content.thinking)), ); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index c11d0bf36..d9ad2a5dd 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1362,15 +1362,20 @@ function queuedTextContent(message: AgentMessage): string | undefined { if (!("content" in message)) return undefined; const content = message.content; if (typeof content === "string") return content; - return content.find((part): part is TextContent => part.type === "text")?.text; + for (const part of content) { + if (part.type === "text") return part.text; + } + return undefined; } function queuedImageContent(message: AgentMessage): ImageContent[] | undefined { if (!("content" in message) || typeof message.content === "string") return undefined; - const images = message.content.filter( - (part): part is ImageContent => - part.type === "image" && typeof part.data === "string" && typeof part.mimeType === "string", - ); + const images: ImageContent[] = []; + for (const part of message.content) { + if (part.type === "image" && typeof part.data === "string" && typeof part.mimeType === "string") { + images.push(part); + } + } return images.length > 0 ? images : undefined; } diff --git a/packages/coding-agent/src/session/session-listing.ts b/packages/coding-agent/src/session/session-listing.ts index 4b554fddb..586cbd825 100644 --- a/packages/coding-agent/src/session/session-listing.ts +++ b/packages/coding-agent/src/session/session-listing.ts @@ -1,6 +1,6 @@ import * as os from "node:os"; import * as path from "node:path"; -import type { Message, TextContent } from "@oh-my-pi/pi-ai"; +import type { Message } from "@oh-my-pi/pi-ai"; import { getAgentDir as getDefaultAgentDir, logger, parseJsonlLenient, toError } from "@oh-my-pi/pi-utils"; import { computeDefaultSessionDir } from "./session-paths"; import { FileSessionStorage, type SessionStorage } from "./session-storage"; @@ -108,10 +108,11 @@ function sessionDisplayName(info: SessionInfo): string { function extractTextFromContent(content: Message["content"]): string { if (typeof content === "string") return content; - return content - .filter((block): block is TextContent => block.type === "text") - .map(block => block.text) - .join(" "); + const text: string[] = []; + for (const block of content) { + if (block.type === "text") text.push(block.text); + } + return text.join(" "); } /** diff --git a/packages/coding-agent/src/session/session-loader.ts b/packages/coding-agent/src/session/session-loader.ts index 0915db7b5..9950568ac 100644 --- a/packages/coding-agent/src/session/session-loader.ts +++ b/packages/coding-agent/src/session/session-loader.ts @@ -252,6 +252,15 @@ async function resolvePersistedBlobRefs(value: unknown, blobStore: BlobStore, ke } if (typeof value !== "object" || value === null) return; + if ( + "type" in value && + value.type === "image_generation_call" && + "result" in value && + typeof value.result === "string" && + isBlobRef(value.result) + ) { + value.result = await resolveImageData(blobStore, value.result); + } if (hasImageUrl(value) && isBlobRef(value.image_url)) { value.image_url = await resolveImageDataUrl(blobStore, value.image_url); diff --git a/packages/coding-agent/src/session/session-persistence.ts b/packages/coding-agent/src/session/session-persistence.ts index 68a2e3455..cc2d6fdfb 100644 --- a/packages/coding-agent/src/session/session-persistence.ts +++ b/packages/coding-agent/src/session/session-persistence.ts @@ -79,6 +79,17 @@ function isNonEmptyString(value: unknown): value is string { */ function truncateForPersistence(obj: unknown, blobStore: BlobStore, key?: string): unknown { if (obj === null || obj === undefined) return obj; + if ( + typeof obj === "object" && + "type" in obj && + obj.type === "image_generation_call" && + "result" in obj && + typeof obj.result === "string" && + !isBlobRef(obj.result) && + obj.result.length >= BLOB_EXTERNALIZE_THRESHOLD + ) { + return { ...obj, result: externalizeImageDataSync(blobStore, obj.result) }; + } if (shouldExternalizeImagePayload(obj, key)) { return { ...obj, data: externalizeImageDataSync(blobStore, obj.data, obj.mimeType) }; } diff --git a/packages/coding-agent/test/agent-session-eager-compaction.test.ts b/packages/coding-agent/test/agent-session-eager-compaction.test.ts index a4d0c2db2..0b2644e07 100644 --- a/packages/coding-agent/test/agent-session-eager-compaction.test.ts +++ b/packages/coding-agent/test/agent-session-eager-compaction.test.ts @@ -57,10 +57,11 @@ function getMessageText(message: AgentMessage): string { if (!("content" in message)) return ""; if (typeof message.content === "string") return message.content; if (!Array.isArray(message.content)) return ""; - return message.content - .filter(isTextContentBlock) - .map(content => content.text) - .join("\n"); + const text: string[] = []; + for (const content of message.content) { + if (isTextContentBlock(content)) text.push(content.text); + } + return text.join("\n"); } function createAssistantResponse(text: string) { diff --git a/packages/coding-agent/test/agent-session-eager-task.test.ts b/packages/coding-agent/test/agent-session-eager-task.test.ts index 566e75306..4cbab7a1e 100644 --- a/packages/coding-agent/test/agent-session-eager-task.test.ts +++ b/packages/coding-agent/test/agent-session-eager-task.test.ts @@ -55,10 +55,11 @@ function getMessageText(message: AgentMessage): string { if (!Array.isArray(message.content)) { return ""; } - return message.content - .filter(isTextContentBlock) - .map(content => content.text) - .join("\n"); + const text: string[] = []; + for (const content of message.content) { + if (isTextContentBlock(content)) text.push(content.text); + } + return text.join("\n"); } describe("AgentSession eager task prelude", () => { diff --git a/packages/coding-agent/test/agent-session-eager-todo.test.ts b/packages/coding-agent/test/agent-session-eager-todo.test.ts index baa6d83f2..5e1c42f7b 100644 --- a/packages/coding-agent/test/agent-session-eager-todo.test.ts +++ b/packages/coding-agent/test/agent-session-eager-todo.test.ts @@ -88,10 +88,11 @@ function getMessageText(message: AgentMessage): string { if (!Array.isArray(message.content)) { return ""; } - return message.content - .filter(isTextContentBlock) - .map(content => content.text) - .join("\n"); + const text: string[] = []; + for (const content of message.content) { + if (isTextContentBlock(content)) text.push(content.text); + } + return text.join("\n"); } describe("AgentSession eager todo enforcement", () => { diff --git a/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts b/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts index c00be9666..32002bf1f 100644 --- a/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts +++ b/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts @@ -130,7 +130,10 @@ function getMessageEntries(sessionManager: SessionManager): SessionMessageEntry[ function getTextContent(message: Message): string | undefined { if (typeof message.content === "string") return message.content; - return message.content.find(block => block.type === "text")?.text; + for (const block of message.content) { + if (block.type === "text") return block.text; + } + return undefined; } function findPersistedMessageEntry( diff --git a/packages/coding-agent/test/agent-session-plan-mode-convergence.test.ts b/packages/coding-agent/test/agent-session-plan-mode-convergence.test.ts index cb6ceaec5..7d375818f 100644 --- a/packages/coding-agent/test/agent-session-plan-mode-convergence.test.ts +++ b/packages/coding-agent/test/agent-session-plan-mode-convergence.test.ts @@ -55,10 +55,11 @@ function messageText(message: AgentMessage): string { const content = message.content; if (typeof content === "string") return content; if (!Array.isArray(content)) return ""; - return content - .filter(block => block.type === "text") - .map(block => block.text) - .join("\n"); + const text: string[] = []; + for (const block of content) { + if (block.type === "text") text.push(block.text); + } + return text.join("\n"); } function countReminders(messages: readonly AgentMessage[]): number { diff --git a/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts b/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts index 2885fc6e5..67b942741 100644 --- a/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts +++ b/packages/coding-agent/test/agent-session-plan-reference-compaction.test.ts @@ -50,10 +50,11 @@ function getMessageText(message: AgentMessage): string { if (!("content" in message)) return ""; if (typeof message.content === "string") return message.content; if (!Array.isArray(message.content)) return ""; - return message.content - .filter(isTextContentBlock) - .map(content => content.text) - .join("\n"); + const text: string[] = []; + for (const content of message.content) { + if (isTextContentBlock(content)) text.push(content.text); + } + return text.join("\n"); } function createAssistantResponse(text: string) { diff --git a/packages/coding-agent/test/agent-session-skill-keywords.test.ts b/packages/coding-agent/test/agent-session-skill-keywords.test.ts index 895b4d1d6..3e3419238 100644 --- a/packages/coding-agent/test/agent-session-skill-keywords.test.ts +++ b/packages/coding-agent/test/agent-session-skill-keywords.test.ts @@ -1,7 +1,6 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as path from "node:path"; import { Agent } from "@oh-my-pi/pi-agent-core"; -import type { TextContent } from "@oh-my-pi/pi-ai"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -53,10 +52,11 @@ describe("AgentSession skill prompt keyword steering", () => { const content = message.content; if (typeof content === "string") return content; if (!Array.isArray(content)) return ""; - return content - .filter((block): block is TextContent => block.type === "text") - .map(block => block.text) - .join("\n"); + const text: string[] = []; + for (const block of content) { + if (block.type === "text") text.push(block.text); + } + return text.join("\n"); }), }); const stream = new AssistantMessageEventStream(); diff --git a/packages/coding-agent/test/modes/components/assistant-message-mermaid.test.ts b/packages/coding-agent/test/modes/components/assistant-message-mermaid.test.ts index 33582e8bd..e375ec0ac 100644 --- a/packages/coding-agent/test/modes/components/assistant-message-mermaid.test.ts +++ b/packages/coding-agent/test/modes/components/assistant-message-mermaid.test.ts @@ -259,7 +259,19 @@ describe("AssistantMessageComponent thinking renderers", () => { }); }); -describe("AssistantMessageComponent tool images", () => { +describe("AssistantMessageComponent images", () => { + it("renders native assistant images and honors image visibility", () => { + const message: AssistantMessage = { + ...createAssistantMessage(""), + content: [{ type: "image", data: "aW1hZ2U=", mimeType: "image/png" }], + }; + const component = new AssistantMessageComponent(message); + + expect(Bun.stripANSI(component.render(80).join("\n"))).toContain("[Image: image/png]"); + component.setImagesVisible(false); + expect(Bun.stripANSI(component.render(80).join("\n"))).not.toContain("[Image: image/png]"); + }); + it("converts WebP tool images for Kitty terminal rendering", async () => { const webpBase64 = Buffer.from( await Bun.file(path.join(import.meta.dir, "../../../../../assets/python.webp")).arrayBuffer(), diff --git a/packages/coding-agent/test/session-persistence-images.test.ts b/packages/coding-agent/test/session-persistence-images.test.ts index 60b123b89..4b1c39441 100644 --- a/packages/coding-agent/test/session-persistence-images.test.ts +++ b/packages/coding-agent/test/session-persistence-images.test.ts @@ -68,4 +68,61 @@ describe("session image persistence", () => { expect(resolvedDetails.images[0]?.data).toBe(generatedImageData); expect(resolvedDetails.images[1]?.data).toBe(typedDetailImageData); }); + + it("externalizes and restores native Responses images in assistant content and provider history", async () => { + using tempDir = TempDir.createSync("@session-native-image-persistence-"); + const blobStore = new BlobStore(tempDir.path()); + const data = Buffer.alloc(1500, 4).toString("base64"); + const original: SessionMessageEntry = { + type: "message", + id: "entry-native-image", + parentId: null, + timestamp: new Date(0).toISOString(), + message: { + role: "assistant", + content: [png(data)], + api: "openai-responses", + provider: "openai", + model: "gpt-image-test", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + providerPayload: { + type: "openaiResponsesHistory", + provider: "openai", + items: [{ type: "image_generation_call", id: "ig_1", status: "completed", result: data }], + }, + timestamp: Date.now(), + }, + }; + + const persisted = prepareEntryForPersistence(original, blobStore); + if (persisted.type !== "message" || persisted.message.role !== "assistant") { + throw new Error("expected persisted assistant message"); + } + const persistedImage = persisted.message.content.find(block => block.type === "image"); + const persistedItem = persisted.message.providerPayload?.items[0]; + if (!persistedItem || typeof persistedItem.result !== "string") { + throw new Error("expected persisted image generation item"); + } + expect(isBlobRef(persistedImage?.data ?? "")).toBe(true); + expect(isBlobRef(persistedItem.result)).toBe(true); + + const loaded: FileEntry[] = [structuredClone(persisted)]; + await resolveBlobRefsInEntries(loaded, blobStore); + const resolved = loaded[0]; + if (resolved?.type !== "message" || resolved.message.role !== "assistant") { + throw new Error("expected resolved assistant message"); + } + const resolvedImage = resolved.message.content.find(block => block.type === "image"); + const resolvedItem = resolved.message.providerPayload?.items[0]; + expect(resolvedImage?.data).toBe(data); + expect(resolvedItem?.result).toBe(data); + }); }); From 2faa345d1cf9afea66837a2574c1856383a2614e Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:16:22 +0000 Subject: [PATCH 006/293] fix(ai): classified anthropic spend-limit as persistent usage limit Anthropic returns a `rate_limit_error` when the account's monthly spend cap is hit. Its message ("This request would exceed your account's monthly spend limit.") matched neither USAGE_LIMIT_PATTERN nor ACCOUNT_RATE_LIMIT_PATTERN, so it classified as a transient rate limit: isProviderRetryableError returned true and streamAnthropicOnce's provider retry loop kept retrying (honoring the minutes-long retry-after) until the local deadline fired, surfacing "Deadline exceeded" with zero tokens. Add a `spend limit` alternative to USAGE_LIMIT_PATTERN so the message is classified as a persistent account usage cap: it now surfaces immediately and rotates to a sibling credential instead of looping in backoff. Fixes #4787 --- packages/ai/CHANGELOG.md | 4 ++++ packages/ai/src/error/rate-limit.ts | 2 +- packages/ai/test/anthropic-retry.test.ts | 10 ++++++++++ packages/ai/test/rate-limit-utils.test.ts | 13 +++++++++++++ 4 files changed, 28 insertions(+), 1 deletion(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9c99508fd..373e84aa9 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Anthropic account quota exhaustion (`This request would exceed your account's monthly spend limit`) hanging until the local deadline instead of surfacing the error: the `rate_limit_error` "spend limit" wording is now classified as a persistent usage limit, so it fails fast and rotates to a sibling credential rather than looping in the provider retry backoff. ([#4787](https://github.com/can1357/oh-my-pi/issues/4787)) + ## [16.3.11] - 2026-07-06 ### Fixed diff --git a/packages/ai/src/error/rate-limit.ts b/packages/ai/src/error/rate-limit.ts index fb5678188..bfc0edb7a 100644 --- a/packages/ai/src/error/rate-limit.ts +++ b/packages/ai/src/error/rate-limit.ts @@ -100,7 +100,7 @@ export function calculateRateLimitBackoffMs(reason: RateLimitReason): number { /** Detect usage/quota limit errors in error messages (persistent, requires credential switch). */ const USAGE_LIMIT_PATTERN = - /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?(?:exceeded|reached|insufficient)|额度不足|额度耗尽|resource.?exhausted|exhausted your capacity|quota will reset|insufficient.?(?:balance|quota)/i; + /usage.?limit|usage_limit_reached|usage_not_included|limit_reached|quota.?(?:exceeded|reached|insufficient)|额度不足|额度耗尽|resource.?exhausted|exhausted your capacity|quota will reset|insufficient.?(?:balance|quota)|spend.?limit/i; /** * HTTP status codes that, absent richer body classification, represent an diff --git a/packages/ai/test/anthropic-retry.test.ts b/packages/ai/test/anthropic-retry.test.ts index 03fe45c0d..9f20af608 100644 --- a/packages/ai/test/anthropic-retry.test.ts +++ b/packages/ai/test/anthropic-retry.test.ts @@ -83,6 +83,16 @@ describe("isProviderRetryableError", () => { ).toBe(false); expect(isProviderRetryableError(new Error("usage_limit_reached"))).toBe(false); expect(isProviderRetryableError(new Error("You have hit your ChatGPT usage limit"))).toBe(false); + // Anthropic monthly spend-cap 429 (issue #4787): must not retry, or the + // provider loop burns its budget on minutes-long retry-after backoff and + // surfaces "Deadline exceeded" instead of the quota error. + expect( + isProviderRetryableError( + new Error( + '429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s monthly spend limit. Please try again later."}}', + ), + ), + ).toBe(false); // A generic transient rate limit (no account/usage framing) still retries. expect(isProviderRetryableError(new Error("Rate limit exceeded"))).toBe(true); }); diff --git a/packages/ai/test/rate-limit-utils.test.ts b/packages/ai/test/rate-limit-utils.test.ts index ef0713a45..f6e4bbcce 100644 --- a/packages/ai/test/rate-limit-utils.test.ts +++ b/packages/ai/test/rate-limit-utils.test.ts @@ -120,6 +120,19 @@ describe("isUsageLimit", () => { ).toBe(true); }); + // Anthropic returns a `rate_limit_error` when the account's monthly spend + // cap is hit ("This request would exceed your account's monthly spend + // limit."). Without the `spend limit` branch the message classifies as a + // transient rate limit, so `isProviderRetryableError` retries it until the + // local deadline instead of surfacing the quota error (issue #4787). + it("detects Anthropic monthly spend-limit as a credential-rotatable usage limit", () => { + expect( + isUsageLimit( + '429 {"type":"error","error":{"type":"rate_limit_error","message":"This request would exceed your account\'s monthly spend limit. Please try again later."}}', + ), + ).toBe(true); + }); + it("detects bare 'quota reached' phrasing", () => { expect(isUsageLimit("quota reached")).toBe(true); expect(isUsageLimit("quota_reached")).toBe(true); From 953859d9563ce70b26ce9c0e1c15c009387eb7e9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:20:48 +0000 Subject: [PATCH 007/293] fix(acp): awaited teardown on stdio disconnect Registered ACP session disposal with postmortem and replaced the hard EOF exit with the awaited graceful shutdown path. Classified stdio-write EPIPE separately from worker IPC EPIPE so ACP peer loss exits successfully after cleanup. Fixes #4788 --- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/modes/acp/acp-agent.ts | 19 ++--- .../coding-agent/src/modes/acp/acp-mode.ts | 23 +++++- .../coding-agent/test/acp-disconnect.test.ts | 53 ++++++++++++ packages/utils/CHANGELOG.md | 4 + packages/utils/src/postmortem.ts | 45 ++++++++--- packages/utils/test/postmortem-epipe.test.ts | 81 ++++++++++++------- 7 files changed, 179 insertions(+), 50 deletions(-) create mode 100644 packages/coding-agent/test/acp-disconnect.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..a580f7b74 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed ACP stdio EOF/EPIPE disconnects bypassing awaited session teardown and leaving in-flight tool calls pending in persisted rollouts ([#4788](https://github.com/can1357/oh-my-pi/issues/4788)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index 8d4083c6c..0497b5eef 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -42,7 +42,7 @@ import { } from "@agentclientprotocol/sdk"; import type { AgentToolResult } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai"; -import { getBlobsDir, isEnoent, logger, VERSION } from "@oh-my-pi/pi-utils"; +import { getBlobsDir, isEnoent, logger, type postmortem, VERSION } from "@oh-my-pi/pi-utils"; import { disableProvider, enableProvider, reset as resetCapabilities } from "../../capability"; import { Settings } from "../../config/settings"; import { clearPluginRootsAndCaches, resolveActiveProjectRegistryPath } from "../../discovery/helpers"; @@ -1006,7 +1006,7 @@ export class AcpAgent implements Agent { this.#connection.signal.addEventListener( "abort", () => { - void this.#disposeAllSessions(); + void this.dispose(); }, { once: true }, ); @@ -2316,7 +2316,7 @@ export class AcpAgent implements Agent { } } - async #disposeSessionRecord(record: ManagedSessionRecord): Promise { + async #disposeSessionRecord(record: ManagedSessionRecord, reason?: postmortem.Reason): Promise { record.lifetimeUnsubscribe?.(); if (record.mcpManager) { try { @@ -2327,21 +2327,22 @@ export class AcpAgent implements Agent { record.mcpManager = undefined; } try { - await record.session.dispose(); + await record.session.dispose({ reason }); } catch (error) { logger.warn("Failed to dispose ACP session", { error }); } } - async #disposeStandaloneSession(session: AgentSession): Promise { + async #disposeStandaloneSession(session: AgentSession, reason?: postmortem.Reason): Promise { try { - await session.dispose(); + await session.dispose({ reason }); } catch (error) { logger.warn("Failed to dispose ACP session", { error }); } } - async #disposeAllSessions(): Promise { + /** Dispose every session owned by this ACP connection and await persisted teardown. */ + async dispose(reason?: postmortem.Reason): Promise { if (this.#disposePromise) { await this.#disposePromise; return; @@ -2357,7 +2358,7 @@ export class AcpAgent implements Agent { "ACP agent disposed before queued prompt could run", ); await this.#cancelPromptForClose(record); - await this.#disposeSessionRecord(record); + await this.#disposeSessionRecord(record, reason); } catch (error) { logger.warn("Failed to clean up ACP session", { sessionId, error }); } @@ -2367,7 +2368,7 @@ export class AcpAgent implements Agent { const initialSession = this.#initialSession; this.#initialSession = undefined; if (initialSession) { - await this.#disposeStandaloneSession(initialSession); + await this.#disposeStandaloneSession(initialSession, reason); } })(); diff --git a/packages/coding-agent/src/modes/acp/acp-mode.ts b/packages/coding-agent/src/modes/acp/acp-mode.ts index 8aaf05409..5fc87e09c 100644 --- a/packages/coding-agent/src/modes/acp/acp-mode.ts +++ b/packages/coding-agent/src/modes/acp/acp-mode.ts @@ -1,23 +1,38 @@ import * as stream from "node:stream"; import { AgentSideConnection, ndJsonStream, type Stream } from "@agentclientprotocol/sdk"; +import { postmortem } from "@oh-my-pi/pi-utils"; import type { AgentSession } from "../../session/agent-session"; import { AcpAgent } from "./acp-agent"; +/** Creates sessions requested by an ACP client. */ export type AcpSessionFactory = (cwd: string) => Promise; +/** Creates an ACP connection and exposes its agent when process-level teardown must own it. */ export function createAcpConnection( transport: Stream, createSession: AcpSessionFactory, initialSession?: AgentSession, + onAgent?: (agent: AcpAgent) => void, ): AgentSideConnection { - return new AgentSideConnection(conn => new AcpAgent(conn, createSession, initialSession), transport); + return new AgentSideConnection(connection => { + const agent = new AcpAgent(connection, createSession, initialSession); + onAgent?.(agent); + return agent; + }, transport); } -export async function runAcpMode(createSession: AcpSessionFactory, initialSession?: AgentSession): Promise { +/** Serves ACP over stdio until the peer disconnects, then awaits session teardown before exit. */ +export async function runAcpMode(createSession: AcpSessionFactory, initialSession?: AgentSession): Promise { + let agent: AcpAgent | undefined; + postmortem.register("acp-session-teardown", reason => agent?.dispose(reason)); + postmortem.registerStdioDisconnectHandling(); + const input = stream.Writable.toWeb(process.stdout); const output = stream.Readable.toWeb(process.stdin); const transport = ndJsonStream(input, output); - const connection = createAcpConnection(transport, createSession, initialSession); + const connection = createAcpConnection(transport, createSession, initialSession, createdAgent => { + agent = createdAgent; + }); await connection.closed; - process.exit(0); + await postmortem.quit(0); } diff --git a/packages/coding-agent/test/acp-disconnect.test.ts b/packages/coding-agent/test/acp-disconnect.test.ts new file mode 100644 index 000000000..5b14b2a6c --- /dev/null +++ b/packages/coding-agent/test/acp-disconnect.test.ts @@ -0,0 +1,53 @@ +import { describe, expect, it } from "bun:test"; +import { runAcpMode } from "@oh-my-pi/pi-coding-agent/modes/acp/acp-mode"; +import { postmortem } from "@oh-my-pi/pi-utils"; + +const childFlag = "--acp-eof-child"; +const childFlagIndex = process.argv.indexOf(childFlag); +if (childFlagIndex >= 0) { + const marker = process.argv[childFlagIndex + 1]; + if (!marker) throw new Error("Missing cleanup marker path"); + const releaseCleanup = Promise.withResolvers(); + process.once("SIGUSR2", releaseCleanup.resolve); + postmortem.register("acp-eof-test", async () => { + process.stderr.write("cleanup started\n"); + await releaseCleanup.promise; + await Bun.write(marker, "cleanup complete"); + }); + await runAcpMode(async () => { + throw new Error("Session factory is unused by the EOF harness"); + }); +} + +describe("ACP stdio disconnect", () => { + it("awaits postmortem cleanup before exiting on client EOF", async () => { + const marker = `/tmp/omp-acp-eof-${process.pid}-${Date.now()}`; + const child = Bun.spawn([process.execPath, import.meta.path, childFlag, marker], { + stdin: "pipe", + stdout: "pipe", + stderr: "pipe", + }); + try { + child.stdin.end(); + const stderrReader = child.stderr.getReader(); + const started = await stderrReader.read(); + stderrReader.releaseLock(); + expect(new TextDecoder().decode(started.value)).toBe("cleanup started\n"); + child.kill("SIGUSR2"); + const [exitCode, stdout] = await Promise.all([child.exited, new Response(child.stdout).text()]); + expect(stdout).toBe(""); + expect(exitCode).toBe(0); + expect(await Bun.file(marker).text()).toBe("cleanup complete"); + } finally { + try { + child.kill("SIGUSR2"); + } catch { + // Already exited after completing teardown. + } + await child.exited; + await Bun.file(marker) + .delete() + .catch(() => {}); + } + }); +}); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 80ebafb3a..ff70fe508 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Added scoped graceful handling for stdio-write EPIPE rejections so protocol servers can await postmortem cleanup when their peer disconnects ([#4788](https://github.com/can1357/oh-my-pi/issues/4788)). + ## [16.3.10] - 2026-07-06 ### Added diff --git a/packages/utils/src/postmortem.ts b/packages/utils/src/postmortem.ts index 01d208e9e..126e3388a 100644 --- a/packages/utils/src/postmortem.ts +++ b/packages/utils/src/postmortem.ts @@ -26,6 +26,7 @@ const callbackList: ((reason: Reason) => Promise | void)[] = []; // Tracks cleanup run state (to prevent recursion/reentry issues) let cleanupStage: "idle" | "running" | "complete" = "idle"; const CLEANUP_DEADLINE_MS = 10_000; +let stdioDisconnectRegistrations = 0; /** * Internal: runs all registered cleanup callbacks for the given reason. @@ -76,17 +77,35 @@ function runCleanup(reason: Reason): Promise { // Worker thread: exit only (workers use self.addEventListener for exceptions) let inspectorOpened = false; +/** Origin of an EPIPE raised by a process communication channel. */ +export type BrokenPipeSource = "ipc-send" | "stdio-write"; + /** - * Detect an EPIPE rejection that originated from an IPC `send()` to a worker - * subprocess (`syscall: "send"`), as opposed to a stdin/stdout pipe write - * (`syscall: "write"`). Only the IPC-send path can break an optional worker - * subsystem without affecting the main process, so only this shape is safe to - * swallow at the global `unhandledRejection` level. See issue #2997. + * Classify EPIPE errors from worker IPC and stdio without treating unrelated + * broken pipes as globally recoverable. */ -export function isIpcSendEpipe(err: Error): boolean { - const code = (err as { code?: unknown }).code; - const syscall = (err as { syscall?: unknown }).syscall; - return code === "EPIPE" && syscall === "send"; +export function classifyBrokenPipe(err: Error): BrokenPipeSource | undefined { + if (!("code" in err) || err.code !== "EPIPE" || !("syscall" in err)) return undefined; + if (err.syscall === "send") return "ipc-send"; + if (err.syscall === "write") return "stdio-write"; + return undefined; +} + +/** + * Treat unhandled stdout EPIPE rejections as a graceful peer disconnect. + * + * Stdio protocol servers call this for their process lifetime so a closed + * client pipe runs registered cleanup callbacks instead of the fatal path. + * The returned callback removes the registration. + */ +export function registerStdioDisconnectHandling(): () => void { + let registered = true; + stdioDisconnectRegistrations++; + return () => { + if (!registered) return; + registered = false; + stdioDisconnectRegistrations--; + }; } // Well-known key marking an error as an *expected* teardown artifact (e.g. a @@ -155,6 +174,7 @@ if (isMainThread) { }) .on("unhandledRejection", async reason => { const err = reason instanceof Error ? reason : new Error(String(reason)); + const brokenPipeSource = classifyBrokenPipe(err); // EPIPE from an IPC `send()` (`syscall: "send"`) originates from a // worker subprocess whose pipe broke between the exit being observed // and the next `proc.send()` — a race window that Bun surfaces as an @@ -164,10 +184,15 @@ if (isMainThread) { // send pipe must never take down the whole session. Log and continue // instead of exiting; the owning client detects the dead worker via // its own `onExit`/error path and respawns or disables it. See #2997. - if (isIpcSendEpipe(err)) { + if (brokenPipeSource === "ipc-send") { logger.warn("Ignoring EPIPE from worker IPC send; optional subsystem will self-recover", { err }); return; } + if (brokenPipeSource === "stdio-write" && stdioDisconnectRegistrations > 0) { + logger.warn("Stdio peer disconnected; shutting down gracefully", { err }); + await quit(0); + return; + } if (isExpectedCleanupError(reason)) { logger.warn("Ignoring expected cleanup rejection", { err }); return; diff --git a/packages/utils/test/postmortem-epipe.test.ts b/packages/utils/test/postmortem-epipe.test.ts index 178281a11..25fdcd8a1 100644 --- a/packages/utils/test/postmortem-epipe.test.ts +++ b/packages/utils/test/postmortem-epipe.test.ts @@ -1,42 +1,69 @@ import { describe, expect, it } from "bun:test"; import { postmortem } from "@oh-my-pi/pi-utils"; -/** - * Contract for issue #2997: an EPIPE rejection from an IPC `send()` to a worker - * subprocess (`syscall: "send"`) must be recognizable as a non-fatal, optional- - * subsystem failure so the global `unhandledRejection` handler can swallow it - * instead of terminating the session. The predicate must be narrow: a bare - * EPIPE, or an EPIPE from a stdin/stdout write (`syscall: "write"`), is NOT - * swallowed — those may signal a real broken pipe to a critical stream. - */ -describe("postmortem.isIpcSendEpipe", () => { +const childFlag = "--stdio-epipe-child"; +const childFlagIndex = process.argv.indexOf(childFlag); +if (childFlagIndex >= 0) { + const marker = process.argv[childFlagIndex + 1]; + if (!marker) throw new Error("Missing cleanup marker path"); + postmortem.registerStdioDisconnectHandling(); + postmortem.register("stdio-epipe-test", async () => { + process.stderr.write("cleanup started\n"); + await new Response(Bun.stdin.stream()).text(); + await Bun.write(marker, "cleanup complete"); + }); + const err = Object.assign(new Error("broken pipe"), { code: "EPIPE", syscall: "write" }); + void Promise.reject(err); + const keepAlive = Promise.withResolvers(); + await keepAlive.promise; +} + +describe("postmortem broken-pipe handling", () => { function makeErr(props: { code?: string; syscall?: string; message?: string }): Error { const err = new Error(props.message ?? "broken pipe"); Object.assign(err, { code: props.code, syscall: props.syscall }); return err; } - it("matches EPIPE with syscall 'send' (worker IPC send)", () => { - expect(postmortem.isIpcSendEpipe(makeErr({ code: "EPIPE", syscall: "send" }))).toBe(true); + it("classifies worker IPC and stdio EPIPE errors", () => { + expect(postmortem.classifyBrokenPipe(makeErr({ code: "EPIPE", syscall: "send" }))).toBe("ipc-send"); + expect(postmortem.classifyBrokenPipe(makeErr({ code: "EPIPE", syscall: "write" }))).toBe("stdio-write"); }); - it("does not match EPIPE from a stdin/stdout write (syscall 'write')", () => { - expect(postmortem.isIpcSendEpipe(makeErr({ code: "EPIPE", syscall: "write" }))).toBe(false); + it("does not classify unrelated errors as recoverable broken pipes", () => { + expect(postmortem.classifyBrokenPipe(makeErr({ code: "EPIPE" }))).toBeUndefined(); + expect(postmortem.classifyBrokenPipe(makeErr({ code: "ENOENT", syscall: "send" }))).toBeUndefined(); + expect(postmortem.classifyBrokenPipe(new Error("boom"))).toBeUndefined(); + expect(postmortem.classifyBrokenPipe(makeErr({ code: undefined, syscall: undefined }))).toBeUndefined(); }); - it("does not match a bare EPIPE without a syscall", () => { - expect(postmortem.isIpcSendEpipe(makeErr({ code: "EPIPE" }))).toBe(false); - }); - - it("does not match a non-EPIPE error even with syscall 'send'", () => { - expect(postmortem.isIpcSendEpipe(makeErr({ code: "ENOENT", syscall: "send" }))).toBe(false); - }); - - it("does not match a plain Error with no code/syscall", () => { - expect(postmortem.isIpcSendEpipe(new Error("boom"))).toBe(false); - }); - - it("does not match nullish/missing errno-style fields gracefully", () => { - expect(postmortem.isIpcSendEpipe(makeErr({ code: undefined, syscall: undefined }))).toBe(false); + it("awaits cleanup and exits successfully when a registered stdio peer disconnects", async () => { + const marker = `/tmp/omp-postmortem-stdio-${process.pid}-${Date.now()}`; + const child = Bun.spawn([process.execPath, import.meta.path, childFlag, marker], { + stdin: "pipe", + stdout: "pipe", + stderr: "pipe", + }); + try { + const stderrReader = child.stderr.getReader(); + const started = await stderrReader.read(); + stderrReader.releaseLock(); + expect(new TextDecoder().decode(started.value)).toBe("cleanup started\n"); + child.stdin.end(); + const [exitCode, stdout] = await Promise.all([child.exited, new Response(child.stdout).text()]); + expect(stdout).toBe(""); + expect(exitCode).toBe(0); + expect(await Bun.file(marker).text()).toBe("cleanup complete"); + } finally { + try { + child.stdin.end(); + } catch { + // Already closed after the cleanup gate was released. + } + await child.exited; + await Bun.file(marker) + .delete() + .catch(() => {}); + } }); }); From 0eda288b99668127bd802341602005c9f53f94a6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:22:55 +0000 Subject: [PATCH 008/293] fix(tui): updated nerd session icon codepoint Replaced the removed Nerd Fonts v2 Material Design glyph with its Nerd Fonts v3 mapping and covered the preset contract. Fixes #4795 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/modes/theme/theme.ts | 4 +- .../test/theme-nerd-symbols.test.ts | 41 +++++++++++++++++++ 3 files changed, 47 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/test/theme-nerd-symbols.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..905964f22 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed the `nerd` status-line preset's session icon using a removed Nerd Fonts v2 codepoint instead of the current Nerd Fonts v3 mapping ([#4795](https://github.com/can1357/oh-my-pi/issues/4795)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index 77140221b..eabba4e45 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -607,8 +607,8 @@ const NERD_SYMBOLS: SymbolMap = { "icon.throughput": "\uf0e4", // pick:  | alt:   "icon.host": "\uf109", - // pick:  | alt:   - "icon.session": "\uf550", + // pick: 󰁑 (nf-md-arrow_left_bold_hexagon_outline) | alt:   + "icon.session": "\u{f0051}", // pick:  | alt:  "icon.package": "\uf487", // pick:  | alt:   diff --git a/packages/coding-agent/test/theme-nerd-symbols.test.ts b/packages/coding-agent/test/theme-nerd-symbols.test.ts new file mode 100644 index 000000000..d2855258e --- /dev/null +++ b/packages/coding-agent/test/theme-nerd-symbols.test.ts @@ -0,0 +1,41 @@ +import { afterEach, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { getAgentDir, getCustomThemesDir, removeWithRetries, setAgentDir } from "@oh-my-pi/pi-utils"; + +const DARK_THEME_PATH = path.join(import.meta.dir, "..", "src", "modes", "theme", "dark.json"); + +let tempAgentDir: string | undefined; +let originalAgentDir = ""; +let originalAgentDirEnv: string | undefined; + +afterEach(async () => { + if (tempAgentDir === undefined) return; + setAgentDir(originalAgentDir); + if (originalAgentDirEnv === undefined) { + delete process.env.PI_CODING_AGENT_DIR; + } else { + process.env.PI_CODING_AGENT_DIR = originalAgentDirEnv; + } + await removeWithRetries(tempAgentDir); + tempAgentDir = undefined; +}); + +it("uses the Nerd Fonts v3 Material Design session icon", async () => { + originalAgentDir = getAgentDir(); + originalAgentDirEnv = process.env.PI_CODING_AGENT_DIR; + tempAgentDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-nerd-symbols-")); + setAgentDir(tempAgentDir); + + const dark = await Bun.file(DARK_THEME_PATH).json(); + const customThemeName = "nerd-symbols"; + await Bun.write( + path.join(getCustomThemesDir(), `${customThemeName}.json`), + JSON.stringify({ ...dark, name: customThemeName, symbols: { ...dark.symbols, preset: "nerd" } }), + ); + + const theme = await getThemeByName(customThemeName); + expect(theme?.symbol("icon.session")).toBe("\u{f0051}"); +}); From f69783765938054abb344db52d6a99f52bad4f0b Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:26:46 +0000 Subject: [PATCH 009/293] fix(tools): stopped column cap from faking window truncation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per-line column cap trims individual lines with a `…` marker but does not truncate the output window. OutputSink.dump() nonetheless set truncated=true whenever a line was capped, and truncationFromSummary then reported a byte tail-window truncation, appending a bogus "Showing lines X-Y of Z (…B limit). Read artifact://N for full output" footer even though every line was shown. - OutputSink no longer flips #truncated on column-cap-only drops. - OutputSummary carries columnMax; truncationFromSummary surfaces it as the "Some lines truncated to N chars" limit notice regardless of window state. Fixes #4735 --- packages/coding-agent/CHANGELOG.md | 4 +++ .../src/session/streaming-output.ts | 4 ++- .../coding-agent/src/tools/output-meta.ts | 7 +++++ .../test/streaming-output.test.ts | 29 ++++++++++++++++++- packages/coding-agent/test/tools.test.ts | 8 ++--- 5 files changed, 46 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 459ec14a8..681a8c2cc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed bash/eval/ssh output that was only per-line column-capped being misreported as byte-window truncation, which appended a bogus `Showing lines X-Y of Z (…B limit). Read artifact://N for full output` footer even though every line was shown. Column-cap trimming now surfaces solely as the `Some lines truncated to N chars` notice ([#4735](https://github.com/can1357/oh-my-pi/issues/4735)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/session/streaming-output.ts b/packages/coding-agent/src/session/streaming-output.ts index 5eafeb528..1d01f5667 100644 --- a/packages/coding-agent/src/session/streaming-output.ts +++ b/packages/coding-agent/src/session/streaming-output.ts @@ -43,6 +43,8 @@ export interface OutputSummary { columnDroppedBytes?: number; /** Number of distinct lines that hit the per-line column cap. */ columnTruncatedLines?: number; + /** Configured per-line column cap in effect (chars), when > 0. */ + columnMax?: number; /** Artifact ID for internal URL access (artifact://) when truncated */ artifactId?: string; } @@ -837,7 +839,6 @@ export class OutputSink { const capped = this.#maxColumns > 0 ? this.#applyColumnCap(chunk) : chunk; const cappedBytes = capped === chunk ? rawBytes : Buffer.byteLength(capped, "utf-8"); const cappedThisChunk = cappedBytes < rawBytes; - if (cappedThisChunk) this.#truncated = true; // Mirror RAW chunk to the artifact file so the on-disk record is the full // uncapped stream. Mirror triggers on: in-memory overflow OR this chunk's @@ -1248,6 +1249,7 @@ export class OutputSink { elidedLines, columnDroppedBytes: this.#columnDroppedBytes > 0 ? this.#columnDroppedBytes : undefined, columnTruncatedLines: this.#columnTruncatedLines > 0 ? this.#columnTruncatedLines : undefined, + columnMax: this.#columnTruncatedLines > 0 ? this.#maxColumns : undefined, artifactId: this.#file?.artifactId, }; } diff --git a/packages/coding-agent/src/tools/output-meta.ts b/packages/coding-agent/src/tools/output-meta.ts index ce5459f17..154ee9fb3 100644 --- a/packages/coding-agent/src/tools/output-meta.ts +++ b/packages/coding-agent/src/tools/output-meta.ts @@ -191,6 +191,13 @@ export class OutputMetaBuilder { /** Add truncation info from OutputSummary. No-op if not truncated. */ truncationFromSummary(summary: OutputSummary, options: TruncationSummaryOptions): this { + // A per-line column cap only trims individual lines (with a `…` marker); + // it is not a window/byte truncation, so surface it as its own limit + // notice rather than a "Showing lines X-Y … limit" range. This runs even + // when the output is otherwise complete (`truncated === false`). + if (summary.columnMax != null && summary.columnMax > 0 && (summary.columnTruncatedLines ?? 0) > 0) { + this.columnTruncated(summary.columnMax); + } if (!summary.truncated) return this; const { direction, startLine = 1, totalFileLines } = options; diff --git a/packages/coding-agent/test/streaming-output.test.ts b/packages/coding-agent/test/streaming-output.test.ts index f18d05fa9..e97d31aa5 100644 --- a/packages/coding-agent/test/streaming-output.test.ts +++ b/packages/coding-agent/test/streaming-output.test.ts @@ -15,6 +15,7 @@ import { truncateTail, truncateTailBytes, } from "@oh-my-pi/pi-coding-agent/session/streaming-output"; +import { formatOutputNotice, outputMeta } from "@oh-my-pi/pi-coding-agent/tools/output-meta"; import { removeWithRetries } from "@oh-my-pi/pi-utils"; const createdTempDirs: string[] = []; @@ -636,7 +637,11 @@ describe("OutputSink maxColumns (per-line cap)", () => { await sink.push(`short\n${"x".repeat(50)}\nfooter`); const dumped = await sink.dump(); - expect(dumped.truncated).toBe(true); + // A per-line column cap trims individual lines but does not truncate the + // output window: every line is still present, so `truncated` stays false. + // (Regression: column-cap-only output was misreported as a byte-window + // truncation, producing a bogus "Showing lines X-Y … limit" footer — #4735.) + expect(dumped.truncated).toBe(false); expect(dumped.output).toContain("short\n"); expect(dumped.output).toContain("\nfooter"); expect(dumped.output).toContain("…"); @@ -644,10 +649,32 @@ describe("OutputSink maxColumns (per-line cap)", () => { expect(dumped.output).not.toContain("x".repeat(50)); expect(dumped.columnTruncatedLines).toBe(1); expect(dumped.columnDroppedBytes ?? 0).toBeGreaterThan(0); + expect(dumped.columnMax).toBe(8); // totalBytes still reflects the raw stream, not the post-cap view. expect(dumped.totalBytes).toBe(byteLength(`short\n${"x".repeat(50)}\nfooter`)); }); + test("column-cap-only output surfaces a column notice, not a window/byte truncation footer", async () => { + // Regression for #4735: fully-shown output whose only trimming was the + // per-line column cap must not emit "Showing lines X-Y of Z (…B limit). + // Read artifact://… for full output" — every line is present. + const sink = new OutputSink({ maxColumns: 8, spillThreshold: 100_000 }); + const lines = ["a", "b", "c", "x".repeat(50), "d"]; + await sink.push(`${lines.join("\n")}\n`); + const dumped = await sink.dump(); + + const meta = outputMeta().truncationFromSummary(dumped, { direction: "tail" }).get(); + // No window truncation → no styled TUI warning and no range/limit footer. + expect(meta?.truncation).toBeUndefined(); + expect(meta?.limits?.columnTruncated).toEqual({ maxColumn: 8 }); + + const notice = formatOutputNotice(meta); + expect(notice).toContain("Some lines truncated to 8 chars"); + expect(notice).not.toContain("Showing lines"); + expect(notice).not.toContain("limit"); + expect(notice).not.toContain("artifact://"); + }); + test("persists per-line state across chunk boundaries", async () => { const sink = new OutputSink({ maxColumns: 4, spillThreshold: 1000 }); await sink.push("ab"); // 2 bytes into the current line diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index a556a4351..650e1ca23 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -1328,10 +1328,10 @@ function b() { it("should write truncated output to artifacts", async () => { const result = await bashTool.execute("test-call-8-artifact", { - // A single line past the 768-byte column cap is the minimal output - // that trips truncation + artifact spill; the old 60K-arg brace - // expansion paid ~60ms of shell time to prove the same path. - command: "printf 'a%.0s' {1..2000}", + // Emit well past the ~50KB inline window across many lines so the + // output is genuinely window-truncated (not merely column-capped), + // which is what allocates the spill artifact. + command: "seq 1 30000", }); const artifactId = result.details?.meta?.truncation?.artifactId; From e4a10450ec269af12382beb36d4918c49fd79e75 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:31:01 +0000 Subject: [PATCH 010/293] fix(natives): distinguish process-stale from disk-stale sentinel mismatch A long-lived process that survives an in-place upgrade keeps the previous pi-natives NAPI addon resident. A tab worker spawned afterwards runs the new JS loader, which expects the new version sentinel, but require returns the resident old exports carrying the prior sentinel. validateLoadedBindings previously reported "reinstall to re-sync" for this case even though disk was already consistent, so only a restart helped. Detect a versioned __piNativesV* export other than the expected one on the loaded bindings and report that omp was upgraded mid-session and must be restarted, reserving the reinstall guidance for genuinely disk-stale addons. Fixes #4812 --- packages/natives/CHANGELOG.md | 1 + packages/natives/native/loader-state.d.ts | 12 ++++ packages/natives/native/loader-state.js | 26 ++++++- .../natives/test/issue-4812-repro.test.ts | 71 +++++++++++++++++++ 4 files changed, 109 insertions(+), 1 deletion(-) create mode 100644 packages/natives/test/issue-4812-repro.test.ts diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index c8c4ffef6..d180e2be6 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed the native build script failing to locate the `@napi-rs/cli` `napi` binary on Windows because the `PATH` lookup joined entries with a Unix `:` separator instead of the platform delimiter (`path.delimiter`). +- Fixed the pi-natives version sentinel emitting "reinstall to re-sync" when a long-lived process survives an in-place upgrade: the loader now detects that the resident addon exposes a *prior* release's sentinel and reports "omp was upgraded while this session was running — restart to pick up the new version (disk is already consistent)" instead of misdiagnosing it as a stale on-disk file ([#4812](https://github.com/can1357/oh-my-pi/issues/4812)). ## [16.3.6] - 2026-07-04 diff --git a/packages/natives/native/loader-state.d.ts b/packages/natives/native/loader-state.d.ts index 73e119ee1..d6fe483be 100644 --- a/packages/natives/native/loader-state.d.ts +++ b/packages/natives/native/loader-state.d.ts @@ -86,4 +86,16 @@ export interface SelectCpuVariantResult { export function selectCpuVariant(input: SelectCpuVariantInput): SelectCpuVariantResult; +export interface ValidateLoadedBindingsContext { + isWorkspaceLoad: boolean; + packageVersion: string; + versionSentinelExport: string; +} + +export function validateLoadedBindings( + ctx: ValidateLoadedBindingsContext, + bindings: Record, + candidate: string, +): void; + export function loadNative(): Record; diff --git a/packages/natives/native/loader-state.js b/packages/natives/native/loader-state.js index 486bed688..44033a61e 100644 --- a/packages/natives/native/loader-state.js +++ b/packages/natives/native/loader-state.js @@ -583,7 +583,7 @@ function maybeStageNodeModulesAddon(ctx, errors) { return stagedPath; } -function validateLoadedBindings(ctx, bindings, candidate) { +export function validateLoadedBindings(ctx, bindings, candidate) { // In workspace dev (running out of `packages/natives/native/` rather than a // `node_modules` install or a compiled bundle) the local `.node` only gains // the renamed sentinel after `bun --cwd=packages/natives run build`. Skip @@ -591,6 +591,30 @@ function validateLoadedBindings(ctx, bindings, candidate) { // completes; install and compiled-binary paths still validate. if (ctx.isWorkspaceLoad) return; if (typeof bindings[ctx.versionSentinelExport] === "function") return; + + // The expected sentinel is missing. Distinguish two failure modes by the + // sentinel the bindings DO carry: + // - disk stale: the `.node` on disk predates this loader (its own build); + // reinstalling re-syncs the file. + // - process stale: an in-place upgrade landed a new release on disk while + // this process still holds the previous addon generation resident in the + // dynamic-loader's native-module cache. `require` returns those old + // exports, which carry the PRIOR sentinel — disk is already consistent, + // so reinstall is a no-op and only restarting the process re-syncs. + const residentSentinel = Object.keys(bindings).find( + key => key !== ctx.versionSentinelExport && /^__piNativesV[A-Za-z0-9_]+$/.test(key), + ); + if (residentSentinel) { + const residentVersion = residentSentinel.slice("__piNativesV".length).replace(/_/g, "."); + throw new Error( + `Loaded ${candidate}, which exposes the @oh-my-pi/pi-natives@${residentVersion} version ` + + `sentinel \`${residentSentinel}\` but not the @${ctx.packageVersion} sentinel ` + + `\`${ctx.versionSentinelExport}\` this loader expects. omp was upgraded to ` + + `${ctx.packageVersion} while this session was running; the ${residentVersion} addon is ` + + "still resident in this process. Disk is already consistent — restart omp to pick up " + + `${ctx.packageVersion} (reinstalling changes nothing).`, + ); + } throw new Error( `Loaded ${candidate} but it does not expose the @oh-my-pi/pi-natives@${ctx.packageVersion} ` + `version sentinel \`${ctx.versionSentinelExport}\`. The .node file on disk is from a different ` + diff --git a/packages/natives/test/issue-4812-repro.test.ts b/packages/natives/test/issue-4812-repro.test.ts new file mode 100644 index 000000000..2d18924d0 --- /dev/null +++ b/packages/natives/test/issue-4812-repro.test.ts @@ -0,0 +1,71 @@ +/** + * Repro for https://github.com/can1357/oh-my-pi/issues/4812 + * + * A long-lived omp session that survives an in-place `bun install -g` upgrade + * keeps the previous pi-natives NAPI addon resident in the process. A tab + * worker spawned afterwards runs the freshly-installed JS loader, which expects + * the new sentinel (e.g. `__piNativesV16_3_11`), but `require` returns the + * resident old exports carrying the PRIOR sentinel (`__piNativesV16_3_10`). + * + * The contract this test pins down: `validateLoadedBindings` distinguishes a + * process-stale mix (disk consistent — restart to re-sync) from a genuinely + * disk-stale addon (reinstall to re-sync), and never tells the operator to + * reinstall when the bindings already carry a versioned sentinel. + */ +import { describe, expect, it } from "bun:test"; +import { validateLoadedBindings } from "../native/loader-state.js"; + +const candidate = "/home/u/.bun/install/global/node_modules/@oh-my-pi/pi-natives-linux-x64/pi_natives.linux-x64.node"; + +function ctxFor(version: string) { + return { + isWorkspaceLoad: false, + packageVersion: version, + versionSentinelExport: `__piNativesV${version.replace(/[^A-Za-z0-9]/g, "_")}`, + }; +} + +describe("issue 4812: pi-natives sentinel process-stale diagnosis", () => { + it("accepts bindings that expose the expected sentinel", () => { + const ctx = ctxFor("16.3.11"); + expect(() => + validateLoadedBindings(ctx, { __piNativesV16_3_11: () => {}, grep: () => {} }, candidate), + ).not.toThrow(); + }); + + it("reports a mid-session upgrade (restart) when bindings carry an older sentinel", () => { + const ctx = ctxFor("16.3.11"); + const resident = { __piNativesV16_3_10: () => {}, grep: () => {} }; + let message = ""; + try { + validateLoadedBindings(ctx, resident, candidate); + } catch (err) { + message = err instanceof Error ? err.message : String(err); + } + expect(message).toContain("16.3.10"); + expect(message).toContain("restart omp"); + expect(message).toContain("Disk is already consistent"); + // The disk-stale advice must NOT appear for a process-stale mix. + expect(message).not.toContain("reinstall to re-sync"); + expect(message).not.toContain("from a different release than this loader"); + }); + + it("still reports disk-stale (reinstall) when no versioned sentinel is present", () => { + const ctx = ctxFor("16.3.11"); + const stale = { grep: () => {}, astGrep: () => {} }; + let message = ""; + try { + validateLoadedBindings(ctx, stale, candidate); + } catch (err) { + message = err instanceof Error ? err.message : String(err); + } + expect(message).toContain("from a different release than this loader"); + expect(message).toContain("reinstall to re-sync"); + expect(message).not.toContain("restart omp"); + }); + + it("skips validation entirely in workspace dev", () => { + const ctx = { ...ctxFor("16.3.11"), isWorkspaceLoad: true }; + expect(() => validateLoadedBindings(ctx, { grep: () => {} }, candidate)).not.toThrow(); + }); +}); From d3f4830ceb9797d0e86d0a5cd61fe77dec0e1bc2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:34:47 +0000 Subject: [PATCH 011/293] fix(tui): deferred command output during streaming - Queued transcript command panels until the active agent turn ends. - Added regression coverage for slash-command output mounting exactly once. Fixes #4806 --- packages/coding-agent/CHANGELOG.md | 4 + .../controllers/command-controller-shared.ts | 2 +- .../modes/controllers/command-controller.ts | 2 +- .../src/modes/controllers/event-controller.ts | 1 + .../src/modes/interactive-mode.ts | 19 +++++ packages/coding-agent/src/modes/types.ts | 8 ++ .../test/issue-4806-command-output.test.ts | 81 +++++++++++++++++++ .../coding-agent/test/issue-956-repro.test.ts | 4 + .../test/mcp-command-reauth.test.ts | 1 + .../test/mcp-command-toggle.test.ts | 1 + 10 files changed, 121 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/test/issue-4806-command-output.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..9708423e5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed `/mcp`, `/mcp list`, and `/tools` output duplicating in terminal scrollback when invoked during agent streaming by deferring command panels until the active turn ends ([#4806](https://github.com/can1357/oh-my-pi/issues/4806)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/modes/controllers/command-controller-shared.ts b/packages/coding-agent/src/modes/controllers/command-controller-shared.ts index 2e3eb6471..c51203c21 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller-shared.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller-shared.ts @@ -105,5 +105,5 @@ export function showCommandMessage(ctx: InteractiveModeContext, text: string): v block.addChild(new DynamicBorder()); block.addChild(new Text(text, 1, 1)); block.addChild(new DynamicBorder()); - ctx.present(block); + ctx.presentCommandOutput(block); } diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index 6fa6c1265..5c3b47a33 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -59,7 +59,7 @@ function showMarkdownPanel(ctx: InteractiveModeContext, title: string, markdown: block.addChild(new Spacer(1)); block.addChild(new Markdown(markdown.trim(), 1, 1, getMarkdownTheme())); block.addChild(new DynamicBorder()); - ctx.present(block); + ctx.presentCommandOutput(block); } export class CommandController { diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index 88a66022b..b997e6c97 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -1116,6 +1116,7 @@ export class EventController { // final history — seal it instead of letting its spinner tick while idle. this.#resolveDisplaceablePoll(); this.#resolveDisplaceableTodo(); + this.ctx.flushPendingCommandOutput(); this.#lastAssistantComponent = undefined; this.ctx.ui.requestRender(); this.#scheduleIdleCompaction(); diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 8e064d8ee..b94ddf4fd 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -515,6 +515,7 @@ export class InteractiveMode implements InteractiveModeContext { collabHost?: CollabHost; collabGuest?: CollabGuestLink; + #pendingCommandOutput: Component[] = []; #pendingSlashCommands: SlashCommand[] = []; #cleanupUnsubscribe?: () => void; #signalTeardown?: SessionTeardown; @@ -3490,6 +3491,24 @@ export class InteractiveMode implements InteractiveModeContext { this.ui.requestRender(); } + /** Defer transcript command panels until the active turn can no longer grow above them. */ + presentCommandOutput(content: Component | readonly Component[]): void { + if (!this.session.isStreaming) { + this.present(content); + return; + } + const items = Array.isArray(content) ? content : [content as Component]; + this.#pendingCommandOutput.push(...items); + } + + /** Mount every command panel queued while the agent was streaming. */ + flushPendingCommandOutput(): void { + if (this.#pendingCommandOutput.length === 0) return; + const pending = this.#pendingCommandOutput; + this.#pendingCommandOutput = []; + this.present(pending); + } + #mountChatChild(item: Component): void { this.chatContainer.addChild(item); if (item instanceof ChatBlock) item.mount(this.#chatHost); diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 6eb2c0f4a..9449373aa 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -231,6 +231,14 @@ export interface InteractiveModeContext { * runs) so their timers/subscriptions start. */ present(content: Component | readonly Component[]): void; + /** + * Mount command output immediately while idle, or defer it until the active + * agent turn ends so a growing live block cannot push duplicate rows into + * native scrollback. + */ + presentCommandOutput(content: Component | readonly Component[]): void; + /** Mount command output deferred by {@link presentCommandOutput}. */ + flushPendingCommandOutput(): void; /** * Dispose every live block in the transcript (stopping timers/subscriptions) * and clear it. Used before a full rebuild so animated/streaming blocks do not diff --git a/packages/coding-agent/test/issue-4806-command-output.test.ts b/packages/coding-agent/test/issue-4806-command-output.test.ts new file mode 100644 index 000000000..adb1eac1c --- /dev/null +++ b/packages/coding-agent/test/issue-4806-command-output.test.ts @@ -0,0 +1,81 @@ +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as path from "node:path"; +import { Agent } from "@oh-my-pi/pi-agent-core"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { InteractiveMode } from "@oh-my-pi/pi-coding-agent/modes/interactive-mode"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { HistoryStorage } from "@oh-my-pi/pi-coding-agent/session/history-storage"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { Text } from "@oh-my-pi/pi-tui"; +import { TempDir } from "@oh-my-pi/pi-utils"; + +describe("issue #4806 command output during streaming", () => { + let authStorage: AuthStorage; + let mode: InteractiveMode; + let session: AgentSession; + let streaming = true; + let tempDir: TempDir; + + beforeAll(() => { + initTheme(); + }); + + beforeEach(async () => { + vi.spyOn(process.stdout, "write").mockReturnValue(true); + vi.spyOn(process.stdin, "resume").mockReturnValue(process.stdin); + vi.spyOn(process.stdin, "pause").mockReturnValue(process.stdin); + vi.spyOn(process.stdin, "setEncoding").mockReturnValue(process.stdin); + if (typeof process.stdin.setRawMode === "function") { + vi.spyOn(process.stdin, "setRawMode").mockReturnValue(process.stdin); + } + + resetSettingsForTest(); + tempDir = TempDir.createSync("@pi-issue-4806-"); + await Settings.init({ inMemory: true, cwd: tempDir.path() }); + authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); + const modelRegistry = new ModelRegistry(authStorage); + const model = modelRegistry.find("anthropic", "claude-sonnet-4-5"); + if (!model) throw new Error("Expected claude-sonnet-4-5 test model"); + session = new AgentSession({ + agent: new Agent({ initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] } }), + sessionManager: SessionManager.create(tempDir.path(), tempDir.path()), + settings: Settings.isolated(), + modelRegistry, + }); + streaming = true; + Object.defineProperty(session, "isStreaming", { configurable: true, get: () => streaming }); + mode = new InteractiveMode(session, "test"); + mode.isInitialized = true; + mode.ui.requestRender = vi.fn(); + }); + + afterEach(async () => { + mode?.stop(); + HistoryStorage.resetInstance(); + vi.restoreAllMocks(); + await session?.dispose(); + authStorage?.close(); + tempDir?.removeSync(); + resetSettingsForTest(); + }); + + it("mounts slash-command output once after the active turn ends", async () => { + const streamedReply = new Text("agent is streaming", 0, 0); + mode.chatContainer.addChild(streamedReply); + + mode.handleToolsCommand(); + + expect(mode.chatContainer.children).toEqual([streamedReply]); + + streaming = false; + await mode.eventController.handleEvent({ type: "agent_end", messages: [] } as AgentSessionEvent); + + expect(mode.chatContainer.children).toHaveLength(2); + const transcript = mode.chatContainer.render(80).join("\n"); + expect(transcript.match(/Available Tools/g)).toHaveLength(1); + }); +}); diff --git a/packages/coding-agent/test/issue-956-repro.test.ts b/packages/coding-agent/test/issue-956-repro.test.ts index f5f7f39fd..8ee3138a8 100644 --- a/packages/coding-agent/test/issue-956-repro.test.ts +++ b/packages/coding-agent/test/issue-956-repro.test.ts @@ -84,6 +84,10 @@ describe("issue #956: interactive /mcp test", () => { for (const item of Array.isArray(content) ? content : [content]) addChild(item); requestRender(); }, + presentCommandOutput: (content: unknown) => { + for (const item of Array.isArray(content) ? content : [content]) addChild(item); + requestRender(); + }, ui: { requestRender }, editor: {}, showError, diff --git a/packages/coding-agent/test/mcp-command-reauth.test.ts b/packages/coding-agent/test/mcp-command-reauth.test.ts index 740554a71..ac9157a99 100644 --- a/packages/coding-agent/test/mcp-command-reauth.test.ts +++ b/packages/coding-agent/test/mcp-command-reauth.test.ts @@ -59,6 +59,7 @@ function createController(authStorage: AuthStorage, mcpManagerOverrides: Record< const controller = new MCPCommandController({ chatContainer: { addChild: vi.fn() }, present, + presentCommandOutput: present, ui: { requestRender: vi.fn() }, editor, showError, diff --git a/packages/coding-agent/test/mcp-command-toggle.test.ts b/packages/coding-agent/test/mcp-command-toggle.test.ts index 05593d8e8..2b948b4b3 100644 --- a/packages/coding-agent/test/mcp-command-toggle.test.ts +++ b/packages/coding-agent/test/mcp-command-toggle.test.ts @@ -53,6 +53,7 @@ function createController() { const controller = new MCPCommandController({ chatContainer: { addChild: vi.fn() }, present: vi.fn(), + presentCommandOutput: vi.fn(), ui: { requestRender: vi.fn() }, editor: {}, showError: vi.fn(), From 0c5458a248590ec7a2a7be532a4c9e4a62533409 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:39:28 +0000 Subject: [PATCH 012/293] fix(ai): recognized OpenRouter daily key limits - Classified free-models-per-day failures as credential-scoped quota exhaustion. - Added regression coverage proving auth retries switch to a healthy sibling key. Fixes #4832 --- packages/ai/CHANGELOG.md | 4 ++++ packages/ai/src/error/rate-limit.ts | 11 ++++++++++- packages/ai/test/auth-retry.test.ts | 20 ++++++++++++++++++++ 3 files changed, 34 insertions(+), 1 deletion(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9c99508fd..d982144ba 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenRouter daily free-model allowance errors (`free-models-per-day`) being treated as transient rate limits, so requests rotate from an exhausted API key to a healthy sibling credential. ([#4832](https://github.com/can1357/oh-my-pi/issues/4832)) + ## [16.3.11] - 2026-07-06 ### Fixed diff --git a/packages/ai/src/error/rate-limit.ts b/packages/ai/src/error/rate-limit.ts index fb5678188..a5f700c41 100644 --- a/packages/ai/src/error/rate-limit.ts +++ b/packages/ai/src/error/rate-limit.ts @@ -19,6 +19,7 @@ const SERVER_ERROR_BACKOFF_MS = 20 * 1000; // 20s const ACCOUNT_RATE_LIMIT_PATTERN = /\baccount(?:'s)?\b[^\n]{0,80}\brate.?limit\b|\brate.?limit\b[^\n]{0,80}\baccount\b/i; const INSUFFICIENT_BALANCE_PATTERN = /insufficient.?balance/i; +const OPENROUTER_DAILY_FREE_LIMIT_PATTERN = /\bfree[-_ ]models[-_ ]per[-_ ]day\b/i; /** * Classify a rate-limit error message into a reason category. @@ -54,6 +55,10 @@ export function parseRateLimitReason(errorMessage: string): RateLimitReason { return "QUOTA_EXHAUSTED"; } + if (OPENROUTER_DAILY_FREE_LIMIT_PATTERN.test(errorMessage)) { + return "QUOTA_EXHAUSTED"; + } + if ( lower.includes("per minute") || lower.includes("rate limit") || @@ -157,5 +162,9 @@ export function isOpaqueStatusBody(message: string): boolean { * {@link isUsageLimitOutcome} uses it for the account-rotation decision. */ export function matchesUsageLimitText(errorMessage: string): boolean { - return USAGE_LIMIT_PATTERN.test(errorMessage) || ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage); + return ( + USAGE_LIMIT_PATTERN.test(errorMessage) || + ACCOUNT_RATE_LIMIT_PATTERN.test(errorMessage) || + OPENROUTER_DAILY_FREE_LIMIT_PATTERN.test(errorMessage) + ); } diff --git a/packages/ai/test/auth-retry.test.ts b/packages/ai/test/auth-retry.test.ts index 73b1f66d1..35a39335a 100644 --- a/packages/ai/test/auth-retry.test.ts +++ b/packages/ai/test/auth-retry.test.ts @@ -99,6 +99,26 @@ describe("withAuth", () => { ]); }); + it("switches credentials when OpenRouter exhausts the daily free-model allowance", async () => { + const keys: string[] = []; + const result = await withAuth( + ctx => (ctx.error === undefined || !ctx.lastChance ? "exhausted-key" : "healthy-key"), + async key => { + keys.push(key); + if (key === "healthy-key") return "success"; + throw Object.assign( + new Error( + "429 Rate limit exceeded: free-models-per-day. Add 10 credits to unlock 1000 free model requests per day", + ), + { status: 429 }, + ); + }, + ); + + expect(result).toBe("success"); + expect(keys).toEqual(["exhausted-key", "healthy-key"]); + }); + it("stops retrying when the resolver returns undefined", async () => { const keys: string[] = []; const original = authError(); From 29c8cae9ba5c1ad2064dbc7aa964e92055dcfea6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:39:48 +0000 Subject: [PATCH 013/293] fix(catalog): extend stream idle floor to kimi k2.7 code Native Kimi K2.7 Code (kimi-k2.7-code / kimi-k2.7-code-highspeed) reasons for minutes before the first stream event like K2.6, but the streamIdleTimeoutMs branch in buildOpenAICompat gated only on isKimiK26ModelId, so K2.7 Code fell through to the 120s default and aborted on long reasoning turns. Match matchesKimiK27CodeFamily in the same branch and rename the constant to KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS. Fixes #4836 --- packages/catalog/CHANGELOG.md | 4 ++++ packages/catalog/src/compat/openai.ts | 8 ++++---- packages/catalog/test/build.test.ts | 21 +++++++++++++++++++++ 3 files changed, 29 insertions(+), 4 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 42e39218a..3ae8f9833 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Extended the reasoning `streamIdleTimeoutMs` floor (300s) to native Kimi K2.7 Code (`kimi-k2.7-code` / `kimi-k2.7-code-highspeed`), which previously fell through to the 120s default and aborted on long reasoning turns ([#4836](https://github.com/can1357/oh-my-pi/issues/4836)). + ## [16.3.11] - 2026-07-06 ### Added diff --git a/packages/catalog/src/compat/openai.ts b/packages/catalog/src/compat/openai.ts index d5627ccd5..8198f93b8 100644 --- a/packages/catalog/src/compat/openai.ts +++ b/packages/catalog/src/compat/openai.ts @@ -37,8 +37,8 @@ const GLM_CODING_PLAN_MODEL_PATTERN = /(^|\/)glm-5(?:[.-]|$)/i; const GLM_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS = 600_000; /** Direct DeepSeek reasoning models stall between thinking and answer phases. */ const DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; -/** Kimi K2.6 can spend several minutes reasoning before the first visible token. */ -const KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; +/** Kimi K2.6 and native K2.7 Code can spend several minutes reasoning before the first visible token. */ +const KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS = 300_000; /** * Native Kimi K2.7 Code requires `thinking.type: "enabled"` and rejects * disabled thinking. Match the public id, its Fast variant, and the @@ -383,8 +383,8 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv ? ALIBABA_CODING_PLAN_STREAM_IDLE_TIMEOUT_MS : isXiaomiMimo ? XIAOMI_MIMO_STREAM_IDLE_TIMEOUT_MS - : spec.reasoning && isKimiK26ModelId(spec.id) - ? KIMI_K26_REASONING_STREAM_IDLE_TIMEOUT_MS + : spec.reasoning && (isKimiK26ModelId(spec.id) || matchesKimiK27CodeFamily(spec)) + ? KIMI_REASONING_STREAM_IDLE_TIMEOUT_MS : spec.reasoning && isDirectDeepseekApi ? DEEPSEEK_REASONING_STREAM_IDLE_TIMEOUT_MS : isLocalOpenAICompatBackend diff --git a/packages/catalog/test/build.test.ts b/packages/catalog/test/build.test.ts index 706321869..628d01cd3 100644 --- a/packages/catalog/test/build.test.ts +++ b/packages/catalog/test/build.test.ts @@ -282,6 +282,27 @@ describe("openai-completions wire-quirk compat detection", () => { expect(buildOpenAICompat(completionsSpec()).reasoningDeltasMayBeCumulative).toBe(false); }); + it("extends the reasoning stream idle floor to Kimi K2.6 and K2.7 Code, not other reasoning models", () => { + const kimiOverrides = { + provider: "moonshot", + baseUrl: "https://api.moonshot.ai/v1", + reasoning: true, + } as const; + expect(buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.6" })).streamIdleTimeoutMs).toBe( + 300_000, + ); + expect(buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.7-code" })).streamIdleTimeoutMs).toBe( + 300_000, + ); + expect( + buildOpenAICompat(completionsSpec({ ...kimiOverrides, id: "kimi-k2.7-code-highspeed" })).streamIdleTimeoutMs, + ).toBe(300_000); + // A non-Kimi reasoning model on a generic host keeps the runtime default. + expect( + buildOpenAICompat(completionsSpec({ id: "some-reasoner", reasoning: true })).streamIdleTimeoutMs, + ).toBeUndefined(); + }); + it("maps the remaining provider-keyed wire quirks", () => { expect(buildOpenAICompat(completionsSpec({ provider: "ollama" })).emptyLengthFinishIsContextError).toBe(true); expect(buildOpenAICompat(completionsSpec()).emptyLengthFinishIsContextError).toBe(false); From 2997037ffaa2d69e80a74c889f361e2943adaa44 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:52:41 +0000 Subject: [PATCH 014/293] fix(mnemopi): pinned Windows ORT DLL path - Resolved ONNX Runtime from fastembed's own dependency graph. - Prepended the cached DLL directory before loading the native binding. - Added regression coverage for inherited stale runtime paths. Fixes #4849 --- packages/mnemopi/CHANGELOG.md | 4 + .../mnemopi/src/core/fastembed-runtime.ts | 92 ++++++++++++++++--- .../mnemopi/test/fastembed-runtime.test.ts | 24 ++++- 3 files changed, 105 insertions(+), 15 deletions(-) diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index 21170cfd9..f673ea666 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Mnemopi local embeddings on Windows loading an unrelated `onnxruntime.dll` from the inherited system path instead of fastembed's cached ORT runtime. ([#4849](https://github.com/can1357/oh-my-pi/issues/4849)) + ## [16.3.9] - 2026-07-06 ### Fixed diff --git a/packages/mnemopi/src/core/fastembed-runtime.ts b/packages/mnemopi/src/core/fastembed-runtime.ts index 3a8978fed..272fa0f76 100644 --- a/packages/mnemopi/src/core/fastembed-runtime.ts +++ b/packages/mnemopi/src/core/fastembed-runtime.ts @@ -48,6 +48,67 @@ export function fastembedRuntimeInstallPlan(): FastembedRuntimeInstallPlan { } let fastembedLoad: Promise | null = null; +/** Inputs for selecting the Windows DLL directory paired with a fastembed installation. */ +export interface WindowsFastembedRuntimeOptions { + /** Resolved fastembed package entry whose dependency graph owns the ORT binding. */ + fastembedEntry: string; + /** Directory containing fastembed's manifest and nested dependency graph. */ + fastembedPackageDir: string; + /** Native architecture to select; defaults to the current process architecture. */ + arch?: string; + /** Environment receiving the DLL search path; defaults to the subprocess environment. */ + env?: NodeJS.ProcessEnv; +} + +/** The ORT module and DLL directory selected from fastembed's own dependency graph. */ +export interface WindowsFastembedRuntime { + /** Resolved entry for fastembed's own ONNX Runtime dependency. */ + ortEntry: string; + /** Package directory containing the selected ORT manifest and native assets. */ + ortPackageDir: string; + /** Directory prepended to `PATH` so Windows finds the paired native DLL. */ + dllDir: string; +} + +/** + * Prepend the ORT DLL directory paired with fastembed before Bun loads its + * native binding. Compiled Windows binaries extract `.node` files to a + * temporary directory, so the default DLL search can otherwise select an + * unrelated `onnxruntime.dll` from the inherited system path. + */ +export async function prepareWindowsFastembedRuntime({ + fastembedEntry, + fastembedPackageDir, + arch = process.arch, + env = process.env, +}: WindowsFastembedRuntimeOptions): Promise { + const nestedNodeModules = path.join(fastembedPackageDir, "node_modules"); + const rootNodeModules = path.dirname(fastembedPackageDir); + const nestedOrtEntry = resolveRuntimeModule(nestedNodeModules, "onnxruntime-node"); + const ortEntry = nestedOrtEntry ?? resolveRuntimeModule(rootNodeModules, "onnxruntime-node"); + const ortPackageDir = path.join(nestedOrtEntry ? nestedNodeModules : rootNodeModules, "onnxruntime-node"); + if (!ortEntry) { + throw new Error(`Cannot find module onnxruntime-node beside ${fastembedEntry}`); + } + const dllGlob = new Bun.Glob(`bin/napi-*/win32/${arch}/onnxruntime.dll`); + let dllDir: string | undefined; + for await (const dll of dllGlob.scan({ cwd: ortPackageDir, absolute: true, onlyFiles: true })) { + dllDir = path.dirname(dll); + break; + } + if (!dllDir) { + throw new Error(`Cannot find module onnxruntime-node Windows DLL for ${arch} beside ${ortEntry}`); + } + + const currentPath = env.PATH; + const normalizedDllDir = path.resolve(dllDir).toLowerCase(); + const alreadyPresent = currentPath + ?.split(path.delimiter) + .some(entry => path.resolve(entry).toLowerCase() === normalizedDllDir); + if (!alreadyPresent) env.PATH = currentPath ? `${dllDir}${path.delimiter}${currentPath}` : dllDir; + return { ortEntry, ortPackageDir, dllDir }; +} + export function loadFastembed(): Promise { fastembedLoad ??= loadFastembedOnce().catch(error => { fastembedLoad = null; @@ -57,16 +118,14 @@ export function loadFastembed(): Promise { } async function loadFastembedOnce(): Promise { - // Dynamic imports: both packages are optional peers that eagerly load - // native addons and may be absent at runtime — a static import would load - // the addon at module-init and crash every consumer without the peers. try { - // Preload the pinned ORT before fastembed's nested ORT — only on Windows, - // where loading the older binding first triggers a DLL-reuse crash. - if (process.platform === "win32") { - await import("onnxruntime-node"); + const requireDirect = createRequire(import.meta.url); + const manifestPath = requireDirect.resolve("fastembed/package.json"); + const manifest: { version?: unknown } = requireDirect(manifestPath); + if (manifest.version !== FASTEMBED_SPEC) { + throw new Error(`Cannot find package fastembed@${FASTEMBED_SPEC}; resolved ${String(manifest.version)}`); } - return await import("fastembed"); + return loadResolvedFastembed(requireDirect.resolve("fastembed"), path.dirname(manifestPath)); } catch (error) { if (!isRecoverableFastembedLoadError(error)) throw error; logger.debug("mnemopi: fastembed not loadable, using on-demand runtime install", { @@ -76,6 +135,16 @@ async function loadFastembedOnce(): Promise { } } +async function loadResolvedFastembed(entry: string, fastembedPackageDir: string): Promise { + const requireFastembed = createRequire(entry); + if (process.platform === "win32") { + const { ortEntry } = await prepareWindowsFastembedRuntime({ fastembedEntry: entry, fastembedPackageDir }); + requireFastembed(ortEntry); + } + const loaded: FastembedModule = requireFastembed(entry); + return loaded; +} + async function loadFromRuntimeInstall(): Promise { const plan = fastembedRuntimeInstallPlan(); const runtimeDir = await ensureRuntimeInstalled({ @@ -89,14 +158,9 @@ async function loadFromRuntimeInstall(): Promise { // onnxruntime-node, @anush008/tokenizers → platform binding, …) through // the runtime cache. installRuntimeModuleResolver({ runtimeNodeModules: nodeModules }); - if (process.platform === "win32") { - const ortEntry = resolveRuntimeModule(nodeModules, "onnxruntime-node"); - if (ortEntry) createRequire(ortEntry)(ortEntry); - } const entry = resolveRuntimeModule(nodeModules, "fastembed"); if (!entry) throw new Error(`fastembed runtime install at ${runtimeDir} has no loadable entry`); - const requireRuntime = createRequire(entry); - return requireRuntime(entry) as FastembedModule; + return loadResolvedFastembed(entry, path.join(nodeModules, "fastembed")); } function isRecoverableFastembedLoadError(error: unknown): boolean { diff --git a/packages/mnemopi/test/fastembed-runtime.test.ts b/packages/mnemopi/test/fastembed-runtime.test.ts index 7b123cb72..1e2206329 100644 --- a/packages/mnemopi/test/fastembed-runtime.test.ts +++ b/packages/mnemopi/test/fastembed-runtime.test.ts @@ -1,7 +1,9 @@ import { describe, expect, test } from "bun:test"; +import { createRequire } from "node:module"; +import * as path from "node:path"; import rootManifest from "../../../package.json" with { type: "json" }; import packageManifest from "../package.json" with { type: "json" }; -import { fastembedRuntimeInstallPlan } from "../src/core/fastembed-runtime"; +import { fastembedRuntimeInstallPlan, prepareWindowsFastembedRuntime } from "../src/core/fastembed-runtime"; // The fastembed peer is pinned as an exact version (not `catalog:`) because // `core/fastembed-runtime.ts` reads it to `bun install` the on-demand embedding @@ -35,4 +37,24 @@ describe("fastembed runtime version pins", () => { expect(plan.versionKey).toContain("transitive-ort"); expect(plan.versionKey).not.toContain("forced-ort"); }); + + test("Windows preload selects fastembed's ORT DLL before inherited paths", async () => { + const requireTest = createRequire(import.meta.url); + const fastembedManifest = requireTest.resolve("fastembed/package.json"); + const fastembedEntry = requireTest.resolve("fastembed"); + const inheritedPath = ["/stale-ort", "/system"].join(path.delimiter); + const env: NodeJS.ProcessEnv = { PATH: inheritedPath }; + const { ortEntry, ortPackageDir, dllDir } = await prepareWindowsFastembedRuntime({ + fastembedEntry, + fastembedPackageDir: path.dirname(fastembedManifest), + arch: "x64", + env, + }); + const ortManifest: { version?: unknown } = requireTest(path.join(ortPackageDir, "package.json")); + + expect(ortManifest.version).toBe(packageManifest.peerDependencies["onnxruntime-node"]); + expect(ortEntry.startsWith(`${ortPackageDir}${path.sep}`)).toBe(true); + expect(await Bun.file(path.join(dllDir, "onnxruntime.dll")).exists()).toBe(true); + expect(env.PATH).toBe(`${dllDir}${path.delimiter}${inheritedPath}`); + }); }); From d866eb8589789de30622dc873554fe6fbe6430bd Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:54:33 +0000 Subject: [PATCH 015/293] fix(mnemopi): protect durable working memory from trim and cascade linked artifacts Working-memory TTL trim treated every consolidated_at IS NULL row as scratch, so restored or imported durable rows disappeared on the next write, and the trim delete left annotations, embeddings, facts, and memoria projections orphaned. - Exclude IMPORTED-tier rows from the trim eligibility query so restored banks survive a later remember()/rememberBatch(). - Stamp imported working-memory rows as consolidated in importFromDict so restored backups are durable regardless of trust tier. - Add purgeWorkingMemoryArtifacts() and route trim, forgetWorking, and force-import overwrite through it to cascade annotations, embeddings, facts (source_msg_id), memoria_* (source_memory_id), gists, and the graph edges tied to those memory/gist/fact node ids. Fixes #4819 --- packages/mnemopi/CHANGELOG.md | 4 + packages/mnemopi/src/core/beam/store.ts | 120 ++++++++++++++--- packages/mnemopi/test/beam-store.test.ts | 163 +++++++++++++++++++++++ 3 files changed, 269 insertions(+), 18 deletions(-) diff --git a/packages/mnemopi/CHANGELOG.md b/packages/mnemopi/CHANGELOG.md index 21170cfd9..938ea7844 100644 --- a/packages/mnemopi/CHANGELOG.md +++ b/packages/mnemopi/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed working-memory TTL trim silently deleting restored or imported durable rows: rows keeping `consolidated_at = NULL` with an old `timestamp` are no longer trimmed when flagged `IMPORTED`, `importFromDict` stamps imported rows as consolidated, and every working-memory delete path (trim, `forgetWorking`, force-import overwrite) now cascades linked annotations, embeddings, facts, memoria projections, gists, and graph edges instead of leaving orphans. ([#4819](https://github.com/can1357/oh-my-pi/issues/4819)) + ## [16.3.9] - 2026-07-06 ### Fixed diff --git a/packages/mnemopi/src/core/beam/store.ts b/packages/mnemopi/src/core/beam/store.ts index 5d0d2f9dd..e541d0ddb 100644 --- a/packages/mnemopi/src/core/beam/store.ts +++ b/packages/mnemopi/src/core/beam/store.ts @@ -163,27 +163,106 @@ function findDuplicate(beam: BeamMemoryState, content: string): string | null { return row?.id ?? null; } +function tableExists(db: BeamMemoryState["db"], table: string): boolean { + return ( + db + .prepare("SELECT 1 FROM sqlite_master WHERE type IN ('table','virtual table') AND name = ? LIMIT 1") + .get(table) !== null + ); +} + +/** Tables whose rows point back to a `working_memory` id via `source_memory_id`. */ +const MEMORIA_SOURCE_TABLES = [ + "memoria_facts", + "memoria_instructions", + "memoria_kg", + "memoria_preferences", + "memoria_timelines", +] as const; + +/** + * Remove every artifact linked to the given `working_memory` ids so no deletion + * path leaves orphans behind. Covers annotations, embeddings, extracted facts + * (`facts.source_msg_id`), memoria projections (`*.source_memory_id`), episodic + * gists, and the graph edges tied to those memory / gist / fact node ids. + * + * Idempotent and schema-tolerant: `gists` / `graph_edges` only exist once an + * `EpisodicGraph` has initialised, so they are guarded. Callers own the + * transaction and the base `working_memory` delete. + */ +function purgeWorkingMemoryArtifacts(db: BeamMemoryState["db"], ids: readonly string[]): void { + if (ids.length === 0) return; + const placeholders = ids.map(() => "?").join(", "); + + const graphRefs = new Set(ids); + for (const id of ids) graphRefs.add(`gist_${id}`); + if (tableExists(db, "facts")) { + const factRows = db.prepare(`SELECT fact_id FROM facts WHERE source_msg_id IN (${placeholders})`).all(...ids) as { + fact_id: string; + }[]; + for (const row of factRows) graphRefs.add(row.fact_id); + db.prepare(`DELETE FROM facts WHERE source_msg_id IN (${placeholders})`).run(...ids); + } + + db.prepare(`DELETE FROM annotations WHERE memory_id IN (${placeholders})`).run(...ids); + db.prepare(`DELETE FROM memory_embeddings WHERE memory_id IN (${placeholders})`).run(...ids); + for (const table of MEMORIA_SOURCE_TABLES) { + db.prepare(`DELETE FROM ${table} WHERE source_memory_id IN (${placeholders})`).run(...ids); + } + + if (tableExists(db, "gists")) { + db.prepare(`DELETE FROM gists WHERE memory_id IN (${placeholders})`).run(...ids); + } + if (tableExists(db, "graph_edges")) { + const refs = [...graphRefs]; + const refPlaceholders = refs.map(() => "?").join(", "); + db.prepare(`DELETE FROM graph_edges WHERE source IN (${refPlaceholders}) OR target IN (${refPlaceholders})`).run( + ...refs, + ...refs, + ); + } +} + +/** + * TTL / overflow trim for transient working memory. Only genuine scratch is + * eligible: `consolidated_at IS NULL` no longer suffices on its own, since + * restored or imported durable rows legitimately carry a NULL consolidation + * marker with an old event timestamp (issue #4819). Rows flagged `IMPORTED` + * are treated as durable and never trimmed, and trimmed rows cascade all linked + * artifacts via `purgeWorkingMemoryArtifacts`. + */ function trimWorkingMemory(beam: BeamMemoryState): void { const limit = beam.config.workingMemoryLimit; if (!Number.isFinite(limit) || limit <= 0) return; const ttlHours = beam.config.workingMemoryTtlHours; const cutoff = toUtcIso(new Date(Date.now() - ttlHours * 3_600_000)); - beam.db - .prepare(` - DELETE FROM working_memory - WHERE session_id = ? - AND consolidated_at IS NULL - AND ( - timestamp < ? OR - id NOT IN ( - SELECT id FROM working_memory - WHERE session_id = ? AND consolidated_at IS NULL - ORDER BY timestamp DESC - LIMIT ? - ) - ) - `) - .run(beam.sessionId, cutoff, beam.sessionId, limit); + const ids = ( + beam.db + .prepare(` + SELECT id FROM working_memory + WHERE session_id = ? + AND consolidated_at IS NULL + AND trust_tier IS NOT 'IMPORTED' + AND ( + timestamp < ? OR + id NOT IN ( + SELECT id FROM working_memory + WHERE session_id = ? AND consolidated_at IS NULL AND trust_tier IS NOT 'IMPORTED' + ORDER BY timestamp DESC + LIMIT ? + ) + ) + `) + .all(beam.sessionId, cutoff, beam.sessionId, limit) as { id: string }[] + ).map(row => row.id); + if (ids.length === 0) return; + const placeholders = ids.map(() => "?").join(", "); + transaction(beam.db, () => { + beam.db + .prepare(`DELETE FROM working_memory WHERE id IN (${placeholders}) AND session_id = ?`) + .run(...ids, beam.sessionId); + purgeWorkingMemoryArtifacts(beam.db, ids); + }); } function addTemporalAnnotations(beam: BeamMemoryState, memoryId: string, timestamp: string, source: string): void { @@ -676,7 +755,7 @@ export function forgetWorking(beam: BeamMemoryState, memoryId: string): boolean .run(memoryId, beam.sessionId); deleted = result.changes; if (deleted > 0) { - beam.db.prepare("DELETE FROM annotations WHERE memory_id = ?").run(memoryId); + purgeWorkingMemoryArtifacts(beam.db, [memoryId]); } }); if (deleted > 0) invalidateCaches(beam); @@ -772,6 +851,10 @@ export function importFromDict(beam: BeamMemoryState, data: Record(); transaction(db, () => { @@ -786,6 +869,7 @@ export function importFromDict(beam: BeamMemoryState, data: Record { expect(dest.db.prepare("SELECT COUNT(*) AS count FROM scratchpad").get()).toEqual({ count: 1 }); expect(scratchpadRead(dest).map(row => row.content)).toEqual([]); }); + + it("keeps restored durable rows and cascades linked artifacts on trim, force-import, and forget (issue #4819)", () => { + const beam = makeState("trim-4819"); + // EpisodicGraph owns the `gists` / `graph_edges` schema; init it so the + // cascade can be exercised end to end on the shared connection. + new EpisodicGraph({ db: beam.db, dbPath: ":memory:" }); + const oldTimestamp = new Date(Date.now() - 1000 * 3_600_000).toISOString(); + const countOf = (sql: string, ...params: (string | number | null)[]): number => { + const row = beam.db.prepare(sql).get(...params) as { count: number }; + return row.count; + }; + const seedArtifacts = (memoryId: string): void => { + beam.db + .prepare("INSERT INTO annotations (memory_id, kind, value) VALUES (?, 'mentions', 'Alice')") + .run(memoryId); + beam.db + .prepare("INSERT INTO memory_embeddings (memory_id, embedding_json, model) VALUES (?, '[0.1]', 't')") + .run(memoryId); + beam.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, source_msg_id) VALUES (?, 'trim-4819', 'Alice', 'is', 'User', ?)", + ) + .run(`fact-${memoryId}`, memoryId); + beam.db + .prepare( + "INSERT INTO memoria_facts (session_id, fact_type, key, value, source_memory_id) VALUES ('trim-4819', 'name', 'name', 'Alice', ?)", + ) + .run(memoryId); + beam.db + .prepare("INSERT INTO gists (id, text, memory_id) VALUES (?, 'g', ?)") + .run(`gist_${memoryId}`, memoryId); + beam.db + .prepare("INSERT INTO graph_edges (source, target, edge_type) VALUES (?, ?, 'ctx')") + .run(memoryId, `gist_${memoryId}`); + beam.db + .prepare("INSERT INTO graph_edges (source, target, edge_type) VALUES (?, ?, 'rel')") + .run(`gist_${memoryId}`, `fact-${memoryId}`); + }; + const artifactCount = (memoryId: string): number => + countOf("SELECT COUNT(*) AS count FROM annotations WHERE memory_id = ?", memoryId) + + countOf("SELECT COUNT(*) AS count FROM memory_embeddings WHERE memory_id = ?", memoryId) + + countOf("SELECT COUNT(*) AS count FROM facts WHERE source_msg_id = ?", memoryId) + + countOf("SELECT COUNT(*) AS count FROM memoria_facts WHERE source_memory_id = ?", memoryId) + + countOf("SELECT COUNT(*) AS count FROM gists WHERE memory_id = ?", memoryId) + + countOf( + "SELECT COUNT(*) AS count FROM graph_edges WHERE source = ? OR target = ? OR source = ? OR target = ?", + memoryId, + memoryId, + `gist_${memoryId}`, + `gist_${memoryId}`, + ); + + // (1) Restored durable row: old timestamp, consolidated_at NULL, IMPORTED tier. + const durableId = "restored-durable"; + beam.db + .prepare( + "INSERT INTO working_memory (id, content, source, timestamp, session_id, importance, trust_tier, consolidated_at) VALUES (?, 'canonical fact', 'backup', ?, 'trim-4819', 0.9, 'IMPORTED', NULL)", + ) + .run(durableId, oldTimestamp); + seedArtifacts(durableId); + + // (2) Transient scratch row: old timestamp, consolidated_at NULL, STATED tier. + const transientId = "transient-scratch"; + beam.db + .prepare( + "INSERT INTO working_memory (id, content, source, timestamp, session_id, importance, trust_tier, consolidated_at) VALUES (?, 'idle chatter', 'conversation', ?, 'trim-4819', 0.2, 'STATED', NULL)", + ) + .run(transientId, oldTimestamp); + seedArtifacts(transientId); + + // A normal write triggers the automatic trim. + remember(beam, "a fresh conversational note", { source: "conversation" }); + + // Durable row survives with all artifacts intact. + expect(get(beam, durableId)?.content).toBe("canonical fact"); + expect(artifactCount(durableId)).toBe(7); + + // Transient old row is trimmed and every linked artifact cascades. + expect(get(beam, durableId) === null).toBe(false); + expect(countOf("SELECT COUNT(*) AS count FROM working_memory WHERE id = ?", transientId)).toBe(0); + expect(artifactCount(transientId)).toBe(0); + + // (3) forgetWorking cascades every linked artifact, not just annotations. + expect(forgetWorking(beam, durableId)).toBe(true); + expect(artifactCount(durableId)).toBe(0); + }); + + it("marks imported working memory as consolidated so restored banks survive trim (issue #4819)", () => { + const dest = makeState("import-4819"); + const oldTimestamp = new Date(Date.now() - 1000 * 3_600_000).toISOString(); + importFromDict( + dest, + { + working_memory: [ + { + id: "restored-import", + content: "durable restored fact", + timestamp: oldTimestamp, + session_id: "import-4819", + trust_tier: "STATED", + consolidated_at: null, + }, + ], + }, + true, + ); + const importedRow = dest.db + .prepare("SELECT consolidated_at FROM working_memory WHERE id = 'restored-import'") + .get() as { consolidated_at: string | null }; + expect(importedRow.consolidated_at).not.toBeNull(); + + remember(dest, "a fresh note", { source: "conversation" }); + expect(get(dest, "restored-import")?.content).toBe("durable restored fact"); + }); + + it("force-import overwrite cleans stale linked artifacts of the replaced row (issue #4819)", () => { + const dest = makeState("import-overwrite-4819"); + const id = "overwrite-me"; + dest.db + .prepare( + "INSERT INTO working_memory (id, content, source, timestamp, session_id, importance, trust_tier) VALUES (?, 'stale', 'backup', '2020-01-01T00:00:00.000Z', 'import-overwrite-4819', 0.5, 'IMPORTED')", + ) + .run(id); + dest.db.prepare("INSERT INTO annotations (memory_id, kind, value) VALUES (?, 'mentions', 'Stale')").run(id); + dest.db + .prepare("INSERT INTO memory_embeddings (memory_id, embedding_json, model) VALUES (?, '[0.9]', 'old')") + .run(id); + dest.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, source_msg_id) VALUES ('stale-fact', 'import-overwrite-4819', 'a', 'is', 'b', ?)", + ) + .run(id); + + importFromDict( + dest, + { + working_memory: [ + { id, content: "fresh replacement", session_id: "import-overwrite-4819", trust_tier: "IMPORTED" }, + ], + }, + true, + ); + + expect(get(dest, id)?.content).toBe("fresh replacement"); + const staleArtifacts = + ( + dest.db.prepare("SELECT COUNT(*) AS count FROM annotations WHERE memory_id = ?").get(id) as { + count: number; + } + ).count + + ( + dest.db.prepare("SELECT COUNT(*) AS count FROM memory_embeddings WHERE memory_id = ?").get(id) as { + count: number; + } + ).count + + ( + dest.db.prepare("SELECT COUNT(*) AS count FROM facts WHERE source_msg_id = ?").get(id) as { + count: number; + } + ).count; + expect(staleArtifacts).toBe(0); + }); }); From df047effe336141e6d32990f43e281c764bdf1ef Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:55:58 +0000 Subject: [PATCH 016/293] fix(cli): guarded documented marketplace verbs from launch leak `omp marketplace add xyz` and the other documented `omp plugin ` verbs (marketplace, discover, upgrade, uninstall, enable, disable) were never registered top-level commands, so `resolveCliArgv` forwarded the whole argv to `launch` as an LLM prompt. The #2935 guard only caught bare single-word verbs, missing the multi-word documented grammar. Extended `reservedTopLevelWordMessage` to hint at the real `omp plugin ` command for these verbs when used bare, with a marketplace sub-action, or with a `name@marketplace` plugin id, while genuine prose prompts beginning with the same words still route to `launch`. Fixes #4845 --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/cli-commands.ts | 77 +++++++++++++------ .../test/plugin-verb-launch-leak.test.ts | 59 ++++++++++++-- 3 files changed, 107 insertions(+), 30 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bde1313a3..a27217aef 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -11,6 +11,7 @@ - Fixed advisor turns hammering the same usage-limited account: a failed advisor turn now marks the exhausted credential blocked (with the provider's retry hint and usage-report reset time), so the next retry rotates to a sibling instead of re-picking the blocked account every few seconds. Previously the in-stream auth retry rotated within a request but never blocked the last failing credential, and the advisor loop — unlike the primary retry pipeline — never called `markUsageLimitReached`. - Added the account key to the `codex-auto-reset: skipped` debug log so skip reasons (e.g. `weekly-not-exhausted`) can be attributed to the evaluated account. +- Fixed documented `omp marketplace`/`discover`/`upgrade`/`uninstall`/`enable`/`disable` CLI verbs silently leaking to the model as a launch prompt instead of managing plugins. `omp marketplace add xyz` (and similar multi-word invocations following the documented `omp plugin ` grammar) now surface a hint pointing at the real `omp plugin ` command, while genuine prose prompts beginning with these words still route to `launch` ([#4845](https://github.com/can1357/oh-my-pi/issues/4845)). ## [16.3.11] - 2026-07-06 diff --git a/packages/coding-agent/src/cli-commands.ts b/packages/coding-agent/src/cli-commands.ts index fb2b04072..d141c7d2c 100644 --- a/packages/coding-agent/src/cli-commands.ts +++ b/packages/coding-agent/src/cli-commands.ts @@ -46,31 +46,62 @@ export const commands: CommandEntry[] = [ { name: "search", load: () => import("./commands/web-search").then(m => m.default), aliases: ["q"] }, ]; -// Documented-looking plugin-management verbs that are NOT registered top-level -// commands. Without a guard `resolveCliArgv` rewrites e.g. `omp list` to -// `omp launch list`, silently forwarding the bare verb to the model as a prompt -// instead of managing plugins (#2935; same class as the `install` leak fixed in -// #1496/#1498). A bare (single-arg) use gets a hint pointing at the real -// `omp plugin ` command; multi-word invocations still fall through to -// `launch`, so genuine prompts that merely begin with one of these words work. -const RESERVED_TOP_LEVEL_WORDS = new Map([ - [ - "extensions", +// Documented-looking plugin/marketplace verbs that are NOT registered top-level +// commands. Without a guard `resolveCliArgv` rewrites e.g. `omp marketplace add +// xyz` to `omp launch marketplace add xyz`, silently forwarding the argv to the +// model as a prompt instead of managing plugins (#4845; same class as the +// `list`/`remove` leak fixed in #2935 and the `install` leak in #1496/#1498). +// The real commands live under `omp plugin `; each entry maps a verb to +// a hint pointing there. See {@link reservedTopLevelWordMessage} for when a hint +// fires vs. when the argv still falls through to `launch`. +const RESERVED_TOP_LEVEL_WORDS: Record = { + extensions: '`omp extensions` is not a management command. Use `omp plugin list` / `omp plugin install`, or run `omp launch extensions` if you meant to send "extensions" as a prompt.', - ], - [ - "list", - '`omp list` is not a top-level command. Use `omp plugin list` to list installed plugins, or run `omp launch list` if you meant to send "list" as a prompt.', - ], - [ - "remove", + list: '`omp list` is not a top-level command. Use `omp plugin list` to list installed plugins, or run `omp launch list` if you meant to send "list" as a prompt.', + remove: '`omp remove` is not a top-level command. Use `omp plugin uninstall ` to remove a plugin, or run `omp launch remove` if you meant to send "remove" as a prompt.', - ], -]); + uninstall: + '`omp uninstall` is not a top-level command. Use `omp plugin uninstall ` to remove a plugin, or run `omp launch uninstall` if you meant to send "uninstall" as a prompt.', + marketplace: + '`omp marketplace` is not a top-level command. Use `omp plugin marketplace ` to manage marketplaces, or run `omp launch marketplace` if you meant to send "marketplace" as a prompt.', + discover: + '`omp discover` is not a top-level command. Use `omp plugin discover [marketplace]` to browse available plugins, or run `omp launch discover` if you meant to send "discover" as a prompt.', + upgrade: + '`omp upgrade` is not a top-level command. Use `omp plugin upgrade [name@marketplace]` to upgrade plugins, or run `omp launch upgrade` if you meant to send "upgrade" as a prompt.', + enable: + '`omp enable` is not a top-level command. Use `omp plugin enable ` to enable a plugin, or run `omp launch enable` if you meant to send "enable" as a prompt.', + disable: + '`omp disable` is not a top-level command. Use `omp plugin disable ` to disable a plugin, or run `omp launch disable` if you meant to send "disable" as a prompt.', +}; -export function reservedTopLevelWordMessage(first: string | undefined, argc = 1): string | undefined { - if (argc !== 1 || !first || first.startsWith("-") || first.startsWith("@")) return undefined; - return RESERVED_TOP_LEVEL_WORDS.get(first); +// Sub-actions that make `omp marketplace ` unambiguously a management +// command even when multi-word (the reporter's `omp marketplace add xyz`, +// #4845). Mirrors the switch in `handleMarketplace` (cli/plugin-cli.ts). +const MARKETPLACE_SUBCOMMANDS: Record = { add: true, remove: true, rm: true, update: true, list: true }; + +/** + * Hint for a reserved plugin/marketplace verb used as a top-level command, or + * `undefined` when the argv should fall through to `launch`. + * + * A bare verb (`omp marketplace`) always hints. A multi-word invocation only + * hints when the arguments follow the documented plugin grammar — a marketplace + * sub-action (`omp marketplace add …`) or a `name@marketplace` plugin id + * (`omp uninstall foo@bar`) — so genuine prompts that merely begin with one of + * these words (`omp list all my files`, `omp upgrade the deps`) still launch. + * + * Flags (`-…`) and `@file` arguments in the verb slot are never management + * commands; those fall through to the default `launch` command. + */ +export function reservedTopLevelWordMessage(argv: readonly string[]): string | undefined { + const first = argv[0]; + if (!first || first.startsWith("-") || first.startsWith("@")) return undefined; + const hint = RESERVED_TOP_LEVEL_WORDS[first]; + if (!hint) return undefined; + const second = argv[1]; + if (second === undefined) return hint; + if (first === "marketplace" && MARKETPLACE_SUBCOMMANDS[second]) return hint; + if (second.includes("@")) return hint; + return undefined; } /** @@ -112,7 +143,7 @@ function leadingSubcommandIndex(argv: string[]): number { */ export function resolveCliArgv(argv: string[]): ResolvedCliArgv { const first = argv[0]; - const reservedMessage = reservedTopLevelWordMessage(first, argv.length); + const reservedMessage = reservedTopLevelWordMessage(argv); if (reservedMessage) return { error: reservedMessage }; if (first === "--help" || first === "-h" || first === "--version" || first === "-v" || first === "help") { return { argv }; diff --git a/packages/coding-agent/test/plugin-verb-launch-leak.test.ts b/packages/coding-agent/test/plugin-verb-launch-leak.test.ts index a13759fe7..1714ae26c 100644 --- a/packages/coding-agent/test/plugin-verb-launch-leak.test.ts +++ b/packages/coding-agent/test/plugin-verb-launch-leak.test.ts @@ -1,16 +1,20 @@ /** - * Regression test for #2935: the plugins docs advertise `omp list` / `omp remove` + * Regression test for #2935 and #4845: the plugins/marketplace docs advertise + * `omp list` / `omp remove` / `omp marketplace ` / `omp uninstall …` etc. * as top-level commands, but only `omp install` is registered. Before the fix, * `resolveCliArgv(["list"])` rewrote the bare verb to `["launch", "list"]`, so * `omp list` silently started an interactive agent session with "list" as the * initial LLM prompt instead of managing plugins (the real command is - * `omp plugin list`). Same footgun for `omp remove`. + * `omp plugin list`). #4845 extended the same footgun to the multi-word + * documented grammar: `omp marketplace add xyz` leaked the whole argv to the + * model as a prompt. * - * These tests pin the chosen bugfix: a bare, single-arg documented plugin verb - * yields a helpful hint pointing at the real `omp plugin ` command - * rather than leaking the word to the model — while multi-word invocations that - * merely happen to begin with one of these verbs still fall through to `launch` - * so genuine prompts are unaffected. + * These tests pin the chosen bugfix: a documented plugin/marketplace verb that + * is bare, or that follows the documented grammar (a marketplace sub-action or a + * `name@marketplace` plugin id), yields a helpful hint pointing at the real + * `omp plugin ` command rather than leaking to the model — while + * genuine prose prompts that merely begin with one of these words still fall + * through to `launch`. * * Imported via a relative path (not the `@oh-my-pi/pi-coding-agent` alias) so the * assertions exercise this checkout's `cli-commands.ts` directly. @@ -52,4 +56,45 @@ describe("documented-but-unregistered plugin verbs do not leak to launch (#2935) expect(isSubcommand("list")).toBe(false); expect(isSubcommand("remove")).toBe(false); }); + + test("multi-word `omp marketplace add xyz` hints at `omp plugin marketplace` instead of leaking to the prompt (#4845)", () => { + const result = resolveCliArgv(["marketplace", "add", "xyz"]); + expect(result).not.toEqual({ argv: ["launch", "marketplace", "add", "xyz"] }); + expect(result).not.toHaveProperty("argv"); + expect(result).toHaveProperty("error"); + expect("error" in result && result.error).toContain("omp plugin marketplace"); + }); + + test("bare marketplace-family verbs hint at their `omp plugin` command (#4845)", () => { + for (const [verb, hint] of [ + ["marketplace", "omp plugin marketplace"], + ["discover", "omp plugin discover"], + ["upgrade", "omp plugin upgrade"], + ["uninstall", "omp plugin uninstall"], + ["enable", "omp plugin enable"], + ["disable", "omp plugin disable"], + ] as const) { + const result = resolveCliArgv([verb]); + expect(result).not.toHaveProperty("argv"); + expect("error" in result && result.error).toContain(hint); + } + }); + + test("`name@marketplace` plugin ids hint instead of launching (#4845)", () => { + for (const verb of ["uninstall", "upgrade", "enable", "disable"] as const) { + const result = resolveCliArgv([verb, "code-review@claude-plugins-official"]); + expect(result).not.toHaveProperty("argv"); + expect(result).toHaveProperty("error"); + } + }); + + test("prose prompts beginning with the new verbs still route to launch (#4845)", () => { + expect(resolveCliArgv(["upgrade", "the", "deps"])).toEqual({ + argv: ["launch", "upgrade", "the", "deps"], + }); + // `marketplace` followed by a non-subcommand word is a genuine prompt. + expect(resolveCliArgv(["marketplace", "research", "for", "me"])).toEqual({ + argv: ["launch", "marketplace", "research", "for", "me"], + }); + }); }); From 2d519d8f89f672a0247440f70602e4c39f1f1cd3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 16:56:20 +0000 Subject: [PATCH 017/293] fix(tui): guard plugin tool renderer components against render crashes Plugin/custom tool renderers can return a component whose render() throws (e.g. a plugin styling its header off an object without a bold method, producing TypeError: th.bold is not a function). ToolExecutionComponent only caught the renderCall/renderResult factory, not the child component's later render() pass, so the exception escaped and crashed the transcript. Wrap every renderer-returned call/result component in SafeToolRendererComponent, which catches render() errors and falls back to the tool label (call) or raw result text (result), logging once per component. Fixes #4978 --- packages/coding-agent/CHANGELOG.md | 4 + .../modes/components/tool-execution.test.ts | 101 ++++++++++++++ .../src/modes/components/tool-execution.ts | 126 ++++++++++++++++-- 3 files changed, 222 insertions(+), 9 deletions(-) create mode 100644 packages/coding-agent/src/modes/components/tool-execution.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 78d22734a..5355980a9 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed a crash when a plugin/custom tool renderer returns a component that throws during its later `render()` pass (e.g. `TypeError: th.bold is not a function` from a plugin that styles its header off an object without a `bold` method). `ToolExecutionComponent` now wraps every renderer-returned call/result component so a throwing `render()` degrades to the safe fallback (tool label or raw result text) instead of taking down the transcript ([#4978](https://github.com/can1357/oh-my-pi/issues/4978)). + ## [16.3.14] - 2026-07-09 ### Fixed diff --git a/packages/coding-agent/src/modes/components/tool-execution.test.ts b/packages/coding-agent/src/modes/components/tool-execution.test.ts new file mode 100644 index 000000000..1408f80d7 --- /dev/null +++ b/packages/coding-agent/src/modes/components/tool-execution.test.ts @@ -0,0 +1,101 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import type { AgentTool } from "@oh-my-pi/pi-agent-core"; +import { type Component, Text } from "@oh-my-pi/pi-tui"; +import { Settings } from "../../config/settings"; +import { getThemeByName, setThemeInstance, theme } from "../theme/theme"; +import { ToolExecutionComponent, type ToolExecutionUi } from "./tool-execution"; + +class BoldTypeErrorComponent implements Component { + render(_width: number): readonly string[] { + throw new TypeError("th.bold is not a function"); + } +} + +function visibleText(lines: readonly string[]): string { + let text = lines.join("\n"); + text = text.replace(/\x1b\]8;[^\x1b\x07]*(?:\x07|\x1b\\)/g, ""); + text = text.replace(/\x1b\[[0-9;]*m/g, ""); + return text; +} + +describe("ToolExecutionComponent custom renderer failures", () => { + beforeAll(async () => { + await Settings.init({ inMemory: true }); + const loaded = await getThemeByName("dark"); + if (!loaded) throw new Error("theme unavailable"); + setThemeInstance(loaded); + }); + + it("falls back to the custom tool label when a renderCall child component throws during render", () => { + const tool: AgentTool = { + name: "graphify_graph", + label: "Graphify Graph", + description: "renders a graph", + parameters: { type: "object", additionalProperties: true }, + renderCall() { + return new BoldTypeErrorComponent(); + }, + async execute() { + return { content: [{ type: "text", text: "ok" }] }; + }, + }; + const ui: ToolExecutionUi = { + requestRender() {}, + requestComponentRender(_component: Component) {}, + resetDisplay() {}, + }; + const component = new ToolExecutionComponent( + "graphify_graph", + {}, + { showImages: false }, + tool, + ui, + process.cwd(), + ); + let text = ""; + + expect(() => { + text = visibleText(component.render(80)); + }).not.toThrow(); + expect(text).toContain("Graphify Graph"); + }); + + it("preserves raw result text when a renderResult child component throws during render", () => { + const rawResultText = "raw result survives child renderer failure"; + const tool: AgentTool = { + name: "crashy_result_renderer", + label: "Crashy Result Renderer", + description: "renders result output", + parameters: { type: "object", additionalProperties: true }, + renderCall() { + return new Text(theme.fg("toolTitle", theme.bold("Crashy Result Renderer")), 0, 0); + }, + renderResult() { + return new BoldTypeErrorComponent(); + }, + async execute() { + return { content: [{ type: "text", text: rawResultText }] }; + }, + }; + const ui: ToolExecutionUi = { + requestRender() {}, + requestComponentRender(_component: Component) {}, + resetDisplay() {}, + }; + const component = new ToolExecutionComponent( + "crashy_result_renderer", + {}, + { showImages: false }, + tool, + ui, + process.cwd(), + ); + component.updateResult({ content: [{ type: "text", text: rawResultText }] }, false); + let text = ""; + + expect(() => { + text = visibleText(component.render(80)); + }).not.toThrow(); + expect(text).toContain(rawResultText); + }); +}); diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index de9f9c483..fae15856e 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -40,7 +40,7 @@ import { } from "../../tools/render-utils"; import { type FirstResultViewportRepaint, toolRenderers } from "../../tools/renderers"; import { TODO_STRIKE_TOTAL_FRAMES, type TodoToolDetails } from "../../tools/todo"; -import { isFramedBlockComponent, renderStatusLine, WidthAwareText } from "../../tui"; +import { isFramedBlockComponent, markFramedBlockComponent, renderStatusLine, WidthAwareText } from "../../tui"; import { sanitizeWithOptionalSixelPassthrough } from "../../utils/sixel"; import { renderDiff } from "./diff"; @@ -148,6 +148,68 @@ function getArgsWithStreamedTextInput(args: unknown): unknown { return input === undefined ? args : { ...record, input }; } +type ToolRendererStage = "call" | "result"; + +class SafeToolRendererComponent implements Component { + #toolName: string; + #stage: ToolRendererStage; + #component: Component; + #fallback: () => Component | undefined; + #warned = false; + readonly wantsKeyRelease: boolean | undefined; + + constructor( + toolName: string, + stage: ToolRendererStage, + component: Component, + fallback: () => Component | undefined, + ) { + this.#toolName = toolName; + this.#stage = stage; + this.#component = component; + this.#fallback = fallback; + this.wantsKeyRelease = component.wantsKeyRelease; + if (isFramedBlockComponent(component)) { + markFramedBlockComponent(this); + } + } + + render(width: number): readonly string[] { + try { + return this.#component.render(width); + } catch (err) { + if (!this.#warned) { + this.#warned = true; + logger.warn("Tool renderer failed", { tool: this.#toolName, stage: this.#stage, error: String(err) }); + } + return this.#fallback()?.render(width) ?? []; + } + } + + handleInput(data: string): void { + const handleInput = this.#component.handleInput; + if (handleInput === undefined) return; + handleInput.call(this.#component, data); + } + + invalidate(): void { + const invalidate = this.#component.invalidate; + if (invalidate === undefined) return; + invalidate.call(this.#component); + } + + setIgnoreTight(ignore: boolean): void { + const setIgnoreTight = this.#component.setIgnoreTight; + if (setIgnoreTight === undefined) return; + setIgnoreTight.call(this.#component, ignore); + } + + dispose(): void { + const dispose = this.#component.dispose; + if (dispose === undefined) return; + dispose.call(this.#component); + } +} /** * Transcript-side probe telling a block whether it is still inside the live * (repaintable) region. Implemented by `TranscriptContainer`; injected rather @@ -157,6 +219,14 @@ export interface TranscriptLiveRegionProbe { isBlockInLiveRegion(component: Component): boolean; } +/** Minimal TUI surface ToolExecutionComponent uses to schedule repaints and share image budget. */ +export interface ToolExecutionUi { + requestRender(): void; + requestComponentRender(component: Component): void; + resetDisplay(): void; + imageBudget?: TUI["imageBudget"]; +} + export interface ToolExecutionOptions { snapshots?: SnapshotStore; showImages?: boolean; // default: true (only used if terminal supports images) @@ -238,7 +308,7 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac // forcing the common image-free result to re-shape on every resize tick. #renderedImageCount = 0; #tool?: AgentTool; - #ui: TUI; + #ui: ToolExecutionUi; #cwd: string; #result?: { content: Array<{ type: string; text?: string; data?: string; mimeType?: string }>; @@ -306,7 +376,7 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac args: any, options: ToolExecutionOptions = {}, tool: AgentTool | undefined, - ui: TUI, + ui: ToolExecutionUi, cwd: string = getProjectDir(), _toolCallId?: string, ) { @@ -901,8 +971,17 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac if (tool.renderCall) { try { const callArgs = this.#getCallArgsForRender(); - const callComponent = tool.renderCall(callArgs, this.#renderState, theme); - if (callComponent) this.#contentBox.addChild(callComponent as Component); + const callComponent = tool.renderCall(callArgs, this.#renderState, theme) as Component | undefined; + if (callComponent) { + this.#contentBox.addChild( + new SafeToolRendererComponent( + this.#toolName, + "call", + callComponent, + () => new Text(theme.fg("toolTitle", theme.bold(this.#toolLabel)), 0, 0), + ), + ); + } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); // Fall back to default on error @@ -933,7 +1012,15 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac theme, this.#args, ); - if (resultComponent) this.#contentBox.addChild(resultComponent); + if (resultComponent) { + this.#contentBox.addChild( + new SafeToolRendererComponent(this.#toolName, "result", resultComponent, () => { + const output = this.#getTextOutput(); + if (!output) return undefined; + return new Text(theme.fg("toolOutput", replaceTabs(output)), 0, 0); + }), + ); + } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); // Fall back to showing raw output on error @@ -990,7 +1077,11 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac this.#renderState, theme, ); - if (resultComponent) fileBox.addChild(resultComponent); + if (resultComponent) { + fileBox.addChild( + new SafeToolRendererComponent(this.#toolName, "result", resultComponent, () => undefined), + ); + } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); } @@ -1037,7 +1128,16 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac try { const callArgs = this.#getCallArgsForRender(); const callComponent = renderer.renderCall(callArgs, this.#renderState, theme); - if (callComponent) this.#contentBox.addChild(callComponent); + if (callComponent) { + this.#contentBox.addChild( + new SafeToolRendererComponent( + this.#toolName, + "call", + callComponent, + () => new Text(theme.fg("toolTitle", theme.bold(this.#toolLabel)), 0, 0), + ), + ); + } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); // Fall back to default on error @@ -1058,7 +1158,15 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac theme, this.#getCallArgsForRender(), ); - if (resultComponent) this.#contentBox.addChild(resultComponent); + if (resultComponent) { + this.#contentBox.addChild( + new SafeToolRendererComponent(this.#toolName, "result", resultComponent, () => { + const output = this.#getTextOutput(); + if (!output) return undefined; + return new Text(theme.fg("toolOutput", replaceTabs(output)), 0, 0); + }), + ); + } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); // Fall back to showing raw output on error From 9ef5538f5c2c088f0d81d73db2b25fadbd29d21a Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:11:48 +0000 Subject: [PATCH 018/293] fix(rpc): tolerated malformed stdin lines instead of crashing RPC mode iterated readJsonl(Bun.stdin.stream()), where JSON parsing runs inside the generator. A parse error thrown from the generator escaped the frame loop's try/catch (which only wrapped dispatch), unwound runRpcMode, and exited the process on any non-JSON stdin line. Read raw lines via readLines and JSON.parse each inside the existing try/catch so a malformed line emits a Failed to parse command error frame and the loop keeps running. Shared readJsonl stays strict for session-file reads. Fixes #5194 --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/modes/rpc/rpc-mode.ts | 17 ++++-- .../test/rpc-malformed-input.test.ts | 54 +++++++++++++++++++ 3 files changed, 67 insertions(+), 5 deletions(-) create mode 100644 packages/coding-agent/test/rpc-malformed-input.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 049ccb186..82f911106 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -31,6 +31,7 @@ - Fixed visible per-keystroke lag while searching in the `/resume` session picker. Literal matches now rank synchronously from a cached per-session haystack, fuzzy scoring runs in bounded background chunks that converge to the same ranking (large listings previously rebuilt a fuzzy index per token per session on every keystroke), and the prompt-history SQLite lookup — an FTS query plus a LIKE scan over every stored prompt — is debounced off the keystroke path. - Fixed compiled Linux binary extension loading when bundled web-search header generation cannot read `header-generator` data files from the build-time path. ([#5178](https://github.com/can1357/oh-my-pi/issues/5178)) +- Fixed RPC mode (`--mode rpc`) crashing the whole process with an uncaught `SyntaxError: Failed to parse JSONL` on any non-JSON stdin line. Malformed lines are now reported via a `Failed to parse command` error frame and the frame loop keeps running. ([#5194](https://github.com/can1357/oh-my-pi/issues/5194)) ## [16.4.4] - 2026-07-11 diff --git a/packages/coding-agent/src/modes/rpc/rpc-mode.ts b/packages/coding-agent/src/modes/rpc/rpc-mode.ts index 16d1cf1b3..2e1c630b4 100644 --- a/packages/coding-agent/src/modes/rpc/rpc-mode.ts +++ b/packages/coding-agent/src/modes/rpc/rpc-mode.ts @@ -12,7 +12,7 @@ */ import { getOAuthProviders } from "@oh-my-pi/pi-ai/oauth"; import { isZodSchema, zodToWireSchema } from "@oh-my-pi/pi-ai/utils/schema"; -import { $env, isRecord, readJsonl, Snowflake } from "@oh-my-pi/pi-utils"; +import { $env, isRecord, readLines, Snowflake } from "@oh-my-pi/pi-utils"; import { reset as resetCapabilities } from "../../capability"; import { clearPluginRootsAndCaches, resolveActiveProjectRegistryPath } from "../../discovery/helpers"; import { @@ -1279,11 +1279,18 @@ export async function runRpcMode( onHostUriResult: frame => hostUriBridge.handleResult(frame), }; - // Listen for JSON input using Bun's stdin. Frame dispatch lives in - // dispatchRpcInputFrame so it can be exercised directly by tests; see the - // helper's docstring for the concurrency contract. - for await (const parsed of readJsonl(Bun.stdin.stream())) { + // Listen for JSON input using Bun's stdin. Frames are read line-by-line and + // parsed here (not via readJsonl) so a single malformed line is reported as + // an error frame and the loop keeps running instead of throwing out of the + // generator and killing the whole process (issue #5194). Frame dispatch + // lives in dispatchRpcInputFrame so it can be exercised directly by tests; + // see the helper's docstring for the concurrency contract. + const decoder = new TextDecoder(); + for await (const line of readLines(Bun.stdin.stream())) { + const text = decoder.decode(line).trim(); + if (!text) continue; try { + const parsed = JSON.parse(text); const awaited = dispatchRpcInputFrame(parsed, dispatchFrameDeps); if (awaited) { await awaited; diff --git a/packages/coding-agent/test/rpc-malformed-input.test.ts b/packages/coding-agent/test/rpc-malformed-input.test.ts new file mode 100644 index 000000000..31bd39507 --- /dev/null +++ b/packages/coding-agent/test/rpc-malformed-input.test.ts @@ -0,0 +1,54 @@ +import { describe, expect, test } from "bun:test"; +import * as path from "node:path"; +import { isRecord, readJsonl } from "@oh-my-pi/pi-utils"; + +/** + * Regression test for issue #5194: a non-JSON stdin line crashed the whole RPC + * process with an uncaught `SyntaxError: Failed to parse JSONL` escaping the + * frame loop. A malformed line must instead be reported as an error frame and + * the process must keep reading subsequent frames. + */ +describe("RPC mode malformed stdin", () => { + test("reports a bad line as an error frame and keeps serving subsequent commands", async () => { + const cliPath = path.join(import.meta.dir, "..", "src", "cli.ts"); + const child = Bun.spawn( + ["bun", cliPath, "--mode", "rpc", "--provider", "anthropic", "--model", "claude-sonnet-4-5"], + { + cwd: path.join(import.meta.dir, ".."), + env: { ...Bun.env, PI_NO_TITLE: "1" }, + stdin: "pipe", + stdout: "pipe", + stderr: "pipe", + }, + ); + + // A non-JSON line followed by a valid command. Pre-fix the first line + // crashed the generator before the second was ever read. + child.stdin.write("this is not json\n"); + child.stdin.write(`${JSON.stringify({ type: "get_state", id: "probe" })}\n`); + await child.stdin.flush(); + + let parseError: Record | undefined; + let stateResponse: Record | undefined; + + for await (const frame of readJsonl(child.stdout as ReadableStream)) { + if (!isRecord(frame)) continue; + if (frame.type === "response" && frame.command === "parse" && frame.success === false) { + parseError = frame; + } + if (frame.type === "response" && frame.id === "probe") { + stateResponse = frame; + break; + } + } + + child.stdin.end(); + child.kill(); + await child.exited.catch(() => {}); + + expect(parseError).toBeDefined(); + expect(String(parseError?.error)).toContain("Failed to parse command"); + expect(stateResponse).toBeDefined(); + expect(stateResponse?.success).toBe(true); + }, 30000); +}); From 12c693a8576131dc12f5a58cbaf3dedd7ca9643a Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:10:23 +0000 Subject: [PATCH 019/293] fix(tools): preferred active image provider with fallback Preferred the active session provider after any explicit image preference and retained the configured auto order for remaining candidates. Continued to the next credentialed image provider after HTTP failures. Fixes #5218 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/config/settings-schema.ts | 3 +- packages/coding-agent/src/tools/image-gen.ts | 1160 +++++++++-------- .../coding-agent/test/tools/image-gen.test.ts | 88 ++ 4 files changed, 689 insertions(+), 563 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 049ccb186..3cd38130a 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -31,6 +31,7 @@ - Fixed visible per-keystroke lag while searching in the `/resume` session picker. Literal matches now rank synchronously from a cached per-session haystack, fuzzy scoring runs in bounded background chunks that converge to the same ranking (large listings previously rebuilt a fuzzy index per token per session on every keystroke), and the prompt-history SQLite lookup — an FTS query plus a LIKE scan over every stored prompt — is debounced off the keystroke path. - Fixed compiled Linux binary extension loading when bundled web-search header generation cannot read `header-generator` data files from the build-time path. ([#5178](https://github.com/can1357/oh-my-pi/issues/5178)) +- Fixed `generate_image` preferring Antigravity over the active session provider and stopping instead of trying the next credentialed provider after an image HTTP failure. ([#5218](https://github.com/can1357/oh-my-pi/issues/5218)) ## [16.4.4] - 2026-07-11 diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 7a7d5b6c9..54e34e5d3 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -4402,7 +4402,8 @@ export const SETTINGS_SCHEMA = { { value: "auto", label: "Auto", - description: "Priority: GPT model image tool > Antigravity > xAI > OpenRouter > Gemini", + description: + "Priority: active session provider > GPT model image tool > Antigravity > xAI > OpenRouter > Gemini", }, { value: "openai", label: "OpenAI", description: "Uses the active GPT Responses/Codex model" }, { diff --git a/packages/coding-agent/src/tools/image-gen.ts b/packages/coding-agent/src/tools/image-gen.ts index ab8a332fc..039855495 100644 --- a/packages/coding-agent/src/tools/image-gen.ts +++ b/packages/coding-agent/src/tools/image-gen.ts @@ -58,6 +58,7 @@ const COMMON_IMAGE_ASPECT_RATIOS = ["1:1", "3:4", "4:3", "9:16", "16:9"] as cons const XAI_IMAGE_ASPECT_RATIOS = [...COMMON_IMAGE_ASPECT_RATIOS, "3:2", "2:3"] as const; const COMMON_IMAGE_ASPECT_RATIO_SET = new Set(COMMON_IMAGE_ASPECT_RATIOS); const IMAGE_PROVIDER_PREFERENCES = new Set(["auto", "antigravity", "gemini", "openai", "openrouter", "xai"]); +const AUTO_IMAGE_PROVIDER_ORDER = ["openai", "antigravity", "xai", "openrouter", "gemini"] as const; const responseModalitySchema = type('"IMAGE" | "TEXT"'); @@ -547,53 +548,58 @@ async function findOpenAIHostedImageCredentials( }; } +function activeImageProvider(model: Model | undefined): Exclude | null { + switch (model?.provider) { + case "openai": + case "openai-codex": + return "openai"; + case "google-antigravity": + return "antigravity"; + case "xai": + case "xai-oauth": + return "xai"; + case "openrouter": + return "openrouter"; + case "google": + return "gemini"; + default: + return null; + } +} + +function imageProviderOrder(activeModel: Model | undefined): Array> { + const providers: Array> = []; + const added = new Set>(); + const add = (provider: Exclude | null): void => { + if (!provider || added.has(provider)) return; + added.add(provider); + providers.push(provider); + }; + + if (preferredImageProvider !== "auto") add(preferredImageProvider); + add(activeImageProvider(activeModel)); + for (const provider of AUTO_IMAGE_PROVIDER_ORDER) add(provider); + return providers; +} + async function findImageApiKey( + provider: Exclude, modelRegistry?: ModelRegistry, activeModel?: Model, sessionId?: string, ): Promise { - // If a specific provider is preferred, try it first. - if (preferredImageProvider === "openai") { - const openAI = await findOpenAIHostedImageCredentials(modelRegistry, activeModel, sessionId); - if (openAI) return openAI; - // Fall through to auto-detect if preferred provider key not found. - } else if (preferredImageProvider === "antigravity" && modelRegistry) { - const antigravity = await findAntigravityCredentials(modelRegistry, sessionId); - if (antigravity) return antigravity; - // Fall through to auto-detect if preferred provider key not found. - } else if (preferredImageProvider === "gemini") { - const gemini = await findGeminiImageCredentials(modelRegistry, sessionId); - if (gemini) return gemini; - // Fall through to auto-detect if preferred provider key not found. - } else if (preferredImageProvider === "openrouter") { - const openRouter = await findOpenRouterImageCredentials(modelRegistry, sessionId); - if (openRouter) return openRouter; - // Fall through to auto-detect if preferred provider key not found. - } else if (preferredImageProvider === "xai") { - const xai = await findXAIImageCredentials(modelRegistry); - if (xai) return xai; - // Fall through to auto-detect if preferred provider key not found. + switch (provider) { + case "openai": + return findOpenAIHostedImageCredentials(modelRegistry, activeModel, sessionId); + case "antigravity": + return modelRegistry ? findAntigravityCredentials(modelRegistry, sessionId) : null; + case "xai": + return findXAIImageCredentials(modelRegistry); + case "openrouter": + return findOpenRouterImageCredentials(modelRegistry, sessionId); + case "gemini": + return findGeminiImageCredentials(modelRegistry, sessionId); } - - // Auto-detect: GPT hosted image generation, then Antigravity, xAI, OpenRouter, Gemini. - const openAI = await findOpenAIHostedImageCredentials(modelRegistry, activeModel, sessionId); - if (openAI) return openAI; - - if (modelRegistry) { - const antigravity = await findAntigravityCredentials(modelRegistry, sessionId); - if (antigravity) return antigravity; - } - - const xai = await findXAIImageCredentials(modelRegistry); - if (xai) return xai; - - const openRouter = await findOpenRouterImageCredentials(modelRegistry, sessionId); - if (openRouter) return openRouter; - - const gemini = await findGeminiImageCredentials(modelRegistry, sessionId); - if (gemini) return gemini; - - return null; } async function loadImageFromPath(imagePath: string, cwd: string): Promise { @@ -1040,533 +1046,563 @@ export const imageGenTool: CustomTool { const sessionId = ctx.sessionManager.getSessionId(); - const apiKey = await findImageApiKey(ctx.modelRegistry, ctx.model, sessionId); - if (!apiKey) { + const providerOrder = imageProviderOrder(ctx.model); + const cwd = ctx.sessionManager.getCwd(); + const requestSignal = ptree.combineSignals(signal, IMAGE_TIMEOUT); + const fetchImpl = ctx.fetch ?? fetch; + const failures: Array<{ provider: ImageProvider; error: ProviderHttpError }> = []; + let foundCredentials = false; + let resolvedImageCache: InlineImageData[] | undefined; + + for (const preferredProvider of providerOrder) { + const apiKey = await findImageApiKey(preferredProvider, ctx.modelRegistry, ctx.model, sessionId); + if (!apiKey) continue; + foundCredentials = true; + if (!resolvedImageCache) { + resolvedImageCache = []; + if (params.input?.length) { + for (const input of params.input) { + resolvedImageCache.push(await resolveInputImage(input, cwd)); + } + } + } + const resolvedImages = resolvedImageCache; + + const provider = apiKey.provider; + try { + const model = + provider === "openai" || provider === "openai-codex" + ? (apiKey.model?.id ?? "gpt") + : provider === "antigravity" + ? DEFAULT_ANTIGRAVITY_MODEL + : provider === "openrouter" + ? DEFAULT_OPENROUTER_MODEL + : provider === "xai" + ? DEFAULT_XAI_IMAGE_MODEL + : DEFAULT_MODEL; + const resolvedModel = provider === "openrouter" ? resolveOpenRouterModel(model) : model; + assertImageAspectRatioSupported(provider, params.aspect_ratio); + if (provider === "openai" || provider === "openai-codex") { + if (!apiKey.model) { + throw new Error("Missing active GPT model for OpenAI image generation"); + } + + const hostedModel = apiKey.model; + const hostedKey: ApiKey = ctx.modelRegistry.resolver(hostedModel, sessionId); + + const parsed = await withAuth( + hostedKey, + key => + generateOpenAIHostedImage( + key, + hostedModel, + params, + resolvedImages, + fetchImpl, + requestSignal, + sessionId, + ), + { signal: requestSignal }, + ); + + if (parsed.images.length === 0) { + const messageText = parsed.responseText ? `\n\n${parsed.responseText}` : ""; + return { + content: [{ type: "text", text: `No image data returned.${messageText}` }], + details: { + provider, + model, + imageCount: 0, + imagePaths: [], + images: [], + responseText: parsed.responseText, + revisedPrompt: parsed.revisedPrompt, + usage: parsed.usage, + }, + }; + } + + const imagePaths = await saveImagesToTemp(parsed.images); + + return { + content: [ + { type: "text", text: buildResponseSummary(provider, model, imagePaths, parsed.responseText) }, + ], + details: { + provider, + model, + imageCount: parsed.images.length, + imagePaths, + images: parsed.images, + responseText: parsed.responseText, + revisedPrompt: parsed.revisedPrompt, + usage: parsed.usage, + }, + }; + } + + if (provider === "antigravity") { + if (!apiKey.projectId) { + throw new Error("Missing projectId in antigravity credentials"); + } + + const prompt = assemblePrompt(params); + const antigravityKey: ApiKey = ctx.modelRegistry.resolver("google-antigravity", { + sessionId, + modelId: DEFAULT_ANTIGRAVITY_MODEL, + }); + + const response = await withAuth( + antigravityKey, + async key => { + // On a retry the resolver yields the raw stored credential JSON + // ({ token, projectId }); the initial seed is the already-parsed + // access token. Tolerate both, falling back to the seed projectId. + const rotated = parseAntigravityCredentials(key); + const bearer = rotated?.accessToken ?? key; + const projectId = rotated?.projectId ?? apiKey.projectId!; + const requestBody = buildAntigravityRequest( + prompt, + model, + projectId, + params.aspect_ratio, + params.image_size, + resolvedImages, + ); + + let endpoints = [DEFAULT_ANTIGRAVITY_ENDPOINT_PROD, DEFAULT_ANTIGRAVITY_ENDPOINT_SANDBOX]; + try { + const mode = settings.get("providers.antigravityEndpoint"); + if (mode === "production") { + endpoints = [DEFAULT_ANTIGRAVITY_ENDPOINT_PROD]; + } else if (mode === "sandbox") { + endpoints = [DEFAULT_ANTIGRAVITY_ENDPOINT_SANDBOX]; + } + } catch { + // Ignored + } + + let resp: Response | undefined; + let lastError: Error | undefined; + + for (let i = 0; i < endpoints.length; i++) { + const endpoint = endpoints[i]; + const isLastEndpoint = i === endpoints.length - 1; + try { + resp = await fetchImpl(`${endpoint}/v1internal:streamGenerateContent?alt=sse`, { + method: "POST", + headers: { + Authorization: `Bearer ${bearer}`, + "Content-Type": "application/json", + Accept: "text/event-stream", + "User-Agent": getAntigravityUserAgent(), + }, + body: JSON.stringify(requestBody), + signal: requestSignal, + }); + + if (resp.ok) { + break; + } + + const errorText = await resp.text(); + let message = errorText; + try { + const parsedErr = JSON.parse(errorText) as { error?: { message?: string } }; + message = parsedErr.error?.message ?? message; + } catch { + // Keep raw text. + } + + lastError = new ProviderHttpError( + `Antigravity image request failed (${resp.status}): ${message}`, + resp.status, + { headers: resp.headers }, + ); + + if (resp.status === 429 || (resp.status >= 500 && resp.status < 600)) { + if (!isLastEndpoint) { + continue; + } + } + break; + } catch (error) { + lastError = error as Error; + if (isLastEndpoint) { + break; + } + } + } + + if (!resp?.ok) { + throw lastError ?? new Error("Antigravity image generation failed"); + } + + return resp; + }, + { signal: requestSignal }, + ); + + const parsed = await parseAntigravitySseForImage(response, requestSignal); + const responseText = parsed.text.length > 0 ? parsed.text.join(" ") : undefined; + + if (parsed.images.length === 0) { + const messageText = responseText ? `\n\n${responseText}` : ""; + return { + content: [{ type: "text", text: `No image data returned.${messageText}` }], + details: { + provider, + model, + imageCount: 0, + imagePaths: [], + images: [], + responseText, + usage: parsed.usage, + }, + }; + } + + const imagePaths = await saveImagesToTemp(parsed.images); + + return { + content: [{ type: "text", text: buildResponseSummary(provider, model, imagePaths, responseText) }], + details: { + provider, + model, + imageCount: parsed.images.length, + imagePaths, + images: parsed.images, + responseText, + usage: parsed.usage, + }, + }; + } + + if (provider === "xai") { + if (!ctx.modelRegistry) { + throw new Error("Missing modelRegistry for xAI image generation"); + } + const xaiCreds = await resolveXAIHttpCredentials(ctx.modelRegistry, resolvedModel); + if (!xaiCreds) { + throw new Error( + "No xAI credentials. Run /login → xAI Grok OAuth (SuperGrok or X Premium+) or set XAI_API_KEY.", + ); + } + + const prompt = assemblePrompt(params); + const aspectRatio = params.aspect_ratio ?? "1:1"; + const xaiResolution = resolveXAIResolution(params.image_size); + + const isEdit = resolvedImages.length > 0; + if (isEdit && resolvedImages.length > XAI_MAX_EDIT_IMAGES) { + throw new Error( + `xAI image edits accept up to ${XAI_MAX_EDIT_IMAGES} reference images; got ${resolvedImages.length}.`, + ); + } + + const xaiBaseBody: XAIImageRequestBase = { + model: resolvedModel, + prompt, + aspect_ratio: aspectRatio, + resolution: xaiResolution, + n: 1, + response_format: "b64_json", + }; + const xaiBody: XAIImageRequestBody = isEdit + ? buildXAIEditPayload(xaiBaseBody, resolvedImages) + : xaiBaseBody; + const xaiEndpoint = isEdit ? "/images/edits" : "/images/generations"; + + const xaiKey: ApiKey = ctx.modelRegistry.resolver(xaiCreds.provider, { + sessionId, + baseUrl: xaiCreds.baseURL, + }); + + const xaiRawText = await withAuth( + xaiKey, + async key => { + const resp = await fetchImpl(`${xaiCreds.baseURL}${xaiEndpoint}`, { + method: "POST", + headers: { + Authorization: `Bearer ${key}`, + "Content-Type": "application/json", + "User-Agent": ohMyPiXAIUserAgent(), + }, + body: JSON.stringify(xaiBody), + signal: requestSignal, + }); + const rawText = await resp.text(); + if (!resp.ok) { + let message = rawText; + try { + const parsedErr = JSON.parse(rawText) as { error?: { message?: string } }; + message = parsedErr.error?.message ?? message; + } catch { + // Keep raw text. + } + throw new ProviderHttpError( + `xAI image request failed (${resp.status}): ${message}`, + resp.status, + { + headers: resp.headers, + }, + ); + } + return rawText; + }, + { signal: requestSignal }, + ); + + const xaiData = JSON.parse(xaiRawText) as { + data?: Array<{ b64_json?: string; url?: string }>; + }; + const xaiInlineImages: InlineImageData[] = []; + for (const entry of xaiData.data ?? []) { + if (entry.b64_json) { + const bytes = Buffer.from(entry.b64_json, "base64"); + const mimeType = parseImageMetadata(bytes)?.mimeType ?? "image/png"; + xaiInlineImages.push({ data: entry.b64_json, mimeType }); + } else if (entry.url) { + xaiInlineImages.push(await loadImageFromUrl(entry.url, fetchImpl, requestSignal)); + } + } + + if (xaiInlineImages.length === 0) { + return { + content: [{ type: "text", text: "No image data returned." }], + details: { + provider, + model: resolvedModel, + imageCount: 0, + imagePaths: [], + images: [], + }, + }; + } + + const xaiImagePaths = await saveImagesToTemp(xaiInlineImages); + + return { + content: [ + { type: "text", text: buildResponseSummary(provider, resolvedModel, xaiImagePaths, undefined) }, + ], + details: { + provider, + model: resolvedModel, + imageCount: xaiInlineImages.length, + imagePaths: xaiImagePaths, + images: xaiInlineImages, + }, + }; + } + + if (provider === "openrouter") { + const prompt = assemblePrompt(params); + const contentParts: OpenRouterContentPart[] = [{ type: "text", text: prompt }]; + for (const image of resolvedImages) { + contentParts.push({ type: "image_url", image_url: { url: toDataUrl(image) } }); + } + + const requestBody = { + model: resolvedModel, + messages: [{ role: "user" as const, content: contentParts }], + }; + + const rawText = await withAuth( + apiKey.apiKey, + async key => { + const resp = await fetchImpl("https://openrouter.ai/api/v1/chat/completions", { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${key}`, + "HTTP-Referer": "https://omp.sh/", + "X-OpenRouter-Title": "Oh-My-Pi", + "X-OpenRouter-Categories": "cli-agent", + }, + body: JSON.stringify(requestBody), + signal: requestSignal, + }); + const text = await resp.text(); + if (!resp.ok) { + let message = text; + try { + const parsed = JSON.parse(text) as { error?: { message?: string } }; + message = parsed.error?.message ?? message; + } catch { + // Keep raw text. + } + throw new ProviderHttpError( + `OpenRouter image request failed (${resp.status}): ${message}`, + resp.status, + { headers: resp.headers }, + ); + } + return text; + }, + { signal: requestSignal }, + ); + + const data = JSON.parse(rawText) as OpenRouterResponse; + const message = data.choices?.[0]?.message; + const responseText = collectOpenRouterResponseText(message); + const imageUrls = extractOpenRouterImageUrls(message); + const inlineImages: InlineImageData[] = []; + for (const imageUrl of imageUrls) { + inlineImages.push(await loadImageFromUrl(imageUrl, fetchImpl, requestSignal)); + } + + if (inlineImages.length === 0) { + const messageText = responseText ? `\n\n${responseText}` : ""; + return { + content: [{ type: "text", text: `No image data returned.${messageText}` }], + details: { + provider, + model: resolvedModel, + imageCount: 0, + imagePaths: [], + images: [], + responseText, + }, + }; + } + + const imagePaths = await saveImagesToTemp(inlineImages); + + return { + content: [ + { type: "text", text: buildResponseSummary(provider, resolvedModel, imagePaths, responseText) }, + ], + details: { + provider, + model: resolvedModel, + imageCount: inlineImages.length, + imagePaths, + images: inlineImages, + responseText, + }, + }; + } + + const parts = [] as Array<{ text?: string; inlineData?: InlineImageData }>; + for (const image of resolvedImages) { + parts.push({ inlineData: image }); + } + parts.push({ text: assemblePrompt(params) }); + + const generationConfig: { + responseModalities: GeminiResponseModality[]; + imageConfig?: { aspectRatio?: string; imageSize?: string }; + } = { + responseModalities: ["IMAGE"], + }; + + if (params.aspect_ratio || params.image_size) { + generationConfig.imageConfig = { + aspectRatio: params.aspect_ratio, + imageSize: params.image_size, + }; + } + + const requestBody = { + contents: [{ role: "user" as const, parts }], + generationConfig, + }; + + const rawText = await withAuth( + apiKey.apiKey, + async key => { + const resp = await fetchImpl( + `https://generativelanguage.googleapis.com/v1beta/models/${encodeURIComponent(model)}:generateContent`, + { + method: "POST", + headers: { + "Content-Type": "application/json", + "x-goog-api-key": key, + }, + body: JSON.stringify(requestBody), + signal: requestSignal, + }, + ); + const text = await resp.text(); + if (!resp.ok) { + let message = text; + try { + const parsed = JSON.parse(text) as { error?: { message?: string } }; + message = parsed.error?.message ?? message; + } catch { + // Keep raw text. + } + throw new ProviderHttpError( + `Gemini image request failed (${resp.status}): ${message}`, + resp.status, + { + headers: resp.headers, + }, + ); + } + return text; + }, + { signal: requestSignal }, + ); + + const data = JSON.parse(rawText) as GeminiGenerateContentResponse; + const responseParts = combineParts(data); + const responseText = collectResponseText(responseParts); + const inlineImages = collectInlineImages(responseParts); + + if (inlineImages.length === 0) { + const blocked = data.promptFeedback?.blockReason + ? `Blocked: ${data.promptFeedback.blockReason}` + : "No image data returned."; + return { + content: [{ type: "text", text: `${blocked}${responseText ? `\n\n${responseText}` : ""}` }], + details: { + provider, + model, + imageCount: 0, + imagePaths: [], + images: [], + responseText, + promptFeedback: data.promptFeedback, + usage: data.usageMetadata, + }, + }; + } + + const imagePaths = await saveImagesToTemp(inlineImages); + + return { + content: [{ type: "text", text: buildResponseSummary(provider, model, imagePaths, responseText) }], + details: { + provider, + model, + imageCount: inlineImages.length, + imagePaths, + images: inlineImages, + responseText, + promptFeedback: data.promptFeedback, + usage: data.usageMetadata, + }, + }; + } catch (error) { + if (!(error instanceof ProviderHttpError) || requestSignal?.aborted) { + throw error; + } + failures.push({ provider, error }); + } + } + + if (!foundCredentials) { throw new Error( "No image API credentials found. Use a GPT Responses/Codex model with OpenAI credentials, login with google-antigravity or xAI Grok OAuth, or set XAI_API_KEY, OPENROUTER_API_KEY, GEMINI_API_KEY, or GOOGLE_API_KEY.", ); } - const provider = apiKey.provider; - const model = - provider === "openai" || provider === "openai-codex" - ? (apiKey.model?.id ?? "gpt") - : provider === "antigravity" - ? DEFAULT_ANTIGRAVITY_MODEL - : provider === "openrouter" - ? DEFAULT_OPENROUTER_MODEL - : provider === "xai" - ? DEFAULT_XAI_IMAGE_MODEL - : DEFAULT_MODEL; - const resolvedModel = provider === "openrouter" ? resolveOpenRouterModel(model) : model; - assertImageAspectRatioSupported(provider, params.aspect_ratio); - const cwd = ctx.sessionManager.getCwd(); - - const resolvedImages: InlineImageData[] = []; - if (params.input?.length) { - for (const input of params.input) { - resolvedImages.push(await resolveInputImage(input, cwd)); - } - } - - const requestSignal = ptree.combineSignals(signal, IMAGE_TIMEOUT); - const fetchImpl = ctx.fetch ?? fetch; - - if (provider === "openai" || provider === "openai-codex") { - if (!apiKey.model) { - throw new Error("Missing active GPT model for OpenAI image generation"); - } - - const hostedModel = apiKey.model; - const hostedKey: ApiKey = ctx.modelRegistry.resolver(hostedModel, sessionId); - - const parsed = await withAuth( - hostedKey, - key => - generateOpenAIHostedImage( - key, - hostedModel, - params, - resolvedImages, - fetchImpl, - requestSignal, - sessionId, - ), - { signal: requestSignal }, - ); - - if (parsed.images.length === 0) { - const messageText = parsed.responseText ? `\n\n${parsed.responseText}` : ""; - return { - content: [{ type: "text", text: `No image data returned.${messageText}` }], - details: { - provider, - model, - imageCount: 0, - imagePaths: [], - images: [], - responseText: parsed.responseText, - revisedPrompt: parsed.revisedPrompt, - usage: parsed.usage, - }, - }; - } - - const imagePaths = await saveImagesToTemp(parsed.images); - - return { - content: [ - { type: "text", text: buildResponseSummary(provider, model, imagePaths, parsed.responseText) }, - ], - details: { - provider, - model, - imageCount: parsed.images.length, - imagePaths, - images: parsed.images, - responseText: parsed.responseText, - revisedPrompt: parsed.revisedPrompt, - usage: parsed.usage, - }, - }; - } - - if (provider === "antigravity") { - if (!apiKey.projectId) { - throw new Error("Missing projectId in antigravity credentials"); - } - - const prompt = assemblePrompt(params); - const antigravityKey: ApiKey = ctx.modelRegistry.resolver("google-antigravity", { - sessionId, - modelId: DEFAULT_ANTIGRAVITY_MODEL, - }); - - const response = await withAuth( - antigravityKey, - async key => { - // On a retry the resolver yields the raw stored credential JSON - // ({ token, projectId }); the initial seed is the already-parsed - // access token. Tolerate both, falling back to the seed projectId. - const rotated = parseAntigravityCredentials(key); - const bearer = rotated?.accessToken ?? key; - const projectId = rotated?.projectId ?? apiKey.projectId!; - const requestBody = buildAntigravityRequest( - prompt, - model, - projectId, - params.aspect_ratio, - params.image_size, - resolvedImages, - ); - - let endpoints = [DEFAULT_ANTIGRAVITY_ENDPOINT_PROD, DEFAULT_ANTIGRAVITY_ENDPOINT_SANDBOX]; - try { - const mode = settings.get("providers.antigravityEndpoint"); - if (mode === "production") { - endpoints = [DEFAULT_ANTIGRAVITY_ENDPOINT_PROD]; - } else if (mode === "sandbox") { - endpoints = [DEFAULT_ANTIGRAVITY_ENDPOINT_SANDBOX]; - } - } catch { - // Ignored - } - - let resp: Response | undefined; - let lastError: Error | undefined; - - for (let i = 0; i < endpoints.length; i++) { - const endpoint = endpoints[i]; - const isLastEndpoint = i === endpoints.length - 1; - try { - resp = await fetchImpl(`${endpoint}/v1internal:streamGenerateContent?alt=sse`, { - method: "POST", - headers: { - Authorization: `Bearer ${bearer}`, - "Content-Type": "application/json", - Accept: "text/event-stream", - "User-Agent": getAntigravityUserAgent(), - }, - body: JSON.stringify(requestBody), - signal: requestSignal, - }); - - if (resp.ok) { - break; - } - - const errorText = await resp.text(); - let message = errorText; - try { - const parsedErr = JSON.parse(errorText) as { error?: { message?: string } }; - message = parsedErr.error?.message ?? message; - } catch { - // Keep raw text. - } - - lastError = new ProviderHttpError( - `Antigravity image request failed (${resp.status}): ${message}`, - resp.status, - { headers: resp.headers }, - ); - - if (resp.status === 429 || (resp.status >= 500 && resp.status < 600)) { - if (!isLastEndpoint) { - continue; - } - } - break; - } catch (error) { - lastError = error as Error; - if (isLastEndpoint) { - break; - } - } - } - - if (!resp?.ok) { - throw lastError ?? new Error("Antigravity image generation failed"); - } - - return resp; - }, - { signal: requestSignal }, - ); - - const parsed = await parseAntigravitySseForImage(response, requestSignal); - const responseText = parsed.text.length > 0 ? parsed.text.join(" ") : undefined; - - if (parsed.images.length === 0) { - const messageText = responseText ? `\n\n${responseText}` : ""; - return { - content: [{ type: "text", text: `No image data returned.${messageText}` }], - details: { - provider, - model, - imageCount: 0, - imagePaths: [], - images: [], - responseText, - usage: parsed.usage, - }, - }; - } - - const imagePaths = await saveImagesToTemp(parsed.images); - - return { - content: [{ type: "text", text: buildResponseSummary(provider, model, imagePaths, responseText) }], - details: { - provider, - model, - imageCount: parsed.images.length, - imagePaths, - images: parsed.images, - responseText, - usage: parsed.usage, - }, - }; - } - - if (provider === "xai") { - if (!ctx.modelRegistry) { - throw new Error("Missing modelRegistry for xAI image generation"); - } - const xaiCreds = await resolveXAIHttpCredentials(ctx.modelRegistry, resolvedModel); - if (!xaiCreds) { - throw new Error( - "No xAI credentials. Run /login → xAI Grok OAuth (SuperGrok or X Premium+) or set XAI_API_KEY.", - ); - } - - const prompt = assemblePrompt(params); - const aspectRatio = params.aspect_ratio ?? "1:1"; - const xaiResolution = resolveXAIResolution(params.image_size); - - const isEdit = resolvedImages.length > 0; - if (isEdit && resolvedImages.length > XAI_MAX_EDIT_IMAGES) { - throw new Error( - `xAI image edits accept up to ${XAI_MAX_EDIT_IMAGES} reference images; got ${resolvedImages.length}.`, - ); - } - - const xaiBaseBody: XAIImageRequestBase = { - model: resolvedModel, - prompt, - aspect_ratio: aspectRatio, - resolution: xaiResolution, - n: 1, - response_format: "b64_json", - }; - const xaiBody: XAIImageRequestBody = isEdit - ? buildXAIEditPayload(xaiBaseBody, resolvedImages) - : xaiBaseBody; - const xaiEndpoint = isEdit ? "/images/edits" : "/images/generations"; - - const xaiKey: ApiKey = ctx.modelRegistry.resolver(xaiCreds.provider, { - sessionId, - baseUrl: xaiCreds.baseURL, - }); - - const xaiRawText = await withAuth( - xaiKey, - async key => { - const resp = await fetchImpl(`${xaiCreds.baseURL}${xaiEndpoint}`, { - method: "POST", - headers: { - Authorization: `Bearer ${key}`, - "Content-Type": "application/json", - "User-Agent": ohMyPiXAIUserAgent(), - }, - body: JSON.stringify(xaiBody), - signal: requestSignal, - }); - const rawText = await resp.text(); - if (!resp.ok) { - let message = rawText; - try { - const parsedErr = JSON.parse(rawText) as { error?: { message?: string } }; - message = parsedErr.error?.message ?? message; - } catch { - // Keep raw text. - } - throw new ProviderHttpError(`xAI image request failed (${resp.status}): ${message}`, resp.status, { - headers: resp.headers, - }); - } - return rawText; - }, - { signal: requestSignal }, - ); - - const xaiData = JSON.parse(xaiRawText) as { - data?: Array<{ b64_json?: string; url?: string }>; - }; - const xaiInlineImages: InlineImageData[] = []; - for (const entry of xaiData.data ?? []) { - if (entry.b64_json) { - const bytes = Buffer.from(entry.b64_json, "base64"); - const mimeType = parseImageMetadata(bytes)?.mimeType ?? "image/png"; - xaiInlineImages.push({ data: entry.b64_json, mimeType }); - } else if (entry.url) { - xaiInlineImages.push(await loadImageFromUrl(entry.url, fetchImpl, requestSignal)); - } - } - - if (xaiInlineImages.length === 0) { - return { - content: [{ type: "text", text: "No image data returned." }], - details: { - provider, - model: resolvedModel, - imageCount: 0, - imagePaths: [], - images: [], - }, - }; - } - - const xaiImagePaths = await saveImagesToTemp(xaiInlineImages); - - return { - content: [ - { type: "text", text: buildResponseSummary(provider, resolvedModel, xaiImagePaths, undefined) }, - ], - details: { - provider, - model: resolvedModel, - imageCount: xaiInlineImages.length, - imagePaths: xaiImagePaths, - images: xaiInlineImages, - }, - }; - } - - if (provider === "openrouter") { - const prompt = assemblePrompt(params); - const contentParts: OpenRouterContentPart[] = [{ type: "text", text: prompt }]; - for (const image of resolvedImages) { - contentParts.push({ type: "image_url", image_url: { url: toDataUrl(image) } }); - } - - const requestBody = { - model: resolvedModel, - messages: [{ role: "user" as const, content: contentParts }], - }; - - const rawText = await withAuth( - apiKey.apiKey, - async key => { - const resp = await fetchImpl("https://openrouter.ai/api/v1/chat/completions", { - method: "POST", - headers: { - "Content-Type": "application/json", - Authorization: `Bearer ${key}`, - "HTTP-Referer": "https://omp.sh/", - "X-OpenRouter-Title": "Oh-My-Pi", - "X-OpenRouter-Categories": "cli-agent", - }, - body: JSON.stringify(requestBody), - signal: requestSignal, - }); - const text = await resp.text(); - if (!resp.ok) { - let message = text; - try { - const parsed = JSON.parse(text) as { error?: { message?: string } }; - message = parsed.error?.message ?? message; - } catch { - // Keep raw text. - } - throw new ProviderHttpError( - `OpenRouter image request failed (${resp.status}): ${message}`, - resp.status, - { headers: resp.headers }, - ); - } - return text; - }, - { signal: requestSignal }, - ); - - const data = JSON.parse(rawText) as OpenRouterResponse; - const message = data.choices?.[0]?.message; - const responseText = collectOpenRouterResponseText(message); - const imageUrls = extractOpenRouterImageUrls(message); - const inlineImages: InlineImageData[] = []; - for (const imageUrl of imageUrls) { - inlineImages.push(await loadImageFromUrl(imageUrl, fetchImpl, requestSignal)); - } - - if (inlineImages.length === 0) { - const messageText = responseText ? `\n\n${responseText}` : ""; - return { - content: [{ type: "text", text: `No image data returned.${messageText}` }], - details: { - provider, - model: resolvedModel, - imageCount: 0, - imagePaths: [], - images: [], - responseText, - }, - }; - } - - const imagePaths = await saveImagesToTemp(inlineImages); - - return { - content: [ - { type: "text", text: buildResponseSummary(provider, resolvedModel, imagePaths, responseText) }, - ], - details: { - provider, - model: resolvedModel, - imageCount: inlineImages.length, - imagePaths, - images: inlineImages, - responseText, - }, - }; - } - - const parts = [] as Array<{ text?: string; inlineData?: InlineImageData }>; - for (const image of resolvedImages) { - parts.push({ inlineData: image }); - } - parts.push({ text: assemblePrompt(params) }); - - const generationConfig: { - responseModalities: GeminiResponseModality[]; - imageConfig?: { aspectRatio?: string; imageSize?: string }; - } = { - responseModalities: ["IMAGE"], - }; - - if (params.aspect_ratio || params.image_size) { - generationConfig.imageConfig = { - aspectRatio: params.aspect_ratio, - imageSize: params.image_size, - }; - } - - const requestBody = { - contents: [{ role: "user" as const, parts }], - generationConfig, - }; - - const rawText = await withAuth( - apiKey.apiKey, - async key => { - const resp = await fetchImpl( - `https://generativelanguage.googleapis.com/v1beta/models/${encodeURIComponent(model)}:generateContent`, - { - method: "POST", - headers: { - "Content-Type": "application/json", - "x-goog-api-key": key, - }, - body: JSON.stringify(requestBody), - signal: requestSignal, - }, - ); - const text = await resp.text(); - if (!resp.ok) { - let message = text; - try { - const parsed = JSON.parse(text) as { error?: { message?: string } }; - message = parsed.error?.message ?? message; - } catch { - // Keep raw text. - } - throw new ProviderHttpError(`Gemini image request failed (${resp.status}): ${message}`, resp.status, { - headers: resp.headers, - }); - } - return text; - }, - { signal: requestSignal }, + throw new AggregateError( + failures.map(failure => failure.error), + `Image generation failed for all credentialed providers: ${failures.map(failure => failure.provider).join(", ")}`, ); - - const data = JSON.parse(rawText) as GeminiGenerateContentResponse; - const responseParts = combineParts(data); - const responseText = collectResponseText(responseParts); - const inlineImages = collectInlineImages(responseParts); - - if (inlineImages.length === 0) { - const blocked = data.promptFeedback?.blockReason - ? `Blocked: ${data.promptFeedback.blockReason}` - : "No image data returned."; - return { - content: [{ type: "text", text: `${blocked}${responseText ? `\n\n${responseText}` : ""}` }], - details: { - provider, - model, - imageCount: 0, - imagePaths: [], - images: [], - responseText, - promptFeedback: data.promptFeedback, - usage: data.usageMetadata, - }, - }; - } - - const imagePaths = await saveImagesToTemp(inlineImages); - - return { - content: [{ type: "text", text: buildResponseSummary(provider, model, imagePaths, responseText) }], - details: { - provider, - model, - imageCount: inlineImages.length, - imagePaths, - images: inlineImages, - responseText, - promptFeedback: data.promptFeedback, - usage: data.usageMetadata, - }, - }; }); }, }; diff --git a/packages/coding-agent/test/tools/image-gen.test.ts b/packages/coding-agent/test/tools/image-gen.test.ts index 1e102b835..8a70c5ea0 100644 --- a/packages/coding-agent/test/tools/image-gen.test.ts +++ b/packages/coding-agent/test/tools/image-gen.test.ts @@ -24,6 +24,37 @@ afterEach(async () => { setPreferredImageProvider("auto"); }); +function createAntigravityXAIContext(model: Model | undefined, fetchMock: typeof fetch): CustomToolContext { + const antigravityCredentials = JSON.stringify({ token: "test-antigravity-token", projectId: "test-project" }); + return { + fetch: fetchMock, + sessionManager: { + getCwd: () => "/tmp", + getSessionId: () => "test-session", + } as unknown as ReadonlySessionManager, + modelRegistry: { + getApiKey: async () => undefined, + getApiKeyForProvider: async (provider: string) => { + if (provider === "google-antigravity") return antigravityCredentials; + if (provider === "xai-oauth") return "test-xai-token"; + return undefined; + }, + getProviderBaseUrl: () => undefined, + getAll: () => [], + authStorage: { + hasNonEnvCredential: (provider: string) => provider === "xai-oauth", + rotateSessionCredential: async () => false, + }, + resolver: (provider: string) => async () => + provider === "google-antigravity" ? antigravityCredentials : "test-xai-token", + } as unknown as ModelRegistry, + model, + isIdle: () => true, + hasQueuedMessages: () => false, + abort: () => {}, + }; +} + describe("imageGenTool", () => { it("registers without resolving image provider credentials", async () => { const modelRegistry = { @@ -333,4 +364,61 @@ describe("imageGenTool", () => { if (!savedPath) throw new Error("Expected generated image path"); expect(await Bun.file(savedPath).bytes()).toEqual(Buffer.from("fake-xai-image")); }); + + it("prefers the active xAI provider over unrelated credentialed providers", async () => { + const requestUrls: string[] = []; + const fetchMock = (async (input: string | URL | Request) => { + const url = input.toString(); + requestUrls.push(url); + if (!url.startsWith("https://api.x.ai/")) { + throw new Error(`Unexpected provider request: ${url}`); + } + return new Response( + JSON.stringify({ data: [{ b64_json: Buffer.from("active-xai-image").toString("base64") }] }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + }) as unknown as typeof fetch; + const model = { + api: "openai-completions", + provider: "xai-oauth", + id: "grok-4.5", + name: "Grok 4.5", + baseUrl: "https://api.x.ai/v1", + } as Model; + const ctx = createAntigravityXAIContext(model, fetchMock); + + const result = await imageGenTool.execute("call-active-xai", { subject: "a cat" }, undefined, ctx); + generatedImagePaths.push(...(result.details?.imagePaths ?? [])); + + expect(requestUrls).toEqual(["https://api.x.ai/v1/images/generations"]); + expect(result.details?.provider).toBe("xai"); + }); + + it("falls back to xAI after an earlier provider HTTP failure", async () => { + const requestUrls: string[] = []; + const fetchMock = (async (input: string | URL | Request) => { + const url = input.toString(); + requestUrls.push(url); + if (url.includes("streamGenerateContent")) { + return new Response(JSON.stringify({ error: { message: "image endpoint unavailable" } }), { + status: 404, + headers: { "content-type": "application/json" }, + }); + } + return new Response( + JSON.stringify({ data: [{ b64_json: Buffer.from("fallback-xai-image").toString("base64") }] }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + }) as unknown as typeof fetch; + const ctx = createAntigravityXAIContext(undefined, fetchMock); + + const result = await imageGenTool.execute("call-fallback-xai", { subject: "a cat" }, undefined, ctx); + generatedImagePaths.push(...(result.details?.imagePaths ?? [])); + + expect(requestUrls).toEqual([ + "https://daily-cloudcode-pa.googleapis.com/v1internal:streamGenerateContent?alt=sse", + "https://api.x.ai/v1/images/generations", + ]); + expect(result.details?.provider).toBe("xai"); + }); }); From dc2235fa9c0271aa3f111df2d804c9a463d8097f Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:16:12 +0000 Subject: [PATCH 020/293] fix(tui): preserved plan review scroll position Retained relative body progress across transient non-scrollable render frames and added regression coverage for terminal-height changes. Fixes #5232 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../modes/components/plan-review-overlay.ts | 59 +++++++++++++++---- .../components/plan-review-overlay.test.ts | 32 ++++++++++ 3 files changed, 85 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 30fc7c72b..84abe32ea 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed fullscreen Plan Review jumping to the top while scrolling when a transient terminal resize or Markdown reflow made the body temporarily non-scrollable. ([#5232](https://github.com/can1357/oh-my-pi/issues/5232)) + ## [16.4.5] - 2026-07-11 ### Breaking Changes diff --git a/packages/coding-agent/src/modes/components/plan-review-overlay.ts b/packages/coding-agent/src/modes/components/plan-review-overlay.ts index 11f7a96d2..7be362d6c 100644 --- a/packages/coding-agent/src/modes/components/plan-review-overlay.ts +++ b/packages/coding-agent/src/modes/components/plan-review-overlay.ts @@ -136,6 +136,8 @@ export class PlanReviewOverlay implements Component { #tocCursor = 0; #sidebarShown = false; #pendingScrollToToc = false; + /** Last meaningful relative body position, retained while a frame cannot scroll. */ + #scrollProgress = 0; // Click hit-testing, rebuilt every render. Keys are 0-based rendered-line // indices (== screen rows, since the fullscreen overlay paints from row 0). @@ -194,6 +196,7 @@ export class PlanReviewOverlay implements Component { setPlanContent(planContent: string): void { this.#setSections(planContent); this.#scrollView.scrollToTop(); + this.#scrollProgress = 0; this.#tocCursor = 0; // A wholesale external-editor swap supersedes prior in-overlay deletions. this.#deleted = []; @@ -337,6 +340,7 @@ export class PlanReviewOverlay implements Component { if (event.wheel !== null) { // Scroll wheel: three rows per notch. this.#scrollView.scroll(event.wheel * 3); + this.#captureScrollProgress(); return true; } if (event.release) return true; @@ -439,13 +443,21 @@ export class PlanReviewOverlay implements Component { // drops into the actions ("next step"); scrolling off the top steps back up // to the ToC. if (matchesSelectUp(data) || data === "k") { - if (this.#scrollView.getScrollOffset() <= 0 && this.#sidebarShown) this.#setFocus("toc"); - else this.#scrollView.scroll(-1); + if (this.#scrollView.getScrollOffset() <= 0 && this.#sidebarShown) { + this.#setFocus("toc"); + } else { + this.#scrollView.scroll(-1); + this.#captureScrollProgress(); + } return; } if (matchesSelectDown(data) || data === "j") { - if (this.#scrollView.getScrollOffset() >= this.#scrollView.getMaxScrollOffset()) this.#setFocus("actions"); - else this.#scrollView.scroll(1); + if (this.#scrollView.getScrollOffset() >= this.#scrollView.getMaxScrollOffset()) { + this.#setFocus("actions"); + } else { + this.#scrollView.scroll(1); + this.#captureScrollProgress(); + } return; } this.#handleBodyScroll(data); @@ -458,9 +470,17 @@ export class PlanReviewOverlay implements Component { * before this runs, so here it only ever sees the paging/fast keys. */ #handleBodyScroll(data: string): void { - if (this.#scrollView.handleScrollKey(data)) return; - if (data === "g") this.#scrollView.scrollToTop(); - else if (data === "G") this.#scrollView.scrollToBottom(); + if (this.#scrollView.handleScrollKey(data)) { + this.#captureScrollProgress(); + return; + } + if (data === "g") { + this.#scrollView.scrollToTop(); + this.#scrollProgress = 0; + } else if (data === "G") { + this.#scrollView.scrollToBottom(); + this.#scrollProgress = 1; + } } #handleToc(data: string): void { @@ -511,7 +531,10 @@ export class PlanReviewOverlay implements Component { const sectionIndex = this.#toc[this.#tocCursor]; if (sectionIndex === undefined) return; const offset = this.#sectionOffsets[sectionIndex]; - if (offset !== undefined) this.#scrollView.setScrollOffset(offset); + if (offset !== undefined) { + this.#scrollView.setScrollOffset(offset); + this.#captureScrollProgress(); + } } /** Greatest ToC position whose section starts at or above the scroll offset. */ @@ -683,6 +706,23 @@ export class PlanReviewOverlay implements Component { return parts.join(sep); } + /** + * Retain relative progress across reflow frames. A non-scrollable intermediate + * frame has no meaningful offset, so it must not erase the last scroll position. + */ + #captureScrollProgress(): void { + const maxOffset = this.#scrollView.getMaxScrollOffset(); + if (maxOffset > 0) this.#scrollProgress = this.#scrollView.getScrollOffset() / maxOffset; + } + + #layoutBody(lines: readonly string[], height: number): void { + this.#captureScrollProgress(); + this.#scrollView.setLines(lines); + this.#scrollView.setHeight(height); + const maxOffset = this.#scrollView.getMaxScrollOffset(); + if (maxOffset > 0) this.#scrollView.setScrollOffset(Math.round(this.#scrollProgress * maxOffset)); + } + /** Build the concatenated body lines and record each section's start row. */ #buildBody(bodyContentWidth: number): string[] { const lines: string[] = []; @@ -800,8 +840,7 @@ export class PlanReviewOverlay implements Component { const regionRows = Math.max(MIN_BODY_ROWS, termHeight - chrome); const bodyLines = this.#buildBody(bodyContentWidth); - this.#scrollView.setLines(bodyLines); - this.#scrollView.setHeight(regionRows); + this.#layoutBody(bodyLines, regionRows); if (this.#pendingScrollToToc) { this.#pendingScrollToToc = false; this.#scrubBodyToToc(); diff --git a/packages/coding-agent/test/modes/components/plan-review-overlay.test.ts b/packages/coding-agent/test/modes/components/plan-review-overlay.test.ts index ef9841fd4..9d19e9d2f 100644 --- a/packages/coding-agent/test/modes/components/plan-review-overlay.test.ts +++ b/packages/coding-agent/test/modes/components/plan-review-overlay.test.ts @@ -180,6 +180,38 @@ describe("PlanReviewOverlay", () => { expect(backToTop).not.toContain("para 199"); }); + it("preserves scroll progress across a transient non-scrollable render", () => { + const originalRows = Object.getOwnPropertyDescriptor(process.stdout, "rows"); + const setRows = (rows: number): void => { + Object.defineProperty(process.stdout, "rows", { configurable: true, value: rows }); + }; + const codeRows = Array.from({ length: 400 }, (_, i) => `L${String(i).padStart(3, "0")}`).join("\n"); + const overlay = new PlanReviewOverlay( + `# Plan\n\n\`\`\`\n${codeRows}\n\`\`\`\n`, + { promptTitle: "next", options: APPROVAL_OPTIONS }, + { onPick: vi.fn(), onCancel: vi.fn() }, + ); + + try { + setRows(40); + render(overlay); + overlay.handleInput("G"); + const bottom = render(overlay); + expect(bottom).toContain("L399"); + expect(bottom).not.toContain("L000"); + + setRows(1000); + render(overlay); + setRows(40); + const restored = render(overlay); + expect(restored).toContain("L399"); + expect(restored).not.toContain("L000"); + } finally { + if (originalRows) Object.defineProperty(process.stdout, "rows", originalRows); + else Reflect.deleteProperty(process.stdout, "rows"); + } + }); + it("swaps the displayed plan and resets scroll on setPlanContent", () => { const longPlan = Array.from({ length: 200 }, (_, i) => `para ${i}`).join("\n\n"); const overlay = new PlanReviewOverlay( From 8dfbe8e09d3cc84dcff2bdd70c7580e4322872e4 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:21:19 +0000 Subject: [PATCH 021/293] fix(model-resolver): strip thinking suffix before fuzzy model match parseModelPatternWithContext ran matchModel on the whole pattern (including a trailing :level thinking suffix) and only stripped the suffix if that first pass missed. matchModel's provider-scoped fuzzy match normalizes colons away and does subsequence matching, so kimi-for-coding:high matched the longer sibling kimi-for-coding-highspeed before the suffix was recognized as a thinking level, silently switching model and billing tier. Match the full pattern exactly first (new exactOnly mode skips the fuzzy/substring fallbacks), then strip a valid :level suffix and recurse before any fuzzy match; fuzzy-match the whole pattern only as a last resort. Literal ids ending in :max still win via the exact pass. Fixes #5151 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../coding-agent/src/config/model-resolver.ts | 35 ++++++++++-- .../coding-agent/test/model-resolver.test.ts | 56 +++++++++++++++++++ 3 files changed, 90 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d8041c95c..153d3d19c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed a role with a `:high` thinking suffix resolving to a longer sibling model whose id embeds the tier name (e.g. `kimi-for-coding:high` → `kimi-for-coding-highspeed`). The thinking suffix is now stripped before any fuzzy match, so `provider/model:high` keeps the exact model at high effort ([#5151](https://github.com/can1357/oh-my-pi/issues/5151)). + ## [16.4.2] - 2026-07-10 ### Fixed diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index 32ed6116f..2ab5f9153 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -617,11 +617,18 @@ function findExactModelReferenceMatch(modelReference: string, availableModels: M * 4. provider-scoped fuzzy match, * 5. substring match with the alias-vs-dated pick. * Returns the matched model or undefined if no match found. + * + * `exactOnly` stops after the exact phases (1-3), skipping the fuzzy/substring + * fallbacks (4-5). Callers use it to resolve the full selector exactly before + * a trailing `:` thinking suffix is split off, so the suffix can never + * be fuzzily absorbed into a longer sibling id (e.g. `kimi-for-coding:high` + * must not match `kimi-for-coding-highspeed`). */ function matchModel( modelPattern: string, availableModels: Model[], context: ModelPreferenceContext, + options?: { exactOnly?: boolean }, ): Model | undefined { const exactRefMatch = findExactModelReferenceMatch(modelPattern, availableModels); if (exactRefMatch) { @@ -657,6 +664,14 @@ function matchModel( return pickPreferredModel(preferred.length > 0 ? preferred : aliasMatches, context); } } + + // Exact phases exhausted. Fuzzy/substring fallbacks (below) subsequence-match + // the whole pattern and would let a trailing `:` thinking suffix bleed + // into a longer sibling id; callers that still hold an unstripped suffix ask + // for exact-only so the suffix is split off before any fuzzy attempt. + if (options?.exactOnly) { + return undefined; + } // Check for provider/modelId format — fuzzy match within provider only. const slashIndex = modelPattern.indexOf("/"); if (slashIndex !== -1) { @@ -760,15 +775,18 @@ function parseModelPatternWithContext( context: ModelPreferenceContext, options?: { allowInvalidThinkingSelectorFallback?: boolean }, ): ParsedModelResult { - // Try exact match first - const exactMatch = matchModel(pattern, availableModels, context); + // Exact match on the full pattern first (no fuzzy): a literal id that + // contains a colon (`coding-router:max`) wins over any suffix split. + const exactMatch = matchModel(pattern, availableModels, context, { exactOnly: true }); if (exactMatch) { return { model: exactMatch, thinkingLevel: undefined, warning: undefined, explicitThinkingLevel: false }; } - // No match - try stripping a valid thinking suffix and recursing. - // `max` is accepted only after the full pattern failed, so literal model IDs - // ending in `:max` keep winning over the thinking suffix. + // Strip a valid thinking suffix and recurse BEFORE any fuzzy match, so a + // `:` suffix can never be subsequence-absorbed into a longer sibling + // id (e.g. `kimi-for-coding:high` must not match `kimi-for-coding-highspeed`). + // `max` is accepted only after the exact match above failed, so literal model + // IDs ending in `:max` keep winning over the thinking suffix. const { base, level } = splitThinkingSuffix(pattern, -1, MAX_THINKING_SUFFIX_OPTIONS); if (level) { const result = parseModelPatternWithContext(base, availableModels, context, options); @@ -785,6 +803,13 @@ function parseModelPatternWithContext( return result; } + // No valid thinking suffix: fall back to fuzzy/substring matching on the + // whole pattern. + const fallbackMatch = matchModel(pattern, availableModels, context); + if (fallbackMatch) { + return { model: fallbackMatch, thinkingLevel: undefined, warning: undefined, explicitThinkingLevel: false }; + } + const lastColonIndex = pattern.lastIndexOf(":"); if (lastColonIndex === -1) { // No colons, pattern simply doesn't match any model diff --git a/packages/coding-agent/test/model-resolver.test.ts b/packages/coding-agent/test/model-resolver.test.ts index 4e2e5d4fc..deaaeca60 100644 --- a/packages/coding-agent/test/model-resolver.test.ts +++ b/packages/coding-agent/test/model-resolver.test.ts @@ -142,6 +142,38 @@ const mockMaxSuffixModels: Model[] = [ }), ]; +// Sibling models where one id is a prefix of the other AND the longer id embeds +// a thinking-tier token (`-highspeed` contains `high`). Regression fixture for +// the fuzzy match swallowing a `:high` thinking suffix into the longer id. +const mockThinkingSuffixSiblingModels: Model<"openai-completions">[] = [ + buildModel({ + id: "kimi-for-coding", + name: "K2.7 Code", + api: "openai-completions", + provider: "kimi-code", + baseUrl: "https://api.kimi.com/coding/v1", + reasoning: true, + thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 262144, + maxTokens: 32000, + }), + buildModel({ + id: "kimi-for-coding-highspeed", + name: "K2.7 Code Highspeed", + api: "openai-completions", + provider: "kimi-code", + baseUrl: "https://api.kimi.com/coding/v1", + reasoning: true, + thinking: { mode: "effort", efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High] }, + input: ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 262144, + maxTokens: 32000, + }), +]; + const mockAutoSuffixModels: Model[] = [ buildModel({ id: "runtime:auto", @@ -468,6 +500,30 @@ describe("parseModelPattern", () => { expect(result.explicitThinkingLevel).toBe(false); expect(result.warning).toBeUndefined(); }); + + test("thinking suffix is stripped before fuzzy match, never absorbed into a longer sibling id", () => { + // `kimi-for-coding:high` must resolve to the standard model at high effort, + // not fuzzy-match `kimi-for-coding-highspeed` (issue #5151). + const result = parseModelPattern("kimi-code/kimi-for-coding:high", mockThinkingSuffixSiblingModels); + expect(result.model?.id).toBe("kimi-for-coding"); + expect(result.thinkingLevel).toBe(Effort.High); + expect(result.explicitThinkingLevel).toBe(true); + expect(result.warning).toBeUndefined(); + }); + + test("bare id thinking suffix is stripped before fuzzy match against a longer sibling", () => { + const result = parseModelPattern("kimi-for-coding:high", mockThinkingSuffixSiblingModels); + expect(result.model?.id).toBe("kimi-for-coding"); + expect(result.thinkingLevel).toBe(Effort.High); + expect(result.explicitThinkingLevel).toBe(true); + }); + + test("the longer sibling still resolves exactly with its own thinking suffix", () => { + const result = parseModelPattern("kimi-code/kimi-for-coding-highspeed:high", mockThinkingSuffixSiblingModels); + expect(result.model?.id).toBe("kimi-for-coding-highspeed"); + expect(result.thinkingLevel).toBe(Effort.High); + expect(result.explicitThinkingLevel).toBe(true); + }); }); describe("patterns with invalid thinking levels", () => { From 9978404c630f3bce0fde4c97f158cce044a34bd3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:34:16 +0000 Subject: [PATCH 022/293] test(coding-agent): guarded compiled header fallback regression Restored the regression test for browser-headers.ts lazy header-generator init. The guard (lazy getHeaderGenerator + static Chrome fallback when data_files are absent) landed in #5178 but its test was deleted as redundant, leaving the fix undefended. Without the guard a compiled single-file binary resolves header-generator data_files to the build-machine node_modules path, which is absent at runtime, so the module throws ENOENT at import time. This poisons the Bing web_search provider import (undefined is not a constructor) and the plugin extension loader (extension validation / omp plugin install). The restored subprocess probe hides header-generator/data_files and asserts the module imports cleanly and returns the fallback profile; verified it fails against the pre-guard source. Fixes #5256 --- packages/coding-agent/CHANGELOG.md | 4 + .../tools/web-search-browser-headers.test.ts | 86 +++++++++++++++++++ 2 files changed, 90 insertions(+) create mode 100644 packages/coding-agent/test/tools/web-search-browser-headers.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 87cd5f7e2..276fd6d33 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Restored the regression test guarding compiled-binary web-search header generation: `browser-headers.ts` lazily constructs `header-generator` and falls back to a static Chrome profile when its `data_files` are absent, but the test defending that contract had been removed. Without the guard, a compiled binary threw `ENOENT` at import time, breaking the Bing `web_search` provider (`undefined is not a constructor`) and extension loading / `omp plugin install`. ([#5256](https://github.com/can1357/oh-my-pi/issues/5256)) + ## [16.4.6] - 2026-07-12 ### Added diff --git a/packages/coding-agent/test/tools/web-search-browser-headers.test.ts b/packages/coding-agent/test/tools/web-search-browser-headers.test.ts new file mode 100644 index 000000000..ca9c428f4 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-browser-headers.test.ts @@ -0,0 +1,86 @@ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { fileURLToPath } from "node:url"; +import { buildBrowserNavigationHeaders } from "@oh-my-pi/pi-coding-agent/web/search/providers/browser-headers"; + +// The header-generator dependency reads its `data_files/*.json` via +// `readFileSync(`${__dirname}/data_files/...`)`. In a compiled single-file binary +// those assets resolve to the build-machine node_modules path, which is absent at +// runtime — the module used to construct HeaderGenerator eagerly and threw ENOENT +// at import time, poisoning the Bing provider import ("undefined is not a +// constructor") and the plugin extension loader (issue #5256). These tests guard +// the lazy-init + fallback contract so that regression cannot silently return. + +const CHROME_FALLBACK_HEADERS: Record = { + Accept: + "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7", + "Accept-Encoding": "gzip, deflate, br, zstd", + "Accept-Language": "en-US,en;q=0.9", + "Cache-Control": "max-age=0", + Priority: "u=0, i", + "Sec-Ch-Ua": '"Google Chrome";v="149", "Chromium";v="149", ";Not A Brand";v="99"', + "Sec-Ch-Ua-Mobile": "?0", + "Sec-Ch-Ua-Platform": '"macOS"', + "Sec-Fetch-Dest": "document", + "Sec-Fetch-Mode": "navigate", + "Sec-Fetch-Site": "none", + "Sec-Fetch-User": "?1", + "Upgrade-Insecure-Requests": "1", + "User-Agent": + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", +}; + +const packageRoot = path.join(import.meta.dir, "../.."); +const headerGeneratorRoot = path.dirname(fileURLToPath(import.meta.resolve("header-generator"))); + +describe("browser navigation headers", () => { + it("returns the stable Mac Chrome profile when randomization is disabled", () => { + const headers = buildBrowserNavigationHeaders({ randomized: false }); + + expect(headers["User-Agent"]).toContain("Chrome/149.0.0.0"); + expect(headers["User-Agent"]).toContain("Macintosh; Intel Mac OS X 10_15_7"); + expect(headers["Sec-Ch-Ua"]).toContain('v="149"'); + expect(headers["Sec-Ch-Ua-Platform"]).toBe('"macOS"'); + }); + + it("imports cleanly and falls back when header-generator data files are absent", async () => { + // Simulate the compiled-binary condition: the fs-loaded data_files that + // header-generator resolves at `${__dirname}/data_files` are missing at + // runtime. A fresh subprocess ensures we exercise module import, not a + // cached singleton from this test process. + const dataFilesDir = path.join(headerGeneratorRoot, "data_files"); + const unavailableDataFilesDir = path.join( + headerGeneratorRoot, + `.data_files-unavailable-${process.pid}-${Date.now()}`, + ); + + await fs.rename(dataFilesDir, unavailableDataFilesDir); + try { + const script = [ + 'import { buildBrowserNavigationHeaders } from "@oh-my-pi/pi-coding-agent/web/search/providers/browser-headers";', + "const headers = buildBrowserNavigationHeaders();", + "process.stdout.write(JSON.stringify(headers));", + ].join("\n"); + const proc = Bun.spawn([process.execPath, "--no-install", "--eval", script], { + cwd: packageRoot, + stdout: "pipe", + stderr: "pipe", + }); + + const [exitCode, stdout, stderr] = await Promise.all([ + proc.exited, + new Response(proc.stdout).text(), + new Response(proc.stderr).text(), + ]); + + if (exitCode !== 0) { + throw new Error(`browser header import failed with exit ${exitCode}:\n${stderr}`); + } + + expect(JSON.parse(stdout)).toEqual(CHROME_FALLBACK_HEADERS); + } finally { + await fs.rename(unavailableDataFilesDir, dataFilesDir); + } + }); +}); From 7b399b32a20e9407a2ef947159551a6eb08827ee Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:36:41 +0000 Subject: [PATCH 023/293] fix(browser): bounded tab teardown waits - Applied close deadlines to cmux surfaces, orphan targets, and browser handles. - Surfaced the backend, tab name, and pending cleanup resource on timeout. - Forced stuck headless browser processes down after Browser.close timed out. Fixes #5259 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/tools/browser.ts | 7 +- .../src/tools/browser/registry.ts | 34 +++++++-- .../src/tools/browser/tab-supervisor.ts | 75 ++++++++++++++++--- .../test/tools/browser-lifecycle-leak.test.ts | 35 ++++++++- 5 files changed, 136 insertions(+), 19 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 87cd5f7e2..98089ec10 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed browser tabs hanging indefinitely at `Closing ` when a worker, CDP target, browser process, or cmux surface stalls during teardown; close deadlines now release the operation with backend, tab, and pending-resource diagnostics. ([#5259](https://github.com/can1357/oh-my-pi/issues/5259)) + ## [16.4.6] - 2026-07-12 ### Added diff --git a/packages/coding-agent/src/tools/browser.ts b/packages/coding-agent/src/tools/browser.ts index 0f7d9e093..32ef2186d 100644 --- a/packages/coding-agent/src/tools/browser.ts +++ b/packages/coding-agent/src/tools/browser.ts @@ -199,7 +199,7 @@ export class BrowserTool implements AgentTool> { const kill = !!params.kill; if (params.all) { - const count = await untilAborted(signal, () => releaseAllTabs({ kill })); + const count = await untilAborted(signal, () => releaseAllTabs({ kill, timeoutMs })); details.result = `Closed ${count} tab(s)`; return toolResult(details).text(details.result).done(); } - const closed = await untilAborted(signal, () => releaseTab(name, { kill })); + const closed = await untilAborted(signal, () => releaseTab(name, { kill, timeoutMs })); details.result = closed ? `Closed tab ${JSON.stringify(name)}` : `No tab named ${JSON.stringify(name)}`; return toolResult(details).text(details.result).done(); } diff --git a/packages/coding-agent/src/tools/browser/registry.ts b/packages/coding-agent/src/tools/browser/registry.ts index 7a6e95ad8..7cf87b962 100644 --- a/packages/coding-agent/src/tools/browser/registry.ts +++ b/packages/coding-agent/src/tools/browser/registry.ts @@ -1,5 +1,5 @@ import * as path from "node:path"; -import { logger } from "@oh-my-pi/pi-utils"; +import { logger, withTimeout } from "@oh-my-pi/pi-utils"; import type { Subprocess } from "bun"; import type { Browser, CDPSession } from "puppeteer-core"; import { ToolAbortError, ToolError } from "../tool-errors"; @@ -40,6 +40,15 @@ export interface CmuxBrowserHandle extends BrowserHandleCommon { export type BrowserHandle = PuppeteerBrowserHandle | CmuxBrowserHandle; +/** Controls bounded browser-handle teardown and identifies the owning resource in timeout diagnostics. */ +export interface ReleaseBrowserOptions { + kill: boolean; + timeoutMs?: number; + resource?: string; +} + +const DEFAULT_BROWSER_CLOSE_TIMEOUT_MS = 5_000; + const browsers = new Map(); function browserKey(kind: BrowserKind): string { @@ -214,7 +223,7 @@ export function holdBrowser(handle: BrowserHandle): void { handle.refCount++; } -export async function releaseBrowser(handle: BrowserHandle, opts: { kill: boolean }): Promise { +export async function releaseBrowser(handle: BrowserHandle, opts: ReleaseBrowserOptions): Promise { handle.refCount = Math.max(0, handle.refCount - 1); if (handle.refCount === 0) { // Only evict if the registry still points at THIS handle. After a disconnect, @@ -225,17 +234,32 @@ export async function releaseBrowser(handle: BrowserHandle, opts: { kill: boolea } } -async function disposeBrowserHandle(handle: BrowserHandle, opts: { kill: boolean }): Promise { +async function disposeBrowserHandle(handle: BrowserHandle, opts: ReleaseBrowserOptions): Promise { if ("client" in handle) { handle.client.close(); return; } if (handle.kind.kind === "headless") { if (handle.browser.connected) { + const timeoutMs = opts.timeoutMs ?? DEFAULT_BROWSER_CLOSE_TIMEOUT_MS; + const resource = opts.resource ?? handle.key; + const timeoutMessage = `Timed out after ${timeoutMs}ms closing headless browser for ${resource}; pending resource: Puppeteer Browser.close()`; try { - await handle.browser.close(); + await withTimeout(handle.browser.close(), timeoutMs, timeoutMessage); } catch (err) { - logger.debug("Failed to close headless browser", { error: (err as Error).message }); + if (err instanceof Error && err.message === timeoutMessage) { + const process = handle.browser.process(); + try { + handle.browser.disconnect(); + } catch {} + try { + process?.kill(); + } catch {} + throw new ToolError(timeoutMessage); + } + logger.debug("Failed to close headless browser", { + error: err instanceof Error ? err.message : String(err), + }); } } return; diff --git a/packages/coding-agent/src/tools/browser/tab-supervisor.ts b/packages/coding-agent/src/tools/browser/tab-supervisor.ts index 921fb23ff..e1ed94066 100644 --- a/packages/coding-agent/src/tools/browser/tab-supervisor.ts +++ b/packages/coding-agent/src/tools/browser/tab-supervisor.ts @@ -1,4 +1,4 @@ -import { getPuppeteerDir, logger, postmortem, Snowflake, workerHostEntry } from "@oh-my-pi/pi-utils"; +import { getPuppeteerDir, logger, postmortem, Snowflake, withTimeout, workerHostEntry } from "@oh-my-pi/pi-utils"; import type { Page, Target } from "puppeteer-core"; import { callSessionTool } from "../../eval/js/tool-bridge"; import { webpExclusionForModel } from "../../utils/image-loading"; @@ -123,6 +123,8 @@ export interface RunInTabOptions { export interface ReleaseTabOptions { kill?: boolean; + /** Maximum time for each asynchronous cleanup resource before close fails with diagnostics. */ + timeoutMs?: number; } const tabs = new Map(); @@ -131,6 +133,22 @@ const tabs = new Map(); // awaits) cannot interleave and leak a worker + browser refCount. const acquireChains = new Map>(); const GRACE_MS = 750; +const DEFAULT_TAB_CLOSE_TIMEOUT_MS = 5_000; + +async function waitForTabCleanup( + tab: TabSession, + timeoutMs: number, + pendingResource: string, + promise: Promise, +): Promise { + const message = `Timed out after ${timeoutMs}ms closing ${tab.kindTag} browser tab ${JSON.stringify(tab.name)}; pending resource: ${pendingResource}`; + try { + return await withTimeout(promise, timeoutMs, message); + } catch (error) { + if (error instanceof Error && error.message === message) throw new ToolError(message); + throw error; + } +} export function getTab(name: string): TabSession | undefined { return tabs.get(name); @@ -507,26 +525,42 @@ export async function releaseTab(name: string, opts: ReleaseTabOptions = {}): Pr pending.reject(closeError); } tab.pending.clear(); + const timeoutMs = opts.timeoutMs ?? DEFAULT_TAB_CLOSE_TIMEOUT_MS; if (tab.backend === "cmux") { - let nonLastCloseError: unknown; + let closeError: unknown; if (wasAlive && tab.cmuxOwnsSurface) { try { - await tab.browser.client.request("surface.close", { surface_id: tab.targetId }); + await waitForTabCleanup( + tab, + timeoutMs, + `cmux surface ${JSON.stringify(tab.targetId)} (surface.close)`, + tab.browser.client.request("surface.close", { surface_id: tab.targetId }, { timeoutMs }), + ); } catch (err) { if (isLastSurfaceCloseError(err)) { logger.debug("Leaving cmux browser surface open because it is the last surface in the workspace", { error: err instanceof Error ? err.message : String(err), }); } else { - nonLastCloseError = err; + closeError = err; } } } - await releaseBrowser(tab.browser, { kill: opts.kill ?? false }); - tabs.delete(name); - if (nonLastCloseError) throw nonLastCloseError; + try { + await releaseBrowser(tab.browser, { + kill: opts.kill ?? false, + timeoutMs, + resource: `tab ${JSON.stringify(name)}`, + }); + } catch (error) { + closeError ??= error; + } finally { + tabs.delete(name); + } + if (closeError) throw closeError; return true; } + let cleanupError: unknown; let forced = false; if (wasAlive) { try { @@ -537,9 +571,30 @@ export async function releaseTab(name: string, opts: ReleaseTabOptions = {}): Pr } } await tab.worker.terminate().catch(() => undefined); - if (forced && tab.kindTag === "headless") await closeOrphanTarget(tab); - await releaseBrowser(tab.browser, { kill: opts.kill ?? false }); - tabs.delete(name); + if (forced && tab.kindTag === "headless") { + try { + await waitForTabCleanup( + tab, + timeoutMs, + `orphan CDP target ${JSON.stringify(tab.targetId)} (Page.close)`, + closeOrphanTarget(tab), + ); + } catch (error) { + cleanupError = error; + } + } + try { + await releaseBrowser(tab.browser, { + kill: opts.kill ?? false, + timeoutMs, + resource: `tab ${JSON.stringify(name)}`, + }); + } catch (error) { + cleanupError ??= error; + } finally { + tabs.delete(name); + } + if (cleanupError) throw cleanupError; return true; } diff --git a/packages/coding-agent/test/tools/browser-lifecycle-leak.test.ts b/packages/coding-agent/test/tools/browser-lifecycle-leak.test.ts index be333c19a..f9e469b67 100644 --- a/packages/coding-agent/test/tools/browser-lifecycle-leak.test.ts +++ b/packages/coding-agent/test/tools/browser-lifecycle-leak.test.ts @@ -16,7 +16,7 @@ * spied so no real cmux socket / puppeteer process is needed. */ -import { afterEach, describe, expect, it, spyOn } from "bun:test"; +import { afterEach, describe, expect, it, spyOn, vi } from "bun:test"; import type { CmuxKind } from "@oh-my-pi/pi-coding-agent/tools/browser/cmux/rpc"; import { CmuxSocketClient } from "@oh-my-pi/pi-coding-agent/tools/browser/cmux/socket-client"; import { acquireBrowser, getBrowsersMapForTest } from "@oh-my-pi/pi-coding-agent/tools/browser/registry"; @@ -173,3 +173,36 @@ describe("browser lifecycle — session-scoped teardown reaps owned tabs", () => expect(getTabsMapForTest().has("reuse-tab")).toBe(false); }); }); + +describe("browser lifecycle — close deadlines", () => { + afterEach(async () => { + vi.useRealTimers(); + vi.restoreAllMocks(); + await drainAllTabs(); + }); + + it("rejects a stuck close with the backend, tab, and pending resource", async () => { + vi.useFakeTimers(); + spyOn(CmuxSocketClient.prototype, "connect").mockResolvedValue(undefined); + spyOn(CmuxSocketClient.prototype, "close").mockImplementation(() => undefined); + const stuck = Promise.withResolvers>(); + spyOn(CmuxSocketClient.prototype, "request").mockImplementation(async method => { + if (method === "browser.open_split") return { surface_id: "probe-surface", url: "about:blank" }; + if (method === "surface.close") return await stuck.promise; + return {}; + }); + + const kind: CmuxKind = { kind: "cmux", socketPath: "/tmp/omp-close-deadline.sock" }; + const browser = await acquireBrowser(kind, { cwd: "/tmp" }); + await acquireTab("probe", browser, { timeoutMs: 1_000 }); + + const close = releaseTab("probe", { timeoutMs: 100 }); + vi.advanceTimersByTime(100); + + await expect(close).rejects.toThrow( + 'Timed out after 100ms closing cmux browser tab "probe"; pending resource: cmux surface "probe-surface" (surface.close)', + ); + expect(getTabsMapForTest().has("probe")).toBe(false); + expect(getBrowsersMapForTest().size).toBe(0); + }); +}); From 8c6b2fb450adc4222e53d1707ebd52bc8904c8d6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:51:33 +0000 Subject: [PATCH 024/293] fix(coding-agent): resolved agent:// slash form for nested subagent output The agent:// path segment was always treated as a jq JSON-extraction key against .md, so agent://Parent/Child loaded Parent.md and applied .Child instead of resolving the nested child capsule Parent.Child.md. A precise-planner reading its own scout child therefore got Not found. The slash is now a hierarchy separator first: agent://Parent/Child resolves Parent.Child.md (subagent children are allocated as dot-qualified ids). It falls back to JSON extraction only when no nested output matches the path; the ?q= query form is always extraction. Fixes #5238 --- docs/tools/task.md | 2 +- packages/coding-agent/CHANGELOG.md | 4 + .../__tests__/agent-protocol-nested.test.ts | 71 +++++++++++ .../src/internal-urls/agent-protocol.ts | 112 ++++++++++++------ .../src/prompts/system/system-prompt.md | 2 +- 5 files changed, 150 insertions(+), 41 deletions(-) diff --git a/docs/tools/task.md b/docs/tools/task.md index 4bdf27a99..1266d721f 100644 --- a/docs/tools/task.md +++ b/docs/tools/task.md @@ -68,7 +68,7 @@ Settled response (`async.enabled=false`, no job manager, every item's agent `blo Artifacts and side channels: - Every subagent with an artifacts dir writes `.md`; `agent://` resolves to that file. -- If the output file is JSON, `agent:///` and `agent://?q=` perform JSON extraction. +- A subagent's own children are dot-qualified (`.`); `agent:///` reads that nested output. When the path names no nested output and the file is JSON, `agent:///` and `agent://?q=` perform JSON extraction. - Each subagent gets `.jsonl` session history when the parent persists artifacts; `history://` renders it as a concise transcript (works for live and parked agents). - Isolated patch mode writes `.patch` before merge. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 30fc7c72b..ab90b6bc5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the `agent:///` slash form failing to resolve a nested subagent's output: the path segment was always treated as a jq JSON-extraction key against `.md`, so a precise-planner reading its own scout child (`agent://Plan/Scout`) got `Not found`. The slash is now a hierarchy separator first (`agent://Parent/Child` → `Parent.Child.md`), falling back to JSON extraction only when no nested output matches the path. ([#5238](https://github.com/can1357/oh-my-pi/issues/5238)) + ## [16.4.5] - 2026-07-11 ### Breaking Changes diff --git a/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts b/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts index 7829557f5..7108a84e8 100644 --- a/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts +++ b/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts @@ -66,3 +66,74 @@ it("agent:// resolves a depth-2 subagent's .md output while its session is live const resource = await new AgentProtocolHandler().resolve(new URL(`agent://${grandchildId}`) as never); expect(resource.content).toBe("full report content"); }); + +it("agent:// slash form resolves a nested subagent child (hierarchy separator)", async () => { + const root = tempDir.path(); + const rootSessionFile = path.join(root, "slash-session.jsonl"); + const rootArtifactsDir = rootSessionFile.slice(0, -6); + await fs.mkdir(rootArtifactsDir, { recursive: true }); + const sharedArtifactManager = new ArtifactManager(rootArtifactsDir); + + // Parent subagent adopts the root ArtifactManager; its own children are + // written one level deeper under its sessionFile-derived dir, dot-qualified. + const parentSessionFile = path.join(rootArtifactsDir, "Parent.jsonl"); + const parentOwnDir = parentSessionFile.slice(0, -6); + await fs.mkdir(parentOwnDir, { recursive: true }); + await fs.writeFile(path.join(parentOwnDir, "Parent.Child.md"), "child capsule"); + + const fakeSession = { + sessionManager: { getArtifactsDir: () => sharedArtifactManager.dir }, + } as unknown as AgentSession; + const registry = AgentRegistry.global(); + registry.register({ + id: "Main", + displayName: "main", + kind: "main", + session: fakeSession, + sessionFile: rootSessionFile, + }); + registry.register({ + id: "Parent", + displayName: "sub", + kind: "sub", + parentId: "Main", + session: fakeSession, + sessionFile: parentSessionFile, + }); + + const handler = new AgentProtocolHandler(); + // Slash form is a hierarchy hop, not a jq extraction. + const slash = await handler.resolve(new URL("agent://Parent/Child") as never); + expect(slash.content).toBe("child capsule"); + expect(slash.contentType).toBe("text/markdown"); + // The canonical dotted id resolves to the same output. + const dotted = await handler.resolve(new URL("agent://Parent.Child") as never); + expect(dotted.content).toBe("child capsule"); +}); + +it("agent:// path form falls back to JSON extraction when no nested output matches", async () => { + const root = tempDir.path(); + const rootSessionFile = path.join(root, "json-session.jsonl"); + const rootArtifactsDir = rootSessionFile.slice(0, -6); + await fs.mkdir(rootArtifactsDir, { recursive: true }); + const sharedArtifactManager = new ArtifactManager(rootArtifactsDir); + await fs.writeFile(path.join(rootArtifactsDir, "Worker.md"), JSON.stringify({ result: { ok: true } })); + + const fakeSession = { + sessionManager: { getArtifactsDir: () => sharedArtifactManager.dir }, + } as unknown as AgentSession; + const registry = AgentRegistry.global(); + registry.register({ + id: "Main", + displayName: "main", + kind: "main", + session: fakeSession, + sessionFile: rootSessionFile, + }); + + const handler = new AgentProtocolHandler(); + // `result` names no nested output, so the path extracts JSON from Worker.md. + const extracted = await handler.resolve(new URL("agent://Worker/result") as never); + expect(extracted.contentType).toBe("application/json"); + expect(JSON.parse(extracted.content)).toEqual({ ok: true }); +}); diff --git a/packages/coding-agent/src/internal-urls/agent-protocol.ts b/packages/coding-agent/src/internal-urls/agent-protocol.ts index 00add4d66..43ebf9956 100644 --- a/packages/coding-agent/src/internal-urls/agent-protocol.ts +++ b/packages/coding-agent/src/internal-urls/agent-protocol.ts @@ -8,7 +8,11 @@ * * URL forms: * - agent:// - Full output content - * - agent:/// - JSON extraction via path form + * - agent:/// - Nested subagent output (hierarchy separator; the + * registry allocates a subagent's own children as dot-qualified ids, so + * `agent://Parent/Child` resolves `Parent.Child.md`) + * - agent:/// - JSON extraction via path form (fallback when no + * nested output matches the path) * - agent://?q= - JSON extraction via query form */ import * as fs from "node:fs/promises"; @@ -44,65 +48,56 @@ export class AgentProtocolHandler implements ProtocolHandler { } const dirs = artifactsDirsFromRegistry(); - if (dirs.length === 0) { throw new Error("No session - agent outputs unavailable"); } - let foundPath: string | undefined; - let anyDirExists = false; - const availableIds = new Set(); - - for (const dir of dirs) { + // A subagent allocates its own children as dot-qualified ids + // (`Parent.Child`), so the slash path form is first tried as a hierarchy + // separator: `agent://Parent/Child` resolves `Parent.Child.md`. Only when + // no such nested output exists does the path fall back to jq-style JSON + // extraction on `.md`. Query form (`?q=`) is always extraction. + const pathSegments = hasPathExtraction ? urlPath.split("/").filter(Boolean) : []; + const decodedSegments = pathSegments.map(segment => { try { - await fs.stat(dir); - anyDirExists = true; - } catch (err) { - if (isEnoent(err)) continue; - throw err; + return decodeURIComponent(segment); + } catch { + return segment; } - const candidate = path.join(dir, `${outputId}.md`); - try { - await fs.stat(candidate); - foundPath = candidate; - break; - } catch (err) { - if (!isEnoent(err)) throw err; - try { - const files = await fs.readdir(dir); - for (const f of files) { - if (f.endsWith(".md")) availableIds.add(f.replace(/\.md$/, "")); - } - } catch { - // Listing failures are non-fatal; continue searching. - } - } - } + }); + const nestedId = + decodedSegments.length > 0 && decodedSegments.every(segment => !segment.includes(".")) + ? [outputId, ...decodedSegments].join(".") + : undefined; - if (!anyDirExists) { + const scan = await this.#findOutput(dirs, nestedId ? [nestedId, outputId] : [outputId]); + if (!scan.anyDirExists) { throw new Error("No artifacts directory found"); } - - if (!foundPath) { - const availableStr = availableIds.size > 0 ? [...availableIds].join(", ") : "none"; - throw new Error(`Not found: ${outputId}\nAvailable: ${availableStr}`); + if (!scan.foundPath) { + const target = nestedId ?? outputId; + const availableStr = scan.availableIds.size > 0 ? [...scan.availableIds].join(", ") : "none"; + throw new Error(`Not found: ${target}\nAvailable: ${availableStr}`); } - const rawContent = await Bun.file(foundPath).text(); + const rawContent = await Bun.file(scan.foundPath).text(); const notes: string[] = []; let content = rawContent; let contentType: InternalResource["contentType"] = "text/markdown"; - if (hasPathExtraction || hasQueryExtraction) { + // Extraction applies only when the URL did NOT resolve to a nested output + // (a slash that named a real child is a hierarchy hop, not a jq path). + const extract = hasQueryExtraction || (hasPathExtraction && scan.matchedId !== nestedId); + if (extract) { let jsonValue: unknown; try { jsonValue = JSON.parse(rawContent); } catch (err) { const message = err instanceof Error ? err.message : String(err); - throw new Error(`Output ${outputId} is not valid JSON: ${message}`); + throw new Error(`Output ${scan.matchedId} is not valid JSON: ${message}`); } - const query = hasPathExtraction ? pathToQuery(urlPath) : queryParam!; + const query = hasQueryExtraction ? queryParam! : pathToQuery(urlPath); if (query) { const extracted = applyQuery(jsonValue, query); try { @@ -122,11 +117,50 @@ export class AgentProtocolHandler implements ProtocolHandler { content, contentType, size: Buffer.byteLength(content, "utf-8"), - sourcePath: foundPath, + sourcePath: scan.foundPath, notes, }; } + /** + * Scan every registered artifacts dir for the first `.md` among + * `candidateIds` (tried in order, so a hierarchy match wins over the base + * id). Returns the resolved path and the id it matched, plus the set of + * available ids gathered from the scanned dirs for the not-found message. + */ + async #findOutput( + dirs: string[], + candidateIds: string[], + ): Promise<{ foundPath?: string; matchedId?: string; anyDirExists: boolean; availableIds: Set }> { + // Build a full id→path map across every registered dir before picking, so + // candidate priority is global: a nested id in a deeper dir must win over + // the base id even when the base id's dir is scanned first. + const byId = new Map(); + let anyDirExists = false; + for (const dir of dirs) { + let files: string[]; + try { + files = await fs.readdir(dir); + } catch (err) { + if (isEnoent(err)) continue; + throw err; + } + anyDirExists = true; + for (const f of files) { + if (!f.endsWith(".md")) continue; + const id = f.slice(0, -3); + if (!byId.has(id)) byId.set(id, path.join(dir, f)); + } + } + for (const id of candidateIds) { + const foundPath = byId.get(id); + if (foundPath) { + return { foundPath, matchedId: id, anyDirExists, availableIds: new Set(byId.keys()) }; + } + } + return { anyDirExists, availableIds: new Set(byId.keys()) }; + } + async complete(): Promise { const ids = new Set(); for (const dir of artifactsDirsFromRegistry()) { diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 1e45733ed..1ee689faa 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -56,7 +56,7 @@ Special URLs for internal resources; with most FS/bash tools they auto-resolve t {{#if hasMemoryRoot}} - `memory://root`: project memory summary {{/if}} -- `agent://`: agent output artifact; `/` extracts a JSON field +- `agent://`: agent output artifact; `/` reads a nested subagent's output, else `/` extracts a JSON field - `artifact://`: artifact content - `local://.md`: plan artifacts or shared content for subagents {{#if hasObsidian}} From cc9977cce43d7e33735dc99608002a8815b83294 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 17:58:19 +0000 Subject: [PATCH 025/293] fix(advisor): anchored context maintenance on provider usage - Anchored advisor compaction on provider-reported context usage (cached input + generated output) floored by a full local estimate including the advisor system prompt and tool schemas, so a near-full cached context is no longer undercounted by the per-message estimate. - Rejected stale provider usage retained across advisor compaction via a runtime-only usage-anchor boundary recorded on the summary message. - Recovered provider overflow by clearing only the advisor's own context at the current primary cursor, retrying the bounded failing batch once against a fresh context without replaying old primary history, and keeping later updates eligible. - Threaded the selected dashboard range through the stats Recent Errors UI, API, and database timestamp filter before ordering and the 50-row limit. Fixes #5282 --- bun.lock | 1 + packages/coding-agent/CHANGELOG.md | 1 + .../src/advisor/__tests__/advisor.test.ts | 307 +++++++++++++++++- packages/coding-agent/src/advisor/runtime.ts | 224 ++++++++----- .../coding-agent/src/session/agent-session.ts | 93 ++++-- .../test/advisor-context-maintenance.test.ts | 199 ++++++++++++ packages/stats/CHANGELOG.md | 4 + packages/stats/package.json | 1 + packages/stats/src/aggregator.ts | 5 +- packages/stats/src/client/api.ts | 10 +- .../stats/src/client/routes/ErrorsRoute.tsx | 4 +- packages/stats/src/db.ts | 11 +- packages/stats/src/server.ts | 4 +- packages/stats/test/errors-range.test.ts | 79 +++++ .../stats/test/errors-route-range.test.tsx | 81 +++++ packages/stats/tsconfig.client.json | 3 +- 16 files changed, 906 insertions(+), 121 deletions(-) create mode 100644 packages/coding-agent/test/advisor-context-maintenance.test.ts create mode 100644 packages/stats/test/errors-range.test.ts create mode 100644 packages/stats/test/errors-route-range.test.tsx diff --git a/bun.lock b/bun.lock index 185d6908c..d0634b4b1 100644 --- a/bun.lock +++ b/bun.lock @@ -217,6 +217,7 @@ "@types/bun": "catalog:", "@types/react": "catalog:", "@types/react-dom": "catalog:", + "linkedom": "catalog:", "postcss": "catalog:", }, }, diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7de11cc4d..0af06bb6f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,6 +4,7 @@ ### Fixed +- Fixed advisor context maintenance undercounting the provider context: the compaction decision now anchors on the advisor's provider-reported context usage (cached input + generated output) floored by a full local estimate that includes the advisor system prompt and tool schemas, rejects stale provider usage retained across advisor compaction, and recovers a provider overflow by clearing only the advisor's own context at the current primary cursor — retrying the bounded failing batch once against a fresh context without replaying old primary history and keeping later updates eligible ([#5282](https://github.com/can1357/oh-my-pi/issues/5282)) - Improved search reliability for Perplexity provider by forcing retrieval for all queries - Fixed JS eval cells losing top-level `function` and `var` declarations across cells when the defining cell contained top-level `await` — the async wrapper scoped them to the cell's IIFE instead of publishing them to the worker global diff --git a/packages/coding-agent/src/advisor/__tests__/advisor.test.ts b/packages/coding-agent/src/advisor/__tests__/advisor.test.ts index 179756d80..82a9085c4 100644 --- a/packages/coding-agent/src/advisor/__tests__/advisor.test.ts +++ b/packages/coding-agent/src/advisor/__tests__/advisor.test.ts @@ -1,5 +1,7 @@ import { describe, expect, it, vi } from "bun:test"; import type { AgentMessage, AgentTelemetryConfig } from "@oh-my-pi/pi-agent-core"; +import type { AssistantMessage } from "@oh-my-pi/pi-ai"; +import * as AIError from "@oh-my-pi/pi-ai/error"; import type { TUI } from "@oh-my-pi/pi-tui"; import { type } from "arktype"; import type { ModelRegistry } from "../../config/model-registry"; @@ -978,7 +980,7 @@ describe("advisor", () => { expect(promptInputs[1]).toContain("summary-bbb"); }); - it("triggers a re-prime and full replay when maintainContext returns true", async () => { + it("clears advisor context without replaying primary history when maintenance requests recovery", async () => { const promptInputs: string[] = []; let resetCount = 0; const agent: AdvisorAgent = { @@ -992,37 +994,326 @@ describe("advisor", () => { state: { messages: [] }, }; const messages: AgentMessage[] = [{ role: "user", content: "aaa", timestamp: 1 } as AgentMessage]; - let shouldRePrime = false; + let shouldResetContext = false; const host: AdvisorRuntimeHost = { snapshotMessages: () => messages, enqueueAdvice: () => {}, maintainContext: async tokens => { expect(tokens).toBeGreaterThan(0); - return shouldRePrime; + return shouldResetContext; }, }; const runtime = new AdvisorRuntime(agent, host); - // First turn: normal incremental prompt runtime.onTurnEnd(messages); await Promise.resolve(); expect(promptInputs).toHaveLength(1); expect(promptInputs[0]).toContain("aaa"); expect(resetCount).toBe(0); - // Second turn: maintainContext resolves true, triggering a re-prime - shouldRePrime = true; + shouldResetContext = true; messages.push({ role: "user", content: "bbb", timestamp: 2 } as AgentMessage); runtime.onTurnEnd(messages); await Promise.resolve(); await Promise.resolve(); - // The reset cleared history and prompted a full replay (so the batch contains both aaa and bbb) expect(promptInputs).toHaveLength(2); - expect(promptInputs[1]).toContain("aaa"); expect(promptInputs[1]).toContain("bbb"); + expect(promptInputs[1]).not.toContain("aaa"); expect(resetCount).toBe(1); }); + + it("preserves updates queued while async maintenance resets the advisor context", async () => { + const promptInputs: string[] = []; + let resetCount = 0; + const agent: AdvisorAgent = { + prompt: async input => { + promptInputs.push(input); + }, + abort: () => {}, + reset: () => { + resetCount++; + }, + state: { messages: [] }, + }; + const maintenanceStarted = Promise.withResolvers(); + const maintenanceFinished = Promise.withResolvers(); + let maintenanceCalls = 0; + const messages: AgentMessage[] = [{ role: "user", content: "bbb", timestamp: 1 } as AgentMessage]; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + maintainContext: async () => { + maintenanceCalls++; + if (maintenanceCalls !== 1) return false; + maintenanceStarted.resolve(); + return await maintenanceFinished.promise; + }, + }; + const runtime = new AdvisorRuntime(agent, host); + + runtime.onTurnEnd(messages); + await maintenanceStarted.promise; + messages.push({ role: "user", content: "ccc", timestamp: 2 } as AgentMessage); + runtime.onTurnEnd(messages); + maintenanceFinished.resolve(true); + await runtime.waitForCatchup(1000, 1); + + expect(promptInputs).toHaveLength(2); + expect(promptInputs[0]).toContain("bbb"); + expect(promptInputs[0]).not.toContain("ccc"); + expect(promptInputs[1]).toContain("ccc"); + expect(promptInputs[1]).not.toContain("bbb"); + expect(resetCount).toBe(1); + }); + + it("re-expands active primary context when maintenance clears advisor history", async () => { + const promptInputs: string[] = []; + const agent = makeAgent(promptInputs); + const planRule = + "Plan mode is active. You MUST remain read-only except for the approved plan file at local://PLAN.md."; + const messages: AgentMessage[] = [ + { role: "user", content: "aaa", timestamp: 1 } as AgentMessage, + { + role: "custom", + customType: "plan-mode-context", + content: planRule, + display: false, + timestamp: 2, + } as AgentMessage, + ]; + let shouldResetContext = false; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + maintainContext: async () => shouldResetContext, + }; + const runtime = new AdvisorRuntime(agent, host); + + runtime.onTurnEnd(messages); + await runtime.waitForCatchup(1000, 1); + expect(promptInputs[0]).toContain(planRule); + + shouldResetContext = true; + messages.push({ role: "user", content: "bbb", timestamp: 3 } as AgentMessage); + messages.push({ + role: "custom", + customType: "plan-mode-context", + content: planRule, + display: false, + timestamp: 4, + } as AgentMessage); + runtime.onTurnEnd(messages); + await runtime.waitForCatchup(1000, 1); + + expect(promptInputs).toHaveLength(2); + expect(promptInputs[1]).toContain("bbb"); + expect(promptInputs[1]).not.toContain("aaa"); + expect(promptInputs[1]).toContain(planRule); + expect(promptInputs[1]).not.toContain("unchanged — still in effect"); + }); + + it("recovers a provider overflow at the current cursor without replaying primary history", async () => { + const overflowMessage = "context_length_exceeded: Your input exceeds the context window of this model."; + const promptInputs: string[] = []; + const state: { messages: AgentMessage[]; error?: string } = { + messages: [{ role: "user", content: "existing advisor context", timestamp: 1 } as AgentMessage], + }; + let promptCalls = 0; + let resetCount = 0; + const agent: AdvisorAgent = { + prompt: async input => { + promptInputs.push(input); + promptCalls++; + state.error = promptCalls === 1 ? overflowMessage : undefined; + }, + abort: () => {}, + reset: () => { + resetCount++; + state.messages.length = 0; + state.error = undefined; + }, + state, + }; + const messages: AgentMessage[] = [ + { role: "user", content: "ancient-primary-one", timestamp: 1 } as AgentMessage, + { + role: "assistant", + content: [{ type: "text", text: "ancient-primary-two" }], + timestamp: 2, + } as AgentMessage, + ]; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + }; + const runtime = new AdvisorRuntime(agent, host, 0); + runtime.seedTo(messages.length); + + messages.push({ role: "user", content: "overflowing-current-update", timestamp: 3 } as AgentMessage); + runtime.onTurnEnd(messages); + await runtime.waitForCatchup(1000, 1); + + expect(promptInputs).toHaveLength(2); + for (const input of promptInputs) { + expect(input).toContain("overflowing-current-update"); + expect(input).not.toContain("ancient-primary-one"); + expect(input).not.toContain("ancient-primary-two"); + } + expect(resetCount).toBe(1); + + messages.push({ role: "user", content: "post-recovery-update", timestamp: 4 } as AgentMessage); + runtime.onTurnEnd(messages); + await runtime.waitForCatchup(1000, 1); + + expect(promptInputs).toHaveLength(3); + expect(promptInputs[2]).toContain("post-recovery-update"); + expect(promptInputs[2]).not.toContain("overflowing-current-update"); + expect(promptInputs[2]).not.toContain("ancient-primary-one"); + expect(promptInputs[2]).not.toContain("ancient-primary-two"); + expect(resetCount).toBe(1); + }); + + it("classifies structured overflow metadata before rolling back the failed turn", async () => { + const promptInputs: string[] = []; + const state: { messages: AgentMessage[]; error?: string } = { + messages: [{ role: "user", content: "existing advisor context", timestamp: 1 } as AgentMessage], + }; + let promptCalls = 0; + let resetCount = 0; + const agent: AdvisorAgent = { + prompt: async input => { + promptInputs.push(input); + promptCalls++; + if (promptCalls !== 1) { + state.error = undefined; + return; + } + state.messages.push({ role: "user", content: input, timestamp: 2 } as AgentMessage); + const failure: AssistantMessage = { + role: "assistant", + content: [], + api: "openai-responses", + provider: "openai", + model: "structured-overflow-model", + usage: { + input: 1, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 1, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "error", + errorMessage: "opaque provider rejection", + errorStatus: 400, + errorId: AIError.create(AIError.Flag.ContextOverflow), + timestamp: 3, + }; + state.messages.push(failure); + state.error = "opaque provider rejection"; + }, + abort: () => {}, + reset: () => { + resetCount++; + state.messages.length = 0; + state.error = undefined; + }, + rollbackTo: count => { + state.messages.length = Math.min(count, state.messages.length); + state.error = undefined; + }, + state, + }; + const messages: AgentMessage[] = [{ role: "user", content: "ancient-primary", timestamp: 1 } as AgentMessage]; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + }; + const runtime = new AdvisorRuntime(agent, host, 0); + runtime.seedTo(messages.length); + + messages.push({ role: "user", content: "structured-current-update", timestamp: 2 } as AgentMessage); + runtime.onTurnEnd(messages); + await runtime.waitForCatchup(1000, 1); + + expect(promptInputs).toHaveLength(2); + for (const input of promptInputs) { + expect(input).toContain("structured-current-update"); + expect(input).not.toContain("ancient-primary"); + } + expect(resetCount).toBe(1); + }); + + it("drops only a double-overflowing batch and continues queued and later updates", async () => { + const overflowMessage = "context_length_exceeded: Your input exceeds the context window of this model."; + const promptInputs: string[] = []; + const failures: unknown[] = []; + const secondAttemptStarted = Promise.withResolvers(); + const finishSecondAttempt = Promise.withResolvers(); + const state: { messages: AgentMessage[]; error?: string } = { + messages: [{ role: "user", content: "existing advisor context", timestamp: 1 } as AgentMessage], + }; + let failingAttempts = 0; + const agent: AdvisorAgent = { + prompt: async input => { + promptInputs.push(input); + if (!input.includes("first-overflow")) { + state.error = undefined; + return; + } + failingAttempts++; + if (failingAttempts === 2) { + secondAttemptStarted.resolve(); + await finishSecondAttempt.promise; + } + state.error = overflowMessage; + }, + abort: () => {}, + reset: () => { + state.messages.length = 0; + state.error = undefined; + }, + state, + }; + const messages: AgentMessage[] = [{ role: "user", content: "ancient-history", timestamp: 1 } as AgentMessage]; + const host: AdvisorRuntimeHost = { + snapshotMessages: () => messages, + enqueueAdvice: () => {}, + notifyFailure: error => failures.push(error), + }; + const runtime = new AdvisorRuntime(agent, host, 0); + runtime.seedTo(messages.length); + + messages.push({ role: "user", content: "first-overflow", timestamp: 2 } as AgentMessage); + runtime.onTurnEnd(messages); + await secondAttemptStarted.promise; + + messages.push({ role: "user", content: "queued-small-update", timestamp: 3 } as AgentMessage); + runtime.onTurnEnd(messages); + finishSecondAttempt.resolve(); + await runtime.waitForCatchup(1000, 1); + + expect(failingAttempts).toBe(2); + expect(promptInputs).toHaveLength(3); + for (const input of promptInputs.slice(0, 2)) { + expect(input).toContain("first-overflow"); + expect(input).not.toContain("ancient-history"); + } + expect(promptInputs[2]).toContain("queued-small-update"); + expect(promptInputs[2]).not.toContain("first-overflow"); + expect(promptInputs[2]).not.toContain("ancient-history"); + expect(failures).toHaveLength(1); + expect(runtime.backlog).toBe(0); + + messages.push({ role: "user", content: "later-small-update", timestamp: 4 } as AgentMessage); + runtime.onTurnEnd(messages); + await runtime.waitForCatchup(1000, 1); + + expect(promptInputs).toHaveLength(4); + expect(promptInputs[3]).toContain("later-small-update"); + expect(promptInputs[3]).not.toContain("first-overflow"); + }); it("tracks backlog and blocks until caught up", async () => { const promptInputs: string[] = []; const { promise: promptStarted, resolve: startPrompt } = Promise.withResolvers(); diff --git a/packages/coding-agent/src/advisor/runtime.ts b/packages/coding-agent/src/advisor/runtime.ts index 8c88f86ae..3937a7e6a 100644 --- a/packages/coding-agent/src/advisor/runtime.ts +++ b/packages/coding-agent/src/advisor/runtime.ts @@ -1,6 +1,7 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import { estimateTokens } from "@oh-my-pi/pi-agent-core/compaction"; import type { AssistantMessage, ImageContent, TextContent } from "@oh-my-pi/pi-ai"; +import * as AIError from "@oh-my-pi/pi-ai/error"; import { logger } from "@oh-my-pi/pi-utils"; import { obfuscateToolArguments, type SecretObfuscator } from "../secrets/obfuscator"; import { formatSessionHistoryMarkdown, PRIMARY_CONTEXT_CUSTOM_TYPES } from "../session/session-history-format"; @@ -35,10 +36,10 @@ export interface AdvisorRuntimeHost { * Pre-prompt context maintenance for the advisor's own append-only context. * Promotes the advisor model to a larger sibling when its context nears the * window (mirroring the primary's promote-first policy) and resolves `true` - * when the advisor should re-prime — reset and replay the current - * primary-bounded transcript — because promotion did not free enough room. - * Optional: hosts that omit it get no maintenance (context only shrinks when - * the primary's next compaction triggers {@link AdvisorRuntime.reset}). + * when the advisor must clear its own context before sending the current + * incremental update. The cursor stays at the current primary position: this + * recovery path must never replay the full primary transcript. + * Optional: hosts that omit it get no proactive maintenance. */ maintainContext?(incomingTokens: number): Promise; /** @@ -65,7 +66,10 @@ export interface AdvisorRuntimeHost { interface PendingDelta { text: string; + rawMessages: AgentMessage[]; + renderRevision: number; turns: number; + overflowRecovery?: boolean; } interface CatchupWaiter { @@ -83,6 +87,8 @@ export class AdvisorRuntime { * marker so the advisor isn't re-fed the full ~1k-token rules each turn. * Cleared on every re-prime/seed and when a failed batch is dropped. */ #seenContext = new Map(); + /** Incremented whenever the advisor loses context so queued raw deltas are re-rendered against fresh dedupe state. */ + #renderRevision = 0; #pending: PendingDelta[] = []; #busy = false; #backlog = 0; @@ -111,9 +117,9 @@ export class AdvisorRuntime { if (this.disposed) return; const all = messages ?? this.host.snapshotMessages(); this.#latestMessages = all; - const render = this.#renderDelta(all); - if (render) { - this.#pending.push({ text: render, turns: 1 }); + const rendered = this.#renderDelta(all); + if (rendered) { + this.#pending.push({ ...rendered, turns: 1 }); this.#backlog++; this.#notifyWaiters(); void this.#drain(); @@ -153,18 +159,15 @@ export class AdvisorRuntime { } catch {} } - #resetAdvisorContext(clearBacklog: boolean, wakeWaiters: boolean): void { - this.#lastCount = 0; - this.#pending = []; + #clearSeenContext(): void { + this.#seenContext.clear(); + this.#renderRevision++; + } + + #clearAdvisorContextAtCurrentCursor(): void { this.#consecutiveFailures = 0; this.#failureNotified = false; - this.#seenContext.clear(); - if (clearBacklog) { - this.#backlog = 0; - } - if (wakeWaiters) { - this.#wakeAllWaiters(); - } + this.#clearSeenContext(); try { this.agent.reset(); } catch {} @@ -173,6 +176,18 @@ export class AdvisorRuntime { } catch {} } + #resetAdvisorContext(clearBacklog: boolean, wakeWaiters: boolean): void { + this.#lastCount = 0; + this.#pending = []; + this.#clearAdvisorContextAtCurrentCursor(); + if (clearBacklog) { + this.#backlog = 0; + } + if (wakeWaiters) { + this.#wakeAllWaiters(); + } + } + /** * Re-prime the advisor after a history rewrite (compaction, session * switch/resume, branch). Clears the advisor's own (non-persisted) context @@ -196,22 +211,14 @@ export class AdvisorRuntime { this.#backlog = 0; this.#consecutiveFailures = 0; this.#failureNotified = false; - this.#seenContext.clear(); + this.#clearSeenContext(); this.#wakeAllWaiters(); } - #renderDelta(messages?: AgentMessage[]): string | null { - const all = messages ?? this.#latestMessages ?? this.host.snapshotMessages(); - if (all.length < this.#lastCount) { - this.#lastCount = all.length; - this.#seenContext.clear(); - return null; - } - const delta = all - .slice(this.#lastCount) - .filter(m => !(m.role === "custom" && (m as { customType?: string }).customType === "advisor")) - .map(m => this.#dedupContextMessage(m)); - this.#lastCount = all.length; + #formatRawDelta(rawMessages: AgentMessage[]): string | null { + const delta = rawMessages + .filter(message => !(message.role === "custom" && message.customType === "advisor")) + .map(message => this.#dedupContextMessage(message)); if (delta.length === 0) return null; const obfuscator = this.host.obfuscator; const formattedDelta = obfuscator?.hasSecrets() ? obfuscateAdvisorDelta(obfuscator, delta) : delta; @@ -226,6 +233,19 @@ export class AdvisorRuntime { return `### Session update\n\n${md}`; } + #renderDelta(messages?: AgentMessage[]): Omit | null { + const all = messages ?? this.#latestMessages ?? this.host.snapshotMessages(); + if (all.length < this.#lastCount) { + this.#lastCount = all.length; + this.#clearSeenContext(); + return null; + } + const rawMessages = all.slice(this.#lastCount); + this.#lastCount = all.length; + const text = this.#formatRawDelta(rawMessages); + return text ? { text, rawMessages, renderRevision: this.#renderRevision } : null; + } + /** * Collapse a re-injected primary-context prompt (plan/goal mode rules, the * approved plan) to a short marker when its body is byte-identical to the @@ -236,12 +256,12 @@ export class AdvisorRuntime { */ #dedupContextMessage(msg: AgentMessage): AgentMessage { if (msg.role !== "custom") return msg; - const type = (msg as { customType?: string }).customType; - if (!type || !PRIMARY_CONTEXT_CUSTOM_TYPES.has(type)) return msg; - const content = (msg as { content?: unknown }).content; + const type = msg.customType; + if (!PRIMARY_CONTEXT_CUSTOM_TYPES.has(type)) return msg; + const content = msg.content; if (typeof content !== "string") return msg; if (this.#seenContext.get(type) === content) { - return { ...(msg as object), content: "(unchanged — still in effect)" } as AgentMessage; + return { ...msg, content: "(unchanged — still in effect)" }; } this.#seenContext.set(type, content); return msg; @@ -284,27 +304,61 @@ export class AdvisorRuntime { } } + #terminalAssistantFailure(snapshot: number): AssistantMessage | undefined { + const messages = this.agent.state.messages; + for (let i = messages.length - 1; i >= snapshot; i--) { + const message = messages[i]; + if (message.role === "assistant" && message.stopReason === "error") return message; + } + return undefined; + } + + #notifyFailureOnce(error: unknown): void { + if (this.#failureNotified) return; + this.#failureNotified = true; + try { + this.host.notifyFailure?.(error); + } catch (notifyErr) { + logger.warn("advisor failure notification failed", { err: String(notifyErr) }); + } + } + async #drain(): Promise { if (this.#busy) return; this.#busy = true; try { while (!this.disposed && this.#pending.length) { - const popped = this.#pending.splice(0); + let popped: PendingDelta[]; + if (this.#pending[0]?.overflowRecovery) { + const recovery = this.#pending.shift(); + if (!recovery) continue; + popped = [recovery]; + } else { + popped = this.#pending.splice(0); + } const epoch = this.#epoch; + for (const delta of popped) { + if (delta.renderRevision === this.#renderRevision) continue; + const refreshed = this.#formatRawDelta(delta.rawMessages); + if (refreshed) delta.text = refreshed; + delta.renderRevision = this.#renderRevision; + } + const rawMessages = popped.flatMap(delta => delta.rawMessages); // Each delta already opens with a `### Session update` heading, so // join with a blank line rather than a `---` rule. - const candidateBatch = popped.map(b => b.text).join("\n\n"); - const turnsCovered = popped.reduce((sum, b) => sum + b.turns, 0); + let batch = popped.map(delta => delta.text).join("\n\n"); + const finalTurns = popped.reduce((sum, delta) => sum + delta.turns, 0); + const recoveringOverflow = popped.some(delta => delta.overflowRecovery === true); const incomingTokens = estimateTokens({ role: "user", - content: candidateBatch, + content: batch, timestamp: Date.now(), }); - let shouldReprime = false; + let shouldResetContext = false; if (this.host.maintainContext) { try { - shouldReprime = await this.host.maintainContext(incomingTokens); + shouldResetContext = await this.host.maintainContext(incomingTokens); } catch (err) { logger.debug("advisor context maintenance failed", { err: String(err) }); } @@ -312,20 +366,16 @@ export class AdvisorRuntime { // A reset/dispose during context maintenance invalidates this batch. if (this.#epoch !== epoch) continue; - let batch: string | null; - let finalTurns: number; - if (shouldReprime) { - // Promotion could not fit the advisor's context — re-prime. - const newTurns = this.#pending.reduce((sum, b) => sum + b.turns, 0); - this.#resetAdvisorContext(false, false); - batch = this.#renderDelta(this.#latestMessages); - finalTurns = turnsCovered + newTurns; - } else { - batch = candidateBatch; - finalTurns = turnsCovered; + if (shouldResetContext) { + // Reset only the advisor Agent/log. The primary cursor, queued deltas, + // backlog, waiters, latest snapshot, and epoch stay untouched. Re-render + // only this already-popped raw batch so active plan/reference bodies are + // restored without replaying any older primary transcript. + this.#clearAdvisorContextAtCurrentCursor(); + batch = this.#formatRawDelta(rawMessages) ?? batch; } - if (this.disposed || batch === null) { + if (this.disposed) { this.#backlog = Math.max(0, this.#backlog - finalTurns); this.#notifyWaiters(); continue; @@ -338,6 +388,7 @@ export class AdvisorRuntime { // failed batch on top of the stale turns and the dropped-after-3 path // would leak orphan failures into the next successful run's context. const messageSnapshot = this.agent.state.messages.length; + const contextWasFresh = shouldResetContext || recoveringOverflow || messageSnapshot === 0; try { // Reset the host's per-update advisor state (one-advise-per-update // gate) before each model cycle, so the new batch starts with a @@ -356,11 +407,14 @@ export class AdvisorRuntime { this.#consecutiveFailures = 0; this.#failureNotified = false; } catch (err) { - // reset()/dispose() aborts the in-flight prompt; the rejection is the - // reset itself, not a transient advisor failure. Drop the stale batch - // (reset already cleared #pending and rewound the cursor) instead of - // requeuing it into the post-reset conversation. + // An external reset/dispose invalidates the in-flight bounded batch; + // never requeue it into the post-reset conversation. if (this.#epoch !== epoch) continue; + const terminalFailure = this.#terminalAssistantFailure(messageSnapshot); + const contextOverflow = + (terminalFailure !== undefined && + AIError.is(AIError.classifyMessage(terminalFailure), AIError.Flag.ContextOverflow)) || + AIError.is(AIError.classify(err), AIError.Flag.ContextOverflow); this.#rollbackFailedTurn(messageSnapshot); logger.debug("advisor turn failed", { err: String(err) }); try { @@ -371,26 +425,48 @@ export class AdvisorRuntime { // The hook awaits; a reset during it invalidates this batch like the // prompt await above — drop it instead of requeueing stale content. if (this.#epoch !== epoch) continue; - this.#consecutiveFailures++; - if (this.#consecutiveFailures >= 3) { - logger.warn("advisor failed consecutively 3 times; dropping backlog to prevent stall"); - if (!this.#failureNotified) { - this.#failureNotified = true; - try { - this.host.notifyFailure?.(err); - } catch (notifyErr) { - logger.warn("advisor failure notification failed", { err: String(notifyErr) }); - } + if (contextOverflow) { + this.#clearAdvisorContextAtCurrentCursor(); + if (contextWasFresh) { + // The bounded update cannot fit even with no advisor history. Drop + // only this batch after its one fresh-context retry; pending and later + // deltas remain eligible so one oversized update cannot disable the advisor. + logger.warn("advisor update overflowed a fresh context; dropping bounded batch"); + this.#notifyFailureOnce(err); + success = true; + } else { + // Retry once against the fresh advisor context, using only the same + // bounded raw batch. Pending updates remain queued behind it. + const recoveryBatch = this.#formatRawDelta(rawMessages) ?? batch; + this.#pending.unshift({ + text: recoveryBatch, + rawMessages, + renderRevision: this.#renderRevision, + turns: finalTurns, + overflowRecovery: true, + }); + logger.debug("advisor context overflow recovered at current primary cursor"); } - this.#consecutiveFailures = 0; - // The dropped batch may carry primary-context we never delivered; drop - // the seen-state too so the next turn re-expands it instead of marking - // it "unchanged" against content the advisor never received. - this.#seenContext.clear(); - success = true; } else { - this.#pending.unshift({ text: batch, turns: finalTurns }); - await Bun.sleep(this.retryDelayMs); + this.#consecutiveFailures++; + if (this.#consecutiveFailures >= 3) { + logger.warn("advisor failed consecutively 3 times; dropping backlog to prevent stall"); + this.#notifyFailureOnce(err); + this.#consecutiveFailures = 0; + // The dropped batch may carry primary-context we never delivered; drop + // the seen-state too so queued raw deltas re-expand before delivery. + this.#clearSeenContext(); + success = true; + } else { + this.#pending.unshift({ + text: batch, + rawMessages, + renderRevision: this.#renderRevision, + turns: finalTurns, + overflowRecovery: recoveringOverflow || undefined, + }); + await Bun.sleep(this.retryDelayMs); + } } } diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 31d6b1383..5e035c8a2 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -248,7 +248,11 @@ import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate"; import { theme } from "../modes/theme/theme"; import { parseTurnBudget } from "../modes/turn-budget"; import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink"; -import { computeNonMessageBreakdown, computeNonMessageTokens } from "../modes/utils/context-usage"; +import { + computeNonMessageBreakdown, + computeNonMessageTokens, + estimateToolSchemaTokens, +} from "../modes/utils/context-usage"; import { containsWorkflow, renderWorkflowNotice } from "../modes/workflow"; import { createPlanReadMatcher } from "../plan-mode/plan-protection"; import type { PlanModeState } from "../plan-mode/state"; @@ -1037,6 +1041,13 @@ interface ActiveAdvisor { signature: string; } +/** Runtime-only advisor compaction metadata. It never enters the model-facing summary text. */ +interface AdvisorCompactionSummaryMessage extends CompactionSummaryMessage { + firstKeptEntryId?: string; + /** First message index eligible to anchor provider usage after this compaction. */ + advisorUsageAnchorStartIndex?: number; +} + /** Resolved advisor config ready to instantiate as an {@link ActiveAdvisor}. */ interface AdvisorRuntimeDescriptor { config: AdvisorConfig; @@ -2849,10 +2860,23 @@ export class AgentSession { if (contextWindow <= 0) return false; const messages = agent.state.messages; - let contextTokens = incomingTokens; + const estimateOptions = { excludeEncryptedReasoning: true } as const; + let storedConversationTokens = 0; for (const message of messages) { - contextTokens += estimateTokens(message); + storedConversationTokens += estimateTokens(message, estimateOptions); } + // Provider usage (including cache reads and generated output) is the + // trustworthy anchor for accumulated context. Add only the trailing incoming + // delta to that arm. Floor it by a full local estimate — fixed advisor system + // prompt, tool schemas, stored messages, and incoming delta — so provider + // under-reporting or payload transforms cannot suppress maintenance. + const providerContextTokens = this.#estimateAdvisorContextTokens(messages) + incomingTokens; + const localContextTokens = + countTokens(agent.state.systemPrompt) + + estimateToolSchemaTokens(agent.state.tools) + + storedConversationTokens + + incomingTokens; + const contextTokens = compactionContextTokens(providerContextTokens, localContextTokens); if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) { return false; @@ -2876,6 +2900,7 @@ export class AgentSession { const timestamp = String(message.timestamp || Date.now()); if (message.role === "compactionSummary") { + const advisorSummary = message as AdvisorCompactionSummaryMessage; return { type: "compaction", id, @@ -2883,9 +2908,7 @@ export class AgentSession { timestamp, summary: message.summary, shortSummary: message.shortSummary, - firstKeptEntryId: - (message as CompactionSummaryMessage & { firstKeptEntryId?: string }).firstKeptEntryId || - `msg-${i + 1}`, + firstKeptEntryId: advisorSummary.firstKeptEntryId || `msg-${i + 1}`, tokensBefore: message.tokensBefore, } satisfies CompactionEntry; } @@ -2979,11 +3002,15 @@ export class AgentSession { const firstKeptEntryId = compactResult.firstKeptEntryId; const tokensBefore = compactResult.tokensBefore; - // Rebuild messages with the compaction summary + // The retained messages still carry provider usage from before this + // compaction. Record their exact array boundary on the in-memory summary so + // only assistants appended afterward can become the next usage anchor. + const advisorUsageAnchorStartIndex = preparation.recentMessages.length + 1; const summaryMessage = { ...createCompactionSummaryMessage(summary, tokensBefore, new Date().toISOString(), shortSummary), firstKeptEntryId, - } as CompactionSummaryMessage & { firstKeptEntryId?: string }; + advisorUsageAnchorStartIndex, + } satisfies AdvisorCompactionSummaryMessage; agent.replaceMessages([summaryMessage, ...preparation.recentMessages]); return false; @@ -16426,37 +16453,51 @@ export class AgentSession { } /** - * Estimate the advisor's current context tokens. When the advisor has a - * recent non-aborted assistant message with usage, use that prompt's token - * count and add a trailing estimate for messages after it. Otherwise estimate - * every message. + * Estimate the advisor's current context tokens. A successful provider usage + * after the latest advisor compaction is ground truth for the prompt plus its + * generated output; only messages after that anchor are estimated. Usage from + * retained pre-compaction messages is stale and must not immediately retrigger + * maintenance on the newly compacted context. */ #estimateAdvisorContextTokens(messages: AgentMessage[]): number { - let lastUsageIndex: number | null = null; - let lastUsage: AssistantMessage["usage"] | undefined; + let usageAnchorStartIndex = 0; for (let i = messages.length - 1; i >= 0; i--) { - const msg = messages[i]; - if (msg.role === "assistant") { - const assistantMsg = msg as AssistantMessage; - if (assistantMsg.stopReason !== "aborted" && assistantMsg.stopReason !== "error" && assistantMsg.usage) { - lastUsage = assistantMsg.usage; - lastUsageIndex = i; - break; - } + const message = messages[i]; + if (message.role !== "compactionSummary") continue; + const advisorSummary = message as AdvisorCompactionSummaryMessage; + // Advisor summaries created before this runtime-only boundary existed have + // no trustworthy way to distinguish retained from newly appended messages. + // Conservatively ignore every current assistant until the next compaction. + usageAnchorStartIndex = advisorSummary.advisorUsageAnchorStartIndex ?? messages.length; + break; + } + + let lastUsageIndex: number | undefined; + let lastUsage: AssistantMessage["usage"] | undefined; + for (let i = messages.length - 1; i >= usageAnchorStartIndex; i--) { + const message = messages[i]; + if (message.role !== "assistant") continue; + const assistant = message as AssistantMessage; + if (assistant.stopReason !== "aborted" && assistant.stopReason !== "error" && assistant.usage) { + lastUsage = assistant.usage; + lastUsageIndex = i; + break; } } - if (!lastUsage || lastUsageIndex === null) { + + const estimateOptions = { excludeEncryptedReasoning: true } as const; + if (!lastUsage || lastUsageIndex === undefined) { let estimated = 0; for (const message of messages) { - estimated += estimateTokens(message); + estimated += estimateTokens(message, estimateOptions); } return estimated; } let trailingTokens = 0; for (let i = lastUsageIndex + 1; i < messages.length; i++) { - trailingTokens += estimateTokens(messages[i]); + trailingTokens += estimateTokens(messages[i], estimateOptions); } - return calculatePromptTokens(lastUsage) + trailingTokens; + return calculateContextTokens(lastUsage) + trailingTokens; } /** diff --git a/packages/coding-agent/test/advisor-context-maintenance.test.ts b/packages/coding-agent/test/advisor-context-maintenance.test.ts new file mode 100644 index 000000000..c121501ec --- /dev/null +++ b/packages/coding-agent/test/advisor-context-maintenance.test.ts @@ -0,0 +1,199 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { Agent, type AgentMessage, type CompactionSummaryMessage, countTokens } from "@oh-my-pi/pi-agent-core"; +import { calculateContextTokens, estimateTokens, resolveThresholdTokens } from "@oh-my-pi/pi-agent-core/compaction"; +import type { AssistantMessage } from "@oh-my-pi/pi-ai"; +import { createMockModel, type MockModel } from "@oh-my-pi/pi-ai/providers/mock"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { estimateToolSchemaTokens } from "@oh-my-pi/pi-coding-agent/modes/utils/context-usage"; +import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { TempDir } from "@oh-my-pi/pi-utils"; + +const CONTEXT_WINDOW = 372_000; +const CACHE_READ_TOKENS = 371_200; +const INPUT_TOKENS = 200; +const OUTPUT_TOKENS = 150; + +interface MaintenanceHarness { + advisor: Agent; + advisorMock: MockModel; + settings: Settings; +} + +interface AdvisorCompactionSummaryFixture extends CompactionSummaryMessage { + advisorUsageAnchorStartIndex?: number; +} + +describe("AgentSession advisor context maintenance", () => { + let tempDir: TempDir; + let authStorage: AuthStorage; + let session: AgentSession; + + beforeEach(async () => { + tempDir = TempDir.createSync("@pi-advisor-context-maintenance-"); + authStorage = await AuthStorage.create(tempDir.join("auth.db")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + await session?.dispose(); + authStorage.close(); + await tempDir.remove(); + }); + + function createHarness(): MaintenanceHarness { + const primaryMock = createMockModel({ + provider: "anthropic", + responses: [{ content: ["primary complete"] }], + }); + const advisorMock = createMockModel({ + provider: "anthropic", + contextWindow: CONTEXT_WINDOW, + responses: [{ content: ["advisor reviewed current update"] }], + }); + const modelRegistry = new ModelRegistry(authStorage, tempDir.join("models.yml")); + const settings = Settings.isolated({ + "advisor.syncBacklog": "1", + "compaction.enabled": true, + "compaction.strategy": "context-full", + "contextPromotion.enabled": false, + }); + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model: primaryMock, systemPrompt: [], tools: [] }, + streamFn: primaryMock.stream, + }); + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + advisorTools: [], + advisorStreamFn: advisorMock.stream, + }); + settings.setModelRole("advisor", "anthropic/claude-sonnet-4-5"); + expect(session.setAdvisorEnabled(true)).toBe(true); + const advisor = session.getAdvisorAgent(); + if (!advisor) throw new Error("Expected advisor agent to be active"); + advisor.setModel(advisorMock); + + // Keep maintenance on the no-summary recovery branch without blocking the + // primary prompt's own credential preflight. + vi.spyOn(modelRegistry, "getApiKey").mockImplementation(async model => + model === primaryMock ? "test-key" : undefined, + ); + return { advisor, advisorMock, settings }; + } + + function usageAnchor(advisorMock: MockModel, timestamp: number): AssistantMessage { + return { + role: "assistant", + content: [{ type: "text", text: "prior advisor output" }], + api: advisorMock.api, + provider: advisorMock.provider, + model: advisorMock.id, + usage: { + input: INPUT_TOKENS, + output: OUTPUT_TOKENS, + cacheRead: CACHE_READ_TOKENS, + cacheWrite: 0, + totalTokens: CACHE_READ_TOKENS + INPUT_TOKENS + OUTPUT_TOKENS, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp, + }; + } + + function compactionSummary(timestamp: number): AdvisorCompactionSummaryFixture { + return { + role: "compactionSummary", + summary: "bounded advisor summary", + tokensBefore: CACHE_READ_TOKENS + INPUT_TOKENS + OUTPUT_TOKENS, + timestamp, + // `[summary, retained]` is the compacted array; index 2 is the first + // position eligible for a newly appended provider-usage anchor. + advisorUsageAnchorStartIndex: 2, + }; + } + + it("maintains a 371,200-token cached advisor context before the 372,000-token window", async () => { + const { advisor, advisorMock, settings } = createHarness(); + const anchor = usageAnchor(advisorMock, Date.now() - 1_000); + advisor.state.messages.push(anchor); + + await session.prompt("small current update"); + + expect(advisorMock.calls).toHaveLength(1); + const advisorCall = advisorMock.calls[0]; + const update = advisorCall.context.messages.find(message => message.role === "user"); + if (!update) throw new Error("Expected the advisor's incremental update"); + const threshold = resolveThresholdTokens(CONTEXT_WINDOW, settings.getGroup("compaction")); + const providerAndUpdateTokens = calculateContextTokens(anchor.usage) + estimateTokens(update as AgentMessage); + expect(calculateContextTokens(anchor.usage)).toBe(CACHE_READ_TOKENS + INPUT_TOKENS + OUTPUT_TOKENS); + expect(providerAndUpdateTokens).toBeGreaterThan(threshold); + + // Provider usage triggers maintenance, but recovery sends only the bounded + // current update into the reset advisor context. + expect(JSON.stringify(advisorCall.context.messages)).toContain("small current update"); + expect(JSON.stringify(advisor.state.messages)).not.toContain("prior advisor output"); + }); + + it("includes advisor system prompt and tool schemas in the local maintenance floor", async () => { + const { advisor, advisorMock, settings } = createHarness(); + const seed: AgentMessage = { role: "user", content: "small stored advisor message", timestamp: 1 }; + advisor.state.messages.push(seed); + const storedTokens = estimateTokens(seed, { excludeEncryptedReasoning: true }); + const fixedPrefixTokens = countTokens(advisor.state.systemPrompt) + estimateToolSchemaTokens(advisor.state.tools); + const threshold = storedTokens + Math.floor(fixedPrefixTokens / 2); + settings.set("compaction.thresholdTokens", threshold); + + await session.prompt("tiny local-floor update"); + + const advisorCall = advisorMock.calls[0]; + const update = advisorCall.context.messages.find(message => message.role === "user"); + if (!update) throw new Error("Expected the advisor's incremental update"); + const messagesOnlyTokens = storedTokens + estimateTokens(update as AgentMessage); + expect(messagesOnlyTokens).toBeLessThan(threshold); + expect(messagesOnlyTokens + fixedPrefixTokens).toBeGreaterThan(threshold); + expect(JSON.stringify(advisor.state.messages)).not.toContain("small stored advisor message"); + }); + + it("ignores retained provider usage that predates the latest advisor compaction", async () => { + const { advisor, advisorMock } = createHarness(); + const compactedAt = Date.now(); + const summary = compactionSummary(compactedAt); + const retained = usageAnchor(advisorMock, compactedAt); + retained.content = [{ type: "text", text: "retained pre-compaction output" }]; + advisor.state.messages.push(summary, retained); + + await session.prompt("post-compaction update"); + + expect(advisorMock.calls).toHaveLength(1); + const sentContext = JSON.stringify(advisorMock.calls[0].context.messages); + expect(sentContext).toContain("retained pre-compaction output"); + expect(sentContext).toContain("post-compaction update"); + }); + + it("accepts equal-timestamp usage appended after the explicit compaction boundary", async () => { + const { advisor, advisorMock } = createHarness(); + const compactedAt = Date.now(); + const summary = compactionSummary(compactedAt); + const retained = usageAnchor(advisorMock, compactedAt); + retained.content = [{ type: "text", text: "retained pre-compaction output" }]; + const fresh = usageAnchor(advisorMock, compactedAt); + fresh.content = [{ type: "text", text: "fresh post-compaction output" }]; + advisor.state.messages.push(summary, retained, fresh); + + await session.prompt("equal-timestamp post-compaction update"); + + expect(advisorMock.calls).toHaveLength(1); + const sentContext = JSON.stringify(advisorMock.calls[0].context.messages); + expect(sentContext).toContain("equal-timestamp post-compaction update"); + expect(sentContext).not.toContain("retained pre-compaction output"); + expect(sentContext).not.toContain("fresh post-compaction output"); + }); +}); diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index 7cf491436..d9c35128b 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Recent Errors now honors the selected dashboard time range before returning the newest 50 failures ([#5282](https://github.com/can1357/oh-my-pi/issues/5282)) + ## [16.4.7] - 2026-07-12 ### Fixed diff --git a/packages/stats/package.json b/packages/stats/package.json index 73f9a7cc3..005cacdc0 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -55,6 +55,7 @@ "@types/bun": "catalog:", "@types/react": "catalog:", "@types/react-dom": "catalog:", + "linkedom": "catalog:", "postcss": "catalog:" }, "engines": { diff --git a/packages/stats/src/aggregator.ts b/packages/stats/src/aggregator.ts index 553033314..fb37e9305 100644 --- a/packages/stats/src/aggregator.ts +++ b/packages/stats/src/aggregator.ts @@ -444,9 +444,10 @@ export async function getRecentRequests(limit?: number): Promise return dbGetRecentRequests(limit); } -export async function getRecentErrors(limit?: number): Promise { +export async function getRecentErrors(range?: string | null, limit?: number): Promise { await initDb(); - return dbGetRecentErrors(limit); + const { cutoff } = getTimeRangeConfig(range); + return dbGetRecentErrors(limit, cutoff); } export async function getRequestDetails(id: number): Promise { diff --git a/packages/stats/src/client/api.ts b/packages/stats/src/client/api.ts index eef020b20..bf7985073 100644 --- a/packages/stats/src/client/api.ts +++ b/packages/stats/src/client/api.ts @@ -59,8 +59,14 @@ export async function getRecentRequests(limit = 50, signal?: AbortSignal): Promi return fetchJson(`${API_BASE}/stats/recent?limit=${limit}`, { signal }); } -export async function getRecentErrors(limit = 50, signal?: AbortSignal): Promise { - return fetchJson(`${API_BASE}/stats/errors?limit=${limit}`, { signal }); +export async function getRecentErrors( + range: TimeRange = "24h", + limit = 50, + signal?: AbortSignal, +): Promise { + return fetchJson(`${API_BASE}/stats/errors?range=${encodeURIComponent(range)}&limit=${limit}`, { + signal, + }); } export async function getRequestDetails(id: number, signal?: AbortSignal): Promise { diff --git a/packages/stats/src/client/routes/ErrorsRoute.tsx b/packages/stats/src/client/routes/ErrorsRoute.tsx index eb469bcca..273e531af 100644 --- a/packages/stats/src/client/routes/ErrorsRoute.tsx +++ b/packages/stats/src/client/routes/ErrorsRoute.tsx @@ -12,12 +12,12 @@ export interface ErrorsRouteProps { onRequestClick: (id: number) => void; } -export function ErrorsRoute({ active, refreshTrigger, onRequestClick }: ErrorsRouteProps) { +export function ErrorsRoute({ active, range, refreshTrigger, onRequestClick }: ErrorsRouteProps) { const { data: recentErrors, error, loading, - } = useResource(["recent-errors-dense", refreshTrigger], signal => getRecentErrors(50, signal), { + } = useResource(["recent-errors-dense", range, refreshTrigger], signal => getRecentErrors(range, 50, signal), { pollMs: 30000, enabled: active, }); diff --git a/packages/stats/src/db.ts b/packages/stats/src/db.ts index e913989a3..7f24d1e72 100644 --- a/packages/stats/src/db.ts +++ b/packages/stats/src/db.ts @@ -832,15 +832,18 @@ export function getRecentRequests(limit = 100): MessageStats[] { return (stmt.all(limit) as any[]).map(rowToMessageStats); } -export function getRecentErrors(limit = 100): MessageStats[] { +export function getRecentErrors(limit = 100, cutoff?: number | null): MessageStats[] { if (!db) return []; + const hasCutoff = cutoff !== undefined && cutoff !== null; const stmt = db.prepare(` - SELECT * FROM messages + SELECT * FROM messages WHERE stop_reason = 'error' - ORDER BY timestamp DESC + ${hasCutoff ? "AND timestamp >= ?" : ""} + ORDER BY timestamp DESC LIMIT ? `); - return (stmt.all(limit) as any[]).map(rowToMessageStats); + const rows = hasCutoff ? stmt.all(cutoff, limit) : stmt.all(limit); + return rows.map(rowToMessageStats); } export function getMessageById(id: number): MessageStats | null { diff --git a/packages/stats/src/server.ts b/packages/stats/src/server.ts index 17fc9fc40..607de3f88 100644 --- a/packages/stats/src/server.ts +++ b/packages/stats/src/server.ts @@ -184,7 +184,7 @@ const ensureClientBuild = async () => { /** * Handle API requests. */ -async function handleApi(req: Request): Promise { +export async function handleApi(req: Request): Promise { const url = new URL(req.url); const path = url.pathname; @@ -229,7 +229,7 @@ async function handleApi(req: Request): Promise { if (path === "/api/stats/errors") { const limit = url.searchParams.get("limit"); - const stats = await getRecentErrors(limit ? parseInt(limit, 10) : undefined); + const stats = await getRecentErrors(range, limit ? parseInt(limit, 10) : undefined); return Response.json(stats); } diff --git a/packages/stats/test/errors-range.test.ts b/packages/stats/test/errors-range.test.ts new file mode 100644 index 000000000..ce341c166 --- /dev/null +++ b/packages/stats/test/errors-range.test.ts @@ -0,0 +1,79 @@ +import { describe, expect, it } from "bun:test"; +import { initDb, insertMessageStats } from "../src/db"; +import { handleApi } from "../src/server"; +import type { MessageStats } from "../src/types"; +import { installStatsTestIsolation } from "./helpers/temp-agent"; + +const HOUR_MS = 60 * 60 * 1000; + +installStatsTestIsolation("@pi-stats-errors-range-"); + +function makeError(timestamp: number, entryId: string): MessageStats { + return { + sessionFile: "/tmp/errors-range-session.jsonl", + entryId, + folder: "/tmp/project", + model: "gpt-5.4", + provider: "openai-codex", + api: "openai-codex-responses", + timestamp, + duration: 1000, + ttft: 100, + stopReason: "error", + errorMessage: `failure ${entryId}`, + usage: { + input: 1000, + output: 500, + cacheRead: 200, + cacheWrite: 0, + totalTokens: 1700, + cost: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + total: 0, + }, + }, + agentType: "main", + }; +} + +async function readMessages(response: Response): Promise { + expect(response.status).toBe(200); + return response.json() as Promise; +} + +describe("Recent Errors range", () => { + it("filters by the mapped range before returning the newest 50 errors", async () => { + await initDb(); + const now = Date.now(); + const recentErrors = Array.from({ length: 50 }, (_, index) => makeError(now - index * 1000, `recent-${index}`)); + const oldError = makeError(now - 48 * HOUR_MS, "outside-24h"); + insertMessageStats([...recentErrors, oldError]); + + const dayErrors = await readMessages( + await handleApi(new Request("http://stats.test/api/stats/errors?range=24h&limit=50")), + ); + expect(dayErrors).toHaveLength(50); + expect(dayErrors.map(error => error.entryId)).toEqual(recentErrors.map(error => error.entryId)); + expect(dayErrors.some(error => error.entryId === oldError.entryId)).toBe(false); + + const allErrors = await readMessages( + await handleApi(new Request("http://stats.test/api/stats/errors?range=all&limit=51")), + ); + expect(allErrors).toHaveLength(51); + expect(allErrors.at(-1)?.entryId).toBe(oldError.entryId); + + const defaultErrors = await readMessages( + await handleApi(new Request("http://stats.test/api/stats/errors?limit=51")), + ); + expect(defaultErrors).toHaveLength(50); + expect(defaultErrors.some(error => error.entryId === oldError.entryId)).toBe(false); + + const fallbackErrors = await readMessages( + await handleApi(new Request("http://stats.test/api/stats/errors?range=unknown&limit=51")), + ); + expect(fallbackErrors.map(error => error.entryId)).toEqual(defaultErrors.map(error => error.entryId)); + }); +}); diff --git a/packages/stats/test/errors-route-range.test.tsx b/packages/stats/test/errors-route-range.test.tsx new file mode 100644 index 000000000..05d168f22 --- /dev/null +++ b/packages/stats/test/errors-route-range.test.tsx @@ -0,0 +1,81 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { parseHTML } from "linkedom"; +import { act } from "react"; +import { createRoot, type Root } from "react-dom/client"; +import { ErrorsRoute } from "../src/client/routes/ErrorsRoute"; + +type FetchInput = string | URL | Request; +type FetchInit = RequestInit | BunFetchRequestInit; + +const originalGlobals = new Map(); +let root: Root | null = null; + +function installGlobal(name: string, value: unknown): void { + originalGlobals.set(name, Object.getOwnPropertyDescriptor(globalThis, name)); + Object.defineProperty(globalThis, name, { configurable: true, value, writable: true }); +} + +function restoreGlobals(): void { + for (const [name, descriptor] of originalGlobals) { + if (descriptor) { + Object.defineProperty(globalThis, name, descriptor); + } else { + Reflect.deleteProperty(globalThis, name); + } + } + originalGlobals.clear(); +} + +afterEach(async () => { + const activeRoot = root; + if (activeRoot) { + await act(async () => { + activeRoot.unmount(); + }); + root = null; + } + vi.restoreAllMocks(); + restoreGlobals(); +}); + +describe("ErrorsRoute range", () => { + it("requests the selected range again when the range changes", async () => { + const domWindow = parseHTML('
').window; + installGlobal("window", domWindow); + installGlobal("document", domWindow.document); + installGlobal("navigator", domWindow.navigator); + installGlobal("Node", domWindow.Node); + installGlobal("Element", domWindow.Element); + installGlobal("HTMLElement", domWindow.HTMLElement); + installGlobal("HTMLIFrameElement", domWindow.HTMLIFrameElement); + installGlobal("SVGElement", domWindow.SVGElement); + installGlobal("IS_REACT_ACT_ENVIRONMENT", true); + + const requestedUrls: string[] = []; + const fetchStub = Object.assign( + async (input: FetchInput, _init?: FetchInit) => { + requestedUrls.push(input instanceof Request ? input.url : input.toString()); + return Response.json([]); + }, + { preconnect: globalThis.fetch.preconnect }, + ); + vi.spyOn(globalThis, "fetch").mockImplementation(fetchStub); + + const container = domWindow.document.getElementById("root"); + if (!container) throw new Error("Expected test root"); + root = createRoot(container as unknown as Element); + + await act(async () => { + root?.render( {}} />); + }); + expect(requestedUrls).toEqual(["/api/stats/errors?range=24h&limit=50"]); + + await act(async () => { + root?.render( {}} />); + }); + expect(requestedUrls).toEqual([ + "/api/stats/errors?range=24h&limit=50", + "/api/stats/errors?range=7d&limit=50", + ]); + }); +}); diff --git a/packages/stats/tsconfig.client.json b/packages/stats/tsconfig.client.json index 2a4594ea1..7b97643c6 100644 --- a/packages/stats/tsconfig.client.json +++ b/packages/stats/tsconfig.client.json @@ -1,7 +1,8 @@ { "extends": "../tsconfig.workspace.json", "include": [ - "src/client" + "src/client", + "test/errors-route-range.test.tsx" ], "compilerOptions": { "jsx": "react-jsx", From 23c78e74dfd6ec33303ba362d6e5f110fbecc604 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 18:14:14 +0000 Subject: [PATCH 026/293] fix(browser): use debugCatchError in stealth acquire paths The stealth puppeteer-core patch re-implements world acquisition without Runtime.enable and used the bare debugError logger in its new FrameManager/WebWorker catch handlers. Puppeteer leaves debugError undefined when the puppeteer:error debug channel is disabled (the default), so a transient CDP failure during world re-acquire threw TypeError: debugError is not a function, escaped as an unhandledRejection, and the postmortem handler killed the whole process along with every subagent. Replace every bare debugError catch handler added by the patch with the safe debugCatchError (already imported for upstream handlers) so a disabled logger can never throw a secondary error. Fixes #5296 --- packages/coding-agent/CHANGELOG.md | 4 + ...browser-stealth-acquire-debugerror.test.ts | 117 ++++++++++++++++++ patches/puppeteer-core@25.3.0.patch | 16 +-- 3 files changed, 129 insertions(+), 8 deletions(-) create mode 100644 packages/coding-agent/test/tools/browser-stealth-acquire-debugerror.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 10f7883ed..4a621165f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the browser tool crashing the whole process (parent session and every subagent) when a CDP world re-acquire failed mid-navigation: the stealth `puppeteer-core` patch called the bare `debugError` logger, which is `undefined` while the `puppeteer:error` debug channel is disabled (the default), turning a transient acquire failure into a fatal `TypeError` unhandled rejection. The patched `FrameManager`/`WebWorker` acquire paths now use `debugCatchError` ([#5296](https://github.com/can1357/oh-my-pi/issues/5296)) + ## [16.4.8] - 2026-07-12 ### Fixed diff --git a/packages/coding-agent/test/tools/browser-stealth-acquire-debugerror.test.ts b/packages/coding-agent/test/tools/browser-stealth-acquire-debugerror.test.ts new file mode 100644 index 000000000..dec0af52e --- /dev/null +++ b/packages/coding-agent/test/tools/browser-stealth-acquire-debugerror.test.ts @@ -0,0 +1,117 @@ +/** + * Regression test for issue #5296: the stealth `puppeteer-core` patch + * (`patches/puppeteer-core@25.3.0.patch`) re-implements world acquisition + * without `Runtime.enable`. Its new catch handlers in `FrameManager` called the + * bare `debugError` logger, which puppeteer leaves `undefined` when the + * `puppeteer:error` debug channel is disabled (the default). A transient CDP + * failure during world re-acquire then threw `TypeError: debugError is not a + * function` from `#doAcquireWorlds`, escaped as an `unhandledRejection`, and the + * postmortem handler killed the whole OMP process (parent session + every + * subagent). + * + * The test drives the real patched `FrameManager` with a `send()` that always + * rejects (a mid-flight CDP failure) and asserts the acquire path emits no + * unhandled `TypeError`. + * + * Real timers are deliberate here (see repo rule ts-no-test-timers): the fatal + * path is the coalescing acquirer's fire-and-forget `void this.#acquireWorlds()` + * retrigger, whose rejection escapes only to the global `unhandledRejection` + * handler — there is no promise or event the test can await, and fake timers + * serialise the two concurrent acquires so the retrigger (and thus the bug) + * never fires. Short real delays let the event loop interleave the acquires the + * way it does in production. + */ + +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { CdpFrame } from "puppeteer-core/lib/puppeteer/cdp/Frame.js"; +import { FrameManager } from "puppeteer-core/lib/puppeteer/cdp/FrameManager.js"; +import { MAIN_WORLD, PUPPETEER_WORLD } from "puppeteer-core/lib/puppeteer/cdp/IsolatedWorlds.js"; +import { EventEmitter } from "puppeteer-core/lib/puppeteer/common/EventEmitter.js"; +import { TimeoutSettings } from "puppeteer-core/lib/puppeteer/common/TimeoutSettings.js"; +import { debugError } from "puppeteer-core/lib/puppeteer/common/util.js"; + +const ACQUIRE_TIMEOUT_MS = 40; + +// A CDP session double whose every `send` rejects, modelling a navigation that +// tears the target's execution contexts down mid-acquire. +class RejectingSession extends EventEmitter> { + constructor(readonly sessionId: string) { + super(); + } + id(): string { + return this.sessionId; + } + send(): Promise { + return Promise.reject(new Error("mid-flight CDP failure")); + } + target(): unknown { + return { _targetId: "T", type: () => "page" }; + } +} + +function makeFrameManager(session: RejectingSession): FrameManager { + const browser = { isNetworkEnabled: () => false, isIssuesEnabled: () => false, connected: true }; + const page = { browser: () => browser, isClosed: () => false, emit() {}, once() {}, off() {} }; + const timeoutSettings = new TimeoutSettings(); + timeoutSettings.setDefaultTimeout(ACQUIRE_TIMEOUT_MS); + // The patched FrameManager only touches the members exercised here; the + // puppeteer-internal `CdpCDPSession` / `CdpPage` types are far wider than the + // acquire path needs, so the doubles cross the boundary with a cast. + return new FrameManager(session as never, page as never, timeoutSettings); +} + +describe("stealth FrameManager world acquire — issue #5296", () => { + const rejections: unknown[] = []; + const onUnhandled = (reason: unknown) => rejections.push(reason); + + beforeEach(() => { + rejections.length = 0; + process.on("unhandledRejection", onUnhandled); + }); + + afterEach(() => { + process.off("unhandledRejection", onUnhandled); + }); + + it("keeps disabled debugError undefined so bare calls would crash", () => { + // The precondition that makes the bug fatal: with the puppeteer:error + // channel off, the logger the patch used is not callable. + expect(debugError).toBeUndefined(); + }); + + it("does not emit an unhandled TypeError when acquire fails mid-flight", async () => { + const session = new RejectingSession("S1"); + const frameManager = makeFrameManager(session); + const frame = new CdpFrame(frameManager, "F1", undefined, session as never); + frameManager._frameTree.addFrame(frame); + + // Navigation installs the lazy context providers and invalidates the old + // contexts; the async handler must settle before we pull a context. + session.emit("Page.frameNavigated", { + frame: { id: "F1", parentId: undefined, url: "about:blank" }, + type: "Navigation", + }); + await Bun.sleep(20); + + // Concurrent pulls on both worlds force the coalescing acquirer to + // re-run (`void this.#acquireWorlds` in its `finally`), which is the exact + // path where `#doAcquireWorlds`'s catch previously threw a bare + // `debugError(error)`. + const main = frame.worlds[MAIN_WORLD]; + const util = frame.worlds[PUPPETEER_WORLD]; + const results = await Promise.allSettled([main.evaluate(() => 1), util.evaluate(() => 1)]); + + // Let the re-triggered acquire settle and any stray rejection surface. + await Bun.sleep(ACQUIRE_TIMEOUT_MS + 40); + + const typeErrors = rejections.filter( + (reason): reason is TypeError => reason instanceof Error && reason.name === "TypeError", + ); + expect(typeErrors).toHaveLength(0); + expect(rejections).toHaveLength(0); + + // The failure is still observable as an ordinary, recoverable evaluate + // error rather than a silent process death. + expect(results.every(r => r.status === "rejected")).toBe(true); + }); +}); diff --git a/patches/puppeteer-core@25.3.0.patch b/patches/puppeteer-core@25.3.0.patch index bc93ba914..0bf86a5b2 100644 --- a/patches/puppeteer-core@25.3.0.patch +++ b/patches/puppeteer-core@25.3.0.patch @@ -352,7 +352,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + client.send('Page.addScriptToEvaluateOnNewDocument', { + source: `//# sourceURL=${PuppeteerURL.INTERNAL_URL}`, + worldName: UTILITY_WORLD_NAME, -+ }).catch(debugError), + }).catch(debugCatchError), ...(frame ? Array.from(this.#scriptsToEvaluateOnNewDocument.values()) : []).map(script => { @@ -445,7 +445,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + worldName: UTILITY_WORLD_NAME, + grantUniveralAccess: true, + }) -+ .catch(debugError); + .catch(debugCatchError); + const utilityId = iso && typeof iso.executionContextId === 'number' ? iso.executionContextId : undefined; + if (utilityId !== undefined) { + this.#onExecutionContextCreated({ @@ -486,7 +486,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + } + } + catch (error) { -+ debugError(error); + debugCatchError(error); + } + } + // xxx-stealth: resolve a frame's MAIN-world execution context id without @@ -511,7 +511,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + expression: 'globalThis', + serializationOptions: { serialization: 'idOnly' }, + }) -+ .catch(debugError); + .catch(debugCatchError); + return parse(globalThis?.result?.objectId); + } + if (utilityId === undefined) { @@ -523,21 +523,21 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + contextId: utilityId, + serializationOptions: { serialization: 'idOnly' }, + }) -+ .catch(debugError); + .catch(debugCatchError); + const utilDocObjectId = utilDoc?.result?.objectId; + if (typeof utilDocObjectId !== 'string') { + return undefined; + } + const described = await session + .send('DOM.describeNode', { objectId: utilDocObjectId }) -+ .catch(debugError); + .catch(debugCatchError); + const backendNodeId = described?.node?.backendNodeId; + if (typeof backendNodeId !== 'number') { + return undefined; + } + const mainNode = await session + .send('DOM.resolveNode', { backendNodeId }) -+ .catch(debugError); + .catch(debugCatchError); + return parse(mainNode?.object?.objectId); } async #createIsolatedWorld(session, name) { @@ -605,7 +605,7 @@ index 3d68f887920ded269eb641273a5a13dee235ae1d..dcdd86c8697c0dbd2dd2162c9a739dd9 + this.#world.setContext(new ExecutionContext(client, { id }, this.#world)); + } + }) -+ .catch(debugError); + .catch(debugCatchError); this.#client.once('Inspector.workerScriptLoaded', () => { this.#workerLoaded.resolve(); }); From d77a3e154a27015d60dbb753be32cb69f595d7b3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 18:23:30 +0000 Subject: [PATCH 027/293] fix(tui): aligned ask "Other" custom-input chrome to prompt gutter The prompt-style HookEditorComponent (used by the ask tool's "Other" custom-input flow) rendered its title, option list, and hint via Text(padX=1) while the borderless editor beneath renders its `> ` gutter at column 0, leaving the input row one column left of everything else. Pad the prompt-style chrome at column 0 to match the gutter; hook-style (bordered) chrome keeps its 1-column indent that lines up with the bordered editor body. Fixes #5313 --- packages/coding-agent/CHANGELOG.md | 4 ++++ .../src/modes/components/hook-editor.ts | 9 +++++--- .../coding-agent/test/hook-editor.test.ts | 22 ++++++++++++++++++- 3 files changed, 31 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e0b1b1608..7aca25bc2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -10,6 +10,10 @@ - Updated status event log to prioritize the most recent entries in the display window +### Fixed + +- Fixed the ask tool's "Other" custom-input dialog rendering the title, options, and hint one column to the right of the `> ` input gutter; the prompt-style editor chrome now aligns to column 0 ([#5313](https://github.com/can1357/oh-my-pi/issues/5313)) + ### Removed - Removed the unreliable Bing and Yahoo HTML-scraping web search providers diff --git a/packages/coding-agent/src/modes/components/hook-editor.ts b/packages/coding-agent/src/modes/components/hook-editor.ts index 19fe74c18..aa06b759d 100644 --- a/packages/coding-agent/src/modes/components/hook-editor.ts +++ b/packages/coding-agent/src/modes/components/hook-editor.ts @@ -47,8 +47,11 @@ export class HookEditorComponent extends Container { this.addChild(new DynamicBorder()); this.addChild(new Spacer(1)); - // Title - this.addChild(new Text(theme.fg("accent", title), 1, 0)); + // Title. Prompt-style renders the borderless editor's `> ` gutter at + // column 0, so pad the title to match; hook-style keeps the 1-col indent + // that lines up with its bordered editor body (#5313). + const chromePadX = this.#promptStyle ? 0 : 1; + this.addChild(new Text(theme.fg("accent", title), chromePadX, 0)); this.addChild(new Spacer(1)); // Editor @@ -69,7 +72,7 @@ export class HookEditorComponent extends Container { const hint = this.#promptStyle ? "enter or ctrl+q submit esc cancel ctrl+g external editor" : "ctrl+q/ctrl+enter submit esc cancel ctrl+g external editor"; - this.addChild(new Text(theme.fg("dim", hint), 1, 0)); + this.addChild(new Text(theme.fg("dim", hint), chromePadX, 0)); this.addChild(new Spacer(1)); this.addChild(new DynamicBorder()); diff --git a/packages/coding-agent/test/hook-editor.test.ts b/packages/coding-agent/test/hook-editor.test.ts index a42ace9fc..a816391d6 100644 --- a/packages/coding-agent/test/hook-editor.test.ts +++ b/packages/coding-agent/test/hook-editor.test.ts @@ -359,7 +359,7 @@ describe("HookEditorComponent prompt-style mode", () => { expect(lines[0]).toMatch(/^─+$/); expect(lines.at(-1)).toMatch(/^─+$/); expect(lines[4]?.startsWith("> ")).toBe(true); - expect(rendered).toContain(" enter or ctrl+q submit esc cancel"); + expect(rendered).toContain("enter or ctrl+q submit esc cancel"); expect(rendered).not.toContain("shift+enter newline"); expect(rendered).toContain("ctrl+g external editor"); }); @@ -419,6 +419,26 @@ describe("HookEditorComponent prompt-style mode", () => { expect(onCancel).toHaveBeenCalledTimes(1); expect(onSubmit).not.toHaveBeenCalled(); }); + + it("aligns the title and hint with the editor prompt gutter at column zero (#5313)", () => { + const title = "◆ Other (type your own)\nEnter your response:"; + const component = new HookEditorComponent(createTui(), title, "不太清楚,", vi.fn(), vi.fn(), { + promptStyle: true, + }); + const lines = renderLines(component); + + const titleRow = lines.find(line => line.includes("Enter your response:")); + const gutterRow = lines.find(line => line.startsWith("> ")); + const hintRow = lines.find(line => line.includes("esc cancel")); + + expect(titleRow).toBeDefined(); + expect(gutterRow).toBeDefined(); + expect(hintRow).toBeDefined(); + // The borderless prompt-style editor renders `> ` starting at column 0, so + // the surrounding title/hint chrome must not carry a leading indent. + expect(titleRow!.startsWith("Enter your response:")).toBe(true); + expect(hintRow!.startsWith(" ")).toBe(false); + }); }); describe("ExtensionUiController hook editor abort", () => { From 68f84d7c2054e20353b70c7746f8f4f88dec4119 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 18:22:33 +0000 Subject: [PATCH 028/293] fix(tui): prevented stale-buffer flicker - Kept fullscreen replacement overlays mounted through asynchronous transcript rebuilds. - Fused alternate-screen exit with destructive repaint and removed resize-time buffer switches. - Preserved statically detected synchronized output when DECRQM probing is inconclusive. Fixes #5319 --- docs/environment-variables.md | 2 +- docs/tui-core-renderer.md | 2 +- docs/tui-runtime-internals.md | 4 +- packages/coding-agent/CHANGELOG.md | 1 + .../modes/controllers/selector-controller.ts | 17 ++-- .../src/modes/interactive-mode.ts | 10 ++- .../src/prompts/system/tan-context-switch.md | 2 +- .../selector-controller-overlay-focus.test.ts | 69 +++++++++++++- packages/tui/CHANGELOG.md | 4 + packages/tui/src/terminal.ts | 27 +++--- packages/tui/src/tui.ts | 72 ++++++--------- packages/tui/test/render-regressions.test.ts | 89 +++++++++++++++++++ .../tui/test/resize-viewport-defer.test.ts | 45 +++++----- packages/tui/test/terminal-appearance.test.ts | 7 +- 14 files changed, 252 insertions(+), 99 deletions(-) diff --git a/docs/environment-variables.md b/docs/environment-variables.md index adf3d310c..cb34ed6b8 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -407,7 +407,7 @@ These are read as runtime signals; they are usually set by the terminal/OS rathe | `PI_NO_DECCARA` | If set (truthy), disables Kitty DECCARA rectangular-SGR background fills (forces padded-string rendering) | | `PI_DEBUG_REDRAW` | If `1`, enables redraw debug logging | | `PI_FORCE_IMAGE_PROTOCOL` | Forces terminal image protocol detection (`kitty`, `iterm2`/`iterm`, `sixel`, `none`) | -| `PI_TUI_RESIZE_IN_PLACE` | `1`/`true` force in-place resize (no alt-screen borrow, no ED3 rewrap); `0`/`false` force the alt-screen fast path. Default-on for Warp, which re-reports its size on alt-screen toggles | +| `PI_TUI_RESIZE_IN_PLACE` | `1`/`true` preserves terminal-managed history and repaints after resize settle; `0`/`false` uses viewport-only drag paints followed by one ED3 history rewrap. Neither path switches terminal buffers. Default-on for Warp and multiplexers | --- diff --git a/docs/tui-core-renderer.md b/docs/tui-core-renderer.md index 25dfb4ad0..d56b18cda 100644 --- a/docs/tui-core-renderer.md +++ b/docs/tui-core-renderer.md @@ -349,7 +349,7 @@ default-on only for kitty/ghostty (`PI_NO_KITTY_PLACEHOLDERS` / | `PI_HARDWARE_CURSOR=1` | Show the real hardware cursor instead of a rendered one. | | `PI_NOTIFICATIONS=off\|0\|false` | Suppress terminal notifications. | | `PI_DEBUG_REDRAW=1` | Log the chosen render intent + ledger state per frame to the debug log. | -| `PI_TUI_RESIZE_IN_PLACE=1\|0` | Force resize to repaint in place (no alt-screen borrow, no ED3 rewrap) on / off. Default-on for terminals that re-report size on alt-screen toggles (Warp). | +| `PI_TUI_RESIZE_IN_PLACE=1\|0` | `1` preserves terminal-managed history and repaints after settle; `0` uses viewport-only drag paints plus one settled ED3 history rewrap. Neither path borrows the alternate screen. Default-on for terminals that re-report size on buffer toggles (Warp). | Removed with the old engine: `PI_TUI_ED3_SAFE` (no ED3-risk lever exists), `PI_CLEAR_ON_SHRINK` (shrinks always clear exactly), `PI_TUI_DEBUG` (per-render diff --git a/docs/tui-runtime-internals.md b/docs/tui-runtime-internals.md index 8040cdc15..d8960f0f0 100644 --- a/docs/tui-runtime-internals.md +++ b/docs/tui-runtime-internals.md @@ -141,9 +141,9 @@ Resize events are event-driven from `ProcessTerminal` to `TUI.requestRender()`. Effects: -- A resize is an explicit user gesture: outside multiplexers the engine erases and replays (`ED3` + full paint) so history rewraps at the new geometry; the commit ledger restarts from the replayed frame. +- A resize is an explicit user gesture: outside multiplexers the engine rewrites only the visible viewport during the drag, directly on the normal buffer, then erases and replays once (`ED3` + full paint) after the drag settles so history rewraps at the new geometry. Avoiding alternate-screen switches prevents the saved pre-TUI normal buffer from flashing at settle on terminals without effective synchronized output. - Inside terminal multiplexers, resize repaints the visible window in place after a settle debounce (issue #2088); pane history keeps its old wrap, like any shell output, because pane scrollback cannot be erased safely. -- Terminals that re-report their size when the alternate screen buffer is toggled (Warp reports a height one row different for the alt buffer) take the in-place path too. The non-multiplexer fast path borrows the alternate screen for drag frames, so on these terminals each alt enter/leave emits a fresh resize event, which re-enters the fast path — a self-sustaining loop that floods ED3 full repaints with stable geometry. `resizeRepaintsInPlace()` (covering multiplexers and these terminals; overridable via `PI_TUI_RESIZE_IN_PLACE`) routes them through the in-place repaint, which never touches the alt buffer. +- Terminals that re-report their size when the alternate screen buffer is toggled (Warp reports a height one row different for the alt buffer) take the same history-preserving in-place path. `resizeRepaintsInPlace()` covers multiplexers and these terminals and remains overridable via `PI_TUI_RESIZE_IN_PLACE`; the viewport-only direct-terminal path no longer toggles buffers, so the override controls settled history rewrap only. - Overlay visibility can depend on terminal dimensions (`OverlayOptions.visible`); focus is corrected when overlays become non-visible after resize. ## Streaming and incremental UI updates diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c25af880a..e069b35d5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -25,6 +25,7 @@ ### Fixed +- Fixed `/resume` and plan approval exposing the previous session while their asynchronous session replacement was still loading by keeping fullscreen overlays mounted until the rebuilt transcript is ready ([#5319](https://github.com/can1357/oh-my-pi/issues/5319)). - Fixed inconsistent history rendering when toggling the display setting for compacted items - Fixed configured `retry.fallbackChains` never engaging on non-retryable provider errors (e.g. "Cloud Code Assist API returned an empty response"): a hard error on a model covered by a fallback chain now switches to the next candidate instead of failing the turn, while still never backoff-retrying the failing model itself - Fixed transcript rebuilds (compaction, `/compact`, and toggling history display) repainting content below stale scrollback when collapsing history; rebuilds now correctly clear the scrollback buffer when history is collapsed diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index adbd74d38..dcd3f9d9a 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -1102,12 +1102,10 @@ export class SelectorController { // every project's history when the cwd has nothing to resume. See #3099. const historyStorage = this.ctx.historyStorage; const historyMatcher = historyStorage ? (query: string) => historyStorage.matchingSessionIds(query) : undefined; - // Fullscreen session picker on the alternate screen (the /settings idiom): - // the overlay borrows the alt buffer and enables mouse tracking (wheel - // scroll + click-to-resume) for its lifetime, leaving the transcript - // untouched underneath. Anchored top-left at full size so a mouse row maps - // directly to a rendered line (the overlay paints from screen row 0), and - // `fillHeight` pads the body so the footer pins to the screen bottom. + // Keep the fullscreen picker on the alternate buffer while a selected + // session is loaded and its transcript is rebuilt. Closing it first exposes + // the stale normal buffer for the entire async switch on terminals without + // effective synchronized output. let overlayHandle: OverlayHandle | undefined; const done = () => { overlayHandle?.hide(); @@ -1117,8 +1115,11 @@ export class SelectorController { const selector = new SessionSelectorComponent( sessions, async (session: SessionInfo) => { - done(); - await this.handleResumeSession(session.path); + try { + await this.handleResumeSession(session.path); + } finally { + done(); + } }, () => { done(); diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 8311acbc2..3dae7aa2e 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -2503,8 +2503,6 @@ export class InteractiveMode implements InteractiveModeContext { const finish = (choice: string | undefined): void => { if (settled) return; settled = true; - this.#hidePlanReview(); - this.ui.requestRender(); resolve(choice); }; const overlay = new PlanReviewOverlay( @@ -3424,6 +3422,10 @@ export class InteractiveMode implements InteractiveModeContext { }, { slider }, ); + const closePlanReview = (): void => { + this.#hidePlanReview(); + this.ui.requestRender(); + }; if (choice === "Approve and execute" || choice === "Approve and compact context" || choice === keepContextLabel) { try { @@ -3436,6 +3438,7 @@ export class InteractiveMode implements InteractiveModeContext { } if (!latestPlanContent) { this.showError(`Plan file not found at ${planFilePath}`); + closePlanReview(); return; } // Capture the operator's tier choice and hand it to #approvePlan, which @@ -3478,6 +3481,7 @@ export class InteractiveMode implements InteractiveModeContext { `Failed to finalize approved plan: ${error instanceof Error ? error.message : String(error)}`, ); } + closePlanReview(); return; } @@ -3496,8 +3500,10 @@ export class InteractiveMode implements InteractiveModeContext { } catch (error) { this.showError(`Failed to refine plan: ${error instanceof Error ? error.message : String(error)}`); } + closePlanReview(); return; } + closePlanReview(); } /** diff --git a/packages/coding-agent/src/prompts/system/tan-context-switch.md b/packages/coding-agent/src/prompts/system/tan-context-switch.md index 55468b15a..88cd57291 100644 --- a/packages/coding-agent/src/prompts/system/tan-context-switch.md +++ b/packages/coding-agent/src/prompts/system/tan-context-switch.md @@ -1,5 +1,5 @@ -The conversation above belongs to your parent session. +The conversation above belongs to your parent session. You are a fork created solely to handle the user's request below. Your parent agent is still working on the original task — that responsibility is diff --git a/packages/coding-agent/test/modes/controllers/selector-controller-overlay-focus.test.ts b/packages/coding-agent/test/modes/controllers/selector-controller-overlay-focus.test.ts index 539dfb3c7..6017e8c51 100644 --- a/packages/coding-agent/test/modes/controllers/selector-controller-overlay-focus.test.ts +++ b/packages/coding-agent/test/modes/controllers/selector-controller-overlay-focus.test.ts @@ -1,12 +1,19 @@ -import { beforeAll, describe, expect, it, vi } from "bun:test"; +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; +import type { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector"; import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; +import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-listing"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; beforeAll(async () => { await initTheme(); }); +afterEach(() => { + vi.restoreAllMocks(); +}); + interface EditorSlot { children: unknown[]; clear: () => void; @@ -83,3 +90,63 @@ describe("SelectorController.focusActiveEditorArea", () => { expect(setFocus).toHaveBeenCalledWith(editor); }); }); + +describe("SelectorController session replacement overlay", () => { + it("keeps the fullscreen selector visible until the resumed transcript is ready", async () => { + const session: SessionInfo = { + path: "/tmp/resume.jsonl", + id: "resume", + cwd: "/tmp", + title: "Resume target", + created: new Date("2026-01-01T00:00:00Z"), + modified: new Date("2026-01-02T00:00:00Z"), + messageCount: 2, + size: 1, + firstMessage: "first", + allMessagesText: "first second", + }; + vi.spyOn(SessionManager, "list").mockResolvedValue([session]); + + const overlayHidden = Promise.withResolvers(); + const hide = vi.fn(() => overlayHidden.resolve()); + let selector: SessionSelectorComponent | undefined; + const editor = { id: "editor" }; + const editorContainer = createEditorSlot(editor); + const ctx = { + editor, + editorContainer, + sessionManager: { + getCwd: () => "/tmp", + getSessionDir: () => "/tmp", + }, + ui: { + showOverlay: vi.fn(component => { + selector = component as SessionSelectorComponent; + return { hide, setHidden: vi.fn(), isHidden: () => false }; + }), + setFocus: vi.fn(), + requestRender: vi.fn(), + terminal: { rows: 24 }, + }, + } as unknown as InteractiveModeContext; + const controller = new SelectorController(ctx); + const resumeStarted = Promise.withResolvers(); + const resumed = Promise.withResolvers(); + const handleResume = vi.spyOn(controller, "handleResumeSession").mockImplementation(() => { + resumeStarted.resolve(); + return resumed.promise; + }); + + await controller.showSessionSelector(); + expect(selector).toBeDefined(); + selector!.handleInput("\n"); + await resumeStarted.promise; + + expect(handleResume).toHaveBeenCalledWith(session.path); + expect(hide).not.toHaveBeenCalled(); + + resumed.resolve(); + await overlayHidden.promise; + expect(hide).toHaveBeenCalledTimes(1); + }); +}); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 3143afbfc..df7609ee9 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed fullscreen session-replacement overlays and resize drags exposing stale normal-buffer frames on terminals without effective DEC 2026: asynchronous replacements now keep their overlay visible until the rebuilt transcript is ready, overlay exit is fused into the destructive paint, and resize viewport frames rewrite the normal buffer without alternate-screen switches. Inconclusive DECRQM probes also no longer disable statically detected synchronized output ([#5319](https://github.com/can1357/oh-my-pi/issues/5319)). + ## [16.4.7] - 2026-07-12 ### Fixed diff --git a/packages/tui/src/terminal.ts b/packages/tui/src/terminal.ts index b35dfa1f5..982db597d 100644 --- a/packages/tui/src/terminal.ts +++ b/packages/tui/src/terminal.ts @@ -390,9 +390,11 @@ export interface Terminal { get appearance(): TerminalAppearance | undefined; /** * Register a callback fired once per DEC private mode when its DECRQM support - * status resolves. Optional: only real terminals implement capability probing. + * status resolves. `confirmed` is false when the terminal answered the DA1 + * sentinel without answering DECRQM, which proves only that querying support + * is unavailable — not that the private mode itself is unsupported. */ - onPrivateModeReport?(callback: (mode: number, supported: boolean) => void): void; + onPrivateModeReport?(callback: (mode: number, supported: boolean, confirmed?: boolean) => void): void; } /** @@ -480,7 +482,7 @@ export class ProcessTerminal implements Terminal { #da1SentinelOwners: Da1SentinelOwner[] = []; /** Resolved DECRQM support per private mode (mode → supported). */ #privateModeSupport = new Map(); - #privateModeCallbacks: Array<(mode: number, supported: boolean) => void> = []; + #privateModeCallbacks: Array<(mode: number, supported: boolean, confirmed: boolean) => void> = []; /** Whether DEC 2048 in-band resize notifications are currently enabled. */ #inBandResizeActive = false; /** Reassembly buffer for a DEC 2048 in-band resize report split across stdin reads. */ @@ -531,7 +533,7 @@ export class ProcessTerminal implements Terminal { } } - onPrivateModeReport(callback: (mode: number, supported: boolean) => void): void { + onPrivateModeReport(callback: (mode: number, supported: boolean, confirmed?: boolean) => void): void { this.#privateModeCallbacks.push(callback); } @@ -866,8 +868,10 @@ export class ProcessTerminal implements Terminal { break; } case "privateMode": { - // DA1 beat the DECRPM reply for this mode → treat as unsupported. - this.#resolvePrivateMode(owner.mode, false); + // DA1 beat the DECRPM reply. The terminal cannot report this + // capability, but may still implement it; keep that distinction + // so static terminal detection is not incorrectly downgraded. + this.#resolvePrivateMode(owner.mode, false, false); break; } case "keyboard": { @@ -1129,7 +1133,7 @@ export class ProcessTerminal implements Terminal { } #handlePrivateModeReport(mode: number, status: string): void { - this.#resolvePrivateMode(mode, isPrivateModeSupported(status)); + this.#resolvePrivateMode(mode, isPrivateModeSupported(status), true); if (isXtermScrollToBottomMode(mode) && isPrivateModeSet(status)) { this.#disableXtermScrollToBottomMode(mode); } @@ -1137,15 +1141,16 @@ export class ProcessTerminal implements Terminal { /** * Record DECRQM support for a private mode (idempotent — first result wins) - * and notify subscribers. Enables DEC 2048 in-band resize when 2048 resolves - * supported. + * and notify subscribers. `confirmed` distinguishes an explicit DECRPM + * unsupported response from an absent response followed by the DA1 sentinel. + * Enables DEC 2048 in-band resize only after positive confirmation. */ - #resolvePrivateMode(mode: number, supported: boolean): void { + #resolvePrivateMode(mode: number, supported: boolean, confirmed: boolean): void { if (this.#privateModeSupport.has(mode)) return; this.#privateModeSupport.set(mode, supported); for (const cb of this.#privateModeCallbacks) { try { - cb(mode, supported); + cb(mode, supported, confirmed); } catch { // Ignore subscriber errors — capability reporting must not crash input. } diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 885f07af2..ac60d8330 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -83,8 +83,6 @@ const CURSOR_END_NO_SYNC = ""; // coordinates so columns/rows past 223 are reported. const MOUSE_TRACKING_ON = "\x1b[?1000h\x1b[?1003h\x1b[?1006h"; const MOUSE_TRACKING_OFF = "\x1b[?1006l\x1b[?1003l\x1b[?1000l"; -const ALT_SCREEN_ENTER = "\x1b[?1049h"; -const ALT_SCREEN_EXIT = "\x1b[?1049l"; type InputListenerResult = { consume?: boolean; data?: string } | undefined; type InputListener = (data: string) => InputListenerResult; @@ -1028,11 +1026,6 @@ export class TUI extends Container { // `#fullRedrawCount`: these never enter native scrollback and exist only for // the lifetime of the drag. Exposed for tests/diagnostics. #resizeViewportPaintCount = 0; - // During a live resize drag the terminal's normal buffer may reflow full-width - // rows before our repaint lands. Borrow the alternate screen for throwaway - // resize frames so width changes truncate the transient viewport instead of - // pushing wrapped fragments into native scrollback. - #resizeAltActive = false; #stopped = false; // Always-on event-loop lag probe. The high default threshold keeps it quiet; // it only logs `ui.loop-blocked` (with the current loop phase) when a frame @@ -1454,13 +1447,15 @@ export class TUI extends Container { this.#watchdog.start(); this.#ghosttyInitialImageDelayDone = false; this.#ghosttyImageReadyAtMs = this.#renderScheduler.now() + TUI.#GHOSTTY_INITIAL_IMAGE_DELAY_MS; - // A DECRQM report for mode 2026 is authoritative: enable synchronized - // output when the terminal reports support (upgrading conservatively - // defaulted-off hosts like zellij/tmux-master/foot) and disable it when - // the terminal reports it unsupported. An explicit user opt-out/force - // (resolved at construction) still wins, so skip the probe in that case. - this.terminal.onPrivateModeReport?.((mode, supported) => { - if (mode !== 2026) return; + // A confirmed DECRPM report for mode 2026 is authoritative: enable + // synchronized output when the terminal reports support and disable it for + // an explicit unsupported status. A DA1 sentinel without a DECRPM reply is + // inconclusive: many terminals implement synchronized output without + // implementing DECRQM, so retain the statically detected default instead of + // exposing destructive full paints. An explicit user opt-out/force still + // wins, so skip every probe result in that case. + this.terminal.onPrivateModeReport?.((mode, supported, confirmed = true) => { + if (mode !== 2026 || !confirmed) return; if (synchronizedOutputUserOverride() !== null) return; this.#setSynchronizedOutput(supported); }); @@ -1678,11 +1673,6 @@ export class TUI extends Container { } stop(): void { - // Leave the alt buffer first so the teardown cursor math below runs against - // the restored normal screen (which #previousLines still describes). - if (this.#resizeAltActive) { - this.terminal.write(this.#leaveResizeAltSequence()); - } if (this.#altActive) { const enhancementExit = this.#keyboardEnhancementExit(); this.terminal.write(`${MOUSE_TRACKING_OFF}${enhancementExit}\x1b[?1049l`); @@ -2644,6 +2634,7 @@ export class TUI extends Container { // Fullscreen alt-screen short-circuit. While the topmost visible overlay // requests it, borrow the terminal's alternate buffer and paint only the // modal there; the normal screen and all accounting stay untouched. + let deferredAltExit = ""; const wantAlt = this.#wantsAltScreen(); if (wantAlt && !this.#altActive) { // Enhanced keyboard modes can be buffer-local: re-push the active @@ -2661,7 +2652,13 @@ export class TUI extends Container { this.#altEnterHeight = height; } else if (!wantAlt && this.#altActive) { const enhancementExit = this.#keyboardEnhancementExit(); - this.terminal.write(`${MOUSE_TRACKING_OFF}${enhancementExit}\x1b[?1049l`); + const exitSequence = `${MOUSE_TRACKING_OFF}${enhancementExit}\x1b[?1049l`; + // Session replacement can finish while a fullscreen selector is still + // covering the old normal buffer. Keep the overlay visible until the + // replacement is ready, then fuse the buffer restore into that full paint; + // a standalone exit exposes the stale session for one terminal frame. + if (this.#clearScrollbackOnNextRender) deferredAltExit = exitSequence; + else this.terminal.write(exitSequence); setAltScreenActive(false); this.#forgetHardwareCursorState(); this.#altActive = false; @@ -2955,6 +2952,7 @@ export class TUI extends Container { chunkTo, windowTop, cursorTrackingLineCount, + leadingSequence: deferredAltExit, }); this.#committedPrefix = rawFrame.slice(0, chunkTo); this.#committedPrefixAuditRows = Math.min(chunkTo, finalBoundary); @@ -3334,6 +3332,7 @@ export class TUI extends Container { chunkTo: number; windowTop: number; cursorTrackingLineCount: number; + leadingSequence: string; }, ): void { this.#fullRedrawCount += 1; @@ -3371,7 +3370,7 @@ export class TUI extends Container { paintCursorPos = paint.cursorPos; } } - let buffer = this.#paintBeginSequence + this.#leaveResizeAltSequence() + purgeSequence; + let buffer = this.#paintBeginSequence + options.leadingSequence + purgeSequence; if (options.clearScrollback) { // Clear native history without blanking the live viewport first. The // replay below rewrites every visible row from home, including blanks, @@ -3571,34 +3570,15 @@ export class TUI extends Container { return this.terminal.kittyEnableSequence ? "\x1b[ 0) buffer += "\r\n"; buffer += this.#lineRewriteSequence(window[r] ?? "", width); diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index 526c73d6d..d9814dc7b 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -127,6 +127,18 @@ class LegacyKeyboardVirtualTerminal extends VirtualTerminal { } } +class PrivateModeProbeTerminal extends VirtualTerminal { + #callback: ((mode: number, supported: boolean, confirmed?: boolean) => void) | undefined; + + onPrivateModeReport(callback: (mode: number, supported: boolean, confirmed?: boolean) => void): void { + this.#callback = callback; + } + + reportPrivateMode(mode: number, supported: boolean, confirmed: boolean): void { + this.#callback?.(mode, supported, confirmed); + } +} + function rows(prefix: string, count: number): string[] { return Array.from({ length: count }, (_v, i) => `${prefix}${i}`); } @@ -1384,6 +1396,83 @@ describe("TUI terminal-state regressions", () => { setTerminalScreenToScrollback(saved); } }); + + it("keeps destructive paints synchronized when DECRQM is unavailable", async () => { + await withEnvPatch( + { + TERM_FEATURES: "Sy", + PI_NO_SYNC_OUTPUT: undefined, + PI_FORCE_SYNC_OUTPUT: undefined, + PI_TUI_SYNC_OUTPUT: undefined, + }, + async () => { + const term = new PrivateModeProbeTerminal(20, 3); + const component = new MutableLinesComponent(rows("old-", 6)); + const tui = new TUI(term); + tui.addChild(component); + + try { + tui.start(); + await settle(term); + const writes = captureWrites(term); + + // A DA1 sentinel without DECRPM is inconclusive. Terminals such as + // xterm.js can implement synchronized output without implementing + // the query, so the static TERM_FEATURES capability must survive. + term.reportPrivateMode(2026, false, false); + expect(tui.synchronizedOutput).toBe(true); + + component.setLines(rows("resumed-", 8)); + tui.requestRender(true, { clearScrollback: true }); + await settle(term); + + const paint = writes.find(write => write.includes("\x1b[3J")); + expect(paint).toBeDefined(); + expect(paint).toContain("\x1b[?2026h"); + expect(paint).toContain("\x1b[?2026l"); + expect(visible(term)).toEqual(["resumed-5", "resumed-6", "resumed-7"]); + } finally { + tui.stop(); + } + }, + ); + }); + + it("fuses fullscreen overlay exit into a pending session replacement paint", async () => { + const term = new VirtualTerminal(24, 4); + const component = new MutableLinesComponent(rows("old-session-", 8)); + const tui = new TUI(term); + tui.addChild(component); + + try { + tui.start(); + await settle(term); + const overlay = tui.showOverlay(new MutableLinesComponent(["session selector"]), { + width: "100%", + maxHeight: "100%", + fullscreen: true, + }); + await settle(term); + + // Session loading finishes behind the still-visible selector. The forced + // replacement remains pending while the fullscreen path owns the frame. + component.setLines(rows("resumed-", 9)); + tui.requestRender(true, { clearScrollback: true }); + await settle(term); + + const writes = captureWrites(term); + overlay.hide(); + await settle(term); + + const exits = writes.filter(write => write.includes("\x1b[?1049l")); + expect(exits).toHaveLength(1); + expect(exits[0]).toContain("\x1b[3J"); + expect(exits[0]).toContain("resumed-8"); + expect(visible(term)).toEqual(["resumed-5", "resumed-6", "resumed-7", "resumed-8"]); + } finally { + tui.stop(); + } + }); }); describe("scrollback integrity", () => { diff --git a/packages/tui/test/resize-viewport-defer.test.ts b/packages/tui/test/resize-viewport-defer.test.ts index 622f8d212..2ef431906 100644 --- a/packages/tui/test/resize-viewport-defer.test.ts +++ b/packages/tui/test/resize-viewport-defer.test.ts @@ -21,9 +21,8 @@ const NO_MULTIPLEXER_ENV: Record = { TMUX: undefined, STY: undefined, ZELLIJ: undefined, - // Pin terminal identity so the alt-screen fast-path assertions below are - // deterministic even when the suite runs inside Warp (which otherwise takes - // the in-place path — see the Warp describe block at the bottom). + // Pin terminal identity so resize classification is deterministic even when + // the suite runs inside Warp (which takes the debounced in-place path below). TERM_PROGRAM: undefined, PI_TUI_RESIZE_IN_PLACE: undefined, }; @@ -302,7 +301,7 @@ describe("non-multiplexer resize viewport fast path", () => { tui.start(); await scheduler.flushImmediates(term); - // One drag SIGWINCH enters the fast path and borrows the alt screen. + // One drag SIGWINCH enters the viewport-only fast path. term.resize(60, 10); await scheduler.flushImmediates(term); expect(tui.resizeViewportActive).toBe(true); @@ -314,15 +313,13 @@ describe("non-multiplexer resize viewport fast path", () => { // A live block keeps animating mid-drag: a spinner tick / streamed // token fires an ordinary (non-forced) render before the 120ms settle // elapses. It must stay on the viewport fast path. Without the guard it - // falls through to the geometry-rebuild full paint, which leaves the - // borrowed alternate screen (ALT_SCREEN_EXIT) and erases native - // scrollback (ED3) to repaint the whole transcript on the normal screen - // for one frame — the flash — before the next SIGWINCH hides it again. + // falls through to an authoritative full paint and erases/replays the + // whole transcript for one frame before the next resize event. tui.requestRender(); await scheduler.flushOrdinaryRenders(term); - // Still mid-drag, still on the alternate screen: a viewport-only paint, - // no authoritative full redraw, no scrollback erase, no alt-screen exit. + // Still mid-drag: a viewport-only paint, no authoritative full redraw, + // no scrollback erase, and no terminal buffer switch. expect(tui.resizeViewportActive).toBe(true); expect(tui.resizeViewportPaints).toBeGreaterThan(baselinePaints); expect(tui.fullRedraws).toBe(baselineFull); @@ -360,7 +357,7 @@ describe("non-multiplexer resize viewport fast path", () => { }); }); - it("uses the alternate screen during width-drag frames so terminal reflow cannot show wrapped fragments", async () => { + it("repaints the normal screen during width drags without switching buffers", async () => { await withEnvPatch(NO_MULTIPLEXER_ENV, async () => { const term = new VirtualTerminal(40, 10, 1000); const scheduler = new DeferScheduler(); @@ -377,17 +374,16 @@ describe("non-multiplexer resize viewport fast path", () => { const writes = captureWrites(term); - // Shrinking full-width normal-screen rows makes Ghostty reflow them - // into wrapped fragments before the app writes again. The resize - // handler must synchronously switch to the alternate screen and - // repaint the new-width viewport in that same write. + // The resize handler rewrites the new-width viewport synchronously on + // the normal buffer. Borrowing the alternate buffer exposes the saved + // pre-TUI screen when the drag settles on terminals without DEC 2026. term.resize(20, 10); await term.flush(); expect(tui.resizeViewportActive).toBe(true); expect(tui.resizeViewportPaints).toBe(1); const drag = writes.join(""); - expect(drag).toContain(ALT_SCREEN_ENTER); + expect(drag).not.toContain(ALT_SCREEN_ENTER); expect(drag).not.toContain("\x1b[2J"); expect(drag).not.toContain("\x1b[3J"); expect(visible(term)).toEqual(expected); @@ -396,8 +392,8 @@ describe("non-multiplexer resize viewport fast path", () => { await scheduler.flushAll(term); const settle = writes.slice(dragWrites).join(""); - expect(settle).toContain(ALT_SCREEN_EXIT); - expect(settle.indexOf(ALT_SCREEN_EXIT)).toBeLessThan(settle.indexOf("\x1b[3J")); + expect(settle).not.toContain(ALT_SCREEN_EXIT); + expect(settle).toContain("\x1b[3J"); expect(visible(term)).toEqual(expected); } finally { tui.stop(); @@ -420,11 +416,10 @@ describe("non-multiplexer resize viewport fast path", () => { expect(tui.resizeViewportActive).toBe(true); const drag = writes.join(""); - // The drag frame borrows the alternate screen and performs per-row - // self-clearing rewrites there. It must not clear/replay the normal - // screen, so even terminals that expose resize reflow between app - // writes cannot show a blanked normal-screen frame. - expect(drag).toContain(ALT_SCREEN_ENTER); + // The drag frame performs per-row self-clearing rewrites directly on + // the normal screen. It must not clear/replay or switch buffers, because + // either transition is visible on terminals without synchronized output. + expect(drag).not.toContain(ALT_SCREEN_ENTER); expect(drag).not.toContain("\x1b[2J"); expect(drag).not.toContain("\x1b[3J"); expect(drag).toContain("\x1b[H"); @@ -501,7 +496,7 @@ describe("resize repaints in place on terminals that re-report size on alt-scree }); }); - it("PI_TUI_RESIZE_IN_PLACE=0 opts Warp back into the alt-screen fast path", async () => { + it("PI_TUI_RESIZE_IN_PLACE=0 opts Warp into the viewport-only fast path", async () => { await withEnvPatch({ ...WARP_ENV, PI_TUI_RESIZE_IN_PLACE: "0" }, async () => { const term = new VirtualTerminal(40, 10, 1000); const { tui, scheduler } = makeTui(term); @@ -514,7 +509,7 @@ describe("resize repaints in place on terminals that re-report size on alt-scree await scheduler.flushImmediates(term); expect(tui.resizeViewportActive).toBe(true); - expect(writes.join("")).toContain(ALT_SCREEN_ENTER); + expect(writes.join("")).not.toContain(ALT_SCREEN_ENTER); } finally { tui.stop(); } diff --git a/packages/tui/test/terminal-appearance.test.ts b/packages/tui/test/terminal-appearance.test.ts index ea1067c1b..d54f79d79 100644 --- a/packages/tui/test/terminal-appearance.test.ts +++ b/packages/tui/test/terminal-appearance.test.ts @@ -562,13 +562,18 @@ describe("ProcessTerminal DECRQM + in-band resize (DEC 2026/2048)", () => { expect(writes).not.toContain("\x1b[?2048l"); }); - it("falls back to unsupported when the DA1 sentinel beats the DECRPM reply", () => { + it("marks a missing DECRPM response as inconclusive when the DA1 sentinel arrives", () => { const { terminal, reports } = setup(); + const confirmations: boolean[] = []; + terminal.onPrivateModeReport?.((mode, _supported, confirmed) => { + if (mode === 2026) confirmations.push(confirmed ?? true); + }); // Drain keyboard + osc11 sentinels, then 2026's DA1 (no DECRPM arrived). process.stdin.emit("data", "\x1b[?1;2c"); process.stdin.emit("data", "\x1b[?1;2c"); process.stdin.emit("data", "\x1b[?1;2c"); expect(reports).toContainEqual({ mode: 2026, supported: false }); + expect(confirmations).toEqual([false]); terminal.stop(); }); From 4882c9b011aac3dfd4ed6e11d03350eab9d7ba0f Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 18:43:29 +0000 Subject: [PATCH 029/293] fix(auth): mount login dialog input for paste-code fallback URL MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Paste-code OAuth providers (Codex, Anthropic, Gemini CLI, GitLab Duo, Antigravity, Devin) need the user to paste the fallback redirect URL when the loopback callback cannot complete (headless/remote/Windows). The login dialog took focus and cleared the editor but only rendered the auth URL plus a tip pointing at `/login ` — a command only reachable through the now-hidden, unfocused editor. The dialog never mounted an Input, so a pasted URL was silently dropped and login stalled. Route onManualCodeInput through the focused dialog's showManualInput so the paste lands in a visible, submittable field. Make showManualInput idempotent so the OAuth callback retry loop reuses the mounted input instead of stacking duplicate prompts. Fixes #5339 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/components/login-dialog.test.ts | 63 +++++++++++++++++++ .../src/modes/components/login-dialog.ts | 10 ++- .../modes/controllers/selector-controller.ts | 17 +++-- .../src/prompts/system/tan-context-switch.md | 2 +- .../agent-session-prune-persistence.test.ts | 4 +- 6 files changed, 82 insertions(+), 15 deletions(-) create mode 100644 packages/coding-agent/src/modes/components/login-dialog.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9fe2494f3..7cc2bbacb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -32,6 +32,7 @@ ### Fixed +- Fixed `/login` for paste-code providers (Codex, Anthropic, Gemini CLI, GitLab Duo, Antigravity, Devin) dropping the pasted fallback redirect URL: the login dialog captured focus but never mounted an input, and the "complete pairing with `/login `" tip pointed at the hidden, unfocused editor. The dialog now mounts a focused input for the manual code/URL paste ([#5339](https://github.com/can1357/oh-my-pi/issues/5339)). - Fixed `/tan` and `/fork` clones cold-missing the provider prompt cache: the per-turn supersede/useless-result prune rewrote the live context without persisting it, so file-based forks and resume rebuilt a divergent (un-pruned) prefix and re-wrote the entire cache - Fixed `/tan` pinning the clone's prompt-cache key to the parent's session id instead of the parent's effective cache key, dropping shard affinity when the parent was itself a fork or tan - Fixed inconsistent history rendering when toggling the display setting for compacted items diff --git a/packages/coding-agent/src/modes/components/login-dialog.test.ts b/packages/coding-agent/src/modes/components/login-dialog.test.ts new file mode 100644 index 000000000..099fd6f62 --- /dev/null +++ b/packages/coding-agent/src/modes/components/login-dialog.test.ts @@ -0,0 +1,63 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import type { TUI } from "@oh-my-pi/pi-tui"; +import { initTheme } from "../theme/theme"; +import { LoginDialogComponent } from "./login-dialog"; + +const BRACKETED_PASTE_START = "\x1b[200~"; +const BRACKETED_PASTE_END = "\x1b[201~"; + +function bracketedPaste(text: string): string { + return `${BRACKETED_PASTE_START}${text}${BRACKETED_PASTE_END}`; +} + +/** Minimal TUI stub — the dialog only calls requestRender/setFocus. */ +function makeDialog(): LoginDialogComponent { + const tui = { requestRender() {}, setFocus() {} } as unknown as TUI; + return new LoginDialogComponent(tui, "openai-codex", () => {}); +} + +describe("LoginDialogComponent manual code input", () => { + beforeAll(async () => { + await initTheme(); + }); + + it("captures a pasted fallback redirect URL and resolves on submit", async () => { + // Regression for #5339: paste-code providers (Codex) route the fallback + // URL through the focused dialog. Without a mounted input, the paste is + // dropped and login never completes. + const dialog = makeDialog(); + dialog.showAuth("https://auth.openai.com/oauth/authorize?state=abc", "instructions"); + + const pending = dialog.showManualInput("Paste the authorization code:"); + expect(dialog.render(80).join("\n")).toContain("Paste the authorization code"); + + const url = "http://localhost:1455/auth/callback?code=THECODE&state=abc"; + dialog.handleInput(bracketedPaste(url)); + dialog.handleInput("\r"); + + expect(await pending).toBe(url); + }); + + it("reuses the mounted input across re-prompts instead of stacking duplicates", async () => { + // The OAuth callback loop re-invokes onManualCodeInput after an invalid + // paste; the second prompt must not append a duplicate input/hint block. + const dialog = makeDialog(); + dialog.showAuth("https://auth.openai.com/oauth/authorize?state=abc"); + + const first = dialog.showManualInput("Paste the code:"); + dialog.handleInput("garbage"); + dialog.handleInput("\r"); + expect(await first).toBe("garbage"); + + const second = dialog.showManualInput("Paste the code:"); + const rendered = dialog.render(80).join("\n"); + expect(rendered.split("Paste the code:").length - 1).toBe(1); + // A stale value from the first attempt must not leak into the retry. + expect(rendered).not.toContain("garbage"); + + const url = "http://localhost:1455/auth/callback?code=OK&state=abc"; + dialog.handleInput(url); + dialog.handleInput("\r"); + expect(await second).toBe(url); + }); +}); diff --git a/packages/coding-agent/src/modes/components/login-dialog.ts b/packages/coding-agent/src/modes/components/login-dialog.ts index 5048162e8..078bf18e5 100644 --- a/packages/coding-agent/src/modes/components/login-dialog.ts +++ b/packages/coding-agent/src/modes/components/login-dialog.ts @@ -108,12 +108,16 @@ export class LoginDialogComponent extends Container { * Show input for manual code/URL entry (for callback server providers) */ showManualInput(prompt: string): Promise { - this.#contentContainer.addChild(new Spacer(1)); - this.#contentContainer.addChild(new Text(theme.fg("dim", prompt), 1, 0)); + // Invalid pastes re-prompt (the OAuth callback loop calls this again), so + // reuse the already-mounted input instead of stacking duplicate prompt and + // hint lines beneath the dialog. Reset the value so each retry starts clean. if (!this.#contentContainer.children.includes(this.#input)) { + this.#contentContainer.addChild(new Spacer(1)); + this.#contentContainer.addChild(new Text(theme.fg("dim", prompt), 1, 0)); this.#contentContainer.addChild(this.#input); + this.#contentContainer.addChild(new Text(theme.fg("dim", "(Escape to cancel)"), 1, 0)); } - this.#contentContainer.addChild(new Text(theme.fg("dim", "(Escape to cancel)"), 1, 0)); + this.#input.setValue(""); this.#tui.requestRender(); const { promise, resolve, reject } = Promise.withResolvers(); diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index adbd74d38..e2ca07e73 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -82,7 +82,7 @@ import { UserMessageSelectorComponent } from "../components/user-message-selecto import type { SessionObserverRegistry } from "../session-observer-registry"; import { buildCopyTargets } from "../utils/copy-targets"; -const MANUAL_LOGIN_TIP = "Tip: You can complete pairing with /login ."; +const MANUAL_LOGIN_PROMPT = "Paste the authorization code (or full redirect URL), then press Enter:"; export class SelectorController { constructor(private ctx: InteractiveModeContext) {} @@ -1265,7 +1265,6 @@ export class SelectorController { */ async #handleOAuthLogin(providerId: string): Promise { this.ctx.showStatus(`Logging in to ${providerId}…`); - const manualInput = this.ctx.oauthManualInput; const useManualInput = PASTE_CODE_LOGIN_PROVIDERS.has(providerId); let restored = false; const restoreEditor = () => { @@ -1293,16 +1292,19 @@ export class SelectorController { // The dialog renders the full URL (SSH-safe copy target) and // opens the browser best-effort. dialog.showAuth(info.url, info.instructions, info.launchUrl); - if (useManualInput) { - dialog.showProgress(MANUAL_LOGIN_TIP); - } }, onPrompt: (prompt: { message: string; placeholder?: string }) => dialog.showPrompt(prompt.message, prompt.placeholder), onProgress: (message: string) => { dialog.showProgress(message); }, - onManualCodeInput: useManualInput ? () => manualInput.waitForInput(providerId) : undefined, + // Paste-code providers (e.g. Codex) may need the user to paste the + // fallback redirect URL when the loopback callback can't complete + // (headless/remote/Windows). Mount a focused input in the dialog so + // the paste lands somewhere the OAuth flow consumes — the hidden + // editor's `/login ` path is unreachable while the dialog holds + // focus (#5339). + onManualCodeInput: useManualInput ? () => dialog.showManualInput(MANUAL_LOGIN_PROMPT) : undefined, }); this.ctx.session.modelRegistry.refreshInBackground(); const block = new TranscriptBlock(); @@ -1321,9 +1323,6 @@ export class SelectorController { this.ctx.showError(`Login failed: ${error instanceof Error ? error.message : String(error)}`); return false; } finally { - if (useManualInput) { - manualInput.clear(`Manual OAuth input cleared for ${providerId}`); - } restoreEditor(); } } diff --git a/packages/coding-agent/src/prompts/system/tan-context-switch.md b/packages/coding-agent/src/prompts/system/tan-context-switch.md index 55468b15a..88cd57291 100644 --- a/packages/coding-agent/src/prompts/system/tan-context-switch.md +++ b/packages/coding-agent/src/prompts/system/tan-context-switch.md @@ -1,5 +1,5 @@ -The conversation above belongs to your parent session. +The conversation above belongs to your parent session. You are a fork created solely to handle the user's request below. Your parent agent is still working on the original task — that responsibility is diff --git a/packages/coding-agent/test/agent-session-prune-persistence.test.ts b/packages/coding-agent/test/agent-session-prune-persistence.test.ts index 3b78c3657..438f0de4e 100644 --- a/packages/coding-agent/test/agent-session-prune-persistence.test.ts +++ b/packages/coding-agent/test/agent-session-prune-persistence.test.ts @@ -114,7 +114,7 @@ describe("AgentSession per-turn prune persistence", () => { const message = session.agent.state.messages.find( candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID, ); - if (!message || message.role !== "toolResult" || !Array.isArray(message.content)) { + if (message?.role !== "toolResult" || !Array.isArray(message.content)) { throw new Error("Expected the seeded tool result in live agent state"); } const text = message.content.find(block => block.type === "text"); @@ -156,7 +156,7 @@ describe("AgentSession per-turn prune persistence", () => { const rebuilt = reloaded .buildSessionContext() .messages.find(candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID); - if (!rebuilt || rebuilt.role !== "toolResult" || !Array.isArray(rebuilt.content)) { + if (rebuilt?.role !== "toolResult" || !Array.isArray(rebuilt.content)) { throw new Error("Expected the seeded tool result in the from-disk rebuild"); } const rebuiltText = rebuilt.content.find(block => block.type === "text"); From bcca907e59eaac3a8daf0aba0c5a2cd54363c67b Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 18:45:55 +0000 Subject: [PATCH 030/293] fix(cli): aliased clear to new session Added /clear as a /new alias so exact slash completion outranks /autoresearch description matches. Fixes #5349 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/prompts/system/tan-context-switch.md | 2 +- .../src/slash-commands/builtin-registry.ts | 1 + .../test/agent-session-prune-persistence.test.ts | 4 ++-- .../test/slash-commands/clear-alias.test.ts | 16 ++++++++++++++++ 5 files changed, 21 insertions(+), 3 deletions(-) create mode 100644 packages/coding-agent/test/slash-commands/clear-alias.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9fe2494f3..1493b9679 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -32,6 +32,7 @@ ### Fixed +- Fixed `/clear` autocomplete selecting `/autoresearch`; `/clear` now starts a new session as an alias for `/new` ([#5349](https://github.com/can1357/oh-my-pi/issues/5349)) - Fixed `/tan` and `/fork` clones cold-missing the provider prompt cache: the per-turn supersede/useless-result prune rewrote the live context without persisting it, so file-based forks and resume rebuilt a divergent (un-pruned) prefix and re-wrote the entire cache - Fixed `/tan` pinning the clone's prompt-cache key to the parent's session id instead of the parent's effective cache key, dropping shard affinity when the parent was itself a fork or tan - Fixed inconsistent history rendering when toggling the display setting for compacted items diff --git a/packages/coding-agent/src/prompts/system/tan-context-switch.md b/packages/coding-agent/src/prompts/system/tan-context-switch.md index 55468b15a..88cd57291 100644 --- a/packages/coding-agent/src/prompts/system/tan-context-switch.md +++ b/packages/coding-agent/src/prompts/system/tan-context-switch.md @@ -1,5 +1,5 @@ -The conversation above belongs to your parent session. +The conversation above belongs to your parent session. You are a fork created solely to handle the user's request below. Your parent agent is still working on the original task — that responsibility is diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 72ec7cfb9..ba06a6968 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -1384,6 +1384,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ }, { name: "new", + aliases: ["clear"], description: "Start a new session", handleTui: async (_command, runtime) => { runtime.ctx.editor.setText(""); diff --git a/packages/coding-agent/test/agent-session-prune-persistence.test.ts b/packages/coding-agent/test/agent-session-prune-persistence.test.ts index 3b78c3657..438f0de4e 100644 --- a/packages/coding-agent/test/agent-session-prune-persistence.test.ts +++ b/packages/coding-agent/test/agent-session-prune-persistence.test.ts @@ -114,7 +114,7 @@ describe("AgentSession per-turn prune persistence", () => { const message = session.agent.state.messages.find( candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID, ); - if (!message || message.role !== "toolResult" || !Array.isArray(message.content)) { + if (message?.role !== "toolResult" || !Array.isArray(message.content)) { throw new Error("Expected the seeded tool result in live agent state"); } const text = message.content.find(block => block.type === "text"); @@ -156,7 +156,7 @@ describe("AgentSession per-turn prune persistence", () => { const rebuilt = reloaded .buildSessionContext() .messages.find(candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID); - if (!rebuilt || rebuilt.role !== "toolResult" || !Array.isArray(rebuilt.content)) { + if (rebuilt?.role !== "toolResult" || !Array.isArray(rebuilt.content)) { throw new Error("Expected the seeded tool result in the from-disk rebuild"); } const rebuiltText = rebuilt.content.find(block => block.type === "text"); diff --git a/packages/coding-agent/test/slash-commands/clear-alias.test.ts b/packages/coding-agent/test/slash-commands/clear-alias.test.ts new file mode 100644 index 000000000..5964a3af7 --- /dev/null +++ b/packages/coding-agent/test/slash-commands/clear-alias.test.ts @@ -0,0 +1,16 @@ +import { describe, expect, it } from "bun:test"; +import { BUILTIN_SLASH_COMMANDS } from "@oh-my-pi/pi-coding-agent/slash-commands/builtin-registry"; +import { CombinedAutocompleteProvider } from "@oh-my-pi/pi-tui/autocomplete"; + +describe("/clear slash command alias", () => { + it("ranks the new-session action above fuzzy description matches", async () => { + const provider = new CombinedAutocompleteProvider([...BUILTIN_SLASH_COMMANDS], process.cwd()); + + const suggestions = await provider.getSuggestions(["/clear"], 0, 6); + + expect(suggestions?.items[0]).toMatchObject({ + value: "clear", + description: "Start a new session", + }); + }); +}); From 86824c94e96ed9961db1240acca11820529c9343 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 19:01:47 +0000 Subject: [PATCH 031/293] fix(ttsr): registered rules with inline regex flags and malformed scope Rules whose condition led with a PCRE-style inline flag group (e.g. `(?i)`) never registered: `new RegExp("(?i)...")` throws in Bun/JS, so the condition failed to compile and `TtsrManager.addRule` dropped the rule as having zero usable conditions. - Add `compileRuleCondition` in capability/rule.ts translating a leading `(?i)`/`(?m)`/`(?s)` group into native RegExp flags; wire it into the TtsrManager and both ttsr-cli compile sites. - Strip surrounding quotes from scope tokens so a malformed `scope: "text","thinking"` recovers to canonical `text`/`thinking`. - Reparse each value in parseFrontmatter's YAML fallback so one bad line can't leave sibling values wrapped in literal quotes. Fixes #4796 --- packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/capability/rule.ts | 39 ++++++++++- packages/coding-agent/src/cli/ttsr-cli.ts | 6 +- packages/coding-agent/src/export/ttsr.ts | 4 +- .../test/ttsr-inline-flags-scope.test.ts | 66 +++++++++++++++++++ packages/utils/CHANGELOG.md | 4 ++ packages/utils/src/frontmatter.ts | 19 +++++- packages/utils/test/frontmatter.test.ts | 22 +++++++ 8 files changed, 155 insertions(+), 9 deletions(-) create mode 100644 packages/coding-agent/test/ttsr-inline-flags-scope.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3a03a1977..c4621e80b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,10 @@ - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. +### Fixed + +- Fixed TTSR rules with a leading `(?i)`/`(?m)`/`(?s)` inline regex flag never registering: `new RegExp("(?i)...")` throws in Bun/JS, so the condition failed to compile and the rule was silently dropped. Leading inline flag groups are now translated to native `RegExp` flags. Also recover `scope` tokens and sibling values from malformed frontmatter (e.g. `scope: "text","thinking"`), which previously left literal quotes on the parsed values ([#4796](https://github.com/can1357/oh-my-pi/issues/4796)). + ## [16.3.11] - 2026-07-06 ### Changed diff --git a/packages/coding-agent/src/capability/rule.ts b/packages/coding-agent/src/capability/rule.ts index cdce087d3..831b8468d 100644 --- a/packages/coding-agent/src/capability/rule.ts +++ b/packages/coding-agent/src/capability/rule.ts @@ -158,7 +158,18 @@ function normalizeScopeField(value: unknown): string[] | undefined { return undefined; } - const tokens = normalized.flatMap(splitScopeTokens).filter(item => item.length > 0); + const tokens = normalized + .flatMap(splitScopeTokens) + .map(token => { + // Tolerate malformed frontmatter (e.g. `scope: "text","thinking"`) whose + // YAML-fallback parse leaves per-token quotes intact (issue #4796). + const quote = token[0]; + if (token.length >= 2 && (quote === '"' || quote === "'") && token[token.length - 1] === quote) { + return token.slice(1, -1).trim(); + } + return token; + }) + .filter(item => item.length > 0); if (tokens.length === 0) { return undefined; } @@ -226,6 +237,32 @@ export function parseRuleConditionAndScope( }; } +/** Leading PCRE-style inline flag group, e.g. `(?i)` or `(?ims)`. */ +const INLINE_FLAG_PREFIX = /^\(\?([a-z]+)\)/; + +/** Inline flags that map cleanly onto native `RegExp` flags. */ +const TRANSLATABLE_INLINE_FLAGS = /^[ims]+$/; + +/** + * Compile a rule `condition` into a `RegExp`, translating a leading PCRE-style + * inline flag group into native `RegExp` flags. + * + * JS/Bun `RegExp` rejects inline flag prefixes such as `(?i)`, so a rule written + * `condition: "(?i)pre.existing"` would otherwise throw at compile time and be + * silently dropped (see issue #4796). Only a *leading* group of `i`/`m`/`s` + * flags is translated; anything else — mid-pattern groups, unsupported flags — + * is passed through verbatim so the native error still surfaces for genuinely + * invalid patterns. + */ +export function compileRuleCondition(pattern: string): RegExp { + const match = INLINE_FLAG_PREFIX.exec(pattern); + if (match && TRANSLATABLE_INLINE_FLAGS.test(match[1])) { + const flags = Array.from(new Set(match[1])).join(""); + return new RegExp(pattern.slice(match[0].length), flags); + } + return new RegExp(pattern); +} + let activeRules: readonly Rule[] = []; /** diff --git a/packages/coding-agent/src/cli/ttsr-cli.ts b/packages/coding-agent/src/cli/ttsr-cli.ts index 108a05004..c20629bf7 100644 --- a/packages/coding-agent/src/cli/ttsr-cli.ts +++ b/packages/coding-agent/src/cli/ttsr-cli.ts @@ -15,7 +15,7 @@ import * as path from "node:path"; import { AstMatchStrictness, astMatch, FileType, type GlobMatch, glob } from "@oh-my-pi/pi-natives"; import { getProjectDir } from "@oh-my-pi/pi-utils/dirs"; import chalk from "chalk"; -import { BUILTIN_DEFAULTS_PROVIDER_ID, type Rule, ruleCapability } from "../capability/rule"; +import { BUILTIN_DEFAULTS_PROVIDER_ID, compileRuleCondition, type Rule, ruleCapability } from "../capability/rule"; import { bucketRules } from "../capability/rule-buckets"; import { Settings } from "../config/settings"; import type { TtsrSettings } from "../config/settings-schema"; @@ -173,7 +173,7 @@ async function regexMatches(rule: Rule, snippet: string): Promise { const out: string[] = []; for (const pattern of rule.condition ?? []) { try { - if (new RegExp(pattern).test(snippet)) out.push(pattern); + if (compileRuleCondition(pattern).test(snippet)) out.push(pattern); } catch { // Invalid regex — skip; the manager already warned at registration. } @@ -569,7 +569,7 @@ function compileScanRulePlans(rules: Rule[]): ScanRulePlan[] { const regexConditions: ScanRegexCondition[] = []; for (const pattern of rule.condition ?? []) { try { - regexConditions.push({ pattern, regex: new RegExp(pattern) }); + regexConditions.push({ pattern, regex: compileRuleCondition(pattern) }); } catch { // Same behavior as TtsrManager: invalid regex conditions are unusable. } diff --git a/packages/coding-agent/src/export/ttsr.ts b/packages/coding-agent/src/export/ttsr.ts index 27b885c40..ad3ba23ab 100644 --- a/packages/coding-agent/src/export/ttsr.ts +++ b/packages/coding-agent/src/export/ttsr.ts @@ -8,7 +8,7 @@ import * as path from "node:path"; import { AstMatchStrictness, astMatch } from "@oh-my-pi/pi-natives"; import { logger } from "@oh-my-pi/pi-utils"; -import type { Rule } from "../capability/rule"; +import { compileRuleCondition, type Rule } from "../capability/rule"; import type { TtsrSettings } from "../config/settings"; export type TtsrMatchSource = "text" | "thinking" | "tool"; @@ -103,7 +103,7 @@ export class TtsrManager { const compiled: RegExp[] = []; for (const pattern of rule.condition ?? []) { try { - compiled.push(new RegExp(pattern)); + compiled.push(compileRuleCondition(pattern)); } catch (error) { logger.warn("TTSR condition has invalid regex pattern, skipping condition", { ruleName: rule.name, diff --git a/packages/coding-agent/test/ttsr-inline-flags-scope.test.ts b/packages/coding-agent/test/ttsr-inline-flags-scope.test.ts new file mode 100644 index 000000000..aeffd8d73 --- /dev/null +++ b/packages/coding-agent/test/ttsr-inline-flags-scope.test.ts @@ -0,0 +1,66 @@ +import { describe, expect, it } from "bun:test"; +import { compileRuleCondition } from "@oh-my-pi/pi-coding-agent/capability/rule"; +import { buildRuleFromMarkdown, createSourceMeta } from "@oh-my-pi/pi-coding-agent/discovery/helpers"; +import { TtsrManager } from "@oh-my-pi/pi-coding-agent/export/ttsr"; + +/** + * Regression coverage for issue #4796: a rule with a leading `(?i)` inline regex + * flag and (separately) malformed `scope` frontmatter silently failed to + * register, so it could never fire. + */ +describe("TTSR inline flags + scope quoting (#4796)", () => { + it("translates leading (?i) into a case-insensitive RegExp", () => { + const regex = compileRuleCondition("(?i)pre.existing"); + expect(regex.flags).toBe("i"); + expect(regex.test("These are Pre-existing failures")).toBe(true); + }); + + it("passes through patterns without a leading inline flag group verbatim", () => { + const regex = compileRuleCondition("pre.existing"); + expect(regex.flags).toBe(""); + expect(regex.test("pre-existing")).toBe(true); + expect(regex.test("PRE-EXISTING")).toBe(false); + }); + + it("does not treat a mid-pattern (?...) group as an inline flag prefix", () => { + // `(?:...)` is a non-capturing group, not an inline flag directive. + const regex = compileRuleCondition("foo(?:bar)"); + expect(regex.flags).toBe(""); + expect(regex.test("foobar")).toBe(true); + }); + + it("registers and fires the reporter's exact rule end-to-end", () => { + // Reporter's frontmatter verbatim: leading (?i) condition + malformed + // `scope: "text","thinking"` (not valid YAML, forces the fallback path). + const content = [ + "---", + "name: fix-failures-now", + "description: prohibits pre-existing classification.", + 'condition: "(?i)(pre.existing|also fails on master|check.*master.*first)"', + 'scope: "text","thinking"', + "---", + "body", + ].join("\n"); + + const source = createSourceMeta("test", "fix-failures-now.md", "project"); + const rule = buildRuleFromMarkdown("fix-failures-now.md", content, "fix-failures-now.md", source); + + // Malformed scope recovers to canonical tokens (no literal quotes). + expect(rule.scope).toEqual(["text", "thinking"]); + // Condition survives the fallback without literal surrounding quotes. + expect(rule.condition).toEqual(["(?i)(pre.existing|also fails on master|check.*master.*first)"]); + + const manager = new TtsrManager(); + expect(manager.addRule(rule)).toBe(true); + + expect( + manager + .checkSnapshot("The CI failure was 4 pre-existing GPS map VR mismatches", { source: "thinking" }) + .map(r => r.name), + ).toEqual(["fix-failures-now"]); + + expect( + manager.checkSnapshot("Everything also fails on master anyway", { source: "text" }).map(r => r.name), + ).toEqual(["fix-failures-now"]); + }); +}); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 80ebafb3a..fae865822 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `parseFrontmatter`'s malformed-YAML fallback corrupting sibling values: one unparseable line (e.g. `scope: "text","thinking"`) forced every value through a raw key/value split that kept literal quotes. Each value is now reparsed independently as YAML, falling back to the raw trimmed string only for the lines that genuinely don't parse ([#4796](https://github.com/can1357/oh-my-pi/issues/4796)). + ## [16.3.10] - 2026-07-06 ### Added diff --git a/packages/utils/src/frontmatter.ts b/packages/utils/src/frontmatter.ts index 061ba87a3..8178172f1 100644 --- a/packages/utils/src/frontmatter.ts +++ b/packages/utils/src/frontmatter.ts @@ -149,12 +149,25 @@ export function parseFrontmatter( throw err; } - // Simple YAML parsing - just key: value pairs + // Simple key: value fallback. Reparse each value on its own so one + // malformed line (e.g. `scope: "text","thinking"`) can't leave sibling + // values wrapped in literal quotes; values that don't parse as YAML fall + // back to the raw trimmed string (issue #4796). for (const line of metadata.split("\n")) { const match = line.match(/^([\w-]+):\s*(.*)$/); - if (match) { - frontmatter[match[1]] = match[2].trim(); + if (!match) continue; + const raw = match[2].trim(); + let value: unknown = raw; + if (raw.length > 0) { + try { + const parsed = YAML.parse(raw); + if (parsed !== null && typeof parsed !== "object") value = parsed; + else if (Array.isArray(parsed)) value = parsed; + } catch { + // keep the raw string + } } + frontmatter[match[1]] = value; } return { frontmatter: normalizeKeys(frontmatter) as Record, body }; diff --git a/packages/utils/test/frontmatter.test.ts b/packages/utils/test/frontmatter.test.ts index a3adc4510..400b115a7 100644 --- a/packages/utils/test/frontmatter.test.ts +++ b/packages/utils/test/frontmatter.test.ts @@ -43,4 +43,26 @@ Body content`; expect.objectContaining({ err: expect.stringContaining("broken.md") }), ); }); + + it("reparses each fallback value so one malformed line can't corrupt its siblings", () => { + // `scope: "text","thinking"` is not valid YAML, forcing the line-by-line + // fallback. The sibling `condition` value must not inherit literal quotes, + // and `enabled` must reparse to a boolean (issue #4796). + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + const content = `--- +condition: "(?i)pre.existing" +scope: "text","thinking" +enabled: true +--- +Body`; + + const result = parseFrontmatter(content, { source: "rule.md" }); + + expect(result.frontmatter.condition).toBe("(?i)pre.existing"); + expect(result.frontmatter.enabled).toBe(true); + // The unrecoverable line survives as its raw trimmed string. + expect(result.frontmatter.scope).toBe('"text","thinking"'); + expect(result.body).toBe("Body"); + expect(warnSpy).toHaveBeenCalled(); + }); }); From cad30f8c29df17db88c50eab57f4153a30aaf9d5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 19:24:35 +0000 Subject: [PATCH 032/293] fix(review): fall back to per-file API when PR diff exceeds 20k lines GitHub rejects the aggregate PR diff endpoint with HTTP 406 once the diff exceeds 20,000 lines, which made `fetchPrDiffFresh` throw and aborted the entire /review workflow. Detect the 406 (diff-too-large) specifically and fall back to the paginated per-file endpoint, reassembling a synthetic unified diff. Files whose patch is omitted (binary or too large) stay visible with an explicit marker instead of being dropped. Fixes #5350 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/prompts/system/tan-context-switch.md | 2 +- packages/coding-agent/src/tools/gh.ts | 121 +++++++++++++++++- .../agent-session-prune-persistence.test.ts | 4 +- packages/coding-agent/test/tools/gh.test.ts | 81 ++++++++++++ 5 files changed, 204 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9fe2494f3..68a4c885b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -32,6 +32,7 @@ ### Fixed +- Fixed `/review` aborting entirely when GitHub rejects a pull request's aggregate diff with HTTP 406 for exceeding the 20,000-line limit: `gh pr diff` now falls back to the paginated per-file endpoint (`/repos/{owner}/{repo}/pulls/{n}/files`) and reassembles a synthetic unified diff, keeping files with omitted (binary/too-large) patches visible with an explicit marker ([#5350](https://github.com/can1357/oh-my-pi/issues/5350)) - Fixed `/tan` and `/fork` clones cold-missing the provider prompt cache: the per-turn supersede/useless-result prune rewrote the live context without persisting it, so file-based forks and resume rebuilt a divergent (un-pruned) prefix and re-wrote the entire cache - Fixed `/tan` pinning the clone's prompt-cache key to the parent's session id instead of the parent's effective cache key, dropping shard affinity when the parent was itself a fork or tan - Fixed inconsistent history rendering when toggling the display setting for compacted items diff --git a/packages/coding-agent/src/prompts/system/tan-context-switch.md b/packages/coding-agent/src/prompts/system/tan-context-switch.md index 55468b15a..88cd57291 100644 --- a/packages/coding-agent/src/prompts/system/tan-context-switch.md +++ b/packages/coding-agent/src/prompts/system/tan-context-switch.md @@ -1,5 +1,5 @@ -The conversation above belongs to your parent session. +The conversation above belongs to your parent session. You are a fork created solely to handle the user's request below. Your parent agent is still working on the original task — that responsibility is diff --git a/packages/coding-agent/src/tools/gh.ts b/packages/coding-agent/src/tools/gh.ts index a4c016b60..f88865ff0 100644 --- a/packages/coding-agent/src/tools/gh.ts +++ b/packages/coding-agent/src/tools/gh.ts @@ -10,7 +10,7 @@ import type { ToolApprovalDecision, } from "@oh-my-pi/pi-agent-core"; -import { getWorktreeDir, hashPath, isEnoent, prompt, untilAborted } from "@oh-my-pi/pi-utils"; +import { getWorktreeDir, hashPath, isEnoent, logger, prompt, untilAborted } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import type { Settings } from "../config/settings"; import githubDescription from "../prompts/tools/github.md" with { type: "text" }; @@ -239,6 +239,7 @@ const RUN_WATCH_TAIL_DEFAULT = 15; const RUN_WATCH_TAIL_MAX = 200; const REVIEW_COMMENTS_PAGE_SIZE = 100; const RUN_JOBS_PAGE_SIZE = 100; +const PR_DIFF_FILES_PAGE_SIZE = 100; const PR_URL_PATTERN = /^https:\/\/github\.com\/([^/]+\/[^/]+)\/pull\/(\d+)(?:\/.*)?$/; const ISSUE_URL_PATTERN = /^https:\/\/github\.com\/([^/]+\/[^/]+)\/issues\/(\d+)(?:\/.*)?$/; const RUN_URL_PATTERN = /^https:\/\/github\.com\/([^/]+\/[^/]+)\/actions\/runs\/(\d+)(?:\/.*)?$/; @@ -2931,6 +2932,111 @@ function parsePrDiffSection(section: string, startOffset: number, endOffset: num return file; } +/** + * A single entry from `GET /repos/{owner}/{repo}/pulls/{n}/files`. `patch` is + * absent for binary files and for individual file diffs GitHub deems too large + * to render. + */ +interface GhPrFileApi { + filename?: string; + previous_filename?: string; + status?: string; + additions?: number; + deletions?: number; + patch?: string; +} + +/** + * GitHub rejects the aggregate PR diff endpoint with HTTP 406 once the diff + * exceeds 20,000 lines. Detect that specific failure so the caller can fall + * back to the per-file endpoint instead of aborting the whole review. + */ +function isPrDiffTooLargeError(err: unknown): boolean { + const message = err instanceof Error ? err.message : String(err); + return ( + /\bHTTP 406\b/.test(message) || + /exceeded the maximum number of lines/i.test(message) || + /\btoo_large\b/.test(message) + ); +} + +/** + * Reconstruct a `diff --git` section from a single files-API entry. The API's + * `patch` field carries only the hunk body, so the `diff --git`/`---`/`+++` + * headers are synthesized to match `gh pr diff` output — this keeps + * {@link parsePrUnifiedDiff} and the review parser producing identical section + * boundaries and byte offsets. Files whose `patch` is omitted (binary or + * too-large) stay visible with an explicit marker rather than being dropped. + */ +function buildSyntheticDiffSection(file: GhPrFileApi): string | undefined { + const newPath = file.filename; + if (!newPath) return undefined; + const status = file.status ?? "modified"; + const oldPath = file.previous_filename ?? newPath; + const lines: string[] = [`diff --git a/${oldPath} b/${newPath}`]; + if (status === "added") { + lines.push("new file mode 100644"); + } else if (status === "removed") { + lines.push("deleted file mode 100644"); + } else if (status === "renamed" || file.previous_filename) { + lines.push(`rename from ${oldPath}`, `rename to ${newPath}`); + } + if (typeof file.patch === "string" && file.patch.length > 0) { + lines.push(status === "added" ? "--- /dev/null" : `--- a/${oldPath}`); + lines.push(status === "removed" ? "+++ /dev/null" : `+++ b/${newPath}`); + lines.push(file.patch); + } else { + lines.push( + `* patch unavailable (binary or too large); additions ${file.additions ?? 0}, deletions ${file.deletions ?? 0}`, + ); + } + return lines.join("\n"); +} + +/** + * Fallback PR diff retrieval via the paginated per-file endpoint, used when the + * aggregate `gh pr diff` is rejected for exceeding GitHub's 20,000-line limit. + * The per-file patches are not subject to that aggregate cap, so even very + * large PRs can be reassembled into a synthetic unified diff. + */ +async function fetchPrDiffViaFilesApi( + cwd: string, + repo: string, + number: number, + signal: AbortSignal | undefined, +): Promise { + const sections: string[] = []; + let page = 1; + while (true) { + const response = await git.github.json( + cwd, + [ + "api", + "--method", + "GET", + `/repos/${repo}/pulls/${number}/files`, + "-F", + `per_page=${PR_DIFF_FILES_PAGE_SIZE}`, + "-F", + `page=${page}`, + ], + signal, + { repoProvided: true }, + ); + for (const file of response) { + const section = buildSyntheticDiffSection(file); + if (section) sections.push(section); + } + if (response.length < PR_DIFF_FILES_PAGE_SIZE) { + break; + } + page += 1; + } + // Trailing newline mirrors `gh pr diff` so downstream parsers splitting on + // `^diff --git ` see identical boundaries. + return sections.length > 0 ? `${sections.join("\n")}\n` : ""; +} + async function fetchPrDiffFresh( cwd: string, repo: string, @@ -2939,7 +3045,18 @@ async function fetchPrDiffFresh( ): Promise<{ rendered: string; sourceUrl: string | undefined; payload: PrDiffPayload }> { const args = ["pr", "diff", String(number), "--color", "never"]; appendRepoFlag(args, repo, String(number)); - const text = await git.github.text(cwd, args, signal, { repoProvided: true, trimOutput: false }); + let text: string; + try { + text = await git.github.text(cwd, args, signal, { repoProvided: true, trimOutput: false }); + } catch (err) { + if (!isPrDiffTooLargeError(err)) throw err; + logger.debug("gh pr diff exceeded GitHub's aggregate line limit; falling back to per-file API", { + repo, + number, + err: String(err), + }); + text = await fetchPrDiffViaFilesApi(cwd, repo, number, signal); + } const payload = parsePrUnifiedDiff(text); // `rendered` already carries the verbatim diff; blank the payload copy so // the cache row stores a potentially huge diff once instead of twice. diff --git a/packages/coding-agent/test/agent-session-prune-persistence.test.ts b/packages/coding-agent/test/agent-session-prune-persistence.test.ts index 3b78c3657..438f0de4e 100644 --- a/packages/coding-agent/test/agent-session-prune-persistence.test.ts +++ b/packages/coding-agent/test/agent-session-prune-persistence.test.ts @@ -114,7 +114,7 @@ describe("AgentSession per-turn prune persistence", () => { const message = session.agent.state.messages.find( candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID, ); - if (!message || message.role !== "toolResult" || !Array.isArray(message.content)) { + if (message?.role !== "toolResult" || !Array.isArray(message.content)) { throw new Error("Expected the seeded tool result in live agent state"); } const text = message.content.find(block => block.type === "text"); @@ -156,7 +156,7 @@ describe("AgentSession per-turn prune persistence", () => { const rebuilt = reloaded .buildSessionContext() .messages.find(candidate => candidate.role === "toolResult" && candidate.toolCallId === BIG_CALL_ID); - if (!rebuilt || rebuilt.role !== "toolResult" || !Array.isArray(rebuilt.content)) { + if (rebuilt?.role !== "toolResult" || !Array.isArray(rebuilt.content)) { throw new Error("Expected the seeded tool result in the from-disk rebuild"); } const rebuiltText = rebuilt.content.find(block => block.type === "text"); diff --git a/packages/coding-agent/test/tools/gh.test.ts b/packages/coding-agent/test/tools/gh.test.ts index ddb885130..da0cd5a8c 100644 --- a/packages/coding-agent/test/tools/gh.test.ts +++ b/packages/coding-agent/test/tools/gh.test.ts @@ -8,6 +8,7 @@ import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { buildSearchDateQualifier, GithubTool, + getOrFetchPrDiff, parsePrUnifiedDiff, parseSearchDateBound, resolveDefaultRepoMemoized, @@ -273,6 +274,86 @@ describe("parsePrUnifiedDiff", () => { }); }); +describe("getOrFetchPrDiff diff-too-large fallback", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + function http406(): Error { + return new Error( + "could not find pull request diff: HTTP 406: Sorry, the diff exceeded the maximum number of lines (20000)", + ); + } + + it("reassembles a unified diff from the per-file API when gh pr diff returns HTTP 406", async () => { + vi.spyOn(git.github, "text").mockRejectedValue(http406()); + const jsonSpy = vi + .spyOn(git.github, "json") + .mockResolvedValueOnce([ + { + filename: "src/big.ts", + status: "modified", + additions: 2, + deletions: 1, + patch: "@@ -1,2 +1,3 @@\n-old\n+new one\n+new two", + }, + { + filename: "src/added.ts", + status: "added", + additions: 1, + deletions: 0, + patch: "@@ -0,0 +1 @@\n+brand new", + }, + ] as unknown as never) + .mockResolvedValueOnce([] as unknown as never); + + const result = await getOrFetchPrDiff({ + cwd: "/tmp/test", + repo: "owner/repo", + number: 79, + cacheAuthKey: null, + }); + + expect(result.payload.files.map(f => f.path)).toEqual(["src/big.ts", "src/added.ts"]); + expect(result.payload.files[0]).toMatchObject({ additions: 2, deletions: 1, changeType: "modified" }); + expect(result.payload.files[1]).toMatchObject({ additions: 1, deletions: 0, changeType: "added" }); + // The reassembled diff parses through parsePrUnifiedDiff identically. + expect(result.payload.unified).toContain("diff --git a/src/big.ts b/src/big.ts"); + expect(result.payload.unified).toContain("new file mode"); + // The files endpoint should have been hit; the first arg after `api` is GET. + expect(jsonSpy.mock.calls[0]?.[1]).toContain("/repos/owner/repo/pulls/79/files"); + }); + + it("keeps files with omitted patches visible instead of dropping them", async () => { + vi.spyOn(git.github, "text").mockRejectedValue(http406()); + vi.spyOn(git.github, "json") + .mockResolvedValueOnce([ + { filename: "assets/logo.png", status: "modified", additions: 0, deletions: 0 }, + ] as unknown as never) + .mockResolvedValueOnce([] as unknown as never); + + const result = await getOrFetchPrDiff({ + cwd: "/tmp/test", + repo: "owner/repo", + number: 80, + cacheAuthKey: null, + }); + + expect(result.payload.files.map(f => f.path)).toEqual(["assets/logo.png"]); + expect(result.payload.unified).toContain("patch unavailable"); + }); + + it("propagates non-406 errors without hitting the files endpoint", async () => { + vi.spyOn(git.github, "text").mockRejectedValue(new Error("authentication required")); + const jsonSpy = vi.spyOn(git.github, "json"); + + await expect( + getOrFetchPrDiff({ cwd: "/tmp/test", repo: "owner/repo", number: 81, cacheAuthKey: null }), + ).rejects.toThrow("authentication required"); + expect(jsonSpy).not.toHaveBeenCalled(); + }); +}); + describe("github tool", () => { beforeAll(async () => { prFixtureTemplate = await buildPrFixtureTemplate(); From f029e536ba5166561bb4c39d28e59b9cf4d4ad85 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 19:50:39 +0000 Subject: [PATCH 033/293] fix(ai): honored proxies for codex websockets Passed provider-specific PI_PROXY settings and standard HTTPS/ALL proxy variables to Bun WebSocket connections while preserving NO_PROXY bypasses. Fixes #5384 --- packages/ai/CHANGELOG.md | 4 + .../src/providers/openai-codex-responses.ts | 19 ++- packages/ai/test/openai-codex-stream.test.ts | 119 +++++++++++++++++- 3 files changed, 140 insertions(+), 2 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d3c2273a8..26ae7c0d3 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed OpenAI Codex WebSocket connections ignoring `PI_PROXY`, provider-specific proxy settings, and standard HTTPS/ALL proxy variables ([#5384](https://github.com/can1357/oh-my-pi/issues/5384)). + ## [16.5.0] - 2026-07-13 ### Added diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 64458dfe9..e762b5dd8 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -60,6 +60,7 @@ import { getOpenAIStreamIdleTimeoutMs, iterateWithIdleTimeout, } from "../utils/idle-iterator"; +import { getProxyForProvider, shouldBypassProxy } from "../utils/proxy"; import { createRequestDebugSession, isRequestDebugEnabled, type RequestDebugResponseLog } from "../utils/request-debug"; import { adaptSchemaForStrict, NO_STRICT, sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema"; import { notifyRawSseEvent } from "../utils/sse-debug"; @@ -1464,6 +1465,7 @@ async function openCodexWebSocketTransport( websocketState, toWebSocketUrl(requestContext.url), websocketHeaders, + model.provider, requestSetup.requestSignal, ); const eventStream = websocketConnection.streamRequest( @@ -2587,6 +2589,7 @@ export async function prewarmOpenAICodexResponses( state, toWebSocketUrl(url), headers, + model.provider, options?.signal, ); state.prewarmed = true; @@ -3049,11 +3052,13 @@ interface CodexWebSocketRequestTimeouts { interface CodexWebSocketConnectionOptions { onHandshakeHeaders?: (headers: Headers) => void; + proxy?: string; } class CodexWebSocketConnection { #url: string; #headers: Record; + #proxy?: string; #onHandshakeHeaders?: (headers: Headers) => void; #socket: Bun.WebSocket | null = null; #queue: Array | Error | null> = []; @@ -3084,6 +3089,7 @@ class CodexWebSocketConnection { constructor(url: string, headers: Record, options: CodexWebSocketConnectionOptions) { this.#url = url; this.#headers = headers; + this.#proxy = options.proxy; this.#onHandshakeHeaders = options.onHandshakeHeaders; } @@ -3145,7 +3151,7 @@ class CodexWebSocketConnection { this.#connectPromise = promise; const socket = new (WebSocket as unknown as new (url: string, opts: Bun.WebSocketOptions) => Bun.WebSocket)( this.#url, - { headers: this.#headers }, + { headers: this.#headers, proxy: this.#proxy }, ); socket.binaryType = "nodebuffer"; this.#socket = socket; @@ -3636,8 +3642,18 @@ async function getOrCreateCodexWebSocketConnection( state: CodexWebSocketSessionState, url: string, headers: Headers, + provider: string, signal?: AbortSignal, ): Promise { + const targetUrl = new URL(url); + const proxy = shouldBypassProxy(targetUrl) + ? undefined + : (getProxyForProvider(provider) ?? + (targetUrl.protocol === "wss:" + ? Bun.env.HTTPS_PROXY || Bun.env.https_proxy + : Bun.env.HTTP_PROXY || Bun.env.http_proxy) ?? + Bun.env.ALL_PROXY ?? + Bun.env.all_proxy); const headerRecord = headersToRecord(headers); // Join an in-flight handshake instead of tearing it down: closing a // CONNECTING socket rejects the concurrent caller (prewarm racing the first @@ -3680,6 +3696,7 @@ async function getOrCreateCodexWebSocketConnection( onHandshakeHeaders: handshakeHeaders => { updateCodexSessionMetadataFromHeaders(state, handshakeHeaders); }, + proxy, }); await state.connection.connect(signal); return state.connection; diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 32b0b52ec..cbd9e9589 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -15,6 +15,7 @@ import type { ModelSpec, ProviderSessionState, } from "@oh-my-pi/pi-ai/types"; +import { __resetProxyCache } from "@oh-my-pi/pi-ai/utils/proxy"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import * as piUtils from "@oh-my-pi/pi-utils"; @@ -23,6 +24,16 @@ const { getAgentDir, setAgentDir, TempDir } = piUtils; const originalAgentDir = getAgentDir(); const originalWebSocket = global.WebSocket; const originalCodexWebSocketV2 = Bun.env.PI_CODEX_WEBSOCKET_V2; +const originalProxyEnv: Record = { + PI_PROXY: Bun.env.PI_PROXY, + PI_PROXY_CODEX_PROXY_TEST: Bun.env.PI_PROXY_CODEX_PROXY_TEST, + HTTPS_PROXY: Bun.env.HTTPS_PROXY, + https_proxy: Bun.env.https_proxy, + ALL_PROXY: Bun.env.ALL_PROXY, + all_proxy: Bun.env.all_proxy, + NO_PROXY: Bun.env.NO_PROXY, + no_proxy: Bun.env.no_proxy, +}; const TEST_INSTALLATION_ID = "00000000-0000-4000-8000-000000000001"; function restoreEnv(name: string, value: string | undefined): void { @@ -34,6 +45,8 @@ function restoreEnv(name: string, value: string | undefined): void { } beforeEach(() => { + for (const key in originalProxyEnv) delete Bun.env[key]; + __resetProxyCache(); vi.spyOn(piUtils, "getInstallId").mockReturnValue(TEST_INSTALLATION_ID); }); @@ -41,6 +54,8 @@ afterEach(() => { global.WebSocket = originalWebSocket; setAgentDir(originalAgentDir); restoreEnv("PI_CODEX_WEBSOCKET_V2", originalCodexWebSocketV2); + for (const key in originalProxyEnv) restoreEnv(key, originalProxyEnv[key]); + __resetProxyCache(); vi.restoreAllMocks(); }); @@ -169,6 +184,7 @@ function encodeWebSocketMessage(value: Record): Uint8Array { } type WsHeaders = Record; +type WsOptions = { headers?: WsHeaders; proxy?: string }; type WsEventType = "open" | "message" | "error" | "close"; type CodexTestUsage = { @@ -208,7 +224,7 @@ class MockWebSocket { constructor( public readonly url: string, - public readonly options?: { headers?: WsHeaders }, + public readonly options?: WsOptions, ) {} send(_data: string): void {} @@ -1225,6 +1241,107 @@ describe("openai-codex streaming", () => { expect(Object.keys(capturedHeaders ?? {}).filter(key => key.toLowerCase() === "openai-beta")).toHaveLength(1); }); + it("passes the provider proxy to websocket handshakes", async () => { + const proxy = "socks5://127.0.0.1:7890"; + Bun.env.PI_PROXY_CODEX_PROXY_TEST = proxy; + __resetProxyCache(); + let capturedProxy: string | undefined; + class ProxyCaptureWebSocket extends MockWebSocket { + constructor(url: string, options?: WsOptions) { + super(url, options); + capturedProxy = options?.proxy; + this.scheduleOpen(); + } + } + global.WebSocket = ProxyCaptureWebSocket as unknown as typeof WebSocket; + const model = { + ...createCodexTestModel("https://chatgpt.com/backend-api"), + provider: "codex-proxy-test", + }; + const providerSessionState = new Map(); + + try { + await prewarmOpenAICodexResponses(model, { + apiKey: createCodexTestToken(), + sessionId: "ws-proxy-session", + providerSessionState, + }); + expect(capturedProxy).toBe(proxy); + } finally { + for (const state of providerSessionState.values()) state.close(); + delete Bun.env.PI_PROXY_CODEX_PROXY_TEST; + } + }); + + it("falls back to standard proxy variables for websocket handshakes", async () => { + const cases: Array<{ env: string; proxy: string }> = [ + { env: "HTTPS_PROXY", proxy: "http://127.0.0.1:7890" }, + { env: "ALL_PROXY", proxy: "socks5://127.0.0.1:7891" }, + ]; + + for (const { env, proxy } of cases) { + delete Bun.env.HTTPS_PROXY; + delete Bun.env.ALL_PROXY; + Bun.env[env] = proxy; + let capturedProxy: string | undefined; + class StandardProxyWebSocket extends MockWebSocket { + constructor(url: string, options?: WsOptions) { + super(url, options); + capturedProxy = options?.proxy; + this.scheduleOpen(); + } + } + global.WebSocket = StandardProxyWebSocket as unknown as typeof WebSocket; + const model = { + ...createCodexTestModel("https://chatgpt.com/backend-api"), + provider: `codex-${env.toLowerCase()}-test`, + }; + const providerSessionState = new Map(); + + try { + await prewarmOpenAICodexResponses(model, { + apiKey: createCodexTestToken(), + sessionId: `ws-${env.toLowerCase()}-proxy-session`, + providerSessionState, + }); + expect(capturedProxy).toBe(proxy); + } finally { + for (const state of providerSessionState.values()) state.close(); + } + } + }); + + it("bypasses configured proxies for NO_PROXY websocket targets", async () => { + Bun.env.PI_PROXY_CODEX_PROXY_TEST = "http://127.0.0.1:7890"; + Bun.env.NO_PROXY = "chatgpt.com"; + __resetProxyCache(); + let capturedProxy: string | undefined; + class NoProxyWebSocket extends MockWebSocket { + constructor(url: string, options?: WsOptions) { + super(url, options); + capturedProxy = options?.proxy; + this.scheduleOpen(); + } + } + global.WebSocket = NoProxyWebSocket as unknown as typeof WebSocket; + const model = { + ...createCodexTestModel("https://chatgpt.com/backend-api"), + provider: "codex-proxy-test", + }; + const providerSessionState = new Map(); + + try { + await prewarmOpenAICodexResponses(model, { + apiKey: createCodexTestToken(), + sessionId: "ws-no-proxy-session", + providerSessionState, + }); + expect(capturedProxy).toBeUndefined(); + } finally { + for (const state of providerSessionState.values()) state.close(); + } + }); + it("sends the Responses Lite marker on the upgrade and in response.create client_metadata", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); From 19674b8dfafef5e3395f2a4ed8d8f337ab164790 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 19:59:47 +0000 Subject: [PATCH 034/293] fix(skills): reloaded runtime skill state - Rediscovered enabled skills across TUI, ACP, and RPC plugin reloads. - Rebuilt skill commands, system prompts, tool snapshots, and skill URL resolution. - Hot-refreshed managed skills after manage_skill create, update, or delete. Fixes #4996 --- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/modes/acp/acp-agent.ts | 1 + .../modes/controllers/selector-controller.ts | 1 + .../src/modes/interactive-mode.ts | 32 +++++-- .../coding-agent/src/modes/rpc/rpc-mode.ts | 1 + packages/coding-agent/src/modes/types.ts | 2 + packages/coding-agent/src/sdk.ts | 8 +- .../coding-agent/src/session/agent-session.ts | 33 ++++++- .../src/slash-commands/builtin-registry.ts | 2 + .../coding-agent/src/slash-commands/types.ts | 8 +- packages/coding-agent/src/system-prompt.ts | 4 +- packages/coding-agent/src/tools/index.ts | 4 +- .../coding-agent/src/tools/manage-skill.ts | 8 +- packages/coding-agent/test/sdk-skills.test.ts | 90 +++++++++++++++++++ 14 files changed, 176 insertions(+), 22 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6c5d526e2..bb0d4896c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `/reload-plugins`, plugin setting changes, and `manage_skill` writes leaving runtime skills and `skill://` resolution stale until restart; sessions now rediscover enabled skills and rebuild `/skill:` commands before the next prompt ([#4996](https://github.com/can1357/oh-my-pi/issues/4996)). + ## [16.3.15] - 2026-07-09 ### Changed diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index 6d76f2721..fb77af77b 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -1845,6 +1845,7 @@ export class AcpAgent implements Agent { const projectPath = await resolveActiveProjectRegistryPath(cwd); clearPluginRootsAndCaches(projectPath ? [projectPath] : undefined); resetCapabilities(); + await record.session.refreshSkills(); const fileCommands = await loadSlashCommands({ cwd }); record.session.setSlashCommands(fileCommands); await record.session.refreshSshTool({ activateIfAvailable: true }); diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index 8e06c5887..7fe56c296 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -184,6 +184,7 @@ export class SelectorController { onPluginsChanged: async () => { const projectPath = await resolveActiveProjectRegistryPath(this.ctx.sessionManager.getCwd()); clearPluginRootsAndCaches(projectPath ? [projectPath] : undefined); + await this.ctx.refreshSkillState(); await this.ctx.refreshSlashCommandState(); await this.ctx.session.refreshSshTool({ activateIfAvailable: true }); this.ctx.ui.requestRender(); diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 5bd64e097..aa4dbde0d 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -725,15 +725,7 @@ export class InteractiveMode implements InteractiveModeContext { description: `${loaded.command.description} (${loaded.source})`, })); - // Build skill commands from session.skills (if enabled) - const skillCommandList: SlashCommand[] = []; - if (settings.get("skills.enableSkillCommands")) { - for (const skill of this.session.skills) { - const commandName = `skill:${skill.name}`; - this.skillCommands.set(commandName, skill); - skillCommandList.push({ name: commandName, description: skill.description }); - } - } + const skillCommandList = this.#rebuildSkillCommandsFromSession(); const builtinCommands = buildTuiBuiltinSlashCommands({ ctx: this }); // Store pending commands for init() where file commands are loaded async @@ -1069,6 +1061,27 @@ export class InteractiveMode implements InteractiveModeContext { this.session.setTitleSystemPrompt(resolved); } + #rebuildSkillCommandsFromSession(): SlashCommand[] { + const commands: SlashCommand[] = []; + this.skillCommands.clear(); + if (this.session.skillsSettings?.enableSkillCommands !== false) { + for (const skill of this.session.skills) { + const commandName = `skill:${skill.name}`; + this.skillCommands.set(commandName, skill); + commands.push({ name: commandName, description: skill.description }); + } + } + return commands; + } + + /** Reload session skills and the `/skill:` command list. */ + async refreshSkillState(): Promise { + await this.session.refreshSkills(); + const retainedCommands = this.#pendingSlashCommands.filter(command => !command.name.startsWith("skill:")); + const skillCommands = this.#rebuildSkillCommandsFromSession(); + this.#pendingSlashCommands = [...retainedCommands, ...skillCommands]; + } + /** Reload slash commands and autocomplete for the provided working directory. */ async refreshSlashCommandState(cwd?: string): Promise { const basePath = cwd ?? this.sessionManager.getCwd(); @@ -1167,6 +1180,7 @@ export class InteractiveMode implements InteractiveModeContext { clearClaudePluginRootsCache(); await this.refreshTitleSystemPrompt(newCwd); resetCapabilities(); + await this.refreshSkillState(); await this.refreshSlashCommandState(newCwd); await this.session.refreshSshTool({ activateIfAvailable: true }); setSessionTerminalTitle(this.sessionManager.getSessionName(), this.sessionManager.getCwd()); diff --git a/packages/coding-agent/src/modes/rpc/rpc-mode.ts b/packages/coding-agent/src/modes/rpc/rpc-mode.ts index 16d1cf1b3..b6e0037c4 100644 --- a/packages/coding-agent/src/modes/rpc/rpc-mode.ts +++ b/packages/coding-agent/src/modes/rpc/rpc-mode.ts @@ -828,6 +828,7 @@ export async function runRpcMode( const projectPath = await resolveActiveProjectRegistryPath(cwd); clearPluginRootsAndCaches(projectPath ? [projectPath] : undefined); resetCapabilities(); + await session.refreshSkills(); session.setSlashCommands(await loadSlashCommands({ cwd })); await session.refreshSshTool({ activateIfAvailable: true }); await emitAvailableCommandsUpdate(); diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 8c66eb8b7..5776c6ec1 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -344,6 +344,8 @@ export interface InteractiveModeContext { ): Promise; openInBrowser(urlOrPath: string): void; refreshSlashCommandState(cwd?: string): Promise; + /** Reload session skills and derived `/skill:` commands. */ + refreshSkillState(): Promise; applyCwdChange(newCwd: string): Promise; // Selector handling diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 0d27eac07..fc0a3d201 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1547,7 +1547,10 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} skipPythonPreflight: options.skipPythonPreflight, contextFiles, workspaceTree: resolvedWorkspaceTree, - skills, + get skills() { + return session?.skills ?? skills; + }, + refreshSkills: () => session.refreshSkills(), rules: allRules, eventBus, outputSchema: options.outputSchema, @@ -2399,7 +2402,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} const defaultPrompt = await buildSystemPromptInternal({ cwd, resolvedCustomPrompt: options.customSystemPrompt, - skills, + skills: session?.skills ?? skills, contextFiles, tools: promptTools, toolNames, @@ -2847,6 +2850,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} customCommands: customCommandsResult.commands, skills, skillWarnings, + skillsReloadable: options.skills === undefined, skillsSettings: settings.getGroup("skills"), modelRegistry, toolRegistry, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 3b4274e5f..ca1d3d20a 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -230,7 +230,7 @@ import type { CompactOptions, ContextUsage } from "../extensibility/extensions/t import { ExtensionToolWrapper } from "../extensibility/extensions/wrapper"; import type { HookCommandContext } from "../extensibility/hooks/types"; import type { RecoveredRetryError } from "../extensibility/shared-events"; -import type { Skill, SkillWarning } from "../extensibility/skills"; +import { loadSkills, type Skill, type SkillWarning, setActiveSkills } from "../extensibility/skills"; import { expandSlashCommand, type FileSlashCommand } from "../extensibility/slash-commands"; import { GoalRuntime } from "../goals/runtime"; import type { Goal, GoalModeState } from "../goals/state"; @@ -686,6 +686,8 @@ export interface AgentSessionConfig { skills?: Skill[]; /** Skill loading warnings (already captured by SDK) */ skillWarnings?: SkillWarning[]; + /** Whether runtime reloads may rediscover disk-backed skills for this session. */ + skillsReloadable?: boolean; /** Custom commands (TypeScript slash commands) */ customCommands?: LoadedCustomCommand[]; skillsSettings?: SkillsSettings; @@ -1714,6 +1716,7 @@ export class AgentSession { #mcpPromptCommands: LoadedCustomCommand[] = []; #skillsSettings: SkillsSettings | undefined; + #skillsReloadable: boolean; // Model registry for API key resolution #modelRegistry: ModelRegistry; @@ -2066,6 +2069,7 @@ export class AgentSession { this.#skills = config.skills ?? []; this.#skillWarnings = config.skillWarnings ?? []; this.#customCommands = config.customCommands ?? []; + this.#skillsReloadable = config.skillsReloadable ?? true; this.#skillsSettings = config.skillsSettings; this.#modelRegistry = config.modelRegistry; // Resolve the wire service-tier per request so the Fireworks Priority @@ -6427,6 +6431,33 @@ export class AgentSession { await this.#applyActiveToolsByName(nextActive); } + /** + * Rediscover disk-backed skills and rebuild prompt-facing state without + * recreating the session. Explicit skill snapshots (`--no-skills`, + * SDK-provided `skills`) remain fixed for the lifetime of the session. + */ + async refreshSkills(): Promise { + if (!this.#skillsReloadable) { + return; + } + + resetCapabilities(); + const skillsSettings = this.settings.getGroup("skills"); + const discovered = await loadSkills({ + ...skillsSettings, + cwd: this.sessionManager.getCwd(), + disabledExtensions: this.settings.get("disabledExtensions") ?? [], + }); + this.#skills = discovered.skills; + this.#skillWarnings = discovered.warnings; + this.#skillsSettings = skillsSettings; + + if (this.#agentKind === "main") { + setActiveSkills(this.#skills); + } + await this.refreshBaseSystemPrompt(); + } + /** * Set active tools by name. * Only tools in the registry can be enabled. Unknown tool names are ignored. diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 5f9bdfc2e..5995485a6 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -2198,6 +2198,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ // listClaudePluginRoots re-reads from disk on next access. const projectPath = await resolveActiveProjectRegistryPath(runtime.ctx.sessionManager.getCwd()); clearPluginRootsAndCaches(projectPath ? [projectPath] : undefined); + await runtime.ctx.refreshSkillState(); await runtime.ctx.refreshSlashCommandState(); await runtime.ctx.session.refreshSshTool({ activateIfAvailable: true }); runtime.ctx.showStatus("Plugins reloaded."); @@ -2537,6 +2538,7 @@ export async function executeBuiltinSlashCommand( reloadPlugins: async () => { const projectPath = await resolveActiveProjectRegistryPath(ctx.sessionManager.getCwd()); clearPluginRootsAndCaches(projectPath ? [projectPath] : undefined); + await ctx.refreshSkillState(); await ctx.refreshSlashCommandState(); await ctx.session.refreshSshTool({ activateIfAvailable: true }); }, diff --git a/packages/coding-agent/src/slash-commands/types.ts b/packages/coding-agent/src/slash-commands/types.ts index 3dfbe66f1..f5fcb22b8 100644 --- a/packages/coding-agent/src/slash-commands/types.ts +++ b/packages/coding-agent/src/slash-commands/types.ts @@ -64,10 +64,10 @@ export interface SlashCommandRuntime { /** Re-advertise the available command list (no-op outside ACP). */ refreshCommands: () => Promise | void; /** - * Reload plugin state (caches, slash command registry, project registries) - * and re-emit available commands. Used by `/reload-plugins`, `/move`, and - * `/marketplace`/`/plugins` mutations so the session sees a consistent view - * after plugin or project-scope changes. + * Reload plugin state (caches, skills, slash command registry, project + * registries) and re-emit available commands. Used by `/reload-plugins`, + * `/move`, and `/marketplace`/`/plugins` mutations so the session sees a + * consistent view after plugin or project-scope changes. */ reloadPlugins: () => Promise; notifyTitleChanged?: () => Promise | void; diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index f8ddd76ca..6ba43bead 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -468,7 +468,7 @@ export interface BuildSystemPromptOptions { /** Pre-loaded context files (skips discovery if provided). */ contextFiles?: Array<{ path: string; content: string; depth?: number }>; /** Skills provided directly to system prompt construction. */ - skills?: Skill[]; + skills?: readonly Skill[]; /** Pre-loaded rulebook rules (descriptions, excluding TTSR and always-apply). */ rules?: Array<{ name: string; description?: string; path: string; globs?: string[] }>; /** Intent field name injected into every tool schema. If set, explains the field in the prompt. */ @@ -623,7 +623,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): totalLines: 0, agentsMdFiles: [], }); - const skillsPromise: Promise = + const skillsPromise: Promise = providedSkills !== undefined ? Promise.resolve(providedSkills) : skillsSettings?.enabled !== false diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 8ea30c85d..8e4b53e63 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -174,7 +174,9 @@ export interface ToolSession { /** Pre-loaded workspace tree (forwarded to subagents to skip re-scanning) */ workspaceTree?: WorkspaceTree; /** Pre-loaded skills */ - skills?: Skill[]; + skills?: readonly Skill[]; + /** Rediscover live session skills after a tool mutates their backing files. */ + refreshSkills?: () => Promise; /** Pre-loaded prompt templates */ promptTemplates?: PromptTemplate[]; /** Pre-loaded rules (forwarded to subagents to skip re-discovery). */ diff --git a/packages/coding-agent/src/tools/manage-skill.ts b/packages/coding-agent/src/tools/manage-skill.ts index 405c1ed14..ae0e9a167 100644 --- a/packages/coding-agent/src/tools/manage-skill.ts +++ b/packages/coding-agent/src/tools/manage-skill.ts @@ -45,16 +45,17 @@ export class ManageSkillTool implements AgentTool { readonly loadMode = "essential" as const; readonly summary = "Create, update, or delete an isolated managed skill"; - // No session state needed: createIf reads settings; writes target the - // home-based managed-skills dir directly. + constructor(private readonly refreshSkills?: () => Promise) {} + static createIf(session: ToolSession): ManageSkillTool | null { if (!session.settings.get("autolearn.enabled")) return null; - return new ManageSkillTool(); + return new ManageSkillTool(session.refreshSkills); } async execute(_id: string, params: ManageSkillParams): Promise { if (params.action === "delete") { await deleteManagedSkill(params.name); + await this.refreshSkills?.(); return { content: [{ type: "text", text: `Deleted managed skill "${params.name}".` }], details: { action: "delete", name: params.name }, @@ -90,6 +91,7 @@ export class ManageSkillTool implements AgentTool { description: params.description, body: params.body, }); + await this.refreshSkills?.(); const relativePath = path.relative(getManagedSkillsDir(), skillPath); const verb = params.action === "create" ? "Created" : "Updated"; return { diff --git a/packages/coding-agent/test/sdk-skills.test.ts b/packages/coding-agent/test/sdk-skills.test.ts index 47be1a42e..f42d0fa45 100644 --- a/packages/coding-agent/test/sdk-skills.test.ts +++ b/packages/coding-agent/test/sdk-skills.test.ts @@ -4,11 +4,13 @@ import * as os from "node:os"; import * as path from "node:path"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { getActiveSkills } from "@oh-my-pi/pi-coding-agent/extensibility/skills"; import type { Skill } from "@oh-my-pi/pi-coding-agent/sdk"; import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { removeSyncWithRetries } from "@oh-my-pi/pi-utils"; +import { getAgentDir, setAgentDir } from "@oh-my-pi/pi-utils/dirs"; import { cleanupTempHome } from "./helpers/temp-home-cleanup"; function createIsolatedSkillsSettings(): Settings { @@ -132,6 +134,94 @@ Loaded via symbolic link. expect(session.skills.some((s: Skill) => s.name === "test-skill")).toBe(true); }); + + it("refreshSkills reloads project skills on an existing session", async () => { + const { session } = await createAgentSession({ + cwd: tempDir, + agentDir: tempDir, + sessionManager: SessionManager.inMemory(tempDir), + modelRegistry: sharedModelRegistry, + settings: createIsolatedSkillsSettings(), + }); + + expect(session.skills.some((s: Skill) => s.name === "runtime-added-skill")).toBe(false); + + const runtimeSkillDir = path.join(tempDir, ".omp", "skills", "runtime-added-skill"); + fs.mkdirSync(runtimeSkillDir, { recursive: true }); + fs.writeFileSync( + path.join(runtimeSkillDir, "SKILL.md"), + `--- +name: runtime-added-skill +description: Added after the session is created. +--- + +# Runtime Added Skill + +This skill is added after session creation. +`, + ); + + await session.refreshSkills(); + + expect(session.skills.some((s: Skill) => s.name === "runtime-added-skill")).toBe(true); + + removeSyncWithRetries(runtimeSkillDir); + + await session.refreshSkills(); + + expect(session.skills.some((s: Skill) => s.name === "runtime-added-skill")).toBe(false); + }); + + it("manage_skill hot-registers managed skills in the active session", async () => { + const originalAgentDir = getAgentDir(); + const managedAgentDir = path.join(tempHomeDir, ".omp", "agent"); + setAgentDir(managedAgentDir); + const settings = createIsolatedSkillsSettings(); + settings.set("autolearn.enabled", true); + const { session } = await createAgentSession({ + cwd: tempDir, + agentDir: managedAgentDir, + sessionManager: SessionManager.inMemory(tempDir), + modelRegistry: sharedModelRegistry, + settings, + }); + + try { + const manageSkill = session.getToolByName("manage_skill"); + expect(manageSkill).toBeDefined(); + await manageSkill!.execute("manage-skill-create", { + action: "create", + name: "runtime-managed-skill", + description: "Created by manage_skill during the session.", + body: "# Runtime Managed Skill\n\nUse this immediately.", + }); + + expect(session.skills.some(skill => skill.name === "runtime-managed-skill")).toBe(true); + expect(getActiveSkills().some(skill => skill.name === "runtime-managed-skill")).toBe(true); + expect(session.agent.state.systemPrompt.join("\n")).toContain("runtime-managed-skill"); + const readSkill = session.getToolByName("read"); + expect(readSkill).toBeDefined(); + const readResult = await readSkill!.execute("read-managed-skill", { path: "skill://runtime-managed-skill" }); + expect( + readResult.content.some(part => part.type === "text" && part.text.includes("# Runtime Managed Skill")), + ).toBe(true); + + await manageSkill!.execute("manage-skill-delete", { + action: "delete", + name: "runtime-managed-skill", + }); + expect(session.skills.some(skill => skill.name === "runtime-managed-skill")).toBe(false); + expect(getActiveSkills().some(skill => skill.name === "runtime-managed-skill")).toBe(false); + expect(session.agent.state.systemPrompt.join("\n")).not.toContain("runtime-managed-skill"); + await expect( + readSkill!.execute("read-deleted-managed-skill", { path: "skill://runtime-managed-skill" }), + ).rejects.toThrow(/Unknown skill/); + } finally { + await session.dispose(); + setAgentDir(originalAgentDir); + } + }); + it("should have empty skills when options.skills is empty array (--no-skills)", async () => { const { session } = await createAgentSession({ cwd: tempDir, From da99a06a444260d0632141b1a73614e703440cf7 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 20:11:41 +0000 Subject: [PATCH 035/293] fix(tui): accepted same-line skill completions - Matched trailing mid-prompt slash tokens during stale-prefix validation. - Exercised Enter acceptance with prose on the same line. Fixes #4773 --- packages/tui/src/components/editor.ts | 20 ++++++++++--------- .../test/editor-autocomplete-actions.test.ts | 4 ++-- 2 files changed, 13 insertions(+), 11 deletions(-) diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 84cc1b144..2c4c33676 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -2876,12 +2876,12 @@ export class Editor implements Component, Focusable { * - Exact match → always safe. * - Path branch is safe when the prefix is still a live suffix of the text; the * provider's default slice at `cursorCol - prefix.length` then hits the right span. - * - Slash branch re-anchors when both the prefix and the current text carry a - * leading slash command and the current slash token is clean (no whitespace or - * inner slash), matching `applyCompletion`'s slash-branch guard. It only - * engages for command-shaped selections: absolute-path completions (`/tmp/fo` - * via the no-command-match fall-through) share the leading-slash prefix shape - * but must use the live-suffix path rule so the apply slice stays anchored. + * - Slash branch re-anchors when the prefix is command-shaped and the current + * text carries either a leading submitted command or a trailing mid-prompt + * slash token. The live token must remain clean (no whitespace or inner slash), + * matching `applyCompletion`'s slash-branch guard. Absolute-path completions + * (`/tmp/fo` via the no-command-match fall-through) share the leading-slash + * prefix shape but use the live-suffix path rule instead. * - `@`-file branch re-anchors via `#extractAtPrefix`; safe when the current text * still ends in a whitespace-anchored `@`. * - Everything else is stale — accepting it would corrupt the buffer (issue #4295). @@ -2890,9 +2890,11 @@ export class Editor implements Component, Focusable { if (currentTextBeforeCursor === this.#autocompletePrefix) return true; if (findLeadingSlashCommandStart(this.#autocompletePrefix) !== null && !this.#selectedCompletionIsPath()) { - const currentLeadingStart = findLeadingSlashCommandStart(currentTextBeforeCursor); - if (currentLeadingStart !== null) { - const token = currentTextBeforeCursor.slice(currentLeadingStart); + const currentSlashStart = + findLeadingSlashCommandStart(currentTextBeforeCursor) ?? + findTrailingSlashCommandStart(currentTextBeforeCursor); + if (currentSlashStart !== null) { + const token = currentTextBeforeCursor.slice(currentSlashStart); if (!token.includes(" ") && !token.slice(1).includes("/")) return true; } return false; diff --git a/packages/tui/test/editor-autocomplete-actions.test.ts b/packages/tui/test/editor-autocomplete-actions.test.ts index 3d39ed516..6c66f06db 100644 --- a/packages/tui/test/editor-autocomplete-actions.test.ts +++ b/packages/tui/test/editor-autocomplete-actions.test.ts @@ -248,7 +248,7 @@ describe("Editor Enter handler sync slash completion", () => { submitted = text; }; - editor.setText("explain this\n"); + editor.setText("fix bug "); editor.handleInput("/"); await Promise.resolve(); @@ -257,7 +257,7 @@ describe("Editor Enter handler sync slash completion", () => { editor.handleInput("security"); editor.handleInput("\r"); - expect(editor.getText()).toBe("explain this\n/skill:security-scan "); + expect(editor.getText()).toBe("fix bug /skill:security-scan "); expect(submitted).toBeUndefined(); }); From e858c1be65a0ce325a0bfc3e30e370ac76c6038b Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 20:12:20 +0000 Subject: [PATCH 036/293] fix(auth): serialized provider oauth refreshes - Routed provider refreshes through durable SQLite leases and fenced follow-up writes against peer rotations. - Kept background usage probes from disabling credentials when refresh fails. - Added multi-instance and post-lease rotation regressions. Fixes #5396 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/auth-storage.ts | 153 ++++++++++-------- .../auth-storage-oauth-refresh-race.test.ts | 107 ++++++++++++ .../ai/test/auth-storage-usage-cache.test.ts | 73 ++------- packages/coding-agent/CHANGELOG.md | 1 + 5 files changed, 212 insertions(+), 126 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index d3c2273a8..a695dc165 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed concurrent provider OAuth refreshes by serializing rotating-token updates across processes, fencing stale writes, and preventing background usage probes from disabling otherwise usable credentials ([#5396](https://github.com/can1357/oh-my-pi/issues/5396)). + ## [16.5.0] - 2026-07-13 ### Added diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 245c118c6..19d6b6831 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -746,6 +746,8 @@ export interface InvalidateCredentialMatchingOptions { /** Options for refreshing one stored OAuth row through durable ownership. */ export interface StoredOAuthRefreshOptions { + /** Stable row id when a provider has multiple OAuth credentials. */ + credentialId?: number; observedCredential?: T; credentialFromRow: (credential: OAuthCredential) => T | undefined; forceRefresh?: boolean; @@ -1763,17 +1765,32 @@ export class AuthStorage { } /** - * Persist a refreshed credential addressed by id, not a positional index. - * A concurrent disable can reorder/shrink the provider's row array while an - * async refresh is in flight, so a pre-await index is unsafe; resolving the - * row by id at write time lands the rotated token on the correct row. Returns - * the row's current index, or -1 when it was disabled/removed mid-refresh. + * Persist a refreshed credential by id only while the row still matches this + * process's snapshot. A peer rotation wins the CAS and is reloaded instead of + * being overwritten after this process releases its refresh lease. + * + * Returns the row's current index, or -1 when it was disabled or removed. */ #replaceCredentialById(provider: string, id: number, credential: AuthCredential): number { const entries = this.#getStoredCredentials(provider); const index = entries.findIndex(entry => entry.id === id); if (index === -1) return -1; - this.#store.updateAuthCredential(id, credential); + const expected = serializeCredential(provider, entries[index]!.credential); + if ( + expected && + this.#store.tryUpdateAuthCredentialIfMatches && + !this.#store.tryUpdateAuthCredentialIfMatches(id, expected.data, credential) + ) { + const latest = this.#store.listAuthCredentials(provider); + this.#setStoredCredentials( + provider, + latest.map(row => ({ id: row.id, credential: row.credential })), + ); + return latest.findIndex(row => row.id === id); + } + if (!expected || !this.#store.tryUpdateAuthCredentialIfMatches) { + this.#store.updateAuthCredential(id, credential); + } const updated = [...entries]; updated[index] = { id, credential }; this.#setStoredCredentials(provider, updated); @@ -1906,7 +1923,11 @@ export class AuthStorage { provider, rows.map(row => ({ id: row.id, credential: row.credential })), ); - const row = rows.find(entry => entry.credential.type === "oauth"); + const row = rows.find( + entry => + entry.credential.type === "oauth" && + (options.credentialId === undefined || entry.id === options.credentialId), + ); if (row?.credential.type !== "oauth") { return { credential: undefined, refreshed: false, removed: false }; } @@ -1945,7 +1966,11 @@ export class AuthStorage { provider, rows.map(row => ({ id: row.id, credential: row.credential })), ); - const row = rows.find(entry => entry.credential.type === "oauth"); + const row = rows.find( + entry => + entry.credential.type === "oauth" && + (options.credentialId === undefined || entry.id === options.credentialId), + ); if (row?.credential.type !== "oauth") { return { credential: undefined, refreshed: false, removed: false }; } @@ -2526,25 +2551,15 @@ export class AuthStorage { } #persistRefreshedUsageCredential(provider: Provider, previous: UsageCredential, next: UsageCredential): void { - const entries = this.#getStoredCredentials(provider); - const index = entries.findIndex(entry => { - if (entry.credential.type !== "oauth") return false; - if (previous.refreshToken && entry.credential.refresh === previous.refreshToken) return true; - if (previous.accessToken && entry.credential.access === previous.accessToken) return true; - return ( - entry.credential.accountId === previous.accountId && - entry.credential.email === previous.email && - entry.credential.projectId === previous.projectId - ); - }); - if (index === -1) return; - const existing = entries[index]!.credential; - if (existing.type !== "oauth") return; - this.#replaceCredentialAt(provider, index, { + const credentialId = this.#findStoredCredentialIdForUsageCredential(provider, previous); + if (credentialId === undefined) return; + const entry = this.#getStoredCredentials(provider).find(candidate => candidate.id === credentialId); + if (entry?.credential.type !== "oauth") return; + this.#replaceCredentialById(provider, credentialId, { type: "oauth", - access: next.accessToken ?? existing.access, - refresh: next.refreshToken ?? existing.refresh, - expires: next.expiresAt ?? existing.expires, + access: next.accessToken ?? entry.credential.access, + refresh: next.refreshToken ?? entry.credential.refresh, + expires: next.expiresAt ?? entry.credential.expires, accountId: next.accountId, projectId: next.projectId, email: next.email, @@ -2598,46 +2613,9 @@ export class AuthStorage { }; } catch (error) { const errorMsg = String(error); - // Definitive failure (invalid_grant / 401 not from a network blip) means - // the refresh token itself is dead — probing with the original credential - // will 401, the catch below will return null, and #fetchUsageCached's - // last-good fallback will surface yesterday's report indefinitely - // (including its already-elapsed `resetsAt`). CAS-disable the row and - // clear the cache so the credential drops out of the report instead of - // freezing in place until the user notices and re-logs in. - if (AIError.isDefinitiveOAuthFailure(errorMsg)) { - const credentialId = this.#findStoredCredentialIdForUsageCredential( - request.provider, - request.credential, - ); - if (credentialId !== undefined) { - const entries = this.#getStoredCredentials(request.provider); - const index = entries.findIndex(entry => entry.id === credentialId); - if (index !== -1) { - const disabled = this.#tryDisableCredentialAtIfMatches( - request.provider, - index, - refreshableCredential, - `oauth refresh failed during usage probe: ${errorMsg}`, - ); - if (disabled) { - this.#usageLogger?.warn( - "Usage credential refresh failed definitively; credential disabled", - { provider: request.provider, credentialId, error: errorMsg }, - ); - // Neutralize last-good for this cache key: write a null - // entry with an immediately-elapsed expiry so a future - // getStale lookup (e.g. on re-login under the same - // account identity) can't replay the stale report. - this.#usageCache.set(this.#buildUsageReportCacheKey(request), { - value: null, - expiresAt: 0, - }); - return null; - } - } - } - } + // Usage polling is advisory. A refresh can fail while the current + // access token remains valid inside the refresh skew, so probe with + // that token and never mutate credential state from this path. this.#usageLogger?.debug("Usage credential refresh failed, using original credential", { provider: request.provider, error: errorMsg, @@ -3960,6 +3938,42 @@ export class AuthStorage { credential: OAuthCredential, credentialId: number | undefined, signal?: AbortSignal, + ): Promise { + const hasDurableLease = + !!this.#store.tryAcquireCredentialRefreshLease && + !!this.#store.getCredentialRefreshLeaseExpiresAt && + !!this.#store.releaseCredentialRefreshLease && + !!this.#store.renewCredentialRefreshLease; + if (credentialId !== undefined && hasDurableLease) { + const forceRefresh = credential.expires === 0; + const result = await this.refreshStoredOAuthCredential(provider, { + credentialId, + observedCredential: forceRefresh ? undefined : credential, + credentialFromRow: row => row, + forceRefresh, + signal, + refresh: (current, refreshSignal) => + this.#requestOAuthCredentialRefresh( + provider, + current, + credentialId, + signal && refreshSignal ? AbortSignal.any([signal, refreshSignal]) : (signal ?? refreshSignal), + ), + }); + if (result.credential) return result.credential; + throw new AIError.OAuthError(`OAuth credential no longer exists for provider: ${provider}`, { + kind: "token-refresh", + provider, + }); + } + return this.#requestOAuthCredentialRefresh(provider, credential, credentialId, signal); + } + + async #requestOAuthCredentialRefresh( + provider: Provider, + credential: OAuthCredential, + credentialId: number | undefined, + signal?: AbortSignal, ): Promise { let refreshPromise: Promise; // Caller override > store-level hook > local per-provider refresh. @@ -3986,10 +4000,9 @@ export class AuthStorage { // Bound the refresh so a slow/hanging token endpoint cannot stall credential selection. // Caller-driven abort jumps the gun on the timeout — the agent's ESC must // take priority over the floor timeout. - let timeout: NodeJS.Timeout | undefined; - let onAbort: (() => void) | undefined; const cancellation = Promise.withResolvers(); - timeout = setTimeout( + let onAbort: (() => void) | undefined; + const timeout = setTimeout( () => cancellation.reject( new AIError.OAuthError(`OAuth token refresh timed out for provider: ${provider}`, { @@ -4010,7 +4023,7 @@ export class AuthStorage { try { return await Promise.race([refreshPromise, cancellation.promise]); } finally { - if (timeout) clearTimeout(timeout); + clearTimeout(timeout); if (signal && onAbort) signal.removeEventListener("abort", onAbort); } } diff --git a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts index 312a073dd..c64c1ea1e 100644 --- a/packages/ai/test/auth-storage-oauth-refresh-race.test.ts +++ b/packages/ai/test/auth-storage-oauth-refresh-race.test.ts @@ -285,6 +285,113 @@ describe("AuthStorage OAuth refresh race", () => { expect(refreshCalls).toBe(1); }); + test("serializes rotating provider refresh tokens across AuthStorage instances", async () => { + if (!authStorage) throw new Error("test setup failed"); + + const expires = Date.now() - 60_000; + const refreshedExpires = Date.now() + 60 * 60_000; + const usedRefreshTokens = new Set(); + let refreshCalls = 0; + + oauthUtils.registerOAuthProvider({ + id: "unit-oauth-cross-process", + name: "Unit OAuth Cross Process", + sourceId: "auth-storage-oauth-refresh-race-test", + async login() { + return { access: "unused", refresh: "unused", expires: refreshedExpires }; + }, + async refreshToken(credentials) { + refreshCalls += 1; + if (usedRefreshTokens.has(credentials.refresh)) { + throw new Error('HTTP 400 invalid_grant {"error":"invalid_grant"}'); + } + usedRefreshTokens.add(credentials.refresh); + await Bun.sleep(50); + return { + ...credentials, + access: "access-rotated", + refresh: "refresh-rotated", + expires: refreshedExpires, + }; + }, + getApiKey(credentials) { + return credentials.access; + }, + }); + + await authStorage.set("unit-oauth-cross-process", [ + { type: "oauth", access: "access-old", refresh: "refresh-old", expires }, + ]); + + const secondStore = await SqliteAuthCredentialStore.open(path.join(tempDir, "agent.db")); + const secondStorage = new AuthStorage(secondStore); + await secondStorage.reload(); + try { + const [first, second] = await Promise.all([ + authStorage.getApiKey("unit-oauth-cross-process", "session-first"), + secondStorage.getApiKey("unit-oauth-cross-process", "session-second"), + ]); + + expect(first).toBe("access-rotated"); + expect(second).toBe("access-rotated"); + expect(refreshCalls).toBe(1); + expect(secondStore.listAuthCredentials("unit-oauth-cross-process")).toHaveLength(1); + } finally { + secondStorage.close(); + } + }); + + test("does not overwrite a peer rotation after releasing the refresh lease", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + const sharedStore = store; + + const expires = Date.now() - 60_000; + const refreshedExpires = Date.now() + 60 * 60_000; + let credentialId: number | undefined; + + oauthUtils.registerOAuthProvider({ + id: "unit-oauth-post-lease-race", + name: "Unit OAuth Post-Lease Race", + sourceId: "auth-storage-oauth-refresh-race-test", + async login() { + return { access: "unused", refresh: "unused", expires: refreshedExpires }; + }, + async refreshToken(credentials) { + return { + ...credentials, + access: "access-from-this-process", + refresh: "refresh-from-this-process", + expires: refreshedExpires, + }; + }, + getApiKey(credentials) { + if (credentialId === undefined) throw new Error("credential id not initialized"); + sharedStore.updateAuthCredential(credentialId, { + type: "oauth", + access: "access-from-peer", + refresh: "refresh-from-peer", + expires: refreshedExpires, + }); + return credentials.access; + }, + }); + + await authStorage.set("unit-oauth-post-lease-race", [ + { type: "oauth", access: "access-old", refresh: "refresh-old", expires }, + ]); + credentialId = store.listAuthCredentials("unit-oauth-post-lease-race")[0]?.id; + expect(credentialId).toBeDefined(); + + const apiKey = await authStorage.getApiKey("unit-oauth-post-lease-race", "session-post-lease"); + expect(apiKey).toBe("access-from-this-process"); + const persisted = store.listAuthCredentials("unit-oauth-post-lease-race")[0]?.credential; + expect(persisted?.type).toBe("oauth"); + if (persisted?.type === "oauth") { + expect(persisted.refresh).toBe("refresh-from-peer"); + expect(persisted.access).toBe("access-from-peer"); + } + }); + test("syncs peer-updated SQLite OAuth rows before returning access tokens", async () => { if (!authStorage || !store) throw new Error("test setup failed"); diff --git a/packages/ai/test/auth-storage-usage-cache.test.ts b/packages/ai/test/auth-storage-usage-cache.test.ts index 5974381df..b7c58f8ae 100644 --- a/packages/ai/test/auth-storage-usage-cache.test.ts +++ b/packages/ai/test/auth-storage-usage-cache.test.ts @@ -531,39 +531,23 @@ describe("AuthStorage usage cache: header ingestion", () => { }); describe("AuthStorage usage cache: terminal refresh failure", () => { - // Regression: a revoked refresh token used to fail the in-line OAuth refresh - // inside the usage probe, get silently swallowed, then trigger the upstream - // 401 → null → last-good fallback chain. The credential was therefore never - // removed from the candidate set and the /usage TUI kept rendering yesterday's - // report — including its now-elapsed `resetsAt`, which the renderer printed - // as e.g. `(-612090ms)`. The fix CAS-disables the row on a definitive refresh - // failure and clears the cache, so the credential drops out cleanly. - it("disables credential and suppresses last-good when OAuth refresh fails with invalid_grant", async () => { - // Row whose access token has just expired — within the 60s refresh skew so - // the usage probe is forced to refresh before issuing the upstream call. + // Usage polling is non-critical: refresh failure must not disable a + // credential whose current access token can still satisfy the probe. + it("keeps credential and probes with current access after a definitive refresh failure", async () => { const row = oauthRow(1, "a@example.com"); - (row.credential as { expires: number }).expires = Date.now() - 1000; + if (row.credential.type !== "oauth") throw new Error("expected OAuth test credential"); + row.credential.expires = Date.now() + 30_000; const rows = [row]; - - // `makeStore` returns `false` from `tryDisableAuthCredentialIfMatches`, - // which would short-circuit our disable. Use a local store that actually - // performs the soft-delete so we can observe the AuthStorage-side effects. const cache = new Map(); let disableCalls = 0; const store: ObservableStore = { cache, close() {}, - listAuthCredentials: () => rows.filter(r => !r.disabledCause), + listAuthCredentials: () => rows.filter(candidate => !candidate.disabledCause), updateAuthCredential() {}, - deleteAuthCredential(id: number, cause: string) { - const target = rows.find(r => r.id === id); - if (target) target.disabledCause = cause; - }, - tryDisableAuthCredentialIfMatches(id: number, _data: string, cause: string) { + deleteAuthCredential() {}, + tryDisableAuthCredentialIfMatches() { disableCalls += 1; - const target = rows.find(r => r.id === id); - if (!target) return false; - target.disabledCause = cause; return true; }, replaceAuthCredentialsForProvider: () => rows, @@ -581,16 +565,6 @@ describe("AuthStorage usage cache: terminal refresh failure", () => { cleanExpiredCache() {}, }; - // Pre-populate the cache with a "last good" report whose inner expiresAt - // is in the past (so `get()` misses) but the entry is still reachable via - // `getStale()`. Mirrors what the prior poll would have written. - const lastGood = makeReport("a@example.com"); - const cacheKey = "usage_cache:report:anthropic:default:oauth|account:account-1|email:a@example.com"; - cache.set(cacheKey, { - value: JSON.stringify({ value: lastGood, expiresAt: 1 }), - expiresAtSec: Math.floor((Date.now() + 24 * 60 * 60_000) / 1000), - }); - const storage = new AuthStorage(store, { usageProviderResolver: provider => (provider === "anthropic" ? claudeUsage.claudeUsageProvider : undefined), refreshOAuthCredential: async () => { @@ -599,31 +573,18 @@ describe("AuthStorage usage cache: terminal refresh failure", () => { }); await storage.reload(); - const fetchSpy = vi.spyOn(claudeUsage.claudeUsageProvider, "fetchUsage"); - + const fetchSpy = vi + .spyOn(claudeUsage.claudeUsageProvider, "fetchUsage") + .mockResolvedValue(makeReport("a@example.com")); try { const reports = anthropicReports(await storage.fetchUsageReports()); - // No last-good fallback: the row was disabled before lastGood could leak. - expect(reports).toHaveLength(0); - // CAS disable was attempted exactly once on the failing row. - expect(disableCalls).toBe(1); - expect(rows[0].disabledCause).toContain("invalid_grant"); - // Upstream probe is short-circuited — no point asking the provider - // with a credential we've just torn down. - expect(fetchSpy).not.toHaveBeenCalled(); - // Cache entry was neutralized: a future `getStale` lookup (e.g. on - // re-login under the same account identity) returns null, not the - // stale report with its already-elapsed `resetsAt`. - const rawAfter = cache.get(cacheKey); - expect(rawAfter).toBeDefined(); - const parsedAfter = JSON.parse(rawAfter!.value); - expect(parsedAfter.value).toBeNull(); - // And a second poll surfaces nothing — the credential is gone from - // `listAuthCredentials`, so `#collectUsageRequests` doesn't even - // look it up. - const secondPoll = anthropicReports(await storage.fetchUsageReports()); - expect(secondPoll).toHaveLength(0); + expect(reports).toHaveLength(1); + expect(reports[0]?.metadata?.email).toBe("a@example.com"); + expect(disableCalls).toBe(0); + expect(rows[0]?.disabledCause).toBeNull(); + expect(fetchSpy).toHaveBeenCalledTimes(1); + expect(fetchSpy.mock.calls[0]?.[0].credential.accessToken).toBe("oat-1"); } finally { storage.close(); vi.restoreAllMocks(); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index da3550531..842271aa5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -39,6 +39,7 @@ - Fixed backgrounded Bash blocks continuing to repaint with live output; they now freeze with a compact job notice while completion is delivered separately. - Fixed rendering, status display, and PTY control sequence formatting issues in the `launch` tool. - Fixed in-process shell builtins (including `stat`, `date`, `sed`, `mktemp`, `tail`, `find`, `base64`, and `ln`) to correctly detect and translate macOS/BSD-style arguments and flags, preventing failures caused by GNU-only assumptions. +- Fixed concurrent provider OAuth refreshes from invalidating Anthropic's rotating refresh token, and prevented background usage probes from permanently disabling credentials after refresh failures ([#5396](https://github.com/can1357/oh-my-pi/issues/5396)). ### Removed From 56fb4c0145b90bb4a64f5a01989f90c8f17be58a Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 20:16:37 +0000 Subject: [PATCH 037/293] fix(tools): recover from active OpenAI image HTTP failure Threw ProviderHttpError from generateOpenAIHostedImage so a failing active OpenAI/Codex image call records the failure and continues the provider fallback chain instead of aborting the tool call. Fixes #5218 --- packages/coding-agent/src/tools/image-gen.ts | 7 +-- .../coding-agent/test/tools/image-gen.test.ts | 53 +++++++++++++++++++ 2 files changed, 57 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/tools/image-gen.ts b/packages/coding-agent/src/tools/image-gen.ts index 039855495..8d3ece8e4 100644 --- a/packages/coding-agent/src/tools/image-gen.ts +++ b/packages/coding-agent/src/tools/image-gen.ts @@ -906,9 +906,10 @@ async function generateOpenAIHostedImage( if (!response.ok) { const errorText = await response.text(); - throw Object.assign( - new Error(`OpenAI image request failed (${response.status}): ${getOpenAIResponseErrorMessage(errorText)}`), - { status: response.status }, + throw new ProviderHttpError( + `OpenAI image request failed (${response.status}): ${getOpenAIResponseErrorMessage(errorText)}`, + response.status, + { headers: response.headers }, ); } diff --git a/packages/coding-agent/test/tools/image-gen.test.ts b/packages/coding-agent/test/tools/image-gen.test.ts index 8a70c5ea0..e6e82be4d 100644 --- a/packages/coding-agent/test/tools/image-gen.test.ts +++ b/packages/coding-agent/test/tools/image-gen.test.ts @@ -394,6 +394,59 @@ describe("imageGenTool", () => { expect(result.details?.provider).toBe("xai"); }); + it("falls back to xAI after the active OpenAI provider HTTP failure", async () => { + const requestUrls: string[] = []; + const fetchMock = (async (input: string | URL | Request) => { + const url = input.toString(); + requestUrls.push(url); + if (url.startsWith("https://api.openai.com/")) { + return new Response(JSON.stringify({ error: { message: "model unavailable" } }), { + status: 404, + headers: { "content-type": "application/json" }, + }); + } + return new Response( + JSON.stringify({ data: [{ b64_json: Buffer.from("openai-fallback-xai-image").toString("base64") }] }), + { status: 200, headers: { "content-type": "application/json" } }, + ); + }) as unknown as typeof fetch; + const model = { + api: "openai-responses", + provider: "openai", + id: "gpt-5.5", + name: "GPT 5.5", + baseUrl: "https://api.openai.com/v1", + } as Model; + const ctx: CustomToolContext = { + fetch: fetchMock, + sessionManager: { + getCwd: () => "/tmp", + getSessionId: () => "test-session", + } as unknown as ReadonlySessionManager, + modelRegistry: { + getApiKey: async () => "test-openai-key", + getApiKeyForProvider: async (provider: string) => (provider === "xai-oauth" ? "test-xai-token" : undefined), + getProviderBaseUrl: () => undefined, + getAll: () => [], + authStorage: { + hasNonEnvCredential: (provider: string) => provider === "xai-oauth", + rotateSessionCredential: async () => false, + }, + resolver: () => async () => "test-openai-key", + } as unknown as ModelRegistry, + model, + isIdle: () => true, + hasQueuedMessages: () => false, + abort: () => {}, + }; + + const result = await imageGenTool.execute("call-openai-fallback-xai", { subject: "a cat" }, undefined, ctx); + generatedImagePaths.push(...(result.details?.imagePaths ?? [])); + + expect(requestUrls).toEqual(["https://api.openai.com/v1/responses", "https://api.x.ai/v1/images/generations"]); + expect(result.details?.provider).toBe("xai"); + }); + it("falls back to xAI after an earlier provider HTTP failure", async () => { const requestUrls: string[] = []; const fetchMock = (async (input: string | URL | Request) => { From 3e5a1f9e34cb399f2a7e97b9152806d3a3d2fc74 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 20:26:03 +0000 Subject: [PATCH 038/293] fix(tui): restricted trailing slash re-anchoring - Allowed trailing slash re-anchoring only for selected skill completions. - Covered stale non-skill popups accepted with Enter and Tab. Fixes #4773 --- packages/tui/src/components/editor.ts | 18 +++++---- .../test/editor-autocomplete-actions.test.ts | 40 +++++++++++++++++++ 2 files changed, 50 insertions(+), 8 deletions(-) diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 2c4c33676..a7ba0ea4a 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -2877,11 +2877,11 @@ export class Editor implements Component, Focusable { * - Path branch is safe when the prefix is still a live suffix of the text; the * provider's default slice at `cursorCol - prefix.length` then hits the right span. * - Slash branch re-anchors when the prefix is command-shaped and the current - * text carries either a leading submitted command or a trailing mid-prompt - * slash token. The live token must remain clean (no whitespace or inner slash), - * matching `applyCompletion`'s slash-branch guard. Absolute-path completions - * (`/tmp/fo` via the no-command-match fall-through) share the leading-slash - * prefix shape but use the live-suffix path rule instead. + * text carries either a leading submitted command or, for a selected skill, + * a trailing mid-prompt slash token. The live token must remain clean (no + * whitespace or inner slash), matching `applyCompletion`'s slash-branch guard. + * Absolute-path completions (`/tmp/fo` via the no-command-match fall-through) + * share the leading-slash prefix shape but use the live-suffix path rule instead. * - `@`-file branch re-anchors via `#extractAtPrefix`; safe when the current text * still ends in a whitespace-anchored `@`. * - Everything else is stale — accepting it would corrupt the buffer (issue #4295). @@ -2890,9 +2890,11 @@ export class Editor implements Component, Focusable { if (currentTextBeforeCursor === this.#autocompletePrefix) return true; if (findLeadingSlashCommandStart(this.#autocompletePrefix) !== null && !this.#selectedCompletionIsPath()) { - const currentSlashStart = - findLeadingSlashCommandStart(currentTextBeforeCursor) ?? - findTrailingSlashCommandStart(currentTextBeforeCursor); + const selected = this.#autocompleteList?.getSelectedItem(); + const currentTrailingStart = selected?.value.startsWith("skill:") + ? findTrailingSlashCommandStart(currentTextBeforeCursor) + : null; + const currentSlashStart = findLeadingSlashCommandStart(currentTextBeforeCursor) ?? currentTrailingStart; if (currentSlashStart !== null) { const token = currentTextBeforeCursor.slice(currentSlashStart); if (!token.includes(" ") && !token.slice(1).includes("/")) return true; diff --git a/packages/tui/test/editor-autocomplete-actions.test.ts b/packages/tui/test/editor-autocomplete-actions.test.ts index 6c66f06db..c0b9501b0 100644 --- a/packages/tui/test/editor-autocomplete-actions.test.ts +++ b/packages/tui/test/editor-autocomplete-actions.test.ts @@ -205,6 +205,28 @@ class SyncSlashProvider implements AutocompleteProvider { } describe("Editor Enter handler sync slash completion", () => { + async function createRelocatedModelPopup() { + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider([{ name: "model", description: "Switch AI model" }], "/tmp"), + ); + const submissions: string[] = []; + editor.onSubmit = text => { + submissions.push(text); + }; + + editor.handleInput("/mo"); + await Promise.resolve(); + expect(editor.isShowingAutocomplete()).toBe(true); + + editor.handleInput("\x01"); // Ctrl+A: move before the command. + editor.handleInput("fix "); + editor.handleInput("\x05"); // Ctrl+E: return to the stale command prefix. + expect(editor.getText()).toBe("fix /mo"); + + return { editor, submissions }; + } + it("opens mid-prompt skill autocomplete and inserts the skill token without wiping the draft on Tab", async () => { const editor = new Editor(defaultEditorTheme); editor.setAutocompleteProvider( @@ -261,6 +283,24 @@ describe("Editor Enter handler sync slash completion", () => { expect(submitted).toBeUndefined(); }); + it("submits the raw draft when Enter sees a relocated non-skill popup", async () => { + const { editor, submissions } = await createRelocatedModelPopup(); + + editor.handleInput("\r"); + + expect(submissions).toEqual(["fix /mo"]); + expect(editor.getText()).toBe(""); + }); + + it("leaves the raw draft when Tab sees a relocated non-skill popup", async () => { + const { editor, submissions } = await createRelocatedModelPopup(); + + editor.handleInput("\t"); + + expect(submissions).toEqual([]); + expect(editor.getText()).toBe("fix /mo"); + }); + it("preserves Tab file completion for an absolute path token after prose", async () => { let forceFileCalls = 0; const editor = new Editor(defaultEditorTheme); From 86b188cd41286c02c09cecb802d462d3fe04fafc Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 20:31:38 +0000 Subject: [PATCH 039/293] fix(patches): restored dropped + markers in puppeteer-core patch Eight closing lines of added `.send(...).catch(debugCatchError)` chains in the FrameManager, WebWorker, and #doAcquireWorlds hunks had lost their unified-diff `+` prefix, so they read as context lines. A fresh `bun patch` application searched upstream for those lines in the wrong place and rejected the patch, leaving the un-stealthed puppeteer-core installed. Fixes #5296 --- patches/puppeteer-core@25.3.0.patch | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/patches/puppeteer-core@25.3.0.patch b/patches/puppeteer-core@25.3.0.patch index 0bf86a5b2..7c47c0c1d 100644 --- a/patches/puppeteer-core@25.3.0.patch +++ b/patches/puppeteer-core@25.3.0.patch @@ -352,7 +352,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + client.send('Page.addScriptToEvaluateOnNewDocument', { + source: `//# sourceURL=${PuppeteerURL.INTERNAL_URL}`, + worldName: UTILITY_WORLD_NAME, - }).catch(debugCatchError), ++ }).catch(debugCatchError), ...(frame ? Array.from(this.#scriptsToEvaluateOnNewDocument.values()) : []).map(script => { @@ -445,7 +445,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + worldName: UTILITY_WORLD_NAME, + grantUniveralAccess: true, + }) - .catch(debugCatchError); ++ .catch(debugCatchError); + const utilityId = iso && typeof iso.executionContextId === 'number' ? iso.executionContextId : undefined; + if (utilityId !== undefined) { + this.#onExecutionContextCreated({ @@ -486,7 +486,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + } + } + catch (error) { - debugCatchError(error); ++ debugCatchError(error); + } + } + // xxx-stealth: resolve a frame's MAIN-world execution context id without @@ -511,7 +511,7 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + expression: 'globalThis', + serializationOptions: { serialization: 'idOnly' }, + }) - .catch(debugCatchError); ++ .catch(debugCatchError); + return parse(globalThis?.result?.objectId); + } + if (utilityId === undefined) { @@ -523,21 +523,21 @@ index 2322aa136a47b446e2a7b2c4f0bc751f2fb821d0..2116367ee34bb86948bf956c2afa2f69 + contextId: utilityId, + serializationOptions: { serialization: 'idOnly' }, + }) - .catch(debugCatchError); ++ .catch(debugCatchError); + const utilDocObjectId = utilDoc?.result?.objectId; + if (typeof utilDocObjectId !== 'string') { + return undefined; + } + const described = await session + .send('DOM.describeNode', { objectId: utilDocObjectId }) - .catch(debugCatchError); ++ .catch(debugCatchError); + const backendNodeId = described?.node?.backendNodeId; + if (typeof backendNodeId !== 'number') { + return undefined; + } + const mainNode = await session + .send('DOM.resolveNode', { backendNodeId }) - .catch(debugCatchError); ++ .catch(debugCatchError); + return parse(mainNode?.object?.objectId); } async #createIsolatedWorld(session, name) { @@ -605,7 +605,7 @@ index 3d68f887920ded269eb641273a5a13dee235ae1d..dcdd86c8697c0dbd2dd2162c9a739dd9 + this.#world.setContext(new ExecutionContext(client, { id }, this.#world)); + } + }) - .catch(debugCatchError); ++ .catch(debugCatchError); this.#client.once('Inspector.workerScriptLoaded', () => { this.#workerLoaded.resolve(); }); From 539c132094c8cfa54a576c1a8d6344bbfacb69c2 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 15 Jul 2026 00:17:00 +0200 Subject: [PATCH 040/293] feat(python/robomp): refined issue triage prompts with stricter bug classification rules - Updated the system prompt to require additional bug-gate checks, including repo-owned-defect and premise-verification before labeling a report as `bug`. - Added non-bug routing guidance for upstream-caused failures, environment/user errors, duplicate audit batches, and out-of-scope or already-possible scenarios. - Adjusted host-tool and issue kick-off prompt instructions to call out upstream vs this-repo cause checks and expanded wontfix rationale wording. --- python/robomp/src/prompts/host_tools.toml | 2 +- python/robomp/src/prompts/kickoff_issue.md | 3 ++- python/robomp/src/prompts/system_append.md | 15 ++++++++++----- 3 files changed, 13 insertions(+), 7 deletions(-) diff --git a/python/robomp/src/prompts/host_tools.toml b/python/robomp/src/prompts/host_tools.toml index bca47df2b..32ab0773f 100644 --- a/python/robomp/src/prompts/host_tools.toml +++ b/python/robomp/src/prompts/host_tools.toml @@ -92,7 +92,7 @@ branch_slug = "Kebab-case slug, 1-50 chars `[a-z0-9-]`, no leading/trailing/doub [classify_issue.next_steps] bug = "reproduce → diagnose → fix → PR" -wontfix = "post one gh_post_comment explaining the design rationale and what evidence would change the assessment; no repro, no PR; the maintainer decides whether to close" +wontfix = "post one gh_post_comment explaining the design rationale or upstream cause and what evidence would change the assessment; no repro, no PR; the maintainer decides whether to close" documentation = "fix the docs and open a PR using the four-section template" question = "answer in a single gh_post_comment; no PR, no repro" enhancement = "post one thoughtful gh_post_comment on feasibility/scope; no PR" diff --git a/python/robomp/src/prompts/kickoff_issue.md b/python/robomp/src/prompts/kickoff_issue.md index a5a851394..20f92a48b 100644 --- a/python/robomp/src/prompts/kickoff_issue.md +++ b/python/robomp/src/prompts/kickoff_issue.md @@ -19,7 +19,8 @@ the classification calls for code. Drive the todo list to completion: `fetch_issue_thread`, then call `classify_issue(primary=..., priority=..., functional=[...], rationale=...)`. Apply the **merit gate** from the system prompt before picking `bug`: - broken contract + demonstrated impact + not a deliberate tradeoff. + broken contract, demonstrated impact, deliberate-tradeoff check, upstream + vs this-repo cause, and premise verification must ALL pass. You NEVER post a comment, push, or open a PR before this step. 2. **Follow the workflow branch** the classification dictates — see the system diff --git a/python/robomp/src/prompts/system_append.md b/python/robomp/src/prompts/system_append.md index f0010dab0..bbf45111b 100644 --- a/python/robomp/src/prompts/system_append.md +++ b/python/robomp/src/prompts/system_append.md @@ -15,7 +15,7 @@ Pick exactly ONE primary label per issue: | Label | When | |---|---| | `bug` | Existing behavior is broken: crashes, errors, regressions, "doesn't work". Repro + fix + PR. | -| `wontfix` | Report is technically accurate but the behavior is intentional design, a documented tradeoff, or the fix costs more than the problem it solves. Explain; no PR. | +| `wontfix` | Report may be technically accurate but the behavior is intentional design, a documented tradeoff, an upstream defect (model/provider/runtime/dependency), or the fix costs more than the problem it solves. Explain; no PR. | | `documentation` | Docs are missing, incorrect, or outdated. Fix + PR (treat the doc as the code). | | `enhancement` | Feature request or improvement to existing behavior. Discuss; do NOT implement uninvited. | | `proposal` | Design/process proposal requiring maintainer decision. Comment with thoughts; no PR. | @@ -25,17 +25,22 @@ Pick exactly ONE primary label per issue: ## Merit gate — `bug` vs `wontfix` vs `enhancement` -A report earns `bug` ONLY when ALL THREE hold. Address them in the `rationale`: +A report earns `bug` ONLY when ALL of these hold. Address each in the `rationale`: 1. **Broken contract.** The behavior contradicts documented behavior or what a reasonable user doing real work would expect — not merely what a spec, standard, or filesystem *permits*. "Paths may legally contain `:`, therefore the tool must parse them" is spec-lawyering, not a broken contract. 2. **Demonstrated impact.** The reporter hit this doing real work, or users plausibly will. An input constructed solely to trigger the report is not impact, and neither is a failure mode discovered by *reading source code* rather than running the tool. Elaborate analysis — tables, line-cited "Evidence" sections, N-of-N repro counts, "Acceptance criteria" — measures the reporter's effort, NEVER the problem's severity. A meticulous report about a non-problem is still a non-problem. -3. **Not a deliberate tradeoff.** Check whether the current behavior was *chosen* — docs, code comments, git history, prior issues. Prompt policies, UX decisions, and guardrails against known failure modes are design, not defects, even when a user dislikes the consequence. Behavior originating upstream (a model's RLHF quirks, a provider API, a dependency) is not this repo's bug. +3. **Not a deliberate tradeoff.** Check whether the current behavior was *chosen* — docs, code comments, git history, prior issues. Prompt policies, UX decisions, guardrails against known failure modes, even joke assets are design, not defects, when a user dislikes the consequence. +4. **This repo's defect.** The cause lives in this codebase — not in a model's behavior (looping, garbage output, ignoring tools: RLHF quirks are the model vendor's problem), a provider outage, npm/mirror lag, a runtime or terminal/font bug, or a dependency. When the defect is upstream, classify `wontfix` even when a client-side workaround is feasible — this repo does not accumulate workarounds for other people's bugs uninvited. +5. **True premise.** Verify the reporter's core factual claims against the repo before accepting them: the "bundled" component actually ships, the "wrong" number is actually wrong, the cited code exists and does what the report says. AI-generated reports and automated security scanners routinely hallucinate components, code paths, and vulnerabilities. False premise → `invalid`, stating plainly which claim failed verification. Common shapes that fail the gate: -- **Audit reports.** Issue reads like a code review: exhaustive citations, hypothetical failure paths, "Open questions", no first-person failure. Classify by what the finding *is* (`wontfix` for by-design, `enhancement` for hardening ideas) — never `bug` on citation volume alone. +- **Audit / batch reports.** Issue reads like a code review: exhaustive citations, hypothetical failure paths, "Open questions", no first-person failure — or arrives as one of several near-identical filings from the same author (`[audit]` prefixes, serial-numbered bodies). The maintainer does not accept batch issues. Classify by what the finding *is* (`wontfix` for by-design, `enhancement` for hardening ideas, `duplicate` citing the sibling for repeat filings) — never `bug` on citation volume alone. - **Niche config + trivial workaround.** Non-default option, exotic environment, and a one-line workaround exists → `wontfix`, whatever the claimed severity. - **Design complaints dressed as bugs.** Reporter wants *different* behavior → `enhancement` / `proposal`, even when the title screams "bug". The reporter's framing NEVER binds your classification. +- **Environment / user error.** Unsupported runtime version, stale package cache, registry lag, feature misuse (e.g. exiting a mode never entered) → `question` when you can name the remedy, `invalid` when there is nothing actionable. One comment stating cause and fix on *their* side; never a code change. +- **Already possible.** The ask is served by existing config, settings, or the extension API → `question`; point at the exact mechanism. +- **Out of scope.** Belongs in a different project or an extension → `wontfix` / `enhancement`; name where it belongs. A maintainer's "PRs welcome" on a prior similar issue is an invitation to *contributors*, NEVER authorization for you to implement. Torn between `bug` + `prio:p3` and `wontfix`? Pick `wontfix`: a maintainer flips it with one comment ("@{{bot_login}} fix it anyway"), but an unwanted PR wastes review time and lands code nobody asked for. @@ -89,7 +94,7 @@ ONE `gh_post_comment` engaging with the request: ONE `gh_post_comment`: - Acknowledge what is technically accurate in the report — no strawmanning. -- Explain the design rationale or tradeoff that makes the current behavior intentional. Cite code/docs by path. +- Explain why it will not be fixed here: the design rationale or tradeoff that makes the behavior intentional, or the upstream component that actually owns the defect. Cite code/docs by path. - Name what evidence WOULD change the assessment (a real failing workflow, a documented contract the behavior violates). - Defer the final call to the maintainer; do not close the issue. From ea320d745beee2653c399032f5c7eae7f41396f7 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 14 Jul 2026 22:21:53 +0000 Subject: [PATCH 041/293] fix(bash): resolved nested internal URLs - Tracked quote context independently inside command substitutions. - Covered unquoted skill URLs nested under outer double quotes. Fixes #5535 --- packages/coding-agent/CHANGELOG.md | 4 +++ .../coding-agent/src/tools/bash-skill-urls.ts | 28 ++++++++++++++++++- .../test/tools/bash-skill-urls.test.ts | 10 +++++++ 3 files changed, 41 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 9d79d33d6..05bfa6114 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Bash internal URLs remaining unresolved when used as unquoted arguments inside command substitutions ([#5535](https://github.com/can1357/oh-my-pi/issues/5535)). + ## [16.5.2] - 2026-07-14 ### Breaking Changes diff --git a/packages/coding-agent/src/tools/bash-skill-urls.ts b/packages/coding-agent/src/tools/bash-skill-urls.ts index 135990b21..b4de3bd4b 100644 --- a/packages/coding-agent/src/tools/bash-skill-urls.ts +++ b/packages/coding-agent/src/tools/bash-skill-urls.ts @@ -142,7 +142,14 @@ function unquoteToken(token: string): string { } function isInsideShellQuote(command: string, index: number): boolean { - let quote: "'" | '"' | undefined; + type ShellQuote = "'" | '"' | undefined; + interface CommandSubstitution { + outerQuote: ShellQuote; + depth: number; + } + + let quote: ShellQuote; + const substitutions: CommandSubstitution[] = []; for (let i = 0; i < index; i++) { const char = command[i]; if (char === "\\" && quote !== "'") { @@ -155,6 +162,25 @@ function isInsideShellQuote(command: string, index: number): boolean { } if (char === '"' && quote !== "'") { quote = quote === '"' ? undefined : '"'; + continue; + } + if (char === "$" && command[i + 1] === "(" && quote !== "'") { + substitutions.push({ outerQuote: quote, depth: 1 }); + quote = undefined; + i++; + continue; + } + if (quote !== undefined) continue; + + const substitution = substitutions.at(-1); + if (!substitution) continue; + if (char === "(") { + substitution.depth++; + } else if (char === ")") { + substitution.depth--; + if (substitution.depth === 0) { + quote = substitutions.pop()?.outerQuote; + } } } return quote !== undefined; diff --git a/packages/coding-agent/test/tools/bash-skill-urls.test.ts b/packages/coding-agent/test/tools/bash-skill-urls.test.ts index 9ea65e242..d84a41409 100644 --- a/packages/coding-agent/test/tools/bash-skill-urls.test.ts +++ b/packages/coding-agent/test/tools/bash-skill-urls.test.ts @@ -206,6 +206,16 @@ describe("expandInternalUrls", () => { ); }); + it("expands an unquoted URL inside a double-quoted command substitution", async () => { + const skills = [createSkill("valid-skill", "/tmp/skills/valid-skill")]; + const command = 'echo "$(realpath skill://valid-skill/SKILL.md 2>&1)"'; + const expectedPath = path.join(skills[0].baseDir, "SKILL.md"); + + await expect(expandInternalUrls(command, { skills })).resolves.toBe( + `echo "$(realpath ${shellEscape(expectedPath)} 2>&1)"`, + ); + }); + it("leaves literal internal URLs embedded in quoted text unchanged", async () => { const router = createInternalRouter({ "memory://root/summary.md": { sourcePath: "/tmp/memories/summary.md" }, From 3e6ae36c7c4bb3544021d320fb6f876c81e5048c Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 15 Jul 2026 00:30:15 +0200 Subject: [PATCH 042/293] feat(robomp): added repo-scoped issue search to issue triage flow - Added `search_issues` support to the GitHub backend and client, including `state_reason` and `is_pull_request` in issue summaries. - Added the `gh_search_issues` host tool with repo-prefixed query handling, non-empty/restricted `repo:` validation, default and bounded `limit` values, and inbound issue filtering. - Added proxy integration for issue search with a new `/gh/v1/search_issues` endpoint and matching proxy-client method/response parsing. - Updated triage prompts to perform pre-`classify_issue` duplicate and already-fixed checks via search, and added tests for search query formatting, validation, and state-aware match rendering. --- python/robomp/src/github_backend.py | 2 + python/robomp/src/github_client.py | 58 ++++++++++++----- python/robomp/src/host_tools.py | 70 ++++++++++++++++++++ python/robomp/src/prompts/host_tools.toml | 7 ++ python/robomp/src/prompts/kickoff_issue.md | 4 +- python/robomp/src/prompts/system_append.md | 9 ++- python/robomp/src/prompts/todo_phases.toml | 1 + python/robomp/src/proxy/server.py | 10 +++ python/robomp/src/proxy_client.py | 10 +++ python/robomp/src/worker.py | 2 +- python/robomp/tests/test_host_tools.py | 76 ++++++++++++++++++++++ python/robomp/tests/test_proxy_client.py | 27 ++++++++ 12 files changed, 257 insertions(+), 19 deletions(-) diff --git a/python/robomp/src/github_backend.py b/python/robomp/src/github_backend.py index 8f9604077..b9b09ba6f 100644 --- a/python/robomp/src/github_backend.py +++ b/python/robomp/src/github_backend.py @@ -46,6 +46,8 @@ class GitHubBackend(Protocol): limit: int = 30, ) -> list[IssueSummary]: ... + async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]: ... + async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: ... async def list_review_comments(self, repo: str, pr_number: int) -> list[ReviewCommentInfo]: ... diff --git a/python/robomp/src/github_client.py b/python/robomp/src/github_client.py index 49053f925..6ad2bf222 100644 --- a/python/robomp/src/github_client.py +++ b/python/robomp/src/github_client.py @@ -116,6 +116,10 @@ class IssueSummary: updated_at: str created_at: str html_url: str + # `completed` / `not_planned` / `reopened` when closed; empty otherwise. + state_reason: str = "" + # Search results mix issues and PRs; list_issues always yields issues. + is_pull_request: bool = False @dataclass(slots=True, frozen=True) @@ -335,24 +339,26 @@ class GitHubClient: for item in data or []: if "pull_request" in item: continue # GitHub's /issues endpoint also returns PRs; skip them. - user = item.get("user") or {} - labels_raw = item.get("labels") or [] - out.append( - IssueSummary( - repo=repo, - number=int(item["number"]), - title=str(item.get("title") or ""), - state=str(item.get("state") or "open"), - author=str(user.get("login") or ""), - labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw), - comments=int(item.get("comments") or 0), - updated_at=str(item.get("updated_at") or ""), - created_at=str(item.get("created_at") or ""), - html_url=str(item.get("html_url") or ""), - ) - ) + out.append(_summary_from_item(repo, item)) return out + async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]: + """Search issues AND pull requests in `repo` using GitHub issue-search syntax. + + `query` takes bare keywords plus qualifiers (`is:pr`, `is:closed`, + `label:bug`, `in:title`, …); the `repo:` scope is applied here. Results + come back in GitHub's best-match order. `limit` is capped at 30 — this + serves triage lookups (duplicates, prior fixes), not pagination. + """ + per_page = max(1, min(int(limit), 30)) + data = await self.request( + "GET", + "/search/issues", + params={"q": f"repo:{repo} {query}".strip(), "per_page": per_page}, + ) + items = (data or {}).get("items") or [] + return [_summary_from_item(repo, item) for item in items] + async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: data = await self.request("GET", f"/repos/{repo}/issues/{number}/comments", params={"per_page": 100}) return [_comment_from_payload(item) for item in (data or [])] @@ -576,6 +582,26 @@ def _pr_review_from_payload(data: Mapping[str, Any]) -> PullRequestReviewInfo: ) +def _summary_from_item(repo: str, item: Mapping[str, Any]) -> IssueSummary: + """Build an `IssueSummary` from a REST issue object (list or search shape).""" + user = item.get("user") or {} + labels_raw = item.get("labels") or [] + return IssueSummary( + repo=repo, + number=int(item["number"]), + title=str(item.get("title") or ""), + state=str(item.get("state") or "open"), + author=str(user.get("login") or ""), + labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw), + comments=int(item.get("comments") or 0), + updated_at=str(item.get("updated_at") or ""), + created_at=str(item.get("created_at") or ""), + html_url=str(item.get("html_url") or ""), + state_reason=str(item.get("state_reason") or ""), + is_pull_request="pull_request" in item, + ) + + def _pr_file_from_payload(data: Mapping[str, Any]) -> PullRequestFileInfo: return PullRequestFileInfo( path=str(data.get("filename") or data.get("path") or ""), diff --git a/python/robomp/src/host_tools.py b/python/robomp/src/host_tools.py index 3fbf84af4..ac4bf3633 100644 --- a/python/robomp/src/host_tools.py +++ b/python/robomp/src/host_tools.py @@ -1256,6 +1256,75 @@ def _build_fetch_thread(bindings: ToolBindings) -> HostTool[Any, Any]: ) +# ---------- gh_search_issues ---------- +_REPO_QUALIFIER_RE = re.compile(r"(?i)\brepo:") + + +def _build_search_issues(bindings: ToolBindings) -> HostTool[Any, Any]: + """Read-only issue/PR search scoped to the current repo. + + Exists so triage can find duplicates and already-merged fixes instead of + classifying blind; the inbound issue itself is filtered out of results. + """ + + def execute(args: dict[str, Any], _ctx: HostToolContext[Any]) -> str: + query = args.get("query") + if not isinstance(query, str) or not query.strip(): + msg = "gh_search_issues requires a non-empty 'query'." + _audit(bindings, "gh_search_issues", args, error=msg) + _raise_command(msg) + query = query.strip() + if _REPO_QUALIFIER_RE.search(query): + msg = "gh_search_issues scopes to the current repo automatically; drop the 'repo:' qualifier." + _audit(bindings, "gh_search_issues", args, error=msg) + _raise_command(msg) + limit_raw = args.get("limit") + limit = max(1, min(int(limit_raw), 20)) if isinstance(limit_raw, int) else 10 + try: + results = _run_coro( + bindings.loop, + bindings.github.search_issues(bindings.repo.full_name, query, limit=limit), + ) + except GitHubError as exc: + _audit(bindings, "gh_search_issues", args, error=str(exc)) + _raise_command(f"GitHub search failed: {exc.status} {exc.message}") + results = [s for s in results if s.is_pull_request or s.number != bindings.issue.number] + if not results: + _audit(bindings, "gh_search_issues", args, result={"matches": 0}) + return f"No issues or PRs in {bindings.repo.full_name} match {query!r}." + lines = [f"# {len(results)} match(es) for {query!r} in {bindings.repo.full_name}"] + for s in results: + kind = "PR" if s.is_pull_request else "issue" + state = f"{s.state} ({s.state_reason})" if s.state_reason else s.state + labels = f" [{', '.join(s.labels)}]" if s.labels else "" + lines.append( + f"- #{s.number} ({kind}, {state}) {s.title} — @{s.author}, updated {s.updated_at[:10]}{labels}" + ) + _audit(bindings, "gh_search_issues", args, result={"matches": len(results)}) + return "\n".join(lines) + + return host_tool( + name="gh_search_issues", + description=persona.host_tool_description("gh_search_issues"), + parameters={ + "type": "object", + "properties": { + "query": { + "type": "string", + "description": persona.host_tool_parameter_description("gh_search_issues", "query"), + }, + "limit": { + "type": "integer", + "description": persona.host_tool_parameter_description("gh_search_issues", "limit"), + }, + }, + "required": ["query"], + "additionalProperties": False, + }, + execute=execute, + ) + + _PRIMARY_TYPES = ("bug", "enhancement", "question", "proposal", "documentation", "wontfix", "invalid", "duplicate") _AUTO_PR_CLASSIFICATIONS = frozenset({"bug", "documentation"}) _PRIORITIES = ("prio:p0", "prio:p1", "prio:p2", "prio:p3") @@ -1835,6 +1904,7 @@ def build(bindings: ToolBindings) -> tuple[HostTool[Any, Any], ...]: _build_mark_unable(bindings), _build_abort_task(bindings), _build_fetch_thread(bindings), + _build_search_issues(bindings), ) diff --git a/python/robomp/src/prompts/host_tools.toml b/python/robomp/src/prompts/host_tools.toml index 32ab0773f..80b0aa211 100644 --- a/python/robomp/src/prompts/host_tools.toml +++ b/python/robomp/src/prompts/host_tools.toml @@ -72,6 +72,13 @@ reason = "Internal diagnosis for the operator. Concrete, specific, blameless. NE [fetch_issue_thread] description = "Refetch the originating issue and its comments. Use sparingly." +[gh_search_issues] +description = "Search issues AND pull requests in the current repo (GitHub issue-search syntax; repo scope applied automatically). Use during triage to find duplicates and to check whether a merged PR already fixed the reported problem before classifying." + +[gh_search_issues.parameters] +query = "GitHub issue-search syntax: bare keywords plus qualifiers like `is:pr`, `is:closed`, `is:merged`, `label:bug`, `in:title`, `author:`. NEVER include a `repo:` qualifier — scope is applied automatically." +limit = "Max results, 1-20. Default 10." + [set_issue_labels] description = "Append labels to the originating issue/PR. NEVER removes existing labels." diff --git a/python/robomp/src/prompts/kickoff_issue.md b/python/robomp/src/prompts/kickoff_issue.md index 20f92a48b..4d5b6bd18 100644 --- a/python/robomp/src/prompts/kickoff_issue.md +++ b/python/robomp/src/prompts/kickoff_issue.md @@ -16,7 +16,9 @@ Worktree is at cwd; the branch above is checked out and ready for commits **if** the classification calls for code. Drive the todo list to completion: 1. **Triage first.** Read the body and any comments via `read` / - `fetch_issue_thread`, then call + `fetch_issue_thread`. Run `gh_search_issues` for duplicates and + already-merged fixes — the reporter may be on an older release than your + worktree. Then call `classify_issue(primary=..., priority=..., functional=[...], rationale=...)`. Apply the **merit gate** from the system prompt before picking `bug`: broken contract, demonstrated impact, deliberate-tradeoff check, upstream diff --git a/python/robomp/src/prompts/system_append.md b/python/robomp/src/prompts/system_append.md index bbf45111b..5212c9ed9 100644 --- a/python/robomp/src/prompts/system_append.md +++ b/python/robomp/src/prompts/system_append.md @@ -21,7 +21,14 @@ Pick exactly ONE primary label per issue: | `proposal` | Design/process proposal requiring maintainer decision. Comment with thoughts; no PR. | | `question` | How-to, clarification, or usage question. Answer in one comment. | | `invalid` | Spam, off-topic, or not actionable. One brief explanatory comment. | -| `duplicate` | Clear duplicate of another issue. Cite the original; no PR. | +| `duplicate` | Duplicate of another issue, or already fixed by a merged PR / newer release. Cite the original or the fixing PR; no new PR. | + +## Duplicate & already-fixed check + +Before `classify_issue`, run `gh_search_issues` with the report's key terms (retry with synonyms and an `is:pr` variant — one search proves nothing): + +- **Prior issue on the same problem** → `duplicate`, cite it. A prior closure as not-planned/`wontfix` on the same complaint is binding precedent — adopt that verdict; NEVER relitigate it. +- **Already fixed.** Your worktree is the CURRENT default branch; reporters often run older releases. When the reported version lags the latest release (topmost released section of the relevant `packages/*/CHANGELOG.md`), check the changelog and merged PRs (`is:pr is:merged `) for an existing fix, and try the repro against the worktree — failing on the reporter's version but passing here means it is already fixed. Classify `duplicate`: cite the fixing PR, name the release carrying it (or say it ships in the next release when still under `[Unreleased]`), and tell the reporter to update. NEVER re-fix what main already fixed. ## Merit gate — `bug` vs `wontfix` vs `enhancement` diff --git a/python/robomp/src/prompts/todo_phases.toml b/python/robomp/src/prompts/todo_phases.toml index 660eb8039..3a451a15a 100644 --- a/python/robomp/src/prompts/todo_phases.toml +++ b/python/robomp/src/prompts/todo_phases.toml @@ -2,6 +2,7 @@ name = "Classify" tasks = [ "Read the issue body + every prior comment", + "gh_search_issues for duplicates and already-merged fixes", "Call classify_issue with primary type + labels", ] diff --git a/python/robomp/src/proxy/server.py b/python/robomp/src/proxy/server.py index bf6ba4cf7..923753b40 100644 --- a/python/robomp/src/proxy/server.py +++ b/python/robomp/src/proxy/server.py @@ -513,6 +513,16 @@ def create_proxy_app(settings: Settings) -> FastAPI: return _gh_error_response(exc) return JSONResponse({"items": [_serialize(s) for s in items]}) + @app.get("/gh/v1/search_issues") + async def search_issues(request: Request, repo: str, q: str, limit: int = 10) -> JSONResponse: + await _authenticate(request) + github: GitHubClient = request.app.state.github + try: + items = await github.search_issues(repo, q, limit=limit) + except GitHubError as exc: + return _gh_error_response(exc) + return JSONResponse({"items": [_serialize(s) for s in items]}) + @app.get("/gh/v1/comments") async def list_comments(request: Request, repo: str, number: int) -> JSONResponse: await _authenticate(request) diff --git a/python/robomp/src/proxy_client.py b/python/robomp/src/proxy_client.py index e65e85311..a2069c950 100644 --- a/python/robomp/src/proxy_client.py +++ b/python/robomp/src/proxy_client.py @@ -204,6 +204,14 @@ class GitHubProxyClient: ) return [_issue_summary_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []] + async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]: + data = await self._request( + "GET", + "/gh/v1/search_issues", + params={"repo": repo, "q": query, "limit": limit}, + ) + return [_issue_summary_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []] + async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: data = await self._request("GET", "/gh/v1/comments", params={"repo": repo, "number": number}) return [_comment_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []] @@ -498,6 +506,8 @@ def _issue_summary_from(data: Any) -> IssueSummary: updated_at=str(data.get("updated_at") or ""), created_at=str(data.get("created_at") or ""), html_url=str(data.get("html_url") or ""), + state_reason=str(data.get("state_reason") or ""), + is_pull_request=bool(data.get("is_pull_request")), ) diff --git a/python/robomp/src/worker.py b/python/robomp/src/worker.py index 335f15f0d..ca14349eb 100644 --- a/python/robomp/src/worker.py +++ b/python/robomp/src/worker.py @@ -198,7 +198,7 @@ def _ensure_agent_run_dir() -> None: return try: run_dir.mkdir(parents=True, exist_ok=True) - for root, dirs, files in os.walk(run_dir): + for root, _dirs, files in os.walk(run_dir): root_path = Path(root) os.chown(root_path, -1, gid) root_path.chmod(0o2770) diff --git a/python/robomp/tests/test_host_tools.py b/python/robomp/tests/test_host_tools.py index 54f9d4c5e..102074c90 100644 --- a/python/robomp/tests/test_host_tools.py +++ b/python/robomp/tests/test_host_tools.py @@ -861,6 +861,82 @@ def test_classify_issue_wontfix_takes_comment_only_path(db: Database, tmp_path: assert row is not None and row.classification == "wontfix" +def test_gh_search_issues_scopes_repo_and_renders_matches(db: Database, tmp_path: Path) -> None: + """Search auto-prefixes the repo scope, surfaces PR/state_reason so triage can + spot prior fixes and not-planned precedents, and filters the inbound issue.""" + captured: dict[str, Any] = {} + + def handler(request: httpx.Request) -> httpx.Response: + captured["q"] = request.url.params["q"] + return httpx.Response( + 200, + json={ + "total_count": 3, + "items": [ + { + "number": 42, # the inbound issue itself — must be filtered + "title": "boom", + "state": "open", + "user": {"login": "alice"}, + "labels": [], + "comments": 0, + "updated_at": "2026-07-01T00:00:00Z", + "created_at": "2026-07-01T00:00:00Z", + "html_url": "https://example/42", + }, + { + "number": 30, + "title": "same crash on resize", + "state": "closed", + "state_reason": "not_planned", + "user": {"login": "bob"}, + "labels": [{"name": "wontfix"}], + "comments": 3, + "updated_at": "2026-06-01T00:00:00Z", + "created_at": "2026-05-01T00:00:00Z", + "html_url": "https://example/30", + }, + { + "number": 31, + "title": "fix: resize crash", + "state": "closed", + "state_reason": "completed", + "user": {"login": "bot"}, + "labels": [], + "comments": 1, + "updated_at": "2026-06-02T00:00:00Z", + "created_at": "2026-06-02T00:00:00Z", + "html_url": "https://example/pull/31", + "pull_request": {"url": "https://example/pull/31"}, + }, + ], + }, + ) + + bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(handler)) + try: + tool = next(x for x in build(bindings) if x.name == "gh_search_issues") + result = tool.execute({"query": "resize crash"}, _ctx()) + finally: + _stop_loop(loop, t) + assert captured["q"] == "repo:octo/widget resize crash" + assert "#42" not in result # inbound issue filtered out + assert "#30 (issue, closed (not_planned))" in result + assert "#31 (PR, closed (completed))" in result + + +def test_gh_search_issues_rejects_repo_qualifier_and_empty_query(db: Database, tmp_path: Path) -> None: + bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500))) + try: + tool = next(x for x in build(bindings) if x.name == "gh_search_issues") + with pytest.raises(RpcCommandError): + tool.execute({"query": "repo:evil/elsewhere secrets"}, _ctx()) + with pytest.raises(RpcCommandError): + tool.execute({"query": " "}, _ctx()) + finally: + _stop_loop(loop, t) + + def test_classify_issue_rejects_bug_without_priority(db: Database, tmp_path: Path) -> None: bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500))) try: diff --git a/python/robomp/tests/test_proxy_client.py b/python/robomp/tests/test_proxy_client.py index ed10b1d48..ff9c03eb1 100644 --- a/python/robomp/tests/test_proxy_client.py +++ b/python/robomp/tests/test_proxy_client.py @@ -251,6 +251,29 @@ def round_trip_app(proxy_settings: Settings): } ], ) + if path == "/search/issues" and req.method == "GET": + assert req.url.params["q"].startswith("repo:octo/widget ") + return httpx.Response( + 200, + json={ + "total_count": 1, + "items": [ + { + "number": 9, + "title": "fixed it", + "state": "closed", + "state_reason": "completed", + "user": {"login": "bob"}, + "labels": [{"name": "bug"}], + "comments": 2, + "updated_at": "2026-02-01T00:00:00Z", + "created_at": "2026-01-15T00:00:00Z", + "html_url": "https://example/9", + "pull_request": {"url": "https://example/pull/9"}, + } + ], + }, + ) if path == "/repos/octo/widget/issues/1/comments" and req.method == "GET": return httpx.Response( 200, @@ -365,6 +388,10 @@ async def test_round_trip_all_endpoints(round_trip_app) -> None: issues = await client.list_issues("octo/widget") assert len(issues) == 1 and isinstance(issues[0], IssueSummary) + found = await client.search_issues("octo/widget", "colon selector is:pr") + assert len(found) == 1 and isinstance(found[0], IssueSummary) + assert found[0].is_pull_request and found[0].state_reason == "completed" + comments = await client.list_comments("octo/widget", 1) assert len(comments) == 1 and isinstance(comments[0], CommentInfo) From e426186e4690812b5c91f88a89f1df8ad00b1b00 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 15 Jul 2026 00:46:08 +0200 Subject: [PATCH 043/293] feat(robomp): added local issue indexing and commit-search tool support - Added `ROBOMP_ISSUE_INDEX_SYNC_SECONDS` configuration and lifecycle-managed issue indexing through startup/shutdown hooks. - Added issue/PR index tables with FTS5 triggers plus upsert and keyword/filter search helpers for indexed records. - Added GitHub backend/proxy support for `IssueIndexEntry` and issue-index page retrieval, including webhook ingestion and periodic watermark-driven sync. - Updated `gh_search_issues` to prefer local index queries when synchronized and added `search_commits` host tool with query modes and validation. --- python/robomp/.env.example | 10 + python/robomp/src/config.py | 6 + python/robomp/src/db.py | 179 +++++++++++++++ python/robomp/src/github_backend.py | 9 + python/robomp/src/github_client.py | 106 +++++++++ python/robomp/src/host_tools.py | 187 +++++++++++++-- python/robomp/src/issue_index.py | 251 +++++++++++++++++++++ python/robomp/src/prompts/host_tools.toml | 13 +- python/robomp/src/prompts/system_append.md | 4 +- python/robomp/src/proxy/server.py | 16 ++ python/robomp/src/proxy_client.py | 36 +++ python/robomp/src/server.py | 17 +- python/robomp/tests/conftest.py | 4 + python/robomp/tests/test_host_tools.py | 95 +++++++- python/robomp/tests/test_issue_index.py | 189 ++++++++++++++++ python/robomp/tests/test_server.py | 36 +++ 16 files changed, 1129 insertions(+), 29 deletions(-) create mode 100644 python/robomp/src/issue_index.py create mode 100644 python/robomp/tests/test_issue_index.py diff --git a/python/robomp/.env.example b/python/robomp/.env.example index 692206eec..a338cdf82 100644 --- a/python/robomp/.env.example +++ b/python/robomp/.env.example @@ -164,6 +164,16 @@ ROBOMP_QUESTION_AUTOCLOSE_HOURS=4 # multi-hour close window. ROBOMP_QUESTION_AUTOCLOSE_SCAN_SECONDS=60 +# ============================================================================= +# --- Local issue search index --- +# ============================================================================= +# `gh_search_issues` answers from a local SQLite FTS mirror of every issue/PR +# in the allowlisted repos. Webhooks keep it fresh in real time; this interval +# controls the periodic GitHub reconcile (first pass backfills each repo). +# Set <= 0 to disable the reconciler — the tool then uses the GitHub search +# API until a repo has been backfilled. +ROBOMP_ISSUE_INDEX_SYNC_SECONDS=900 + # Path or command name for the omp binary inside the container. The shipped # image installs a shim that invokes Bun against the mounted pi checkout. ROBOMP_OMP_COMMAND=omp diff --git a/python/robomp/src/config.py b/python/robomp/src/config.py index 765507905..046e7c656 100644 --- a/python/robomp/src/config.py +++ b/python/robomp/src/config.py @@ -147,6 +147,12 @@ class Settings(BaseSettings): question_autoclose_enabled: bool = Field(True, alias="ROBOMP_QUESTION_AUTOCLOSE_ENABLED") question_autoclose_hours: float = Field(4.0, alias="ROBOMP_QUESTION_AUTOCLOSE_HOURS") question_autoclose_scan_seconds: float = Field(60.0, alias="ROBOMP_QUESTION_AUTOCLOSE_SCAN_SECONDS") + # Local issue/PR search index. Webhooks keep it fresh in real time; this + # interval controls the periodic GitHub reconcile (first pass = full + # backfill of every allowlisted repo). <= 0 disables the reconciler — + # `gh_search_issues` then falls back to the remote search API until the + # repo has a sync watermark. + issue_index_sync_seconds: float = Field(900.0, alias="ROBOMP_ISSUE_INDEX_SYNC_SECONDS") # pi-natives build-output cache. Hardlinks pre-built # `packages/natives/native/*.node` (and its companions) into new diff --git a/python/robomp/src/db.py b/python/robomp/src/db.py index ffbbf2d2c..7fa13940d 100644 --- a/python/robomp/src/db.py +++ b/python/robomp/src/db.py @@ -12,6 +12,8 @@ from datetime import UTC, datetime, timedelta from pathlib import Path from typing import Any, Literal +from robomp.github_client import IssueIndexEntry + EventState = Literal["queued", "running", "done", "failed", "skipped"] INACTIVE_EVENT_STATES: tuple[EventState, ...] = ("done", "failed", "skipped") @@ -113,6 +115,53 @@ CREATE TABLE IF NOT EXISTS pending_closures ( ); CREATE INDEX IF NOT EXISTS pending_closures_state_close_at ON pending_closures(state, close_at); + +-- Local mirror of every issue/PR in allowlisted repos, kept fresh by webhook +-- upserts plus the periodic `IssueIndexSync` reconciler. `gh_search_issues` +-- serves from here so triage lookups cost no GitHub API calls. +CREATE TABLE IF NOT EXISTS issue_index ( + repo TEXT NOT NULL, + number INTEGER NOT NULL, + is_pr INTEGER NOT NULL DEFAULT 0, + title TEXT NOT NULL DEFAULT '', + body TEXT NOT NULL DEFAULT '', + state TEXT NOT NULL DEFAULT 'open', + state_reason TEXT NOT NULL DEFAULT '', + merged_at TEXT NOT NULL DEFAULT '', + author TEXT NOT NULL DEFAULT '', + labels_json TEXT NOT NULL DEFAULT '[]', + comments INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL DEFAULT '', + updated_at TEXT NOT NULL DEFAULT '', + html_url TEXT NOT NULL DEFAULT '', + PRIMARY KEY (repo, number) +); +CREATE INDEX IF NOT EXISTS issue_index_repo_updated + ON issue_index(repo, updated_at); + +CREATE VIRTUAL TABLE IF NOT EXISTS issue_index_fts USING fts5( + title, body, content='issue_index', content_rowid='rowid' +); + +CREATE TRIGGER IF NOT EXISTS issue_index_ai AFTER INSERT ON issue_index BEGIN + INSERT INTO issue_index_fts(rowid, title, body) VALUES (new.rowid, new.title, new.body); +END; +CREATE TRIGGER IF NOT EXISTS issue_index_ad AFTER DELETE ON issue_index BEGIN + INSERT INTO issue_index_fts(issue_index_fts, rowid, title, body) + VALUES ('delete', old.rowid, old.title, old.body); +END; +CREATE TRIGGER IF NOT EXISTS issue_index_au AFTER UPDATE ON issue_index BEGIN + INSERT INTO issue_index_fts(issue_index_fts, rowid, title, body) + VALUES ('delete', old.rowid, old.title, old.body); + INSERT INTO issue_index_fts(rowid, title, body) VALUES (new.rowid, new.title, new.body); +END; + +-- Per-repo reconcile watermark: the max `updated_at` the sync has fully +-- ingested. Absent row = repo never backfilled. +CREATE TABLE IF NOT EXISTS issue_index_sync ( + repo TEXT PRIMARY KEY, + last_synced TEXT NOT NULL +); """ @@ -1163,6 +1212,136 @@ class Database: ).fetchone() return _pending_closure_from_row(row) if row is not None else None + # ---- issue search index ---- + def upsert_issue_index(self, entry: IssueIndexEntry) -> None: + """Insert or refresh one issue/PR in the local search index.""" + with self._lock: + self._conn.execute( + """ + INSERT INTO issue_index + (repo, number, is_pr, title, body, state, state_reason, merged_at, + author, labels_json, comments, created_at, updated_at, html_url) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ON CONFLICT(repo, number) DO UPDATE SET + is_pr = excluded.is_pr, + title = excluded.title, + body = excluded.body, + state = excluded.state, + state_reason = excluded.state_reason, + merged_at = excluded.merged_at, + author = excluded.author, + labels_json = excluded.labels_json, + comments = excluded.comments, + created_at = excluded.created_at, + updated_at = excluded.updated_at, + html_url = excluded.html_url + """, + ( + entry.repo, + entry.number, + 1 if entry.is_pull_request else 0, + entry.title, + entry.body, + entry.state, + entry.state_reason, + entry.merged_at, + entry.author, + json.dumps(list(entry.labels), separators=(",", ":")), + entry.comments, + entry.created_at, + entry.updated_at, + entry.html_url, + ), + ) + + def search_issue_index( + self, + repo: str, + *, + keywords: Iterable[str] = (), + is_pr: bool | None = None, + state: str | None = None, + merged: bool | None = None, + label: str | None = None, + author: str | None = None, + limit: int = 10, + ) -> list[IssueIndexEntry]: + """Query the local index. Keywords go through FTS5 (bm25-ranked, AND + semantics); the remaining filters are exact. With no keywords, results + order by `updated_at` descending. + """ + conds = ["i.repo = ?"] + params: list[Any] = [repo] + if is_pr is not None: + conds.append("i.is_pr = ?") + params.append(1 if is_pr else 0) + if state is not None: + conds.append("i.state = ?") + params.append(state) + if merged is not None: + conds.append("i.merged_at != ''" if merged else "i.merged_at = ''") + if label is not None: + conds.append("EXISTS (SELECT 1 FROM json_each(i.labels_json) WHERE json_each.value = ?)") + params.append(label) + if author is not None: + conds.append("i.author = ?") + params.append(author) + terms = [t for t in keywords if t.strip()] + limit = max(1, min(int(limit), 50)) + with self._lock: + if terms: + # Quote every term so reporter text can never inject FTS5 syntax. + match = " ".join('"' + t.replace('"', '""') + '"' for t in terms) + sql = ( + "SELECT i.* FROM issue_index_fts f JOIN issue_index i ON i.rowid = f.rowid " + f"WHERE issue_index_fts MATCH ? AND {' AND '.join(conds)} " + "ORDER BY bm25(issue_index_fts) LIMIT ?" + ) + rows = self._conn.execute(sql, (match, *params, limit)).fetchall() + else: + sql = f"SELECT i.* FROM issue_index i WHERE {' AND '.join(conds)} ORDER BY i.updated_at DESC LIMIT ?" + rows = self._conn.execute(sql, (*params, limit)).fetchall() + return [_index_entry_from_row(row) for row in rows] + + def issue_index_watermark(self, repo: str) -> str | None: + """Max `updated_at` fully ingested for `repo`; None = never backfilled.""" + with self._lock: + row = self._conn.execute("SELECT last_synced FROM issue_index_sync WHERE repo = ?", (repo,)).fetchone() + return str(row["last_synced"]) if row is not None else None + + def set_issue_index_watermark(self, repo: str, last_synced: str) -> None: + with self._lock: + self._conn.execute( + """ + INSERT INTO issue_index_sync (repo, last_synced) VALUES (?, ?) + ON CONFLICT(repo) DO UPDATE SET last_synced = excluded.last_synced + """, + (repo, last_synced), + ) + + +def _index_entry_from_row(row: sqlite3.Row) -> IssueIndexEntry: + try: + labels = tuple(str(x) for x in json.loads(row["labels_json"])) + except (ValueError, TypeError): + labels = () + return IssueIndexEntry( + repo=str(row["repo"]), + number=int(row["number"]), + is_pull_request=bool(row["is_pr"]), + title=str(row["title"]), + body=str(row["body"]), + state=str(row["state"]), + state_reason=str(row["state_reason"]), + merged_at=str(row["merged_at"]), + author=str(row["author"]), + labels=labels, + comments=int(row["comments"]), + created_at=str(row["created_at"]), + updated_at=str(row["updated_at"]), + html_url=str(row["html_url"]), + ) + _DB_SINGLETON: Database | None = None _DB_LOCK = threading.Lock() diff --git a/python/robomp/src/github_backend.py b/python/robomp/src/github_backend.py index b9b09ba6f..38e78c19b 100644 --- a/python/robomp/src/github_backend.py +++ b/python/robomp/src/github_backend.py @@ -13,6 +13,7 @@ from typing import Any, Protocol from robomp.github_client import ( CommentInfo, + IssueIndexEntry, IssueInfo, IssueSummary, PullRequestFileInfo, @@ -47,6 +48,14 @@ class GitHubBackend(Protocol): ) -> list[IssueSummary]: ... async def search_issues(self, repo: str, query: str, *, limit: int = 10) -> list[IssueSummary]: ... + async def list_issue_index_entries( + self, + repo: str, + *, + since: str | None = None, + page: int = 1, + per_page: int = 100, + ) -> list[IssueIndexEntry]: ... async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: ... diff --git a/python/robomp/src/github_client.py b/python/robomp/src/github_client.py index 6ad2bf222..f6b03d088 100644 --- a/python/robomp/src/github_client.py +++ b/python/robomp/src/github_client.py @@ -122,6 +122,30 @@ class IssueSummary: is_pull_request: bool = False +@dataclass(slots=True, frozen=True) +class IssueIndexEntry: + """Full projection of an issue/PR for the local search index (includes body). + + Produced by `GitHubClient.list_issue_index_entries` / webhook payloads and + stored verbatim in the orchestrator's `issue_index` table. + """ + + repo: str + number: int + is_pull_request: bool + title: str + body: str + state: str # open | closed + state_reason: str # completed | not_planned | reopened | "" + merged_at: str # ISO timestamp for merged PRs; "" otherwise + author: str + labels: tuple[str, ...] + comments: int + created_at: str + updated_at: str + html_url: str + + @dataclass(slots=True, frozen=True) class ReactionInfo: """A reaction on an issue/comment. @@ -359,6 +383,31 @@ class GitHubClient: items = (data or {}).get("items") or [] return [_summary_from_item(repo, item) for item in items] + async def list_issue_index_entries( + self, + repo: str, + *, + since: str | None = None, + page: int = 1, + per_page: int = 100, + ) -> list[IssueIndexEntry]: + """One page of issues AND PRs (with bodies) for the local search index. + + `since` is GitHub's ISO `updated_at` lower bound; omit for a full + backfill. Callers page from 1 until a short page comes back. + """ + params: dict[str, Any] = { + "state": "all", + "per_page": max(1, min(int(per_page), 100)), + "page": max(1, int(page)), + "sort": "updated", + "direction": "asc", + } + if since: + params["since"] = since + data = await self.request("GET", f"/repos/{repo}/issues", params=params) + return [index_entry_from_issue_object(repo, item) for item in (data or [])] + async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: data = await self.request("GET", f"/repos/{repo}/issues/{number}/comments", params={"per_page": 100}) return [_comment_from_payload(item) for item in (data or [])] @@ -602,6 +651,60 @@ def _summary_from_item(repo: str, item: Mapping[str, Any]) -> IssueSummary: ) +def index_entry_from_issue_object(repo: str, item: Mapping[str, Any]) -> IssueIndexEntry: + """Build an `IssueIndexEntry` from a REST *issue-shaped* object. + + Accepts both plain issues and the issue representation of a PR (webhook + `issues`/`issue_comment` payloads, `/repos/{repo}/issues` items): PRs carry + a `pull_request` sub-object holding `merged_at`. + """ + user = item.get("user") or {} + labels_raw = item.get("labels") or [] + pr_obj = item.get("pull_request") + is_pr = pr_obj is not None + merged_at = str(pr_obj.get("merged_at") or "") if isinstance(pr_obj, Mapping) else "" + return IssueIndexEntry( + repo=repo, + number=int(item["number"]), + is_pull_request=is_pr, + title=str(item.get("title") or ""), + body=str(item.get("body") or ""), + state=str(item.get("state") or "open"), + state_reason=str(item.get("state_reason") or ""), + merged_at=merged_at, + author=str(user.get("login") or ""), + labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw), + comments=int(item.get("comments") or 0), + created_at=str(item.get("created_at") or ""), + updated_at=str(item.get("updated_at") or ""), + html_url=str(item.get("html_url") or ""), + ) + + +def index_entry_from_pr_object(repo: str, item: Mapping[str, Any]) -> IssueIndexEntry: + """Build an `IssueIndexEntry` from a REST *pull-request-shaped* object + (webhook `pull_request*` payloads), where `merged_at` sits at the top level. + """ + user = item.get("user") or {} + labels_raw = item.get("labels") or [] + return IssueIndexEntry( + repo=repo, + number=int(item["number"]), + is_pull_request=True, + title=str(item.get("title") or ""), + body=str(item.get("body") or ""), + state=str(item.get("state") or "open"), + state_reason="", + merged_at=str(item.get("merged_at") or ""), + author=str(user.get("login") or ""), + labels=tuple(str(lbl["name"]) if isinstance(lbl, dict) else str(lbl) for lbl in labels_raw), + comments=int(item.get("comments") or 0), + created_at=str(item.get("created_at") or ""), + updated_at=str(item.get("updated_at") or ""), + html_url=str(item.get("html_url") or ""), + ) + + def _pr_file_from_payload(data: Mapping[str, Any]) -> PullRequestFileInfo: return PullRequestFileInfo( path=str(data.get("filename") or data.get("path") or ""), @@ -663,6 +766,7 @@ __all__ = [ "CommentInfo", "GitHubClient", "GitHubError", + "IssueIndexEntry", "IssueInfo", "IssueSummary", "PullRequestFileInfo", @@ -671,5 +775,7 @@ __all__ = [ "ReactionInfo", "RepoInfo", "ReviewCommentInfo", + "index_entry_from_issue_object", + "index_entry_from_pr_object", "parse_issue_payload", ] diff --git a/python/robomp/src/host_tools.py b/python/robomp/src/host_tools.py index ac4bf3633..92cdd42be 100644 --- a/python/robomp/src/host_tools.py +++ b/python/robomp/src/host_tools.py @@ -27,6 +27,7 @@ from robomp.db import Database, IssueState, issue_key from robomp.git_ops import GitCommandError, HeadDriftError from robomp.github_backend import GitHubBackend from robomp.github_client import GitHubError, IssueInfo, PullRequestFileInfo, RepoInfo +from robomp.issue_index import parse_search_query from robomp.sandbox import ( GitTransport, Workspace, @@ -1260,11 +1261,25 @@ def _build_fetch_thread(bindings: ToolBindings) -> HostTool[Any, Any]: _REPO_QUALIFIER_RE = re.compile(r"(?i)\brepo:") +def _render_search_matches( + query: str, repo: str, rows: list[tuple[bool, int, str, str, str, tuple[str, ...], str]] +) -> str: + """Render (is_pr, number, state_display, title, author, labels, updated) rows.""" + lines = [f"# {len(rows)} match(es) for {query!r} in {repo}"] + for is_pr, number, state, title, author, labels, updated in rows: + kind = "PR" if is_pr else "issue" + label_sfx = f" [{', '.join(labels)}]" if labels else "" + lines.append(f"- #{number} ({kind}, {state}) {title} — @{author}, updated {updated[:10]}{label_sfx}") + return "\n".join(lines) + + def _build_search_issues(bindings: ToolBindings) -> HostTool[Any, Any]: - """Read-only issue/PR search scoped to the current repo. + """Issue/PR search scoped to the current repo, served from the local index. Exists so triage can find duplicates and already-merged fixes instead of - classifying blind; the inbound issue itself is filtered out of results. + classifying blind. Queries hit the webhook-fed SQLite FTS index (zero API + cost); the GitHub search API is only used before the repo's first + reconcile completes. The inbound issue is filtered out of results. """ def execute(args: dict[str, Any], _ctx: HostToolContext[Any]) -> str: @@ -1280,28 +1295,61 @@ def _build_search_issues(bindings: ToolBindings) -> HostTool[Any, Any]: _raise_command(msg) limit_raw = args.get("limit") limit = max(1, min(int(limit_raw), 20)) if isinstance(limit_raw, int) else 10 - try: - results = _run_coro( - bindings.loop, - bindings.github.search_issues(bindings.repo.full_name, query, limit=limit), + repo = bindings.repo.full_name + + rows: list[tuple[bool, int, str, str, str, tuple[str, ...], str]] + if bindings.db.issue_index_watermark(repo) is not None: + parsed = parse_search_query(query) + entries = bindings.db.search_issue_index( + repo, + keywords=parsed.keywords, + is_pr=parsed.is_pr, + state=parsed.state, + merged=parsed.merged, + label=parsed.label, + author=parsed.author, + limit=limit + 1, # headroom for the self-filter below ) - except GitHubError as exc: - _audit(bindings, "gh_search_issues", args, error=str(exc)) - _raise_command(f"GitHub search failed: {exc.status} {exc.message}") - results = [s for s in results if s.is_pull_request or s.number != bindings.issue.number] - if not results: - _audit(bindings, "gh_search_issues", args, result={"matches": 0}) - return f"No issues or PRs in {bindings.repo.full_name} match {query!r}." - lines = [f"# {len(results)} match(es) for {query!r} in {bindings.repo.full_name}"] - for s in results: - kind = "PR" if s.is_pull_request else "issue" - state = f"{s.state} ({s.state_reason})" if s.state_reason else s.state - labels = f" [{', '.join(s.labels)}]" if s.labels else "" - lines.append( - f"- #{s.number} ({kind}, {state}) {s.title} — @{s.author}, updated {s.updated_at[:10]}{labels}" - ) - _audit(bindings, "gh_search_issues", args, result={"matches": len(results)}) - return "\n".join(lines) + entries = [e for e in entries if e.is_pull_request or e.number != bindings.issue.number][:limit] + rows = [] + for e in entries: + if e.is_pull_request and e.merged_at: + state = "merged" + elif e.state_reason: + state = f"{e.state} ({e.state_reason})" + else: + state = e.state + rows.append((e.is_pull_request, e.number, state, e.title, e.author, e.labels, e.updated_at)) + source = "local" + else: + # Index not backfilled yet — fall through to the GitHub search API. + try: + found = _run_coro( + bindings.loop, + bindings.github.search_issues(repo, query, limit=limit), + ) + except GitHubError as exc: + _audit(bindings, "gh_search_issues", args, error=str(exc)) + _raise_command(f"GitHub search failed: {exc.status} {exc.message}") + found = [s for s in found if s.is_pull_request or s.number != bindings.issue.number] + rows = [ + ( + s.is_pull_request, + s.number, + f"{s.state} ({s.state_reason})" if s.state_reason else s.state, + s.title, + s.author, + s.labels, + s.updated_at, + ) + for s in found + ] + source = "remote" + if not rows: + _audit(bindings, "gh_search_issues", args, result={"matches": 0, "source": source}) + return f"No issues or PRs in {repo} match {query!r}." + _audit(bindings, "gh_search_issues", args, result={"matches": len(rows), "source": source}) + return _render_search_matches(query, repo, rows) return host_tool( name="gh_search_issues", @@ -1325,6 +1373,98 @@ def _build_search_issues(bindings: ToolBindings) -> HostTool[Any, Any]: ) +# ---------- search_commits ---------- +_COMMIT_SEARCH_TIMEOUT_SECONDS = 120.0 + + +def _build_search_commits(bindings: ToolBindings) -> HostTool[Any, Any]: + """Local `git log` search over the default branch's history. + + Two modes: `message` greps commit subjects/bodies (case-insensitive + regex), `patch` runs the pickaxe (`-S`) to find commits whose diff adds or + removes the literal string — the sharp tool for "was this already fixed". + The search interface (query in, ranked commits out) is deliberately opaque + about its backend so a semantic index can replace git plumbing later. + """ + + def execute(args: dict[str, Any], _ctx: HostToolContext[Any]) -> str: + query = args.get("query") + if not isinstance(query, str) or not query.strip(): + msg = "search_commits requires a non-empty 'query'." + _audit(bindings, "search_commits", args, error=msg) + _raise_command(msg) + query = query.strip() + mode = args.get("mode") or "message" + if mode not in ("message", "patch"): + msg = "search_commits 'mode' must be 'message' or 'patch'." + _audit(bindings, "search_commits", args, error=msg) + _raise_command(msg) + limit_raw = args.get("limit") + limit = max(1, min(int(limit_raw), 30)) if isinstance(limit_raw, int) else 10 + paths = [p for p in (args.get("paths") or ()) if isinstance(p, str) and p.strip()] + + rev = f"origin/{bindings.repo.default_branch}" + probe = _run_repo_command(bindings, ["git", "rev-parse", "--verify", "--quiet", rev], timeout=30.0) + if probe.returncode != 0: + rev = "HEAD" + cmd = ["git", "log", rev, "-n", str(limit), "--date=short", "--pretty=format:%h %ad %an — %s"] + if mode == "message": + cmd += [f"--grep={query}", "--regexp-ignore-case"] + else: + cmd += ["-S", query] + if paths: + cmd += ["--", *paths] + try: + proc = _run_repo_command(bindings, cmd, timeout=_COMMIT_SEARCH_TIMEOUT_SECONDS) + except subprocess.TimeoutExpired: + msg = f"search_commits timed out after {_COMMIT_SEARCH_TIMEOUT_SECONDS:.0f}s; narrow with 'paths' or a shorter history window." + _audit(bindings, "search_commits", args, error=msg) + _raise_command(msg) + if proc.returncode != 0: + msg = f"git log failed: {(proc.stderr or proc.stdout).strip()[:500]}" + _audit(bindings, "search_commits", args, error=msg) + _raise_command(msg) + out = proc.stdout.strip() + if not out: + _audit(bindings, "search_commits", args, result={"matches": 0}) + return f"No commits on {rev} match {query!r} (mode={mode})." + matches = out.splitlines() + _audit(bindings, "search_commits", args, result={"matches": len(matches)}) + header = f"# {len(matches)} commit(s) on {rev} matching {query!r} (mode={mode})" + return "\n".join([header, *matches]) + + return host_tool( + name="search_commits", + description=persona.host_tool_description("search_commits"), + parameters={ + "type": "object", + "properties": { + "query": { + "type": "string", + "description": persona.host_tool_parameter_description("search_commits", "query"), + }, + "mode": { + "type": "string", + "enum": ["message", "patch"], + "description": persona.host_tool_parameter_description("search_commits", "mode"), + }, + "paths": { + "type": "array", + "items": {"type": "string"}, + "description": persona.host_tool_parameter_description("search_commits", "paths"), + }, + "limit": { + "type": "integer", + "description": persona.host_tool_parameter_description("search_commits", "limit"), + }, + }, + "required": ["query"], + "additionalProperties": False, + }, + execute=execute, + ) + + _PRIMARY_TYPES = ("bug", "enhancement", "question", "proposal", "documentation", "wontfix", "invalid", "duplicate") _AUTO_PR_CLASSIFICATIONS = frozenset({"bug", "documentation"}) _PRIORITIES = ("prio:p0", "prio:p1", "prio:p2", "prio:p3") @@ -1905,6 +2045,7 @@ def build(bindings: ToolBindings) -> tuple[HostTool[Any, Any], ...]: _build_abort_task(bindings), _build_fetch_thread(bindings), _build_search_issues(bindings), + _build_search_commits(bindings), ) diff --git a/python/robomp/src/issue_index.py b/python/robomp/src/issue_index.py new file mode 100644 index 000000000..98f966018 --- /dev/null +++ b/python/robomp/src/issue_index.py @@ -0,0 +1,251 @@ +"""Local issue/PR search index: webhook ingest, periodic reconcile, query parsing. + +The `issue_index` table mirrors every issue and PR of the allowlisted repos so +`gh_search_issues` answers from SQLite FTS5 instead of the GitHub search API. +Freshness comes from two directions: + + - `ingest_webhook_payload` upserts on every `issues` / `issue_comment` / + `pull_request*` delivery, keeping the hot path current in real time. + - `IssueIndexSync` reconciles each repo every `issue_index_sync_seconds` + (and backfills on first run) via `/repos/{repo}/issues?since=…`, catching + anything webhooks missed while the orchestrator was down. + +`parse_search_query` translates the GitHub-search-flavored tool query into the +structured filters `Database.search_issue_index` takes, so the agent keeps one +query language whether the lookup is served locally or remotely. +""" + +from __future__ import annotations + +import asyncio +import logging +from collections.abc import Mapping +from dataclasses import dataclass +from datetime import UTC, datetime, timedelta + +from robomp.config import Settings +from robomp.db import Database +from robomp.github_backend import GitHubBackend +from robomp.github_client import ( + GitHubError, + index_entry_from_issue_object, + index_entry_from_pr_object, +) + +log = logging.getLogger(__name__) + +# Overlap subtracted from the watermark on every reconcile so a sync that +# raced a concurrent update can never permanently skip it. +_SYNC_OVERLAP = timedelta(minutes=2) +_PAGE_SIZE = 100 +_MAX_PAGES_PER_TICK = 30 + + +@dataclass(slots=True, frozen=True) +class ParsedSearchQuery: + """Structured form of a GitHub-issue-search style query string.""" + + keywords: tuple[str, ...] + is_pr: bool | None = None + state: str | None = None + merged: bool | None = None + label: str | None = None + author: str | None = None + + +def parse_search_query(query: str) -> ParsedSearchQuery: + """Split a GitHub-search style string into keywords + structured filters. + + Supported qualifiers: `is:pr` / `is:issue` / `is:open` / `is:closed` / + `is:merged`, `label:`, `author:`. Unrecognized `key:value` + qualifiers are dropped rather than fed to FTS5 (a bare `in:title` token + would otherwise be a syntax error). Everything else is a keyword. + """ + keywords: list[str] = [] + is_pr: bool | None = None + state: str | None = None + merged: bool | None = None + label: str | None = None + author: str | None = None + for token in query.split(): + key, sep, value = token.partition(":") + if not sep or not value or " " in key: + keywords.append(token) + continue + key = key.lower() + if key == "is": + v = value.lower() + if v == "pr": + is_pr = True + elif v == "issue": + is_pr = False + elif v in ("open", "closed"): + state = v + elif v == "merged": + is_pr = True + merged = True + elif key == "label": + label = value.strip('"') + elif key == "author": + author = value.lstrip("@") + # Any other qualifier (in:, sort:, created:, …) is intentionally dropped. + return ParsedSearchQuery( + keywords=tuple(keywords), + is_pr=is_pr, + state=state, + merged=merged, + label=label, + author=author, + ) + + +def ingest_webhook_payload(db: Database, repo: str, event_type: str, payload: Mapping[str, object]) -> bool: + """Upsert the issue/PR carried by a webhook delivery into the index. + + Returns True when the payload contained an indexable object. Runs before + routing so even deliveries the router skips (bot comments, unhandled + actions) still refresh the index. + """ + if event_type in ("issues", "issue_comment"): + obj = payload.get("issue") + if isinstance(obj, Mapping) and obj.get("number") is not None: + db.upsert_issue_index(index_entry_from_issue_object(repo, obj)) + return True + return False + if event_type.startswith("pull_request"): + obj = payload.get("pull_request") + if isinstance(obj, Mapping) and obj.get("number") is not None: + db.upsert_issue_index(index_entry_from_pr_object(repo, obj)) + return True + return False + return False + + +def _utcnow_iso() -> str: + return datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ") + + +def _overlapped(watermark: str) -> str: + """Rewind an ISO watermark by the sync overlap; fall back to the raw value.""" + try: + parsed = datetime.strptime(watermark, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=UTC) + except ValueError: + return watermark + return (parsed - _SYNC_OVERLAP).strftime("%Y-%m-%dT%H:%M:%SZ") + + +class IssueIndexSync: + """Background reconciler for the local issue index. + + First tick per repo backfills everything (`since=None`); later ticks pull + only issues updated after the stored watermark (minus a small overlap). + Pagination is bounded per tick — a huge backlog finishes across ticks + rather than hogging one. + """ + + def __init__(self, *, settings: Settings, db: Database, github: GitHubBackend) -> None: + self._settings = settings + self._db = db + self._github = github + self._task: asyncio.Task[None] | None = None + self._stop_event: asyncio.Event | None = None + + @property + def enabled(self) -> bool: + return self._settings.issue_index_sync_seconds > 0 + + async def start(self) -> None: + """Spawn the background loop. No-op when disabled.""" + if not self.enabled: + log.info("issue index sync disabled") + return + if self._task is not None: + return + self._stop_event = asyncio.Event() + self._task = asyncio.create_task(self._run(), name="issue-index-sync") + log.info( + "issue index sync started", + extra={"interval_seconds": self._settings.issue_index_sync_seconds}, + ) + + async def stop(self) -> None: + """Signal the loop to exit and await its termination.""" + if self._task is None: + return + assert self._stop_event is not None + self._stop_event.set() + try: + await asyncio.wait_for(self._task, timeout=5.0) + except TimeoutError: + self._task.cancel() + try: + await self._task + except (asyncio.CancelledError, Exception): + pass + finally: + self._task = None + self._stop_event = None + + async def _run(self) -> None: + assert self._stop_event is not None + interval = float(self._settings.issue_index_sync_seconds) + while not self._stop_event.is_set(): + try: + await self.tick() + except Exception: + log.exception("issue index sync tick failed") + try: + await asyncio.wait_for(self._stop_event.wait(), timeout=interval) + except TimeoutError: + continue + + async def tick(self) -> None: + """Reconcile every allowlisted repo once.""" + for repo in self._settings.repo_allowlist: + try: + await self.sync_repo(repo) + except GitHubError as exc: + log.warning( + "issue index sync failed; will retry next tick", + extra={"repo": repo, "status": exc.status, "gh_message": exc.message}, + ) + + async def sync_repo(self, repo: str) -> int: + """Pull updated issues/PRs for one repo into the index. Returns count ingested.""" + started_at = _utcnow_iso() + watermark = self._db.issue_index_watermark(repo) + since = _overlapped(watermark) if watermark else None + ingested = 0 + exhausted = False + last_seen = "" + for page in range(1, _MAX_PAGES_PER_TICK + 1): + batch = await self._github.list_issue_index_entries(repo, since=since, page=page, per_page=_PAGE_SIZE) + for entry in batch: + self._db.upsert_issue_index(entry) + if entry.updated_at > last_seen: + last_seen = entry.updated_at + ingested += len(batch) + if len(batch) < _PAGE_SIZE: + exhausted = True + break + if exhausted: + # Everything up to tick start is ingested; later updates are the + # next tick's problem (or a webhook's). + self._db.set_issue_index_watermark(repo, started_at) + elif last_seen: + # Page budget hit mid-backfill: advance the watermark only to the + # newest updated_at actually ingested, so the next tick resumes there. + self._db.set_issue_index_watermark(repo, last_seen) + log.info( + "issue index synced", + extra={"repo": repo, "ingested": ingested, "backfill": watermark is None, "complete": exhausted}, + ) + return ingested + + +__all__ = [ + "IssueIndexSync", + "ParsedSearchQuery", + "ingest_webhook_payload", + "parse_search_query", +] diff --git a/python/robomp/src/prompts/host_tools.toml b/python/robomp/src/prompts/host_tools.toml index 80b0aa211..82a84a32d 100644 --- a/python/robomp/src/prompts/host_tools.toml +++ b/python/robomp/src/prompts/host_tools.toml @@ -73,12 +73,21 @@ reason = "Internal diagnosis for the operator. Concrete, specific, blameless. NE description = "Refetch the originating issue and its comments. Use sparingly." [gh_search_issues] -description = "Search issues AND pull requests in the current repo (GitHub issue-search syntax; repo scope applied automatically). Use during triage to find duplicates and to check whether a merged PR already fixed the reported problem before classifying." +description = "Search issues AND pull requests in the current repo. Served from a local index of every issue/PR (webhook-fed, periodically reconciled), so calls are free — search liberally during triage to find duplicates and already-merged fixes before classifying." [gh_search_issues.parameters] -query = "GitHub issue-search syntax: bare keywords plus qualifiers like `is:pr`, `is:closed`, `is:merged`, `label:bug`, `in:title`, `author:`. NEVER include a `repo:` qualifier — scope is applied automatically." +query = "GitHub issue-search syntax: bare keywords plus qualifiers like `is:pr`, `is:closed`, `is:merged`, `label:bug`, `author:`. NEVER include a `repo:` qualifier — scope is applied automatically." limit = "Max results, 1-20. Default 10." +[search_commits] +description = "Search the default branch's commit history. `mode=message` greps commit messages (case-insensitive regex); `mode=patch` finds commits whose diff adds/removes the literal query string — use it to check whether the broken code path was already touched by a fix." + +[search_commits.parameters] +query = "Regex for `mode=message`; literal string for `mode=patch`." +mode = "`message` (default) searches commit messages; `patch` pickaxe-searches diff content." +paths = "Optional path filters (files or directories) to narrow the walk." +limit = "Max commits, 1-30. Default 10." + [set_issue_labels] description = "Append labels to the originating issue/PR. NEVER removes existing labels." diff --git a/python/robomp/src/prompts/system_append.md b/python/robomp/src/prompts/system_append.md index 5212c9ed9..9bbced981 100644 --- a/python/robomp/src/prompts/system_append.md +++ b/python/robomp/src/prompts/system_append.md @@ -25,10 +25,10 @@ Pick exactly ONE primary label per issue: ## Duplicate & already-fixed check -Before `classify_issue`, run `gh_search_issues` with the report's key terms (retry with synonyms and an `is:pr` variant — one search proves nothing): +Before `classify_issue`, run `gh_search_issues` with the report's key terms (retry with synonyms and an `is:pr` variant — searches are served from a local index and cost nothing; one search proves nothing): - **Prior issue on the same problem** → `duplicate`, cite it. A prior closure as not-planned/`wontfix` on the same complaint is binding precedent — adopt that verdict; NEVER relitigate it. -- **Already fixed.** Your worktree is the CURRENT default branch; reporters often run older releases. When the reported version lags the latest release (topmost released section of the relevant `packages/*/CHANGELOG.md`), check the changelog and merged PRs (`is:pr is:merged `) for an existing fix, and try the repro against the worktree — failing on the reporter's version but passing here means it is already fixed. Classify `duplicate`: cite the fixing PR, name the release carrying it (or say it ships in the next release when still under `[Unreleased]`), and tell the reporter to update. NEVER re-fix what main already fixed. +- **Already fixed.** Your worktree is the CURRENT default branch; reporters often run older releases. When the reported version lags the latest release (topmost released section of the relevant `packages/*/CHANGELOG.md`), check the changelog, merged PRs (`is:pr is:merged `), and recent commits (`search_commits` — `mode=message` for symptom keywords, `mode=patch` for the exact broken code) for an existing fix, and try the repro against the worktree — failing on the reporter's version but passing here means it is already fixed. Classify `duplicate`: cite the fixing PR/commit, name the release carrying it (or say it ships in the next release when still under `[Unreleased]`), and tell the reporter to update. NEVER re-fix what main already fixed. ## Merit gate — `bug` vs `wontfix` vs `enhancement` diff --git a/python/robomp/src/proxy/server.py b/python/robomp/src/proxy/server.py index 923753b40..2d3bd96a6 100644 --- a/python/robomp/src/proxy/server.py +++ b/python/robomp/src/proxy/server.py @@ -523,6 +523,22 @@ def create_proxy_app(settings: Settings) -> FastAPI: return _gh_error_response(exc) return JSONResponse({"items": [_serialize(s) for s in items]}) + @app.get("/gh/v1/issue_index_entries") + async def list_issue_index_entries( + request: Request, + repo: str, + since: str | None = None, + page: int = 1, + per_page: int = 100, + ) -> JSONResponse: + await _authenticate(request) + github: GitHubClient = request.app.state.github + try: + items = await github.list_issue_index_entries(repo, since=since, page=page, per_page=per_page) + except GitHubError as exc: + return _gh_error_response(exc) + return JSONResponse({"items": [_serialize(s) for s in items]}) + @app.get("/gh/v1/comments") async def list_comments(request: Request, repo: str, number: int) -> JSONResponse: await _authenticate(request) diff --git a/python/robomp/src/proxy_client.py b/python/robomp/src/proxy_client.py index a2069c950..ddf0c3229 100644 --- a/python/robomp/src/proxy_client.py +++ b/python/robomp/src/proxy_client.py @@ -25,6 +25,7 @@ from robomp.git_ops import GitCommandError, HeadDriftError, PushResult from robomp.github_client import ( CommentInfo, GitHubError, + IssueIndexEntry, IssueInfo, IssueSummary, PullRequestFileInfo, @@ -212,6 +213,20 @@ class GitHubProxyClient: ) return [_issue_summary_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []] + async def list_issue_index_entries( + self, + repo: str, + *, + since: str | None = None, + page: int = 1, + per_page: int = 100, + ) -> list[IssueIndexEntry]: + params: dict[str, Any] = {"repo": repo, "page": page, "per_page": per_page} + if since: + params["since"] = since + data = await self._request("GET", "/gh/v1/issue_index_entries", params=params) + return [_index_entry_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []] + async def list_comments(self, repo: str, number: int) -> list[CommentInfo]: data = await self._request("GET", "/gh/v1/comments", params={"repo": repo, "number": number}) return [_comment_from(item) for item in (data.get("items") if isinstance(data, dict) else None) or []] @@ -511,6 +526,27 @@ def _issue_summary_from(data: Any) -> IssueSummary: ) +def _index_entry_from(data: Any) -> IssueIndexEntry: + if not isinstance(data, dict): + raise GitHubError(500, "proxy returned malformed issue index payload") + return IssueIndexEntry( + repo=str(data["repo"]), + number=int(data["number"]), + is_pull_request=bool(data.get("is_pull_request")), + title=str(data.get("title") or ""), + body=str(data.get("body") or ""), + state=str(data.get("state") or ""), + state_reason=str(data.get("state_reason") or ""), + merged_at=str(data.get("merged_at") or ""), + author=str(data.get("author") or ""), + labels=tuple(str(x) for x in (data.get("labels") or [])), + comments=int(data.get("comments") or 0), + created_at=str(data.get("created_at") or ""), + updated_at=str(data.get("updated_at") or ""), + html_url=str(data.get("html_url") or ""), + ) + + def _comment_from(data: Any) -> CommentInfo: if not isinstance(data, dict): raise GitHubError(500, "proxy returned malformed comment payload") diff --git a/python/robomp/src/server.py b/python/robomp/src/server.py index 6a9371f96..36aa3d5a5 100644 --- a/python/robomp/src/server.py +++ b/python/robomp/src/server.py @@ -14,7 +14,7 @@ from fastapi import Body, FastAPI, Header, HTTPException, Request, status from fastapi.responses import HTMLResponse, JSONResponse from fastapi.staticfiles import StaticFiles -from robomp import github_events +from robomp import github_events, issue_index from robomp.autoclose import AutocloseScheduler from robomp.config import Settings, get_settings from robomp.dashboard import render_index, static_dir, tail_jsonl @@ -29,6 +29,7 @@ from robomp.db import ( ) from robomp.github_backend import GitHubBackend from robomp.github_client import GitHubError, IssueSummary +from robomp.issue_index import IssueIndexSync from robomp.manual_triage import ( InvalidIssueRef, ManualTriageConflict, @@ -252,6 +253,7 @@ def _build_state(settings: Settings) -> dict[str, Any]: ) pool = WorkerPool(settings=settings, db=db, github=github, sandbox=sandbox, git_transport=git_transport) autoclose = AutocloseScheduler(settings=settings, db=db, github=github) + index_sync = IssueIndexSync(settings=settings, db=db, github=github) return { "settings": settings, "db": db, @@ -262,6 +264,7 @@ def _build_state(settings: Settings) -> dict[str, Any]: "pool": pool, "issue_browse_cache": _IssueBrowseCache(), "autoclose": autoclose, + "issue_index_sync": index_sync, } @@ -278,9 +281,12 @@ def create_app(settings: Settings | None = None) -> FastAPI: await pool.start() autoclose: AutocloseScheduler = app.state.bag["autoclose"] await autoclose.start() + index_sync: IssueIndexSync = app.state.bag["issue_index_sync"] + await index_sync.start() try: yield finally: + await index_sync.stop() await autoclose.stop() await pool.stop( drain_timeout=cfg.shutdown_drain_timeout_seconds, @@ -328,6 +334,15 @@ def create_app(settings: Settings | None = None) -> FastAPI: payload=payload, allowlist=cfg.repo_allowlist, ) + # Keep the local search index fresh from every delivery that carries an + # issue/PR object — including ones the router will skip. + if x_github_event in ("issues", "issue_comment") or x_github_event.startswith("pull_request"): + repo_full = str((payload.get("repository") or {}).get("full_name") or "") + if repo_full and repo_full in cfg.repo_allowlist: + try: + issue_index.ingest_webhook_payload(db, repo_full, x_github_event, payload) + except Exception: + log.exception("issue index webhook ingest failed", extra={"repo": repo_full}) def _resolve(repo_full: str, pr_number: int) -> str | None: row = db.find_issue_by_pr(repo_full, pr_number) diff --git a/python/robomp/tests/conftest.py b/python/robomp/tests/conftest.py index 89959e5fb..6fc4395ab 100644 --- a/python/robomp/tests/conftest.py +++ b/python/robomp/tests/conftest.py @@ -97,6 +97,10 @@ def _baseline_env(tmp_path: Path) -> dict[str, str]: # cache flip `ROBOMP_NATIVES_CACHE_ENABLED=true` explicitly. "ROBOMP_NATIVES_CACHE_ROOT": str(tmp_path / "natives-cache"), "ROBOMP_NATIVES_CACHE_ENABLED": "false", + # Same reasoning for the issue-index reconciler: its first tick would + # spin connect-retries against the .invalid proxy URL inside server + # tests. Tests that want it construct IssueIndexSync directly. + "ROBOMP_ISSUE_INDEX_SYNC_SECONDS": "0", } diff --git a/python/robomp/tests/test_host_tools.py b/python/robomp/tests/test_host_tools.py index 102074c90..3e2f2f541 100644 --- a/python/robomp/tests/test_host_tools.py +++ b/python/robomp/tests/test_host_tools.py @@ -4,6 +4,7 @@ from __future__ import annotations import asyncio import json +import subprocess import threading from pathlib import Path from typing import Any @@ -14,7 +15,7 @@ from omp_rpc import HostToolContext, RpcCommandError from robomp import host_tools from robomp.db import Database -from robomp.github_client import GitHubClient, IssueInfo, RepoInfo +from robomp.github_client import GitHubClient, IssueIndexEntry, IssueInfo, RepoInfo from robomp.host_tools import AbortController, ToolBindings, build from robomp.sandbox import LocalGitTransport, Workspace @@ -937,6 +938,98 @@ def test_gh_search_issues_rejects_repo_qualifier_and_empty_query(db: Database, t _stop_loop(loop, t) +def test_gh_search_issues_serves_from_local_index_once_synced(db: Database, tmp_path: Path) -> None: + """With a sync watermark present the tool answers from SQLite: qualifiers + become filters, merged PRs render as `merged`, and NO GitHub call happens.""" + + def handler(_request: httpx.Request) -> httpx.Response: + raise AssertionError("local-index search must not call GitHub") + + bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(handler)) + db.set_issue_index_watermark("octo/widget", "2026-07-01T00:00:00Z") + db.upsert_issue_index( + IssueIndexEntry( + repo="octo/widget", + number=31, + is_pull_request=True, + title="fix: resize crash", + body="handles narrow terminals", + state="closed", + state_reason="", + merged_at="2026-06-02T00:00:00Z", + author="bot", + labels=(), + comments=1, + created_at="2026-06-02T00:00:00Z", + updated_at="2026-06-02T00:00:00Z", + html_url="https://example/pull/31", + ) + ) + db.upsert_issue_index( + IssueIndexEntry( + repo="octo/widget", + number=30, + is_pull_request=False, + title="resize crash report", + body="", + state="closed", + state_reason="not_planned", + merged_at="", + author="bob", + labels=("wontfix",), + comments=3, + created_at="2026-05-01T00:00:00Z", + updated_at="2026-06-01T00:00:00Z", + html_url="https://example/30", + ) + ) + try: + tool = next(x for x in build(bindings) if x.name == "gh_search_issues") + result = tool.execute({"query": "resize crash"}, _ctx()) + pr_only = tool.execute({"query": "resize crash is:merged"}, _ctx()) + finally: + _stop_loop(loop, t) + assert "#30 (issue, closed (not_planned))" in result + assert "#31 (PR, merged)" in result + assert "#31" in pr_only and "#30" not in pr_only + + +def _git_repo_with_commits(bindings) -> None: + """Turn the stub workspace repo_dir into a git repo with two commits.""" + repo = str(bindings.workspace.repo_dir) + ident = ["-c", "user.name=t", "-c", "user.email=t@example.invalid"] + subprocess.run(["git", "init", "-q", "-b", "main", repo], check=True) + Path(repo, "a.txt").write_text("plain start\n", encoding="utf-8") + subprocess.run(["git", "-C", repo, "add", "."], check=True) + subprocess.run(["git", "-C", repo, *ident, "commit", "-q", "-m", "feat: initial import"], check=True) + Path(repo, "a.txt").write_text("plain start\nsplitPathAndSel guard\n", encoding="utf-8") + subprocess.run(["git", "-C", repo, "add", "."], check=True) + subprocess.run( + ["git", "-C", repo, *ident, "commit", "-q", "-m", "fix(tools): colon selector literal paths"], + check=True, + ) + + +def test_search_commits_message_and_patch_modes(db: Database, tmp_path: Path) -> None: + """message mode greps commit messages; patch mode pickaxes diff content. + Without an origin ref the search falls back to HEAD instead of failing.""" + bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500))) + _git_repo_with_commits(bindings) + try: + tool = next(x for x in build(bindings) if x.name == "search_commits") + by_message = tool.execute({"query": "colon selector"}, _ctx()) + by_patch = tool.execute({"query": "splitPathAndSel", "mode": "patch"}, _ctx()) + none = tool.execute({"query": "nonexistent-topic"}, _ctx()) + with pytest.raises(RpcCommandError): + tool.execute({"query": "x", "mode": "bogus"}, _ctx()) + finally: + _stop_loop(loop, t) + assert "fix(tools): colon selector literal paths" in by_message + assert "feat: initial import" not in by_message + assert "fix(tools): colon selector literal paths" in by_patch + assert none.startswith("No commits") + + def test_classify_issue_rejects_bug_without_priority(db: Database, tmp_path: Path) -> None: bindings, loop, t = _bindings(db, tmp_path, httpx.MockTransport(lambda r: httpx.Response(500))) try: diff --git a/python/robomp/tests/test_issue_index.py b/python/robomp/tests/test_issue_index.py new file mode 100644 index 000000000..d3da7021f --- /dev/null +++ b/python/robomp/tests/test_issue_index.py @@ -0,0 +1,189 @@ +"""Local issue index: query parsing, webhook ingest, FTS search, reconcile sync.""" + +from __future__ import annotations + +from pathlib import Path + +from robomp.db import Database +from robomp.github_client import IssueIndexEntry +from robomp.issue_index import IssueIndexSync, ingest_webhook_payload, parse_search_query + + +def _entry(number: int, **overrides) -> IssueIndexEntry: + base = { + "repo": "octo/widget", + "number": number, + "is_pull_request": False, + "title": f"issue {number}", + "body": "", + "state": "open", + "state_reason": "", + "merged_at": "", + "author": "alice", + "labels": (), + "comments": 0, + "created_at": "2026-01-01T00:00:00Z", + "updated_at": "2026-01-01T00:00:00Z", + "html_url": f"https://example/{number}", + } + base.update(overrides) + return IssueIndexEntry(**base) + + +# ---- parse_search_query ---- + + +def test_parse_search_query_extracts_supported_qualifiers() -> None: + parsed = parse_search_query("colon selector is:pr is:merged label:bug author:@alice in:title") + assert parsed.keywords == ("colon", "selector") # `in:title` dropped, not fed to FTS + assert parsed.is_pr is True + assert parsed.merged is True + assert parsed.label == "bug" + assert parsed.author == "alice" + + +def test_parse_search_query_state_and_issue_kind() -> None: + parsed = parse_search_query("is:issue is:closed crash") + assert parsed.is_pr is False + assert parsed.state == "closed" + assert parsed.keywords == ("crash",) + + +# ---- db index: upsert + search ---- + + +def test_search_issue_index_matches_body_text_and_ranks(db: Database) -> None: + db.upsert_issue_index(_entry(1, title="TUI crash on resize", body="stack trace mentions overlay")) + db.upsert_issue_index(_entry(2, title="unrelated docs typo", body="readme wording")) + found = db.search_issue_index("octo/widget", keywords=("resize", "crash")) + assert [e.number for e in found] == [1] + # body-only terms also hit + found = db.search_issue_index("octo/widget", keywords=("overlay",)) + assert [e.number for e in found] == [1] + + +def test_search_issue_index_filters(db: Database) -> None: + db.upsert_issue_index( + _entry(1, title="fix crash", is_pull_request=True, merged_at="2026-02-01T00:00:00Z", state="closed") + ) + db.upsert_issue_index( + _entry(2, title="crash report", state="closed", state_reason="not_planned", labels=("wontfix",)) + ) + db.upsert_issue_index(_entry(3, title="crash report open", state="open")) + + merged_prs = db.search_issue_index("octo/widget", keywords=("crash",), is_pr=True, merged=True) + assert [e.number for e in merged_prs] == [1] + wontfixed = db.search_issue_index("octo/widget", keywords=("crash",), label="wontfix") + assert [e.number for e in wontfixed] == [2] + open_only = db.search_issue_index("octo/widget", keywords=("crash",), state="open") + assert [e.number for e in open_only] == [3] + + +def test_upsert_refreshes_fts_so_stale_text_stops_matching(db: Database) -> None: + """The UPDATE trigger must swap FTS content, not accumulate it.""" + db.upsert_issue_index(_entry(1, title="original scrollback wipe")) + db.upsert_issue_index(_entry(1, title="renamed: alternate screen request", state="closed")) + assert db.search_issue_index("octo/widget", keywords=("scrollback",)) == [] + found = db.search_issue_index("octo/widget", keywords=("alternate",)) + assert len(found) == 1 and found[0].state == "closed" + + +def test_search_issue_index_quotes_fts_metacharacters(db: Database) -> None: + """Reporter text like `"AND (` must never raise an FTS5 syntax error.""" + db.upsert_issue_index(_entry(1, title='crash with "quoted" AND (parens)')) + found = db.search_issue_index("octo/widget", keywords=('"quoted"', "AND", "(parens)")) + assert [e.number for e in found] == [1] + + +def test_issue_index_watermark_roundtrip(db: Database) -> None: + assert db.issue_index_watermark("octo/widget") is None + db.set_issue_index_watermark("octo/widget", "2026-07-01T00:00:00Z") + assert db.issue_index_watermark("octo/widget") == "2026-07-01T00:00:00Z" + db.set_issue_index_watermark("octo/widget", "2026-07-02T00:00:00Z") + assert db.issue_index_watermark("octo/widget") == "2026-07-02T00:00:00Z" + + +# ---- webhook ingest ---- + + +def test_ingest_webhook_issue_and_pr_payloads(db: Database) -> None: + ingested = ingest_webhook_payload( + db, + "octo/widget", + "issues", + {"issue": {"number": 5, "title": "boom", "body": "b", "state": "open", "user": {"login": "alice"}}}, + ) + assert ingested + # PR-flavored issue payload (issue_comment on a PR) carries pull_request.merged_at. + ingest_webhook_payload( + db, + "octo/widget", + "issue_comment", + { + "issue": { + "number": 6, + "title": "fixes boom", + "state": "closed", + "user": {"login": "bob"}, + "pull_request": {"merged_at": "2026-03-01T00:00:00Z"}, + } + }, + ) + # Native pull_request payload: merged_at at top level. + ingest_webhook_payload( + db, + "octo/widget", + "pull_request", + {"pull_request": {"number": 7, "title": "another fix", "state": "closed", "merged_at": "2026-04-01T00:00:00Z"}}, + ) + assert not ingest_webhook_payload(db, "octo/widget", "push", {"ref": "refs/heads/main"}) + + boom = db.search_issue_index("octo/widget", keywords=("boom",)) + assert {e.number for e in boom} == {5, 6} + pr6 = next(e for e in boom if e.number == 6) + assert pr6.is_pull_request and pr6.merged_at == "2026-03-01T00:00:00Z" + pr7 = db.search_issue_index("octo/widget", keywords=("another",))[0] + assert pr7.is_pull_request and pr7.merged_at == "2026-04-01T00:00:00Z" + + +# ---- reconcile sync ---- + + +class _FakeBackend: + """Pages of index entries keyed by page number; records `since` per call.""" + + def __init__(self, pages: dict[int, list[IssueIndexEntry]]) -> None: + self.pages = pages + self.calls: list[tuple[str | None, int]] = [] + + async def list_issue_index_entries( + self, repo: str, *, since: str | None = None, page: int = 1, per_page: int = 100 + ) -> list[IssueIndexEntry]: + self.calls.append((since, page)) + return self.pages.get(page, []) + + +class _SyncSettings: + issue_index_sync_seconds = 900.0 + repo_allowlist = frozenset({"octo/widget"}) + + +async def test_sync_repo_backfills_pages_and_sets_watermark(db: Database, tmp_path: Path) -> None: + full_page = [_entry(n, updated_at=f"2026-06-{n:02d}T00:00:00Z") for n in range(1, 101)] + short_page = [_entry(101, updated_at="2026-07-01T00:00:00Z")] + backend = _FakeBackend({1: full_page, 2: short_page}) + sync = IssueIndexSync(settings=_SyncSettings(), db=db, github=backend) # type: ignore[arg-type] + + ingested = await sync.sync_repo("octo/widget") + assert ingested == 101 + # First run is a backfill: no `since` on any call, pages walked in order. + assert backend.calls == [(None, 1), (None, 2)] + watermark = db.issue_index_watermark("octo/widget") + assert watermark is not None + assert db.search_issue_index("octo/widget", keywords=("issue",), limit=5) + + # Second run is incremental: `since` derives from the stored watermark. + backend.calls.clear() + backend.pages = {1: []} + await sync.sync_repo("octo/widget") + assert backend.calls and backend.calls[0][0] is not None diff --git a/python/robomp/tests/test_server.py b/python/robomp/tests/test_server.py index f4c492b5b..7141fddc3 100644 --- a/python/robomp/tests/test_server.py +++ b/python/robomp/tests/test_server.py @@ -958,6 +958,42 @@ def test_webhook_incoming_pr_comment_without_directive_skips_without_counting_bu assert states == ["queued", "queued", "skipped"] +def test_webhook_delivery_populates_issue_index(settings: Settings) -> None: + """Every issue-carrying delivery upserts the local search index — including + ones the router skips (here: a conversation comment on an incoming PR).""" + app = create_app(settings) + with TestClient(app) as client: + payload = { + "action": "opened", + "issue": { + "number": 501, + "title": "grep misses colon filenames", + "body": "read tool peels the selector suffix", + "state": "open", + "user": {"login": "alice"}, + "author_association": "NONE", + }, + "repository": {"full_name": "octo/widget"}, + } + body = json.dumps(payload).encode() + resp = client.post( + "/webhook/github", + content=body, + headers=_signed_headers("test-webhook-secret", body, event="issues", delivery="idx-1"), + ) + assert resp.status_code == 202 + + skipped = _post_pr_issue_comment(client, delivery="idx-2", user="stranger", pr_number=502) + assert skipped.json()["state"] == "skipped" + + db = get_database(settings.sqlite_path) + by_body = db.search_issue_index("octo/widget", keywords=("selector", "suffix")) + pr_row = db.search_issue_index("octo/widget", is_pr=True) + close_database() + assert [e.number for e in by_body] == [501] + assert [e.number for e in pr_row] == [502] + + def test_webhook_contributor_gets_higher_cap(rate_limited_settings: Settings) -> None: app = create_app(rate_limited_settings) with TestClient(app) as client: From 425e583ae0bafc423da76fb1e7772ae012b9b4e2 Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 15 Jul 2026 00:50:55 +0200 Subject: [PATCH 044/293] feat(coding-agent): added support for task-agent field and model resolution - Added schema and type updates for task-agent fields and model resolver settings. - Extended discovery helper logic to carry resolved task-agent metadata through execution setup. - Updated task/agent registration and execution paths to use the new capability/field data. - Expanded test coverage for agent-field parsing, model resolution, and executor prewalk behavior. --- docs/task-agent-discovery.md | 6 +- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/config/model-resolver.ts | 31 ++ .../src/config/settings-schema.ts | 4 + .../coding-agent/src/discovery/helpers.ts | 22 +- packages/coding-agent/src/main.ts | 6 +- .../src/modes/components/agent-dashboard.ts | 67 ++++- .../src/prompts/agents/frontmatter.md | 1 + packages/coding-agent/src/task/agents.ts | 4 + packages/coding-agent/src/task/executor.ts | 34 ++- packages/coding-agent/src/task/index.ts | 3 + packages/coding-agent/src/task/types.ts | 2 + .../test/discovery/agent-fields.test.ts | 21 ++ .../coding-agent/test/model-resolver.test.ts | 28 ++ .../test/task/executor-prewalk.test.ts | 281 ++++++++++++++++++ .../tools/task-agent-capabilities.test.ts | 8 + 16 files changed, 515 insertions(+), 7 deletions(-) create mode 100644 packages/coding-agent/test/task/executor-prewalk.test.ts diff --git a/docs/task-agent-discovery.md b/docs/task-agent-discovery.md index c725f61f0..be8a4beeb 100644 --- a/docs/task-agent-discovery.md +++ b/docs/task-agent-discovery.md @@ -24,7 +24,7 @@ It covers runtime behavior as implemented today, including precedence, invalid-d Task agents normalize into `AgentDefinition` (`src/task/types.ts`): - `name`, `description`, `systemPrompt` (required for a valid loaded agent) -- optional `tools`, `spawns`, `model`, `thinkingLevel`, `output`, `blocking`, `autoloadSkills`, `readSummarize` +- optional `tools`, `spawns`, `model`, `thinkingLevel`, `output`, `blocking`, `autoloadSkills`, `readSummarize`, `prewalk` - `source`: `"bundled" | "user" | "project"` - optional `filePath` @@ -36,6 +36,7 @@ Parsing comes from frontmatter via `parseAgentFields()` (`src/discovery/helpers. - backward-compat behavior: if `spawns` missing but `tools` includes `task`, `spawns` becomes `*` - `output` is passed through as opaque schema data - `read-summarize: false` (parsed as `readSummarize`) forces the subagent's `read` tool to return verbatim file content instead of structural summaries — `runSubprocess` applies it as a `read.summarize.enabled: false` override on the subagent's isolated settings (`src/task/executor.ts`). `scout` and `librarian` ship with it disabled. Defaults to enabled when the field is absent. +- `prewalk: true` starts the subagent on its resolved model and hands off to the default prewalk target (the `smol` role) at its first edit/write, exactly like the session-level `--prewalk`; a string value (e.g. `prewalk: "@smol"` or `prewalk: "openai/gpt-5-mini"`) picks a custom target. The `task.agentPrewalk` settings record (agent name → `"on"` / `"off"` / pattern, toggled per agent from `/agents` with `P`) overrides the frontmatter. Resolution happens in `runSubprocess` (`src/task/executor.ts`); an unresolvable target or a target equal to the starting model skips the hand-off instead of failing the spawn. ## Bundled agents @@ -44,7 +45,7 @@ Bundled agents are embedded at build time (`src/task/agents.ts`) using text impo `EMBEDDED_AGENT_DEFS` defines: - `scout`, `designer`, `reviewer`, `librarian` from prompt files -- `task` and `sonic` from shared `task.md` body plus injected frontmatter +- `task` and `sonic` from shared `task.md` body plus injected frontmatter; `task` ships with `prewalk: true` (default hand-off to the `smol` role, opt out per agent via `/agents` / `task.agentPrewalk`) Loading path: @@ -184,5 +185,6 @@ When parent plan mode is enabled, `TaskTool.#runSpawn` builds an `effectiveAgent - prepends the plan-mode subagent system prompt - restricts tools to `read`, `search`, `find`, `lsp`, and `web_search`, plus `ast_grep`/`report_finding` when the agent's own tool list declares them (`PLAN_MODE_AGENT_TOOL_ALLOWLIST`) - clears child spawns +- clears `prewalk` (read-only exploration must not receive the prewalk plan/implement nudges) The same `effectiveAgent` is used for subprocess launch, model/thinking overrides, and output-schema selection. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 05bfa6114..327dfefc6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added per-agent prewalk for subagents: a `prewalk` frontmatter field (`true` = hand off to the default prewalk target, a string = custom target model pattern) and a `task.agentPrewalk` settings override toggled per agent from the `/agents` dashboard with `P`. The bundled generic `task` agent ships with prewalk enabled by default (skipped when the target resolves to the subagent's own starting model, and never armed for plan-mode spawns). + ### Fixed - Fixed Bash internal URLs remaining unresolved when used as unquoted arguments inside command substitutions ([#5535](https://github.com/can1357/oh-my-pi/issues/5535)). diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index e7b3249a4..28e2b5a56 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -1055,6 +1055,37 @@ export function resolveAgentModelPatterns(options: AgentModelPatternResolutionOp activeModelPattern?.trim() || fallbackModelPattern?.trim() || settings?.getModelRole("default")?.trim() || ""; return resolveConfiguredModelPatterns(fallback, settings); } +/** Default prewalk hand-off target when no explicit target is configured. */ +export const DEFAULT_PREWALK_TARGET = "@smol"; + +export interface AgentPrewalkResolutionOptions { + /** `task.agentPrewalk` settings value for this agent: `"on"`, `"off"`, or a model pattern. */ + settingsOverride?: string; + /** Agent definition `prewalk` frontmatter: `true` = default target, string = custom target pattern. */ + agentPrewalk?: boolean | string; +} + +/** + * Effective prewalk target pattern for a subagent, or `undefined` when prewalk + * is disabled. The settings override decides enablement first ("off" wins, + * "on" enables with the agent's own target or {@link DEFAULT_PREWALK_TARGET}, + * any other value is a custom target pattern); otherwise the agent + * definition's `prewalk` field applies. Role aliases in the returned pattern + * are expanded later by {@link resolveModelOverride}. + */ +export function resolveAgentPrewalkPattern(options: AgentPrewalkResolutionOptions): string | undefined { + const agentPattern = + typeof options.agentPrewalk === "string" && options.agentPrewalk.trim() ? options.agentPrewalk.trim() : undefined; + const override = options.settingsOverride?.trim(); + if (override) { + const lowered = override.toLowerCase(); + if (lowered === "off" || lowered === "false") return undefined; + if (lowered === "on" || lowered === "true") return agentPattern ?? DEFAULT_PREWALK_TARGET; + return override; + } + if (options.agentPrewalk === true) return DEFAULT_PREWALK_TARGET; + return agentPattern; +} /** * Resolve a model role value into a concrete model and thinking metadata. diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index ae879faaf..8e7b5e6bb 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -4264,6 +4264,10 @@ export const SETTINGS_SCHEMA = { type: "record", default: {} as Record, }, + "task.agentPrewalk": { + type: "record", + default: {} as Record, + }, "tasks.todoClearDelay": { type: "number", diff --git a/packages/coding-agent/src/discovery/helpers.ts b/packages/coding-agent/src/discovery/helpers.ts index 093e1ef5d..26b5fb6da 100644 --- a/packages/coding-agent/src/discovery/helpers.ts +++ b/packages/coding-agent/src/discovery/helpers.ts @@ -232,6 +232,8 @@ export interface ParsedAgentFields { autoloadSkills?: string[]; readSummarize?: boolean; blocking?: boolean; + /** `true` = prewalk into the default target; string = prewalk into that model pattern. */ + prewalk?: boolean | string; } /** @@ -286,10 +288,28 @@ export function parseAgentFields(frontmatter: Record): ParsedAg const model = parseModelList(frontmatter.model); const blocking = parseBoolean(frontmatter.blocking); const readSummarize = parseBoolean(frontmatter.readSummarize); + // prewalk: true → hand off to the default prewalk target; "" → custom target. + let prewalk: boolean | string | undefined = parseBoolean(frontmatter.prewalk); + if (prewalk === undefined && typeof frontmatter.prewalk === "string") { + const trimmed = frontmatter.prewalk.trim(); + if (trimmed) prewalk = trimmed; + } const autoloadSkills = parseArrayOrCSV(frontmatter.autoloadSkills) ?.map(s => s.trim()) .filter(Boolean); - return { name, description, tools, spawns, model, output, thinkingLevel, blocking, autoloadSkills, readSummarize }; + return { + name, + description, + tools, + spawns, + model, + output, + thinkingLevel, + blocking, + autoloadSkills, + readSummarize, + prewalk, + }; } async function globIf( diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index 463a4cc22..8486b6abd 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -31,6 +31,7 @@ import { applyStartupCwd } from "./cli/startup-cwd"; import { findConfigFile } from "./config"; import { ModelRegistry } from "./config/model-registry"; import { + DEFAULT_PREWALK_TARGET, expandRoleAlias, getModelMatchPreferences, resolveCliModel, @@ -136,6 +137,7 @@ const HOST_DEFAULTED_SETTING_PATHS: SettingPath[] = [ "task.maxRecursionDepth", "task.disabledAgents", "task.agentModelOverrides", + "task.agentPrewalk", // Memory subsystems are off-by-default for RPC/ACP hosts; embedders that want // memory should opt in explicitly through their own settings layer. "memory.backend", @@ -938,13 +940,13 @@ export async function buildSessionOptions( ? true : activeSettings.get("prewalk.enabled"); if (prewalkEnabled) { - const rolePattern = expandRoleAlias(parsed.prewalkInto ?? "@smol", activeSettings); + const rolePattern = expandRoleAlias(parsed.prewalkInto ?? DEFAULT_PREWALK_TARGET, activeSettings); const resolved = resolveCliModel({ cliModel: rolePattern, modelRegistry, preferences: modelMatchPreferences }); if (resolved.warning) { process.stderr.write(`${chalk.yellow(`Warning: ${resolved.warning}`)}\n`); } if (resolved.error || !resolved.model) { - throw new Error(resolved.error ?? `Model "${parsed.prewalkInto ?? "@smol"}" not found`); + throw new Error(resolved.error ?? `Model "${parsed.prewalkInto ?? DEFAULT_PREWALK_TARGET}" not found`); } if (!modelRegistry.hasConfiguredAuth(resolved.model)) { throw new Error(`No API key for ${resolved.model.provider}/${resolved.model.id}`); diff --git a/packages/coding-agent/src/modes/components/agent-dashboard.ts b/packages/coding-agent/src/modes/components/agent-dashboard.ts index e7a63c6dd..a8e294ef5 100644 --- a/packages/coding-agent/src/modes/components/agent-dashboard.ts +++ b/packages/coding-agent/src/modes/components/agent-dashboard.ts @@ -40,6 +40,7 @@ import type { ModelRegistry } from "../../config/model-registry"; import { formatModelString, resolveAgentModelPatterns, + resolveAgentPrewalkPattern, resolveConfiguredModelPatterns, resolveModelOverride, } from "../../config/model-resolver"; @@ -72,6 +73,8 @@ interface SourceTab { interface DashboardAgent extends AgentDefinition { disabled: boolean; overrideModel?: string; + /** `task.agentPrewalk` value for this agent: "on", "off", or a model pattern. */ + prewalkOverride?: string; } interface ModelResolution { @@ -105,7 +108,7 @@ const SOURCE_LABEL: Record = { }; const LIST_FOOTER = - " ↑/↓: navigate Space: toggle Enter: model override N: new agent ←/→: source Ctrl+R: reload Esc: close"; + " ↑/↓: navigate Space: toggle Enter: model override P: prewalk N: new agent ←/→: source Ctrl+R: reload Esc: close"; const IDENTIFIER_PATTERN = /^[a-z0-9]+(?:-[a-z0-9]+){1,5}$/; function joinPatterns(patterns: string[]): string { @@ -258,6 +261,8 @@ class AgentInspectorPane implements Component { private readonly defaultResolution: ModelResolution | undefined, private readonly effectivePatterns: string[], private readonly effectiveResolution: ModelResolution | undefined, + private readonly prewalkPattern: string | undefined, + private readonly prewalkResolution: ModelResolution | undefined, ) {} render(width: number): readonly string[] { @@ -287,6 +292,7 @@ class AgentInspectorPane implements Component { lines.push( `${theme.fg("muted", "Effective:")} ${this.effectiveResolution ? this.#formatResolution(this.effectiveResolution) : theme.fg("dim", "(unresolved)")}`, ); + lines.push(`${theme.fg("muted", "Prewalk:")} ${this.#prewalkLabel()}`); if (this.agent.filePath) { lines.push(""); @@ -304,6 +310,23 @@ class AgentInspectorPane implements Component { return lines; } + /** "off", "on → target" (with source: agent default vs override), or the unresolved pattern. */ + #prewalkLabel(): string { + if (!this.agent) return theme.fg("dim", "off"); + const override = this.agent.prewalkOverride?.trim(); + const sourceTag = override + ? theme.fg("warning", " (override)") + : this.agent.prewalk !== undefined && this.agent.prewalk !== false + ? theme.fg("dim", " (agent default)") + : ""; + if (!this.prewalkPattern) { + return `${theme.fg("dim", "off")}${override ? sourceTag : ""}`; + } + const target = this.prewalkResolution + ? this.#formatResolution(this.prewalkResolution) + : theme.fg("dim", "(unresolved)"); + return `${theme.fg("success", "on")} ${theme.fg("dim", `${replaceTabs(this.prewalkPattern)} →`)} ${target}${sourceTag}`; + } #formatResolution(resolution: ModelResolution): string { return formatResolution(resolution); @@ -410,6 +433,7 @@ export class AgentDashboard extends Container { const { agents } = await discoverAgents(this.cwd); const disabled = new Set((this.#settingsManager?.get("task.disabledAgents") as string[] | undefined) ?? []); const overrides = this.#settingsManager?.get("task.agentModelOverrides") ?? {}; + const prewalkOverrides = this.#settingsManager?.get("task.agentPrewalk") ?? {}; this.#allAgents = agents .slice() @@ -422,6 +446,7 @@ export class AgentDashboard extends Container { ...agent, disabled: disabled.has(agent.name), overrideModel: overrides[agent.name]?.trim() || undefined, + prewalkOverride: prewalkOverrides[agent.name]?.trim() || undefined, })); this.#tabs = this.#buildTabs(this.#allAgents); @@ -556,6 +581,33 @@ export class AgentDashboard extends Container { } this.#settingsManager.set("task.agentModelOverrides", overrides); } + #persistPrewalkOverrides(): void { + if (!this.#settingsManager) return; + const overrides: Record = {}; + for (const agent of this.#allAgents) { + const value = agent.prewalkOverride?.trim(); + if (value) { + overrides[agent.name] = value; + } + } + this.#settingsManager.set("task.agentPrewalk", overrides); + } + + /** Cycle the prewalk override for the selected agent: agent default → on → off → agent default. */ + #cyclePrewalkOverride(): void { + const selected = this.#selectedAgent(); + if (!selected) return; + const current = selected.prewalkOverride?.trim().toLowerCase(); + selected.prewalkOverride = current === undefined || current === "" ? "on" : current === "on" ? "off" : undefined; + this.#persistPrewalkOverrides(); + const pattern = resolveAgentPrewalkPattern({ + settingsOverride: selected.prewalkOverride, + agentPrewalk: selected.prewalk, + }); + const state = selected.prewalkOverride ?? "agent default"; + this.#notice = `Prewalk for ${selected.name}: ${state}${pattern ? ` (into ${pattern})` : ""}`; + this.#buildLayout(); + } #toggleSelectedAgent(): void { const selected = this.#selectedAgent(); @@ -1026,6 +1078,13 @@ export class AgentDashboard extends Container { const defaultResolution = selected ? this.#resolvePatterns(defaultPatterns) : undefined; const effectivePatterns = selected ? this.#effectivePatternsFor(selected, selected.overrideModel) : []; const effectiveResolution = selected ? this.#resolvePatterns(effectivePatterns) : undefined; + const prewalkPattern = selected + ? resolveAgentPrewalkPattern({ + settingsOverride: selected.prewalkOverride, + agentPrewalk: selected.prewalk, + }) + : undefined; + const prewalkResolution = prewalkPattern ? this.#resolvePatterns([prewalkPattern]) : undefined; const listPane = new AgentListPane( this.#filteredAgents, @@ -1040,6 +1099,8 @@ export class AgentDashboard extends Container { defaultResolution, effectivePatterns, effectiveResolution, + prewalkPattern, + prewalkResolution, ); const bodyHeight = this.#computeBodyHeight(); this.addChild(new TwoColumnBody(listPane, inspector, bodyHeight)); @@ -1163,6 +1224,10 @@ export class AgentDashboard extends Container { this.#beginCreateFlow(); return; } + if (data.toLowerCase() === "p") { + this.#cyclePrewalkOverride(); + return; + } if (matchesKey(data, "backspace")) { if (this.#searchQuery.length > 0) { diff --git a/packages/coding-agent/src/prompts/agents/frontmatter.md b/packages/coding-agent/src/prompts/agents/frontmatter.md index 941ab049b..f2f715aa7 100644 --- a/packages/coding-agent/src/prompts/agents/frontmatter.md +++ b/packages/coding-agent/src/prompts/agents/frontmatter.md @@ -6,6 +6,7 @@ description: {{jsonStringify description}} {{/if}}{{#if model}}model: {{jsonStringify model}} {{/if}}{{#if thinkingLevel}}thinking-level: {{jsonStringify thinkingLevel}} {{/if}}{{#if blocking}}blocking: true +{{/if}}{{#if prewalk}}prewalk: {{jsonStringify prewalk}} {{/if}}{{#if autoloadSkills}}autoloadSkills: {{jsonStringify autoloadSkills}} {{/if}}--- {{body}} diff --git a/packages/coding-agent/src/task/agents.ts b/packages/coding-agent/src/task/agents.ts index 2b071e4a4..40aade7f0 100644 --- a/packages/coding-agent/src/task/agents.ts +++ b/packages/coding-agent/src/task/agents.ts @@ -25,6 +25,7 @@ interface AgentFrontmatter { model?: string | string[]; thinkingLevel?: string; blocking?: boolean; + prewalk?: boolean | string; } interface EmbeddedAgentDef { @@ -52,6 +53,9 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [ spawns: "*", model: "@task", thinkingLevel: AUTO_THINKING, + // Strong model plans and starts the implementation, then hands off to + // the smol role. Per-agent opt-out via /agents (task.agentPrewalk). + prewalk: true, }, template: taskMd, }, diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 958a7ec85..8f3f33a39 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -14,6 +14,7 @@ import { ModelRegistry } from "../config/model-registry"; import { formatModelSelectorValue, formatModelStringWithRouting, + resolveAgentPrewalkPattern, resolveModelOverride, resolveModelOverrideWithAuthFallback, } from "../config/model-resolver"; @@ -36,7 +37,7 @@ import submitReminderTemplate from "../prompts/system/subagent-yield-reminder.md import { AgentLifecycleManager } from "../registry/agent-lifecycle"; import { AgentRegistry } from "../registry/agent-registry"; import { type CreateAgentSessionOptions, createAgentSession, discoverAuthStorage } from "../sdk"; -import type { AgentSession, AgentSessionEvent } from "../session/agent-session"; +import type { AgentSession, AgentSessionEvent, Prewalk } from "../session/agent-session"; import type { ArtifactManager } from "../session/artifacts"; import type { AuthStorage } from "../session/auth-storage"; import { SKILL_PROMPT_MESSAGE_TYPE, USER_INTERRUPT_LABEL } from "../session/messages"; @@ -2376,6 +2377,36 @@ export async function runSubprocess(options: ExecutorOptions): Promise { test("returns undefined readSummarize when field absent", () => { expect(parseAgentFields({ name: "explore", description: "desc" })?.readSummarize).toBeUndefined(); }); + test("parses prewalk from boolean frontmatter", () => { + expect(parseAgentFields({ name: "worker", description: "desc", prewalk: true })?.prewalk).toBe(true); + expect(parseAgentFields({ name: "worker", description: "desc", prewalk: false })?.prewalk).toBe(false); + }); + + test("parses prewalk boolean strings as booleans", () => { + expect(parseAgentFields({ name: "worker", description: "desc", prewalk: "true" })?.prewalk).toBe(true); + expect(parseAgentFields({ name: "worker", description: "desc", prewalk: "false" })?.prewalk).toBe(false); + }); + + test("parses prewalk model pattern strings", () => { + expect(parseAgentFields({ name: "worker", description: "desc", prewalk: " @smol " })?.prewalk).toBe("@smol"); + expect(parseAgentFields({ name: "worker", description: "desc", prewalk: "openai/gpt-5-mini" })?.prewalk).toBe( + "openai/gpt-5-mini", + ); + }); + + test("ignores empty and absent prewalk values", () => { + expect(parseAgentFields({ name: "worker", description: "desc", prewalk: " " })?.prewalk).toBeUndefined(); + expect(parseAgentFields({ name: "worker", description: "desc" })?.prewalk).toBeUndefined(); + }); }); diff --git a/packages/coding-agent/test/model-resolver.test.ts b/packages/coding-agent/test/model-resolver.test.ts index 5b3152e40..da75db427 100644 --- a/packages/coding-agent/test/model-resolver.test.ts +++ b/packages/coding-agent/test/model-resolver.test.ts @@ -10,6 +10,7 @@ import { parseModelString, pickDefaultAvailableModel, resolveAgentModelPatterns, + resolveAgentPrewalkPattern, resolveAllowedModels, resolveCliModel, resolveModelFromString, @@ -736,6 +737,33 @@ describe("resolveModelRoleValue", () => { expect(result.explicitThinkingLevel).toBe(true); }); }); +describe("resolveAgentPrewalkPattern", () => { + test("agent definition alone decides: true → default target, pattern → custom, false/absent → off", () => { + expect(resolveAgentPrewalkPattern({ agentPrewalk: true })).toBe("@smol"); + expect(resolveAgentPrewalkPattern({ agentPrewalk: "@very-smol" })).toBe("@very-smol"); + expect(resolveAgentPrewalkPattern({ agentPrewalk: false })).toBeUndefined(); + expect(resolveAgentPrewalkPattern({})).toBeUndefined(); + }); + + test("settings override wins over the agent definition", () => { + expect(resolveAgentPrewalkPattern({ settingsOverride: "off", agentPrewalk: true })).toBeUndefined(); + expect(resolveAgentPrewalkPattern({ settingsOverride: "off", agentPrewalk: "@very-smol" })).toBeUndefined(); + expect(resolveAgentPrewalkPattern({ settingsOverride: "on", agentPrewalk: false })).toBe("@smol"); + expect(resolveAgentPrewalkPattern({ settingsOverride: "openai/gpt-4o", agentPrewalk: false })).toBe( + "openai/gpt-4o", + ); + }); + + test("override 'on' keeps the agent's custom target when one is defined", () => { + expect(resolveAgentPrewalkPattern({ settingsOverride: "on", agentPrewalk: "@very-smol" })).toBe("@very-smol"); + expect(resolveAgentPrewalkPattern({ settingsOverride: "on" })).toBe("@smol"); + }); + + test("blank override falls through to the agent definition", () => { + expect(resolveAgentPrewalkPattern({ settingsOverride: " ", agentPrewalk: true })).toBe("@smol"); + expect(resolveAgentPrewalkPattern({ settingsOverride: "", agentPrewalk: false })).toBeUndefined(); + }); +}); describe("resolveAgentModelPatterns", () => { test("falls back to the active session model when @task is unset", () => { const settings = Settings.isolated({ diff --git a/packages/coding-agent/test/task/executor-prewalk.test.ts b/packages/coding-agent/test/task/executor-prewalk.test.ts new file mode 100644 index 000000000..f7b14ba11 --- /dev/null +++ b/packages/coding-agent/test/task/executor-prewalk.test.ts @@ -0,0 +1,281 @@ +/** + * Per-agent prewalk resolution in `runSubprocess`: the agent definition's + * `prewalk` frontmatter and the `task.agentPrewalk` settings override decide + * whether the spawned session gets a `prewalk` hand-off config, which target + * model it resolves to, and when the hand-off is skipped (override off, + * target identical to the starting model). + */ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import type { Model } from "@oh-my-pi/pi-ai"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import type { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { LoadExtensionsResult } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types"; +import { AgentLifecycleManager } from "@oh-my-pi/pi-coding-agent/registry/agent-lifecycle"; +import { AgentRegistry } from "@oh-my-pi/pi-coding-agent/registry/agent-registry"; +import type { CreateAgentSessionResult } from "@oh-my-pi/pi-coding-agent/sdk"; +import * as sdkModule from "@oh-my-pi/pi-coding-agent/sdk"; +import type { AgentSession, AgentSessionEvent, PromptOptions } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { TaskTool } from "@oh-my-pi/pi-coding-agent/task"; +import * as discoveryModule from "@oh-my-pi/pi-coding-agent/task/discovery"; +import * as executorModule from "@oh-my-pi/pi-coding-agent/task/executor"; +import { runSubprocess } from "@oh-my-pi/pi-coding-agent/task/executor"; +import type { AgentDefinition, SingleResult } from "@oh-my-pi/pi-coding-agent/task/types"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { EventBus } from "@oh-my-pi/pi-coding-agent/utils/event-bus"; + +function yieldEmittingSession(): AgentSession { + const listeners: Array<(event: AgentSessionEvent) => void> = []; + const session = { + state: { messages: [] }, + agent: { state: { systemPrompt: ["test"] } }, + model: undefined, + extensionRunner: undefined, + sessionManager: { appendSessionInit: () => {} }, + getActiveToolNames: () => ["read", "yield"], + setActiveToolsByName: async (_toolNames: string[]) => {}, + subscribe: (listener: (event: AgentSessionEvent) => void) => { + listeners.push(listener); + return () => { + const index = listeners.indexOf(listener); + if (index >= 0) listeners.splice(index, 1); + }; + }, + prompt: async (_text: string, _options?: PromptOptions) => { + for (const listener of listeners) { + listener({ + type: "tool_execution_end", + toolCallId: "tool-prewalk", + toolName: "yield", + result: { + content: [{ type: "text", text: "Result submitted." }], + details: { status: "success", data: { ok: true } }, + }, + isError: false, + }); + } + }, + waitForIdle: async () => {}, + getLastAssistantMessage: () => undefined, + abort: async () => {}, + dispose: async () => {}, + }; + return session as unknown as AgentSession; +} + +function createSessionResult(session: AgentSession): CreateAgentSessionResult { + return { + session, + extensionsResult: { extensions: [], errors: [], runtime: {} as unknown } as unknown as LoadExtensionsResult, + setToolUIContext: () => {}, + eventBus: new EventBus(), + }; +} + +function modelOrThrow(id: string): Model { + const model = getBundledModel("anthropic", id); + if (!model) throw new Error(`Expected bundled model ${id}`); + return model; +} + +function createModelRegistry(models: Model[]): ModelRegistry { + return { + authStorage: {}, + refresh: async () => {}, + getAvailable: () => models, + getApiKey: async () => "test-key", + hasConfiguredAuth: () => true, + } as unknown as ModelRegistry; +} + +const baseAgent: AgentDefinition = { + name: "task", + description: "test", + systemPrompt: "test", + source: "bundled", +}; + +describe("runSubprocess per-agent prewalk", () => { + const primary = modelOrThrow("claude-sonnet-4-5"); + const target = modelOrThrow("claude-sonnet-4-6"); + + function baseOptions(id: string, settings: Settings) { + return { + cwd: "/tmp", + task: "do work", + index: 0, + id, + settings, + modelRegistry: createModelRegistry([primary, target]), + enableLsp: false, + }; + } + + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("resolves a frontmatter prewalk pattern to a target for the spawned session", async () => { + const spy = vi + .spyOn(sdkModule, "createAgentSession") + .mockResolvedValue(createSessionResult(yieldEmittingSession())); + + const result = await runSubprocess({ + ...baseOptions("subagent-prewalk-frontmatter", Settings.isolated()), + agent: { + ...baseAgent, + model: [`${primary.provider}/${primary.id}`], + prewalk: `${target.provider}/${target.id}`, + }, + }); + + expect(result.exitCode).toBe(0); + const forwarded = spy.mock.calls[0]?.[0]; + expect(forwarded?.prewalk?.target.id).toBe(target.id); + expect(forwarded?.prewalk?.target.provider).toBe(target.provider); + }); + + it("resolves prewalk: true through the smol role default target", async () => { + const settings = Settings.isolated(); + settings.setModelRole("smol", `${target.provider}/${target.id}`); + const spy = vi + .spyOn(sdkModule, "createAgentSession") + .mockResolvedValue(createSessionResult(yieldEmittingSession())); + + const result = await runSubprocess({ + ...baseOptions("subagent-prewalk-default-target", settings), + agent: { ...baseAgent, model: [`${primary.provider}/${primary.id}`], prewalk: true }, + }); + + expect(result.exitCode).toBe(0); + const forwarded = spy.mock.calls[0]?.[0]; + expect(forwarded?.prewalk?.target.id).toBe(target.id); + }); + + it("task.agentPrewalk 'off' disables a frontmatter-enabled prewalk", async () => { + const settings = Settings.isolated(); + settings.set("task.agentPrewalk", { task: "off" }); + const spy = vi + .spyOn(sdkModule, "createAgentSession") + .mockResolvedValue(createSessionResult(yieldEmittingSession())); + + const result = await runSubprocess({ + ...baseOptions("subagent-prewalk-off", settings), + agent: { + ...baseAgent, + model: [`${primary.provider}/${primary.id}`], + prewalk: `${target.provider}/${target.id}`, + }, + }); + + expect(result.exitCode).toBe(0); + expect(spy.mock.calls[0]?.[0]?.prewalk).toBeUndefined(); + }); + + it("task.agentPrewalk 'on' enables prewalk for an agent without frontmatter", async () => { + const settings = Settings.isolated(); + settings.setModelRole("smol", `${target.provider}/${target.id}`); + settings.set("task.agentPrewalk", { task: "on" }); + const spy = vi + .spyOn(sdkModule, "createAgentSession") + .mockResolvedValue(createSessionResult(yieldEmittingSession())); + + const result = await runSubprocess({ + ...baseOptions("subagent-prewalk-on", settings), + agent: { ...baseAgent, model: [`${primary.provider}/${primary.id}`] }, + }); + + expect(result.exitCode).toBe(0); + expect(spy.mock.calls[0]?.[0]?.prewalk?.target.id).toBe(target.id); + }); + + it("skips prewalk when the target resolves to the starting model", async () => { + const spy = vi + .spyOn(sdkModule, "createAgentSession") + .mockResolvedValue(createSessionResult(yieldEmittingSession())); + + const result = await runSubprocess({ + ...baseOptions("subagent-prewalk-same-model", Settings.isolated()), + agent: { + ...baseAgent, + model: [`${primary.provider}/${primary.id}`], + prewalk: `${primary.provider}/${primary.id}`, + }, + }); + + expect(result.exitCode).toBe(0); + expect(spy.mock.calls[0]?.[0]?.prewalk).toBeUndefined(); + }); +}); +// Plan-mode spawns are read-only exploration: the task tool must strip a +// prewalk-enabled agent definition before spawning so the hidden +// plan/implement nudges never reach an agent without edit tools. +describe("task tool plan-mode prewalk guard", () => { + const prewalkAgent: AgentDefinition = { + name: "task", + description: "General-purpose task agent", + systemPrompt: "You are a task agent.", + source: "bundled", + prewalk: true, + }; + + beforeEach(() => { + AgentRegistry.resetGlobalForTests(); + AgentLifecycleManager.resetGlobalForTests(); + }); + + afterEach(() => { + vi.restoreAllMocks(); + AgentLifecycleManager.resetGlobalForTests(); + AgentRegistry.resetGlobalForTests(); + }); + + function toolSession(planMode: boolean): ToolSession { + return { + cwd: "/tmp", + hasUI: false, + settings: Settings.isolated({ "task.isolation.mode": "none" }), + getSessionFile: () => null, + getSessionSpawns: () => "*", + getPlanModeState: () => (planMode ? { enabled: true, planFilePath: "local://PLAN.md" } : undefined), + } as unknown as ToolSession; + } + + async function spawnedAgentPrewalk(planMode: boolean): Promise { + vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ + agents: [prewalkAgent], + projectAgentsDir: null, + }); + let forwarded: AgentDefinition | undefined; + vi.spyOn(executorModule, "runSubprocess").mockImplementation(async (options): Promise => { + forwarded = options.agent; + return { + index: options.index ?? 0, + id: options.id ?? "X", + agent: "task", + agentSource: "bundled", + task: "t", + assignment: "do the thing", + exitCode: 0, + output: "done", + stderr: "", + truncated: false, + durationMs: 1, + tokens: 0, + requests: 1, + }; + }); + const tool = await TaskTool.create(toolSession(planMode)); + await tool.execute("tc", { task: "explore the thing" }); + expect(forwarded).toBeDefined(); + return forwarded?.prewalk; + } + + it("strips prewalk from the agent definition while plan mode is active", async () => { + expect(await spawnedAgentPrewalk(true)).toBeUndefined(); + }); + + it("keeps the agent definition's prewalk outside plan mode", async () => { + expect(await spawnedAgentPrewalk(false)).toBe(true); + }); +}); diff --git a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts index 022b2f057..574d6ed90 100644 --- a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts +++ b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts @@ -28,4 +28,12 @@ describe("task agent capability descriptions", () => { expect(agentByName(agents, name).readSummarize).toBeUndefined(); } }); + it("ships the generic task agent with prewalk enabled, all other bundled agents without", () => { + const agents = loadBundledAgents(); + + expect(agentByName(agents, "task").prewalk).toBe(true); + for (const name of ["scout", "sonic", "reviewer", "designer", "librarian"]) { + expect(agentByName(agents, name).prewalk).toBeUndefined(); + } + }); }); From b3145170ab9bd4a6eb24306a56c7912808eabac5 Mon Sep 17 00:00:00 2001 From: DarkPhilosophy <19309990+DarkPhilosophy@users.noreply.github.com> Date: Wed, 15 Jul 2026 02:32:37 +0300 Subject: [PATCH 045/293] fix(agent): surface provider stream failures --- packages/agent/CHANGELOG.md | 4 ++++ packages/agent/src/agent.ts | 12 +++--------- packages/agent/test/agent.test.ts | 30 +++++++++++++++++++----------- 3 files changed, 26 insertions(+), 20 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 6f84e20f9..efb1f8666 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Surfaced provider stream failures through the normal assistant message lifecycle so interactive clients show the terminal error instead of leaving users with a silent working spinner. + ## [16.5.2] - 2026-07-14 ### Fixed diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 280109497..896fd5fde 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -64,12 +64,6 @@ function defaultConvertToLlm(messages: AgentMessage[]): Message[] { }); } -const ANTHROPIC_OUTPUT_BLOCKED_PREFIX = "Output blocked by conten"; - -function isAnthropicOutputBlockedError(message: string): boolean { - return message.includes(ANTHROPIC_OUTPUT_BLOCKED_PREFIX); -} - function refreshToolChoiceForActiveTools( toolChoice: ToolChoice | undefined, tools: AgentContext["tools"] = [], @@ -1283,11 +1277,11 @@ export class Agent { : err instanceof Error ? err.message : String(err); - const shouldEmitVisibleOutputBlockedError = !stoppedForAbort && isAnthropicOutputBlockedError(errorMessage); + const shouldEmitVisibleError = !stoppedForAbort; const assistantPartial = partial?.role === "assistant" ? partial : undefined; const hadAssistantStart = assistantPartial !== undefined; const errorMsg: AssistantMessage = - shouldEmitVisibleOutputBlockedError && assistantPartial + shouldEmitVisibleError && assistantPartial ? { ...assistantPartial, stopReason: "error", errorMessage } : { role: "assistant", @@ -1308,7 +1302,7 @@ export class Agent { timestamp: Date.now(), }; - if (shouldEmitVisibleOutputBlockedError) { + if (shouldEmitVisibleError) { if (!hadAssistantStart) { this.#state.streamMessage = errorMsg; this.#emit({ type: "message_start", message: errorMsg }); diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index 178ff84be..95965a5c5 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -184,7 +184,7 @@ describe("Agent", () => { expect(lastMessage.errorMessage).toBe(errorText); }); - it("prompt() keeps unrelated provider stream failures out of the assistant lifecycle", async () => { + it("prompt() emits assistant error lifecycle for provider stream failures", async () => { const mock = createMockModel({ responses: [] }); const errorText = "connection reset"; const agent = new Agent({ @@ -201,17 +201,25 @@ describe("Agent", () => { await agent.prompt("trigger"); unsubscribe(); - expect(events.some(event => event.type === "message_start" && event.message.role === "assistant")).toBe(false); - expect(events.some(event => event.type === "message_end" && event.message.role === "assistant")).toBe(false); - const agentEnd = events.find(event => event.type === "agent_end"); - if (agentEnd?.type !== "agent_end") { - throw new Error("agent_end not emitted"); + const assistantStartIndex = events.findIndex( + event => event.type === "message_start" && event.message.role === "assistant", + ); + const assistantEndIndex = events.findIndex( + event => event.type === "message_end" && event.message.role === "assistant", + ); + const turnEndIndex = events.findIndex(event => event.type === "turn_end"); + const agentEndIndex = events.findIndex(event => event.type === "agent_end"); + expect(assistantStartIndex).toBeGreaterThan(-1); + expect(assistantEndIndex).toBeGreaterThan(assistantStartIndex); + expect(turnEndIndex).toBeGreaterThan(assistantEndIndex); + expect(agentEndIndex).toBeGreaterThan(turnEndIndex); + + const assistantEnd = events[assistantEndIndex]; + if (assistantEnd?.type !== "message_end" || assistantEnd.message.role !== "assistant") { + throw new Error("assistant message_end not emitted"); } - const errorMessage = agentEnd.messages.find(message => message.role === "assistant"); - if (errorMessage?.role !== "assistant") { - throw new Error("assistant error was not included in agent_end"); - } - expect(errorMessage.errorMessage).toBe(errorText); + expect(assistantEnd.message.stopReason).toBe("error"); + expect(assistantEnd.message.errorMessage).toBe(errorText); }); it("prompt() finalizes an existing assistant stream for Anthropic output-blocked stream errors", async () => { From 4086418227b4707f916689403c6ab2836209a157 Mon Sep 17 00:00:00 2001 From: DarkPhilosophy <19309990+DarkPhilosophy@users.noreply.github.com> Date: Wed, 15 Jul 2026 02:47:15 +0300 Subject: [PATCH 046/293] fix(agent): pair tools on failed partial streams --- packages/agent/src/agent-loop.ts | 47 ++++++++++++++++++------------- packages/agent/src/agent.ts | 28 ++++++++++++++++-- packages/agent/test/agent.test.ts | 44 +++++++++++++++++++++++++++++ 3 files changed, 97 insertions(+), 22 deletions(-) diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 937437a74..f508fbd71 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -2286,18 +2286,14 @@ function syntheticDetailsFor( } /** - * Create a tool result for a tool call that was emitted by the assistant but - * never invoked locally. Maintains the tool_use / tool_result pairing the - * provider API requires, and tags {@link SyntheticToolResultDetails} so - * consumers can distinguish this from a real local tool failure without - * string-matching the content (#4321). + * Create the persisted synthetic result for a tool call that was emitted by + * the assistant but never invoked locally. */ -function createAbortedToolResult( +export function createSyntheticToolResultMessage( toolCall: Extract, - stream: EventStream, reason: "aborted" | "error" | "skipped" | "length", errorMessage?: string, -): ToolResultMessage { +): ToolResultMessage { const message = reason === "aborted" ? "Tool execution was aborted" @@ -2307,9 +2303,31 @@ function createAbortedToolResult( ? "Tool call was not executed because the assistant ended its turn" : "Tool call was not executed because the provider stream ended with an error before the tool could run"; const details = syntheticDetailsFor(reason, errorMessage); - const result: AgentToolResult = { + return { + role: "toolResult", + toolCallId: toolCall.id, + toolName: toolCall.name, content: [{ type: "text", text: errorMessage ? `${message}: ${errorMessage}` : `${message}.` }], details, + isError: true, + timestamp: Date.now(), + }; +} + +/** + * Create and emit a tool result for a tool call that was emitted by the + * assistant but never invoked locally. + */ +function createAbortedToolResult( + toolCall: Extract, + stream: EventStream, + reason: "aborted" | "error" | "skipped" | "length", + errorMessage?: string, +): ToolResultMessage { + const toolResultMessage = createSyntheticToolResultMessage(toolCall, reason, errorMessage); + const result: AgentToolResult = { + content: toolResultMessage.content, + details: toolResultMessage.details, }; stream.push({ @@ -2326,17 +2344,6 @@ function createAbortedToolResult( result, isError: true, }); - - const toolResultMessage: ToolResultMessage = { - role: "toolResult", - toolCallId: toolCall.id, - toolName: toolCall.name, - content: result.content, - details, - isError: true, - timestamp: Date.now(), - }; - stream.push({ type: "message_start", message: toolResultMessage }); stream.push({ type: "message_end", message: toolResultMessage }); diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 896fd5fde..22456090f 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -31,6 +31,7 @@ import { abortReasonText, agentLoop, agentLoopContinue, + createSyntheticToolResultMessage, normalizeMessagesForProvider, normalizeTools, resolveOwnedDialectFromEnv, @@ -1311,8 +1312,31 @@ export class Agent { this.appendMessage(errorMsg); this.#state.error = errorMessage; this.#emit({ type: "message_end", message: errorMsg }); - this.#emit({ type: "turn_end", message: errorMsg, toolResults: [] }); - this.#emit({ type: "agent_end", messages: [errorMsg] }); + const toolResults: ToolResultMessage[] = []; + for (const block of errorMsg.content) { + if (block.type !== "toolCall") continue; + const toolResult = createSyntheticToolResultMessage(block, "error", errorMessage); + this.#emit({ + type: "tool_execution_start", + toolCallId: block.id, + toolName: block.name, + args: block.arguments, + intent: block.intent, + }); + this.#emit({ + type: "tool_execution_end", + toolCallId: block.id, + toolName: block.name, + result: { content: toolResult.content, details: toolResult.details }, + isError: true, + }); + this.appendMessage(toolResult); + this.#emit({ type: "message_start", message: toolResult }); + this.#emit({ type: "message_end", message: toolResult }); + toolResults.push(toolResult); + } + this.#emit({ type: "turn_end", message: errorMsg, toolResults }); + this.#emit({ type: "agent_end", messages: [errorMsg, ...toolResults] }); } else { this.appendMessage(errorMsg); this.#state.error = errorMessage; diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index 95965a5c5..897a980e2 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -222,6 +222,50 @@ describe("Agent", () => { expect(assistantEnd.message.errorMessage).toBe(errorText); }); + it("pairs tool calls from failed partial streams with synthetic tool results", async () => { + const mock = createMockModel({ responses: [] }); + const errorText = "connection reset after tool call"; + const started = createAssistantMessage([ + { type: "toolCall", id: "tool-1", name: "alpha", arguments: { value: "hello" } }, + ]); + const agent = new Agent({ + initialState: { model: mock.model, systemPrompt: ["Test"], tools: [], messages: [] }, + streamFn: () => { + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + stream.push({ type: "start", partial: started }); + stream.fail(new Error(errorText)); + }); + return stream; + }, + }); + const events: AgentEvent[] = []; + const unsubscribe = agent.subscribe(event => events.push(event)); + + await agent.prompt("trigger"); + unsubscribe(); + + const toolResult = agent.state.messages.find(message => message.role === "toolResult"); + expect(toolResult).toMatchObject({ + role: "toolResult", + toolCallId: "tool-1", + toolName: "alpha", + isError: true, + details: { + __synthetic: true, + source: "assistant_stop_error", + executed: false, + upstreamError: errorText, + }, + }); + + const turnEnd = events.find(event => event.type === "turn_end"); + expect(turnEnd).toMatchObject({ + type: "turn_end", + toolResults: [{ role: "toolResult", toolCallId: "tool-1", isError: true }], + }); + }); + it("prompt() finalizes an existing assistant stream for Anthropic output-blocked stream errors", async () => { const mock = createMockModel({ responses: [] }); const errorText = "Output blocked by content filtering policy"; From 5a7f107802a1e9487d35e8d79907d590637a3440 Mon Sep 17 00:00:00 2001 From: DarkPhilosophy <19309990+DarkPhilosophy@users.noreply.github.com> Date: Wed, 15 Jul 2026 03:01:09 +0300 Subject: [PATCH 047/293] fix(agent): preserve Cursor results on stream failure --- packages/agent/src/agent.ts | 10 +++++++ packages/agent/test/agent.test.ts | 48 ++++++++++++++++++++++++++++++- 2 files changed, 57 insertions(+), 1 deletion(-) diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 22456090f..6da223551 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -1313,8 +1313,18 @@ export class Agent { this.#state.error = errorMessage; this.#emit({ type: "message_end", message: errorMsg }); const toolResults: ToolResultMessage[] = []; + const bufferedCursorResults = this.#cursorToolResultBuffer.map(({ toolResult }) => toolResult); + this.#cursorToolResultBuffer = []; + const bufferedCursorToolCallIds = new Set(bufferedCursorResults.map(({ toolCallId }) => toolCallId)); + for (const toolResult of bufferedCursorResults) { + this.appendMessage(toolResult); + this.#emit({ type: "message_start", message: toolResult }); + this.#emit({ type: "message_end", message: toolResult }); + toolResults.push(toolResult); + } for (const block of errorMsg.content) { if (block.type !== "toolCall") continue; + if (bufferedCursorToolCallIds.has(block.id)) continue; const toolResult = createSyntheticToolResultMessage(block, "error", errorMessage); this.#emit({ type: "tool_execution_start", diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index 897a980e2..0714823bf 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -1,7 +1,8 @@ import { describe, expect, it } from "bun:test"; import { Agent, type AgentEvent, type AgentTool, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; -import { type SimpleStreamOptions, z } from "@oh-my-pi/pi-ai"; +import { type SimpleStreamOptions, type ToolResultMessage, z } from "@oh-my-pi/pi-ai"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; +import { kCursorExecResolved } from "@oh-my-pi/pi-ai/utils/block-symbols"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { createAssistantMessage } from "./helpers"; @@ -266,6 +267,51 @@ describe("Agent", () => { }); }); + it("preserves buffered Cursor results when a partial stream fails", async () => { + const mock = createMockModel({ responses: [] }); + const errorText = "connection reset after Cursor exec"; + const toolCall = { + type: "toolCall" as const, + id: "cursor-tool-1", + name: "shell", + arguments: { command: "pwd" }, + [kCursorExecResolved]: true, + }; + const started = createAssistantMessage([toolCall]); + const realToolResult: ToolResultMessage = { + role: "toolResult", + toolCallId: toolCall.id, + toolName: toolCall.name, + content: [{ type: "text", text: "/workspace" }], + isError: false, + timestamp: Date.now(), + }; + const agent = new Agent({ + initialState: { model: mock.model, systemPrompt: ["Test"], tools: [], messages: [] }, + cursorOnToolResult: message => message, + streamFn: (_model, _context, options) => { + const stream = new AssistantMessageEventStream(); + queueMicrotask(async () => { + await options?.cursorOnToolResult?.(realToolResult); + stream.push({ type: "start", partial: started }); + stream.fail(new Error(errorText)); + }); + return stream; + }, + }); + + await agent.prompt("trigger"); + + const toolResults = agent.state.messages.filter(message => message.role === "toolResult"); + expect(toolResults).toHaveLength(1); + expect(toolResults[0]).toMatchObject({ + toolCallId: toolCall.id, + toolName: toolCall.name, + content: [{ type: "text", text: "/workspace" }], + isError: false, + }); + }); + it("prompt() finalizes an existing assistant stream for Anthropic output-blocked stream errors", async () => { const mock = createMockModel({ responses: [] }); const errorText = "Output blocked by content filtering policy"; From d00e5548e276105eea639650cb4c78043758de6b Mon Sep 17 00:00:00 2001 From: DarkPhilosophy <19309990+DarkPhilosophy@users.noreply.github.com> Date: Wed, 15 Jul 2026 03:11:25 +0300 Subject: [PATCH 048/293] fix(agent): drop incomplete failed tool calls --- packages/agent/src/agent.ts | 17 +++++++++++++++-- packages/agent/test/agent.test.ts | 30 +++++++++++++++++++++++++++--- 2 files changed, 42 insertions(+), 5 deletions(-) diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 6da223551..0ebdb0f00 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -1200,6 +1200,7 @@ export class Agent { }; let partial: AgentMessage | null = null; + const completedToolCallIds = new Set(); try { const stream = messages @@ -1217,6 +1218,9 @@ export class Agent { case "message_update": partial = event.message; this.#state.streamMessage = event.message; + if (event.assistantMessageEvent.type === "toolcall_end") { + completedToolCallIds.add(event.assistantMessageEvent.toolCall.id); + } break; case "message_end": @@ -1281,9 +1285,19 @@ export class Agent { const shouldEmitVisibleError = !stoppedForAbort; const assistantPartial = partial?.role === "assistant" ? partial : undefined; const hadAssistantStart = assistantPartial !== undefined; + const bufferedCursorResults = this.#cursorToolResultBuffer.map(({ toolResult }) => toolResult); + const retainedToolCallIds = new Set(completedToolCallIds); + for (const { toolCallId } of bufferedCursorResults) retainedToolCallIds.add(toolCallId); const errorMsg: AssistantMessage = shouldEmitVisibleError && assistantPartial - ? { ...assistantPartial, stopReason: "error", errorMessage } + ? { + ...assistantPartial, + content: assistantPartial.content.filter( + block => block.type !== "toolCall" || retainedToolCallIds.has(block.id), + ), + stopReason: "error", + errorMessage, + } : { role: "assistant", content: [{ type: "text", text: "" }], @@ -1313,7 +1327,6 @@ export class Agent { this.#state.error = errorMessage; this.#emit({ type: "message_end", message: errorMsg }); const toolResults: ToolResultMessage[] = []; - const bufferedCursorResults = this.#cursorToolResultBuffer.map(({ toolResult }) => toolResult); this.#cursorToolResultBuffer = []; const bufferedCursorToolCallIds = new Set(bufferedCursorResults.map(({ toolCallId }) => toolCallId)); for (const toolResult of bufferedCursorResults) { diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index 0714823bf..30e24f1b4 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -226,15 +226,15 @@ describe("Agent", () => { it("pairs tool calls from failed partial streams with synthetic tool results", async () => { const mock = createMockModel({ responses: [] }); const errorText = "connection reset after tool call"; - const started = createAssistantMessage([ - { type: "toolCall", id: "tool-1", name: "alpha", arguments: { value: "hello" } }, - ]); + const toolCall = { type: "toolCall" as const, id: "tool-1", name: "alpha", arguments: { value: "hello" } }; + const started = createAssistantMessage([toolCall]); const agent = new Agent({ initialState: { model: mock.model, systemPrompt: ["Test"], tools: [], messages: [] }, streamFn: () => { const stream = new AssistantMessageEventStream(); queueMicrotask(() => { stream.push({ type: "start", partial: started }); + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall, partial: started }); stream.fail(new Error(errorText)); }); return stream; @@ -267,6 +267,30 @@ describe("Agent", () => { }); }); + it("drops incomplete tool calls when a partial stream fails before toolcall_end", async () => { + const mock = createMockModel({ responses: [] }); + const started = createAssistantMessage([{ type: "toolCall", id: "tool-1", name: "alpha", arguments: {} }]); + const agent = new Agent({ + initialState: { model: mock.model, systemPrompt: ["Test"], tools: [], messages: [] }, + streamFn: () => { + const stream = new AssistantMessageEventStream(); + queueMicrotask(() => { + stream.push({ type: "start", partial: started }); + stream.push({ type: "toolcall_start", contentIndex: 0, partial: started }); + stream.push({ type: "toolcall_delta", contentIndex: 0, delta: '{"value":', partial: started }); + stream.fail(new Error("connection reset during tool arguments")); + }); + return stream; + }, + }); + + await agent.prompt("trigger"); + + const assistant = agent.state.messages.find(message => message.role === "assistant"); + expect(assistant?.content.some(block => block.type === "toolCall")).toBe(false); + expect(agent.state.messages.some(message => message.role === "toolResult")).toBe(false); + }); + it("preserves buffered Cursor results when a partial stream fails", async () => { const mock = createMockModel({ responses: [] }); const errorText = "connection reset after Cursor exec"; From 1227727827d0d65d9053fc49f0de7d44680202d8 Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 15 Jul 2026 00:44:40 +0000 Subject: [PATCH 049/293] fix(amazon-bedrock): guarded baseMessage against undefined stringify streamBedrock's catch block computed baseMessage via JSON.stringify(error) for non-Error values, which returns undefined (not "undefined") for undefined, functions, and circular objects. The next line called baseMessage.includes(...), throwing an unhandled TypeError and leaving the stream broken instead of emitting a clean error event. Added a String(error) fallback so baseMessage is always a string. Fixes #5539 --- packages/ai/CHANGELOG.md | 4 ++++ packages/ai/src/providers/amazon-bedrock.ts | 2 +- .../ai/test/bedrock-inference-profile.test.ts | 19 +++++++++++++++++++ 3 files changed, 24 insertions(+), 1 deletion(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3464f09cd..7d69fa57e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Amazon Bedrock stream error handler crashing with `TypeError: undefined is not an object (evaluating 'baseMessage.includes')` when a non-`Error` value is thrown and `JSON.stringify` returns `undefined` ([#5539](https://github.com/can1357/oh-my-pi/issues/5539)). + ## [16.5.2] - 2026-07-14 ### Added diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index b30d8a9aa..93692e8e6 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -510,7 +510,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = ( for (const block of output.content) { if (block.type === "toolCall") clearStreamingPartialJson(block); } - const baseMessage = error instanceof Error ? error.message : JSON.stringify(error); + const baseMessage = error instanceof Error ? error.message : (JSON.stringify(error) ?? String(error)); // Enrich error with thinking block diagnostics for signature-related failures let diagnostics = ""; if (baseMessage.includes("signature") || baseMessage.includes("thinking")) { diff --git a/packages/ai/test/bedrock-inference-profile.test.ts b/packages/ai/test/bedrock-inference-profile.test.ts index ddf0eb1b6..75092e7f4 100644 --- a/packages/ai/test/bedrock-inference-profile.test.ts +++ b/packages/ai/test/bedrock-inference-profile.test.ts @@ -148,3 +148,22 @@ describe("Bedrock cross-region inference-profile geo routing", () => { }); }); }); + +describe("Bedrock error handling", () => { + // Regression (#5539): a non-`Error` thrown inside the stream body where + // `JSON.stringify` returns `undefined` (e.g. `undefined`, a function, a + // circular object) must not crash the catch block via `baseMessage.includes(...)`. + // The stream must close cleanly with an error result instead of an unhandled + // `TypeError: undefined is not an object (evaluating 'baseMessage.includes')`. + test("surfaces a stream error when a non-Error value is thrown", async () => { + const result = await streamBedrock(profileModel, userContext(), { + bearerToken: "test-token", + maxTokens: 16, + onPayload: () => { + throw undefined; + }, + }).result(); + + expect(result.stopReason).toBe("error"); + }); +}); From d600cce75b0b00f4403ffe6130133a8c2802dad2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Wed, 15 Jul 2026 00:49:22 +0000 Subject: [PATCH 050/293] fix(amazon-bedrock): handled stringify exceptions Guard diagnostic JSON serialization so circular objects and BigInt values cannot escape the provider catch block. Expanded the regression coverage to undefined, BigInt, and circular thrown values. Fixes #5539 --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/providers/amazon-bedrock.ts | 7 ++++++- .../ai/test/bedrock-inference-profile.test.ts | 16 +++++++++------- 3 files changed, 16 insertions(+), 9 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 7d69fa57e..d6df74d4b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Amazon Bedrock stream error handler crashing with `TypeError: undefined is not an object (evaluating 'baseMessage.includes')` when a non-`Error` value is thrown and `JSON.stringify` returns `undefined` ([#5539](https://github.com/can1357/oh-my-pi/issues/5539)). +- Fixed Amazon Bedrock stream error handling for non-`Error` values that `JSON.stringify` cannot serialize ([#5539](https://github.com/can1357/oh-my-pi/issues/5539)). ## [16.5.2] - 2026-07-14 diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 93692e8e6..6a3cf1f02 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -510,7 +510,12 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = ( for (const block of output.content) { if (block.type === "toolCall") clearStreamingPartialJson(block); } - const baseMessage = error instanceof Error ? error.message : (JSON.stringify(error) ?? String(error)); + let baseMessage: string; + try { + baseMessage = error instanceof Error ? error.message : (JSON.stringify(error) ?? String(error)); + } catch { + baseMessage = String(error); + } // Enrich error with thinking block diagnostics for signature-related failures let diagnostics = ""; if (baseMessage.includes("signature") || baseMessage.includes("thinking")) { diff --git a/packages/ai/test/bedrock-inference-profile.test.ts b/packages/ai/test/bedrock-inference-profile.test.ts index 75092e7f4..08314bb84 100644 --- a/packages/ai/test/bedrock-inference-profile.test.ts +++ b/packages/ai/test/bedrock-inference-profile.test.ts @@ -150,17 +150,19 @@ describe("Bedrock cross-region inference-profile geo routing", () => { }); describe("Bedrock error handling", () => { - // Regression (#5539): a non-`Error` thrown inside the stream body where - // `JSON.stringify` returns `undefined` (e.g. `undefined`, a function, a - // circular object) must not crash the catch block via `baseMessage.includes(...)`. - // The stream must close cleanly with an error result instead of an unhandled - // `TypeError: undefined is not an object (evaluating 'baseMessage.includes')`. - test("surfaces a stream error when a non-Error value is thrown", async () => { + const circular: Record = {}; + circular.self = circular; + + test.each([ + ["undefined", undefined], + ["BigInt", 1n], + ["circular object", circular], + ])("surfaces a stream error when %s is thrown", async (_name, thrown) => { const result = await streamBedrock(profileModel, userContext(), { bearerToken: "test-token", maxTokens: 16, onPayload: () => { - throw undefined; + throw thrown; }, }).result(); From af1832af1bd36070a814c3bb175c3655ee44e29f Mon Sep 17 00:00:00 2001 From: can1357 Date: Wed, 15 Jul 2026 01:02:01 +0200 Subject: [PATCH 051/293] feat(coding-agent/prompts): refined tool prompts for shell, browser, and eval workflows - Simplified `bash` guidance to tighten allowed command patterns, pipeline limits, and launch-based process handling. - Reworked `browser` instructions into grouped helper sections while preserving selector restrictions and key action semantics. - Harmonized `eval`, `irc`, `read`, and `todo` prompt wording around state reuse, messaging, selector formats, and task operations. --- .../src/prompts/tools/ast-edit.md | 30 +++---- .../src/prompts/tools/ast-grep.md | 32 +++----- .../coding-agent/src/prompts/tools/bash.md | 81 ++++--------------- .../coding-agent/src/prompts/tools/browser.md | 57 +++++-------- .../coding-agent/src/prompts/tools/debug.md | 21 ++--- .../coding-agent/src/prompts/tools/eval.md | 72 +++++------------ .../coding-agent/src/prompts/tools/grep.md | 22 ++--- .../src/prompts/tools/image-gen.md | 6 +- .../coding-agent/src/prompts/tools/irc.md | 29 ++----- .../coding-agent/src/prompts/tools/lsp.md | 33 +++----- .../coding-agent/src/prompts/tools/read.md | 81 ++++--------------- .../coding-agent/src/prompts/tools/task.md | 55 ++++++------- .../coding-agent/src/prompts/tools/todo.md | 14 ++-- .../test/tools/eval-description.test.ts | 10 --- .../coding-agent/test/tools/index.test.ts | 10 +++ .../test/tools/schema-validation.test.ts | 21 +---- 16 files changed, 173 insertions(+), 401 deletions(-) diff --git a/packages/coding-agent/src/prompts/tools/ast-edit.md b/packages/coding-agent/src/prompts/tools/ast-edit.md index 78bd577b6..7627819fd 100644 --- a/packages/coding-agent/src/prompts/tools/ast-edit.md +++ b/packages/coding-agent/src/prompts/tools/ast-edit.md @@ -1,22 +1,10 @@ -Structural AST-aware rewrites via ast-grep. +Structural AST-aware rewrites via ast-grep. Use for codemods where text replace is unsafe. Narrow each call to one language. - -- Use for codemods / structural rewrites where text replace is unsafe -- Narrow each call to one language -- Metavariables captured in `pat` (`$A`, `$$$ARGS`) substitute into that entry's `out` template -- **Patterns match AST structure, not text.** `$NAME` = one node (captured); `$_` = one without binding; `$$$NAME` = zero-or-more; `$$$` = zero-or-more without binding. Use `$$$NAME`, NOT `$$NAME` — the two-dollar form is invalid. Metavariable names are UPPERCASE and MUST be the whole AST node — partial text like `prefix$VAR` or `"hello $NAME"` does NOT work -- Same metavariable twice → both occurrences MUST match identical code (`$A == $A` matches `x == x`, not `x == y`) -- Rewrite patterns MUST parse as a single valid AST node. Non-standalone snippets → wrap in context, e.g. `class $_ { … }` -- TS declarations/methods — tolerate unknown annotations: `async function $NAME($$$ARGS): $_ { $$$BODY }` or `class $_ { method($ARG: $_): $_ { $$$BODY } }` -- Delete matched code with empty `out`: `{"pat":"console.log($$$)","out":""}` -- Each rewrite is a 1:1 substitution — no splitting a capture across nodes or merging captures - - - -- Change diffs: `[src/foo.ts#1A2B]`, `-12:before`, `+12:after` - - - -- Parse issues mean the rewrite is malformed or mis-scoped — fix the pattern before assuming a clean no-op -- For one-off local text edits, you SHOULD prefer the Edit tool - +- Metavariables in `pat` (`$A`, `$$$ARGS`) substitute into `out`. +- **Patterns match AST structure, not text.** `$NAME` = one node; `$_` = unbound; `$$$NAME` = zero-or-more. + - Use `$$$NAME`, NOT `$$NAME` (invalid). Names UPPERCASE, whole node — partial like `prefix$VAR` fails. +- Same metavariable twice → MUST match identical code (`$A == $A` matches `x == x`, not `x == y`). +- Rewrite patterns MUST parse as single AST node. Non-standalone → wrap: `class $_ { … }`. +- TS: tolerate annotations — `async function $NAME($$$ARGS): $_ { $$$BODY }`. Delete with empty `out`: `{"pat":"console.log($$$)","out":""}`. +- 1:1 substitution — no splitting/merging captures. +- Parse issues → malformed rewrite, not clean no-op. For one-off text edits, prefer the Edit tool. diff --git a/packages/coding-agent/src/prompts/tools/ast-grep.md b/packages/coding-agent/src/prompts/tools/ast-grep.md index 8948637a5..e7ee59f86 100644 --- a/packages/coding-agent/src/prompts/tools/ast-grep.md +++ b/packages/coding-agent/src/prompts/tools/ast-grep.md @@ -1,25 +1,19 @@ -Structural code search via ast-grep. +Structural code search via ast-grep. Use when syntax shape matters more than text (calls, declarations, language constructs). -- Use when syntax shape matters more than text (calls, declarations, language constructs) -- Narrow each call to one language -- `pat` is ONE AST pattern; separate calls for unrelated patterns -- `$NAME` captures one node; `$_` matches one without binding; `$$$NAME` captures zero-or-more; `$$$` matches zero-or-more without binding. Use `$$$NAME`, NOT `$$NAME` — the two-dollar form is invalid -- Metavariable names are UPPERCASE and MUST be the whole AST node — partial text like `prefix$VAR`, `"hello $NAME"`, or `a $OP b` does NOT work -- Same metavariable twice → both occurrences MUST match identical code (`$A == $A` matches `x == x`, not `x == y`) -- Patterns MUST parse as a single valid AST node. Non-standalone snippets → wrap in context, e.g. `class $_ { … }` -- C++ expression-statement calls need trailing `;`: `ns::doThing($ARG);`, `$CALLEE($ARG);` -- TS declarations/methods — tolerate unknown annotations: `async function $NAME($$$ARGS): $_ { $$$BODY }` or `class $_ { method($ARG: $_): $_ { $$$BODY } }` -- Declaration forms are distinct shapes — `function foo`, method `foo()`, `const foo = () => {}`; search the right form before concluding absence -- Loosest existence check: `pat: "executeBash"` with narrow `path` +- Narrow each call to one language. `pat` is ONE AST pattern; separate calls for unrelated patterns. +- `$NAME` captures one node; `$_` matches without binding; `$$$NAME` zero-or-more; `$$$` zero-or-more unbound. + - Use `$$$NAME`, NOT `$$NAME` (invalid). Names UPPERCASE, whole node — `prefix$VAR` fails. +- Same metavariable twice → MUST match identical code (`$A == $A` matches `x == x`, not `x == y`). +- Patterns MUST parse as single AST node. Non-standalone → wrap: `class $_ { … }`. +- C++ expression-statement calls need trailing `;`: `ns::doThing($ARG);`, `$CALLEE($ARG);`. +- TS: tolerate annotations — `async function $NAME($$$ARGS): $_ { $$$BODY }`. +- Declaration forms are distinct — `function foo`, method `foo()`, `const foo = () => {}`; search the right form before concluding absence. +- Loosest existence check: `pat: "executeBash"` with narrow `path`. - -- Matches under a snapshot tag header: `[src/foo.ts#1A2B]`, `*42:` matched, ` 43:` context - - -- AVOID repo-root scans — narrow `path` first -- Parse issues = query failure, not absence: fix the pattern or tighten `path` before concluding "no matches" -- Broad cross-subsystem exploration: you SHOULD use the Task tool + scout subagent first +- AVOID repo-root scans — narrow `path` first. +- Parse issues = query failure, not absence: fix pattern or tighten `path` before concluding "no matches". +- Broad cross-subsystem exploration → Task tool + scout subagent first. diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index fd5b6921b..931dfc587 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -1,72 +1,25 @@ -Runs commands in the embedded shell — terminal ops: git, bun, cargo, python. +Runs commands in the embedded shell. NOT full GNU Bash — invokes real binaries with simple args. -# When to use bash — and when not to - -The shell invokes **real binaries** with simple args. It is NOT full GNU Bash. - -Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a fact and does not depend on shell-specific regex/quoting (`wc -l`, `sort | uniq -c`, `comm`, `diff`, a checksum, `git status`). -{{#if hasLaunch}}Long-running service, watcher, debugger, REPL, or process needing later input? MUST use `launch`, not bash.{{/if}} - -{{#if hasEval}}Anything below → `eval` cell, not bash: -- Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language -- Heredocs (`< -- `cwd` sets the working dir, not `cd dir && …` -- `env: { NAME: "…" }` for multiline / quote-heavy / untrusted values; reference `$NAME` -- Quote expansions (`"$NAME"`) to preserve exact content -- `pty: true` only when the command needs a real terminal (`sudo`, `ssh` needing input); default `false` -- `;` only when later commands should run despite earlier failures -- Multiple bash calls per message run concurrently. NEVER split order-dependent commands across parallel calls — chain with `&&` in one call. -- Internal URIs (`skill://`, `agent://`, …) auto-resolve to FS paths -{{#if hasEval}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly.{{else}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Use a checked-in script, purpose-built tool, or single command that owns the output shape.{{/if}} -{{#if asyncEnabled}} -- `async: true` defers reporting for finite commands that need no later input; completion arrives as a follow-up. -{{/if}} +- `cwd` sets working dir (not `cd dir && …`). `env: { NAME: "…" }` for multiline/quote-heavy values; `"$NAME"` to expand. +- `pty: true` only for real terminal needs (`sudo`, `ssh`); default `false`. +- Multiple calls run concurrently; NEVER split order-dependent commands — chain with `&&` in one call (`;` only to continue past failure). +- Internal URIs (`skill://`, `agent://`, …) auto-resolve to FS paths. +{{#if asyncEnabled}}- `async: true` defers reporting for finite commands needing no later input.{{/if}} -{{#if hasEval}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.{{else}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a shell program; use a purpose-built tool or checked-in script instead.{{/if}} -{{#if hasGrep}}- NEVER shell out to search content or files: `grep/rg` → `grep`.{{else}}- Avoid shelling out for broad content search; use an active search/read tool when one is available.{{/if}} -{{#if hasRead}}{{#if hasGlob}}- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing.{{else}}- Prefer `read` for known file and directory reads. Only use shell listing when no file-listing tool is active.{{/if}}{{else}}{{#if hasGlob}}- Prefer `glob` for file discovery; avoid `find` when `glob` is active.{{else}}- If no file read/listing tool is active, keep shell inspection narrow and state that limitation.{{/if}}{{/if}} -- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`. -{{#if hasLaunch}}- NEVER launch daemons, watchers, dev servers, debuggers, or REPLs through bash/background shell syntax — use `launch`.{{/if}} +{{#if hasGrep}}- NEVER shell out to search: `grep`/`rg` → built-in `grep`.{{/if}} +{{#if hasRead}}{{#if hasGlob}}- NEVER use `ls` or `find` — `ls` → `read`, `find` → `glob`. NON-NEGOTIABLE.{{/if}}{{/if}} +- Avoid head/tail/redirections: stderr merged, output auto-truncated, full capture at `artifact://`. +{{#if hasLaunch}}- NEVER launch daemons/watchers/servers/debuggers/REPLs through bash — use `launch`.{{/if}} - -- Returns output (stderr merged into stdout); exit code shown on non-zero exit. -- Truncated output → `artifact://` (linked in metadata). - - -{{#if asyncEnabled}} -# Timeout and async - -- `timeout` is seconds; nonzero values are clamped to `1..3600` and the process is killed on elapse. Set `timeout: 0` only for finite commands whose completion is cancellation-owned. -- `async: true` defers only reporting; it does NOT extend a nonzero timeout. -{{#if hasLaunch}}- Need a service, watcher, debugger, REPL, or later stdin? MUST use `launch`. NEVER use `cmd &`, `nohup`, or async bash as a process supervisor.{{else}}- Need a long-running process or >3600s run? Use an external process supervisor; avoid detached shell jobs you cannot later observe or stop.{{/if}} -{{/if}} -{{#if autoBackgroundEnabled}} - -## Auto-background - -- A long-running foreground call may convert to a background job; the final result arrives as a follow-up tool call. NOT a failure — don't retry or wait synchronously. -- Need the result inline (e.g. piping into another command)? Raise `timeout` above expected duration{{#if asyncEnabled}}, or set `async: true` up front{{/if}}. -{{/if}} - -# Output minimizer - -- Long output truncated; test/lint runner output filtered to failures. When visible text changed, a `[raw output: artifact://]` footer links the full capture — read it if a run looks suspicious or you need exact bytes. -- No footer = what you see is exactly what the command emitted. +{{#if asyncEnabled}}- `timeout`: nonzero clamped 1–3600, killed on elapse. `0` only for cancellation-owned. `async: true` defers reporting only, doesn't extend timeout.{{/if}} +{{#if autoBackgroundEnabled}}- Long foreground calls may auto-background; result arrives as follow-up — NOT a failure. Need inline? Raise timeout{{#if asyncEnabled}} or `async: true`{{/if}}.{{/if}} +- Long output truncated, test/lint filtered to failures. `[raw output: artifact://]` footer links full capture. No footer = what you see is exact output. diff --git a/packages/coding-agent/src/prompts/tools/browser.md b/packages/coding-agent/src/prompts/tools/browser.md index 27121f2b9..efc1b3f12 100644 --- a/packages/coding-agent/src/prompts/tools/browser.md +++ b/packages/coding-agent/src/prompts/tools/browser.md @@ -1,45 +1,26 @@ Drives real Chromium tab; full puppeteer access via JS. -- Static content (articles, docs, issues/PRs, JSON, PDFs, feeds)? `read` the URL. Browser only for JS execution, auth, interactive actions. -- Three actions: - - `open` — acquire/reuse named tab (`name` defaults `"main"`). Optional `url` (navigate once ready), `viewport`, `dialogs: "accept" | "dismiss"` (auto-handle `alert`/`confirm`/`beforeunload`; else page hangs till you wire `page.on('dialog', …)`). - - `close` — release tab by `name`, or all with `all: true`. `kill: true` also kills spawned-app process trees. - - `run` — execute JS in existing tab. `code` = async function body; `page`, `browser`, `tab`, `display`, `assert`, `wait` in scope. Return value JSON-stringified into result; `display(value)` accumulates text/images. `wait(ms)` sleeps; `wait(fn, { timeout?, interval? })` polls `fn` (sync or async) until truthy and resolves with that value (default 100ms interval; deadline min(30s, cell budget − 1s), named error on timeout) — use it instead of in-page polling Promises inside `tab.evaluate`. -- Tabs survive `run` calls and in-process subagents — open once, reuse. -- Browser kinds (`app` on `open`): - - default (no `app`) → headless Chromium with stealth patches. - - `app.path` → spawn absolute binary (Electron/CDP). No stealth patches — NEVER tamper with a real desktop app. - - `app.cdp_url` → connect to existing CDP endpoint (e.g. `http://127.0.0.1:9222`). - - `app.target` (with `path`/`cdp_url`) — substring on url+title picks BrowserWindow. -- `tab` helpers; drop to raw puppeteer `page` for anything uncovered: - - `tab.goto(url, { waitUntil? })` — navigate. A hung load fails ~1s before the cell budget with a named, catchable error and the pending navigation is stopped; for slow pages raise `timeout` or use `waitUntil: "domcontentloaded"`. - - `tab.observe({ includeAll?, viewportOnly? })` — accessibility snapshot: `{ url, title, viewport, scroll, elements: [{ id, role, name, value, states, … }] }`. Ids stable until next observe/goto. - - `tab.ariaSnapshot(selector?, { depth?, boxes? })` — Playwright-format ARIA-tree YAML (nested roles + accessible names + `/url`/`/placeholder`), scoped to `selector` or the whole document. Every node carries a `[ref=eN]` id; `[cursor=pointer]` flags clickables. Captures dense, hierarchical structure/text that `observe()`'s flat list flattens away. Refs renumber from e1 each call and stay valid until the next `ariaSnapshot()`. - - `tab.ref("e5")` — `[ref=eN]` from the last ariaSnapshot → element handle with the common action methods (`.click()`, `.type()`, `.fill()`, `.hover()`, `.evaluate()`, …); the primary way to act on a ref. For convenience `aria-ref=e5` also works inline in `tab.click`/`type`/`fill`/`waitFor`/`scrollIntoView` (e.g. `tab.click("aria-ref=e5")`). - - `tab.id(n)` — id from last observe → element handle with the same action methods (`.click()`, `.type()`, `.fill()`, …). - - `tab.click(selector)` / `tab.type(selector, text)` / `tab.fill(selector, value)` / `tab.press(key, { selector? })` / `tab.scroll(dx, dy)`. - - `tab.waitFor(selector, { timeout? })` / `tab.waitForSelector(selector, { timeout?, visible?, hidden? })` — wait until attached (optionally visible/hidden); returns an action-method handle. - - `tab.drag(from, to)` — endpoints: selector (center-to-center) or `{ x, y }` viewport point (canvases, sliders). - - `tab.scrollIntoView(selector)` — center in viewport; before clicking off-screen elements. - - `tab.select(selector, …values)` — set ``; paths relative to cwd. - - `tab.waitForUrl(pattern, { timeout? })` — substring or `RegExp` (matches SPA pushState nav); returns matched URL. - - `tab.waitForResponse(pattern, { timeout? })` — substring, `RegExp`, or `(response) => boolean`; returns puppeteer `HTTPResponse` (`.text()`/`.json()`/`.status()`/`.headers()`). - - `tab.waitForNavigation({ waitUntil?, timeout? })` — resolves on the next navigation. Start it BEFORE the click/submit that triggers it; after `tab.goto` (which already waits) use `tab.waitForUrl`/`tab.waitForSelector` instead. - - `tab.evaluate(fn, …args)` — run ad-hoc code in the page's MAIN world. DOM and page-defined globals (`window.myFlag`) are visible; mutations affect the page. - - `tab.screenshot({ selector?, fullPage?, save?, silent? })` — capture + attach for viewing (`silent: true` skips). Pass `save` only when a later step needs the file. - - `tab.extract(format = "markdown")` — readable page content (`"markdown"` | `"text"`); throws when nothing readable. -- Selectors: CSS + puppeteer handlers `aria/Sign in`, `text/Continue`, `xpath/…`, `pierce/…`; also Playwright-style `p-aria/…`, `p-text/…`. Playwright-only engines/pseudos (`:has-text()`, `:visible`, …) are rejected — use `text/…` or `aria/…`. A stalled action/wait fails fast with a named `tab.` error carrying a match-count diagnosis, never the whole-cell timeout; a selector matching nothing fails in ~2s (pass an explicit `{ timeout }` to `waitFor`/`waitForSelector` to wait out slow-appearing elements). A whole-cell timeout names the stalled op (including `wait(…)`) and any unhandled dialog blocking the page. +- Static content? `read` the URL. Browser only for JS execution, auth, interactive actions. +- `open` → `run` — tabs survive calls and subagents, open once reuse. +- `run` scope: `page`, `browser`, `tab`, `display`, `assert`, `wait` available. `wait(fn)` polls until truthy — use instead of polling inside `tab.evaluate`. + +- `tab` helpers (drop to raw puppeteer `page` for anything uncovered): + Element handles: `tab.ref("e5")` / `tab.id(n)`. Also `aria-ref=e5` inline. + Simple: `tab.goto`, `tab.click`, `tab.type`, `tab.fill`, `tab.press`, `tab.scroll`, `tab.scrollIntoView`, `tab.drag`, `tab.uploadFile`, `tab.select`, `tab.screenshot`, `tab.extract`, `tab.evaluate`. + Waits: `tab.waitFor`, `tab.waitForSelector`, `tab.waitForUrl`, `tab.waitForResponse`, `tab.waitForNavigation`. + Snapshots: `tab.observe()` → accessibility tree; `tab.ariaSnapshot()` → ARIA YAML with `[ref=eN]`. + + Gotchas: + - `tab.fill` NEVER works for `