From ad444f652cbf42038eddc1f6fc72eeab77e2aaf9 Mon Sep 17 00:00:00 2001 From: DarkPhilosophy <19309990+DarkPhilosophy@users.noreply.github.com> Date: Fri, 5 Jun 2026 20:30:16 +0300 Subject: [PATCH 001/181] fix(coding-agent): retry orphaned toolUse stops to prevent history corruption MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When Anthropic Claude returns stopReason: "toolUse" but the assistant message contains no actual toolCall, the session would append a spurious toolResult entry. This creates tool_result blocks without matching tool_use blocks — structurally invalid for Anthropic's API validator. The symptom is overloaded_error on every subsequent request (in: 0 out: 0), even though servers aren't overloaded. Other providers handle the same corrupted history more leniently, making the problem appear Anthropic-specific. The #isEmptyAssistantStop guard now checks for stopReason === "toolUse" with no text and no toolCall. When the retry cap is hit, tool-use orphans are still removed from active context (unlike regular empty stops) because they corrupt message history. A regression test covers both the retry and cap scenarios. For affected sessions: run /compact to drop the corrupted history tail. Fixes review feedback from chatgpt-codex-connector. --- .../coding-agent/src/session/agent-session.ts | 13 ++++- .../agent-session-empty-stop-guard.test.ts | 47 +++++++++++++++++++ 2 files changed, 59 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 84f78cdce..b29f817c3 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -6443,9 +6443,13 @@ export class AgentSession { this.#retryAttempt = 0; } this.#resolveRetry(); + // Tool-use orphans corrupt Anthropic message history (tool_result without + // matching tool_use). Always remove them even when the retry cap is hit. + if (assistantMessage.stopReason === "toolUse") { + this.#removeEmptyStopFromActiveContext(assistantMessage); + } return true; } - this.#removeEmptyStopFromActiveContext(assistantMessage); this.agent.appendMessage({ role: "developer", @@ -6458,6 +6462,13 @@ export class AgentSession { } #isEmptyAssistantStop(assistantMessage: AssistantMessage): boolean { + const hasText = assistantMessage.content.some( + content => content.type === "text" && content.text.trim().length > 0, + ); + const hasToolCall = assistantMessage.content.some(content => content.type === "toolCall"); + if (assistantMessage.stopReason === "toolUse") { + return !hasText && !hasToolCall; + } if (assistantMessage.stopReason !== "stop") return false; return !assistantMessage.content.some(content => { if (content.type === "text") return content.text.trim().length > 0; diff --git a/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts b/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts index 16cd9ecb1..eb9495ef0 100644 --- a/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts +++ b/packages/coding-agent/test/agent-session-empty-stop-guard.test.ts @@ -51,6 +51,14 @@ function emptyStop(): MockResponse { }; } +function orphanedToolUseStop(): MockResponse { + return { + content: [{ type: "thinking", thinking: "I should call a tool next." }], + stopReason: "toolUse", + usage: { output: 1, cacheRead: 100 }, + }; +} + async function createHarness( responses: MockResponse[], settingsOverrides: SettingsOverrides = {}, @@ -177,6 +185,45 @@ describe("AgentSession empty stop guard", () => { ).toHaveLength(1); }); + it("retries a tool-use stop that has no tool call or text", async () => { + const { session, mock } = await createHarness([ + recordCall("orphan", "call-record-orphan"), + orphanedToolUseStop(), + { content: ["finished after orphaned tool-use retry"], stopReason: "stop" }, + ]); + + await session.prompt("record orphan"); + await session.waitForIdle(); + + expect(mock.calls).toHaveLength(3); + expect(assistantText(session.agent.state.messages)).toContain("finished after orphaned tool-use retry"); + expect(reminderMessages(session.agent.state.messages)).toHaveLength(1); + }); + + it("removes orphaned tool-use stops even when retry cap is hit", async () => { + const { session, mock } = await createHarness([ + recordCall("gamma", "call-record-gamma"), + orphanedToolUseStop(), + orphanedToolUseStop(), + orphanedToolUseStop(), + orphanedToolUseStop(), + ]); + await session.prompt("record gamma"); + await session.waitForIdle(); + expect(mock.calls).toHaveLength(5); + expect(reminderMessages(session.agent.state.messages)).toHaveLength(3); + const activeBranchMessages = session.sessionManager + .getBranch() + .filter(entry => entry.type === "message") + .map(entry => entry.message as AgentMessage); + const orphanedToolUseStops = activeBranchMessages.filter( + message => + message.role === "assistant" && + message.stopReason === "toolUse" && + !message.content.some(content => content.type === "toolCall"), + ); + expect(orphanedToolUseStops).toHaveLength(0); + }); it("caps empty stop retries at three attempts", async () => { const { session, mock } = await createHarness([ recordCall("beta", "call-record-beta"), From e744ce28944cdd2a5da010c0792fe5fbc6137506 Mon Sep 17 00:00:00 2001 From: DarkPhilosophy <19309990+DarkPhilosophy@users.noreply.github.com> Date: Sat, 6 Jun 2026 02:22:36 +0300 Subject: [PATCH 002/181] perf(coding-agent): single-pass empty-stop content scan --- .../coding-agent/src/session/agent-session.ts | 30 +++++++++++-------- 1 file changed, 18 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index c52a65311..31775300d 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -6484,19 +6484,25 @@ export class AgentSession { } #isEmptyAssistantStop(assistantMessage: AssistantMessage): boolean { - const hasText = assistantMessage.content.some( - content => content.type === "text" && content.text.trim().length > 0, - ); - const hasToolCall = assistantMessage.content.some(content => content.type === "toolCall"); - if (assistantMessage.stopReason === "toolUse") { - return !hasText && !hasToolCall; + const { stopReason } = assistantMessage; + if (stopReason !== "stop" && stopReason !== "toolUse") return false; + + // Single pass over content; the three flags cover every emptiness rule below. + let hasText = false; + let hasThinking = false; + let hasToolCall = false; + for (const content of assistantMessage.content) { + if (content.type === "text") hasText ||= content.text.trim().length > 0; + else if (content.type === "thinking") hasThinking ||= content.thinking.trim().length > 0; + else if (content.type === "toolCall") hasToolCall = true; } - if (assistantMessage.stopReason !== "stop") return false; - return !assistantMessage.content.some(content => { - if (content.type === "text") return content.text.trim().length > 0; - if (content.type === "thinking") return content.thinking.trim().length > 0; - return content.type === "toolCall"; - }); + + // An orphaned toolUse stop (no tool_use block) corrupts Anthropic history: + // a later tool_result has nothing to anchor to. Thinking alone cannot anchor + // a tool_result, so it does not rescue a toolUse stop here. + if (stopReason === "toolUse") return !hasText && !hasToolCall; + // A plain stop is empty only when it carries no usable content at all. + return !hasText && !hasThinking && !hasToolCall; } #emptyStopRetryReminder(): string { From 489b83d1ae7e78ca6f0d8156302ae741c04ee058 Mon Sep 17 00:00:00 2001 From: tycronk Date: Sat, 6 Jun 2026 03:15:27 -0400 Subject: [PATCH 003/181] docs: document Pi extension export drift in porting guide MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Section 3: add the @earendil-works/* scope alias and @mariozechner/pi-utils to the import-scope replacements, and note that bare typebox is not an @oh-my-pi/* scope. Section 15 (Extensions divergence): record the dropped/renamed exports that break a straight scope-rename port — StringEnum (pi-ai), formatSize (pi-coding-agent), and the removed DefaultResourceLoader/DefaultPackageManager/ SettingsManager/createEventBus discovery classes — with their OMP replacements. --- docs/porting-from-pi-mono.md | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/docs/porting-from-pi-mono.md b/docs/porting-from-pi-mono.md index 3bfb97bc1..23e4dfcd1 100644 --- a/docs/porting-from-pi-mono.md +++ b/docs/porting-from-pi-mono.md @@ -47,6 +47,9 @@ Upstream uses different package scopes. Replace them consistently. - `@mariozechner/pi-agent-core` → `@oh-my-pi/pi-agent-core` - `@mariozechner/pi-tui` → `@oh-my-pi/pi-tui` - `@mariozechner/pi-ai` → `@oh-my-pi/pi-ai` + - `@mariozechner/pi-utils` → `@oh-my-pi/pi-utils` +- Some upstream packages publish under the `@earendil-works/*` scope instead of `@mariozechner/*`. Map it the same way (`@earendil-works/pi-coding-agent` → `@oh-my-pi/pi-coding-agent`, and so on). +- The bare `typebox` package is not an `@oh-my-pi/*` scope; do not rewrite it as one. See the Extensions divergence in section 15 for how tool-parameter schemas map. ## 4) Use Bun APIs where they improve on Node @@ -353,10 +356,13 @@ Our fork has architectural decisions that differ from upstream. **Do not port th ### Extensions -| Upstream | Our Fork | -| ----------------------------- | ------------------------------------------------- | -| `jiti` for TypeScript loading | Native Bun `import()` | -| `pkg.pi` manifest field | `pkg.omp` preferred; fallback to `pkg.pi` remains | +| Upstream | Our Fork | +| ---------------------------------------------------------------- | ------------------------------------------------------------------------------------------------- | +| `jiti` for TypeScript loading | Native Bun `import()` | +| `pkg.pi` manifest field | `pkg.omp` preferred; fallback to `pkg.pi` remains | +| `StringEnum` from `pi-ai` | `Type.Enum` from the `pi.typebox` shim (or author the schema with `pi.zod`); `pi-ai` no longer exports `StringEnum` | +| `formatSize` from `pi-coding-agent` | `formatBytes` from `@oh-my-pi/pi-utils` | +| `DefaultResourceLoader` / `DefaultPackageManager` / `SettingsManager` / `createEventBus` | Capability-based discovery (`loadCapability(...)`) plus the `Settings` singleton and `EventBus` | ### Skip These Upstream Features From 236bba96b996b6d11575fbe121e3e78279c4ec15 Mon Sep 17 00:00:00 2001 From: Asaf Mahlev Date: Sat, 6 Jun 2026 11:56:23 +0300 Subject: [PATCH 004/181] fix(eval): floor JS worker init timeout to stop terminate-mid-init flake Worker-ready wait reused Bun's 5s default per-test timeout as its floor, so a slow cold-start under --isolate + high CI concurrency was aborted at 5s. The catch then terminate()s a still-initializing Bun worker -- the documented SIGILL/SIGTRAP crash trigger -- crashing the whole test file and intermittently failing unrelated PRs. Introduce WORKER_INIT_TIMEOUT_MS=15s as a fixed infrastructure floor (independent of, still dominated by, a larger per-cell timeout) and set a 20s file-local setDefaultTimeout in js-executor/js-workflow-helpers tests so cold starts complete instead of being torn down. --- packages/coding-agent/CHANGELOG.md | 3 +++ .../coding-agent/src/eval/js/context-manager.ts | 13 +++++++++---- packages/coding-agent/test/core/js-executor.test.ts | 7 ++++++- .../test/core/js-workflow-helpers.test.ts | 7 ++++++- 4 files changed, 24 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e14f2353f..da2371269 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] +### Fixed + +- Fixed a flaky JS eval worker startup that intermittently failed unrelated CI runs. The worker-ready wait reused Bun's 5s default per-test timeout as its floor, so a slow cold-start under `--isolate` + high concurrency was aborted mid-init; terminating a still-initializing Bun worker is the documented SIGILL/SIGTRAP crash trigger, which took down the whole test file. Worker init now floors at a fixed 15s infrastructure budget (independent of, and still dominated by, a larger per-cell `timeout`), and the JS eval test suites set a 20s file-local timeout so cold starts complete instead of being torn down. ## [15.9.5] - 2026-06-05 ### Added diff --git a/packages/coding-agent/src/eval/js/context-manager.ts b/packages/coding-agent/src/eval/js/context-manager.ts index cea0c97e5..8e7da951a 100644 --- a/packages/coding-agent/src/eval/js/context-manager.ts +++ b/packages/coding-agent/src/eval/js/context-manager.ts @@ -53,7 +53,12 @@ interface JsSession { const sessions = new Map(); const startingSessions = new Map>(); const resettingSessions = new Set(); -const READY_TIMEOUT_MS_DEFAULT = 5_000; +// Worker startup (module-graph import + WorkerCore construction) is infrastructure +// cost, not user compute. Floor it independently of Bun's 5s default per-test timeout +// so a slow cold-start under load isn't aborted mid-init — terminating a still- +// initializing Bun worker is the documented SIGILL/SIGTRAP crash trigger (see +// shared/indirect-eval.ts). Callers that pass a larger per-cell budget still dominate. +const WORKER_INIT_TIMEOUT_MS = 15_000; export async function executeInVmContext(options: { sessionKey: string; @@ -191,9 +196,9 @@ async function acquireSession(sessionKey: string, snapshot: SessionSnapshot, tim handleSessionMessage(session, msg); }); try { - // Cold-start can exceed 5s on slow hosts. Let the caller's per-cell timeout dominate so - // users can grant more headroom when they raise `timeout` on a cell. - const readyTimeoutMs = Math.max(READY_TIMEOUT_MS_DEFAULT, timeoutMs ?? 0); + // Init headroom is the fixed infrastructure floor; the caller's per-cell timeout + // dominates when larger so users can grant more by raising `timeout` on a cell. + const readyTimeoutMs = Math.max(WORKER_INIT_TIMEOUT_MS, timeoutMs ?? 0); await raceWithTimeout(readyPromise, readyTimeoutMs, "Timed out initializing JS eval worker"); worker.send({ type: "init", snapshot }); sessions.set(sessionKey, session); diff --git a/packages/coding-agent/test/core/js-executor.test.ts b/packages/coding-agent/test/core/js-executor.test.ts index 3d33d369d..7776bfbb2 100644 --- a/packages/coding-agent/test/core/js-executor.test.ts +++ b/packages/coding-agent/test/core/js-executor.test.ts @@ -1,4 +1,4 @@ -import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, describe, expect, it, setDefaultTimeout, vi } from "bun:test"; import * as path from "node:path"; import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; @@ -8,6 +8,11 @@ import * as z from "zod/v4"; import { disposeAllVmContexts } from "../../src/eval/js/context-manager"; import { executeJs, type JsResult } from "../../src/eval/js/executor"; +// JS eval cold-starts a Bun worker; under --isolate + high CI concurrency that startup +// can exceed Bun's 5s default per-test timeout, flaking the suite. Give the worker-backed +// tests headroom above the worker-init floor (context-manager WORKER_INIT_TIMEOUT_MS). +setDefaultTimeout(20_000); + function createTool( name: string, execute: (toolCallId: string, args: unknown, signal?: AbortSignal) => Promise, diff --git a/packages/coding-agent/test/core/js-workflow-helpers.test.ts b/packages/coding-agent/test/core/js-workflow-helpers.test.ts index 8e005bab8..d8592165a 100644 --- a/packages/coding-agent/test/core/js-workflow-helpers.test.ts +++ b/packages/coding-agent/test/core/js-workflow-helpers.test.ts @@ -1,4 +1,4 @@ -import { afterAll, beforeAll, describe, expect, it } from "bun:test"; +import { afterAll, beforeAll, describe, expect, it, setDefaultTimeout } from "bun:test"; import * as path from "node:path"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; @@ -6,6 +6,11 @@ import { TempDir } from "@oh-my-pi/pi-utils"; import { disposeAllVmContexts } from "../../src/eval/js/context-manager"; import { executeJs, type JsResult } from "../../src/eval/js/executor"; +// JS eval cold-starts a Bun worker; under --isolate + high CI concurrency that startup +// can exceed Bun's 5s default per-test timeout, flaking the suite. Give the worker-backed +// tests headroom above the worker-init floor (context-manager WORKER_INIT_TIMEOUT_MS). +setDefaultTimeout(20_000); + function statusEvents(result: JsResult) { return result.displayOutputs.filter( (output): output is Extract => output.type === "status", From ea8d58d3c696cf48afca46719e886ff8abfda7a2 Mon Sep 17 00:00:00 2001 From: Asaf Mahlev Date: Sat, 6 Jun 2026 12:43:53 +0300 Subject: [PATCH 005/181] docs(eval): clarify indirect-eval cross-reference in worker-init comment Address review: shared/indirect-eval.ts documents the vm.runInContext mid-execution terminate-race, not a mid-init one. Reword so a reader doesn't grep that file for an init-specific note. --- packages/coding-agent/src/eval/js/context-manager.ts | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/src/eval/js/context-manager.ts b/packages/coding-agent/src/eval/js/context-manager.ts index 8e7da951a..c1dcef642 100644 --- a/packages/coding-agent/src/eval/js/context-manager.ts +++ b/packages/coding-agent/src/eval/js/context-manager.ts @@ -56,8 +56,9 @@ const resettingSessions = new Set(); // Worker startup (module-graph import + WorkerCore construction) is infrastructure // cost, not user compute. Floor it independently of Bun's 5s default per-test timeout // so a slow cold-start under load isn't aborted mid-init — terminating a still- -// initializing Bun worker is the documented SIGILL/SIGTRAP crash trigger (see -// shared/indirect-eval.ts). Callers that pass a larger per-cell budget still dominate. +// initializing Bun worker triggers the same kind of terminate-race that motivates +// avoiding `vm.runInContext` (see shared/indirect-eval.ts), here surfacing as a +// SIGILL/SIGSEGV. Callers that pass a larger per-cell budget still dominate. const WORKER_INIT_TIMEOUT_MS = 15_000; export async function executeInVmContext(options: { From 06e157cc242af128b50cbdeb886ad88bdfcfb959 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 17:58:28 +0200 Subject: [PATCH 006/181] ux(coding-agent/edit): inlined edit result stats into the header - Updated edit result rendering to inline diff change statistics in the file header instead of using a separate metadata row. - Removed the redundant standalone metadata line and removed the extra blank line before diff bodies for a tighter single-hunk display. - Added a test asserting the header now contains +/-/hunk stats and that no extra stats row appears before the diff. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/edit/renderer.ts | 44 ++++++------------- .../test/extensions-runner.test.ts | 1 - .../test/tools/edit-renderer.test.ts | 25 +++++++++++ 4 files changed, 40 insertions(+), 31 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6515a5a39..821b00241 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -9,6 +9,7 @@ ### Changed - Changed the default persistent model selector shortcut from `Ctrl+L` to `Alt+M`, leaving `Ctrl+L` for display reset. Existing user remaps in `keybindings.yml` are preserved. +- Changed the edit tool result header to carry the diff change stats (`+N / -M / K hunks`) inline next to the file path, and removed the redundant lone language-icon metadata row and the blank line between the header and the diff body, so a single-hunk edit renders as `✔ Edit: path:LINE ⟨+3 / 1 hunk⟩` immediately followed by the diff. ### Fixed diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index 51f35095d..4a51f9740 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -179,11 +179,6 @@ function countEditFiles(edits: EditRenderEntry[]): number { return new Set(edits.map(edit => filePathFromEditEntry(edit.path)).filter(Boolean)).size; } -function countLines(text: string): number { - if (!text) return 0; - return text.split("\n").length; -} - function getOperationTitle(op: Operation | undefined): string { return op === "create" ? "Create" : op === "delete" ? "Delete" : "Edit"; } @@ -263,14 +258,6 @@ function formatStreamingDiff( return text; } -function formatMetadataLine(lineCount: number | null, language: string | undefined, uiTheme: Theme): string { - const icon = uiTheme.getLangIcon(language); - if (lineCount !== null) { - return uiTheme.fg("dim", `${icon} ${lineCount} lines`); - } - return uiTheme.fg("dim", `${icon}`); -} - function formatMultiFileStreamingDiff(previews: PerFileDiffPreview[], uiTheme: Theme, expanded: boolean): string { const parts: string[] = []; for (const preview of previews) { @@ -387,6 +374,13 @@ function getApplyPatchRenderSummary( } } +function formatDiffStatsSuffix(diff: string, uiTheme: Theme): string { + const { added, removed, hunks } = getDiffStats(diff); + const stats = formatDiffStats(added, removed, hunks, uiTheme); + if (!stats) return ""; + return ` ${uiTheme.fg("dim", uiTheme.format.bracketLeft)}${stats}${uiTheme.fg("dim", uiTheme.format.bracketRight)}`; +} + function renderDiffSection( diff: string, rawPath: string, @@ -394,15 +388,6 @@ function renderDiffSection( uiTheme: Theme, renderDiffFn: (t: string, o?: { filePath?: string }) => string, ): string { - let text = ""; - const diffStats = getDiffStats(diff); - text += `\n${uiTheme.fg("dim", uiTheme.format.bracketLeft)}${formatDiffStats( - diffStats.added, - diffStats.removed, - diffStats.hunks, - uiTheme, - )}${uiTheme.fg("dim", uiTheme.format.bracketRight)}`; - const { text: truncatedDiff, hiddenHunks, @@ -411,7 +396,7 @@ function renderDiffSection( ? { text: diff, hiddenHunks: 0, hiddenLines: 0 } : truncateDiffByHunk(diff, PREVIEW_LIMITS.DIFF_COLLAPSED_HUNKS, PREVIEW_LIMITS.DIFF_COLLAPSED_LINES); - text += `\n\n${renderDiffFn(truncatedDiff, { filePath: rawPath })}`; + let text = `\n${renderDiffFn(truncatedDiff, { filePath: rawPath })}`; if (!expanded && (hiddenHunks > 0 || hiddenLines > 0)) { const remainder: string[] = []; if (hiddenHunks > 0) remainder.push(`${hiddenHunks} more hunks`); @@ -532,11 +517,6 @@ function renderSingleFileResult( ""; const op = args?.op || firstEdit?.op || details?.op; const rename = args?.rename || firstEdit?.rename || firstEdit?.move || details?.move; - const { language } = formatEditDescription(rawPath, uiTheme, { rename }); - - const editTextSource = args?.newText ?? args?.oldText ?? args?.diff ?? args?.patch; - const metadataLineCount = editTextSource ? countLines(editTextSource) : null; - const metadataLine = op !== "delete" ? `\n${formatMetadataLine(metadataLineCount, language, uiTheme)}` : ""; const displayErrorText = isError && details && "displayErrorText" in details ? details.displayErrorText : undefined; const errorText = isError @@ -560,6 +540,11 @@ function renderSingleFileResult( (details && !isError ? details.firstChangedLine : undefined); const { description } = formatEditDescription(rawPath, uiTheme, { rename, firstChangedLine }); + // Change stats ride inline on the header next to the path rather than a separate row. + const previewDiff = editDiffPreview && !("error" in editDiffPreview) ? editDiffPreview.diff : undefined; + const headerDiff = isError ? undefined : details?.diff || previewDiff; + const statsSuffix = headerDiff ? formatDiffStatsSuffix(headerDiff, uiTheme) : ""; + const header = renderStatusLine( { icon: isError ? "error" : "success", @@ -568,8 +553,7 @@ function renderSingleFileResult( }, uiTheme, ); - let text = header; - text += metadataLine; + let text = header + statsSuffix; if (isError) { if (errorText) { diff --git a/packages/coding-agent/test/extensions-runner.test.ts b/packages/coding-agent/test/extensions-runner.test.ts index 1542897a9..33e188b7f 100644 --- a/packages/coding-agent/test/extensions-runner.test.ts +++ b/packages/coding-agent/test/extensions-runner.test.ts @@ -148,7 +148,6 @@ describe("ExtensionRunner", () => { warnSpy.mockRestore(); }); - it("warns when two extensions register same shortcut", async () => { // Use a non-reserved shortcut const extCode1 = ` diff --git a/packages/coding-agent/test/tools/edit-renderer.test.ts b/packages/coding-agent/test/tools/edit-renderer.test.ts index 57f9aef63..20370c582 100644 --- a/packages/coding-agent/test/tools/edit-renderer.test.ts +++ b/packages/coding-agent/test/tools/edit-renderer.test.ts @@ -304,4 +304,29 @@ describe("editToolRenderer", () => { const rendered = Bun.stripANSI(component.render(160).join("\n")); expect(rendered).toContain("plain streamed text"); }); + + it("renders change stats inline on the result header with no separate metadata or stats row", async () => { + const uiTheme = await getUiTheme(); + const diff = [" 115│ ctx", "-116│ old", "+117│ new one", "+118│ new two"].join("\n"); + const component = editToolRenderer.renderResult( + { + content: [{ type: "text", text: "Updated demo.go" }], + details: { diff, op: "update" }, + }, + { expanded: false, isPartial: false, renderContext: { editMode: "hashline" } }, + uiTheme, + { file_path: "demo.go" }, + ); + + const lines = Bun.stripANSI(component.render(160).join("\n")).split("\n"); + // Stats ride on the header line next to the path… + expect(lines[0]).toContain("demo.go"); + expect(lines[0]).toContain("+2"); + expect(lines[0]).toContain("-1"); + expect(lines[0]).toContain("1 hunk"); + // …only there (no standalone stats row), and the diff starts immediately + // below the header (no blank line, no lone lang-icon metadata row). + expect(lines[1]).toContain("115│ ctx"); + expect(lines.filter(line => line.includes("hunk"))).toHaveLength(1); + }); }); From c49d5c99b1c3c46369a219e2fa92cc88756cfa32 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 18:21:09 +0200 Subject: [PATCH 007/181] fix(coding-agent): removed preview line capping on context lines - Rendered full context lines instead of truncating via capPreviewLines. --- packages/coding-agent/src/task/render.ts | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/packages/coding-agent/src/task/render.ts b/packages/coding-agent/src/task/render.ts index 1522511bf..f6cc1fc0e 100644 --- a/packages/coding-agent/src/task/render.ts +++ b/packages/coding-agent/src/task/render.ts @@ -13,7 +13,6 @@ import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { formatContextUsage } from "../modes/components/status-line/context-thresholds"; import type { Theme } from "../modes/theme/theme"; import { - capPreviewLines, formatBadge, formatDuration, formatMoreItems, @@ -566,7 +565,7 @@ export function renderCall( const content = line ? theme.fg("muted", replaceTabs(line)) : ""; return ` ${vertical} ${content}`; }); - lines.push(...capPreviewLines(contextLines, theme, { expanded: options.expanded, prefix: ` ${vertical} ` })); + lines.push(...contextLines); } // `Tasks` is the last child unless the isolation flag follows it. From 75e211415e8e964eafbc20d199be9b982eedfd1e Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 18:28:19 +0200 Subject: [PATCH 008/181] feat(cli): added gallery CLI command for renderer previews and filtering options - Added lazy-loaded `gallery` command registration and new filters for tool, state, width, expanded, and plain output. - Implemented gallery state rendering with terminal-width defaults, state filtering, and unknown-tool fallback handling. - Added shared fixture types and aggregated renderer fixtures for multiple tool families in `galleryFixtures`. - Added tests for renderer state coverage, route-specific output (streaming/progress/success/error), and fixture fallback. --- packages/coding-agent/CHANGELOG.md | 3 +- packages/coding-agent/src/cli-commands.ts | 1 + packages/coding-agent/src/cli/gallery-cli.ts | 165 ++++++++++ .../src/cli/gallery-fixtures/agentic.ts | 291 ++++++++++++++++++ .../src/cli/gallery-fixtures/codeintel.ts | 187 +++++++++++ .../src/cli/gallery-fixtures/edit.ts | 194 ++++++++++++ .../src/cli/gallery-fixtures/fs.ts | 153 +++++++++ .../src/cli/gallery-fixtures/index.ts | 40 +++ .../src/cli/gallery-fixtures/interaction.ts | 49 +++ .../src/cli/gallery-fixtures/memory.ts | 81 +++++ .../src/cli/gallery-fixtures/misc.ts | 221 +++++++++++++ .../src/cli/gallery-fixtures/search.ts | 213 +++++++++++++ .../src/cli/gallery-fixtures/shell.ts | 167 ++++++++++ .../src/cli/gallery-fixtures/types.ts | 32 ++ .../src/cli/gallery-fixtures/web.ts | 158 ++++++++++ packages/coding-agent/src/commands/gallery.ts | 37 +++ 16 files changed, 1991 insertions(+), 1 deletion(-) create mode 100644 packages/coding-agent/src/cli/gallery-cli.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/agentic.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/edit.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/fs.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/index.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/interaction.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/memory.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/misc.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/search.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/shell.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/types.ts create mode 100644 packages/coding-agent/src/cli/gallery-fixtures/web.ts create mode 100644 packages/coding-agent/src/commands/gallery.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 821b00241..b19bd4917 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,9 +1,10 @@ # Changelog ## [Unreleased] - ### Added +- Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states +- Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output - Added `app.display.reset`, bound to `Ctrl+L` by default, to force an immediate terminal display reset/redraw without resizing the window. ### Changed diff --git a/packages/coding-agent/src/cli-commands.ts b/packages/coding-agent/src/cli-commands.ts index c9ac00741..efa68d4fa 100644 --- a/packages/coding-agent/src/cli-commands.ts +++ b/packages/coding-agent/src/cli-commands.ts @@ -22,6 +22,7 @@ export const commands: CommandEntry[] = [ { name: "config", load: () => import("./commands/config").then(m => m.default) }, { name: "dry-balance", load: () => import("./commands/dry-balance").then(m => m.default) }, { name: "grep", load: () => import("./commands/grep").then(m => m.default) }, + { name: "gallery", load: () => import("./commands/gallery").then(m => m.default) }, { name: "grievances", load: () => import("./commands/grievances").then(m => m.default) }, { name: "install", load: () => import("./commands/install").then(m => m.default) }, { name: "plugin", load: () => import("./commands/plugin").then(m => m.default) }, diff --git a/packages/coding-agent/src/cli/gallery-cli.ts b/packages/coding-agent/src/cli/gallery-cli.ts new file mode 100644 index 000000000..744691a15 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-cli.ts @@ -0,0 +1,165 @@ +/** + * `omp gallery` — render every built-in tool's renderer across its lifecycle. + * + * For each tool with a registered renderer, the gallery drives a real + * {@link ToolExecutionComponent} through four states — streaming arguments, + * arguments complete (in progress), success, and failure — and prints the + * rendered output to stdout. It exists for visual QA of tool renderers without + * having to provoke each state through a live agent session. + */ +import type { AgentTool } from "@oh-my-pi/pi-agent-core"; +import type { TUI } from "@oh-my-pi/pi-tui"; +import { getProjectDir } from "@oh-my-pi/pi-utils"; +import { Settings } from "../config/settings"; +import { ToolExecutionComponent } from "../modes/components/tool-execution"; +import { initTheme, theme } from "../modes/theme/theme"; +import { toolRenderers } from "../tools/renderers"; +import { type GalleryFixture, type GalleryResult, galleryFixtures } from "./gallery-fixtures"; + +/** Lifecycle states the gallery renders, in display order. */ +export const GALLERY_STATES = ["streaming", "progress", "success", "error"] as const; +export type GalleryState = (typeof GALLERY_STATES)[number]; + +const STATE_LABELS: Record = { + streaming: "streaming args", + progress: "in progress", + success: "done", + error: "failed", +}; + +export interface GalleryCommandArgs { + /** Render width in columns (defaults to terminal width, clamped). */ + width?: number; + /** Restrict to a single tool name. */ + tool?: string; + /** Restrict to specific lifecycle states. */ + states?: GalleryState[]; + /** Render the expanded variant of each renderer. */ + expanded?: boolean; + /** Strip ANSI styling from the output (useful when redirecting to a file). */ + plain?: boolean; +} + +const GENERIC_ERROR: GalleryResult = { + content: [{ type: "text", text: "Error: operation failed" }], + isError: true, +}; + +/** Build the fake `AgentTool` the component needs for its label and edit mode. */ +function fakeToolFor(name: string, fixture: GalleryFixture | undefined): AgentTool | undefined { + if (!fixture?.label && !fixture?.editMode) return undefined; + return { name, label: fixture.label ?? name, mode: fixture.editMode } as unknown as AgentTool; +} + +/** The curated fixture for a tool, or a generic one for registry tools lacking sample data. */ +export function resolveFixture(name: string): GalleryFixture { + return ( + galleryFixtures[name] ?? + ({ + args: { note: `sample ${name} call` }, + result: { content: [{ type: "text", text: `${name} completed` }] }, + } satisfies GalleryFixture) + ); +} + +/** + * Render a single tool/state pair to lines. Builds a fresh component, drives it + * to the requested state, settles any async edit preview, then snapshots the + * render and stops all animation timers. + */ +export async function renderGalleryState( + name: string, + fixture: GalleryFixture, + state: GalleryState, + width: number, + expanded = false, +): Promise { + const tool = fakeToolFor(name, fixture); + const streamingArgs = state === "streaming" ? (fixture.streamingArgs ?? fixture.args) : fixture.args; + // The component only calls `requestRender` during a static render; + // `imageBudget` is consulted solely when images render, which the gallery + // disables. A cast avoids constructing a real terminal. + const ui = { requestRender() {} } as unknown as TUI; + const component = new ToolExecutionComponent(name, streamingArgs, { showImages: false }, tool, ui, getProjectDir()); + component.setExpanded(expanded); + + if (state !== "streaming") { + component.setArgsComplete(); + } + if (state === "success") { + component.updateResult(fixture.result, false); + } else if (state === "error") { + component.updateResult(fixture.errorResult ?? GENERIC_ERROR, false); + } + + // Edit-like renderers compute their diff preview off the render path; wait + // for it to settle so the snapshot is deterministic instead of racing a tick. + await component.whenPreviewSettled(); + + const lines = component.render(width); + component.stopAnimation(); + return lines; +} + +function resolveWidth(requested: number | undefined): number { + const fallback = process.stdout.columns ?? 100; + const width = requested ?? fallback; + return Math.max(40, Math.min(200, width)); +} + +function sectionRule(label: string, width: number): string { + const prefix = `── ${label} `; + const fill = Math.max(0, width - prefix.length); + return theme.fg("accent", theme.bold(`${prefix}${"─".repeat(fill)}`)); +} + +/** + * Render the gallery to stdout. Iterates the renderer registry (or a single + * tool), printing each requested lifecycle state under a labeled section. + */ +export async function runGalleryCommand(args: GalleryCommandArgs): Promise { + const settingsInstance = await Settings.init(); + await initTheme( + false, + settingsInstance.get("symbolPreset"), + settingsInstance.get("colorBlindMode"), + settingsInstance.get("theme.dark"), + settingsInstance.get("theme.light"), + ); + + const width = resolveWidth(args.width); + const expanded = args.expanded ?? false; + const states = args.states && args.states.length > 0 ? args.states : [...GALLERY_STATES]; + + const allNames = Object.keys(toolRenderers).sort(); + const names = args.tool ? allNames.filter(name => name === args.tool) : allNames; + if (args.tool && names.length === 0) { + process.stdout.write(`Unknown tool '${args.tool}'. Known tools: ${allNames.join(", ")}\n`); + return; + } + + const out: string[] = []; + const push = (line: string) => out.push(args.plain ? Bun.stripANSI(line) : line); + + for (const name of names) { + const fixture = resolveFixture(name); + const heading = fixture.label && fixture.label !== name ? `${name} — ${fixture.label}` : name; + push(""); + push(sectionRule(heading, width)); + + for (const state of states) { + push(""); + push(theme.fg("dim", ` · ${STATE_LABELS[state]}`)); + let lines: string[]; + try { + lines = await renderGalleryState(name, fixture, state, width, expanded); + } catch (err) { + lines = [theme.fg("error", ` render failed: ${String(err)}`)]; + } + for (const line of lines) push(line); + } + } + push(""); + + process.stdout.write(`${out.join("\n")}\n`); +} diff --git a/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts new file mode 100644 index 000000000..f2dd796ef --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts @@ -0,0 +1,291 @@ +// Gallery fixtures for the agentic orchestration tools (task, goal, job). +import type { GalleryFixture } from "./types"; + +export const agenticFixtures: Record = { + task: { + label: "Task", + // Streaming: agent chosen, first task fully arrived, second still landing. + streamingArgs: { + agent: "task", + tasks: [ + { + id: "AuthLoader", + description: "Load auth middleware", + assignment: "Read packages/server/src/auth/*.ts and summarize the session-cookie flow.", + }, + { id: "RateLimiter", description: "Audit rate limiter" }, + ], + }, + args: { + agent: "task", + context: [ + "# Goal", + "Harden the HTTP auth stack before the release cut.", + "# Constraints", + "Touch only files under packages/server/src/auth/. Do not run gates.", + ].join("\n"), + tasks: [ + { + id: "AuthLoader", + description: "Load auth middleware", + assignment: + "Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.", + }, + { + id: "RateLimiter", + description: "Audit rate limiter", + assignment: + "Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.", + }, + { + id: "TokenRotation", + description: "Check token rotation", + assignment: + "Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.", + }, + ], + }, + result: { + content: [ + { + type: "text", + text: "3 agents completed: AuthLoader, RateLimiter, TokenRotation.", + }, + ], + details: { + projectAgentsDir: null, + totalDurationMs: 48_200, + usage: { cost: { total: 0.34 } }, + results: [ + { + index: 0, + id: "AuthLoader", + agent: "task", + agentSource: "bundled", + description: "Load auth middleware", + task: "Read packages/server/src/auth/session.ts and middleware.ts", + assignment: + "Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.", + exitCode: 0, + output: [ + "Session validation runs in middleware.ts:42 via verifySessionCookie().", + "Cookies are HMAC-signed (SHA-256) and checked against the session store.", + "TODO at session.ts:88 — sliding-expiration refresh is stubbed.", + ].join("\n"), + stderr: "", + truncated: false, + durationMs: 41_900, + tokens: 61_400, + contextTokens: 23_100, + contextWindow: 200_000, + resolvedModel: "anthropic/claude-sonnet", + usage: { cost: { total: 0.12 } }, + outputMeta: { lineCount: 3, charCount: 214 }, + }, + { + index: 1, + id: "RateLimiter", + agent: "task", + agentSource: "bundled", + description: "Audit rate limiter", + task: "Inspect packages/server/src/auth/rate-limit.ts", + assignment: + "Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.", + exitCode: 0, + output: [ + "rate-limit.ts uses a fixed-window counter keyed by client IP.", + "429 responses set Retry-After (rate-limit.ts:57).", + "Gap: no per-account limit, so a botnet across IPs bypasses the cap.", + ].join("\n"), + stderr: "", + truncated: false, + durationMs: 38_500, + tokens: 54_800, + contextTokens: 19_700, + contextWindow: 200_000, + resolvedModel: "anthropic/claude-sonnet", + usage: { cost: { total: 0.1 } }, + outputMeta: { lineCount: 3, charCount: 198 }, + }, + { + index: 2, + id: "TokenRotation", + agent: "task", + agentSource: "bundled", + description: "Check token rotation", + task: "Trace refresh-token rotation in packages/server/src/auth/tokens.ts", + assignment: + "Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.", + exitCode: 0, + output: [ + "Refresh tokens rotate on every use (tokens.ts:120) and the old jti is revoked.", + "Reuse of a rotated token triggers full-family revocation — no reuse window found.", + ].join("\n"), + stderr: "", + truncated: false, + durationMs: 48_200, + tokens: 49_200, + contextTokens: 17_500, + contextWindow: 200_000, + resolvedModel: "anthropic/claude-sonnet", + usage: { cost: { total: 0.12 } }, + outputMeta: { lineCount: 2, charCount: 160 }, + }, + ], + }, + }, + errorResult: { + isError: true, + content: [ + { + type: "text", + text: "1 of 3 agents failed: RateLimiter.", + }, + ], + details: { + projectAgentsDir: null, + totalDurationMs: 39_400, + usage: { cost: { total: 0.21 } }, + results: [ + { + index: 0, + id: "AuthLoader", + agent: "task", + agentSource: "bundled", + description: "Load auth middleware", + task: "Read packages/server/src/auth/session.ts and middleware.ts", + assignment: + "Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.", + exitCode: 0, + output: "Session validation runs in middleware.ts:42 via verifySessionCookie().", + stderr: "", + truncated: false, + durationMs: 31_200, + tokens: 58_100, + contextTokens: 21_900, + contextWindow: 200_000, + resolvedModel: "anthropic/claude-sonnet", + usage: { cost: { total: 0.11 } }, + outputMeta: { lineCount: 1, charCount: 70 }, + }, + { + index: 1, + id: "RateLimiter", + agent: "task", + agentSource: "bundled", + description: "Audit rate limiter", + task: "Inspect packages/server/src/auth/rate-limit.ts", + assignment: + "Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.", + exitCode: 1, + output: "", + stderr: "ENOENT: packages/server/src/auth/rate-limit.ts", + truncated: false, + durationMs: 9_800, + tokens: 12_300, + contextTokens: 6_400, + contextWindow: 200_000, + resolvedModel: "anthropic/claude-sonnet", + usage: { cost: { total: 0.1 } }, + error: "Subagent exited 1: target file packages/server/src/auth/rate-limit.ts does not exist.", + outputMeta: { lineCount: 0, charCount: 0 }, + }, + ], + }, + }, + }, + + goal: { + label: "Goal", + // Streaming: op is "create"; objective text still being typed. + streamingArgs: { op: "create", objective: "Ship the auth hardening" }, + args: { + op: "create", + objective: "Ship the auth hardening pass: per-account rate limits and sliding session expiry.", + token_budget: 500_000, + }, + result: { + content: [ + { + type: "text", + text: "Goal set. Working toward: Ship the auth hardening pass.", + }, + ], + details: { + op: "create", + remainingTokens: 451_800, + completionBudgetReport: null, + goal: { + id: "goal_8f2a", + objective: "Ship the auth hardening pass: per-account rate limits and sliding session expiry.", + status: "active", + tokenBudget: 500_000, + tokensUsed: 48_200, + timeUsedSeconds: 312, + createdAt: 1_749_200_000_000, + updatedAt: 1_749_200_312_000, + }, + }, + }, + errorResult: { + isError: true, + content: [{ type: "text", text: "Goal tool failed: objective is required when op=create." }], + details: { op: "create" }, + }, + }, + + job: { + label: "Job", + // Streaming: polling a single job id; the second id is still arriving. + streamingArgs: { poll: ["job_a1"] }, + args: { poll: ["job_a1", "job_b2", "job_c3"] }, + result: { + content: [{ type: "text", text: "3 jobs settled." }], + details: { + jobs: [ + { + id: "job_a1", + type: "bash", + status: "completed", + label: "bun test packages/server/test/auth.test.ts", + durationMs: 18_400, + resultText: "42 pass, 0 fail (18.4s)", + }, + { + id: "job_b2", + type: "task", + status: "completed", + label: "Migrate rate limiter to a sliding window", + durationMs: 96_700, + resultText: "Rewrote rate-limit.ts to a token-bucket; added per-account keys.", + }, + { + id: "job_c3", + type: "bash", + status: "failed", + label: "bunx biome check packages/server/src/auth", + durationMs: 4_100, + errorText: "biome: 2 errors in tokens.ts — noUnusedVariables, useConst", + }, + ], + }, + }, + errorResult: { + isError: true, + content: [{ type: "text", text: "Job cancelled by user." }], + details: { + jobs: [ + { + id: "job_d4", + type: "task", + status: "cancelled", + label: "Refactor the session store to Redis", + durationMs: 52_300, + errorText: "Aborted: superseded by goal re-scope.", + }, + ], + cancelled: [{ id: "job_d4", status: "cancelled" }], + }, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts b/packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts new file mode 100644 index 000000000..5f10a9e2d --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts @@ -0,0 +1,187 @@ +/** Gallery fixtures for the code-intelligence tools (lsp, debug). */ +import type { GalleryFixture } from "./types"; + +export const codeintelFixtures: Record = { + lsp: { + label: "LSP", + streamingArgs: { + action: "references", + file: "src/server/auth.ts", + }, + args: { + action: "references", + file: "src/server/auth.ts", + line: 42, + symbol: "validateToken", + }, + result: { + content: [ + { + type: "text", + text: [ + "Found 6 reference(s):", + " src/server/auth.ts:42:14", + " 41: ", + " 42: export function validateToken(token: string): Claims {", + " 43: const claims = verifyJwt(token);", + " src/server/auth.ts:118:21", + ' 117: if (!header) throw new HttpError(401, "missing token");', + " 118: const claims = validateToken(stripBearer(header));", + " 119: return claims.sub;", + " src/server/middleware/session.ts:57:18", + " 56: const token = req.cookies.session;", + " 57: const claims = validateToken(token);", + " 58: req.userId = claims.sub;", + " src/server/router.ts:153:20", + " 152: router.use(async (req, res, next) => {", + " 153: req.claims = await validateToken(req.token);", + " 154: next();", + " test/auth.test.ts:24:9", + ' 23: it("rejects expired tokens", () => {', + " 24: expect(() => validateToken(expired)).toThrow(/expired/);", + " 25: });", + " test/auth.test.ts:41:9", + ' 40: it("accepts valid tokens", () => {', + " 41: const claims = validateToken(signed);", + ' 42: expect(claims.sub).toBe("u_123");', + ].join("\n"), + }, + ], + details: { + serverName: "typescript-language-server", + action: "references", + success: true, + request: { + action: "references", + file: "src/server/auth.ts", + line: 42, + symbol: "validateToken", + }, + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "No language server found for this file", + }, + ], + isError: true, + details: { + serverName: "typescript-language-server", + action: "references", + success: false, + request: { + action: "references", + file: "src/server/auth.ts", + line: 42, + symbol: "validateToken", + }, + }, + }, + }, + + debug: { + label: "Debug", + streamingArgs: { + action: "stack_trace", + }, + args: { + action: "stack_trace", + levels: 20, + }, + result: { + content: [ + { + type: "text", + text: [ + "Stack trace:", + "- #1000 validate_token @ app/server.py:42:14", + "- #1001 authenticate @ app/server.py:88:9", + "- #1002 handle_request @ app/router.py:153:20", + "- #1003 dispatch @ app/router.py:97:5", + "- #1004 @ app/server.py:212:1", + ].join("\n"), + }, + ], + details: { + action: "stack_trace", + success: true, + snapshot: { + id: "dbg-1", + adapter: "debugpy", + cwd: "/Users/dev/project", + program: "./app/server.py", + status: "stopped", + launchedAt: "2026-06-06T14:21:08.412Z", + lastUsedAt: "2026-06-06T14:22:55.901Z", + threadId: 1, + frameId: 1000, + stopReason: "breakpoint", + stopDescription: "breakpoint 2", + frameName: "validate_token", + instructionPointerReference: "0x00000001000034a8", + source: { name: "server.py", path: "app/server.py" }, + line: 42, + column: 14, + breakpointFiles: 1, + breakpointCount: 2, + functionBreakpointCount: 0, + outputBytes: 248, + outputTruncated: false, + needsConfigurationDone: false, + }, + stackFrames: [ + { + id: 1000, + name: "validate_token", + source: { name: "server.py", path: "app/server.py" }, + line: 42, + column: 14, + }, + { + id: 1001, + name: "authenticate", + source: { name: "server.py", path: "app/server.py" }, + line: 88, + column: 9, + }, + { + id: 1002, + name: "handle_request", + source: { name: "router.py", path: "app/router.py" }, + line: 153, + column: 20, + }, + { + id: 1003, + name: "dispatch", + source: { name: "router.py", path: "app/router.py" }, + line: 97, + column: 5, + }, + { + id: 1004, + name: "", + source: { name: "server.py", path: "app/server.py" }, + line: 212, + column: 1, + }, + ], + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "No active debug session. Launch or attach first.", + }, + ], + isError: true, + details: { + action: "stack_trace", + success: false, + }, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/edit.ts b/packages/coding-agent/src/cli/gallery-fixtures/edit.ts new file mode 100644 index 000000000..b1da61d4b --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/edit.ts @@ -0,0 +1,194 @@ +/** Gallery fixtures for the edit tools (edit, apply_patch, ast_edit). */ +import type { GalleryFixture } from "./types"; + +export const editFixtures: Record = { + edit: { + label: "Edit", + editMode: "replace", + // `previewDiff` is surfaced verbatim by the renderer's call preview, and the + // harness diff strategy skips `{ file_path, previewDiff }` (no `path`/`edits`), + // so the canned diff survives the streaming and progress states. + streamingArgs: { + file_path: "packages/coding-agent/src/tools/read.ts", + previewDiff: [ + "@@ -88,3 +88,4 @@", + " const offset = args.offset ?? 1;", + "- const limit = args.limit ?? 2000;", + "+ const limit = args.limit ?? 4000;", + ].join("\n"), + }, + args: { + file_path: "packages/coding-agent/src/tools/read.ts", + previewDiff: [ + "@@ -88,5 +88,6 @@", + " const offset = args.offset ?? 1;", + "- const limit = args.limit ?? 2000;", + "+ const limit = args.limit ?? 4000;", + " const raw = await Bun.file(path).text();", + "- return raw.slice(offset, offset + limit);", + '+ return raw.split("\\n").slice(offset - 1, offset - 1 + limit).join("\\n");', + ].join("\n"), + }, + result: { + content: [{ type: "text", text: "Edited packages/coding-agent/src/tools/read.ts (1 hunk, +3 -2)" }], + details: { + path: "packages/coding-agent/src/tools/read.ts", + firstChangedLine: 89, + diff: [ + "@@ -88,5 +88,6 @@", + " const offset = args.offset ?? 1;", + "- const limit = args.limit ?? 2000;", + "+ const limit = args.limit ?? 4000;", + " const raw = await Bun.file(path).text();", + "- return raw.slice(offset, offset + limit);", + '+ return raw.split("\\n").slice(offset - 1, offset - 1 + limit).join("\\n");', + ].join("\n"), + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "Edit failed: the search text was not found in packages/coding-agent/src/tools/read.ts", + }, + ], + isError: true, + details: { + path: "packages/coding-agent/src/tools/read.ts", + diff: "", + errorText: + "No match for the search text. Expected `const limit = args.limit ?? 2000;` near line 89, but the file has `const limit = args.limit ?? 1000;`. Re-read the file and retry with the current contents.", + }, + }, + }, + + apply_patch: { + label: "Apply Patch", + editMode: "apply_patch", + streamingArgs: { + file_path: "packages/coding-agent/src/edit/renderer.ts", + previewDiff: [ + "@@ -464,2 +464,2 @@", + "- fileCount = countEditFiles(editArgs.edits);", + "+ fileCount = countDistinctFiles(editArgs.edits);", + ].join("\n"), + }, + args: { + file_path: "packages/coding-agent/src/edit/renderer.ts", + previewDiff: [ + "@@ -177,4 +177,4 @@", + " /** Count distinct file paths in an edits array. */", + "-function countEditFiles(edits: EditRenderEntry[]): number {", + "+function countDistinctFiles(edits: EditRenderEntry[]): number {", + " return new Set(edits.map(edit => filePathFromEditEntry(edit.path)).filter(Boolean)).size;", + " }", + "@@ -467,2 +467,2 @@", + "- fileCount = countEditFiles(editArgs.edits);", + "+ fileCount = countDistinctFiles(editArgs.edits);", + ].join("\n"), + }, + result: { + content: [ + { type: "text", text: "Applied patch to packages/coding-agent/src/edit/renderer.ts (2 hunks, +2 -2)" }, + ], + details: { + op: "update", + path: "packages/coding-agent/src/edit/renderer.ts", + firstChangedLine: 178, + diff: [ + "@@ -177,4 +177,4 @@", + " /** Count distinct file paths in an edits array. */", + "-function countEditFiles(edits: EditRenderEntry[]): number {", + "+function countDistinctFiles(edits: EditRenderEntry[]): number {", + " return new Set(edits.map(edit => filePathFromEditEntry(edit.path)).filter(Boolean)).size;", + " }", + "@@ -467,2 +467,2 @@", + "- fileCount = countEditFiles(editArgs.edits);", + "+ fileCount = countDistinctFiles(editArgs.edits);", + ].join("\n"), + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "Apply patch failed: context does not match at line 177 of packages/coding-agent/src/edit/renderer.ts", + }, + ], + isError: true, + details: { + op: "update", + path: "packages/coding-agent/src/edit/renderer.ts", + diff: "", + errorText: + "Hunk @@ -177,4 +177,4 @@ failed to apply: the context line `function countEditFiles(edits: EditRenderEntry[]): number {` does not match the file. The file may have changed since it was read.", + }, + }, + }, + + ast_edit: { + label: "AST Edit", + streamingArgs: { + ops: [{ pat: "countEditFiles($$$ARGS)" }], + paths: ["packages/coding-agent/src/**/*.ts"], + }, + args: { + ops: [{ pat: "countEditFiles($$$ARGS)", out: "countDistinctFiles($$$ARGS)" }], + paths: ["packages/coding-agent/src/**/*.ts"], + }, + result: { + content: [ + { + type: "text", + text: [ + "# edit/renderer.ts (2 replacements)", + "-468: fileCount = countEditFiles(editArgs.edits);", + "+468: fileCount = countDistinctFiles(editArgs.edits);", + "-488: const totalFiles = args?.edits ? countEditFiles(args.edits) : 0;", + "+488: const totalFiles = args?.edits ? countDistinctFiles(args.edits) : 0;", + "", + "# tools/tool-result.ts (1 replacement)", + "-42: return countEditFiles(files);", + "+42: return countDistinctFiles(files);", + ].join("\n"), + }, + ], + details: { + totalReplacements: 3, + filesTouched: 2, + filesSearched: 214, + applied: false, + limitReached: false, + scopePath: "packages/coding-agent/src", + searchPath: "/Users/dev/Projects/pi/packages/coding-agent/src", + files: ["edit/renderer.ts", "tools/tool-result.ts"], + fileReplacements: [ + { path: "edit/renderer.ts", count: 2 }, + { path: "tools/tool-result.ts", count: 1 }, + ], + displayContent: [ + "# edit/", + "## renderer.ts (2 replacements)", + "-468│ fileCount = countEditFiles(editArgs.edits);", + "+468│ fileCount = countDistinctFiles(editArgs.edits);", + "-488│ const totalFiles = args?.edits ? countEditFiles(args.edits) : 0;", + "+488│ const totalFiles = args?.edits ? countDistinctFiles(args.edits) : 0;", + "", + "# tools/", + "## tool-result.ts (1 replacement)", + "-42│ return countEditFiles(files);", + "+42│ return countDistinctFiles(files);", + ].join("\n"), + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "Pattern parse error in ops[0].pat: unbalanced parenthesis in `countEditFiles($$$ARGS`", + }, + ], + isError: true, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/fs.ts b/packages/coding-agent/src/cli/gallery-fixtures/fs.ts new file mode 100644 index 000000000..290571544 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/fs.ts @@ -0,0 +1,153 @@ +// biome-ignore-all lint/suspicious/noTemplateCurlyInString: sample source-code strings (read fixtures) intentionally contain literal ${...}. +// Gallery fixtures for the filesystem tools (read, write, find). +import type { GalleryFixture } from "./types"; + +const readSnippet = [ + "export const findToolRenderer = {", + "\tinline: true,", + "\trenderCall(args: FindRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component {", + "\t\tconst meta: string[] = [];", + "\t\tif (args.limit !== undefined) meta.push(`limit:${args.limit}`);", + "", + "\t\tconst text = renderStatusLine(", + '\t\t\t{ icon: "pending", title: "Find", description: formatFindRenderPaths(args.paths) || "*", meta },', + "\t\t\tuiTheme,", + "\t\t);", + "\t\treturn new Text(text, 0, 0);", + "\t},", +].join("\n"); + +const writtenContent = [ + 'import { describe, expect, it } from "bun:test";', + 'import { parseSel } from "../src/tools/read";', + "", + 'describe("parseSel", () => {', + '\tit("parses a single line range", () => {', + '\t\texpect(parseSel("42-58")).toEqual({', + '\t\t\tkind: "lines",', + "\t\t\tranges: [{ startLine: 42, endLine: 58 }],", + "\t\t});", + "\t});", + "", + '\tit("treats raw as a verbatim selector", () => {', + '\t\texpect(parseSel("raw")).toEqual({ kind: "raw" });', + "\t});", + "});", + "", +].join("\n"); + +export const fsFixtures: Record = { + read: { + label: "Read", + // Streaming: path still being typed, selector not yet appended. + streamingArgs: { path: "packages/coding-agent/src/tools/find" }, + args: { path: "packages/coding-agent/src/tools/find.ts:437-448" }, + result: { + content: [ + { + type: "text", + text: [ + "[packages/coding-agent/src/tools/find.ts#E48E]", + "437:export const findToolRenderer = {", + "438:\tinline: true,", + "439:\trenderCall(args: FindRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component {", + "440:\t\tconst meta: string[] = [];", + "441:\t\tif (args.limit !== undefined) meta.push(`limit:${args.limit}`);", + "442:", + "443:\t\tconst text = renderStatusLine(", + '444:\t\t\t{ icon: "pending", title: "Find", description: formatFindRenderPaths(args.paths) || "*", meta },', + "445:\t\t\tuiTheme,", + "446:\t\t);", + "447:\t\treturn new Text(text, 0, 0);", + "448:\t},", + ].join("\n"), + }, + ], + details: { + kind: "file", + resolvedPath: "/Users/dev/Projects/pi/packages/coding-agent/src/tools/find.ts", + contentType: "text/typescript", + displayContent: { text: readSnippet, startLine: 437 }, + }, + }, + errorResult: { + isError: true, + content: [ + { + type: "text", + text: "Error: ENOENT: no such file or directory, open 'packages/coding-agent/src/tools/find.ts'", + }, + ], + }, + }, + + write: { + label: "Write", + // Streaming: path known, content still arriving (only the imports so far). + streamingArgs: { + path: "packages/coding-agent/test/parse-sel.test.ts", + content: 'import { describe, expect, it } from "bun:test";\nimport { parseSel } from "../src/tools/read";\n', + }, + args: { + path: "packages/coding-agent/test/parse-sel.test.ts", + content: writtenContent, + }, + result: { + content: [ + { + type: "text", + text: "Created packages/coding-agent/test/parse-sel.test.ts (17 lines, 412 bytes).", + }, + ], + details: {}, + }, + errorResult: { + isError: true, + content: [ + { + type: "text", + text: "Error: EACCES: permission denied, open 'packages/coding-agent/test/parse-sel.test.ts'", + }, + ], + }, + }, + + find: { + label: "Find", + // Streaming: glob half-typed, no limit yet. + streamingArgs: { paths: ["packages/coding-agent/src/tools/*-render"] }, + args: { paths: ["packages/coding-agent/src/**/*.test.ts"], limit: 50 }, + result: { + content: [ + { + type: "text", + text: [ + "packages/coding-agent/src/tools/read.test.ts", + "packages/coding-agent/src/tools/write.test.ts", + "packages/coding-agent/src/tools/find.test.ts", + "packages/coding-agent/src/cli/gallery-cli.test.ts", + "packages/coding-agent/src/edit/edit.test.ts", + ].join("\n"), + }, + ], + details: { + scopePath: "packages/coding-agent/src", + cwd: "/Users/dev/Projects/pi", + fileCount: 5, + truncated: false, + files: [ + "packages/coding-agent/src/cli/gallery-cli.test.ts", + "packages/coding-agent/src/edit/edit.test.ts", + "packages/coding-agent/src/tools/find.test.ts", + "packages/coding-agent/src/tools/read.test.ts", + "packages/coding-agent/src/tools/write.test.ts", + ], + }, + }, + errorResult: { + isError: true, + content: [{ type: "text", text: "Find failed: invalid glob pattern '[unclosed'." }], + details: { error: "invalid glob pattern '[unclosed'" }, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/index.ts b/packages/coding-agent/src/cli/gallery-fixtures/index.ts new file mode 100644 index 000000000..404082263 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/index.ts @@ -0,0 +1,40 @@ +/** + * Aggregated sample data for the `omp gallery` command. + * + * Each fixture drives one tool's renderer through the four lifecycle states the + * gallery showcases: arguments streaming in, arguments complete but awaiting a + * result, a successful result, and a failed result. The data is intentionally + * hand-written (rather than schema-derived) so the gallery reflects what a real + * tool call looks like — the whole point is visual QA of the renderers. + * + * Fixtures are grouped by subsystem into sibling modules and merged here. + * Adding a tool to one of those groups is enough for the gallery to render it. + * Tools present in the renderer registry but missing here fall back to a + * generic fixture (see `gallery-cli.ts`), so the gallery never crashes on a + * newly added tool — it just looks plain until a fixture is supplied. + */ +import { agenticFixtures } from "./agentic"; +import { codeintelFixtures } from "./codeintel"; +import { editFixtures } from "./edit"; +import { fsFixtures } from "./fs"; +import { interactionFixtures } from "./interaction"; +import { memoryFixtures } from "./memory"; +import { miscFixtures } from "./misc"; +import { searchFixtures } from "./search"; +import { shellFixtures } from "./shell"; +import { webFixtures } from "./web"; + +export * from "./types"; + +export const galleryFixtures = { + ...interactionFixtures, + ...shellFixtures, + ...fsFixtures, + ...searchFixtures, + ...editFixtures, + ...agenticFixtures, + ...memoryFixtures, + ...webFixtures, + ...codeintelFixtures, + ...miscFixtures, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/interaction.ts b/packages/coding-agent/src/cli/gallery-fixtures/interaction.ts new file mode 100644 index 000000000..34da85a14 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/interaction.ts @@ -0,0 +1,49 @@ +/** Gallery fixtures for the todo / ask / resolve interaction tools. */ +import type { GalleryFixture } from "./types"; + +export const interactionFixtures: Record = { + todo: { + label: "Todo", + streamingArgs: { + ops: [{ op: "init", list: [{ phase: "Foundation", items: ["Scaffold crate"] }] }], + }, + args: { + ops: [ + { + op: "init", + list: [ + { phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] }, + { phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] }, + ], + }, + ], + }, + result: { + content: [{ type: "text", text: "Initialized 4 tasks across 2 phases" }], + details: { + storage: "session", + phases: [ + { + name: "Foundation", + tasks: [ + { content: "Scaffold crate", status: "done" }, + { content: "Wire workspace", status: "in_progress" }, + ], + }, + { + name: "Auth", + tasks: [ + { content: "Port credential store", status: "pending" }, + { content: "Wire OAuth providers", status: "pending" }, + ], + }, + ], + completedTasks: [{ phase: "Foundation", content: "Scaffold crate" }], + }, + }, + errorResult: { + content: [{ type: "text", text: "Unknown phase 'Auth' — initialize the list first" }], + isError: true, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/memory.ts b/packages/coding-agent/src/cli/gallery-fixtures/memory.ts new file mode 100644 index 000000000..01b70e7e4 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/memory.ts @@ -0,0 +1,81 @@ +// Gallery fixtures for the long-term memory tools (retain, recall, reflect). +import type { GalleryFixture } from "./types"; + +export const memoryFixtures: Record = { + retain: { + label: "Retain", + // Streaming: first item complete, second still arriving without a context. + streamingArgs: { + items: [{ content: "User prefers Bun over Node for all new scripts in this repo." }], + }, + args: { + items: [ + { + content: "User prefers Bun over Node for all new scripts in this repo.", + context: "Established while wiring up the gallery command tooling.", + }, + { + content: "The TUI renderers live in packages/coding-agent/src/tools/*-render.ts.", + context: "Discovered during the gallery-fixtures task.", + }, + ], + }, + result: { + content: [{ type: "text", text: "2 memories stored." }], + details: { count: 2 }, + }, + errorResult: { + isError: true, + content: [{ type: "text", text: "Retain failed: memory store is not initialized." }], + }, + }, + + recall: { + label: "Recall", + // Streaming: query partially typed. + streamingArgs: { query: "bun vs node" }, + args: { query: "Which runtime does the user prefer for scripts?" }, + result: { + content: [ + { + type: "text", + text: [ + "Found 2 relevant memories:", + "", + "1. [0.92] User prefers Bun over Node for all new scripts in this repo.", + " (Established while wiring up the gallery command tooling.)", + "2. [0.78] The TUI renderers live in packages/coding-agent/src/tools/*-render.ts.", + " (Discovered during the gallery-fixtures task.)", + ].join("\n"), + }, + ], + }, + errorResult: { + isError: true, + content: [{ type: "text", text: "Recall failed: vector index unavailable." }], + }, + }, + + reflect: { + label: "Reflect", + streamingArgs: { query: "what have we learned about the user's" }, + args: { query: "What have we learned about the user's tooling preferences?" }, + result: { + content: [ + { + type: "text", + text: [ + "The user consistently favors Bun as the runtime for scripts in this", + "repository, avoiding Node where possible. They also track the location", + "of TUI renderers under packages/coding-agent/src/tools, suggesting an", + "interest in keeping rendering logic discoverable and well-organized.", + ].join("\n"), + }, + ], + }, + errorResult: { + isError: true, + content: [{ type: "text", text: "Reflect failed: no memories matched the query." }], + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/misc.ts b/packages/coding-agent/src/cli/gallery-fixtures/misc.ts new file mode 100644 index 000000000..80200a749 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/misc.ts @@ -0,0 +1,221 @@ +/** Gallery fixtures for the ask / resolve / ssh / github / inspect_image tools. */ +import type { GalleryFixture } from "./types"; + +export const miscFixtures: Record = { + ask: { + label: "Ask", + streamingArgs: { + questions: [ + { + id: "db", + question: "Which database should the new service use?", + options: [{ label: "Postgres" }], + }, + ], + }, + args: { + questions: [ + { + id: "db", + question: "Which database should the new service use?", + options: [ + { label: "Postgres", description: "Relational, strong consistency, JSONB support" }, + { label: "SQLite", description: "Embedded, zero-ops, great for single-node" }, + { label: "MongoDB", description: "Document store, flexible schema" }, + ], + recommended: 0, + }, + { + id: "features", + question: "Which auth flows should ship in v1?", + options: [ + { label: "Email + password" }, + { label: "OAuth (Google, GitHub)" }, + { label: "Magic links" }, + { label: "SAML SSO", description: "Enterprise; can be deferred" }, + ], + multi: true, + }, + ], + }, + result: { + content: [ + { + type: "text", + text: "db: Postgres\nfeatures: Email + password, OAuth (Google, GitHub)", + }, + ], + details: { + results: [ + { + id: "db", + question: "Which database should the new service use?", + options: ["Postgres", "SQLite", "MongoDB"], + multi: false, + selectedOptions: ["Postgres"], + }, + { + id: "features", + question: "Which auth flows should ship in v1?", + options: ["Email + password", "OAuth (Google, GitHub)", "Magic links", "SAML SSO"], + multi: true, + selectedOptions: ["Email + password", "OAuth (Google, GitHub)"], + }, + ], + }, + }, + errorResult: { + content: [{ type: "text", text: "Prompt cancelled by user before any answer was given" }], + isError: true, + }, + }, + + resolve: { + label: "Resolve", + streamingArgs: { + action: "apply", + }, + args: { + action: "apply", + reason: "Rename is mechanical and the staged diff matches the intended refactor.", + extra: { title: "rename-usecredentials-hook" }, + }, + result: { + content: [{ type: "text", text: "Applied pending ast_edit: 7 replacements across 3 files" }], + details: { + action: "apply", + reason: "Rename is mechanical and the staged diff matches the intended refactor.", + extra: { title: "rename-usecredentials-hook" }, + sourceToolName: "ast_edit", + label: "ast_edit: 7 replacements across 3 files", + }, + }, + errorResult: { + content: [{ type: "text", text: "No pending action to resolve" }], + isError: true, + details: { + action: "apply", + reason: "Rename is mechanical and the staged diff matches the intended refactor.", + sourceToolName: "ast_edit", + label: "ast_edit: 7 replacements across 3 files", + }, + }, + }, + + ssh: { + label: "SSH", + streamingArgs: { + host: "deploy@web-01", + command: "systemctl status", + }, + args: { + host: "deploy@web-01", + command: "systemctl status omp-api --no-pager | head -n 12", + cwd: "/srv/omp", + timeout: 60, + }, + result: { + content: [ + { + type: "text", + text: [ + "● omp-api.service - Oh My Pi API", + " Loaded: loaded (/etc/systemd/system/omp-api.service; enabled)", + " Active: active (running) since Sat 2026-06-06 09:14:02 UTC; 3h 21min ago", + " Main PID: 4812 (bun)", + " Tasks: 17 (limit: 4915)", + " Memory: 142.6M", + " CPU: 38.214s", + " CGroup: /system.slice/omp-api.service", + " └─4812 /usr/local/bin/bun run dist/server.js", + ].join("\n"), + }, + ], + }, + errorResult: { + content: [ + { + type: "text", + text: "ssh: connect to host web-01 port 22: Connection timed out", + }, + ], + isError: true, + }, + }, + + github: { + label: "GitHub", + streamingArgs: { + op: "search_prs", + query: "is:open author:@me", + }, + args: { + op: "search_prs", + query: "is:open review-requested:@me sort:updated", + repo: "oh-my-pi/pi", + }, + result: { + content: [ + { + type: "text", + text: [ + "#1842 feat(tui): virtualized scrollback for tool output openyou · 2h ago +312 -47", + "#1839 fix(agent): retry stream on transient 529 dvir · 5h ago +18 -4", + "#1830 refactor(edit): unify hashline + ast_edit previews mira · 1d ago +540 -210", + "#1817 docs: document gallery fixtures contract leo · 2d ago +96 -0", + "", + "4 open pull requests requesting your review", + ].join("\n"), + }, + ], + }, + errorResult: { + content: [ + { + type: "text", + text: "gh: Could not resolve to a Repository with the name 'oh-my-pi/pi'. (HTTP 404)", + }, + ], + isError: true, + }, + }, + + inspect_image: { + label: "Inspect Image", + streamingArgs: { + path: "docs/assets/dashboard-mock.png", + }, + args: { + path: "docs/assets/dashboard-mock.png", + question: "What chart types are shown and roughly what layout does the dashboard use?", + }, + result: { + content: [ + { + type: "text", + text: [ + "The dashboard uses a two-column layout on a dark background.", + "Top row: four KPI cards (Revenue, Active Users, Churn, MRR) with sparklines.", + "Left column: a stacked area chart of weekly sessions over ~3 months.", + "Right column: a horizontal bar chart ranking the top 6 referrers.", + "Bottom: a paginated table of recent transactions with status pills.", + ].join("\n"), + }, + ], + details: { + model: "claude-opus-4", + imagePath: "docs/assets/dashboard-mock.png", + mimeType: "image/png", + }, + }, + errorResult: { + content: [{ type: "text", text: "Image not found: docs/assets/dashboard-mock.png" }], + isError: true, + details: { + model: "claude-opus-4", + imagePath: "docs/assets/dashboard-mock.png", + mimeType: "image/png", + }, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/search.ts b/packages/coding-agent/src/cli/gallery-fixtures/search.ts new file mode 100644 index 000000000..2a0962070 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/search.ts @@ -0,0 +1,213 @@ +/** Gallery fixtures for the search tools (search, search_tool_bm25, ast_grep). */ +import type { GalleryFixture } from "./types"; + +export const searchFixtures: Record = { + search: { + label: "Search", + streamingArgs: { + pattern: "useState", + }, + args: { + pattern: "useState", + paths: ["packages/tui/src"], + }, + result: { + content: [ + { + type: "text", + text: [ + "# packages/tui/src/components/", + "## SearchBox.tsx", + '18: const [query, setQuery] = useState("");', + "19: const [results, setResults] = useState([]);", + "## StatusBar.tsx", + "27: const [expanded, setExpanded] = useState(false);", + "", + "# packages/tui/src/hooks/", + "## useDebounced.ts", + "9: const [value, setValue] = useState(initial);", + "10: const [pending, setPending] = useState(false);", + ].join("\n"), + }, + ], + details: { + scopePath: "packages/tui/src", + searchPath: "/Users/dev/Projects/pi/packages/tui/src", + matchCount: 5, + fileCount: 3, + files: [ + "packages/tui/src/components/SearchBox.tsx", + "packages/tui/src/components/StatusBar.tsx", + "packages/tui/src/hooks/useDebounced.ts", + ], + fileMatches: [ + { path: "packages/tui/src/components/SearchBox.tsx", count: 2 }, + { path: "packages/tui/src/components/StatusBar.tsx", count: 1 }, + { path: "packages/tui/src/hooks/useDebounced.ts", count: 2 }, + ], + truncated: false, + displayContent: [ + "# packages/tui/src/components/", + "## SearchBox.tsx", + '*18│ const [query, setQuery] = useState("");', + "*19│ const [results, setResults] = useState([]);", + "## StatusBar.tsx", + "*27│ const [expanded, setExpanded] = useState(false);", + "", + "# packages/tui/src/hooks/", + "## useDebounced.ts", + " *9│ const [value, setValue] = useState(initial);", + "*10│ const [pending, setPending] = useState(false);", + ].join("\n"), + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "Invalid regex pattern: unclosed group near index 8", + }, + ], + isError: true, + details: { + error: "Invalid regex pattern: unclosed group near index 8", + }, + }, + }, + + search_tool_bm25: { + label: "SearchTools", + streamingArgs: { + query: "read pdf and ext", + }, + args: { + query: "read pdf and extract tables", + limit: 5, + }, + result: { + content: [ + { + type: "text", + text: JSON.stringify({ + query: "read pdf and extract tables", + activated_tools: ["docling_extract_tables", "docling_convert", "pdf_read_text"], + match_count: 4, + total_tools: 142, + }), + }, + ], + details: { + query: "read pdf and extract tables", + limit: 5, + total_tools: 142, + activated_tools: ["docling_extract_tables", "docling_convert", "pdf_read_text"], + active_selected_tools: ["read", "search", "edit", "bash"], + tools: [ + { + name: "docling_extract_tables", + label: "Extract Tables", + description: "Extract tabular data from PDF documents into CSV or JSON rows.", + server_name: "docling", + mcp_tool_name: "extract_tables", + schema_keys: ["path", "pages", "format"], + score: 9.412037, + }, + { + name: "docling_convert", + label: "Convert Document", + description: "Convert PDF, DOCX, or PPTX into structured Markdown with layout preserved.", + server_name: "docling", + mcp_tool_name: "convert", + schema_keys: ["path", "target", "ocr"], + score: 6.83102, + }, + { + name: "pdf_read_text", + label: "Read PDF Text", + description: "Read raw text from a PDF, optionally scoped to a page range.", + server_name: "pdf-tools", + mcp_tool_name: "read_text", + schema_keys: ["path", "page_start", "page_end"], + score: 5.207884, + }, + { + name: "tabula_scan", + label: "Scan Tables", + description: "Detect table bounding boxes on scanned PDF pages before extraction.", + server_name: "pdf-tools", + mcp_tool_name: "scan", + schema_keys: ["path", "dpi"], + score: 3.119556, + }, + ], + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "Tool discovery is disabled. Enable tools.discoveryMode or mcp.discoveryMode to use search_tool_bm25.", + }, + ], + isError: true, + }, + }, + + ast_grep: { + label: "AST Grep", + streamingArgs: { + pat: "useState(", + }, + args: { + pat: "useState($A)", + paths: ["packages/tui/src/components"], + }, + result: { + content: [ + { + type: "text", + text: [ + "# packages/tui/src/components/", + "## SearchBox.tsx", + '18: const [query, setQuery] = useState("");', + ' meta: $A=""', + "## StatusBar.tsx", + "27: const [expanded, setExpanded] = useState(false);", + " meta: $A=false", + ].join("\n"), + }, + ], + details: { + matchCount: 2, + fileCount: 2, + filesSearched: 14, + limitReached: false, + scopePath: "packages/tui/src/components", + searchPath: "/Users/dev/Projects/pi/packages/tui/src/components", + files: ["packages/tui/src/components/SearchBox.tsx", "packages/tui/src/components/StatusBar.tsx"], + fileMatches: [ + { path: "packages/tui/src/components/SearchBox.tsx", count: 1 }, + { path: "packages/tui/src/components/StatusBar.tsx", count: 1 }, + ], + displayContent: [ + "# packages/tui/src/components/", + "## SearchBox.tsx", + '*18│ const [query, setQuery] = useState("");', + ' meta: $A=""', + "## StatusBar.tsx", + "*27│ const [expanded, setExpanded] = useState(false);", + " meta: $A=false", + ].join("\n"), + }, + }, + errorResult: { + content: [ + { + type: "text", + text: "Pattern parse error: incomplete node `useState(` — expected a closing `)`", + }, + ], + isError: true, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/shell.ts b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts new file mode 100644 index 000000000..81bb0739d --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts @@ -0,0 +1,167 @@ +/** Gallery fixtures for the shell tools (bash, eval). */ +import type { GalleryFixture } from "./types"; + +export const shellFixtures: Record = { + bash: { + label: "Bash", + streamingArgs: { + command: "git status --short && git log --on", + }, + args: { + command: "git status --short && git log --oneline -5", + cwd: "packages/coding-agent", + timeout: 30, + }, + result: { + content: [ + { + type: "text", + text: [ + " M src/cli/gallery-cli.ts", + " M src/tools/bash.ts", + "?? src/cli/gallery-fixtures/shell.ts", + "a1b2c3d Wire gallery command into CLI dispatch", + "9f8e7d6 Add ToolExecutionComponent lifecycle states", + "4c5b6a7 Extract createShellRenderer from bashToolRenderer", + "2d3e4f5 Strip LLM-facing notices before TUI render", + "7a8b9c0 Cap preview lines in pending command block", + ].join("\n"), + }, + ], + details: { + exitCode: 0, + wallTimeMs: 184, + timeoutSeconds: 30, + }, + }, + errorResult: { + content: [ + { + type: "text", + text: [ + "src/tools/bash.ts:1142:34 - error TS2339: Property 'requestedTimeoutSeconds' does not exist on type 'BashToolDetails'.", + "", + "1142 const requestedTimeoutSeconds = details?.requestedTimeoutSeconds;", + " ~~~~~~~~~~~~~~~~~~~~~~~~", + "Found 1 error in src/tools/bash.ts:1142", + ].join("\n"), + }, + ], + isError: true, + details: { + exitCode: 2, + wallTimeMs: 5120, + timeoutSeconds: 30, + }, + }, + }, + + eval: { + label: "Eval", + streamingArgs: { + cells: [ + { + language: "py", + code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js', + title: "load config", + }, + ], + }, + args: { + cells: [ + { + language: "py", + title: "load config", + code: [ + "import json", + "from pathlib import Path", + "", + 'data = json.loads(Path("package.json").read_text())', + 'deps = data.get("dependencies", {})', + 'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")', + 'print(f"{len(deps)} dependencies")', + "display(sorted(deps)[:3])", + ].join("\n"), + }, + ], + }, + result: { + content: [ + { + type: "text", + text: ["@oh-my-pi/coding-agent v0.42.0", "37 dependencies"].join("\n"), + }, + ], + details: { + language: "python", + languages: ["python"], + jsonOutputs: [["@ai-sdk/anthropic", "@oh-my-pi/pi-ai", "@oh-my-pi/pi-tui"]], + cells: [ + { + index: 0, + title: "load config", + language: "python", + code: [ + "import json", + "from pathlib import Path", + "", + 'data = json.loads(Path("package.json").read_text())', + 'deps = data.get("dependencies", {})', + 'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")', + 'print(f"{len(deps)} dependencies")', + "display(sorted(deps)[:3])", + ].join("\n"), + output: ["@oh-my-pi/coding-agent v0.42.0", "37 dependencies"].join("\n"), + status: "complete", + durationMs: 64, + exitCode: 0, + }, + ], + }, + }, + errorResult: { + content: [ + { + type: "text", + text: [ + "Traceback (most recent call last):", + ' File "", line 4, in ', + ' data = json.loads(Path("package.json").read_text())', + " ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^", + "json.decoder.JSONDecodeError: Expecting ',' delimiter: line 12 column 3 (char 318)", + ].join("\n"), + }, + ], + isError: true, + details: { + language: "python", + languages: ["python"], + isError: true, + cells: [ + { + index: 0, + title: "load config", + language: "python", + code: [ + "import json", + "from pathlib import Path", + "", + 'data = json.loads(Path("package.json").read_text())', + 'deps = data.get("dependencies", {})', + 'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")', + ].join("\n"), + output: [ + "Traceback (most recent call last):", + ' File "", line 4, in ', + ' data = json.loads(Path("package.json").read_text())', + "json.decoder.JSONDecodeError: Expecting ',' delimiter: line 12 column 3 (char 318)", + ].join("\n"), + status: "error", + durationMs: 41, + exitCode: 1, + }, + ], + }, + }, + }, +}; diff --git a/packages/coding-agent/src/cli/gallery-fixtures/types.ts b/packages/coding-agent/src/cli/gallery-fixtures/types.ts new file mode 100644 index 000000000..77c051ce6 --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/types.ts @@ -0,0 +1,32 @@ +/** + * Types for `omp gallery` sample data. See {@link ./index} for the aggregated + * fixture registry and the contract each fixture must satisfy. + */ +import type { EditMode } from "../../edit"; + +/** A tool result snapshot, matching the shape `ToolExecutionComponent` consumes. */ +export interface GalleryResult { + content: Array<{ type: string; text?: string; data?: string; mimeType?: string }>; + details?: unknown; + isError?: boolean; +} + +export interface GalleryFixture { + /** Display label for the tool header (defaults to the tool name). */ + label?: string; + /** Edit mode for edit-like tools so the streaming preview dispatches correctly. */ + editMode?: EditMode; + /** + * Arguments shown during the streaming state — a partial view of {@link args} + * as if the tool-call JSON were still arriving. May include `__partialJson` + * for renderers (bash, edit) that surface fields before the object closes. + * Defaults to {@link args} when omitted. + */ + streamingArgs?: unknown; + /** Complete arguments shown for the in-progress, success, and error states. */ + args: unknown; + /** Successful result. */ + result: GalleryResult; + /** Failed result. Falls back to a generic error when omitted. */ + errorResult?: GalleryResult; +} diff --git a/packages/coding-agent/src/cli/gallery-fixtures/web.ts b/packages/coding-agent/src/cli/gallery-fixtures/web.ts new file mode 100644 index 000000000..bec37debe --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-fixtures/web.ts @@ -0,0 +1,158 @@ +// Gallery fixtures for the web tools (web_search, browser). +import type { GalleryFixture } from "./types"; + +export const webFixtures: Record = { + web_search: { + label: "Web Search", + // Streaming: query still being typed, no recency/limit yet. + streamingArgs: { query: "bun vs node performance" }, + args: { + query: "Bun vs Node.js performance benchmarks 2026", + recency: "month", + limit: 4, + }, + result: { + content: [ + { + type: "text", + text: [ + "Bun continues to outperform Node.js on raw HTTP throughput and cold-start", + "time thanks to its JavaScriptCore engine and native-Zig runtime, while", + "Node.js retains an edge in ecosystem maturity and long-term stability.", + "For script-heavy workflows Bun's faster startup is the decisive factor.", + ].join("\n"), + }, + ], + details: { + response: { + provider: "perplexity", + model: "sonar-pro", + authMode: "api_key", + requestId: "req_a1b2c3d4e5f6", + answer: [ + "Bun continues to outperform Node.js on raw HTTP throughput and cold-start", + "time thanks to its JavaScriptCore engine and native-Zig runtime, while", + "Node.js retains an edge in ecosystem maturity and long-term stability.", + "For script-heavy workflows Bun's faster startup is the decisive factor.", + ].join("\n"), + searchQueries: ["bun vs node.js performance benchmarks 2026", "bun http throughput vs node"], + sources: [ + { + title: "Bun 1.2 Benchmarks: HTTP, SQLite, and Startup Time", + url: "https://bun.sh/blog/bun-v1.2-benchmarks", + snippet: + "Bun serves roughly 2.5x the requests per second of Node.js on a simple HTTP server and starts in under 10ms.", + ageSeconds: 86400 * 12, + author: "The Bun Team", + }, + { + title: "Node.js vs Bun: A 2026 Performance Deep Dive", + url: "https://blog.platformatic.dev/nodejs-vs-bun-2026", + snippet: + "Across CPU-bound workloads the gap narrows, but Bun's faster module resolution keeps cold starts ahead.", + ageSeconds: 86400 * 3, + author: "Matteo Collina", + }, + { + title: "Real-world API latency: Bun, Deno, and Node compared", + url: "https://www.theregister.com/2026/05/18/js_runtime_latency/", + snippet: + "Under sustained load p99 latencies converge, suggesting runtime choice matters less for steady-state services.", + ageSeconds: 86400 * 19, + }, + { + title: "Why we migrated our CLI tooling from Node to Bun", + url: "https://engineering.example.com/posts/bun-cli-migration", + snippet: + "Startup dropped from 180ms to 22ms, shaving seconds off every developer command invocation.", + ageSeconds: 86400 * 27, + author: "Dana Whitfield", + }, + ], + citations: [ + { + url: "https://bun.sh/blog/bun-v1.2-benchmarks", + title: "Bun 1.2 Benchmarks", + citedText: "Bun serves roughly 2.5x the requests per second of Node.js", + }, + ], + usage: { + inputTokens: 312, + outputTokens: 248, + totalTokens: 560, + searchRequests: 2, + }, + }, + }, + }, + errorResult: { + isError: true, + content: [{ type: "text", text: "Web search failed: provider returned HTTP 429 (rate limited)." }], + details: { + response: { + provider: "perplexity", + sources: [], + }, + error: "Provider returned HTTP 429 (rate limited). Retry after 30s.", + }, + }, + }, + + browser: { + label: "Browser", + // Streaming: code body still arriving for a `run` action. + streamingArgs: { + action: "run", + name: "docs", + code: "const obs = await tab.observe();\n", + }, + args: { + action: "run", + name: "docs", + code: [ + "const obs = await tab.observe();", + "const heading = obs.elements.find(e => e.role === 'heading');", + "display({ url: obs.url, title: obs.title, headings: obs.elements.filter(e => e.role === 'heading').length });", + "return heading?.name ?? 'no heading found';", + ].join("\n"), + }, + result: { + content: [ + { + type: "text", + text: [ + '{ url: "https://bun.sh/docs", title: "Bun Documentation", headings: 14 }', + '"Get started with Bun"', + ].join("\n"), + }, + ], + details: { + action: "run", + name: "docs", + url: "https://bun.sh/docs", + browser: "headless", + viewport: { width: 1280, height: 800, deviceScaleFactor: 1 }, + result: '"Get started with Bun"', + }, + }, + errorResult: { + isError: true, + content: [ + { + type: "text", + text: [ + "TimeoutError: waiting for selector `aria/Sign in` failed: timeout 30000ms exceeded", + " at Tab.waitFor (browser/tab.ts:212:13)", + " at run (eval:3:7)", + ].join("\n"), + }, + ], + details: { + action: "run", + name: "docs", + url: "https://bun.sh/docs", + browser: "headless", + }, + }, + }, +}; diff --git a/packages/coding-agent/src/commands/gallery.ts b/packages/coding-agent/src/commands/gallery.ts new file mode 100644 index 000000000..ee5b69cba --- /dev/null +++ b/packages/coding-agent/src/commands/gallery.ts @@ -0,0 +1,37 @@ +/** + * Render every built-in tool's renderer across its lifecycle states. + */ +import { Command, Flags } from "@oh-my-pi/pi-utils/cli"; +import { GALLERY_STATES, type GalleryState, runGalleryCommand } from "../cli/gallery-cli"; + +export default class Gallery extends Command { + static description = "Preview tool renderers across streaming, in-progress, success, and failure states"; + + static flags = { + tool: Flags.string({ char: "t", description: "Render a single tool by name" }), + state: Flags.string({ + char: "s", + description: "Render only the given lifecycle state(s)", + options: [...GALLERY_STATES], + multiple: true, + }), + width: Flags.integer({ char: "w", description: "Render width in columns" }), + expanded: Flags.boolean({ + char: "e", + description: "Render the expanded variant of each renderer", + default: false, + }), + plain: Flags.boolean({ description: "Strip ANSI styling from the output", default: false }), + }; + + async run(): Promise { + const { flags } = await this.parse(Gallery); + await runGalleryCommand({ + tool: flags.tool, + states: flags.state as GalleryState[] | undefined, + width: flags.width, + expanded: flags.expanded, + plain: flags.plain, + }); + } +} From d1fbb28edc417df2f6c8f4ab15250184e8b006b7 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 18:40:52 +0200 Subject: [PATCH 009/181] fix(coding-agent): removed redundant tool-name line in custom render - Fixed custom-rendered tools with `mergeCallAndResult` (e.g. `lsp`) emitting a redundant tool-name line above the framed result. - Collapsed the leading blank line for self-delimiting framed boxes. - Added gallery fidelity routing `lsp`/`task` through the custom-tool branch via a `customRendered` fixture flag. - Added gallery harness tests guarding state coverage and the custom-branch fallback label. --- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/src/cli/gallery-cli.ts | 22 +++++- .../src/cli/gallery-fixtures/agentic.ts | 1 + .../src/cli/gallery-fixtures/codeintel.ts | 1 + .../src/cli/gallery-fixtures/types.ts | 9 +++ .../src/modes/components/tool-execution.ts | 41 +++++++--- .../coding-agent/test/gallery-cli.test.ts | 79 +++++++++++++++++++ 7 files changed, 139 insertions(+), 16 deletions(-) create mode 100644 packages/coding-agent/test/gallery-cli.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b19bd4917..151342073 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,6 +5,7 @@ - Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states - Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output +- Added `omp gallery` fidelity for tools whose renderers are attached on the tool instance (`lsp`, `task`): the gallery now drives them through the same custom-tool render branch production uses, so regressions in that path surface in the gallery rather than only in a live session. - Added `app.display.reset`, bound to `Ctrl+L` by default, to force an immediate terminal display reset/redraw without resizing the window. ### Changed @@ -17,6 +18,7 @@ - Fixed framed read results rendering with an extra blank row above and below the output block. - Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice. - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. +- Fixed custom-rendered tools that set `mergeCallAndResult` (e.g. `lsp`) rendering a redundant tool-name line above the framed result once a result arrived. `ToolExecutionComponent`'s custom-tool branch now emits the fallback label only when the tool has no `renderCall` and the call is not suppressed by an existing result, matching the built-in renderer branch. ## [15.9.67] - 2026-06-06 ### Added diff --git a/packages/coding-agent/src/cli/gallery-cli.ts b/packages/coding-agent/src/cli/gallery-cli.ts index 744691a15..921e571db 100644 --- a/packages/coding-agent/src/cli/gallery-cli.ts +++ b/packages/coding-agent/src/cli/gallery-cli.ts @@ -45,10 +45,26 @@ const GENERIC_ERROR: GalleryResult = { isError: true, }; -/** Build the fake `AgentTool` the component needs for its label and edit mode. */ +/** + * Build the fake `AgentTool` the component needs for its label, edit mode, and — + * for `customRendered` fixtures — the renderer functions that route it through + * the same custom-tool branch production uses (see {@link GalleryFixture}). + */ function fakeToolFor(name: string, fixture: GalleryFixture | undefined): AgentTool | undefined { - if (!fixture?.label && !fixture?.editMode) return undefined; - return { name, label: fixture.label ?? name, mode: fixture.editMode } as unknown as AgentTool; + if (!fixture?.label && !fixture?.editMode && !fixture?.customRendered) return undefined; + const tool: Record = { name, label: fixture.label ?? name, mode: fixture.editMode }; + if (fixture.customRendered) { + const renderer = toolRenderers[name] as + | { renderCall?: unknown; renderResult?: unknown; mergeCallAndResult?: unknown; inline?: unknown } + | undefined; + if (renderer) { + tool.renderCall = renderer.renderCall; + tool.renderResult = renderer.renderResult; + tool.mergeCallAndResult = renderer.mergeCallAndResult; + tool.inline = renderer.inline; + } + } + return tool as unknown as AgentTool; } /** The curated fixture for a tool, or a generic one for registry tools lacking sample data. */ diff --git a/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts index f2dd796ef..d1c262abe 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/agentic.ts @@ -4,6 +4,7 @@ import type { GalleryFixture } from "./types"; export const agenticFixtures: Record = { task: { label: "Task", + customRendered: true, // Streaming: agent chosen, first task fully arrived, second still landing. streamingArgs: { agent: "task", diff --git a/packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts b/packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts index 5f10a9e2d..0d9faca65 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/codeintel.ts @@ -4,6 +4,7 @@ import type { GalleryFixture } from "./types"; export const codeintelFixtures: Record = { lsp: { label: "LSP", + customRendered: true, streamingArgs: { action: "references", file: "src/server/auth.ts", diff --git a/packages/coding-agent/src/cli/gallery-fixtures/types.ts b/packages/coding-agent/src/cli/gallery-fixtures/types.ts index 77c051ce6..97b7da510 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/types.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/types.ts @@ -16,6 +16,15 @@ export interface GalleryFixture { label?: string; /** Edit mode for edit-like tools so the streaming preview dispatches correctly. */ editMode?: EditMode; + /** + * Set for tools whose real `AgentTool` attaches `renderCall`/`renderResult` + * directly on the instance (e.g. `lsp`, `task`). The harness then attaches + * the registry renderer onto the fake tool so the component routes through + * the custom-tool branch — the same path production takes — instead of the + * built-in registry branch. The two branches can diverge, so exercising the + * real one keeps the gallery honest for these tools. + */ + customRendered?: boolean; /** * Arguments shown during the streaming state — a partial view of {@link args} * as if the tool-call JSON were still arriving. May include `__partialJson` diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index b232bc282..4fd08f692 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -171,6 +171,7 @@ let toolExecutionInstanceSeq = 0; export class ToolExecutionComponent extends Container { #contentBox: Box; // Used for custom tools and bash visual truncation #contentText: Text; // For built-in tools (with its own padding/bg) + #leadingSpacer: Spacer; // Blank line above the content; collapsed for self-delimiting framed boxes #multiFileBoxes: (Box | Spacer)[] = []; // Extra boxes for multi-file edit results #imageComponents: Image[] = []; #imageSpacers: Spacer[] = []; @@ -245,7 +246,8 @@ export class ToolExecutionComponent extends Container { this.#cwd = cwd; this.#args = args; - this.addChild(new Spacer(1)); + this.#leadingSpacer = new Spacer(1); + this.addChild(this.#leadingSpacer); // Always create both - contentBox for custom tools/bash/tools with renderers, contentText for other built-ins this.#contentBox = new Box(1, 1, (text: string) => theme.bg("toolPendingBg", text)); @@ -585,6 +587,8 @@ export class ToolExecutionComponent extends Container { this.#renderState.expanded = this.#expanded; this.#renderState.isPartial = this.#isPartial; this.#renderState.spinnerFrame = this.#spinnerFrame; + // Self-delimiting framed boxes don't need the leading blank line for separation. + let hasFramedBlock = false; // Check for custom tool rendering if (this.#tool && (this.#tool.renderCall || this.#tool.renderResult)) { @@ -601,22 +605,28 @@ export class ToolExecutionComponent extends Container { // call preview once result lines exist. this.#renderState.renderContext = this.#buildRenderContext(); - // Render call component + // Render call component. The fallback label only stands in for a + // missing `renderCall`; when the call is intentionally suppressed + // (mergeCallAndResult once a result exists) we render nothing here so + // the result component isn't preceded by a redundant tool-name line. const shouldRenderCall = !this.#result || !mergeCallAndResult; - if (shouldRenderCall && tool.renderCall) { - try { - const callComponent = tool.renderCall(this.#getCallArgsForRender(), this.#renderState, theme); - if (callComponent) { - contentBoxHasFramedBlock = addBoxChild(this.#contentBox, callComponent) || contentBoxHasFramedBlock; + if (shouldRenderCall) { + if (tool.renderCall) { + try { + const callComponent = tool.renderCall(this.#getCallArgsForRender(), this.#renderState, theme); + if (callComponent) { + contentBoxHasFramedBlock = + addBoxChild(this.#contentBox, callComponent) || contentBoxHasFramedBlock; + } + } catch (err) { + logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); + // Fall back to default on error + addBoxChild(this.#contentBox, new Text(theme.fg("toolTitle", theme.bold(this.#toolLabel)), 0, 0)); } - } catch (err) { - logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); - // Fall back to default on error + } else { + // No custom renderCall, show tool name addBoxChild(this.#contentBox, new Text(theme.fg("toolTitle", theme.bold(this.#toolLabel)), 0, 0)); } - } else { - // No custom renderCall, show tool name - addBoxChild(this.#contentBox, new Text(theme.fg("toolTitle", theme.bold(this.#toolLabel)), 0, 0)); } // Render result component if we have a result @@ -657,6 +667,7 @@ export class ToolExecutionComponent extends Container { } } setBoxPaddingForFramedBlock(this.#contentBox, contentBoxHasFramedBlock); + hasFramedBlock = contentBoxHasFramedBlock; } else if (this.#toolName in toolRenderers) { // Built-in tools with renderers const renderer = toolRenderers[this.#toolName]; @@ -700,6 +711,7 @@ export class ToolExecutionComponent extends Container { if (resultComponent) { const fileBoxHasFramedBlock = addBoxChild(fileBox, resultComponent); setBoxPaddingForFramedBlock(fileBox, fileBoxHasFramedBlock); + if (fileBoxHasFramedBlock) hasFramedBlock = true; } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); @@ -783,6 +795,7 @@ export class ToolExecutionComponent extends Container { } } setBoxPaddingForFramedBlock(this.#contentBox, contentBoxHasFramedBlock); + hasFramedBlock = contentBoxHasFramedBlock; } } else { // Other built-in tools: use Text directly with caching @@ -790,6 +803,8 @@ export class ToolExecutionComponent extends Container { this.#contentText.setText(this.#formatToolExecution()); } + this.#leadingSpacer.setLines(hasFramedBlock ? 0 : 1); + // Handle images (same for both custom and built-in) for (const img of this.#imageComponents) { this.removeChild(img); diff --git a/packages/coding-agent/test/gallery-cli.test.ts b/packages/coding-agent/test/gallery-cli.test.ts new file mode 100644 index 000000000..d7087f0b4 --- /dev/null +++ b/packages/coding-agent/test/gallery-cli.test.ts @@ -0,0 +1,79 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import { GALLERY_STATES, renderGalleryState, resolveFixture } from "../src/cli/gallery-cli"; +import type { GalleryFixture } from "../src/cli/gallery-fixtures"; +import { resetSettingsForTest, Settings } from "../src/config/settings"; +import { initTheme } from "../src/modes/theme/theme"; +import { toolRenderers } from "../src/tools/renderers"; + +beforeAll(async () => { + resetSettingsForTest(); + await Settings.init({ inMemory: true }); + await initTheme(false, undefined, undefined, "dark", "light"); +}); + +describe("gallery harness", () => { + it("renders every registered tool in every lifecycle state without throwing", async () => { + for (const name in toolRenderers) { + const fixture = resolveFixture(name); + for (const state of GALLERY_STATES) { + const lines = await renderGalleryState(name, fixture, state, 100); + // A renderer that produces no lines for a state is a regression: the + // component should always emit at least the call header or result. + expect(lines.length, `${name}/${state} rendered nothing`).toBeGreaterThan(0); + } + } + }); + + it("routes each state to the matching args/result (streaming args vs result, success vs error)", async () => { + const fixture: GalleryFixture = { + label: "Bash", + streamingArgs: { command: "echo STREAM_MARK" }, + args: { command: "echo PROGRESS_MARK" }, + result: { content: [{ type: "text", text: "SUCCESS_OUT" }], details: { exitCode: 0 } }, + errorResult: { content: [{ type: "text", text: "ERROR_OUT" }], isError: true, details: { exitCode: 1 } }, + }; + const render = async (state: (typeof GALLERY_STATES)[number]) => + Bun.stripANSI((await renderGalleryState("bash", fixture, state, 100)).join("\n")); + + const streaming = await render("streaming"); + expect(streaming).toContain("STREAM_MARK"); + expect(streaming).not.toContain("PROGRESS_MARK"); + expect(streaming).not.toContain("SUCCESS_OUT"); + + const progress = await render("progress"); + expect(progress).toContain("PROGRESS_MARK"); + expect(progress).not.toContain("SUCCESS_OUT"); + + const success = await render("success"); + expect(success).toContain("SUCCESS_OUT"); + expect(success).not.toContain("ERROR_OUT"); + + const error = await render("error"); + expect(error).toContain("ERROR_OUT"); + expect(error).not.toContain("SUCCESS_OUT"); + }); + + it("routes customRendered tools (lsp, task) through the custom-tool branch", async () => { + // `lsp`/`task` attach their renderers on the real AgentTool, so the gallery + // must reproduce that path. With a result present and mergeCallAndResult, the + // custom branch must NOT emit a redundant tool-name line above the result box + // (regression guard for tool-execution's custom-branch fallback label). + const lsp = resolveFixture("lsp"); + expect(lsp.customRendered).toBe(true); + const lines = await renderGalleryState("lsp", lsp, "error", 100); + const stripped = lines.map(line => Bun.stripANSI(line).trim()); + // The framed result header is present... + expect(stripped.some(line => line.includes("LSP references"))).toBe(true); + // ...but no standalone "LSP" label line precedes it. + expect(stripped).not.toContain("LSP"); + }); + + it("falls back to a generic fixture for registry tools without curated sample data", () => { + // resolveFixture never returns undefined for a registry tool, even one + // missing from the curated fixtures, so the gallery cannot crash on a newly + // added renderer. + const fixture = resolveFixture("a-tool-that-has-no-fixture"); + expect(fixture.args).toBeDefined(); + expect(fixture.result.content.length).toBeGreaterThan(0); + }); +}); From 9aa10dd92b7c9511babfd2321b2b7ccb02e82fae Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 18:52:46 +0200 Subject: [PATCH 010/181] ux(coding-agent): condensed web_search result rendering - Showed answer text in full in the TUI; kept the `omp q` compact cap. - Rendered each source as a single title/domain/age line with the URL linked on the title. - Collapsed the metadata block to one Provider line plus Usage. - Rendered search errors as a framed panel matching the success layout. --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/components/tool-execution.ts | 11 +-- .../coding-agent/src/web/search/render.ts | 93 ++++++++----------- 3 files changed, 41 insertions(+), 64 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 151342073..3dc36cc36 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -12,6 +12,7 @@ - Changed the default persistent model selector shortcut from `Ctrl+L` to `Alt+M`, leaving `Ctrl+L` for display reset. Existing user remaps in `keybindings.yml` are preserved. - Changed the edit tool result header to carry the diff change stats (`+N / -M / K hunks`) inline next to the file path, and removed the redundant lone language-icon metadata row and the blank line between the header and the diff body, so a single-hunk edit renders as `✔ Edit: path:LINE ⟨+3 / 1 hunk⟩` immediately followed by the diff. +- Changed the `web_search` tool result rendering: the answer text now shows in full instead of being truncated to a "… N more lines" preview (the `omp q` CLI still caps its compact output), each source renders as a single `title (domain) · age` line with the URL linked on the title (dropping the snippet and bare-URL rows), and the metadata block collapses to one `Provider: @ ()` line plus `Usage:` (removing the redundant Sources/Citations/Request/Queries rows). ### Fixed diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index 4fd08f692..5b90a8a6b 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -171,7 +171,6 @@ let toolExecutionInstanceSeq = 0; export class ToolExecutionComponent extends Container { #contentBox: Box; // Used for custom tools and bash visual truncation #contentText: Text; // For built-in tools (with its own padding/bg) - #leadingSpacer: Spacer; // Blank line above the content; collapsed for self-delimiting framed boxes #multiFileBoxes: (Box | Spacer)[] = []; // Extra boxes for multi-file edit results #imageComponents: Image[] = []; #imageSpacers: Spacer[] = []; @@ -246,8 +245,7 @@ export class ToolExecutionComponent extends Container { this.#cwd = cwd; this.#args = args; - this.#leadingSpacer = new Spacer(1); - this.addChild(this.#leadingSpacer); + this.addChild(new Spacer(1)); // Always create both - contentBox for custom tools/bash/tools with renderers, contentText for other built-ins this.#contentBox = new Box(1, 1, (text: string) => theme.bg("toolPendingBg", text)); @@ -587,8 +585,6 @@ export class ToolExecutionComponent extends Container { this.#renderState.expanded = this.#expanded; this.#renderState.isPartial = this.#isPartial; this.#renderState.spinnerFrame = this.#spinnerFrame; - // Self-delimiting framed boxes don't need the leading blank line for separation. - let hasFramedBlock = false; // Check for custom tool rendering if (this.#tool && (this.#tool.renderCall || this.#tool.renderResult)) { @@ -667,7 +663,6 @@ export class ToolExecutionComponent extends Container { } } setBoxPaddingForFramedBlock(this.#contentBox, contentBoxHasFramedBlock); - hasFramedBlock = contentBoxHasFramedBlock; } else if (this.#toolName in toolRenderers) { // Built-in tools with renderers const renderer = toolRenderers[this.#toolName]; @@ -711,7 +706,6 @@ export class ToolExecutionComponent extends Container { if (resultComponent) { const fileBoxHasFramedBlock = addBoxChild(fileBox, resultComponent); setBoxPaddingForFramedBlock(fileBox, fileBoxHasFramedBlock); - if (fileBoxHasFramedBlock) hasFramedBlock = true; } } catch (err) { logger.warn("Tool renderer failed", { tool: this.#toolName, error: String(err) }); @@ -795,7 +789,6 @@ export class ToolExecutionComponent extends Container { } } setBoxPaddingForFramedBlock(this.#contentBox, contentBoxHasFramedBlock); - hasFramedBlock = contentBoxHasFramedBlock; } } else { // Other built-in tools: use Text directly with caching @@ -803,8 +796,6 @@ export class ToolExecutionComponent extends Container { this.#contentText.setText(this.#formatToolExecution()); } - this.#leadingSpacer.setLines(hasFramedBlock ? 0 : 1); - // Handle images (same for both custom and built-in) for (const img of this.#imageComponents) { this.removeChild(img); diff --git a/packages/coding-agent/src/web/search/render.ts b/packages/coding-agent/src/web/search/render.ts index 48ffe0607..2e802f606 100644 --- a/packages/coding-agent/src/web/search/render.ts +++ b/packages/coding-agent/src/web/search/render.ts @@ -15,23 +15,16 @@ import { formatMoreItems, formatStatusIcon, getDomain, - getPreviewLines, PREVIEW_LIMITS, - TRUNCATE_LENGTHS, + replaceTabs, truncateToWidth, } from "../../tools/render-utils"; -import { renderStatusLine, renderTreeList } from "../../tui"; +import { renderStatusLine, renderTreeList, urlHyperlink } from "../../tui"; import { CachedOutputBlock, markFramedBlockComponent } from "../../tui/output-block"; import { getSearchProviderLabel } from "./provider"; import type { SearchResponse } from "./types"; -const MAX_COLLAPSED_ANSWER_LINES = PREVIEW_LIMITS.COLLAPSED_LINES; -const MAX_SNIPPET_LINES = 2; -const MAX_SNIPPET_LINE_LEN = TRUNCATE_LENGTHS.LINE; const MAX_COLLAPSED_ITEMS = PREVIEW_LIMITS.COLLAPSED_ITEMS; -const MAX_QUERY_PREVIEW = 2; -const MAX_QUERY_LEN = 90; -const MAX_REQUEST_ID_LEN = 36; function renderFallbackText(contentText: string, expanded: boolean, theme: Theme): Component { const lines = contentText.split("\n").filter(line => line.trim()); @@ -66,6 +59,21 @@ export interface SearchRenderDetails { error?: string; } +/** Render a web search failure as a framed error panel, matching the success layout. */ +function renderSearchErrorPanel(message: string, providerLabel: string | undefined, theme: Theme): Component { + const header = renderStatusLine({ icon: "error", title: "Web Search", description: providerLabel }, theme); + const body = theme.fg("error", `Error: ${replaceTabs(message)}`); + const outputBlock = new CachedOutputBlock(); + return markFramedBlockComponent({ + render(width: number): string[] { + return outputBlock.render({ header, state: "error", sections: [{ lines: [body] }], width }, theme); + }, + invalidate() { + outputBlock.invalidate(); + }, + }); +} + /** Render web search result with tree-based layout */ export function renderSearchResult( result: { content: Array<{ type: string; text?: string }>; details?: SearchRenderDetails }, @@ -78,9 +86,12 @@ export function renderSearchResult( ): Component { const details = result.details; - // Handle error case + // Handle error case as a framed panel, matching the success layout. if (details?.error) { - return new Text(theme.fg("error", `Error: ${details.error}`), 0, 0); + const errorProvider = details.response?.provider; + const errorProviderLabel = + errorProvider && errorProvider !== "none" ? getSearchProviderLabel(errorProvider) : undefined; + return renderSearchErrorPanel(details.error, errorProviderLabel, theme); } const rawText = result.content?.find(block => block.type === "text")?.text?.trim() ?? ""; @@ -91,8 +102,6 @@ export function renderSearchResult( const sources = Array.isArray(response.sources) ? response.sources : []; const sourceCount = sources.length; - const citations = Array.isArray(response.citations) ? response.citations : []; - const citationCount = citations.length; const searchQueries = Array.isArray(response.searchQueries) ? response.searchQueries.filter(item => typeof item === "string") : []; @@ -118,16 +127,11 @@ export function renderSearchResult( theme, ); - const metaLines: string[] = []; - metaLines.push(`${theme.fg("muted", "Provider:")} ${theme.fg("text", providerLabel)}`); - if (response.authMode) - metaLines.push( - `${theme.fg("muted", "Auth:")} ${theme.fg("text", response.authMode === "oauth" ? "OAuth" : response.authMode === "api_key" ? "API key" : response.authMode)}`, - ); - if (response.model) metaLines.push(`${theme.fg("muted", "Model:")} ${theme.fg("text", response.model)}`); - metaLines.push(`${theme.fg("muted", "Sources:")} ${theme.fg("text", String(sourceCount))}`); - if (citationCount > 0) - metaLines.push(`${theme.fg("muted", "Citations:")} ${theme.fg("text", String(citationCount))}`); + const authShort = + response.authMode === "oauth" ? "OAuth" : response.authMode === "api_key" ? "API" : response.authMode; + let providerInfo = response.model ? `${response.model} @ ${providerLabel}` : providerLabel; + if (authShort) providerInfo += ` (${authShort})`; + const metaLines: string[] = [`${theme.fg("muted", "Provider:")} ${theme.fg("text", providerInfo)}`]; if (response.usage) { const usageParts: string[] = []; if (response.usage.inputTokens !== undefined) usageParts.push(`in ${response.usage.inputTokens}`); @@ -137,17 +141,6 @@ export function renderSearchResult( if (usageParts.length > 0) metaLines.push(`${theme.fg("muted", "Usage:")} ${theme.fg("text", usageParts.join(theme.sep.dot))}`); } - if (response.requestId) { - metaLines.push( - `${theme.fg("muted", "Request:")} ${theme.fg("text", truncateToWidth(response.requestId, MAX_REQUEST_ID_LEN))}`, - ); - } - if (searchQueries.length > 0) { - const queriesPreview = searchQueries.slice(0, MAX_QUERY_PREVIEW); - const queryList = queriesPreview.map(q => truncateToWidth(q, MAX_QUERY_LEN)); - const suffix = searchQueries.length > queriesPreview.length ? "…" : ""; - metaLines.push(`${theme.fg("muted", "Queries:")} ${theme.fg("text", queryList.join("; "))}${suffix}`); - } const answerMarkdown = contentText ? new Markdown(contentText, 0, 0, getMarkdownTheme()) : undefined; const outputBlock = new CachedOutputBlock(); @@ -163,15 +156,15 @@ export function renderSearchResult( let answerLines: string[]; if (renderedAnswer.length === 0) { answerLines = [theme.fg("muted", "No answer text returned")]; - } else if (expanded) { - answerLines = renderedAnswer; - } else { - const collapsedCap = args?.maxAnswerLines ?? MAX_COLLAPSED_ANSWER_LINES; - answerLines = renderedAnswer.slice(0, collapsedCap); + } else if (args?.maxAnswerLines !== undefined && !expanded) { + // CLI compact mode (`omp q`) caps the answer; the TUI passes no cap and shows it in full. + answerLines = renderedAnswer.slice(0, args.maxAnswerLines); const remaining = renderedAnswer.length - answerLines.length; if (remaining > 0) { answerLines.push(theme.fg("muted", formatMoreItems(remaining, "line"))); } + } else { + answerLines = renderedAnswer; } const sourceTree = renderTreeList( @@ -187,30 +180,22 @@ export function renderSearchResult( : typeof src.url === "string" && src.url.trim() ? src.url : "Untitled"; - const title = truncateToWidth(titleText, MAX_SNIPPET_LINE_LEN); const url = typeof src.url === "string" ? src.url : ""; const domain = url ? getDomain(url) : ""; const age = formatAge(src.ageSeconds) || (typeof src.publishedDate === "string" ? src.publishedDate : ""); const metaParts: string[] = []; if (domain) metaParts.push(theme.fg("dim", `(${domain})`)); - if (typeof src.author === "string" && src.author.trim()) - metaParts.push(theme.fg("muted", truncateToWidth(src.author.trim(), 40))); if (age) metaParts.push(theme.fg("muted", age)); const metaSep = theme.fg("dim", theme.sep.dot); const metaSuffix = metaParts.length > 0 ? ` ${metaParts.join(metaSep)}` : ""; - const srcLines: string[] = [ - truncateToWidth(`${theme.fg("accent", title)}${metaSuffix}`, MAX_SNIPPET_LINE_LEN), - ]; - const snippetText = typeof src.snippet === "string" ? src.snippet : ""; - if (snippetText.trim()) { - const snippetLines = getPreviewLines(snippetText, MAX_SNIPPET_LINES, MAX_SNIPPET_LINE_LEN); - for (const snippetLine of snippetLines) { - srcLines.push(theme.fg("muted", `${theme.format.dash} ${snippetLine}`)); - } - } - if (url) srcLines.push(theme.fg("mdLinkUrl", truncateToWidth(url, MAX_SNIPPET_LINE_LEN))); - return srcLines; + // One line per source: the title links to its URL, followed by domain · age. + // Reserve room for the box borders, the tree branch, and the meta suffix. + const lineBudget = Math.max(24, width - 6); + const titleBudget = Math.max(12, lineBudget - Bun.stringWidth(metaSuffix)); + const title = theme.fg("accent", truncateToWidth(titleText, titleBudget)); + const linkedTitle = url ? urlHyperlink(url, title) : title; + return [`${linkedTitle}${metaSuffix}`]; }, }, theme, From c642232266794d38dd8d48f24cd83dcd860dc7b3 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 19:02:16 +0200 Subject: [PATCH 011/181] ux(coding-agent/tools): improved tool error rendering with subordinate detail lines - Added sanitizeErrorText in render-utils to normalize and truncate tool error messages. - Introduced formatErrorDetail for indented subordinate error text without redundant icon or Error prefix. - Updated goal and write tool renderers to use the new detail formatter, with write now handling isError results via a status header plus detail line. --- .../coding-agent/src/goals/tools/goal-tool.ts | 4 ++-- .../coding-agent/src/tools/render-utils.ts | 20 ++++++++++++++++--- packages/coding-agent/src/tools/write.ts | 12 ++++++++++- 3 files changed, 30 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/src/goals/tools/goal-tool.ts b/packages/coding-agent/src/goals/tools/goal-tool.ts index 1c94cb54f..539597945 100644 --- a/packages/coding-agent/src/goals/tools/goal-tool.ts +++ b/packages/coding-agent/src/goals/tools/goal-tool.ts @@ -8,7 +8,7 @@ import type { Theme, ThemeColor } from "../../modes/theme/theme"; import goalDescription from "../../prompts/tools/goal.md" with { type: "text" }; import { formatDuration } from "../../slash-commands/helpers/format"; import type { ToolSession } from "../../tools"; -import { formatErrorMessage, TRUNCATE_LENGTHS } from "../../tools/render-utils"; +import { formatErrorDetail, TRUNCATE_LENGTHS } from "../../tools/render-utils"; import { ToolError } from "../../tools/tool-errors"; import { renderStatusLine, truncateToWidth } from "../../tui"; import { completionBudgetReport, remainingTokens } from "../runtime"; @@ -190,7 +190,7 @@ export const goalToolRenderer = { if (result.isError) { const header = renderStatusLine({ icon: "error", title: "Goal", description }, uiTheme); - const body = formatErrorMessage(fallbackText || "Goal tool failed", uiTheme); + const body = formatErrorDetail(fallbackText || "Goal tool failed", uiTheme); return new Text([header, body].join("\n"), 0, 0); } diff --git a/packages/coding-agent/src/tools/render-utils.ts b/packages/coding-agent/src/tools/render-utils.ts index b44033f4e..b96931dae 100644 --- a/packages/coding-agent/src/tools/render-utils.ts +++ b/packages/coding-agent/src/tools/render-utils.ts @@ -221,10 +221,24 @@ export function formatMeta(meta: string[], theme: Theme): string { return meta.length > 0 ? ` ${theme.fg("muted", meta.join(theme.sep.dot))}` : ""; } -export function formatErrorMessage(message: string | undefined, theme: Theme): string { +function sanitizeErrorText(message: string | undefined): string { const clean = (message ?? "").replace(/^Error:\s*/, "").trim(); - const safe = clean ? replaceTabs(truncateToWidth(clean, TRUNCATE_LENGTHS.LINE)) : "Unknown error"; - return `${theme.styledSymbol("status.error", "error")} ${theme.fg("error", `Error: ${safe}`)}`; + return clean ? replaceTabs(truncateToWidth(clean, TRUNCATE_LENGTHS.LINE)) : "Unknown error"; +} + +export function formatErrorMessage(message: string | undefined, theme: Theme): string { + return `${theme.styledSymbol("status.error", "error")} ${theme.fg("error", `Error: ${sanitizeErrorText(message)}`)}`; +} + +/** + * Error message rendered as a subordinate detail line beneath a status header + * that already carries the error icon (e.g. `✘ Write: `). The header's + * icon already signals failure, so this omits the redundant error symbol and + * "Error:" prefix that `formatErrorMessage` adds for standalone single-line + * errors, indenting two columns to sit under the header title instead. + */ +export function formatErrorDetail(message: string | undefined, theme: Theme): string { + return ` ${theme.fg("error", sanitizeErrorText(message))}`; } export function formatEmptyMessage(message: string, theme: Theme): string { diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index 3b0115498..4c1d92af4 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -37,6 +37,7 @@ import { formatPathRelativeToCwd, isInternalUrlPath } from "./path-utils"; import { enforcePlanModeWrite, resolvePlanPath } from "./plan-mode-guard"; import { formatDiagnostics, + formatErrorDetail, formatExpandHint, formatMoreItems, formatStatusIcon, @@ -1021,7 +1022,7 @@ export const writeToolRenderer = { }, renderResult( - result: { content: Array<{ type: string; text?: string }>; details?: WriteToolDetails }, + result: { content: Array<{ type: string; text?: string }>; details?: WriteToolDetails; isError?: boolean }, options: RenderResultOptions, uiTheme: Theme, args?: WriteRenderArgs, @@ -1032,6 +1033,15 @@ export const writeToolRenderer = { const lang = getLanguageFromPath(rawPath); const langIcon = uiTheme.fg("muted", uiTheme.getLangIcon(lang)); const pathDisplay = filePath ? uiTheme.fg("accent", filePath) : uiTheme.fg("toolOutput", "…"); + + if (result.isError) { + const errorText = result.content?.find(c => c.type === "text")?.text ?? ""; + const errorHeader = renderStatusLine( + { icon: "error", title: "Write", description: `${langIcon} ${pathDisplay}` }, + uiTheme, + ); + return new Text(`${errorHeader}\n${formatErrorDetail(errorText, uiTheme)}`, 0, 0); + } const lineCount = countLines(fileContent); const lineSuffix = formatLineCountSuffix(lineCount, uiTheme); const execSuffix = result.details?.madeExecutable From ead5cf687105580492ad75104fd86a7748e93e64 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 19:05:20 +0200 Subject: [PATCH 012/181] fix(coding-agent/scripts): fixed CLI startup by launching through a bunfig-free shim script - Updated `install:dev` to symlink `packages/coding-agent/scripts/dev-launch` into Bun's global bin directory as `omp`. - Added a `dev-launch` shell script that launches Bun from an isolated directory and preserves the caller's working directory for restoration. - Added a preload shim that restores `OMP_LAUNCH_CWD` before CLI execution so external project `bunfig.toml` preloads are not used. --- package.json | 2 +- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/scripts/dev-launch | 38 +++++++++++++++++++ .../scripts/dev-launch-preload.ts | 19 ++++++++++ 4 files changed, 59 insertions(+), 1 deletion(-) create mode 100755 packages/coding-agent/scripts/dev-launch create mode 100644 packages/coding-agent/scripts/dev-launch-preload.ts diff --git a/package.json b/package.json index a936a6642..542faa00e 100644 --- a/package.json +++ b/package.json @@ -85,7 +85,7 @@ }, "overrides": {}, "scripts": { - "install:dev": "bun install && bun --cwd=packages/coding-agent link && bun --cwd=packages/ai link", + "install:dev": "bun install && bun --cwd=packages/coding-agent link && ln -sfn \"$(pwd)/packages/coding-agent/scripts/dev-launch\" \"$(bun pm -g bin)/omp\"", "dev": "bun --cwd=packages/coding-agent src/cli.ts", "stats": "bun --cwd=packages/coding-agent src/cli.ts stats", "claude:trace": "bun scripts/claude-trace.ts", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3dc36cc36..735d4b643 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -20,6 +20,7 @@ - Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice. - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. - Fixed custom-rendered tools that set `mergeCallAndResult` (e.g. `lsp`) rendering a redundant tool-name line above the framed result once a result arrived. `ToolExecutionComponent`'s custom-tool branch now emits the fallback label only when the tool has no `renderCall` and the call is not suppressed by an existing result, matching the built-in renderer branch. +- Fixed the `write` tool result rendering with a green success checkmark even when the write failed. `writeToolRenderer.renderResult` now branches on `result.isError`, rendering the error status icon plus the failure message instead of the success header and content preview. ## [15.9.67] - 2026-06-06 ### Added diff --git a/packages/coding-agent/scripts/dev-launch b/packages/coding-agent/scripts/dev-launch new file mode 100755 index 000000000..519ccff0f --- /dev/null +++ b/packages/coding-agent/scripts/dev-launch @@ -0,0 +1,38 @@ +#!/bin/sh +# Dev launcher for the omp CLI, installed by `bun run install:dev`. +# +# Problem it solves: Bun reads `bunfig.toml` from the *current working +# directory* at startup and evaluates its `preload` entries before running the +# script. A bun-shebang bin (what `bun link` creates for `src/cli.ts`) +# therefore inherits whatever `preload` the directory you happen to be in +# declares. Running `omp`/`pi` inside an unrelated Bun project can execute — and +# crash on — that project's preload, e.g. +# error: Cannot find module '@v12sh/utils/frontmatter' from '.../loader.ts' +# +# Bun only reads the *exact* cwd (it does not walk parents) and ignores +# `--config`/`BUN_BE_BUN` for this, so the fix is to launch Bun from an empty, +# bunfig-free directory and restore the real cwd inside the process via the +# preload shim alongside this file. +set -e + +# Resolve this script's real location even when invoked through a symlink +# (`$HOME/.bun/bin/omp` -> this file). +self=$0 +while [ -L "$self" ]; do + link=$(readlink "$self") + case $link in + /*) self=$link ;; + *) self=$(dirname "$self")/$link ;; + esac +done +scripts_dir=$(CDPATH= cd -- "$(dirname -- "$self")" && pwd -P) +cli=$scripts_dir/../src/cli.ts +preload=$scripts_dir/dev-launch-preload.ts + +launch_dir=${OMP_DEV_LAUNCH_DIR:-${HOME}/.omp/.dev-cwd} +mkdir -p "$launch_dir" + +OMP_LAUNCH_CWD=$PWD +export OMP_LAUNCH_CWD +cd "$launch_dir" +exec bun --preload "$preload" "$cli" "$@" diff --git a/packages/coding-agent/scripts/dev-launch-preload.ts b/packages/coding-agent/scripts/dev-launch-preload.ts new file mode 100644 index 000000000..5cafa09ad --- /dev/null +++ b/packages/coding-agent/scripts/dev-launch-preload.ts @@ -0,0 +1,19 @@ +/** + * Bun `--preload` shim for the omp dev launcher (`scripts/dev-launch`). + * + * The launcher starts Bun from an empty, bunfig-free directory so a foreign + * project's `bunfig.toml` `preload` cannot run inside the omp CLI: Bun reads + * `bunfig.toml` from the *current working directory* on startup and evaluates + * its `preload` entries before the entrypoint, so a bun-shebang bin inherits + * whatever `preload` the directory you launched from declares (and crashes if + * that preload can't resolve). This shim is loaded before the entrypoint's + * imports run, so it restores the user's real working directory in time for + * import-time snapshots (e.g. `getProjectDir()` in `@oh-my-pi/pi-utils/dirs`). + */ +const launchCwd = process.env.OMP_LAUNCH_CWD; +if (launchCwd) { + delete process.env.OMP_LAUNCH_CWD; + try { + process.chdir(launchCwd); + } catch {} +} From 2887eec37385522fc1b91260f5e8fa49b5b54681 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 19:05:53 +0200 Subject: [PATCH 013/181] feat(cli): added PNG screenshot support for the gallery CLI command - Added `omp gallery --screenshot`, `--out`, `--font`, and `--font-size` flags. - Added a VHS-based screenshot path that captures gallery output as PNG file(s). - Added chunking and naming logic to split tall galleries into multiple numbered captures. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/cli/gallery-cli.ts | 88 ++++-- .../src/cli/gallery-screenshot.ts | 279 ++++++++++++++++++ packages/coding-agent/src/commands/gallery.ts | 15 + 4 files changed, 360 insertions(+), 23 deletions(-) create mode 100644 packages/coding-agent/src/cli/gallery-screenshot.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 735d4b643..794ec90fc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,7 @@ - Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states - Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output - Added `omp gallery` fidelity for tools whose renderers are attached on the tool instance (`lsp`, `task`): the gallery now drives them through the same custom-tool render branch production uses, so regressions in that path surface in the gallery rather than only in a live session. +- Added `omp gallery --screenshot`, which renders the gallery through a real virtual terminal (VHS) and writes PNG screenshot(s) instead of ANSI, so agents (and anything that can only read raw bytes) can actually see the rendered output. The capture forces truecolor and matches the active theme/symbol preset; tall galleries split across multiple images (whole renderers are never cut). Tune with `--out`, `--font`, and `--font-size`; requires `vhs` on `PATH` and fails with install guidance when absent. - Added `app.display.reset`, bound to `Ctrl+L` by default, to force an immediate terminal display reset/redraw without resizing the window. ### Changed diff --git a/packages/coding-agent/src/cli/gallery-cli.ts b/packages/coding-agent/src/cli/gallery-cli.ts index 921e571db..7f52ec956 100644 --- a/packages/coding-agent/src/cli/gallery-cli.ts +++ b/packages/coding-agent/src/cli/gallery-cli.ts @@ -15,6 +15,7 @@ import { ToolExecutionComponent } from "../modes/components/tool-execution"; import { initTheme, theme } from "../modes/theme/theme"; import { toolRenderers } from "../tools/renderers"; import { type GalleryFixture, type GalleryResult, galleryFixtures } from "./gallery-fixtures"; +import { captureGalleryScreenshots } from "./gallery-screenshot"; /** Lifecycle states the gallery renders, in display order. */ export const GALLERY_STATES = ["streaming", "progress", "success", "error"] as const; @@ -38,6 +39,20 @@ export interface GalleryCommandArgs { expanded?: boolean; /** Strip ANSI styling from the output (useful when redirecting to a file). */ plain?: boolean; + /** Capture the rendered gallery as PNG screenshot(s) via VHS instead of printing ANSI. */ + screenshot?: boolean; + /** Screenshot output path (single image) or base path (suffixed when split across images). */ + out?: string; + /** Font family for screenshots (must be installed; Nerd Font recommended for icon glyphs). */ + font?: string; + /** Font size in points for screenshots. */ + fontSize?: number; +} + +/** One tool's rendered lifecycle, as ANSI lines: a leading blank, the section rule, then each state. */ +export interface GallerySection { + heading: string; + lines: string[]; } const GENERIC_ERROR: GalleryResult = { @@ -130,11 +145,45 @@ function sectionRule(label: string, width: number): string { } /** - * Render the gallery to stdout. Iterates the renderer registry (or a single - * tool), printing each requested lifecycle state under a labeled section. + * Render each requested tool's lifecycle into ANSI section blocks. The block + * layout (leading blank, section rule, then a blank + dim label + body per + * state) is shared by the stdout and screenshot paths so both stay identical. + */ +async function renderGallerySections( + names: string[], + states: GalleryState[], + width: number, + expanded: boolean, +): Promise { + const sections: GallerySection[] = []; + for (const name of names) { + const fixture = resolveFixture(name); + const heading = fixture.label && fixture.label !== name ? `${name} — ${fixture.label}` : name; + const lines: string[] = ["", sectionRule(heading, width)]; + for (const state of states) { + lines.push("", theme.fg("dim", ` · ${STATE_LABELS[state]}`)); + try { + for (const line of await renderGalleryState(name, fixture, state, width, expanded)) lines.push(line); + } catch (err) { + lines.push(theme.fg("error", ` render failed: ${String(err)}`)); + } + } + sections.push({ heading, lines }); + } + return sections; +} + +/** + * Render the gallery. Iterates the renderer registry (or a single tool), + * printing each requested lifecycle state under a labeled section — or, with + * `screenshot`, capturing the rendered output as PNG(s) via VHS. */ export async function runGalleryCommand(args: GalleryCommandArgs): Promise { const settingsInstance = await Settings.init(); + // Screenshots must carry exact theme RGB regardless of how the invoking + // terminal advertises its color support, so force truecolor before the theme + // (and therefore every SGR escape it emits) is built. + if (args.screenshot) process.env.COLORTERM = "truecolor"; await initTheme( false, settingsInstance.get("symbolPreset"), @@ -154,28 +203,21 @@ export async function runGalleryCommand(args: GalleryCommandArgs): Promise return; } - const out: string[] = []; - const push = (line: string) => out.push(args.plain ? Bun.stripANSI(line) : line); + const sections = await renderGallerySections(names, states, width, expanded); - for (const name of names) { - const fixture = resolveFixture(name); - const heading = fixture.label && fixture.label !== name ? `${name} — ${fixture.label}` : name; - push(""); - push(sectionRule(heading, width)); - - for (const state of states) { - push(""); - push(theme.fg("dim", ` · ${STATE_LABELS[state]}`)); - let lines: string[]; - try { - lines = await renderGalleryState(name, fixture, state, width, expanded); - } catch (err) { - lines = [theme.fg("error", ` render failed: ${String(err)}`)]; - } - for (const line of lines) push(line); - } + if (args.screenshot) { + const paths = await captureGalleryScreenshots(sections, { + width, + font: args.font, + fontSize: args.fontSize, + out: args.out, + }); + process.stdout.write(`${paths.join("\n")}\n`); + return; } - push(""); - process.stdout.write(`${out.join("\n")}\n`); + const lines = sections.flatMap(section => section.lines); + lines.push(""); + const text = lines.map(line => (args.plain ? Bun.stripANSI(line) : line)).join("\n"); + process.stdout.write(`${text}\n`); } diff --git a/packages/coding-agent/src/cli/gallery-screenshot.ts b/packages/coding-agent/src/cli/gallery-screenshot.ts new file mode 100644 index 000000000..05ba3eafa --- /dev/null +++ b/packages/coding-agent/src/cli/gallery-screenshot.ts @@ -0,0 +1,279 @@ +/** + * Render `omp gallery` output to PNG screenshots via VHS. + * + * ANSI escapes are invisible to anything that can only read raw bytes (e.g. + * agents), so `--screenshot` drives the rendered gallery through a real virtual + * terminal (VHS + ttyd + ffmpeg) and writes the captured frame to disk. The + * gallery is pre-rendered to truecolor ANSI in this process — where the user's + * theme and symbol preset are correct — then `cat`'d inside VHS so the captured + * pixels match exactly what the live TUI would draw. + * + * VHS is a hard dependency of this path: if it is not installed we fail loudly + * rather than degrade to a lossy fallback. + */ +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { $which } from "@oh-my-pi/pi-utils"; +import { theme } from "../modes/theme/theme"; +import type { GallerySection } from "./gallery-cli"; + +/** Nerd Font family so the gallery's icon glyphs (PUA) render instead of tofu. */ +export const DEFAULT_SCREENSHOT_FONT = "JetBrainsMono Nerd Font"; +export const DEFAULT_SCREENSHOT_FONT_SIZE = 18; + +/** Inner padding (px) VHS leaves around the terminal grid. */ +const PADDING = 14; +const LINE_HEIGHT = 1.0; +/** + * Upper-bound cell metrics relative to font size. Real monospace cells are + * smaller, so over-provisioning the canvas guarantees the gallery never + * soft-wraps (too few columns) or scrolls off the top (too few rows). The slack + * shows up only as a modest background margin, which is harmless for review. + */ +const CELL_WIDTH_RATIO = 0.65; +const CELL_HEIGHT_RATIO = 1.5; +/** Keep each image well under headless-Chromium's tall-canvas limits. */ +const MAX_IMAGE_HEIGHT_PX = 8000; + +export interface GalleryScreenshotOptions { + /** Gallery render width in columns (matches the ANSI line width). */ + width: number; + /** VHS `FontFamily`. */ + font?: string; + /** VHS `FontSize`. */ + fontSize?: number; + /** + * Output destination. When omitted, PNGs land in a fresh temp directory. + * With multiple images the path is suffixed (`name-01.png`, `name-02.png`). + */ + out?: string; +} + +/** + * Capture the gallery sections as one or more PNGs and return their absolute + * paths. Tall galleries are split across images so no single capture exceeds + * the terminal-canvas height limit. + */ +export async function captureGalleryScreenshots( + sections: GallerySection[], + options: GalleryScreenshotOptions, +): Promise { + const vhs = $which("vhs"); + if (!vhs) { + throw new Error( + "`omp gallery --screenshot` requires VHS, which is not installed. " + + "Install it (e.g. `brew install vhs`, or see https://github.com/charmbracelet/vhs) and retry.", + ); + } + + const font = options.font ?? DEFAULT_SCREENSHOT_FONT; + const fontSize = options.fontSize ?? DEFAULT_SCREENSHOT_FONT_SIZE; + const cellHeight = fontSize * Math.max(LINE_HEIGHT, 1) * CELL_HEIGHT_RATIO; + const cellWidth = fontSize * CELL_WIDTH_RATIO; + const rowBudget = Math.max(40, Math.floor((MAX_IMAGE_HEIGHT_PX - 2 * PADDING) / cellHeight) - 2); + const chunks = chunkGallerySections(sections, rowBudget); + const themeJson = buildVhsTheme(); + + const baseDir = options.out + ? path.dirname(path.resolve(options.out)) + : fs.mkdtempSync(path.join(os.tmpdir(), "omp-gallery-")); + await fs.promises.mkdir(baseDir, { recursive: true }); + + const outPaths: string[] = []; + for (let i = 0; i < chunks.length; i++) { + if (chunks.length > 1) { + process.stderr.write(`Rendering gallery screenshot ${i + 1}/${chunks.length}…\n`); + } + const outPng = resolveScreenshotOutputPath(options.out, baseDir, i, chunks.length); + const lines = chunks[i].flatMap(section => section.lines); + await renderChunk({ vhs, lines, outPng, font, fontSize, cellWidth, cellHeight, width: options.width, themeJson }); + outPaths.push(outPng); + } + return outPaths; +} + +interface RenderChunkArgs { + vhs: string; + lines: string[]; + outPng: string; + font: string; + fontSize: number; + cellWidth: number; + cellHeight: number; + width: number; + themeJson: string; +} + +async function renderChunk(args: RenderChunkArgs): Promise { + const rows = args.lines.length; + const widthPx = Math.ceil(args.width * args.cellWidth) + 2 * PADDING; + const heightPx = Math.ceil((rows + 2) * args.cellHeight) + 2 * PADDING; + + const dir = path.dirname(args.outPng); + const stem = path.basename(args.outPng, path.extname(args.outPng)); + const ansiPath = path.join(dir, `.${stem}.ansi`); + const tapePath = path.join(dir, `.${stem}.tape`); + const gifPath = path.join(dir, `.${stem}.gif`); + + // CRLF so each gallery line is its own terminal row regardless of how the + // captured shell handles bare LF. + await Bun.write(ansiPath, `${args.lines.join("\r\n")}\r\n`); + await Bun.write( + tapePath, + buildTape({ + gifPath, + outPng: args.outPng, + ansiPath, + widthPx, + heightPx, + font: args.font, + fontSize: args.fontSize, + themeJson: args.themeJson, + }), + ); + + try { + const result = await Bun.$`${args.vhs} ${tapePath}`.quiet().nothrow(); + if (result.exitCode !== 0 || !(await Bun.file(args.outPng).exists())) { + const detail = result.stderr.toString().trim() || result.stdout.toString().trim(); + throw new Error(`VHS failed to render the gallery screenshot${detail ? `: ${detail.slice(-600)}` : ""}`); + } + } finally { + await Promise.all([ + fs.promises.rm(ansiPath, { force: true }), + fs.promises.rm(tapePath, { force: true }), + fs.promises.rm(gifPath, { force: true }), + ]); + } +} + +interface TapeArgs { + gifPath: string; + outPng: string; + ansiPath: string; + widthPx: number; + heightPx: number; + font: string; + fontSize: number; + themeJson: string; +} + +function buildTape(args: TapeArgs): string { + // `Output` (a throwaway GIF) is mandatory for VHS to record; the screenshot + // is captured from the final visible frame. Setup is hidden so the typed + // `cat` command and shell prompt never appear in the capture, and a trailing + // `sleep` keeps the shell from drawing a fresh prompt under the output. + const shellCommand = `clear; cat ${shellSingleQuote(args.ansiPath)}; sleep 120`; + return `${[ + `Output ${JSON.stringify(args.gifPath)}`, + `Set Width ${args.widthPx}`, + `Set Height ${args.heightPx}`, + `Set FontFamily ${JSON.stringify(args.font)}`, + `Set FontSize ${args.fontSize}`, + `Set Padding ${PADDING}`, + `Set LineHeight ${LINE_HEIGHT}`, + `Set Theme ${args.themeJson}`, + "Hide", + `Type ${JSON.stringify(shellCommand)}`, + "Enter", + "Sleep 1.2s", + "Show", + "Sleep 400ms", + `Screenshot ${JSON.stringify(args.outPng)}`, + ].join("\n")}\n`; +} + +/** + * Build the VHS terminal theme. Only background/foreground/cursor matter: the + * gallery emits truecolor (`38;2`/`48;2`) escapes, so the 16-color palette is + * never consulted — it is filler to satisfy VHS's theme schema. + */ +function buildVhsTheme(): string { + const background = parseAnsiRgb(theme.getBgAnsi("statusLineBg")) ?? (theme.isLight ? "#ffffff" : "#1a1a1a"); + const foreground = theme.isLight ? "#1a1a1a" : "#d4d4d4"; + const selection = theme.isLight ? "#c8d6ff" : "#404862"; + return JSON.stringify({ + name: "omp-gallery", + background, + foreground, + cursor: foreground, + selection, + black: "#000000", + red: "#ff5555", + green: "#50fa7b", + yellow: "#f1fa8c", + blue: "#6272ff", + magenta: "#ff79c6", + cyan: "#8be9fd", + white: "#bfbfbf", + brightBlack: "#4d4d4d", + brightRed: "#ff6e6e", + brightGreen: "#69ff94", + brightYellow: "#ffffa5", + brightBlue: "#8aa0ff", + brightMagenta: "#ff92df", + brightCyan: "#a4ffff", + brightWhite: "#ffffff", + }); +} + +/** Extract `#rrggbb` from a truecolor SGR escape (`…38;2;r;g;b…` / `…48;2;…`). */ +function parseAnsiRgb(ansi: string): string | undefined { + const match = /[34]8;2;(\d+);(\d+);(\d+)/.exec(ansi); + if (!match) return undefined; + const hex = (value: string) => Number(value).toString(16).padStart(2, "0"); + return `#${hex(match[1])}${hex(match[2])}${hex(match[3])}`; +} + +/** POSIX single-quote a path for embedding in the VHS shell command. */ +function shellSingleQuote(value: string): string { + return `'${value.replace(/'/g, `'\\''`)}'`; +} + +/** + * Resolve a chunk's PNG path. A single image keeps the bare name (or the exact + * `out`); multiple images gain a zero-padded `-NN` suffix so they sort and never + * collide. + */ +export function resolveScreenshotOutputPath( + out: string | undefined, + baseDir: string, + index: number, + total: number, +): string { + if (total === 1) { + return out ? path.resolve(out) : path.join(baseDir, "gallery.png"); + } + const suffix = String(index + 1).padStart(2, "0"); + if (out) { + const resolved = path.resolve(out); + const ext = path.extname(resolved) || ".png"; + const stem = path.basename(resolved, ext); + return path.join(path.dirname(resolved), `${stem}-${suffix}${ext}`); + } + return path.join(baseDir, `gallery-${suffix}.png`); +} + +/** + * Group whole tool sections into chunks that stay under `rowBudget` rows. A + * single section larger than the budget gets its own (taller) image rather than + * being split mid-renderer. + */ +export function chunkGallerySections(sections: GallerySection[], rowBudget: number): GallerySection[][] { + const chunks: GallerySection[][] = []; + let current: GallerySection[] = []; + let currentRows = 0; + for (const section of sections) { + const rows = section.lines.length; + if (current.length > 0 && currentRows + rows > rowBudget) { + chunks.push(current); + current = []; + currentRows = 0; + } + current.push(section); + currentRows += rows; + } + if (current.length > 0) chunks.push(current); + return chunks.length > 0 ? chunks : [[]]; +} diff --git a/packages/coding-agent/src/commands/gallery.ts b/packages/coding-agent/src/commands/gallery.ts index ee5b69cba..d0b878fac 100644 --- a/packages/coding-agent/src/commands/gallery.ts +++ b/packages/coding-agent/src/commands/gallery.ts @@ -22,6 +22,17 @@ export default class Gallery extends Command { default: false, }), plain: Flags.boolean({ description: "Strip ANSI styling from the output", default: false }), + screenshot: Flags.boolean({ + description: + "Capture the rendered output as PNG screenshot(s) via VHS instead of printing ANSI (requires vhs)", + default: false, + }), + out: Flags.string({ + char: "o", + description: "Screenshot output path (with --screenshot); suffixed per image when split across multiple", + }), + font: Flags.string({ description: "Screenshot font family (default: JetBrainsMono Nerd Font)" }), + "font-size": Flags.integer({ description: "Screenshot font size in points (default: 18)" }), }; async run(): Promise { @@ -32,6 +43,10 @@ export default class Gallery extends Command { width: flags.width, expanded: flags.expanded, plain: flags.plain, + screenshot: flags.screenshot, + out: flags.out, + font: flags.font, + fontSize: flags["font-size"], }); } } From adcb8793b2b09239d6fa4bb65777a2eecc3e07e6 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 17:10:18 +0000 Subject: [PATCH 014/181] fix(tui): gated deccara fills on sync output Prevented DECCARA background-fill optimization from shortening rows unless the active TUI paint is protected by synchronized output, preserving padded background bytes when sync output is disabled. Added regression coverage for the synchronized-output opt-out path and kept existing DECCARA tests forced onto synchronized output so the optimized path remains covered. Fixes #2000 --- packages/tui/CHANGELOG.md | 4 ++ packages/tui/src/tui.ts | 16 +++++- packages/tui/test/deccara.test.ts | 91 +++++++++++++++++++++++++++++++ 3 files changed, 108 insertions(+), 3 deletions(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index c023618be..40abd7bb6 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -7,6 +7,10 @@ - Added `TUI.resetDisplay()` to force an immediate full-frame replay, including native scrollback when the host can safely clear it. - Added `setPaddingY` to `Box` so vertical padding can be updated programmatically after creation. +### Fixed + +- Fixed DECCARA background-fill optimization running when synchronized output is disabled, which could expose default-background gaps during rapidly updating tool-use panels ([#2000](https://github.com/can1357/oh-my-pi/issues/2000)). + ## [15.9.67] - 2026-06-06 ### Added diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index d694d1af9..c568d09ea 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -571,6 +571,11 @@ export class TUI extends Container { get synchronizedOutput(): boolean { return this.#synchronizedOutputEnabled; } + #deccaraFillsEnabled(): boolean { + // DECCARA fill rectangles arrive after shortened row text; synchronized + // output hides that intermediate default-background state from users. + return TERMINAL.deccara && this.#synchronizedOutputEnabled; + } /** * When enabled, live render frames rebuild native scrollback on offscreen and @@ -2404,7 +2409,7 @@ export class TUI extends Container { const visibleStart = Math.max(0, lines.length - height); let fillSequence = ""; let visibleTexts: string[] | null = null; - if (TERMINAL.deccara && visibleStart < lines.length) { + if (this.#deccaraFillsEnabled() && visibleStart < lines.length) { const visible: string[] = new Array(lines.length - visibleStart); for (let k = 0; k < visible.length; k++) { visible[k] = this.#fitLineToWidth(lines[visibleStart + k], width); @@ -2504,7 +2509,7 @@ export class TUI extends Container { for (let screenRow = 0; screenRow < height; screenRow++) { visible[screenRow] = this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width); } - const { texts, sequence } = TERMINAL.deccara + const { texts, sequence } = this.#deccaraFillsEnabled() ? planDeccaraFills(visible, width) : { texts: visible, sequence: "" }; let buffer = `${this.#paintBeginSequence}\x1b[H`; @@ -2814,7 +2819,12 @@ export class TUI extends Container { const fillStart = Math.max(firstChanged, fillViewportTop); let fillSequence = ""; let fillTexts: string[] | null = null; - if (TERMINAL.deccara && !appendStart && moveTargetRow <= prevViewportBottom && renderEnd >= fillStart) { + if ( + this.#deccaraFillsEnabled() && + !appendStart && + moveTargetRow <= prevViewportBottom && + renderEnd >= fillStart + ) { const slice: string[] = new Array(renderEnd - fillStart + 1); for (let i = fillStart; i <= renderEnd; i++) { slice[i - fillStart] = this.#fitLineToWidth(lines[i], width); diff --git a/packages/tui/test/deccara.test.ts b/packages/tui/test/deccara.test.ts index d5dd82e48..f1d0329f2 100644 --- a/packages/tui/test/deccara.test.ts +++ b/packages/tui/test/deccara.test.ts @@ -74,6 +74,41 @@ function countOccurrences(haystack: string, needle: string): number { } } +async function withEnvPatch(patch: Record, run: () => T | Promise): Promise { + const bunSnapshot: Record = {}; + const processSnapshot: Record = {}; + for (const key in patch) { + bunSnapshot[key] = Bun.env[key]; + processSnapshot[key] = process.env[key]; + const value = patch[key]; + if (value === undefined) { + delete Bun.env[key]; + delete process.env[key]; + } else { + Bun.env[key] = value; + process.env[key] = value; + } + } + try { + return await run(); + } finally { + for (const key in patch) { + const bunValue = bunSnapshot[key]; + if (bunValue === undefined) { + delete Bun.env[key]; + } else { + Bun.env[key] = bunValue; + } + const processValue = processSnapshot[key]; + if (processValue === undefined) { + delete process.env[key]; + } else { + process.env[key] = processValue; + } + } + } +} + describe("detectRectangularSgrSupport", () => { it("enables only kitty, which implements the SGR-background extension", () => { expect(detectRectangularSgrSupport("kitty", {})).toBe(true); @@ -240,8 +275,17 @@ describe("planDeccaraFills", () => { describe("TUI DECCARA integration", () => { const savedDeccara = TERMINAL.deccara; + const savedForceSyncOutput = Bun.env.PI_FORCE_SYNC_OUTPUT; + const savedNoSyncOutput = Bun.env.PI_NO_SYNC_OUTPUT; + const savedTuiSyncOutput = Bun.env.PI_TUI_SYNC_OUTPUT; beforeEach(() => { + Bun.env.PI_FORCE_SYNC_OUTPUT = "1"; + process.env.PI_FORCE_SYNC_OUTPUT = "1"; + delete Bun.env.PI_NO_SYNC_OUTPUT; + delete process.env.PI_NO_SYNC_OUTPUT; + delete Bun.env.PI_TUI_SYNC_OUTPUT; + delete process.env.PI_TUI_SYNC_OUTPUT; let monotonic = 0; vi.spyOn(performance, "now").mockImplementation(() => { monotonic += 20; @@ -250,6 +294,27 @@ describe("TUI DECCARA integration", () => { }); afterEach(() => { + if (savedForceSyncOutput === undefined) { + delete Bun.env.PI_FORCE_SYNC_OUTPUT; + delete process.env.PI_FORCE_SYNC_OUTPUT; + } else { + Bun.env.PI_FORCE_SYNC_OUTPUT = savedForceSyncOutput; + process.env.PI_FORCE_SYNC_OUTPUT = savedForceSyncOutput; + } + if (savedNoSyncOutput === undefined) { + delete Bun.env.PI_NO_SYNC_OUTPUT; + delete process.env.PI_NO_SYNC_OUTPUT; + } else { + Bun.env.PI_NO_SYNC_OUTPUT = savedNoSyncOutput; + process.env.PI_NO_SYNC_OUTPUT = savedNoSyncOutput; + } + if (savedTuiSyncOutput === undefined) { + delete Bun.env.PI_TUI_SYNC_OUTPUT; + delete process.env.PI_TUI_SYNC_OUTPUT; + } else { + Bun.env.PI_TUI_SYNC_OUTPUT = savedTuiSyncOutput; + process.env.PI_TUI_SYNC_OUTPUT = savedTuiSyncOutput; + } setTerminalDeccara(savedDeccara); vi.restoreAllMocks(); }); @@ -299,6 +364,32 @@ describe("TUI DECCARA integration", () => { } }); + it("keeps padded fallback bytes when synchronized output is disabled", async () => { + await withEnvPatch( + { PI_NO_SYNC_OUTPUT: "1", PI_FORCE_SYNC_OUTPUT: undefined, PI_TUI_SYNC_OUTPUT: undefined }, + async () => { + setTerminalDeccara(true); + const term = new VirtualTerminal(40, 8); + const tui = new TUI(term); + tui.addChild(new BgPanelComponent(["", "", "", ""])); + const writes = captureWrites(term); + + try { + tui.start(); + await settle(term); + const out = writes.join(""); + + expect(out).not.toContain("$r"); + expect(out).not.toContain(DECSACE_RECT); + expect(out).toContain(`${BG_OPEN}${" ".repeat(40)}`); + expect(term.getViewportRowBackgroundColumns(0)).toHaveLength(40); + } finally { + tui.stop(); + } + }, + ); + }); + it("preserves viewport text identically whether DECCARA is on or off", async () => { const rowsContent = ["", "Hello", "", "World", ""]; const trimmed = (term: VirtualTerminal) => term.getViewport().map(line => line.trimEnd()); From c43eac9d9184e7c06f8c523c4412978fc79fcf37 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 17:10:25 +0000 Subject: [PATCH 015/181] style: bun run fix --- packages/coding-agent/test/extensions-runner.test.ts | 1 - 1 file changed, 1 deletion(-) diff --git a/packages/coding-agent/test/extensions-runner.test.ts b/packages/coding-agent/test/extensions-runner.test.ts index 1542897a9..33e188b7f 100644 --- a/packages/coding-agent/test/extensions-runner.test.ts +++ b/packages/coding-agent/test/extensions-runner.test.ts @@ -148,7 +148,6 @@ describe("ExtensionRunner", () => { warnSpy.mockRestore(); }); - it("warns when two extensions register same shortcut", async () => { // Use a non-reserved shortcut const extCode1 = ` From d63a91bf5eafc6256565489e16c09d58ba85b41c Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 19:36:30 +0200 Subject: [PATCH 016/181] fix(coding-agent/web): sent bare query on perplexity OAuth path - Stopped prepending system_prompt to the consumer ask endpoint, which lacks a system slot and refused the meta-instruction. - Kept system_prompt as a proper system message on the API-key path. --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/web/search/providers/perplexity.ts | 8 +++++++- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 794ec90fc..4ac3f4feb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -22,6 +22,7 @@ - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. - Fixed custom-rendered tools that set `mergeCallAndResult` (e.g. `lsp`) rendering a redundant tool-name line above the framed result once a result arrived. `ToolExecutionComponent`'s custom-tool branch now emits the fallback label only when the tool has no `renderCall` and the call is not suppressed by an existing result, matching the built-in renderer branch. - Fixed the `write` tool result rendering with a green success checkmark even when the write failed. `writeToolRenderer.renderResult` now branches on `result.isError`, rendering the error status icon plus the failure message instead of the success header and content preview. +- Fixed Perplexity OAuth/cookie web search returning a refusal answer ("I don't currently have access to the web-search tools in this turn") despite returning real sources. `callPerplexityOAuth` was prepending the API-style `web-search` system prompt to the query (`query_str = systemPrompt + "\n\n" + query`), but the consumer `www.perplexity.ai/rest/sse/perplexity_ask` endpoint has no system-message slot and reads the prepended instruction as a meta-prompt, making the model decline. The OAuth/cookie path now sends the bare query; the API-key path still passes the system prompt as a proper `system` message. ## [15.9.67] - 2026-06-06 ### Added diff --git a/packages/coding-agent/src/web/search/providers/perplexity.ts b/packages/coding-agent/src/web/search/providers/perplexity.ts index 2bc2ac837..e03112a64 100644 --- a/packages/coding-agent/src/web/search/providers/perplexity.ts +++ b/packages/coding-agent/src/web/search/providers/perplexity.ts @@ -334,7 +334,13 @@ async function callPerplexityOAuth( params: PerplexitySearchParams, ): Promise<{ answer: string; sources: SearchSource[]; model?: string; requestId?: string }> { const requestId = crypto.randomUUID(); - const effectiveQuery = params.system_prompt ? `${params.system_prompt}\n\n${params.query}` : params.query; + // The consumer `perplexity_ask` endpoint is itself a research assistant and + // has no system-message slot. Prepending the API-style system prompt to the + // query makes the model read it as a meta-instruction and refuse with + // "I don't have access to web-search tools in this turn", so OAuth/cookie + // searches send the bare query. (The API-key path still uses system_prompt + // as a proper `system` message.) + const effectiveQuery = params.query; const response = await fetch(PERPLEXITY_OAUTH_ASK_URL, { method: "POST", From 8a5b99a9679f311d5d2ffab451c43e64b25bb80e Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 19:42:16 +0200 Subject: [PATCH 017/181] feat(coding-agent): enabled anonymous Perplexity fallback and updated web-search checks - Added anonymous Perplexity authentication mode for unauthenticated web searches. - Switched web-search setup checks to use `isExplicitlyAvailable` and removed key enforcement in doctor. - Updated Perplexity OAuth flow to reuse auth handling for all non-key searches and anonymous responses. - Updated CLI and provider option help text to mark the Perplexity key optional with fallback. --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/cli/args.ts | 2 +- .../src/extensibility/plugins/doctor.ts | 1 - .../modes/setup-wizard/scenes/web-search.ts | 5 +- .../src/web/search/providers/perplexity.ts | 239 ++++++++++++++---- packages/coding-agent/src/web/search/types.ts | 2 +- 6 files changed, 198 insertions(+), 55 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 4ac3f4feb..dc108df82 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,8 +1,10 @@ # Changelog ## [Unreleased] + ### Added +- Added anonymous fallback for Perplexity web search, allowing `web_search` and explicit Perplexity provider usage when no Perplexity credentials are configured - Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states - Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output - Added `omp gallery` fidelity for tools whose renderers are attached on the tool instance (`lsp`, `task`): the gallery now drives them through the same custom-tool render branch production uses, so regressions in that path surface in the gallery rather than only in a live session. @@ -11,12 +13,14 @@ ### Changed +- Changed Perplexity explicit provider availability checks so the setup wizard can mark `perplexity` as available for manual selection without credentials, while auto provider discovery still requires auth - Changed the default persistent model selector shortcut from `Ctrl+L` to `Alt+M`, leaving `Ctrl+L` for display reset. Existing user remaps in `keybindings.yml` are preserved. - Changed the edit tool result header to carry the diff change stats (`+N / -M / K hunks`) inline next to the file path, and removed the redundant lone language-icon metadata row and the blank line between the header and the diff body, so a single-hunk edit renders as `✔ Edit: path:LINE ⟨+3 / 1 hunk⟩` immediately followed by the diff. - Changed the `web_search` tool result rendering: the answer text now shows in full instead of being truncated to a "… N more lines" preview (the `omp q` CLI still caps its compact output), each source renders as a single `title (domain) · age` line with the URL linked on the title (dropping the snippet and bare-URL rows), and the metadata block collapses to one `Provider: @ ()` line plus `Usage:` (removing the redundant Sources/Citations/Request/Queries rows). ### Fixed +- Fixed Perplexity `perplexity_ask` response parsing so OAuth, cookie, and anonymous searches correctly extract answer text and sources from JSON payloads - Fixed framed read results rendering with an extra blank row above and below the output block. - Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice. - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index cefb84942..0f770c0ce 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -284,7 +284,7 @@ export function getExtraHelpText(): string { ${chalk.dim("# Search & Tools")} EXA_API_KEY - Exa web search BRAVE_API_KEY - Brave web search - PERPLEXITY_API_KEY - Perplexity web search (API) + PERPLEXITY_API_KEY - Perplexity web search API key (optional; anonymous fallback) PERPLEXITY_COOKIES - Perplexity web search (session cookie) TAVILY_API_KEY - Tavily web search ANTHROPIC_SEARCH_API_KEY - Anthropic web search (override; isolates search from main ANTHROPIC_API_KEY) diff --git a/packages/coding-agent/src/extensibility/plugins/doctor.ts b/packages/coding-agent/src/extensibility/plugins/doctor.ts index ff33cc562..5bd600c67 100644 --- a/packages/coding-agent/src/extensibility/plugins/doctor.ts +++ b/packages/coding-agent/src/extensibility/plugins/doctor.ts @@ -25,7 +25,6 @@ export async function runDoctorChecks(): Promise { const apiKeys = [ { name: "ANTHROPIC_API_KEY", description: "Anthropic API" }, { name: "OPENAI_API_KEY", description: "OpenAI API" }, - { name: "PERPLEXITY_API_KEY", description: "Perplexity search" }, { name: "EXA_API_KEY", description: "Exa search" }, ]; diff --git a/packages/coding-agent/src/modes/setup-wizard/scenes/web-search.ts b/packages/coding-agent/src/modes/setup-wizard/scenes/web-search.ts index 906803bfe..221da8b4a 100644 --- a/packages/coding-agent/src/modes/setup-wizard/scenes/web-search.ts +++ b/packages/coding-agent/src/modes/setup-wizard/scenes/web-search.ts @@ -19,7 +19,8 @@ type Availability = "checking" | boolean; /** * "Web search" panel: picks the provider the web_search tool should prefer and * reports whether the highlighted provider is ready to use given current - * credentials (env keys or OAuth sign-ins from the Sign in tab). + * credentials (env keys or OAuth sign-ins from the Sign in tab) or an + * unauthenticated fallback. */ export class WebSearchTab implements SetupTab { readonly id = "web-search"; @@ -91,7 +92,7 @@ export class WebSearchTab implements SetupTab { let ready = false; try { const provider = await getSearchProvider(id); - ready = await provider.isAvailable(this.host.ctx.session.modelRegistry.authStorage); + ready = await provider.isExplicitlyAvailable(this.host.ctx.session.modelRegistry.authStorage); } catch { ready = false; } diff --git a/packages/coding-agent/src/web/search/providers/perplexity.ts b/packages/coding-agent/src/web/search/providers/perplexity.ts index e03112a64..92a19865d 100644 --- a/packages/coding-agent/src/web/search/providers/perplexity.ts +++ b/packages/coding-agent/src/web/search/providers/perplexity.ts @@ -1,10 +1,11 @@ /** * Perplexity Web Search Provider * - * Supports three auth modes: + * Supports four auth modes: * - Cookies (`PERPLEXITY_COOKIES`) via `www.perplexity.ai/rest/sse/perplexity_ask` * - OAuth/session bearer via `AuthStorage` and `www.perplexity.ai/rest/sse/perplexity_ask` * - API key (`PERPLEXITY_API_KEY`) via `api.perplexity.ai/chat/completions` + * - Anonymous via `www.perplexity.ai/rest/sse/perplexity_ask` */ import { type AuthStorage, getEnvApiKey } from "@oh-my-pi/pi-ai"; @@ -32,6 +33,8 @@ const DEFAULT_NUM_SEARCH_RESULTS = 20; const OAUTH_EXPIRY_BUFFER_MS = 5 * 60 * 1000; const OAUTH_API_VERSION = "2.18"; const OAUTH_USER_AGENT = "Perplexity/641 CFNetwork/1568 Darwin/25.2.0"; +const ANONYMOUS_USER_AGENT = + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36"; type PerplexityAuth = | { @@ -45,6 +48,9 @@ type PerplexityAuth = | { type: "cookies"; cookies: string; + } + | { + type: "anonymous"; }; interface PerplexityOAuthStreamMarkdownBlock { @@ -149,6 +155,112 @@ function mergeOAuthEventSnapshot( return merged; } + +function asRecord(value: unknown): Record | null { + if (typeof value !== "object" || value === null || Array.isArray(value)) return null; + return value as Record; +} +} + +function parseJson(text: string): unknown | null { + try { + return JSON.parse(text); + } catch { + return null; + } +} + +function textFromChunks(value: unknown): string | null { + if (!Array.isArray(value) || value.length === 0) return null; + let text = ""; + for (const chunk of value) { + if (typeof chunk !== "string") return null; + text += chunk; + } + return text.length > 0 ? text : null; +} + +function textFromStructuredAnswer(value: unknown): string | null { + if (!Array.isArray(value)) return null; + for (const item of value) { + const record = asRecord(item); + if (!record) continue; + const text = record.text; + if (typeof text === "string" && text.length > 0) return text; + const chunks = textFromChunks(record.chunks); + if (chunks) return chunks; + } + return null; +} + +function answerFromTextPayload(payload: Record): string | null { + const structured = textFromStructuredAnswer(payload.structured_answer); + if (structured) return structured; + const chunks = textFromChunks(payload.chunks); + if (chunks) return chunks; + const answer = payload.answer; + return typeof answer === "string" && answer.length > 0 ? answer : null; +} + +function parseOAuthTextPayload(text: string): Record | null { + const parsed = parseJson(text); + const direct = asRecord(parsed); + if (direct) return direct; + if (!Array.isArray(parsed)) return null; + + for (const item of parsed) { + const step = asRecord(item); + const content = asRecord(step?.content); + const answer = content?.answer; + if (typeof answer !== "string" || answer.length === 0) continue; + const payload = asRecord(parseJson(answer)); + if (payload) return payload; + } + return null; +} + +function parseOAuthTextAnswer(text: string): string { + const payload = parseOAuthTextPayload(text); + if (payload) { + const answer = answerFromTextPayload(payload); + if (answer) return answer; + } + + const parsed = parseJson(text); + if (!Array.isArray(parsed)) return text; + for (const item of parsed) { + const step = asRecord(item); + const content = asRecord(step?.content); + const answer = content?.answer; + if (typeof answer === "string" && answer.length > 0) return answer; + } + return text; +} + +function sourcesFromTextPayload(text: string | undefined): SearchSource[] { + if (!text) return []; + const payload = parseOAuthTextPayload(text); + const webResults = payload?.web_results; + if (!Array.isArray(webResults) || webResults.length === 0) return []; + + const sources: SearchSource[] = []; + for (const value of webResults) { + const result = asRecord(value); + const url = result?.url; + if (typeof url !== "string" || url.length === 0) continue; + const name = result.name; + const snippet = result.snippet; + const timestamp = result.timestamp; + sources.push({ + title: typeof name === "string" && name.length > 0 ? name : url, + url, + snippet: typeof snippet === "string" ? snippet : undefined, + publishedDate: typeof timestamp === "string" ? timestamp : undefined, + ageSeconds: dateToAgeSeconds(typeof timestamp === "string" ? timestamp : undefined), + }); + } + return sources; +} export interface PerplexitySearchParams { signal?: AbortSignal; query: string; @@ -216,7 +328,7 @@ async function findPerplexityAuth( authStorage: AuthStorage, sessionId: string | undefined, signal: AbortSignal | undefined, -): Promise { +): Promise { // 1. PERPLEXITY_COOKIES env var const cookies = $env.PERPLEXITY_COOKIES?.trim(); if (cookies) { @@ -235,7 +347,9 @@ async function findPerplexityAuth( if (apiKey) { return { type: "api_key", token: apiKey }; } - return null; + + // 4. The consumer ask endpoint currently accepts unauthenticated browser-style requests. + return { type: "anonymous" }; } /** Call Perplexity API-key endpoint. */ @@ -284,7 +398,7 @@ function buildOAuthSources(event: PerplexityOAuthStreamEvent): SearchSource[] { })); } - return (event.sources_list ?? []) + const sources = (event.sources_list ?? []) .filter(source => typeof source.url === "string" && source.url.length > 0) .map(source => ({ title: source.title ?? source.url ?? "", @@ -293,11 +407,13 @@ function buildOAuthSources(event: PerplexityOAuthStreamEvent): SearchSource[] { publishedDate: source.date, ageSeconds: dateToAgeSeconds(source.date), })); + if (sources.length > 0) return sources; + return sourcesFromTextPayload(event.text); } function buildOAuthAnswer(event: PerplexityOAuthStreamEvent): string { if (!event.blocks?.length) { - return typeof event.text === "string" ? event.text : ""; + return typeof event.text === "string" ? parseOAuthTextAnswer(event.text) : ""; } const markdownBlock = event.blocks.find( @@ -324,57 +440,74 @@ function buildOAuthAnswer(event: PerplexityOAuthStreamEvent): string { } } if (typeof event.text === "string" && event.text.length > 0) { - return event.text; + return parseOAuthTextAnswer(event.text); } return ""; } async function callPerplexityOAuth( - auth: { type: "oauth"; token: string } | { type: "cookies"; cookies: string }, + auth: + | { type: "oauth"; token: string } + | { type: "cookies"; cookies: string } + | { type: "anonymous" }, params: PerplexitySearchParams, ): Promise<{ answer: string; sources: SearchSource[]; model?: string; requestId?: string }> { const requestId = crypto.randomUUID(); // The consumer `perplexity_ask` endpoint is itself a research assistant and // has no system-message slot. Prepending the API-style system prompt to the // query makes the model read it as a meta-instruction and refuse with - // "I don't have access to web-search tools in this turn", so OAuth/cookie + // "I don't have access to web-search tools in this turn", so ask-endpoint // searches send the bare query. (The API-key path still uses system_prompt // as a proper `system` message.) const effectiveQuery = params.query; + const headers: Record = { + "Content-Type": "application/json", + Accept: "text/event-stream", + Origin: "https://www.perplexity.ai", + Referer: "https://www.perplexity.ai/", + "User-Agent": auth.type === "anonymous" ? ANONYMOUS_USER_AGENT : OAUTH_USER_AGENT, + "X-Request-ID": requestId, + }; + if (auth.type === "oauth") { + headers.Authorization = `Bearer ${auth.token}`; + } else if (auth.type === "cookies") { + headers.Cookie = auth.cookies; + } + if (auth.type !== "anonymous") { + headers["X-App-ApiClient"] = "default"; + headers["X-App-ApiVersion"] = OAUTH_API_VERSION; + headers["X-Perplexity-Request-Reason"] = "submit"; + } + + const requestParams: Record = { + query_str: effectiveQuery, + search_focus: "internet", + mode: "copilot", + model_preference: auth.type === "anonymous" ? "experimental" : "pplx_pro_upgraded", + sources: ["web"], + attachments: [], + frontend_uuid: crypto.randomUUID(), + frontend_context_uuid: crypto.randomUUID(), + version: OAUTH_API_VERSION, + language: "en-US", + timezone: Intl.DateTimeFormat().resolvedOptions().timeZone, + search_recency_filter: params.search_recency_filter ?? null, + is_incognito: true, + use_schematized_api: true, + skip_search_enabled: true, + }; + if (auth.type === "anonymous") { + requestParams.send_back_text_in_streaming_api = true; + requestParams.source = "default"; + } + const response = await fetch(PERPLEXITY_OAUTH_ASK_URL, { method: "POST", - headers: { - ...(auth.type === "cookies" ? { Cookie: auth.cookies } : { Authorization: `Bearer ${auth.token}` }), - "Content-Type": "application/json", - Accept: "text/event-stream", - Origin: "https://www.perplexity.ai", - Referer: "https://www.perplexity.ai/", - "User-Agent": OAUTH_USER_AGENT, - "X-App-ApiClient": "default", - "X-App-ApiVersion": OAUTH_API_VERSION, - "X-Perplexity-Request-Reason": "submit", - "X-Request-ID": requestId, - }, + headers, body: JSON.stringify({ query_str: effectiveQuery, - params: { - query_str: effectiveQuery, - search_focus: "internet", - mode: "copilot", - model_preference: "pplx_pro_upgraded", - sources: ["web"], - attachments: [], - frontend_uuid: crypto.randomUUID(), - frontend_context_uuid: crypto.randomUUID(), - version: OAUTH_API_VERSION, - language: "en-US", - timezone: Intl.DateTimeFormat().resolvedOptions().timeZone, - search_recency_filter: params.search_recency_filter ?? null, - is_incognito: true, - use_schematized_api: true, - skip_search_enabled: true, - }, + params: requestParams, }), signal: withHardTimeout(params.signal), }); @@ -385,13 +518,13 @@ async function callPerplexityOAuth( if (classified) throw classified; throw new SearchProviderError( "perplexity", - `Perplexity OAuth API error (${response.status}): ${errorText}`, + `Perplexity ask API error (${response.status}): ${errorText}`, response.status, ); } if (!response.body) { - throw new SearchProviderError("perplexity", "Perplexity OAuth API returned no response body", 500); + throw new SearchProviderError("perplexity", "Perplexity ask API returned no response body", 500); } let answer = ""; @@ -403,7 +536,7 @@ async function callPerplexityOAuth( for await (const event of readSseJson(response.body, params.signal)) { if (event.error_code) { const message = event.error_message ?? event.error_code; - throw new SearchProviderError("perplexity", `Perplexity OAuth stream error: ${message}`, 400); + throw new SearchProviderError("perplexity", `Perplexity ask stream error: ${message}`, 400); } mergedEvent = mergeOAuthEventSnapshot(mergedEvent, event); @@ -506,20 +639,17 @@ function applySourceLimit(result: SearchResponse, limit?: number): SearchRespons /** Execute Perplexity web search */ export async function searchPerplexity(params: PerplexitySearchParams): Promise { const auth = await findPerplexityAuth(params.authStorage, params.sessionId, params.signal); - if (!auth) { - throw new Error("Perplexity auth not found. Set PERPLEXITY_COOKIES, PERPLEXITY_API_KEY, or login via OAuth."); - } - if (auth.type === "oauth" || auth.type === "cookies") { - const oauthResult = await callPerplexityOAuth(auth, params); + if (auth.type !== "api_key") { + const askResult = await callPerplexityOAuth(auth, params); return applySourceLimit( { provider: "perplexity", - answer: oauthResult.answer || undefined, - sources: oauthResult.sources, - model: oauthResult.model, - requestId: oauthResult.requestId, - authMode: "oauth", + answer: askResult.answer || undefined, + sources: askResult.sources, + model: askResult.model, + requestId: askResult.requestId, + authMode: auth.type === "anonymous" ? "anonymous" : "oauth", }, params.num_results, ); @@ -568,6 +698,15 @@ export class PerplexityProvider extends SearchProvider { return !!$env.PERPLEXITY_COOKIES?.trim() || authStorage.hasAuth("perplexity") || !!findApiKey(); } + /** + * Perplexity accepts anonymous browser-style ask requests, but keep auto + * provider selection credential-gated so a configured provider keeps priority + * over the anonymous fallback. + */ + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + search(params: SearchParams): Promise { return searchPerplexity({ signal: params.signal, diff --git a/packages/coding-agent/src/web/search/types.ts b/packages/coding-agent/src/web/search/types.ts index e37e6d961..946d6cc2d 100644 --- a/packages/coding-agent/src/web/search/types.ts +++ b/packages/coding-agent/src/web/search/types.ts @@ -33,7 +33,7 @@ export const SEARCH_PROVIDER_OPTIONS = [ description: "Automatically uses the first configured web-search provider", }, { value: "tavily", label: "Tavily", description: "Requires TAVILY_API_KEY" }, - { value: "perplexity", label: "Perplexity", description: "Requires PERPLEXITY_COOKIES or PERPLEXITY_API_KEY" }, + { value: "perplexity", label: "Perplexity", description: "Uses auth when configured; explicit selection falls back to anonymous search" }, { value: "brave", label: "Brave", description: "Requires BRAVE_API_KEY" }, { value: "jina", label: "Jina", description: "Requires JINA_API_KEY" }, { value: "kimi", label: "Kimi", description: "Requires MOONSHOT_SEARCH_API_KEY or MOONSHOT_API_KEY" }, From f73892d491921f1ec27b0d7bd570311c79552a89 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 20:01:21 +0200 Subject: [PATCH 018/181] fix(coding-agent): bounded expanded single-file search results MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Stopped expanded view from dumping every match when all hits share one file. - Applied an `EXPANDED_LINES × 2` budget while keeping context rows. - Appended a `… N more matches` summary when truncated. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/tools/search.ts | 39 ++++++++------- .../test/tools/search-renderer.test.ts | 50 +++++++++++++++++++ 3 files changed, 73 insertions(+), 17 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index dc108df82..45b731581 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -23,6 +23,7 @@ - Fixed Perplexity `perplexity_ask` response parsing so OAuth, cookie, and anonymous searches correctly extract answer text and sources from JSON payloads - Fixed framed read results rendering with an extra blank row above and below the output block. - Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice. +- Fixed the expanded search result view dumping every match when all hits live in one file. A single file's matches collapse into one blank-line group, and the expanded tree list ignored the line budget, so a hot file whose matches span its whole length rendered every row (e.g. lines 374–2858). Expanded search output is now bounded by a larger-than-collapsed budget (`EXPANDED_LINES × 2`), keeps surrounding context rows, and appends a `… N more matches` summary when truncated. - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. - Fixed custom-rendered tools that set `mergeCallAndResult` (e.g. `lsp`) rendering a redundant tool-name line above the framed result once a result arrived. `ToolExecutionComponent`'s custom-tool branch now emits the fallback label only when the tool has no `renderCall` and the call is not suppressed by an existing result, matching the built-in renderer branch. - Fixed the `write` tool result rendering with a green success checkmark even when the write failed. `writeToolRenderer.renderResult` now branches on `result.isError`, rendering the error status icon plus the failure message instead of the success header and content preview. diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index ce281fb05..d74a35726 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -1173,6 +1173,10 @@ interface SearchRenderArgs { } const COLLAPSED_TEXT_LIMIT = PREVIEW_LIMITS.COLLAPSED_LINES * 2; +/** Line budget for the expanded view. Larger than collapsed so expanding + * reveals more matches with context, but still bounded so a single hot file + * whose matches span the whole file can't dump its entire length. */ +const EXPANDED_TEXT_LIMIT = PREVIEW_LIMITS.EXPANDED_LINES * 2; const SEARCH_CODE_FRAME_LINE_RE = /^\s*\*?(\d+)│/; @@ -1283,16 +1287,20 @@ function countPreviewMatches(lines: readonly RenderedSearchLine[], hasMarkedMatc return lines.reduce((count, line) => count + (!isSearchHeaderLine(line.raw) && line.raw.length > 0 ? 1 : 0), 0); } -function renderCollapsedSearchGroups( +function renderBudgetedSearchGroups( groups: string[][], maxLines: number, matchCount: number, searchBase: string | undefined, uiTheme: Theme, + compact: boolean, ): string[] { if (maxLines <= 0) return []; const renderedGroups = groups - .map(group => compactSearchPreviewGroup(renderSearchDisplayGroup(group, searchBase, uiTheme))) + .map(group => { + const rendered = renderSearchDisplayGroup(group, searchBase, uiTheme); + return compact ? compactSearchPreviewGroup(rendered) : rendered; + }) .filter(group => group.length > 0); if (renderedGroups.length === 0) return []; @@ -1451,22 +1459,19 @@ export const searchToolRenderer = { return createCachedComponent( () => options.expanded, width => { - const collapsedMatchLineBudget = Math.max(COLLAPSED_TEXT_LIMIT - extraLines.length, 0); + const budget = Math.max( + (options.expanded ? EXPANDED_TEXT_LIMIT : COLLAPSED_TEXT_LIMIT) - extraLines.length, + 0, + ); const searchBase = details?.searchPath; - const matchLines = options.expanded - ? renderTreeList( - { - items: matchGroups, - expanded: true, - maxCollapsed: matchGroups.length, - maxCollapsedLines: collapsedMatchLineBudget, - itemType: "match", - renderItem: group => - renderSearchDisplayGroup(group, searchBase, uiTheme).map(line => line.styled), - }, - uiTheme, - ) - : renderCollapsedSearchGroups(matchGroups, collapsedMatchLineBudget, matchCount, searchBase, uiTheme); + const matchLines = renderBudgetedSearchGroups( + matchGroups, + budget, + matchCount, + searchBase, + uiTheme, + !options.expanded, + ); return [header, ...matchLines, ...extraLines].map(l => truncateToWidth(l, width, Ellipsis.Omit)); }, ); diff --git a/packages/coding-agent/test/tools/search-renderer.test.ts b/packages/coding-agent/test/tools/search-renderer.test.ts index 8f89c31e7..13f1c7f32 100644 --- a/packages/coding-agent/test/tools/search-renderer.test.ts +++ b/packages/coding-agent/test/tools/search-renderer.test.ts @@ -154,4 +154,54 @@ describe("searchToolRenderer", () => { expect(extractLinkUris(rendered)).toContain("file:///tmp/omp-project/file.ts?line=7"); }); + + it("bounds the expanded single-file view instead of dumping every match", async () => { + const theme = await getThemeByName("dark"); + expect(theme).toBeDefined(); + const uiTheme = theme!; + + // One file's matches collapse into a single blank-line group (no `#`/`##` + // headers, `│...` gap separators). Before the fix the expanded renderer + // dumped the entire span because the tree list ignored the line budget. + const clusters = Array.from({ length: 12 }, (_, i) => i * 100 + 1); + const displayContent = clusters + .map((line, idx) => { + const cluster = [` ${line}│ context before`, `*${line + 1}│ MATCH ${idx}`, ` ${line + 2}│ context after`]; + return idx === 0 ? cluster.join("\n") : [" │...", ...cluster].join("\n"); + }) + .join("\n"); + + const result = { + content: [{ type: "text", text: "" }], + details: { + matchCount: clusters.length, + fileCount: 1, + searchPath: "/tmp/omp-project/renderer.ts", + scopePath: "renderer.ts", + displayContent, + }, + }; + + const render = (expanded: boolean) => + sanitizeText( + searchToolRenderer + .renderResult(result as never, { expanded, isPartial: false }, uiTheme, { pattern: "needle" }) + .render(200) + .join("\n"), + ).split("\n"); + + const expanded = render(true); + const expandedBody = expanded.slice(1); + // Bounded: must not render all 12 clusters (36+ lines). + expect(expandedBody.length).toBeLessThan(clusters.length * 3); + expect(expandedBody.some(line => line.includes("more matches"))).toBe(true); + // Expanded keeps surrounding context lines (unlike the compact collapsed view). + expect(expandedBody.some(line => line.includes("context before"))).toBe(true); + + const collapsedBody = render(false).slice(1); + expect(collapsedBody.length).toBeLessThan(expandedBody.length); + // Collapsed compacts to match lines only — no context. + expect(collapsedBody.some(line => line.includes("context before"))).toBe(false); + expect(collapsedBody.some(line => line.includes("more matches"))).toBe(true); + }); }); From 246688e8748ef02a042f491d0381a2d3a86f95d0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 20:01:51 +0200 Subject: [PATCH 019/181] fix(coding-agent/web): unlocked perplexity pro via session cookie - Sent the OAuth token as `__Secure-next-auth.session-token` cookie since the ask endpoint ignores bearer headers and silently downgrades to `turbo`. - Fell back to `result.title` when web results omit `name`. - Renamed `callPerplexityOAuth` to `callPerplexityAsk` and removed a stray brace. - Added tests covering OAuth, API-key, and anonymous request shapes. --- packages/coding-agent/CHANGELOG.md | 1 + .../src/web/search/providers/perplexity.ts | 23 ++- packages/coding-agent/src/web/search/types.ts | 6 +- .../test/web/search/perplexity.test.ts | 171 +++++++++++++++++- 4 files changed, 189 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 45b731581..0b985df8c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -28,6 +28,7 @@ - Fixed custom-rendered tools that set `mergeCallAndResult` (e.g. `lsp`) rendering a redundant tool-name line above the framed result once a result arrived. `ToolExecutionComponent`'s custom-tool branch now emits the fallback label only when the tool has no `renderCall` and the call is not suppressed by an existing result, matching the built-in renderer branch. - Fixed the `write` tool result rendering with a green success checkmark even when the write failed. `writeToolRenderer.renderResult` now branches on `result.isError`, rendering the error status icon plus the failure message instead of the success header and content preview. - Fixed Perplexity OAuth/cookie web search returning a refusal answer ("I don't currently have access to the web-search tools in this turn") despite returning real sources. `callPerplexityOAuth` was prepending the API-style `web-search` system prompt to the query (`query_str = systemPrompt + "\n\n" + query`), but the consumer `www.perplexity.ai/rest/sse/perplexity_ask` endpoint has no system-message slot and reads the prepended instruction as a meta-prompt, making the model decline. The OAuth/cookie path now sends the bare query; the API-key path still passes the system prompt as a proper `system` message. +- Fixed Perplexity OAuth web search always returning the free `turbo` model instead of the account's Pro model (e.g. `pplx_pro_upgraded`/Sonar). The `www.perplexity.ai/rest/sse/perplexity_ask` endpoint authenticates via the `__Secure-next-auth.session-token` cookie and ignores the `Authorization: Bearer` header entirely — so sending the OAuth session token as a bearer was treated as an anonymous request, which silently downgrades to `turbo` regardless of `model_preference`. The stored Perplexity OAuth token is itself the next-auth session JWT (the macOS app injects the same value as that cookie), so `callPerplexityAsk` now sends it as the `__Secure-next-auth.session-token` cookie, unlocking Pro model selection. ## [15.9.67] - 2026-06-06 ### Added diff --git a/packages/coding-agent/src/web/search/providers/perplexity.ts b/packages/coding-agent/src/web/search/providers/perplexity.ts index 92a19865d..7a1579f01 100644 --- a/packages/coding-agent/src/web/search/providers/perplexity.ts +++ b/packages/coding-agent/src/web/search/providers/perplexity.ts @@ -160,7 +160,6 @@ function asRecord(value: unknown): Record | null { if (typeof value !== "object" || value === null || Array.isArray(value)) return null; return value as Record; } -} function parseJson(text: string): unknown | null { try { @@ -246,9 +245,10 @@ function sourcesFromTextPayload(text: string | undefined): SearchSource[] { const sources: SearchSource[] = []; for (const value of webResults) { const result = asRecord(value); - const url = result?.url; + if (!result) continue; + const url = result.url; if (typeof url !== "string" || url.length === 0) continue; - const name = result.name; + const name = result.name ?? result.title; const snippet = result.snippet; const timestamp = result.timestamp; sources.push({ @@ -445,11 +445,8 @@ function buildOAuthAnswer(event: PerplexityOAuthStreamEvent): string { return ""; } -async function callPerplexityOAuth( - auth: - | { type: "oauth"; token: string } - | { type: "cookies"; cookies: string } - | { type: "anonymous" }, +async function callPerplexityAsk( + auth: { type: "oauth"; token: string } | { type: "cookies"; cookies: string } | { type: "anonymous" }, params: PerplexitySearchParams, ): Promise<{ answer: string; sources: SearchSource[]; model?: string; requestId?: string }> { const requestId = crypto.randomUUID(); @@ -470,7 +467,13 @@ async function callPerplexityOAuth( "X-Request-ID": requestId, }; if (auth.type === "oauth") { - headers.Authorization = `Bearer ${auth.token}`; + // The ask endpoint authenticates via the next-auth session cookie, NOT a + // bearer header — a bearer (even a garbage one) is ignored and the request + // silently falls back to the anonymous free `turbo` model regardless of + // `model_preference`. The stored OAuth token IS the Perplexity session JWT + // (the native app injects the same value as this cookie), so sending it as + // the cookie is what unlocks the account's Pro model selection. + headers.Cookie = `__Secure-next-auth.session-token=${auth.token}`; } else if (auth.type === "cookies") { headers.Cookie = auth.cookies; } @@ -641,7 +644,7 @@ export async function searchPerplexity(params: PerplexitySearchParams): Promise< const auth = await findPerplexityAuth(params.authStorage, params.sessionId, params.signal); if (auth.type !== "api_key") { - const askResult = await callPerplexityOAuth(auth, params); + const askResult = await callPerplexityAsk(auth, params); return applySourceLimit( { provider: "perplexity", diff --git a/packages/coding-agent/src/web/search/types.ts b/packages/coding-agent/src/web/search/types.ts index 946d6cc2d..334971033 100644 --- a/packages/coding-agent/src/web/search/types.ts +++ b/packages/coding-agent/src/web/search/types.ts @@ -33,7 +33,11 @@ export const SEARCH_PROVIDER_OPTIONS = [ description: "Automatically uses the first configured web-search provider", }, { value: "tavily", label: "Tavily", description: "Requires TAVILY_API_KEY" }, - { value: "perplexity", label: "Perplexity", description: "Uses auth when configured; explicit selection falls back to anonymous search" }, + { + value: "perplexity", + label: "Perplexity", + description: "Uses auth when configured; explicit selection falls back to anonymous search", + }, { value: "brave", label: "Brave", description: "Requires BRAVE_API_KEY" }, { value: "jina", label: "Jina", description: "Requires JINA_API_KEY" }, { value: "kimi", label: "Kimi", description: "Requires MOONSHOT_SEARCH_API_KEY or MOONSHOT_API_KEY" }, diff --git a/packages/coding-agent/test/web/search/perplexity.test.ts b/packages/coding-agent/test/web/search/perplexity.test.ts index 1d2ef6781..4a7e30923 100644 --- a/packages/coding-agent/test/web/search/perplexity.test.ts +++ b/packages/coding-agent/test/web/search/perplexity.test.ts @@ -1,6 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import type { AuthStorage } from "@oh-my-pi/pi-ai"; -import { searchPerplexity } from "@oh-my-pi/pi-coding-agent/web/search/providers/perplexity"; +import { PerplexityProvider, searchPerplexity } from "@oh-my-pi/pi-coding-agent/web/search/providers/perplexity"; import { hookFetch } from "@oh-my-pi/pi-utils"; const API_URL = "https://api.perplexity.ai/chat/completions"; @@ -97,3 +97,172 @@ describe("Perplexity API-key request shape", () => { expect(response.relatedQuestions).toBeUndefined(); }); }); + +const OAUTH_ASK_URL = "https://www.perplexity.ai/rest/sse/perplexity_ask"; + +// OAuth path: getOAuthAccess returns a bearer (no `.`-delimited exp claim, so it +// is treated as non-expiring), making findPerplexityAuth pick the oauth branch. +const oauthAuthStorage = { + async getOAuthAccess() { + return { accessToken: "test-oauth-token" }; + }, + hasAuth() { + return true; + }, +} as unknown as AuthStorage; + +const anonymousAuthStorage = { + async getOAuthAccess() { + return undefined; + }, + hasAuth() { + return false; + }, +} as unknown as AuthStorage; + +function mockOAuth(capture: (body: Record, headers: Headers) => void) { + const event = { + final: true, + display_model: "turbo", + uuid: "req-oauth", + blocks: [ + { intended_usage: "ask_text", markdown_block: { answer: "OAuth answer" } }, + { + intended_usage: "web_results", + web_result_block: { web_results: [{ name: "T", url: "https://example.com", snippet: "s" }] }, + }, + ], + }; + const sseBody = `data: ${JSON.stringify(event)}\n\n`; + return hookFetch(async (input, init) => { + const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + if (url === OAUTH_ASK_URL) { + capture(JSON.parse(init?.body as string), new Headers(init?.headers)); + return new Response(sseBody, { status: 200, headers: { "Content-Type": "text/event-stream" } }); + } + return new Response("not mocked", { status: 500 }); + }); +} + +function mockAnonymous(capture: (body: Record, headers: Headers) => void) { + const answerPayload = { + answer: "Anonymous answer", + web_results: [{ name: "Example", url: "https://example.com", snippet: "s" }], + chunks: ["Anonymous ", "answer"], + structured_answer: [{ type: "markdown", text: "Anonymous answer", chunks: ["Anonymous ", "answer"] }], + }; + const event = { + final: true, + display_model: "turbo", + uuid: "req-anon", + text: JSON.stringify([{ step_type: "FINAL", content: { answer: JSON.stringify(answerPayload) }, uuid: "" }]), + }; + const sseBody = `data: ${JSON.stringify(event)}\n\n`; + return hookFetch(async (input, init) => { + const url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + if (url === OAUTH_ASK_URL) { + capture(JSON.parse(init?.body as string), new Headers(init?.headers)); + return new Response(sseBody, { status: 200, headers: { "Content-Type": "text/event-stream" } }); + } + return new Response("not mocked", { status: 500 }); + }); +} + +describe("Perplexity OAuth request shape", () => { + const savedCookies = process.env.PERPLEXITY_COOKIES; + + beforeEach(() => { + delete process.env.PERPLEXITY_COOKIES; // cookies take precedence over oauth; keep them out + }); + + afterEach(() => { + vi.restoreAllMocks(); + if (savedCookies === undefined) delete process.env.PERPLEXITY_COOKIES; + else process.env.PERPLEXITY_COOKIES = savedCookies; + }); + + it("sends the bare query, never the API-style system prompt, to the ask endpoint", async () => { + let body: Record | undefined; + let headers: Headers | undefined; + using _hook = mockOAuth((b, h) => { + body = b; + headers = h; + }); + + const response = await searchPerplexity({ + query: "quic vs tcp", + system_prompt: "Research assistant with web search. Synthesize comprehensive answers.", + authStorage: oauthAuthStorage, + }); + + // The consumer ask endpoint has no system slot; prepending the prompt makes + // the model refuse ("I don't have web-search tools in this turn"). + expect(body?.query_str).toBe("quic vs tcp"); + expect((body?.params as Record).query_str).toBe("quic vs tcp"); + // The ask endpoint authenticates via the next-auth session cookie; a bearer + // header is ignored and silently downgrades to the anonymous `turbo` model. + expect(headers?.get("cookie")).toBe("__Secure-next-auth.session-token=test-oauth-token"); + expect(headers?.has("authorization")).toBe(false); + expect(response.authMode).toBe("oauth"); + expect(response.answer).toBe("OAuth answer"); + }); +}); + +describe("Perplexity anonymous fallback", () => { + const savedKey = process.env.PERPLEXITY_API_KEY; + const savedPplxKey = process.env.PPLX_API_KEY; + const savedCookies = process.env.PERPLEXITY_COOKIES; + + beforeEach(() => { + delete process.env.PERPLEXITY_API_KEY; + delete process.env.PPLX_API_KEY; + delete process.env.PERPLEXITY_COOKIES; + }); + + afterEach(() => { + vi.restoreAllMocks(); + if (savedKey === undefined) delete process.env.PERPLEXITY_API_KEY; + else process.env.PERPLEXITY_API_KEY = savedKey; + if (savedPplxKey === undefined) delete process.env.PPLX_API_KEY; + else process.env.PPLX_API_KEY = savedPplxKey; + if (savedCookies === undefined) delete process.env.PERPLEXITY_COOKIES; + else process.env.PERPLEXITY_COOKIES = savedCookies; + }); + + it("uses the browser ask endpoint without credential headers when no key is configured", async () => { + let body: Record | undefined; + let headers: Headers | undefined; + using _hook = mockAnonymous((b, h) => { + body = b; + headers = h; + }); + + const response = await searchPerplexity({ query: "anonymous search", authStorage: anonymousAuthStorage }); + const requestParams = body?.params as Record; + + expect(headers?.has("authorization")).toBe(false); + expect(headers?.has("cookie")).toBe(false); + expect(headers?.get("user-agent")).toContain("Mozilla/5.0"); + expect(requestParams.model_preference).toBe("experimental"); + expect(requestParams.send_back_text_in_streaming_api).toBe(true); + expect(requestParams.source).toBe("default"); + expect(response.authMode).toBe("anonymous"); + expect(response.answer).toBe("Anonymous answer"); + expect(response.sources).toEqual([ + { + title: "Example", + url: "https://example.com", + snippet: "s", + publishedDate: undefined, + ageSeconds: undefined, + }, + ]); + }); + + it("keeps anonymous Perplexity out of auto provider selection but allows explicit selection", () => { + const provider = new PerplexityProvider(); + + expect(provider.isAvailable(anonymousAuthStorage)).toBe(false); + expect(provider.isExplicitlyAvailable(anonymousAuthStorage)).toBe(true); + }); +}); From ce60b6626b85708c0e93cd248f3fb307239b8a58 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:11:42 +0000 Subject: [PATCH 020/181] fix(coding-agent): harden todo renderer against malformed streaming args MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The todo tool's renderCall ran args?.ops?.map(...) directly, which throws TypeError on any non-array ops value. parseStreamingJson surfaces such shapes mid-stream: a partial Anthropic input_json_delta buffer like '{"ops":"[{' becomes { ops: '[{' }, and intermediate states can hand back null entries before object fields arrive. Each crash spammed Tool renderer failed warnings and starved the TUI render loop. Guard against: - ops being any non-array (string, object, primitive) - entries being null / non-object - entry.items being a non-array The fix is in the TUI renderer only — schema validation in the agent loop is unchanged, so any genuinely malformed model output still surfaces an invalid-args tool error to the model. Fixes #2005 --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/tools/todo.ts | 27 +++++++--- packages/coding-agent/test/tools/todo.test.ts | 50 +++++++++++++++++++ 3 files changed, 71 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6515a5a39..101c92e73 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -15,6 +15,7 @@ - Fixed framed read results rendering with an extra blank row above and below the output block. - Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice. - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. +- Fixed the `todo` tool's TUI renderer crashing with `TypeError: args?.ops?.map is not a function` when a streaming tool-call delta surfaces a non-array `ops` field (mid-stream `parseStreamingJson` shapes like `{ ops: "[{" }`, or `[null]` entries before fields arrive). The renderer now treats non-array `ops`, non-object entries, and non-array `items` as missing structure instead of crashing, which also stops the spam-warn/retry cascade that followed each malformed delta ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). ## [15.9.67] - 2026-06-06 ### Added diff --git a/packages/coding-agent/src/tools/todo.ts b/packages/coding-agent/src/tools/todo.ts index 6d7c67ab0..f61d4b419 100644 --- a/packages/coding-agent/src/tools/todo.ts +++ b/packages/coding-agent/src/tools/todo.ts @@ -777,13 +777,26 @@ function renderNoteAttachments(phases: TodoPhase[], uiTheme: Theme): string[] { export const todoToolRenderer = { renderCall(args: TodoRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component { - const ops = args?.ops?.map(entry => { - const parts = [entry.op ?? "update"]; - if (entry.task) parts.push(entry.task); - if (entry.phase) parts.push(entry.phase); - if (entry.items?.length) parts.push(`${entry.items.length} item${entry.items.length === 1 ? "" : "s"}`); - return parts.join(" "); - }) ?? ["update"]; + // `args` here is the raw partially-parsed JSON from the streaming + // tool-call delta and may not satisfy `TodoRenderArgs` at runtime: + // `parseStreamingJson` can hand back `{ ops: "[" }` mid-delta, or + // entries that are `null` / strings before fields stream. Guard + // against non-array `ops` and non-object entries so a malformed + // delta never breaks the TUI render loop (#2005). + const opsList = Array.isArray(args?.ops) ? args.ops : []; + const ops = + opsList.length === 0 + ? ["update"] + : opsList.map(entry => { + const e = entry && typeof entry === "object" ? entry : ({} as NonNullable); + const parts = [e.op ?? "update"]; + if (e.task) parts.push(e.task); + if (e.phase) parts.push(e.phase); + if (Array.isArray(e.items) && e.items.length) { + parts.push(`${e.items.length} item${e.items.length === 1 ? "" : "s"}`); + } + return parts.join(" "); + }); const text = renderStatusLine({ icon: "pending", title: "Todo", meta: ops }, uiTheme); return new Text(text, 0, 0); }, diff --git a/packages/coding-agent/test/tools/todo.test.ts b/packages/coding-agent/test/tools/todo.test.ts index 73d0d6a61..670091e0c 100644 --- a/packages/coding-agent/test/tools/todo.test.ts +++ b/packages/coding-agent/test/tools/todo.test.ts @@ -345,3 +345,53 @@ describe("todoMatchesAnyDescription", () => { expect(todoMatchesAnyDescription("Audit AGENTS.md compliance", ["Audit AGENTS md compliance"])).toBe(true); }); }); + +describe("todoToolRenderer.renderCall malformed-args regression (#2005)", () => { + // Reporter saw `TypeError: args?.ops?.map is not a function` against + // Xiaomi Token Plan's Anthropic protocol because `parseStreamingJson` + // surfaced `{ ops: "[..." }` shapes mid-stream. The renderer is invoked + // on every streaming delta, so any non-array `ops` (string, object, + // number) must NOT crash the TUI render loop and trigger the spam-warn / + // retry cascade. + const renderOptions = { expanded: false, isPartial: true } as const; + + it("does not throw when ops is a streaming-truncated string", () => { + // Mid-stream `partialJson === '{"ops":"[{'` parses into `{ops: "[{"}`. + const args = { ops: '[{"op":"init"' } as unknown as Parameters[0]; + expect(() => todoToolRenderer.renderCall(args, renderOptions, theme)).not.toThrow(); + }); + + it("does not throw when ops entries are null", () => { + // `partialParse` of `'{"ops":[null'` can hand back `{ops: [null]}` in + // intermediate states before the entry object opens. + const args = { ops: [null] } as unknown as Parameters[0]; + expect(() => todoToolRenderer.renderCall(args, renderOptions, theme)).not.toThrow(); + }); + + it("does not throw when an entry's items field is a non-array", () => { + const args = { + ops: [{ op: "append", phase: "Work", items: "Second" as unknown as string[] }], + } as unknown as Parameters[0]; + expect(() => todoToolRenderer.renderCall(args, renderOptions, theme)).not.toThrow(); + }); + + it("still renders ops summary metadata for well-formed args", () => { + const args = { + ops: [ + { op: "init", items: ["a", "b", "c"] }, + { op: "done", task: "a" }, + { op: "append", phase: "Cleanup", items: ["d"] }, + ], + }; + const component = todoToolRenderer.renderCall(args, renderOptions, theme); + // `Text(text, 0, 0)` from `@oh-my-pi/pi-tui` exposes the content via .render(). + const rendered = Bun.stripANSI(component.render(120).join("\n")); + expect(rendered).toContain("init"); + expect(rendered).toContain("3 items"); + expect(rendered).toContain("done"); + expect(rendered).toContain("a"); + expect(rendered).toContain("append"); + expect(rendered).toContain("Cleanup"); + expect(rendered).toContain("1 item"); + }); +}); From 1c5cb37df5ba00bece7471bde72722d8d468bb7e Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:12:08 +0000 Subject: [PATCH 021/181] style: bun run fix --- packages/coding-agent/test/extensions-runner.test.ts | 1 - 1 file changed, 1 deletion(-) diff --git a/packages/coding-agent/test/extensions-runner.test.ts b/packages/coding-agent/test/extensions-runner.test.ts index 1542897a9..33e188b7f 100644 --- a/packages/coding-agent/test/extensions-runner.test.ts +++ b/packages/coding-agent/test/extensions-runner.test.ts @@ -148,7 +148,6 @@ describe("ExtensionRunner", () => { warnSpy.mockRestore(); }); - it("warns when two extensions register same shortcut", async () => { // Use a non-reserved shortcut const extCode1 = ` From 43b22e95646369a72f8ef10c295f1cda03883ad0 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 20:13:06 +0200 Subject: [PATCH 022/181] fix(coding-agent/web): defaulted perplexity ask to experimental model - Forced authenticated ask requests to `experimental`, matching the anonymous fallback since the cookie session ignores pro upgrades. - Kept TUI collapsed search answers full; capping now only applies in compact mode via `maxAnswerLines`. - Preserved full multiline task pending preview instead of bounding it. --- packages/coding-agent/CHANGELOG.md | 1 + .../src/web/search/providers/perplexity.ts | 2 +- .../test/streaming-preview-height.test.ts | 30 +++++++++++-------- .../test/web/search/perplexity.test.ts | 1 + .../test/web/search/render.test.ts | 16 ++++++++-- 5 files changed, 34 insertions(+), 16 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 0b985df8c..1f7b3db85 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,6 +13,7 @@ ### Changed +- Changed authenticated Perplexity ask requests to default to the `experimental` model preference, matching the anonymous fallback. - Changed Perplexity explicit provider availability checks so the setup wizard can mark `perplexity` as available for manual selection without credentials, while auto provider discovery still requires auth - Changed the default persistent model selector shortcut from `Ctrl+L` to `Alt+M`, leaving `Ctrl+L` for display reset. Existing user remaps in `keybindings.yml` are preserved. - Changed the edit tool result header to carry the diff change stats (`+N / -M / K hunks`) inline next to the file path, and removed the redundant lone language-icon metadata row and the blank line between the header and the diff body, so a single-hunk edit renders as `✔ Edit: path:LINE ⟨+3 / 1 hunk⟩` immediately followed by the diff. diff --git a/packages/coding-agent/src/web/search/providers/perplexity.ts b/packages/coding-agent/src/web/search/providers/perplexity.ts index 7a1579f01..70284cdfb 100644 --- a/packages/coding-agent/src/web/search/providers/perplexity.ts +++ b/packages/coding-agent/src/web/search/providers/perplexity.ts @@ -487,7 +487,7 @@ async function callPerplexityAsk( query_str: effectiveQuery, search_focus: "internet", mode: "copilot", - model_preference: auth.type === "anonymous" ? "experimental" : "pplx_pro_upgraded", + model_preference: "experimental", sources: ["web"], attachments: [], frontend_uuid: crypto.randomUUID(), diff --git a/packages/coding-agent/test/streaming-preview-height.test.ts b/packages/coding-agent/test/streaming-preview-height.test.ts index 8321e2b59..5156d8382 100644 --- a/packages/coding-agent/test/streaming-preview-height.test.ts +++ b/packages/coding-agent/test/streaming-preview-height.test.ts @@ -338,7 +338,7 @@ describe("streaming tool call preview height (bounded across renderers)", () => expect(visibleWidth(topBorder ?? "")).toBe(width); }); - test("eval/bash/ssh/task pending previews stay short even with very long multiline args", () => { + test("eval/bash/ssh pending previews stay short even with very long multiline args", () => { const longLines = Array.from({ length: 80 }, (_, i) => `line-${i}`); const cases: Array<{ name: string; @@ -359,7 +359,7 @@ describe("streaming tool call preview height (bounded across renderers)", () => marker: /earlier lines/, }, { - // bash/ssh/task keep a bounded head+tail window: the start and the + // bash/ssh keep a bounded head+tail window: the start and the // latest are both visible, the middle is elided. name: "bash", args: { command: longLines.join("\n") }, @@ -374,17 +374,6 @@ describe("streaming tool call preview height (bounded across renderers)", () => mustHide: ["line-40"], marker: /more lines/, }, - { - name: "task", - args: { - agent: "task", - context: longLines.join("\n"), - tasks: [{ id: "alpha", description: "preview" }], - }, - mustContain: ["line-0", "line-79"], - mustHide: ["line-40"], - marker: /more lines/, - }, ]; for (const testCase of cases) { @@ -399,4 +388,19 @@ describe("streaming tool call preview height (bounded across renderers)", () => expect(text, `${testCase.name} preview should advertise truncation`).toMatch(testCase.marker); } }); + + test("task pending preview preserves full multiline context", () => { + const longLines = Array.from({ length: 80 }, (_, i) => `line-${i}`); + const { lines, text } = renderPending("task", { + agent: "task", + context: longLines.join("\n"), + tasks: [{ id: "alpha", description: "preview" }], + }); + + expect(lines.length, "task preview should not be capped").toBeGreaterThan(80); + expect(text).toContain("line-0"); + expect(text).toContain("line-40"); + expect(text).toContain("line-79"); + expect(text).not.toMatch(/more lines/); + }); }); diff --git a/packages/coding-agent/test/web/search/perplexity.test.ts b/packages/coding-agent/test/web/search/perplexity.test.ts index 4a7e30923..a582141ab 100644 --- a/packages/coding-agent/test/web/search/perplexity.test.ts +++ b/packages/coding-agent/test/web/search/perplexity.test.ts @@ -199,6 +199,7 @@ describe("Perplexity OAuth request shape", () => { // the model refuse ("I don't have web-search tools in this turn"). expect(body?.query_str).toBe("quic vs tcp"); expect((body?.params as Record).query_str).toBe("quic vs tcp"); + expect((body?.params as Record).model_preference).toBe("experimental"); // The ask endpoint authenticates via the next-auth session cookie; a bearer // header is ignored and silently downgrades to the anonymous `turbo` model. expect(headers?.get("cookie")).toBe("__Secure-next-auth.session-token=test-oauth-token"); diff --git a/packages/coding-agent/test/web/search/render.test.ts b/packages/coding-agent/test/web/search/render.test.ts index d9a26ccc5..e2b689dc7 100644 --- a/packages/coding-agent/test/web/search/render.test.ts +++ b/packages/coding-agent/test/web/search/render.test.ts @@ -75,13 +75,25 @@ describe("renderSearchResult", () => { expect(answer).not.toMatch(/more line/); }); - it("truncates the answer with a summary when collapsed", async () => { + it("shows the full answer when collapsed by default", async () => { const uiTheme = (await getThemeByName("dark"))!; const component = renderSearchResult(buildResult(ANSWER), { expanded: false, isPartial: false }, uiTheme, { query: "test query", }); const answer = answerSection(component.render(120).map(l => sanitizeText(l))); - // Collapsed view caps the answer and signals how many lines were hidden. + // TUI collapsed view keeps the answer intact; only explicit compact mode caps it. + expect(answer).toContain("FINAL_UNIQUE_MARKER"); + expect(answer).not.toMatch(/more line/); + }); + + it("truncates the answer only when compact mode provides maxAnswerLines", async () => { + const uiTheme = (await getThemeByName("dark"))!; + const component = renderSearchResult(buildResult(ANSWER), { expanded: false, isPartial: false }, uiTheme, { + query: "test query", + maxAnswerLines: 3, + }); + const answer = answerSection(component.render(120).map(l => sanitizeText(l))); + expect(answer).toMatch(/more line/); expect(answer).not.toContain("FINAL_UNIQUE_MARKER"); }); From a7f5e83067c22638f01e6adef16f14c1a5ed5480 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 20:13:59 +0200 Subject: [PATCH 023/181] chore: bump version to 15.9.69 --- Cargo.lock | 8 ++--- Cargo.toml | 2 +- bun.lock | 46 +++++++++++++++------------ crates/pi-natives/src/lib.rs | 2 +- package.json | 18 +++++------ packages/agent/package.json | 2 +- packages/ai/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 ++ packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 ++ packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- 19 files changed, 56 insertions(+), 48 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index b88543c78..170561fab 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2331,7 +2331,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "15.9.67" +version = "15.9.69" dependencies = [ "anyhow", "ast-grep-core", @@ -2399,7 +2399,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "15.9.67" +version = "15.9.69" dependencies = [ "async-trait", "libc", @@ -2411,7 +2411,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "15.9.67" +version = "15.9.69" dependencies = [ "anyhow", "arboard", @@ -2457,7 +2457,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "15.9.67" +version = "15.9.69" dependencies = [ "anyhow", "brush-builtins", diff --git a/Cargo.toml b/Cargo.toml index 2cf4d2996..5a0405396 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "15.9.67" +version = "15.9.69" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index bff8b3963..78a5b2598 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.67", + "version": "15.9.69", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -30,7 +30,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "15.9.67", + "version": "15.9.69", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -44,7 +44,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.67", + "version": "15.9.69", "bin": { "omp": "src/cli.ts", }, @@ -90,7 +90,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "15.9.67", + "version": "15.9.69", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -101,7 +101,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.67", + "version": "15.9.69", "bin": { "mnemopi": "src/cli.ts", }, @@ -118,7 +118,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "15.9.67", + "version": "15.9.69", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -126,7 +126,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "15.9.67", + "version": "15.9.69", "bin": { "omp-stats": "./src/index.ts", }, @@ -151,7 +151,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "15.9.67", + "version": "15.9.69", "bin": { "omp-swarm": "src/cli.ts", }, @@ -167,7 +167,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "15.9.67", + "version": "15.9.69", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -208,7 +208,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "15.9.67", + "version": "15.9.69", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "beautiful-mermaid": "catalog:", @@ -248,15 +248,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.67", - "@oh-my-pi/omp-stats": "15.9.67", - "@oh-my-pi/pi-agent-core": "15.9.67", - "@oh-my-pi/pi-ai": "15.9.67", - "@oh-my-pi/pi-coding-agent": "15.9.67", - "@oh-my-pi/pi-mnemopi": "15.9.67", - "@oh-my-pi/pi-natives": "15.9.67", - "@oh-my-pi/pi-tui": "15.9.67", - "@oh-my-pi/pi-utils": "15.9.67", + "@oh-my-pi/hashline": "15.9.69", + "@oh-my-pi/omp-stats": "15.9.69", + "@oh-my-pi/pi-agent-core": "15.9.69", + "@oh-my-pi/pi-ai": "15.9.69", + "@oh-my-pi/pi-coding-agent": "15.9.69", + "@oh-my-pi/pi-mnemopi": "15.9.69", + "@oh-my-pi/pi-natives": "15.9.69", + "@oh-my-pi/pi-tui": "15.9.69", + "@oh-my-pi/pi-utils": "15.9.69", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -395,7 +395,7 @@ "@huggingface/jinja": ["@huggingface/jinja@0.5.9", "", {}, "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw=="], - "@huggingface/tasks": ["@huggingface/tasks@0.21.6", "", {}, "sha512-XfLE2clF0uHw7kMb6HHMkpyJ+bmu2T0EZ8O1WxbJXIqdRwKRB0RM7Y639Ph3aYj2GjFLJhNo1Lz8y0jn9k10LQ=="], + "@huggingface/tasks": ["@huggingface/tasks@0.21.7", "", {}, "sha512-GuEXszIkir4j/Oywp4hXP+wfwojo/SKWA/omroNkzWWgqUGiOQ5p6HuyXcDOcinYnLQW1WsO8fwdEvtLTZbA4w=="], "@huggingface/tokenizers": ["@huggingface/tokenizers@0.1.3", "", {}, "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA=="], @@ -1259,7 +1259,7 @@ "string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="], - "string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], + "string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="], "strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="], @@ -1413,6 +1413,8 @@ "string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="], + "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], + "wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="], "xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="], @@ -1427,6 +1429,8 @@ "fastembed/onnxruntime-node/tar": ["tar@7.5.16", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w=="], + "jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], + "log-update/slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], "log-update/wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 5017e2ddd..186fca14d 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -68,5 +68,5 @@ use napi_derive::napi; /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV15_9_67")] +#[napi(js_name = "__piNativesV15_9_69")] pub const fn pi_natives_version_sentinel() {} diff --git a/package.json b/package.json index 542faa00e..78098d0fa 100644 --- a/package.json +++ b/package.json @@ -20,15 +20,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.67", - "@oh-my-pi/omp-stats": "15.9.67", - "@oh-my-pi/pi-agent-core": "15.9.67", - "@oh-my-pi/pi-ai": "15.9.67", - "@oh-my-pi/pi-coding-agent": "15.9.67", - "@oh-my-pi/pi-mnemopi": "15.9.67", - "@oh-my-pi/pi-natives": "15.9.67", - "@oh-my-pi/pi-tui": "15.9.67", - "@oh-my-pi/pi-utils": "15.9.67", + "@oh-my-pi/hashline": "15.9.69", + "@oh-my-pi/omp-stats": "15.9.69", + "@oh-my-pi/pi-agent-core": "15.9.69", + "@oh-my-pi/pi-ai": "15.9.69", + "@oh-my-pi/pi-coding-agent": "15.9.69", + "@oh-my-pi/pi-mnemopi": "15.9.69", + "@oh-my-pi/pi-natives": "15.9.69", + "@oh-my-pi/pi-tui": "15.9.69", + "@oh-my-pi/pi-utils": "15.9.69", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index eccf73aa8..6310a8e04 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.67", + "version": "15.9.69", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/package.json b/packages/ai/package.json index 5c05d9e97..30ed8ddba 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "15.9.67", + "version": "15.9.69", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1f7b3db85..33ff5aaaf 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.9.69] - 2026-06-06 + ### Added - Added anonymous fallback for Perplexity web search, allowing `web_search` and explicit Perplexity provider usage when no Perplexity credentials are configured diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index ffe24bdca..7698d64de 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.67", + "version": "15.9.69", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 2d04460bb..98c20ce7f 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "15.9.67", + "version": "15.9.69", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 170d46cec..2687a9186 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.67", + "version": "15.9.69", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 9b2eaa14a..a008fcbb1 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -136,7 +136,7 @@ export declare class Shell { * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV15_9_67(): void +export declare function __piNativesV15_9_69(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index cb0247d4a..9048b0f2e 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -23,7 +23,7 @@ export const PtySession = nativeBindings.PtySession; export const Shell = nativeBindings.Shell; // functions -export const __piNativesV15_9_67 = nativeBindings.__piNativesV15_9_67; +export const __piNativesV15_9_69 = nativeBindings.__piNativesV15_9_69; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 2f7df44ef..65f55801e 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "15.9.67", + "version": "15.9.69", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/stats/package.json b/packages/stats/package.json index bc1eb2e5a..79d24ca71 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "15.9.67", + "version": "15.9.69", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index c0cb0db20..4e224cccb 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "15.9.67", + "version": "15.9.69", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 40abd7bb6..f5151aefa 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [15.9.69] - 2026-06-06 + ### Added - Added `TUI.resetDisplay()` to force an immediate full-frame replay, including native scrollback when the host can safely clear it. diff --git a/packages/tui/package.json b/packages/tui/package.json index 1e16b084c..2a2140346 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "15.9.67", + "version": "15.9.69", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index 8aef12be3..6a022067f 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "15.9.67", + "version": "15.9.69", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From cab465cba4f7e067d0992b77194bffa43f54a3b2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:15:04 +0000 Subject: [PATCH 024/181] fix(eval): surfaced subagent abort reason through python agent() bridge Python eval agent() collapsed every subagent runtime-limit abort into a generic 'RuntimeError: bridge call __agent__ failed' instead of the real reason. runEvalAgent built its failure message with: result.error ?? result.stderr ?? result.abortReason ?? ? is nullish-coalescing, so result.stderr = "" (the executor's value for a runtime-limit abort) short-circuited the chain and never reached abortReason. The host bridge then shipped {ok: false, error: ""}, and prelude.py's ' or ' picked the named-bridge fallback. Extracted buildSubagentFailureMessage(): aborted subagents prefer the trimmed abortReason; otherwise fall through error, stderr (trimmed), abortReason, and the named-bridge default. Empty/whitespace strings no longer mask anything. The failure-detection condition also accepts result.aborted so an abort with exitCode 0 (theoretically) still flows the abort reason out. Added a regression test asserting that runtime-limit aborts, whitespace stderr/error, and totally blank aborts all produce non-empty messages matching the executor's abortReason text. Fixes #2006 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/eval/__tests__/agent-bridge.test.ts | 51 +++++++++++++++++++ .../coding-agent/src/eval/agent-bridge.ts | 28 ++++++++-- 3 files changed, 75 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 6515a5a39..e9378f066 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -15,6 +15,7 @@ - Fixed framed read results rendering with an extra blank row above and below the output block. - Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice. - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. +- Fixed Python eval `agent()` collapsing subagent runtime-limit aborts (and other empty-stderr aborts) into a generic `RuntimeError: bridge call '__agent__' failed`. `runEvalAgent` coalesced the failure message with `??`, which stopped at the empty `stderr` and never reached `abortReason`, shipping an empty error through the loopback bridge. The bridge now prefers `abortReason` for aborts and trims empty `stderr`/`error` out of the fallback chain, so Python surfaces the actionable reason (e.g. `Subagent runtime limit exceeded (task.maxRuntimeMs=900000)`) ([#2006](https://github.com/can1357/oh-my-pi/issues/2006)). ## [15.9.67] - 2026-06-06 ### Added diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 08bf08401..dd66f44cc 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -231,6 +231,57 @@ describe("runEvalAgent", () => { }); await expect(runEvalAgent({ prompt: "fail" }, { session: makeSession() })).rejects.toThrow("boom"); }); + + // Regression: a runtime-limit abort returns exitCode=1, stderr="", error=undefined, + // aborted=true, abortReason="Subagent runtime limit exceeded (...)". The previous + // failure-message coalesce stopped at the empty `stderr` (since `??` only skips + // nullish values) and shipped an empty error through the bridge — Python then + // surfaced the generic `bridge call '__agent__' failed`. See #2006. + it("surfaces abortReason for aborts that leave stderr empty", async () => { + mockAgents(); + const runSpy = vi.spyOn(taskExecutor, "runSubprocess"); + runSpy.mockImplementationOnce(async options => + singleResult(options, { + exitCode: 1, + output: "", + stderr: "", + error: undefined, + aborted: true, + abortReason: "Subagent runtime limit exceeded (task.maxRuntimeMs=900000)", + }), + ); + runSpy.mockImplementationOnce(async options => + singleResult(options, { + exitCode: 1, + output: "", + stderr: " ", + error: " ", + aborted: true, + abortReason: "Cancelled by caller", + }), + ); + runSpy.mockImplementationOnce(async options => + singleResult(options, { + exitCode: 1, + output: "", + stderr: "", + error: undefined, + }), + ); + + await expect(runEvalAgent({ prompt: "slow" }, { session: makeSession() })).rejects.toThrow( + "Subagent runtime limit exceeded (task.maxRuntimeMs=900000)", + ); + // Whitespace-only stderr/error must not mask abortReason either. + await expect(runEvalAgent({ prompt: "cancelled" }, { session: makeSession() })).rejects.toThrow( + "Cancelled by caller", + ); + // Last resort: still produce a non-empty message even when nothing useful is set, + // so Python never falls back to `bridge call '__agent__' failed`. + await expect(runEvalAgent({ prompt: "blank" }, { session: makeSession() })).rejects.toThrow( + "agent() subagent 'task' failed.", + ); + }); }); describe("agent() through eval runtimes", () => { diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index a97f6c98e..696d9c011 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -13,7 +13,7 @@ import subagentUserPromptTemplate from "../prompts/system/subagent-user-prompt.m import * as taskDiscovery from "../task/discovery"; import * as taskExecutor from "../task/executor"; import { AgentOutputManager } from "../task/output-manager"; -import type { AgentDefinition, AgentProgress } from "../task/types"; +import type { AgentDefinition, AgentProgress, SingleResult } from "../task/types"; import type { ToolSession } from "../tools"; import { ToolError } from "../tools/tool-errors"; import { withBridgeTimeoutPause } from "./bridge-timeout"; @@ -173,6 +173,26 @@ function emitProgressStatus(emitStatus: ((event: JsStatusEvent) => void) | undef }); } +/** + * Coalesce a subagent failure into a non-empty, human-meaningful error message. + * + * When the executor aborts a subagent (runtime limit, parent cancellation, …) + * the actionable explanation lives on `abortReason`, while `error`/`stderr` + * are routinely empty strings. Plain `??` coalescing stops at the empty string + * and ships an empty error through the bridge — Python then surfaces only the + * generic `bridge call '__agent__' failed`. See #2006. + */ +function buildSubagentFailureMessage(agentName: string, result: SingleResult): string { + const abortReason = trimToUndefined(result.abortReason); + if (result.aborted && abortReason) return abortReason; + return ( + trimToUndefined(result.error) ?? + trimToUndefined(result.stderr) ?? + abortReason ?? + `agent() subagent '${agentName}' failed.` + ); +} + /** * Run a single subagent on behalf of an eval cell's `agent()` call. */ @@ -278,10 +298,8 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption }), ); - if (result.exitCode !== 0 || result.error) { - const failureMessage = - result.error ?? result.stderr ?? result.abortReason ?? `agent() subagent '${agentName}' failed.`; - throw new ToolError(failureMessage); + if (result.exitCode !== 0 || result.error || result.aborted) { + throw new ToolError(buildSubagentFailureMessage(agentName, result)); } options.session.recordEvalSubagentUsage?.(result.usage?.output ?? 0); From e83dbc177b33dda19dd9365eeca7f453b619f52f Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:15:09 +0000 Subject: [PATCH 025/181] style: bun run fix --- packages/coding-agent/test/extensions-runner.test.ts | 1 - 1 file changed, 1 deletion(-) diff --git a/packages/coding-agent/test/extensions-runner.test.ts b/packages/coding-agent/test/extensions-runner.test.ts index 1542897a9..33e188b7f 100644 --- a/packages/coding-agent/test/extensions-runner.test.ts +++ b/packages/coding-agent/test/extensions-runner.test.ts @@ -148,7 +148,6 @@ describe("ExtensionRunner", () => { warnSpy.mockRestore(); }); - it("warns when two extensions register same shortcut", async () => { // Use a non-reserved shortcut const extCode1 = ` From 6dcbb07793c412427356975a002fcb3507a0ba24 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:20:27 +0000 Subject: [PATCH 026/181] fix(ai): replay xiaomi mimo anthropic-compat thinking blocks unsigned MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Anthropic-compat endpoints hosted under *.xiaomimimo.com (every Xiaomi MiMo Token Plan region plus api.xiaomimimo.com) emit thinking blocks without a signature. convertAnthropicMessages defaulted to "signing capable" for any endpoint not explicitly allowlisted as non-signing, so MiMo's unsigned thinking blocks were demoted to text on every continuation request. Without its prior reasoning replayed, MiMo destabilized tool-call argument serialization — the root cause behind the args?.ops?.map crash already mitigated at the renderer in #2005. Extend isNonSigningAnthropicEndpoint to cover the xiaomi catalog provider, every xiaomi-token-plan-* provider id, and any baseUrl on xiaomimimo.com so the existing non-signing replay branch fires for MiMo the same way it does for DeepSeek and Z.AI. Fixes #2005 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/anthropic.ts | 20 ++- .../anthropic-xiaomi-thinking-replay.test.ts | 153 ++++++++++++++++++ packages/coding-agent/CHANGELOG.md | 2 +- 4 files changed, 174 insertions(+), 5 deletions(-) create mode 100644 packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2f8dea9f7..ba2e08298 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Xiaomi MiMo Anthropic-compat endpoints (`*.xiaomimimo.com/anthropic`, including every Token Plan region and `api.xiaomimimo.com`) losing prior-turn reasoning on continuation requests. `convertAnthropicMessages` treated all unknown endpoints as signing-capable and demoted MiMo's unsigned `thinking` blocks to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). `isNonSigningAnthropicEndpoint` now recognizes the Xiaomi family (provider id `xiaomi` / `xiaomi-token-plan-*` and the `xiaomimimo.com` host suffix) so unsigned thinking blocks replay as `type: "thinking"`. + ## [15.9.67] - 2026-06-06 ### Fixed diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 6716952fc..b3b72c56d 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2428,18 +2428,30 @@ function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { /** * Returns true for providers whose Anthropic-compatible endpoints do NOT - * implement signature-based thinking-chain integrity (DeepSeek, Z.AI, etc.). - * For these providers, unsigned thinking blocks must be preserved as - * `type: "thinking"` instead of being degraded to text. + * implement signature-based thinking-chain integrity (DeepSeek, Z.AI, + * Xiaomi MiMo Token Plan, …). For these providers, unsigned `thinking` + * blocks emitted on prior assistant turns must be replayed as + * `type: "thinking"` instead of being degraded to `type: "text"` — the + * model relies on seeing its own reasoning chain back in the conversation + * to keep tool-call argument serialization stable (#2005). */ function isNonSigningAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { // Known non-signing providers if (model.provider === "zai" || model.provider === "deepseek") return true; + // Xiaomi MiMo (catalog `xiaomi` + every Token Plan region) exposes an + // Anthropic-compat endpoint that does not sign its `thinking` blocks. + // Match the provider-id pattern used in `openai-completions-compat.ts`. + if (model.provider === "xiaomi" || model.provider.startsWith("xiaomi-token-plan-")) return true; const baseUrl = model.baseUrl; if (!baseUrl) return false; try { const hostname = new URL(baseUrl).hostname.toLowerCase(); - return hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com"); + if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true; + // Cover user-defined providers pointed at any Xiaomi Token Plan host + // (e.g. `token-plan-sgp.xiaomimimo.com/anthropic`), matching the + // existing `xiaomimimo.com` detection in `append-only-context-mode.ts`. + if (hostname === "xiaomimimo.com" || hostname.endsWith(".xiaomimimo.com")) return true; + return false; } catch { return false; } diff --git a/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts b/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts new file mode 100644 index 000000000..aa4e075e1 --- /dev/null +++ b/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts @@ -0,0 +1,153 @@ +import { describe, expect, it } from "bun:test"; +import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; + +/** + * Regression: Xiaomi MiMo's Anthropic-compat endpoint + * (`token-plan-*.xiaomimimo.com/anthropic`, `api.xiaomimimo.com/anthropic`) + * emits `thinking` blocks WITHOUT a `signature`. Before #2005, the conversion + * layer treated any unknown Anthropic endpoint as signing-capable, so those + * unsigned thinking blocks got demoted to `type: "text"` on replay. With the + * reasoning chain stripped, MiMo's next continuation surfaced malformed tool + * arguments (e.g. `todo.ops` arriving as a JSON-string), triggering the + * downstream renderer crash and retry-spam reported in #2005. + * + * Now that `isNonSigningAnthropicEndpoint` recognizes the Xiaomi family, + * unsigned `thinking` blocks must round-trip as `{ type: "thinking", signature: "" }`. + */ +function makeXiaomiModel(overrides: Partial> = {}): Model<"anthropic-messages"> { + return { + api: "anthropic-messages", + provider: "xiaomi-token-plan-sgp", + id: "mimo-v2.5-pro", + name: "MiMo V2.5 Pro (Singapore)", + baseUrl: "https://token-plan-sgp.xiaomimimo.com/anthropic", + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + maxTokens: 131_072, + contextWindow: 1_048_576, + reasoning: true, + ...overrides, + }; +} + +function makeUser(text = "continue"): UserMessage { + return { role: "user", content: text, timestamp: 0 }; +} + +function makeAssistantThinking(thinking: string, tail: AssistantMessage["content"][number][] = []): AssistantMessage { + return { + role: "assistant", + content: [{ type: "thinking", thinking, thinkingSignature: "" }, ...tail], + api: "anthropic-messages", + provider: "xiaomi-token-plan-sgp", + model: "mimo-v2.5-pro", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: 0, + }; +} + +interface WireThinkingBlock { + type: "thinking"; + thinking: string; + signature: string; +} +interface WireTextBlock { + type: "text"; + text: string; +} +type WireBlock = WireThinkingBlock | WireTextBlock | { type: string; [key: string]: unknown }; + +function assistantWireBlocks(messages: Message[], model: Model<"anthropic-messages">): WireBlock[] { + const params = convertAnthropicMessages(messages, model, false); + const assistant = params.find(p => p.role === "assistant"); + return (assistant?.content as WireBlock[] | undefined) ?? []; +} + +describe("Xiaomi MiMo Anthropic-compat thinking replay (#2005)", () => { + it("preserves unsigned thinking blocks as type:'thinking' for xiaomi-token-plan-sgp", () => { + const blocks = assistantWireBlocks( + [ + makeUser("solve x"), + makeAssistantThinking("plan: read the file, then edit", [{ type: "text", text: "Sure." }]), + ], + makeXiaomiModel(), + ); + // Critical: a `thinking` block survives — without the fix it would have + // been demoted to a `text` block and the model would replay garbled args. + expect(blocks[0]).toEqual({ + type: "thinking", + thinking: "plan: read the file, then edit", + signature: "", + }); + expect(blocks[1]).toEqual({ type: "text", text: "Sure." }); + }); + + it("preserves unsigned thinking for every Token Plan region (ams, cn) and api.xiaomimimo.com", () => { + for (const baseUrl of [ + "https://token-plan-ams.xiaomimimo.com/anthropic", + "https://token-plan-cn.xiaomimimo.com/anthropic", + "https://api.xiaomimimo.com/anthropic", + ]) { + const model = makeXiaomiModel({ baseUrl, provider: "user-custom" }); + const blocks = assistantWireBlocks( + [makeUser(), makeAssistantThinking("hidden reasoning")], + model, + ); + expect(blocks[0]?.type).toBe("thinking"); + expect((blocks[0] as WireThinkingBlock).thinking).toBe("hidden reasoning"); + } + }); + + it("still degrades unsigned thinking to text for unknown signing-capable endpoints", () => { + // Sanity guard: don't flip the default for, e.g., api.anthropic.com. + const anthropicModel: Model<"anthropic-messages"> = { + ...makeXiaomiModel(), + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + id: "claude-sonnet-4-6", + }; + const blocks = assistantWireBlocks( + [makeUser(), makeAssistantThinking("internal scratch")], + anthropicModel, + ); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + }); + + it("keeps thinking → tool_use pairing intact across the conversion (continuation contract)", () => { + // The original failure mode: continuation request after a tool call. + // Thinking must precede `tool_use`, both must survive, and the + // `tool_result` must follow as the next user turn. + const toolResult: ToolResultMessage = { + role: "toolResult", + toolCallId: "toolu_xiaomi_1", + toolName: "read", + content: [{ type: "text", text: "file body" }], + isError: false, + timestamp: 0, + }; + const model = makeXiaomiModel(); + const messages: Message[] = [ + makeUser("read README"), + makeAssistantThinking("I need to call the read tool", [ + { type: "toolCall", id: "toolu_xiaomi_1", name: "read", arguments: { path: "README.md" } }, + ]), + toolResult, + ]; + const params = convertAnthropicMessages(messages, model, false); + expect(params.map(p => p.role)).toEqual(["user", "assistant", "user"]); + const assistantBlocks = params[1].content as WireBlock[]; + expect(assistantBlocks[0]?.type).toBe("thinking"); + expect(assistantBlocks[1]?.type).toBe("tool_use"); + expect((assistantBlocks[1] as { id: string }).id).toBe("toolu_xiaomi_1"); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 101c92e73..56b633357 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -15,7 +15,7 @@ - Fixed framed read results rendering with an extra blank row above and below the output block. - Fixed collapsed search result previews that could show only "… N more matches" when the first grouped section exceeded the preview budget. Collapsed search output now compacts to match rows, fills the budget with visible hits before the summary, and keeps truncation details out of the bottom user-visible notice. - Fixed boolean environment flag overrides that were ORed with settings, so `PI_INTENT_TRACING=0`, `PI_AUTO_QA=0`, and per-backend eval flags now take precedence when present while falling back to config when unset. -- Fixed the `todo` tool's TUI renderer crashing with `TypeError: args?.ops?.map is not a function` when a streaming tool-call delta surfaces a non-array `ops` field (mid-stream `parseStreamingJson` shapes like `{ ops: "[{" }`, or `[null]` entries before fields arrive). The renderer now treats non-array `ops`, non-object entries, and non-array `items` as missing structure instead of crashing, which also stops the spam-warn/retry cascade that followed each malformed delta ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). +- Fixed the `todo` tool's TUI renderer crashing with `TypeError: args?.ops?.map is not a function` when a streaming tool-call delta surfaced a non-array `ops` field (mid-stream `parseStreamingJson` shapes like `{ ops: "[{" }`, or `[null]` entries before fields arrive). The renderer now treats non-array `ops`, non-object entries, and non-array `items` as missing structure instead of crashing, which also stops the spam-warn cascade that followed each malformed delta. Paired with the Anthropic-side reasoning-replay fix in `packages/ai` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). ## [15.9.67] - 2026-06-06 ### Added From ae3477abfdf414cf7d8abe3216b7633ae7fcb2c0 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:20:33 +0000 Subject: [PATCH 027/181] style: bun run fix --- .../ai/test/anthropic-xiaomi-thinking-replay.test.ts | 12 +++--------- 1 file changed, 3 insertions(+), 9 deletions(-) diff --git a/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts b/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts index aa4e075e1..629113a05 100644 --- a/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts +++ b/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts @@ -98,10 +98,7 @@ describe("Xiaomi MiMo Anthropic-compat thinking replay (#2005)", () => { "https://api.xiaomimimo.com/anthropic", ]) { const model = makeXiaomiModel({ baseUrl, provider: "user-custom" }); - const blocks = assistantWireBlocks( - [makeUser(), makeAssistantThinking("hidden reasoning")], - model, - ); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("hidden reasoning")], model); expect(blocks[0]?.type).toBe("thinking"); expect((blocks[0] as WireThinkingBlock).thinking).toBe("hidden reasoning"); } @@ -115,10 +112,7 @@ describe("Xiaomi MiMo Anthropic-compat thinking replay (#2005)", () => { baseUrl: "https://api.anthropic.com", id: "claude-sonnet-4-6", }; - const blocks = assistantWireBlocks( - [makeUser(), makeAssistantThinking("internal scratch")], - anthropicModel, - ); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], anthropicModel); expect(blocks[0]?.type).toBe("text"); expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); }); @@ -148,6 +142,6 @@ describe("Xiaomi MiMo Anthropic-compat thinking replay (#2005)", () => { const assistantBlocks = params[1].content as WireBlock[]; expect(assistantBlocks[0]?.type).toBe("thinking"); expect(assistantBlocks[1]?.type).toBe("tool_use"); - expect((assistantBlocks[1] as { id: string }).id).toBe("toolu_xiaomi_1"); + expect((assistantBlocks[1] as unknown as { id: string }).id).toBe("toolu_xiaomi_1"); }); }); From d08e4ebb1f85789996536a5a1b063aac6bd02f94 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:29:43 +0000 Subject: [PATCH 028/181] fix(ai): generalize anthropic unsigned thinking replay Anthropic-compatible reasoning providers commonly emit thinking blocks without first-party Anthropic signatures while still expecting those blocks back as native thinking on continuation. The previous follow-up for #2005 fixed Xiaomi by provider/host allowlist, but the protocol contract is broader and matches the behavior described in #1996. Replace the Xiaomi-specific branch with a protocol-level rule: - official api.anthropic.com keeps demoting unsigned thinking to text - non-official anthropic-messages reasoning models replay unsigned thinking as type: thinking with an empty signature - existing known non-signing DeepSeek/Z.AI compatibility remains The regression test now uses a generic Anthropic-compatible reasoning endpoint, includes the Xiaomi MiMo reporter configuration only as a fixture, and guards non-reasoning unknown endpoints plus official Anthropic behavior. Fixes #2005 --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/providers/anthropic.ts | 51 +++--- ...anthropic-unsigned-thinking-replay.test.ts | 153 ++++++++++++++++++ .../anthropic-xiaomi-thinking-replay.test.ts | 147 ----------------- 4 files changed, 182 insertions(+), 171 deletions(-) create mode 100644 packages/ai/test/anthropic-unsigned-thinking-replay.test.ts delete mode 100644 packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index ba2e08298..bca3d6f7e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Xiaomi MiMo Anthropic-compat endpoints (`*.xiaomimimo.com/anthropic`, including every Token Plan region and `api.xiaomimimo.com`) losing prior-turn reasoning on continuation requests. `convertAnthropicMessages` treated all unknown endpoints as signing-capable and demoted MiMo's unsigned `thinking` blocks to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). `isNonSigningAnthropicEndpoint` now recognizes the Xiaomi family (provider id `xiaomi` / `xiaomi-token-plan-*` and the `xiaomimimo.com` host suffix) so unsigned thinking blocks replay as `type: "thinking"`. +- Fixed Anthropic-compatible reasoning endpoints losing prior-turn reasoning on continuation requests when they emit unsigned `thinking` blocks. `convertAnthropicMessages` treated unknown endpoints as signature-enforcing and demoted unsigned reasoning to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool. Official `api.anthropic.com` keeps the conservative text fallback; non-official `anthropic-messages` reasoning models now replay unsigned reasoning as native `type: "thinking"` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). ## [15.9.67] - 2026-06-06 diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index b3b72c56d..ac75d6798 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2426,37 +2426,42 @@ function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { } } -/** - * Returns true for providers whose Anthropic-compatible endpoints do NOT - * implement signature-based thinking-chain integrity (DeepSeek, Z.AI, - * Xiaomi MiMo Token Plan, …). For these providers, unsigned `thinking` - * blocks emitted on prior assistant turns must be replayed as - * `type: "thinking"` instead of being degraded to `type: "text"` — the - * model relies on seeing its own reasoning chain back in the conversation - * to keep tool-call argument serialization stable (#2005). - */ -function isNonSigningAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { - // Known non-signing providers - if (model.provider === "zai" || model.provider === "deepseek") return true; - // Xiaomi MiMo (catalog `xiaomi` + every Token Plan region) exposes an - // Anthropic-compat endpoint that does not sign its `thinking` blocks. - // Match the provider-id pattern used in `openai-completions-compat.ts`. - if (model.provider === "xiaomi" || model.provider.startsWith("xiaomi-token-plan-")) return true; +function isOfficialAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { const baseUrl = model.baseUrl; if (!baseUrl) return false; try { const hostname = new URL(baseUrl).hostname.toLowerCase(); - if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true; - // Cover user-defined providers pointed at any Xiaomi Token Plan host - // (e.g. `token-plan-sgp.xiaomimimo.com/anthropic`), matching the - // existing `xiaomimimo.com` detection in `append-only-context-mode.ts`. - if (hostname === "xiaomimimo.com" || hostname.endsWith(".xiaomimimo.com")) return true; - return false; + return hostname === "api.anthropic.com"; } catch { return false; } } +/** + * Returns true when unsigned `thinking` blocks from prior assistant turns should + * be replayed as Anthropic-native thinking instead of demoted to text. + * + * Official Anthropic enforces signature-based thinking-chain integrity, so + * unsigned blocks must remain text there. Anthropic-compatible reasoning + * endpoints commonly emit unsigned thinking blocks while still expecting those + * blocks back as `type: "thinking"` on continuation; demoting them loses the + * model's reasoning chain and can destabilize the next tool-call arguments + * (#2005). Known non-signing hosts are also preserved for compatibility. + */ +function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">): boolean { + if (model.provider === "zai" || model.provider === "deepseek") return true; + const baseUrl = model.baseUrl; + if (baseUrl) { + try { + const hostname = new URL(baseUrl).hostname.toLowerCase(); + if (hostname === "api.deepseek.com" || hostname.endsWith(".deepseek.com")) return true; + } catch { + // Fall through to the protocol-level reasoning rule below. + } + } + return model.reasoning && !isOfficialAnthropicEndpoint(model); +} + function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam { const block: ContentBlockParam = { type: "tool_result", @@ -2545,7 +2550,7 @@ export function convertAnthropicMessages( } if (block.thinking.trim().length === 0) continue; if (!block.thinkingSignature || block.thinkingSignature.trim().length === 0) { - if (isNonSigningAnthropicEndpoint(model)) { + if (shouldReplayUnsignedThinking(model)) { blocks.push({ type: "thinking", thinking: block.thinking.toWellFormed(), diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts new file mode 100644 index 000000000..7e33e859f --- /dev/null +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -0,0 +1,153 @@ +import { describe, expect, it } from "bun:test"; +import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; + +/** + * Regression: Anthropic-compatible reasoning endpoints often emit `thinking` + * blocks without a first-party Anthropic signature, but still expect those + * blocks back as native `type: "thinking"` on continuation. Demoting unsigned + * thinking to text strips the reasoning chain and can destabilize follow-up + * tool-call argument serialization (the upstream cause behind #2005's `todo` + * renderer crash). + * + * Official Anthropic remains conservative: unsigned thinking is demoted to text + * there because the first-party API enforces signature-based integrity. + */ +function makeModel(overrides: Partial> = {}): Model<"anthropic-messages"> { + return { + api: "anthropic-messages", + provider: "custom-anthropic", + id: "reasoning-model", + name: "Reasoning Anthropic-Compatible Model", + baseUrl: "https://llm.example.com/anthropic", + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + maxTokens: 8_192, + contextWindow: 200_000, + reasoning: true, + ...overrides, + }; +} + +function makeUser(text = "continue"): UserMessage { + return { role: "user", content: text, timestamp: 0 }; +} + +function makeAssistantThinking(thinking: string, tail: AssistantMessage["content"][number][] = []): AssistantMessage { + return { + role: "assistant", + content: [{ type: "thinking", thinking, thinkingSignature: "" }, ...tail], + api: "anthropic-messages", + provider: "custom-anthropic", + model: "reasoning-model", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: 0, + }; +} + +interface WireThinkingBlock { + type: "thinking"; + thinking: string; + signature: string; +} +interface WireTextBlock { + type: "text"; + text: string; +} +interface WireToolUseBlock { + type: "tool_use"; + id: string; + name: string; + input: Record; +} +type WireBlock = WireThinkingBlock | WireTextBlock | WireToolUseBlock | { type: string; [key: string]: unknown }; + +function assistantWireBlocks(messages: Message[], model: Model<"anthropic-messages">): WireBlock[] { + const params = convertAnthropicMessages(messages, model, false); + const assistant = params.find(p => p.role === "assistant"); + return (assistant?.content as WireBlock[] | undefined) ?? []; +} + +describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { + it("preserves unsigned thinking for non-official reasoning endpoints", () => { + const blocks = assistantWireBlocks( + [ + makeUser("solve x"), + makeAssistantThinking("plan: read the file, then edit", [{ type: "text", text: "Sure." }]), + ], + makeModel(), + ); + expect(blocks[0]).toEqual({ + type: "thinking", + thinking: "plan: read the file, then edit", + signature: "", + }); + expect(blocks[1]).toEqual({ type: "text", text: "Sure." }); + }); + + it("covers the Xiaomi MiMo Anthropic-compatible reporter configuration without provider allowlists", () => { + const model = makeModel({ + provider: "user-custom", + id: "mimo-v2.5-pro", + name: "MiMo V2.5 Pro (Singapore)", + baseUrl: "https://token-plan-sgp.xiaomimimo.com/anthropic", + maxTokens: 131_072, + contextWindow: 1_048_576, + }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("hidden reasoning")], model); + expect(blocks[0]).toEqual({ type: "thinking", thinking: "hidden reasoning", signature: "" }); + }); + + it("preserves legacy known non-signing endpoints even if model.reasoning is false", () => { + const model = makeModel({ provider: "custom", baseUrl: "https://api.deepseek.com/v1", reasoning: false }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("deepseek reasoning")], model); + expect(blocks[0]?.type).toBe("thinking"); + }); + + it("still degrades unsigned thinking to text for official Anthropic", () => { + const model = makeModel({ provider: "anthropic", baseUrl: "https://api.anthropic.com" }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + }); + + it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => { + const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" }); + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("scratch"); + }); + + it("keeps thinking → tool_use pairing intact across continuation conversion", () => { + const toolResult: ToolResultMessage = { + role: "toolResult", + toolCallId: "toolu_reasoning_1", + toolName: "read", + content: [{ type: "text", text: "file body" }], + isError: false, + timestamp: 0, + }; + const model = makeModel(); + const messages: Message[] = [ + makeUser("read README"), + makeAssistantThinking("I need to call the read tool", [ + { type: "toolCall", id: "toolu_reasoning_1", name: "read", arguments: { path: "README.md" } }, + ]), + toolResult, + ]; + const params = convertAnthropicMessages(messages, model, false); + expect(params.map(p => p.role)).toEqual(["user", "assistant", "user"]); + const assistantBlocks = params[1].content as WireBlock[]; + expect(assistantBlocks[0]?.type).toBe("thinking"); + expect(assistantBlocks[1]?.type).toBe("tool_use"); + expect((assistantBlocks[1] as WireToolUseBlock).id).toBe("toolu_reasoning_1"); + }); +}); diff --git a/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts b/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts deleted file mode 100644 index 629113a05..000000000 --- a/packages/ai/test/anthropic-xiaomi-thinking-replay.test.ts +++ /dev/null @@ -1,147 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic"; -import type { AssistantMessage, Message, Model, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai/types"; - -/** - * Regression: Xiaomi MiMo's Anthropic-compat endpoint - * (`token-plan-*.xiaomimimo.com/anthropic`, `api.xiaomimimo.com/anthropic`) - * emits `thinking` blocks WITHOUT a `signature`. Before #2005, the conversion - * layer treated any unknown Anthropic endpoint as signing-capable, so those - * unsigned thinking blocks got demoted to `type: "text"` on replay. With the - * reasoning chain stripped, MiMo's next continuation surfaced malformed tool - * arguments (e.g. `todo.ops` arriving as a JSON-string), triggering the - * downstream renderer crash and retry-spam reported in #2005. - * - * Now that `isNonSigningAnthropicEndpoint` recognizes the Xiaomi family, - * unsigned `thinking` blocks must round-trip as `{ type: "thinking", signature: "" }`. - */ -function makeXiaomiModel(overrides: Partial> = {}): Model<"anthropic-messages"> { - return { - api: "anthropic-messages", - provider: "xiaomi-token-plan-sgp", - id: "mimo-v2.5-pro", - name: "MiMo V2.5 Pro (Singapore)", - baseUrl: "https://token-plan-sgp.xiaomimimo.com/anthropic", - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - maxTokens: 131_072, - contextWindow: 1_048_576, - reasoning: true, - ...overrides, - }; -} - -function makeUser(text = "continue"): UserMessage { - return { role: "user", content: text, timestamp: 0 }; -} - -function makeAssistantThinking(thinking: string, tail: AssistantMessage["content"][number][] = []): AssistantMessage { - return { - role: "assistant", - content: [{ type: "thinking", thinking, thinkingSignature: "" }, ...tail], - api: "anthropic-messages", - provider: "xiaomi-token-plan-sgp", - model: "mimo-v2.5-pro", - usage: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "toolUse", - timestamp: 0, - }; -} - -interface WireThinkingBlock { - type: "thinking"; - thinking: string; - signature: string; -} -interface WireTextBlock { - type: "text"; - text: string; -} -type WireBlock = WireThinkingBlock | WireTextBlock | { type: string; [key: string]: unknown }; - -function assistantWireBlocks(messages: Message[], model: Model<"anthropic-messages">): WireBlock[] { - const params = convertAnthropicMessages(messages, model, false); - const assistant = params.find(p => p.role === "assistant"); - return (assistant?.content as WireBlock[] | undefined) ?? []; -} - -describe("Xiaomi MiMo Anthropic-compat thinking replay (#2005)", () => { - it("preserves unsigned thinking blocks as type:'thinking' for xiaomi-token-plan-sgp", () => { - const blocks = assistantWireBlocks( - [ - makeUser("solve x"), - makeAssistantThinking("plan: read the file, then edit", [{ type: "text", text: "Sure." }]), - ], - makeXiaomiModel(), - ); - // Critical: a `thinking` block survives — without the fix it would have - // been demoted to a `text` block and the model would replay garbled args. - expect(blocks[0]).toEqual({ - type: "thinking", - thinking: "plan: read the file, then edit", - signature: "", - }); - expect(blocks[1]).toEqual({ type: "text", text: "Sure." }); - }); - - it("preserves unsigned thinking for every Token Plan region (ams, cn) and api.xiaomimimo.com", () => { - for (const baseUrl of [ - "https://token-plan-ams.xiaomimimo.com/anthropic", - "https://token-plan-cn.xiaomimimo.com/anthropic", - "https://api.xiaomimimo.com/anthropic", - ]) { - const model = makeXiaomiModel({ baseUrl, provider: "user-custom" }); - const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("hidden reasoning")], model); - expect(blocks[0]?.type).toBe("thinking"); - expect((blocks[0] as WireThinkingBlock).thinking).toBe("hidden reasoning"); - } - }); - - it("still degrades unsigned thinking to text for unknown signing-capable endpoints", () => { - // Sanity guard: don't flip the default for, e.g., api.anthropic.com. - const anthropicModel: Model<"anthropic-messages"> = { - ...makeXiaomiModel(), - provider: "anthropic", - baseUrl: "https://api.anthropic.com", - id: "claude-sonnet-4-6", - }; - const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], anthropicModel); - expect(blocks[0]?.type).toBe("text"); - expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); - }); - - it("keeps thinking → tool_use pairing intact across the conversion (continuation contract)", () => { - // The original failure mode: continuation request after a tool call. - // Thinking must precede `tool_use`, both must survive, and the - // `tool_result` must follow as the next user turn. - const toolResult: ToolResultMessage = { - role: "toolResult", - toolCallId: "toolu_xiaomi_1", - toolName: "read", - content: [{ type: "text", text: "file body" }], - isError: false, - timestamp: 0, - }; - const model = makeXiaomiModel(); - const messages: Message[] = [ - makeUser("read README"), - makeAssistantThinking("I need to call the read tool", [ - { type: "toolCall", id: "toolu_xiaomi_1", name: "read", arguments: { path: "README.md" } }, - ]), - toolResult, - ]; - const params = convertAnthropicMessages(messages, model, false); - expect(params.map(p => p.role)).toEqual(["user", "assistant", "user"]); - const assistantBlocks = params[1].content as WireBlock[]; - expect(assistantBlocks[0]?.type).toBe("thinking"); - expect(assistantBlocks[1]?.type).toBe("tool_use"); - expect((assistantBlocks[1] as unknown as { id: string }).id).toBe("toolu_xiaomi_1"); - }); -}); From 5046f0f47c58d2004db86f6cdefc4e3cd432c486 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:34:17 +0000 Subject: [PATCH 029/181] fix(ai): treat missing anthropic baseUrl as official in thinking replay MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit resolveAnthropicBaseUrl defaults to https://api.anthropic.com when model.baseUrl is absent (e.g. same-id custom overrides that only tweak metadata), and the existing isAnthropicApiBaseUrl helper already treats an empty/undefined baseUrl as official. The new isOfficialAnthropicEndpoint helper classified the same model as non-official, so shouldReplayUnsignedThinking would replay unsigned thinking as type: thinking against the first-party API — which rejects it. Drop the redundant helper and use isAnthropicApiBaseUrl as the single source of truth. Add a regression test that pins the missing-baseUrl case to the text fallback. --- packages/ai/src/providers/anthropic.ts | 28 +++++++------------ ...anthropic-unsigned-thinking-replay.test.ts | 11 ++++++++ 2 files changed, 21 insertions(+), 18 deletions(-) diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index ac75d6798..f96477f4c 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2426,27 +2426,19 @@ function isZaiAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { } } -function isOfficialAnthropicEndpoint(model: Model<"anthropic-messages">): boolean { - const baseUrl = model.baseUrl; - if (!baseUrl) return false; - try { - const hostname = new URL(baseUrl).hostname.toLowerCase(); - return hostname === "api.anthropic.com"; - } catch { - return false; - } -} - /** * Returns true when unsigned `thinking` blocks from prior assistant turns should * be replayed as Anthropic-native thinking instead of demoted to text. * - * Official Anthropic enforces signature-based thinking-chain integrity, so - * unsigned blocks must remain text there. Anthropic-compatible reasoning - * endpoints commonly emit unsigned thinking blocks while still expecting those - * blocks back as `type: "thinking"` on continuation; demoting them loses the - * model's reasoning chain and can destabilize the next tool-call arguments - * (#2005). Known non-signing hosts are also preserved for compatibility. + * Official Anthropic (matched via `isAnthropicApiBaseUrl`, which intentionally + * treats a missing baseUrl as official since `resolveAnthropicBaseUrl` routes + * it to `https://api.anthropic.com`) enforces signature-based thinking-chain + * integrity, so unsigned blocks must remain text there. Anthropic-compatible + * reasoning endpoints commonly emit unsigned thinking blocks while still + * expecting them back as `type: "thinking"` on continuation; demoting them + * loses the model's reasoning chain and can destabilize the next tool-call + * arguments (#2005). Known non-signing hosts are also preserved for + * compatibility. */ function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">): boolean { if (model.provider === "zai" || model.provider === "deepseek") return true; @@ -2459,7 +2451,7 @@ function shouldReplayUnsignedThinking(model: Model<"anthropic-messages">): boole // Fall through to the protocol-level reasoning rule below. } } - return model.reasoning && !isOfficialAnthropicEndpoint(model); + return model.reasoning && !isAnthropicApiBaseUrl(baseUrl); } function buildToolResultBlock(model: Model<"anthropic-messages">, msg: ToolResultMessage): ContentBlockParam { diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts index 7e33e859f..9b3f8dcbc 100644 --- a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -119,6 +119,17 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); }); + it("treats a missing baseUrl as official Anthropic (resolveAnthropicBaseUrl default)", () => { + // `isAnthropicApiBaseUrl(undefined) === true` because the actual HTTP + // dispatch falls back to https://api.anthropic.com. Same-id custom + // overrides that only tweak model metadata (no baseUrl override) must + // not regress to native-thinking replay against the first-party API. + const model = { ...makeModel(), provider: "anthropic", baseUrl: "" }; + const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("internal scratch")], model); + expect(blocks[0]?.type).toBe("text"); + expect((blocks[0] as WireTextBlock).text).toBe("internal scratch"); + }); + it("still degrades unsigned thinking to text for non-reasoning unknown endpoints", () => { const model = makeModel({ reasoning: false, baseUrl: "https://plain.example.com/anthropic" }); const blocks = assistantWireBlocks([makeUser(), makeAssistantThinking("scratch")], model); From c933d34398b8005f042aae2f4e6aec457a642e8b Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 19:22:21 +0200 Subject: [PATCH 030/181] =?UTF-8?q?fix(usage):=20antigravity=20/usage=20di?= =?UTF-8?q?splay=20=E2=80=94=20dedupe=20by=20tier,=20fix=20account=20count?= =?UTF-8?q?,=20add=20projectId=20identity?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Antigravity usage provider now deduplicates model quota entries by tier instead of emitting one bar per model (15+ redundant bars for one account). The upstream API groups quota by tier — models within the same tier share the same quota bucket, so per-model bars were misleading noise. - Reports now carry credential email and accountId in metadata so the /usage display and deduplicator can show meaningful account identities instead of 'account 1'. - formatAggregateAmount no longer uses limits.length as account count. Instead counts unique accountId values from limit scopes — a single account's N incomplete limits no longer display as 'N accts'. - Usage report dedup now considers metadata.projectId for Google Cloud providers so duplicate credential rows with the same project merge. - account labels in both TUI and ACP markdown paths now fall back to metadata.projectId before the generic 'account N' placeholder. --- packages/ai/CHANGELOG.md | 6 ++ packages/ai/src/auth-storage.ts | 2 + packages/ai/src/usage/google-antigravity.ts | 78 +++++++++++++------ packages/coding-agent/CHANGELOG.md | 4 + .../modes/controllers/command-controller.ts | 11 ++- .../slash-commands/helpers/usage-report.ts | 2 + 6 files changed, 79 insertions(+), 24 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2f8dea9f7..cbdaf6c5e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,12 @@ ## [Unreleased] +### Fixed + +- Fixed Antigravity usage provider emitting one bar per model instead of deduplicating by tier — a single account's 15+ model entries now collapse to one bar per tier, matching the shared-quota reality of the upstream API. +- Fixed Antigravity usage reports missing `email` and `accountId` in metadata, so the `/usage` display and the deduplicator can associate reports with their credentials. +- Fixed usage-report dedup ignoring `projectId` for Google Cloud providers, preventing duplicate credential entries from being recognized as the same account. + ## [15.9.67] - 2026-06-06 ### Fixed diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index f03f7b810..35c9a8929 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -2295,6 +2295,8 @@ export class AuthStorage { if (report.provider === "openai-codex" || report.provider === "anthropic") { return identifiers.map(identifier => `${report.provider}:${identifier.toLowerCase()}`); } + const projectId = this.#getUsageReportMetadataValue(report, "projectId"); + if (projectId) identifiers.push(`project:${projectId}`); const accountId = this.#getUsageReportMetadataValue(report, "accountId"); if (accountId) identifiers.push(`account:${accountId}`); const account = this.#getUsageReportMetadataValue(report, "account"); diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index 26963d7b5..0a5ad1a2f 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -148,46 +148,78 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo } const data = (await response.json()) as AntigravityUsageResponse; - const limits: UsageLimit[] = []; + + // The API returns per-model quota entries, but quota is shared across + // models within the same tier. Deduplicate by (tier, windowId) so one + // account doesn't produce 15 redundant bars. + const deduped = new Map< + string, + { amount: UsageAmount; window: UsageWindow | undefined; tier: string | undefined } + >(); let earliestReset: number | undefined; - for (const [modelId, modelInfo] of Object.entries(data.models ?? {})) { + for (const [_modelId, modelInfo] of Object.entries(data.models ?? {})) { const quotaInfos = normalizeQuotaInfos(modelInfo); for (const quotaInfo of quotaInfos) { + if (quotaInfo.remainingFraction === undefined) continue; const amount = buildAmount(quotaInfo); const window = parseWindow(quotaInfo); if (window?.resetsAt) { earliestReset = earliestReset ? Math.min(earliestReset, window.resetsAt) : window.resetsAt; } - const labelBase = modelInfo.displayName || modelId; - const label = quotaInfo.tier ? `${labelBase} (${quotaInfo.tier})` : labelBase; + const tier = quotaInfo.tier ?? "default"; const windowId = window?.id ?? "default"; - limits.push({ - id: `${modelId}:${quotaInfo.tier ?? "default"}:${windowId}`, - label, - scope: { - provider: params.provider, - accountId: credential.accountId, - projectId: credential.projectId, - modelId, - tier: quotaInfo.tier, - windowId, - }, - window, - amount, - status: getUsageStatus(amount.remainingFraction), - }); + const key = `${tier}|${windowId}`; + const existing = deduped.get(key); + if ( + !existing || + (existing.amount.remainingFraction !== undefined && + amount.remainingFraction !== undefined && + amount.remainingFraction < existing.amount.remainingFraction) + ) { + deduped.set(key, { amount, window, tier: quotaInfo.tier }); + } } } + const limits: UsageLimit[] = []; + for (const [key, entry] of deduped) { + const [tier, windowId] = key.split("|") as [string, string]; + const label = entry.tier ?? "Usage"; + limits.push({ + id: `google-antigravity:${tier}:${windowId}`, + label, + scope: { + provider: params.provider, + accountId: credential.accountId, + projectId: credential.projectId, + tier: entry.tier ?? undefined, + windowId, + }, + window: entry.window, + amount: entry.amount, + status: getUsageStatus(entry.amount.remainingFraction), + }); + } + + limits.sort((a, b) => { + const aFraction = a.amount.remainingFraction ?? 1; + const bFraction = b.amount.remainingFraction ?? 1; + return aFraction - bFraction; + }); + + const metadata: UsageReport["metadata"] = { + endpoint: url, + projectId: credential.projectId, + }; + if (credential.email) metadata.email = credential.email; + if (credential.accountId) metadata.accountId = credential.accountId; + const report: UsageReport = { provider: params.provider, fetchedAt: nowMs, limits, - metadata: { - endpoint: url, - projectId: credential.projectId, - }, + metadata, raw: data, }; diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 33ff5aaaf..a90e895a4 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -3,6 +3,10 @@ ## [Unreleased] ## [15.9.69] - 2026-06-06 +### Fixed + +- Fixed `/usage` aggregate amount fallback using raw `limits.length` as account count — now counts unique `accountId` values from limit scopes, so N limits from a single account no longer display as "N accts". +- Fixed `/usage` account labeling falling back to "account N" for providers that use `projectId` as their primary identity (e.g. Google Antigravity, Gemini CLI) — `projectId` from report metadata is now considered before the generic fallback. ### Added diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index f1f857ecb..8d9388597 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -1272,6 +1272,8 @@ function formatAccountLabel(limit: UsageLimit, report: UsageReport, index: numbe if (email) return email; const accountId = (report.metadata?.accountId as string | undefined) ?? limit.scope.accountId; if (accountId) return accountId; + const projectId = report.metadata?.projectId as string | undefined; + if (projectId) return projectId; return `account ${index + 1}`; } @@ -1280,6 +1282,8 @@ function formatUnlimitedReportLabel(report: UsageReport, index: number): string if (email) return email; const accountId = report.metadata?.accountId as string | undefined; if (accountId) return accountId; + const projectId = report.metadata?.projectId as string | undefined; + if (projectId) return projectId; return `account ${index + 1}`; } @@ -1365,7 +1369,12 @@ function formatAggregateAmount(limits: UsageLimit[]): string { return `${formatNumber(remainingPct)}% free`; } - return `${limits.length} accts`; + // Count unique accounts from limit scopes — not limits.length. + const uniqueAccountIds = new Set( + limits.map(limit => limit.scope.accountId).filter((id): id is string => typeof id === "string" && id.length > 0), + ); + if (uniqueAccountIds.size === 0) return ""; + return `${uniqueAccountIds.size} ${uniqueAccountIds.size === 1 ? "acct" : "accts"}`; } function resolveResetRange(limits: UsageLimit[], nowMs: number): string | null { diff --git a/packages/coding-agent/src/slash-commands/helpers/usage-report.ts b/packages/coding-agent/src/slash-commands/helpers/usage-report.ts index bbef96730..71f184550 100644 --- a/packages/coding-agent/src/slash-commands/helpers/usage-report.ts +++ b/packages/coding-agent/src/slash-commands/helpers/usage-report.ts @@ -26,6 +26,8 @@ function formatUsageReportAccount(report: UsageReport, limit: UsageLimit, index: if (typeof email === "string" && email) return email; const accountId = report.metadata?.accountId ?? limit.scope.accountId; if (typeof accountId === "string" && accountId) return accountId; + const projectId = report.metadata?.projectId; + if (typeof projectId === "string" && projectId) return projectId; return `account ${index + 1}`; } From ca24f34044a1a10c991b92927679b635511b0fd8 Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 19:47:28 +0200 Subject: [PATCH 031/181] =?UTF-8?q?fix(usage):=20merge=20antigravity=20ded?= =?UTF-8?q?up=20entries=20=E2=80=94=20keep=20bar=20data=20+=20reset=20time?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When models within the same (tier, windowId) group have complementary data — some carry remainingFraction but no resetTime, others carry resetTime but no remainingFraction — merge them so the displayed entry has both a real bar and the 'resets in…' line. Also lowercases tier names for dedup keys so 'Default' and 'default' are recognized as the same tier. Adds projectId to OAuth credential identity extraction and to the usage-report dedup identifiers so duplicate credential rows (same Google Cloud project, separate login sessions) are pruned and merged at both the store and usage-report levels. --- packages/ai/src/auth-storage.ts | 4 +++ packages/ai/src/usage/google-antigravity.ts | 35 ++++++++++++++++----- 2 files changed, 31 insertions(+), 8 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 35c9a8929..589b5a3e5 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3933,6 +3933,8 @@ function resolveProviderCredentialIdentityKey(provider: string, identifiers: str if ((provider === "openai-codex" || provider === "anthropic") && emailIdentifier) return emailIdentifier; const accountIdentifier = identifiers.find(identifier => identifier.startsWith("account:")); if (accountIdentifier) return accountIdentifier; + const projectIdentifier = identifiers.find(identifier => identifier.startsWith("project:")); + if (projectIdentifier) return projectIdentifier; if (emailIdentifier) return emailIdentifier; return null; } @@ -3969,6 +3971,8 @@ function extractOAuthCredentialIdentifiers(credential: OAuthCredential): string[ if (accountId) identifiers.add(`account:${accountId}`); const email = normalizeStoredEmail(credential.email); if (email) identifiers.add(`email:${email}`); + const projectId = normalizeStoredAccountId(credential.projectId); + if (projectId) identifiers.add(`project:${projectId}`); const accessIdentifiers = extractOAuthTokenIdentifiers(credential.access) ?? []; for (const identifier of accessIdentifiers) { identifiers.add(identifier); diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index 0a5ad1a2f..42bce9164 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -161,24 +161,43 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo for (const [_modelId, modelInfo] of Object.entries(data.models ?? {})) { const quotaInfos = normalizeQuotaInfos(modelInfo); for (const quotaInfo of quotaInfos) { - if (quotaInfo.remainingFraction === undefined) continue; const amount = buildAmount(quotaInfo); const window = parseWindow(quotaInfo); if (window?.resetsAt) { earliestReset = earliestReset ? Math.min(earliestReset, window.resetsAt) : window.resetsAt; } - const tier = quotaInfo.tier ?? "default"; + const tier = (quotaInfo.tier ?? "default").toLowerCase(); const windowId = window?.id ?? "default"; const key = `${tier}|${windowId}`; const existing = deduped.get(key); - if ( - !existing || - (existing.amount.remainingFraction !== undefined && - amount.remainingFraction !== undefined && - amount.remainingFraction < existing.amount.remainingFraction) - ) { + if (!existing) { deduped.set(key, { amount, window, tier: quotaInfo.tier }); + continue; } + // Merge: keep the entry with fraction data for the bar, but + // also keep any window with a reset time so "resets in…" survives. + const eFrac = existing.amount.remainingFraction; + const cFrac = amount.remainingFraction; + const eHasFrac = eFrac !== undefined; + const cHasFrac = cFrac !== undefined; + + let bestAmount = existing.amount; + let bestWindow = existing.window?.resetsAt ? existing.window : (window ?? existing.window); + let bestTier = existing.tier ?? quotaInfo.tier; + + if (!eHasFrac && cHasFrac) { + bestAmount = amount; + bestTier = quotaInfo.tier ?? existing.tier; + } else if (eHasFrac && cHasFrac && cFrac! < eFrac!) { + bestAmount = amount; + bestTier = quotaInfo.tier ?? existing.tier; + } + // Always merge in window with reset time if the current + // best doesn't have one. + if (!bestWindow?.resetsAt && window?.resetsAt) { + bestWindow = window; + } + deduped.set(key, { amount: bestAmount, window: bestWindow, tier: bestTier }); } } From 15c0dff28e973326beb7816acf5bd6ce0a046dc6 Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 19:55:26 +0200 Subject: [PATCH 032/181] fix(usage): fall back to limit.scope.projectId when metadata.projectId is absent Gemini CLI provider stores projectId on limit.scope but not in report metadata, so the metadata-only projectId fallback added earlier missed that case. Now all three lookup sites (dedup identifiers, TUI account label, ACP account label) also check limit.scope.projectId. --- packages/ai/src/auth-storage.ts | 15 +++++++++++++-- .../src/modes/controllers/command-controller.ts | 2 +- .../src/slash-commands/helpers/usage-report.ts | 2 +- 3 files changed, 15 insertions(+), 4 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 589b5a3e5..136f6ceae 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -2288,6 +2288,16 @@ export class AuthStorage { return undefined; } + #getUsageReportScopeProjectId(report: UsageReport): string | undefined { + const ids = new Set(); + for (const limit of report.limits) { + const projectId = limit.scope.projectId?.trim(); + if (projectId) ids.add(projectId); + } + if (ids.size === 1) return [...ids][0]; + return undefined; + } + #getUsageReportIdentifiers(report: UsageReport): string[] { const identifiers: string[] = []; const email = this.#getUsageReportMetadataValue(report, "email"); @@ -2295,9 +2305,10 @@ export class AuthStorage { if (report.provider === "openai-codex" || report.provider === "anthropic") { return identifiers.map(identifier => `${report.provider}:${identifier.toLowerCase()}`); } - const projectId = this.#getUsageReportMetadataValue(report, "projectId"); - if (projectId) identifiers.push(`project:${projectId}`); + const projectId = + this.#getUsageReportMetadataValue(report, "projectId") ?? this.#getUsageReportScopeProjectId(report); const accountId = this.#getUsageReportMetadataValue(report, "accountId"); + if (projectId) identifiers.push(`project:${projectId}`); if (accountId) identifiers.push(`account:${accountId}`); const account = this.#getUsageReportMetadataValue(report, "account"); if (account) identifiers.push(`account:${account}`); diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index 8d9388597..759e99dfa 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -1272,7 +1272,7 @@ function formatAccountLabel(limit: UsageLimit, report: UsageReport, index: numbe if (email) return email; const accountId = (report.metadata?.accountId as string | undefined) ?? limit.scope.accountId; if (accountId) return accountId; - const projectId = report.metadata?.projectId as string | undefined; + const projectId = (report.metadata?.projectId as string | undefined) ?? limit.scope.projectId; if (projectId) return projectId; return `account ${index + 1}`; } diff --git a/packages/coding-agent/src/slash-commands/helpers/usage-report.ts b/packages/coding-agent/src/slash-commands/helpers/usage-report.ts index 71f184550..8b3610f9d 100644 --- a/packages/coding-agent/src/slash-commands/helpers/usage-report.ts +++ b/packages/coding-agent/src/slash-commands/helpers/usage-report.ts @@ -26,7 +26,7 @@ function formatUsageReportAccount(report: UsageReport, limit: UsageLimit, index: if (typeof email === "string" && email) return email; const accountId = report.metadata?.accountId ?? limit.scope.accountId; if (typeof accountId === "string" && accountId) return accountId; - const projectId = report.metadata?.projectId; + const projectId = report.metadata?.projectId ?? limit.scope.projectId; if (typeof projectId === "string" && projectId) return projectId; return `account ${index + 1}`; } From 794a64aae13c350b12651280ef1728bf3de0836b Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 20:01:17 +0200 Subject: [PATCH 033/181] =?UTF-8?q?fix(usage):=20address=20review=20?= =?UTF-8?q?=E2=80=94=20order=20email=20before=20project,=20add=20tests,=20?= =?UTF-8?q?nits?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - BLOCKING: reorder resolveProviderCredentialIdentityKey so email identity takes priority over project — two users with different emails on the same GCP project no longer get merged/hard-deleted. - Added #getUsageReportScopeProjectId helper so Gemini CLI reports (which set projectId on limit.scope but not metadata) still get dedup coverage. Both metadata and scope projectId paths checked. - formatAggregateAmount now falls back to limits.length when no scope.accountId values are present, preserving pre-existing behaviour for providers that don't set accountId on limits. - Added 9 contract tests for the antigravity usage merge logic: tier dedup, worst-fraction-wins, mixed-case collapsing, reset-time-from-other-entry, windowId separation, metadata, sort order, and null-on-no-project. - Nits: label='Usage' (so formatLimitTitle renders 'Usage (Default)' not bare 'Default'), id uses params.provider instead of hardcoded string, tier field drops redundant ?? undefined. --- packages/ai/src/auth-storage.ts | 2 +- packages/ai/src/usage/google-antigravity.ts | 6 +- .../ai/test/google-antigravity-usage.test.ts | 189 ++++++++++++++++++ .../modes/controllers/command-controller.ts | 6 +- 4 files changed, 197 insertions(+), 6 deletions(-) create mode 100644 packages/ai/test/google-antigravity-usage.test.ts diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 136f6ceae..86b627da7 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -3944,9 +3944,9 @@ function resolveProviderCredentialIdentityKey(provider: string, identifiers: str if ((provider === "openai-codex" || provider === "anthropic") && emailIdentifier) return emailIdentifier; const accountIdentifier = identifiers.find(identifier => identifier.startsWith("account:")); if (accountIdentifier) return accountIdentifier; + if (emailIdentifier) return emailIdentifier; const projectIdentifier = identifiers.find(identifier => identifier.startsWith("project:")); if (projectIdentifier) return projectIdentifier; - if (emailIdentifier) return emailIdentifier; return null; } diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index 42bce9164..ef3b59a65 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -204,15 +204,15 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo const limits: UsageLimit[] = []; for (const [key, entry] of deduped) { const [tier, windowId] = key.split("|") as [string, string]; - const label = entry.tier ?? "Usage"; + const label = "Usage"; limits.push({ - id: `google-antigravity:${tier}:${windowId}`, + id: `${params.provider}:${tier}:${windowId}`, label, scope: { provider: params.provider, accountId: credential.accountId, projectId: credential.projectId, - tier: entry.tier ?? undefined, + tier: entry.tier, windowId, }, window: entry.window, diff --git a/packages/ai/test/google-antigravity-usage.test.ts b/packages/ai/test/google-antigravity-usage.test.ts new file mode 100644 index 000000000..e26a09449 --- /dev/null +++ b/packages/ai/test/google-antigravity-usage.test.ts @@ -0,0 +1,189 @@ +/** + * Antigravity usage provider contract tests. The merge logic + * deduplicates per-model quota entries by (tier, windowId), + * preserves reset times when bar data and window data come from + * different model entries, and handles mixed-case tier names. + */ +import { describe, expect, it } from "bun:test"; +import { antigravityUsageProvider } from "../src/usage/google-antigravity"; +import type { UsageFetchParams, UsageFetchContext } from "../src/usage"; + +const accessTokenFixture = (() => { + const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url"); + const body = Buffer.from( + JSON.stringify({ sub: "user-fixture" }), + ).toString("base64url"); + return `${header}.${body}.sig`; +})(); + +function makeCredential(overrides?: Partial) { + return { + type: "oauth" as const, + accessToken: accessTokenFixture, + refresh: "refresh-fixture", + expiresAt: Date.now() + 3600_000, + projectId: "test-project", + email: "test@example.com", + accountId: "acct-1", + ...overrides, + } satisfies UsageFetchParams["credential"]; +} + +function fakeFetch(json: unknown): typeof fetch { + const fn = async () => + new Response(JSON.stringify(json), { + status: 200, + headers: { "content-type": "application/json" }, + }); + return fn as unknown as typeof fetch; +} + +function makeCtx(fetchImpl?: typeof fetch): UsageFetchContext { + return { fetch: fetchImpl ?? fakeFetch({}) }; +} + +// ── helpers ────────────────────────────────────────────────────────── + +function makeApiModel( + displayName: string, + quota: { remainingFraction?: number; resetTime?: string; tier?: string; windowId?: string }, +) { + return { + displayName, + quotaInfo: { + remainingFraction: quota.remainingFraction, + resetTime: quota.resetTime, + tier: quota.tier, + windowId: quota.windowId, + }, + }; +} + +// ── tests ──────────────────────────────────────────────────────────── + +describe("antigravity usage provider", () => { + it("merges two models with same tier into one limit", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.5, tier: "premium" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report).not.toBeNull(); + expect(report!.limits.length).toBe(1); + }); + + it("keeps the worst remainingFraction when merging same tier", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.1, tier: "premium" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.8, tier: "premium" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.1); + }); + + it("merges mixed-case tier names under lowercased key", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "Default" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.6, tier: "default" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + }); + + it("preserves reset time from an entry even when bar data comes from another", async () => { + const now = Date.now(); + const resetTime = new Date(now + 4 * 3600_000).toISOString(); + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "default" }), + modelB: makeApiModel("Model B", { remainingFraction: undefined, tier: "default", resetTime }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(1); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.3); + expect(report!.limits[0]!.window).toBeDefined(); + expect(report!.limits[0]!.window!.resetsAt).toBeGreaterThan(now); + }); + + it("separates models with different windowIds in the same tier", async () => { + const now = Date.now(); + const t1 = new Date(now + 5 * 3600_000).toISOString(); + const t2 = new Date(now + 24 * 3600_000).toISOString(); + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium", windowId: "5h", resetTime: t1 }), + modelB: makeApiModel("Model B", { remainingFraction: 0.7, tier: "premium", windowId: "daily", resetTime: t2 }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(2); + }); + + it("includes email and projectId in report metadata", async () => { + const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential({ email: "user@example.com", projectId: "proj-1" }), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.metadata.email).toBe("user@example.com"); + expect(report!.metadata.projectId).toBe("proj-1"); + }); + + it("does not include email when credential has none", async () => { + const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential({ email: undefined }), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.metadata.email).toBeUndefined(); + }); + + it("sorts limits by remainingFraction ascending (worst first)", async () => { + const payload = { + models: { + modelA: makeApiModel("Model A", { remainingFraction: 0.9, tier: "high" }), + modelB: makeApiModel("Model B", { remainingFraction: 0.2, tier: "low" }), + modelC: makeApiModel("Model C", { remainingFraction: 0.5, tier: "mid" }), + }, + }; + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential(), signal: undefined }, + makeCtx(fakeFetch(payload)), + ); + expect(report!.limits.length).toBe(3); + expect(report!.limits[0]!.amount.remainingFraction).toBe(0.2); + expect(report!.limits[1]!.amount.remainingFraction).toBe(0.5); + expect(report!.limits[2]!.amount.remainingFraction).toBe(0.9); + }); + + it("returns null when credential has no projectId", async () => { + const report = await antigravityUsageProvider.fetchUsage!( + { provider: "google-antigravity", credential: makeCredential({ projectId: undefined }), signal: undefined }, + makeCtx(), + ); + expect(report).toBeNull(); + }); +}); diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index 759e99dfa..a64a254f1 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -1373,8 +1373,10 @@ function formatAggregateAmount(limits: UsageLimit[]): string { const uniqueAccountIds = new Set( limits.map(limit => limit.scope.accountId).filter((id): id is string => typeof id === "string" && id.length > 0), ); - if (uniqueAccountIds.size === 0) return ""; - return `${uniqueAccountIds.size} ${uniqueAccountIds.size === 1 ? "acct" : "accts"}`; + if (uniqueAccountIds.size > 0) return `${uniqueAccountIds.size} ${uniqueAccountIds.size === 1 ? "acct" : "accts"}`; + // No account IDs available — keep the pre-existing fallback so providers + // that don't populate scope.accountId still show a summary. + return `${limits.length} accts`; } function resolveResetRange(limits: UsageLimit[], nowMs: number): string | null { From 60f79199ca99af728b4673afd4f3d1e9aa10cc55 Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 20:05:47 +0200 Subject: [PATCH 034/181] fix(usage): only add project: dedup identifier when no email; preserve raw windowId - #getUsageReportIdentifiers now only pushes project: when no email was found, preventing two users with different emails on the same GCP project from being merged at the usage-report level. - Antigravity dedup key now uses quotaInfo.windowId directly before falling back to parseWindow's id. When resetTime is absent but windowId is set, separate windows no longer collapse to 'default'. --- packages/ai/src/auth-storage.ts | 4 +++- packages/ai/src/usage/google-antigravity.ts | 4 +++- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 86b627da7..b1153bed0 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -2307,8 +2307,10 @@ export class AuthStorage { } const projectId = this.#getUsageReportMetadataValue(report, "projectId") ?? this.#getUsageReportScopeProjectId(report); + // Only add project as a fallback when no email is available — two users + // with different emails on the same GCP project must not merge. + if (projectId && !email) identifiers.push(`project:${projectId}`); const accountId = this.#getUsageReportMetadataValue(report, "accountId"); - if (projectId) identifiers.push(`project:${projectId}`); if (accountId) identifiers.push(`account:${accountId}`); const account = this.#getUsageReportMetadataValue(report, "account"); if (account) identifiers.push(`account:${account}`); diff --git a/packages/ai/src/usage/google-antigravity.ts b/packages/ai/src/usage/google-antigravity.ts index ef3b59a65..54e2e11cf 100644 --- a/packages/ai/src/usage/google-antigravity.ts +++ b/packages/ai/src/usage/google-antigravity.ts @@ -167,7 +167,9 @@ async function fetchAntigravityUsage(params: UsageFetchParams, ctx: UsageFetchCo earliestReset = earliestReset ? Math.min(earliestReset, window.resetsAt) : window.resetsAt; } const tier = (quotaInfo.tier ?? "default").toLowerCase(); - const windowId = window?.id ?? "default"; + // Use quotaInfo.windowId even when parseWindow returns undefined + // (no resetTime) — separate windows must not collapse to "default". + const windowId = quotaInfo.windowId ?? window?.id ?? "default"; const key = `${tier}|${windowId}`; const existing = deduped.get(key); if (!existing) { From f3210ab862d974cd63317ac14bcdcf35dd5e38fb Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 18:47:11 +0200 Subject: [PATCH 035/181] fix(ai): strip type-specific keys when CCA mixed-type collapse picks non-matching type MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When collapseMixedTypeCombinerVariants collapses an anyOf with mixed types (e.g. string | array), it previously picked the first non-null type but indiscriminately copied ALL mergedVariantFields — including type-specific keys like "items" that only belong to array. This produced schemas like {type: "string", items: {...}} which Google Cloud Code Assist API rejects with 400. Fix: filter mergedVariantFields against the chosen types allowed keys (CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS) before copying, so array-only keys are dropped when the winner is string (and vice versa). Fixes 400 error on github tools "pr" parameter (anyOf string/array). --- packages/ai/src/utils/schema/normalize.ts | 8 ++++++++ packages/ai/test/schema-normalization.test.ts | 15 +++++++++++++++ 2 files changed, 23 insertions(+) diff --git a/packages/ai/src/utils/schema/normalize.ts b/packages/ai/src/utils/schema/normalize.ts index 1b21afd67..e10a46270 100644 --- a/packages/ai/src/utils/schema/normalize.ts +++ b/packages/ai/src/utils/schema/normalize.ts @@ -505,8 +505,16 @@ function collapseMixedTypeCombinerVariants(schema: JsonObject, combiner: "anyOf" const nextSchema = copySchemaWithout(schema, combiner); const nonNullTypes = variantTypes.filter(t => t !== "null"); nextSchema.type = nonNullTypes[0] ?? variantTypes[0]; + const chosenTypeAllowedKeys = CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS[nextSchema.type as string] ?? {}; for (const key in mergedVariantFields) { if (!Object.hasOwn(mergedVariantFields, key)) continue; + // Drop type-specific keys that don't belong to the chosen type + if ( + !Object.hasOwn(chosenTypeAllowedKeys, key) && + !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key) + ) { + continue; + } const value = mergedVariantFields[key]; const existingValue = nextSchema[key]; if (existingValue !== undefined && !areJsonValuesEqual(existingValue, value)) { diff --git a/packages/ai/test/schema-normalization.test.ts b/packages/ai/test/schema-normalization.test.ts index 1f6ba7688..c7e55d9d1 100644 --- a/packages/ai/test/schema-normalization.test.ts +++ b/packages/ai/test/schema-normalization.test.ts @@ -952,6 +952,21 @@ describe("normalizeSchemaForCCA", () => { properties: {}, }); }); + + it("strips array-only keys when mixed-type collapse picks a non-array type", () => { + // Regression: anyOf [{type:"string"}, {type:"array", items:{type:"string"}}] + // collapsed to {type:"string", items:{type:"string"}} which is invalid. + // The fix filters mergedVariantFields against the chosen type's allowed keys. + const normalized = normalizeSchemaForCCA({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }], + description: "pr number, url, or branch", + }); + + expect(normalized).toEqual({ + type: "string", + description: "pr number, url, or branch", + }); + }); }); // --------------------------------------------------------------------------- From 2623bd75a279e08018925a511dbdf722b9b6017e Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 18:53:52 +0200 Subject: [PATCH 036/181] test(ai): add stripResidualCombiners regression for mixed-type string|array collapse --- packages/ai/test/schema-normalization.test.ts | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/packages/ai/test/schema-normalization.test.ts b/packages/ai/test/schema-normalization.test.ts index c7e55d9d1..b2c23fbc4 100644 --- a/packages/ai/test/schema-normalization.test.ts +++ b/packages/ai/test/schema-normalization.test.ts @@ -728,6 +728,18 @@ describe("stripResidualCombiners", () => { expect(normalized.anyOf).toBeUndefined(); expect(normalized.oneOf).toBeUndefined(); }); + + it("drops array-only keys when mixed-type collapse picks string from anyOf fixpoint", () => { + const stripped = stripResidualCombiners({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "string" } }], + description: "pr number, url, or branch", + }) as Record; + + expect(stripped.type).toBe("string"); + expect(stripped.items).toBeUndefined(); + expect(stripped.anyOf).toBeUndefined(); + expect(stripped.description).toBe("pr number, url, or branch"); + }); }); // --------------------------------------------------------------------------- From 807df56ba752e347678ac002ea7f78996c7a66be Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 19:00:32 +0200 Subject: [PATCH 037/181] docs(ai): add unreleased changelog entry for CCA mixed-type combiner collapse fix --- packages/ai/CHANGELOG.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2f8dea9f7..dd8e58bbc 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,9 @@ ## [Unreleased] +### Fixed + +- Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#TBD](https://github.com/can1357/oh-my-pi/pull/TBD)) ## [15.9.67] - 2026-06-06 ### Fixed From dee4db16025a07f5b2f5db9406d371b2d84f8710 Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 19:02:49 +0200 Subject: [PATCH 038/181] chore(ai): update changelog with PR number --- packages/ai/CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index dd8e58bbc..01a4ab507 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#TBD](https://github.com/can1357/oh-my-pi/pull/TBD)) +- Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#2002](https://github.com/can1357/oh-my-pi/pull/2002)) ## [15.9.67] - 2026-06-06 ### Fixed From 7c8fb4d8f6875900e46b0a3cd275806e24f6cd45 Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 19:08:07 +0200 Subject: [PATCH 039/181] fix(ai): also strip sibling type-specific keys during CCA mixed-type collapse Address review feedback: - Replace `as string` assertion with typed `chosenType` local - Strip sibling keys from nextSchema that were copied via copySchemaWithout but belong to a type other than the chosen one (e.g. sibling `items` on a now-string-typed schema) - Export ALL_CCA_TYPE_SPECIFIC_KEYS from fields.ts for sibling filtering - Add regression test for the sibling-key edge case --- packages/ai/src/utils/schema/fields.ts | 16 ++++++++++++++ packages/ai/src/utils/schema/normalize.ts | 22 ++++++++++++++++--- packages/ai/test/schema-normalization.test.ts | 15 +++++++++++++ 3 files changed, 50 insertions(+), 3 deletions(-) diff --git a/packages/ai/src/utils/schema/fields.ts b/packages/ai/src/utils/schema/fields.ts index 41e9aacf1..b25006248 100644 --- a/packages/ai/src/utils/schema/fields.ts +++ b/packages/ai/src/utils/schema/fields.ts @@ -154,6 +154,22 @@ export const CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS: Record = buildAllCcaTypeSpecificKeys(); + +function buildAllCcaTypeSpecificKeys(): Record { + const all: Record = {}; + for (const typeKeys of Object.values(CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS)) { + for (const key in typeKeys) { + all[key] = true; + } + } + return all; +} + /** * Cloud Code Assist shared schema keys allowed on any type. * Used alongside CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS for CCA combiner collapsing. diff --git a/packages/ai/src/utils/schema/normalize.ts b/packages/ai/src/utils/schema/normalize.ts index e10a46270..0d2961f26 100644 --- a/packages/ai/src/utils/schema/normalize.ts +++ b/packages/ai/src/utils/schema/normalize.ts @@ -11,6 +11,7 @@ import { dereferenceJsonSchema } from "./dereference"; import { upgradeJsonSchemaTo202012 } from "./draft"; import { areJsonValuesEqual, mergePropertySchemas } from "./equality"; import { + ALL_CCA_TYPE_SPECIFIC_KEYS, CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS, COMBINATOR_KEYS, @@ -501,11 +502,26 @@ function collapseMixedTypeCombinerVariants(schema: JsonObject, combiner: "anyOf" if (variantTypes.length < 2 || variantTypes.every(type => type === "object")) { return schema; } - const nextSchema = copySchemaWithout(schema, combiner); const nonNullTypes = variantTypes.filter(t => t !== "null"); - nextSchema.type = nonNullTypes[0] ?? variantTypes[0]; - const chosenTypeAllowedKeys = CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS[nextSchema.type as string] ?? {}; + const chosenType: string = nonNullTypes[0] ?? variantTypes[0]; + nextSchema.type = chosenType; + const chosenTypeAllowedKeys = CLOUD_CODE_ASSIST_TYPE_SPECIFIC_KEYS[chosenType] ?? {}; + + // Strip sibling keys that were copied from the parent and belong to a + // different type (e.g. `items` sibling on a now-string-typed schema). + for (const key in nextSchema) { + if (!Object.hasOwn(nextSchema, key)) continue; + if (key === "type") continue; + if ( + Object.hasOwn(ALL_CCA_TYPE_SPECIFIC_KEYS, key) && + !Object.hasOwn(chosenTypeAllowedKeys, key) && + !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key) + ) { + delete nextSchema[key]; + } + } + for (const key in mergedVariantFields) { if (!Object.hasOwn(mergedVariantFields, key)) continue; // Drop type-specific keys that don't belong to the chosen type diff --git a/packages/ai/test/schema-normalization.test.ts b/packages/ai/test/schema-normalization.test.ts index b2c23fbc4..22efcb2cd 100644 --- a/packages/ai/test/schema-normalization.test.ts +++ b/packages/ai/test/schema-normalization.test.ts @@ -979,6 +979,21 @@ describe("normalizeSchemaForCCA", () => { description: "pr number, url, or branch", }); }); + + it("strips sibling type-specific keys copied from parent when mixed-type collapse picks opposing type", () => { + // Edge case: parent has a sibling `items` outside the anyOf, + // and the chosen type is string. The sibling must be stripped. + const normalized = normalizeSchemaForCCA({ + anyOf: [{ type: "string" }, { type: "array", items: { type: "number" } }], + items: { type: "string" }, + description: "pr number, url, or branch", + }); + + expect(normalized).toEqual({ + type: "string", + description: "pr number, url, or branch", + }); + }); }); // --------------------------------------------------------------------------- From f8ef2cf8e45899f28fc6475a30a10752786a3f11 Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 20:42:06 +0200 Subject: [PATCH 040/181] fix: format CCA schema normalization --- packages/ai/src/utils/schema/normalize.ts | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/packages/ai/src/utils/schema/normalize.ts b/packages/ai/src/utils/schema/normalize.ts index 0d2961f26..ac50eccb7 100644 --- a/packages/ai/src/utils/schema/normalize.ts +++ b/packages/ai/src/utils/schema/normalize.ts @@ -525,10 +525,7 @@ function collapseMixedTypeCombinerVariants(schema: JsonObject, combiner: "anyOf" for (const key in mergedVariantFields) { if (!Object.hasOwn(mergedVariantFields, key)) continue; // Drop type-specific keys that don't belong to the chosen type - if ( - !Object.hasOwn(chosenTypeAllowedKeys, key) && - !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key) - ) { + if (!Object.hasOwn(chosenTypeAllowedKeys, key) && !Object.hasOwn(CLOUD_CODE_ASSIST_SHARED_SCHEMA_KEYS, key)) { continue; } const value = mergedVariantFields[key]; From f854596f83c68768bd477e2769cdd52d3f597438 Mon Sep 17 00:00:00 2001 From: basedcorp99 Date: Sat, 6 Jun 2026 20:42:03 +0200 Subject: [PATCH 041/181] fix: format antigravity usage tests --- .../ai/test/google-antigravity-usage.test.ts | 27 ++++++++++++------- 1 file changed, 17 insertions(+), 10 deletions(-) diff --git a/packages/ai/test/google-antigravity-usage.test.ts b/packages/ai/test/google-antigravity-usage.test.ts index e26a09449..4fdec1ae0 100644 --- a/packages/ai/test/google-antigravity-usage.test.ts +++ b/packages/ai/test/google-antigravity-usage.test.ts @@ -5,14 +5,12 @@ * different model entries, and handles mixed-case tier names. */ import { describe, expect, it } from "bun:test"; +import type { UsageFetchContext, UsageFetchParams } from "../src/usage"; import { antigravityUsageProvider } from "../src/usage/google-antigravity"; -import type { UsageFetchParams, UsageFetchContext } from "../src/usage"; const accessTokenFixture = (() => { const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url"); - const body = Buffer.from( - JSON.stringify({ sub: "user-fixture" }), - ).toString("base64url"); + const body = Buffer.from(JSON.stringify({ sub: "user-fixture" })).toString("base64url"); return `${header}.${body}.sig`; })(); @@ -20,7 +18,7 @@ function makeCredential(overrides?: Partial) { return { type: "oauth" as const, accessToken: accessTokenFixture, - refresh: "refresh-fixture", + refreshToken: "refresh-fixture", expiresAt: Date.now() + 3600_000, projectId: "test-project", email: "test@example.com", @@ -132,7 +130,12 @@ describe("antigravity usage provider", () => { const payload = { models: { modelA: makeApiModel("Model A", { remainingFraction: 0.3, tier: "premium", windowId: "5h", resetTime: t1 }), - modelB: makeApiModel("Model B", { remainingFraction: 0.7, tier: "premium", windowId: "daily", resetTime: t2 }), + modelB: makeApiModel("Model B", { + remainingFraction: 0.7, + tier: "premium", + windowId: "daily", + resetTime: t2, + }), }, }; const report = await antigravityUsageProvider.fetchUsage!( @@ -145,11 +148,15 @@ describe("antigravity usage provider", () => { it("includes email and projectId in report metadata", async () => { const payload = { models: { m: makeApiModel("M", { remainingFraction: 1 }) } }; const report = await antigravityUsageProvider.fetchUsage!( - { provider: "google-antigravity", credential: makeCredential({ email: "user@example.com", projectId: "proj-1" }), signal: undefined }, + { + provider: "google-antigravity", + credential: makeCredential({ email: "user@example.com", projectId: "proj-1" }), + signal: undefined, + }, makeCtx(fakeFetch(payload)), ); - expect(report!.metadata.email).toBe("user@example.com"); - expect(report!.metadata.projectId).toBe("proj-1"); + expect(report!.metadata?.email).toBe("user@example.com"); + expect(report!.metadata?.projectId).toBe("proj-1"); }); it("does not include email when credential has none", async () => { @@ -158,7 +165,7 @@ describe("antigravity usage provider", () => { { provider: "google-antigravity", credential: makeCredential({ email: undefined }), signal: undefined }, makeCtx(fakeFetch(payload)), ); - expect(report!.metadata.email).toBeUndefined(); + expect(report!.metadata?.email).toBeUndefined(); }); it("sorts limits by remainingFraction ascending (worst first)", async () => { From 2620d3970cec81f993aaf9ef7a96faa7decc51cf Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 18:50:37 +0000 Subject: [PATCH 042/181] docs(web-search): updated kagi description to v1 endpoint The runtime cutover to Kagi's V1 search API landed in #1272 but docs/tools/web_search.md still described the sunset V0 endpoint (GET /api/v0/search with 'Authorization: Bot ...'). Realigned the Querying and Output bullets with the actual implementation in packages/coding-agent/src/web/kagi.ts: - POST https://kagi.com/api/v1/search with Bearer auth and JSON body. - recency maps to filters.after as a UTC YYYY-MM-DD string. - Output now includes the categorized bucket merge (search/video/news/ infobox with title tags), adjacent_question + related_search related questions, direct_answer-derived answer, and meta.trace requestId. Fixes #2009 --- docs/tools/web_search.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 5b3aa380b..c62df385d 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -161,9 +161,9 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Output: `sources`, `requestId`. - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` - Availability: env or `agent.db` credential for `kagi`. - - Querying: GET `https://kagi.com/api/v0/search?q=&limit=` with `Authorization: Bot `. + - Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer ` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`). - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. - - Output: `sources`, `relatedQuestions`, `requestId`. + - Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`). - **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts` - Availability: env or `agent.db` credential for `synthetic`. - Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`. From bdbbfa97789ac244546804e2cd5c04a4c5f53abb Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 20:54:00 +0200 Subject: [PATCH 043/181] fix(eval): surfaced subagent abort reason - Used `||` so empty stderr no longer masks the real abort reason. --- .../src/eval/__tests__/agent-bridge.test.ts | 21 +++++++++++++++++++ .../coding-agent/src/eval/agent-bridge.ts | 2 +- 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 08bf08401..2d8662ad5 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -231,6 +231,27 @@ describe("runEvalAgent", () => { }); await expect(runEvalAgent({ prompt: "fail" }, { session: makeSession() })).rejects.toThrow("boom"); }); + + it("surfaces the abort reason when an aborted subagent has empty stderr", async () => { + // An aborted subagent returns exitCode 1 with stderr "" and error + // undefined; the real reason lives in abortReason. The bridge must not + // collapse the failure message to "" (which the Python prelude renders as + // the info-free "bridge call '__agent__' failed"). + mockAgents(); + vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => + singleResult(options, { + exitCode: 1, + output: "", + stderr: "", + aborted: true, + abortReason: "Subagent runtime limit exceeded (task.maxRuntimeMs=1000)", + }), + ); + + await expect(runEvalAgent({ prompt: "hello" }, { session: makeSession() })).rejects.toThrow( + "Subagent runtime limit exceeded (task.maxRuntimeMs=1000)", + ); + }); }); describe("agent() through eval runtimes", () => { diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index a97f6c98e..e114c59d8 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -280,7 +280,7 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption if (result.exitCode !== 0 || result.error) { const failureMessage = - result.error ?? result.stderr ?? result.abortReason ?? `agent() subagent '${agentName}' failed.`; + result.error || result.stderr || result.abortReason || `agent() subagent '${agentName}' failed.`; throw new ToolError(failureMessage); } From 0bac7012cd76b2e7ad11d1ee89345c7f15625824 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 20:56:31 +0200 Subject: [PATCH 044/181] feat(coding-agent/web): added GitHub Actions run/job scraping - Parsed /actions/runs URLs into run and job render handlers. - Rendered run metadata with per-job breakdown, showing steps for failed jobs. - Fetched job logs via API token, stripping ISO timestamp prefixes. --- packages/coding-agent/CHANGELOG.md | 5 + .../coding-agent/src/web/scrapers/github.ts | 258 +++++++++++++++++- .../tools/web-scrapers/git-hosting.test.ts | 56 +++- 3 files changed, 315 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a90e895a4..211bb67ff 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,11 @@ ## [Unreleased] +### Added + +- Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run). + + ## [15.9.69] - 2026-06-06 ### Fixed diff --git a/packages/coding-agent/src/web/scrapers/github.ts b/packages/coding-agent/src/web/scrapers/github.ts index 50ef8b50c..b7b7990c2 100644 --- a/packages/coding-agent/src/web/scrapers/github.ts +++ b/packages/coding-agent/src/web/scrapers/github.ts @@ -1,14 +1,28 @@ import { $env, ptree } from "@oh-my-pi/pi-utils"; import type { RenderResult, SpecialHandler } from "./types"; -import { buildResult, loadPage } from "./types"; +import { buildResult, formatMediaDuration, loadPage } from "./types"; interface GitHubUrl { - type: "blob" | "tree" | "repo" | "issue" | "issues" | "pull" | "pulls" | "discussion" | "discussions" | "other"; + type: + | "blob" + | "tree" + | "repo" + | "issue" + | "issues" + | "pull" + | "pulls" + | "discussion" + | "discussions" + | "actions-run" + | "actions-job" + | "other"; owner: string; repo: string; ref?: string; path?: string; number?: number; + runId?: number; + jobId?: number; } interface GitHubIssueComment { @@ -20,7 +34,7 @@ interface GitHubIssueComment { /** * Parse GitHub URL into components */ -function parseGitHubUrl(url: string): GitHubUrl | null { +export function parseGitHubUrl(url: string): GitHubUrl | null { try { const parsed = new URL(url); if (parsed.hostname !== "github.com") return null; @@ -54,6 +68,20 @@ function parseGitHubUrl(url: string): GitHubUrl | null { return { type: "pulls", owner, repo }; case "pulls": return { type: "pulls", owner, repo }; + case "actions": { + // /actions/runs/{runId} → run summary + jobs + // /actions/runs/{runId}/job/{jobId} → single job (web URL uses singular "job") + // /actions/runs/{runId}/jobs/{jobId} → single job (API-style plural) + if (subParts[0] === "runs" && /^\d+$/.test(subParts[1] ?? "")) { + const runId = parseInt(subParts[1], 10); + const seg = subParts[2]; + if ((seg === "job" || seg === "jobs") && /^\d+$/.test(subParts[3] ?? "")) { + return { type: "actions-job", owner, repo, runId, jobId: parseInt(subParts[3], 10) }; + } + return { type: "actions-run", owner, repo, runId }; + } + return { type: "other", owner, repo }; + } case "discussions": if (subParts.length > 0 && /^\d+$/.test(subParts[0])) { return { type: "discussion", owner, repo, number: parseInt(subParts[0], 10) }; @@ -371,6 +399,212 @@ async function renderGitHubRepo( return { content: md, ok: true }; } +interface GitHubActionsStep { + name: string; + status: string; + conclusion: string | null; + number: number; + started_at: string | null; + completed_at: string | null; +} + +interface GitHubActionsJob { + id: number; + run_id: number; + name: string; + status: string; + conclusion: string | null; + started_at: string | null; + completed_at: string | null; + html_url: string | null; + steps?: GitHubActionsStep[]; + runner_name?: string | null; + labels?: string[]; + workflow_name?: string | null; + head_branch?: string | null; + head_sha?: string; +} + +interface GitHubActionsRun { + id: number; + name?: string | null; + display_title?: string; + run_number: number; + run_attempt?: number; + event: string; + status: string; + conclusion: string | null; + head_branch?: string | null; + head_sha?: string; + html_url: string; + created_at: string; + updated_at: string; + run_started_at?: string; + actor?: { login: string }; + triggering_actor?: { login: string }; +} + +/** Combine status + conclusion into a single label, e.g. `completed (failure)`. */ +function statusLabel(status: string, conclusion: string | null | undefined): string { + return conclusion ? `${status} (${conclusion})` : status; +} + +/** Wall-clock duration between two ISO timestamps, formatted HH:MM:SS / MM:SS. Empty when unknown. */ +function actionDuration(start?: string | null, end?: string | null): string { + if (!start || !end) return ""; + const ms = Date.parse(end) - Date.parse(start); + if (!Number.isFinite(ms) || ms < 0) return ""; + return formatMediaDuration(Math.round(ms / 1000)); +} + +/** Escape `|` so step/job names can't break a markdown table row. */ +function escapeCell(text: string): string { + return text.replaceAll("|", "\\|"); +} + +/** + * Strip the per-line ISO-8601 timestamp prefix GitHub prepends to every job log line. + * Cuts ~28 bytes/line of noise while preserving the message text. Also drops the leading + * UTF-8 BOM GitHub puts at the start of the log file (otherwise the first line's timestamp + * survives because `^` no longer sits before a digit). + */ +export function stripActionsLogTimestamps(logs: string): string { + return logs.replace(/^\uFEFF/, "").replace(/^\d{4}-\d{2}-\d{2}T[\d:.]+Z /gm, ""); +} + +/** Render a job's steps as a markdown table. Empty string when there are no steps. */ +function renderActionsSteps(steps?: GitHubActionsStep[]): string { + if (!steps || steps.length === 0) return ""; + let md = "| # | Step | Status | Conclusion | Duration |\n"; + md += "|---|------|--------|------------|----------|\n"; + for (const step of steps) { + const dur = actionDuration(step.started_at, step.completed_at) || "-"; + md += `| ${step.number} | ${escapeCell(step.name)} | ${step.status} | ${step.conclusion ?? "-"} | ${dur} |\n`; + } + return `${md}\n`; +} + +/** Run-level metadata lines shared by the run and job renderers. */ +function renderActionsRunMeta(run: GitHubActionsRun): string { + let md = `**Workflow:** ${run.name ?? "(unknown)"}\n`; + md += `**Run:** #${run.run_number}`; + if (run.run_attempt && run.run_attempt > 1) md += ` (attempt ${run.run_attempt})`; + md += ` · ${statusLabel(run.status, run.conclusion)}\n`; + if (run.head_branch) { + md += `**Branch:** ${run.head_branch}${run.head_sha ? ` @ ${run.head_sha.slice(0, 7)}` : ""}\n`; + } + const actor = run.triggering_actor?.login ?? run.actor?.login; + md += `**Event:** ${run.event}${actor ? ` · by @${actor}` : ""}\n`; + const started = run.run_started_at ?? run.created_at; + const dur = actionDuration(started, run.updated_at); + md += `Started: ${started}${dur ? ` · Duration: ${dur}` : ""}\n`; + md += `URL: ${run.html_url}\n`; + return md; +} + +/** Fetch a job's plain-text logs. Returns null when unavailable (no token / expired / private). */ +async function fetchGitHubJobLogs( + owner: string, + repo: string, + jobId: number, + timeout: number, + signal?: AbortSignal, +): Promise { + const headers: Record = { + Accept: "application/vnd.github+json", + "X-GitHub-Api-Version": "2022-11-28", + }; + const token = $env.GITHUB_TOKEN || $env.GH_TOKEN; + if (token) headers.Authorization = `Bearer ${token}`; + + // 302 → signed log URL on a different origin; fetch strips Authorization on the cross-origin hop. + const result = await loadPage(`https://api.github.com/repos/${owner}/${repo}/actions/jobs/${jobId}/logs`, { + timeout, + headers, + signal, + }); + return result.ok && result.content ? result.content : null; +} + +/** + * Render a workflow run: run metadata plus a per-job breakdown. Steps are listed for any job that + * did not succeed (the debugging-relevant ones); successful jobs collapse to a single line. + */ +async function renderGitHubActionsRun( + gh: GitHubUrl, + timeout: number, + signal?: AbortSignal, +): Promise<{ content: string; ok: boolean }> { + const runResult = await fetchGitHubApi(`/repos/${gh.owner}/${gh.repo}/actions/runs/${gh.runId}`, timeout, signal); + if (!runResult.ok || !runResult.data) return { content: "", ok: false }; + + const run = runResult.data as GitHubActionsRun; + let md = `# ${run.display_title || run.name || `Run #${run.run_number}`}\n\n`; + md += renderActionsRunMeta(run); + md += `\n---\n\n`; + + const jobsResult = await fetchGitHubApi( + `/repos/${gh.owner}/${gh.repo}/actions/runs/${gh.runId}/jobs?per_page=100`, + timeout, + signal, + ); + if (jobsResult.ok && jobsResult.data) { + const jobs = (jobsResult.data as { jobs?: GitHubActionsJob[] }).jobs ?? []; + md += `## Jobs (${jobs.length})\n\n`; + for (const job of jobs) { + const dur = actionDuration(job.started_at, job.completed_at); + md += `### ${escapeCell(job.name)} — ${statusLabel(job.status, job.conclusion)}${dur ? ` (${dur})` : ""}\n\n`; + if (job.conclusion !== "success") { + md += renderActionsSteps(job.steps); + } + } + } + + return { content: md, ok: true }; +} + +/** + * Render a single workflow job: run context, step table, and the full job logs. + */ +async function renderGitHubActionsJob( + gh: GitHubUrl, + timeout: number, + signal?: AbortSignal, +): Promise<{ content: string; ok: boolean }> { + const jobResult = await fetchGitHubApi(`/repos/${gh.owner}/${gh.repo}/actions/jobs/${gh.jobId}`, timeout, signal); + if (!jobResult.ok || !jobResult.data) return { content: "", ok: false }; + + const job = jobResult.data as GitHubActionsJob; + + // Best-effort run context for nicer headers; the job render stands on its own without it. + const runResult = await fetchGitHubApi(`/repos/${gh.owner}/${gh.repo}/actions/runs/${job.run_id}`, timeout, signal); + const run = runResult.ok && runResult.data ? (runResult.data as GitHubActionsRun) : null; + + let md = `# ${escapeCell(job.name)}\n\n`; + if (run) { + md += renderActionsRunMeta(run); + } else if (job.workflow_name) { + md += `**Workflow:** ${job.workflow_name}\n`; + if (job.head_branch) md += `**Branch:** ${job.head_branch}\n`; + } + const dur = actionDuration(job.started_at, job.completed_at); + md += `**Job:** ${escapeCell(job.name)} · ${statusLabel(job.status, job.conclusion)}${dur ? ` · ${dur}` : ""}\n`; + if (job.runner_name) md += `**Runner:** ${job.runner_name}\n`; + if (job.html_url) md += `URL: ${job.html_url}\n`; + md += `\n---\n\n`; + + const steps = renderActionsSteps(job.steps); + if (steps) md += `## Steps\n\n${steps}`; + + const logs = await fetchGitHubJobLogs(gh.owner, gh.repo, job.id, timeout, signal); + md += `## Logs\n\n`; + md += logs + ? stripActionsLogTimestamps(logs) + : "*Logs unavailable — requires a GITHUB_TOKEN/GH_TOKEN with read access, or the run's logs have expired.*\n"; + + return { content: md, ok: true }; +} + /** * Handle GitHub URLs specially */ @@ -445,6 +679,24 @@ export const handleGitHub: SpecialHandler = async ( } break; } + + case "actions-run": { + notes.push(`Fetched via GitHub API`); + const result = await renderGitHubActionsRun(gh, timeout, signal); + if (result.ok) { + return buildResult(result.content, { url, method: "github-actions-run", fetchedAt, notes }); + } + break; + } + + case "actions-job": { + notes.push(`Fetched via GitHub API`); + const result = await renderGitHubActionsJob(gh, timeout, signal); + if (result.ok) { + return buildResult(result.content, { url, method: "github-actions-job", fetchedAt, notes }); + } + break; + } } // Fall back to null (let normal rendering handle it) diff --git a/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts b/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts index d54a9ee47..e219e5167 100644 --- a/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts +++ b/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { handleGitHub } from "@oh-my-pi/pi-coding-agent/web/scrapers/github"; +import { handleGitHub, parseGitHubUrl, stripActionsLogTimestamps } from "@oh-my-pi/pi-coding-agent/web/scrapers/github"; import { handleGitHubGist } from "@oh-my-pi/pi-coding-agent/web/scrapers/github-gist"; const SKIP = !Bun.env.WEB_FETCH_INTEGRATION; @@ -203,3 +203,57 @@ describe.skipIf(SKIP)("handleGitHubGist", () => { expect(result).toBeDefined(); }); }); + +// ============================================================================= +// GitHub Actions URL parsing (pure, network-free) +// ============================================================================= + +describe("parseGitHubUrl — Actions", () => { + it("classifies a workflow run URL", () => { + const gh = parseGitHubUrl("https://github.com/can1357/oh-my-pi/actions/runs/27070071296"); + expect(gh).toEqual({ type: "actions-run", owner: "can1357", repo: "oh-my-pi", runId: 27070071296 }); + }); + + it("classifies a job URL using the web-form singular `job` segment", () => { + const gh = parseGitHubUrl("https://github.com/can1357/oh-my-pi/actions/runs/27070071296/job/79897931171"); + expect(gh).toEqual({ + type: "actions-job", + owner: "can1357", + repo: "oh-my-pi", + runId: 27070071296, + jobId: 79897931171, + }); + }); + + it("classifies a job URL using the API-form plural `jobs` segment", () => { + const gh = parseGitHubUrl("https://github.com/can1357/oh-my-pi/actions/runs/27070071296/jobs/79897931171"); + expect(gh?.type).toBe("actions-job"); + expect(gh?.jobId).toBe(79897931171); + }); + + it("does not treat non-run Actions URLs (e.g. workflow files) as runs/jobs", () => { + expect(parseGitHubUrl("https://github.com/can1357/oh-my-pi/actions/workflows/ci.yml")?.type).toBe("other"); + expect(parseGitHubUrl("https://github.com/can1357/oh-my-pi/actions")?.type).toBe("other"); + }); + + it("does not misparse a run URL with a non-numeric id", () => { + expect(parseGitHubUrl("https://github.com/can1357/oh-my-pi/actions/runs/latest")?.type).toBe("other"); + }); + + it("returns null for non-github hosts", () => { + expect(parseGitHubUrl("https://gitlab.com/o/r/actions/runs/1")).toBeNull(); + }); +}); + +describe("stripActionsLogTimestamps", () => { + it("removes the per-line ISO timestamp prefix and a leading BOM", () => { + const raw = + "\uFEFF2026-06-06T18:14:12.8793443Z Current runner version: '2.334.0'\n2026-06-06T18:14:13.0000000Z done\n"; + expect(stripActionsLogTimestamps(raw)).toBe("Current runner version: '2.334.0'\ndone\n"); + }); + + it("leaves grouped/non-timestamped lines untouched", () => { + const raw = "2026-06-06T18:14:12.0000000Z ##[group]Operating System\nUbuntu\n##[endgroup]\n"; + expect(stripActionsLogTimestamps(raw)).toBe("##[group]Operating System\nUbuntu\n##[endgroup]\n"); + }); +}); From 133137c9a672077880c94c7d6065b194ca76d799 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 21:33:07 +0200 Subject: [PATCH 045/181] fix(eval): surfaced subagent abort reasons and disabled runtime cap - Used `||` so empty stderr falls through to abortReason in agent bridge. - Preferred assistant errorMessage over "Cancelled by caller" on internal aborts. - Forced `maxRuntimeMs: 0` for eval subagents via ExecutorOptions override. --- packages/coding-agent/CHANGELOG.md | 8 +++++ .../src/eval/__tests__/agent-bridge.test.ts | 9 ++++++ .../coding-agent/src/eval/agent-bridge.ts | 6 ++++ packages/coding-agent/src/task/executor.ts | 22 ++++++++++++-- .../task/executor-subagent-reminders.test.ts | 29 +++++++++++++++++++ 5 files changed, 72 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 211bb67ff..8aa225b00 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,14 @@ - Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run). +### Changed + +- Changed eval `agent()` subagents so they are never subject to the `task.maxRuntimeMs` wall-clock cap. The parent cell's idle watchdog is already suspended for the entire bridge call (`withBridgeTimeoutPause`), so a long-running fan-out/recovery workflow must not be killed by a per-subagent runtime limit. `runEvalAgent` now passes `maxRuntimeMs: 0` to `runSubprocess`, which honors an explicit `ExecutorOptions.maxRuntimeMs` override over the inherited setting. + +### Fixed + +- Fixed eval `agent()` failures surfacing as an opaque `RuntimeError: bridge call '__agent__' failed` with no reason. When a subagent aborted, `runEvalAgent` built its failure message with `result.error ?? result.stderr ?? result.abortReason ?? …`, but `result.stderr` is the empty string on a clean abort (and `result.error` is gated on a non-empty `stderr`), so the nullish chain stopped at `""` and never reached `abortReason`. The empty string propagated through the loopback bridge and the Python prelude's `RuntimeError(msg or "bridge call … failed")`, discarding the real reason. The chain now uses `||` so an empty `stderr` falls through to `abortReason`. +- Fixed subagent aborts being mislabeled as the generic "Cancelled by caller" when the abort originated inside the subagent's own turn (`stopReason: "aborted"` with no caller signal and no runtime-limit timer). `runSubprocess` now prefers the aborted assistant message's `errorMessage` (e.g. "Request was aborted" or a specific stream error) for that case, while a real caller signal or wall-clock abort still reports its precise reason. ## [15.9.69] - 2026-06-06 ### Fixed diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 2d8662ad5..9f6c5d48c 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -252,6 +252,15 @@ describe("runEvalAgent", () => { "Subagent runtime limit exceeded (task.maxRuntimeMs=1000)", ); }); + + it("disables the wall-clock runtime limit for eval subagents", async () => { + mockAgents(); + const runSpy = vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options)); + + await runEvalAgent({ prompt: "hello" }, { session: makeSession() }); + + expect(runSpy.mock.calls[0]?.[0].maxRuntimeMs).toBe(0); + }); }); describe("agent() through eval runtimes", () => { diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index e114c59d8..618895e91 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -259,6 +259,12 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption authStorage: options.session.authStorage, modelRegistry: options.session.modelRegistry, settings: options.session.settings, + // Eval `agent()` subagents are never wall-clock capped: the parent + // cell's idle watchdog is suspended for the whole bridge call + // (withBridgeTimeoutPause), so a long-running phase/recovery workflow + // must not be killed by `task.maxRuntimeMs`. Force the limit off + // regardless of the inherited session setting. + maxRuntimeMs: 0, mcpManager, contextFiles, skills: availableSkills, diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 94fb9a63b..254a2dc5e 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -166,6 +166,13 @@ export interface ExecutorOptions { outputSchema?: unknown; /** Parent task recursion depth (0 = top-level, 1 = first child, etc.) */ taskDepth?: number; + /** + * Override the `task.maxRuntimeMs` wall-clock cap for this run. When provided + * it wins over the settings value; `0` disables the per-subagent wall-clock + * limit entirely. Used by the eval `agent()` bridge, whose parent cell + * watchdog is already suspended for the call's duration. + */ + maxRuntimeMs?: number; enableLsp?: boolean; signal?: AbortSignal; onProgress?: (progress: AgentProgress) => void; @@ -625,7 +632,10 @@ export async function runSubprocess(options: ExecutorOptions): Promise= 0 && childDepth >= maxRecursionDepth; @@ -1484,7 +1494,15 @@ export async function runSubprocess(options: ExecutorOptions): Promise { expect(result.abortReason).toBe("Cancelled before start"); expect(result.stderr).toBe("Cancelled before start"); }); + + it("surfaces the assistant abort message instead of 'Cancelled by caller' on an internal turn abort", async () => { + // No caller signal and no runtime limit: the subagent's own turn ended with + // stopReason "aborted" (e.g. a merged request-signal abort). abortReason is + // undefined, so the executor must report the assistant's real errorMessage, + // not the generic caller-cancellation fallback. This is also what the eval + // agent() bridge re-raises, so a blank/misleading reason here surfaces as an + // opaque "bridge call '__agent__' failed". + const session = createMockSession(({ emit, state }) => { + const aborted: AssistantMessage = { + ...createAssistantStopMessage(""), + stopReason: "aborted", + errorMessage: "Request was aborted", + }; + state.messages.push(aborted); + emit({ type: "message_end", message: aborted }); + }); + + mockCreateAgentSession(session); + + const result = await runSubprocess({ ...baseOptions, id: "subagent-internal-abort" }); + + expect(result.aborted).toBe(true); + expect(result.exitCode).toBe(1); + expect(result.abortReason).toBe("Request was aborted"); + expect(result.abortReason).not.toBe("Cancelled by caller"); + expect(result.error).toBeUndefined(); + expect(result.stderr).toBe(""); + }); it("uses modelRegistry.authStorage when only options.modelRegistry is provided", async () => { const session = createMockSession(({ emit }) => { emit({ From 5fc443f4af9939266f3324b96a436448f45e98ea Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 21:36:39 +0200 Subject: [PATCH 046/181] fix(ai): adjusted usage ranking comparator for stable metric ordering - Added a tolerance-aware `compareUsageRankingMetric` helper with finite-value handling. - Replaced usage provider sorting comparisons with the new comparator for secondary and primary usage metrics. - Kept existing tie-breaker fields while stabilizing ordering for nearly equal metric values. --- packages/ai/src/auth-storage.ts | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index b1153bed0..dab8e1045 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -36,6 +36,8 @@ import { loginOpenAICodexDevice } from "./utils/oauth/openai-codex"; import type { OAuthController, OAuthCredentials, OAuthProvider, OAuthProviderId } from "./utils/oauth/types"; import { loginXiaomi, loginXiaomiTokenPlan } from "./utils/oauth/xiaomi"; +const USAGE_RANKING_METRIC_EPSILON = 1e-9; + // ───────────────────────────────────────────────────────────────────────────── // Credential Types // ───────────────────────────────────────────────────────────────────────────── @@ -606,6 +608,14 @@ function hasOpenAICodexProPlan(report: UsageReport | null): boolean { return getUsagePlanType(report)?.includes("pro") === true; } +function compareUsageRankingMetric(left: number, right: number): number { + if (left === right) return 0; + if (!Number.isFinite(left) || !Number.isFinite(right)) return left < right ? -1 : 1; + const delta = left - right; + const tolerance = Math.max(USAGE_RANKING_METRIC_EPSILON, Math.max(Math.abs(left), Math.abs(right)) * 0.000001); + return Math.abs(delta) <= tolerance ? 0 : delta; +} + function resolveDefaultUsageProvider(provider: Provider): UsageProvider | undefined { return DEFAULT_USAGE_PROVIDER_MAP.get(provider); } @@ -2796,12 +2806,14 @@ export class AuthStorage { return left.planPriority - right.planPriority; } if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; - if (left.secondaryDrainRate !== right.secondaryDrainRate) { - return left.secondaryDrainRate - right.secondaryDrainRate; - } - if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed; - if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate; - if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed; + let metric = compareUsageRankingMetric(left.secondaryDrainRate, right.secondaryDrainRate); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.secondaryUsed, right.secondaryUsed); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.primaryDrainRate, right.primaryDrainRate); + if (metric !== 0) return metric; + metric = compareUsageRankingMetric(left.primaryUsed, right.primaryUsed); + if (metric !== 0) return metric; return 0; } From 57210347395ad3bb431d440c0a580de76965ea21 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 21:53:22 +0200 Subject: [PATCH 047/181] fix(ui): forced full replay on tool output expand toggle - Replaced viewport-only repaint with resetDisplay so committed scrollback reflects new heights. - Added per-server rust-analyzer workspace-ready timing overrides as a test seam. - Added clearSuppressedSelectors to reset retry-fallback cooldown state. - Removed obsolete shared eval executors test. --- .../coding-agent/src/config/model-registry.ts | 8 + .../eval/__tests__/shared-executors.test.ts | 609 ------------------ packages/coding-agent/src/lsp/client.ts | 16 +- packages/coding-agent/src/lsp/types.ts | 10 + .../src/modes/controllers/input-controller.ts | 10 +- packages/tui/src/tui.ts | 9 + 6 files changed, 47 insertions(+), 615 deletions(-) delete mode 100644 packages/coding-agent/src/eval/__tests__/shared-executors.test.ts diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index a37fdb8e9..ac83a06ce 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -2609,6 +2609,14 @@ export class ModelRegistry { } return true; } + + /** + * Clear all cooldown suppressions recorded via {@link suppressSelector}. + * Used to reset retry-fallback cooldown state without a full {@link refresh}. + */ + clearSuppressedSelectors(): void { + this.#suppressedSelectors.clear(); + } } /** diff --git a/packages/coding-agent/src/eval/__tests__/shared-executors.test.ts b/packages/coding-agent/src/eval/__tests__/shared-executors.test.ts deleted file mode 100644 index a68c896e8..000000000 --- a/packages/coding-agent/src/eval/__tests__/shared-executors.test.ts +++ /dev/null @@ -1,609 +0,0 @@ -import { afterAll, afterEach, describe, expect, it, vi } from "bun:test"; -import * as fs from "node:fs/promises"; -import * as path from "node:path"; -import type { AssistantMessage } from "@oh-my-pi/pi-ai"; -import { TempDir } from "@oh-my-pi/pi-utils"; -import type { ModelRegistry } from "../../config/model-registry"; -import { Settings } from "../../config/settings"; -import type { LoadExtensionsResult } from "../../extensibility/extensions/types"; -import type { CreateAgentSessionOptions, CreateAgentSessionResult } from "../../sdk"; -import * as sdkModule from "../../sdk"; -import type { AgentSession, AgentSessionEvent, PromptOptions } from "../../session/agent-session"; -import { TaskTool } from "../../task"; -import * as discoveryModule from "../../task/discovery"; -import type { AgentDefinition, TaskParams } from "../../task/types"; -import type { ToolSession } from "../../tools"; -import { EventBus } from "../../utils/event-bus"; -import { disposeAllVmContexts } from "../js/context-manager"; -import { executeJs } from "../js/executor"; -import { disposeAllKernelSessions, executePython } from "../py/executor"; - -function createToolSession(cwd: string, sessionFile: string | null, evalSessionId?: string): ToolSession { - const modelRegistry = { - authStorage: undefined, - refresh: async () => {}, - getAvailable: () => [], - getApiKey: async () => null, - } as unknown as ModelRegistry; - return { - cwd, - hasUI: false, - settings: Settings.isolated({ - "async.enabled": false, - "task.isolation.mode": "none", - }), - getSessionFile: () => sessionFile, - getSessionSpawns: () => "*", - getEvalSessionId: evalSessionId ? () => evalSessionId : undefined, - modelRegistry, - } as unknown as ToolSession; -} - -function createBridgeToolSession(resultText: string, calls: unknown[]): ToolSession { - const readTool = { - name: "read", - label: "read", - description: "read", - parameters: { type: "object" }, - async execute(_id: string, args: unknown) { - calls.push(args); - return { content: [{ type: "text" as const, text: resultText }] }; - }, - }; - const tools = new Map([["read", readTool]]); - return { getToolByName: (name: string) => tools.get(name) } as unknown as ToolSession; -} - -function assistantStopMessage(text: string): AssistantMessage { - return { - role: "assistant", - content: text ? [{ type: "text", text }] : [], - api: "openai-responses", - provider: "openai", - model: "mock", - usage: { - input: 0, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 0, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "stop", - timestamp: Date.now(), - }; -} - -function createYieldingSubagentSession(onPrompt: () => Promise): AgentSession { - const listeners: Array<(event: AgentSessionEvent) => void> = []; - const state = { messages: [] as AssistantMessage[] }; - const emit = (event: AgentSessionEvent) => { - for (const listener of listeners) listener(event); - }; - return { - state, - agent: { state: { systemPrompt: ["test"] } }, - model: undefined, - extensionRunner: undefined, - sessionManager: { - appendSessionInit: () => {}, - }, - getActiveToolNames: () => ["eval", "yield"], - setActiveToolsByName: async () => {}, - subscribe: (listener: (event: AgentSessionEvent) => void) => { - listeners.push(listener); - return () => { - const index = listeners.indexOf(listener); - if (index >= 0) listeners.splice(index, 1); - }; - }, - prompt: async (_text: string, _options?: PromptOptions) => { - await onPrompt(); - state.messages.push(assistantStopMessage("done")); - emit({ - type: "tool_execution_end", - toolCallId: "yield-call", - toolName: "yield", - result: { - content: [{ type: "text", text: "Result submitted." }], - details: { status: "success", data: { ok: true } }, - }, - isError: false, - }); - }, - waitForIdle: async () => {}, - getLastAssistantMessage: () => state.messages[state.messages.length - 1], - abort: async () => {}, - dispose: async () => {}, - } as unknown as AgentSession; -} - -const taskAgent: AgentDefinition = { - name: "task", - description: "Task agent", - systemPrompt: "Read eval state and yield.", - source: "bundled", - tools: ["eval", "yield"], -}; - -const taskParams: TaskParams = { - agent: "task", - tasks: [{ id: "ReadEval", description: "Read eval state", assignment: "Read parent eval state." }], -}; - -describe("shared eval executors", () => { - afterEach(() => { - vi.restoreAllMocks(); - }); - - afterAll(async () => { - await disposeAllVmContexts(); - await disposeAllKernelSessions(); - }); - - it("shares JavaScript state across executeJs calls with one session id", async () => { - using tempDir = TempDir.createSync("@omp-eval-js-shared-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-shared:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - - await executeJs("globalThis.x = 41;", { sessionId, session, sessionFile }); - const result = await executeJs("return globalThis.x + 1;", { sessionId, session, sessionFile }); - - expect(result.exitCode).toBe(0); - expect(result.output.trim()).toBe("42"); - }); - - it("treats idleTimeoutMs as caller-owned watchdog metadata, not a fixed timer", async () => { - using tempDir = TempDir.createSync("@omp-eval-js-idle-budget-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-idle-budget:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - - // With no wall-clock deadlineMs/timeoutMs and no aborting signal, a cell that - // runs well past idleTimeoutMs must still complete: the backend must never - // derive a competing fixed timer from the caller-owned watchdog budget. - const result = await executeJs("await Bun.sleep(120); return 'done';", { - sessionId, - session, - sessionFile, - idleTimeoutMs: 30, - }); - - expect(result.cancelled).toBe(false); - expect(result.exitCode).toBe(0); - expect(result.output.trim()).toBe("done"); - }); - - it("shares Python state across executePython calls with one session id", async () => { - using tempDir = TempDir.createSync("@omp-eval-py-shared-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `py-shared:${crypto.randomUUID()}`; - - await executePython("x = 41", { cwd: tempDir.path(), sessionId, sessionFile }); - const result = await executePython("print(x + 1)", { cwd: tempDir.path(), sessionId, sessionFile }); - - expect(result.exitCode).toBe(0); - expect(result.output.trim()).toBe("42"); - }); - - it("deduplicates concurrent first JavaScript session acquisition", async () => { - using tempDir = TempDir.createSync("@omp-eval-js-cold-start-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-cold-start:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - - const [first, second] = await Promise.all([ - executeJs( - "globalThis.sharedMarker ??= crypto.randomUUID(); await Bun.sleep(50); return globalThis.sharedMarker;", - { - sessionId, - session, - sessionFile, - }, - ), - executeJs("globalThis.sharedMarker ??= crypto.randomUUID(); return globalThis.sharedMarker;", { - sessionId, - session, - sessionFile, - }), - ]); - const third = await executeJs("return globalThis.sharedMarker;", { sessionId, session, sessionFile }); - - expect(first.exitCode).toBe(0); - expect(second.exitCode).toBe(0); - expect(third.exitCode).toBe(0); - expect(first.output.trim()).toBe(second.output.trim()); - expect(third.output.trim()).toBe(first.output.trim()); - }); - - it("deduplicates concurrent first Python session acquisition", async () => { - using tempDir = TempDir.createSync("@omp-eval-py-cold-start-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `py-cold-start:${crypto.randomUUID()}`; - - const [first, second] = await Promise.all([ - executePython( - `import asyncio, uuid -shared_marker = globals().get("shared_marker") or str(uuid.uuid4()) -globals()["shared_marker"] = shared_marker -await asyncio.sleep(0.05) -print(shared_marker)`, - { cwd: tempDir.path(), sessionId, sessionFile }, - ), - executePython( - `import uuid -shared_marker = globals().get("shared_marker") or str(uuid.uuid4()) -globals()["shared_marker"] = shared_marker -print(shared_marker)`, - { cwd: tempDir.path(), sessionId, sessionFile }, - ), - ]); - const third = await executePython("print(shared_marker)", { cwd: tempDir.path(), sessionId, sessionFile }); - - expect(first.exitCode).toBe(0); - expect(second.exitCode).toBe(0); - expect(third.exitCode).toBe(0); - expect(first.output.trim()).toBe(second.output.trim()); - expect(third.output.trim()).toBe(first.output.trim()); - }); - - it("splits retained Python kernels by cwd for one shared session id", async () => { - using tempDir = TempDir.createSync("@omp-eval-py-cwd-"); - const dirA = path.join(tempDir.path(), "a"); - const dirB = path.join(tempDir.path(), "b"); - await fs.mkdir(dirA); - await fs.mkdir(dirB); - const realDirA = await fs.realpath(dirA); - const realDirB = await fs.realpath(dirB); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `py-cwd:${crypto.randomUUID()}`; - - const first = await executePython( - `import os -token = "from-a" -print(os.getcwd())`, - { - cwd: dirA, - sessionId, - sessionFile, - }, - ); - const second = await executePython( - `import os -print(os.getcwd()) -print("token" in globals())`, - { - cwd: dirB, - sessionId, - sessionFile, - }, - ); - const third = await executePython("print(token)", { cwd: dirA, sessionId, sessionFile }); - - expect(first.exitCode).toBe(0); - expect(first.output.trim()).toBe(realDirA); - expect(second.exitCode).toBe(0); - expect(second.output.trim().split("\n")).toEqual([realDirB, "False"]); - expect(third.exitCode).toBe(0); - expect(third.output.trim()).toBe("from-a"); - }); - - it("interrupts timed out synchronous Python cells before they mutate shared state", async () => { - using tempDir = TempDir.createSync("@omp-eval-py-sync-timeout-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `py-sync-timeout:${crypto.randomUUID()}`; - - const timedOut = await executePython("import time\ntime.sleep(0.2)\nleaked_after_timeout = True", { - cwd: tempDir.path(), - sessionId, - sessionFile, - timeoutMs: 20, - }); - await Bun.sleep(250); - const probe = await executePython('print("leaked_after_timeout" in globals())', { - cwd: tempDir.path(), - sessionId, - sessionFile, - }); - - expect(timedOut.cancelled).toBe(true); - expect(probe.exitCode).toBe(0); - expect(probe.output.trim()).toBe("False"); - }); - - it("settles Python cells that raise SystemExit", async () => { - using tempDir = TempDir.createSync("@omp-eval-py-system-exit-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `py-system-exit:${crypto.randomUUID()}`; - - const result = await executePython('raise SystemExit("bye")', { - cwd: tempDir.path(), - sessionId, - sessionFile, - timeoutMs: 500, - }); - - expect(result.exitCode).toBe(1); - expect(result.output).toContain("SystemExit"); - expect(result.output).toContain("bye"); - }); - - it("lets a subagent inherit parent JavaScript and Python eval state", async () => { - using tempDir = TempDir.createSync("@omp-eval-subagent-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const evalSessionId = `session:${sessionFile}:cwd:${tempDir.path()}`; - const parentSession = createToolSession(tempDir.path(), sessionFile, evalSessionId); - let seenJs = ""; - let seenPy = ""; - let capturedOptions: CreateAgentSessionOptions | undefined; - - await executeJs('globalThis.parentSecret = "hello-js";', { - sessionId: `js:${evalSessionId}`, - session: parentSession, - sessionFile, - }); - await executePython('parent_secret = "hello-py"', { - cwd: tempDir.path(), - sessionId: `python:${evalSessionId}`, - sessionFile, - }); - - vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ agents: [taskAgent], projectAgentsDir: null }); - vi.spyOn(sdkModule, "createAgentSession").mockImplementation(async (options = {}) => { - capturedOptions = options; - const inherited = options.parentEvalSessionId; - if (!inherited) throw new Error("Missing parent eval session id"); - return { - session: createYieldingSubagentSession(async () => { - const jsResult = await executeJs("return globalThis.parentSecret;", { - sessionId: `js:${inherited}`, - session: parentSession, - sessionFile, - }); - const pyResult = await executePython("print(parent_secret)", { - cwd: tempDir.path(), - sessionId: `python:${inherited}`, - sessionFile, - }); - seenJs = jsResult.output.trim(); - seenPy = pyResult.output.trim(); - }), - extensionsResult: {} as unknown as LoadExtensionsResult, - setToolUIContext: () => {}, - eventBus: new EventBus(), - } satisfies CreateAgentSessionResult; - }); - - const tool = await TaskTool.create(parentSession); - await tool.execute("tool-call", taskParams); - - expect(capturedOptions?.parentEvalSessionId).toBe(evalSessionId); - expect(seenJs).toBe("hello-js"); - expect(seenPy).toBe("hello-py"); - }); - - it("routes interleaved JavaScript display output to the matching run", async () => { - using tempDir = TempDir.createSync("@omp-eval-js-interleave-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-interleave:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - - const first = executeJs('await Bun.sleep(80); display({ label: "A" });', { - sessionId, - session, - sessionFile, - }); - await Bun.sleep(10); - const second = executeJs('display({ label: "B" });', { - sessionId, - session, - sessionFile, - }); - - const [firstResult, secondResult] = await Promise.all([first, second]); - expect(firstResult.exitCode).toBe(0); - expect(secondResult.exitCode).toBe(0); - expect(firstResult.displayOutputs).toEqual([{ type: "json", data: { label: "A" } }]); - expect(secondResult.displayOutputs).toEqual([{ type: "json", data: { label: "B" } }]); - }); - - it("routes interleaved Python display output to the matching run", async () => { - using tempDir = TempDir.createSync("@omp-eval-py-interleave-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `py-interleave:${crypto.randomUUID()}`; - - const first = executePython( - `import asyncio -await asyncio.sleep(0.08) -display({"label": "A"})`, - { - cwd: tempDir.path(), - sessionId, - sessionFile, - }, - ); - await Bun.sleep(10); - const second = executePython('display({"label": "B"})', { - cwd: tempDir.path(), - sessionId, - sessionFile, - }); - - const [firstResult, secondResult] = await Promise.all([first, second]); - expect(firstResult.exitCode).toBe(0); - expect(secondResult.exitCode).toBe(0); - expect(firstResult.displayOutputs).toEqual([{ type: "json", data: { label: "A" } }]); - expect(secondResult.displayOutputs).toEqual([{ type: "json", data: { label: "B" } }]); - }); - it("preserves module-level singleton state across re-imports of an unchanged file", async () => { - using tempDir = TempDir.createSync("@omp-eval-js-mtime-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-mtime:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - const modulePath = path.join(tempDir.path(), "singleton.ts"); - const moduleSpec = JSON.stringify(modulePath); - await Bun.write( - modulePath, - "let value = 0;\nexport function set(v) { value = v; }\nexport function get() { return value; }\n", - ); - - const initResult = await executeJs(`const mod = await import(${moduleSpec}); mod.set(42); return mod.get();`, { - sessionId, - session, - sessionFile, - }); - expect(initResult.exitCode).toBe(0); - expect(initResult.output.trim()).toBe("42"); - - // Unchanged file: re-import must reuse the existing module namespace so the - // counter is still 42. This is the regression — the previous unconditional - // `delete require.cache[target]` reset singletons on every dynamic import. - const reuseResult = await executeJs(`const mod = await import(${moduleSpec}); return mod.get();`, { - sessionId, - session, - sessionFile, - }); - expect(reuseResult.exitCode).toBe(0); - expect(reuseResult.output.trim()).toBe("42"); - - // Bump mtime by 5s to simulate an edit; the next import must evict the cache - // and re-evaluate the file, dropping the counter back to its initializer. - const future = new Date(Date.now() + 5_000); - await fs.utimes(modulePath, future, future); - - const reloadResult = await executeJs(`const mod = await import(${moduleSpec}); return mod.get();`, { - sessionId, - session, - sessionFile, - }); - expect(reloadResult.exitCode).toBe(0); - expect(reloadResult.output.trim()).toBe("0"); - }); - - it("reloads a local re-export when a transitive dependency changes", async () => { - using tempDir = TempDir.createSync("@omp-eval-js-transitive-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-transitive:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - const leafPath = path.join(tempDir.path(), "leaf.ts"); - const entryPath = path.join(tempDir.path(), "entry.ts"); - const entrySpec = JSON.stringify(entryPath); - await Bun.write(leafPath, "export const value = 1;\n"); - await Bun.write(entryPath, 'export { value } from "./leaf.ts";\n'); - - const initial = await executeJs(`const mod = await import(${entrySpec}); return mod.value;`, { - sessionId, - session, - sessionFile, - }); - expect(initial.exitCode).toBe(0); - expect(initial.output.trim()).toBe("1"); - - await Bun.write(leafPath, "export const value = 2;\n"); - const future = new Date(Date.now() + 5_000); - await fs.utimes(leafPath, future, future); - - const reloaded = await executeJs(`const mod = await import(${entrySpec}); return mod.value;`, { - sessionId, - session, - sessionFile, - }); - expect(reloaded.exitCode).toBe(0); - expect(reloaded.output.trim()).toBe("2"); - }); - - it("links a cyclic local module graph without crashing", async () => { - // Regression: the loader used to link()+evaluate() each local module individually - // inside the recursive linker callback. On any import cycle that re-entered Bun's - // node:vm linker mid-instantiation and segfaulted the process (SIGTRAP, - // getImportedModule on a null record) — e.g. `await import("…/edit/streaming.ts")`, - // whose relative-import subtree is cyclic. The graph must now link in a single pass. - using tempDir = TempDir.createSync("@omp-eval-js-cycle-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-cycle:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - const alphaPath = path.join(tempDir.path(), "alpha.ts"); - const betaPath = path.join(tempDir.path(), "beta.ts"); - const alphaSpec = JSON.stringify(alphaPath); - const betaSpec = JSON.stringify(betaPath); - await Bun.write( - alphaPath, - 'import { betaName } from "./beta.ts";\nexport const alphaName = "alpha";\nexport function combined() { return alphaName + ":" + betaName; }\n', - ); - await Bun.write( - betaPath, - 'import { alphaName } from "./alpha.ts";\nexport const betaName = "beta";\nexport function viaAlpha() { return alphaName; }\n', - ); - - const result = await executeJs( - `const a = await import(${alphaSpec});\nconst b = await import(${betaSpec});\nreturn [a.combined(), b.viaAlpha()].join("|");`, - { sessionId, session, sessionFile }, - ); - - expect(result.exitCode).toBe(0); - expect(result.output.trim()).toBe("alpha:beta|alpha"); - }); - - it("loads TypeScript type-only imports in cells and local modules", async () => { - using tempDir = TempDir.createSync("@omp-eval-js-type-imports-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `js-type-imports:${crypto.randomUUID()}`; - const session = createToolSession(tempDir.path(), sessionFile); - const typesPath = path.join(tempDir.path(), "types.ts"); - const valuesPath = path.join(tempDir.path(), "values.ts"); - const entryPath = path.join(tempDir.path(), "entry.ts"); - const typesSpec = JSON.stringify(typesPath); - const entrySpec = JSON.stringify(entryPath); - await Bun.write(typesPath, "export interface TypeOnly { value: number }\n"); - await Bun.write(valuesPath, "export interface InlineOnly { value: number }\nexport const imported = 41;\n"); - await Bun.write( - entryPath, - [ - 'import type { TypeOnly } from "./types.ts";', - 'import { type InlineOnly, imported } from "./values.ts";', - "export const typeOnly = 1;", - "export const inlineType = imported;", - "", - ].join("\n"), - ); - - const result = await executeJs( - `import type { TypeOnly } from ${typesSpec};\nconst mod = await import(${entrySpec});\nreturn mod.typeOnly + mod.inlineType;`, - { - sessionId, - session, - sessionFile, - }, - ); - - expect(result.exitCode).toBe(0); - expect(result.output.trim()).toBe("42"); - }); - - it("refreshes the Python tool proxy when bridge env appears after kernel warm-up", async () => { - using tempDir = TempDir.createSync("@omp-eval-py-tool-proxy-"); - const sessionFile = path.join(tempDir.path(), "session.jsonl"); - const sessionId = `py-tool-proxy:${crypto.randomUUID()}`; - const bridgeCalls: unknown[] = []; - const bridgeSession = createBridgeToolSession("bridge-ok", bridgeCalls); - - const withoutBridge = await executePython( - 'try:\n print(tool.read({"path": "foo.txt"}))\nexcept Exception as exc:\n print(type(exc).__name__)\n print(str(exc))', - { cwd: tempDir.path(), sessionId, sessionFile }, - ); - const withBridge = await executePython('print(tool.read({"path": "foo.txt"}))', { - cwd: tempDir.path(), - sessionId, - sessionFile, - toolSession: bridgeSession, - }); - - expect(withoutBridge.exitCode).toBe(0); - expect(withoutBridge.output).toContain("RuntimeError"); - expect(withoutBridge.output).toContain("tool bridge is unavailable"); - expect(withBridge.exitCode).toBe(0); - expect(withBridge.output.trim()).toBe("bridge-ok"); - expect(bridgeCalls).toEqual([{ path: "foo.txt", _i: "py prelude" }]); - }); -}); diff --git a/packages/coding-agent/src/lsp/client.ts b/packages/coding-agent/src/lsp/client.ts index cf872019c..3080fc083 100644 --- a/packages/coding-agent/src/lsp/client.ts +++ b/packages/coding-agent/src/lsp/client.ts @@ -438,6 +438,7 @@ export const WARMUP_TIMEOUT_MS = 5000; const RUST_ANALYZER_WORKSPACE_READY_TIMEOUT_MS = 5_000; const RUST_ANALYZER_WORKSPACE_READY_POLL_MS = 100; const RUST_ANALYZER_WORKSPACE_READY_SETTLE_MS = 2_000; +const RUST_ANALYZER_STATUS_REQUEST_TIMEOUT_MS = 1_000; const rustAnalyzerReadyClients = new WeakSet(); function commandBasename(command: string): string { @@ -462,29 +463,34 @@ async function waitForRustAnalyzerWorkspace(client: LspClient, signal?: AbortSig if (rustAnalyzerReadyClients.has(client)) { return; } + const timings = client.config.workspaceReadyTimings; + const timeoutMs = timings?.timeoutMs ?? RUST_ANALYZER_WORKSPACE_READY_TIMEOUT_MS; + const pollMs = timings?.pollMs ?? RUST_ANALYZER_WORKSPACE_READY_POLL_MS; + const settleMs = timings?.settleMs ?? RUST_ANALYZER_WORKSPACE_READY_SETTLE_MS; + const statusRequestTimeoutMs = timings?.statusRequestTimeoutMs ?? RUST_ANALYZER_STATUS_REQUEST_TIMEOUT_MS; const started = Date.now(); - const deadline = started + RUST_ANALYZER_WORKSPACE_READY_TIMEOUT_MS; + const deadline = started + timeoutMs; while (true) { throwIfAborted(signal); let status: unknown; try { - status = await sendRequest(client, "rust-analyzer/analyzerStatus", {}, signal, 1_000); + status = await sendRequest(client, "rust-analyzer/analyzerStatus", {}, signal, statusRequestTimeoutMs); } catch (err) { if (!isRustAnalyzerStatusTimeout(err) || Date.now() >= deadline) { return; } - await Bun.sleep(RUST_ANALYZER_WORKSPACE_READY_POLL_MS); + await Bun.sleep(pollMs); continue; } const ready = typeof status === "string" && !status.startsWith("No workspaces"); - if (ready && Date.now() - started >= RUST_ANALYZER_WORKSPACE_READY_SETTLE_MS) { + if (ready && Date.now() - started >= settleMs) { rustAnalyzerReadyClients.add(client); return; } if (Date.now() >= deadline) { return; } - await Bun.sleep(RUST_ANALYZER_WORKSPACE_READY_POLL_MS); + await Bun.sleep(pollMs); } } diff --git a/packages/coding-agent/src/lsp/types.ts b/packages/coding-agent/src/lsp/types.ts index 96b6a1f6d..42028047a 100644 --- a/packages/coding-agent/src/lsp/types.ts +++ b/packages/coding-agent/src/lsp/types.ts @@ -356,6 +356,16 @@ export interface ServerConfig { disabled?: boolean; /** Per-server warmup timeout in milliseconds. Overrides the global WARMUP_TIMEOUT_MS for this server during startup. */ warmupTimeoutMs?: number; + /** + * Per-server overrides for rust-analyzer workspace-ready polling. When omitted, the module + * defaults are used. Primarily a tuning/test seam to bound the multi-second settle window. + */ + workspaceReadyTimings?: { + timeoutMs?: number; + pollMs?: number; + settleMs?: number; + statusRequestTimeoutMs?: number; + }; capabilities?: ServerCapabilities; /** If true, this is a linter/formatter server (e.g., Biome) - used only for diagnostics/actions, not type intelligence */ isLinter?: boolean; diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index 8c3fbd1b4..253998228 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -846,7 +846,15 @@ export class InputController { child.setExpanded(expanded); } } - this.ctx.ui.requestRender(false, { allowUnknownViewportMutation: true }); + // Toggling expansion mutates every block, but on ED3-risk terminals the + // transcript freezes a snapshot of each block once it scrolls past the live + // region (committed native scrollback is immutable there). A plain repaint + // replays those stale snapshots, so the toggle appears to do nothing above + // the live block. resetDisplay() invalidates the snapshots and forces a + // full clear + replay — the keyboard-accessible resize-reset equivalent — + // which is the only path that re-emits the whole transcript at its new + // heights. + this.ctx.ui.resetDisplay(); } toggleThinkingBlockVisibility(): void { diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index c568d09ea..732af473c 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -984,9 +984,18 @@ export class TUI extends Container { * scrollback. This is the keyboard-accessible equivalent of the resize reset: * no queued diff frame or terminal scrollback probe can downgrade it to a * viewport-only repaint. + * + * Invalidates every component first so the replay reflects current state. A + * geometry-driven reset thaws frozen scrollback snapshots implicitly (the new + * width misses every cached snapshot), but a same-width reset would otherwise + * replay stale snapshots — leaving host-frozen blocks (e.g. a transcript whose + * committed rows are immutable on ED3-risk terminals) showing pre-mutation + * content. Invalidation is the generic signal those containers use to retire + * their snapshots, which is exactly what a user-driven display reset wants. */ resetDisplay(): void { if (this.#stopped) return; + this.invalidate(); this.#prepareForcedRender(!isMultiplexerSession(), true); this.#resizeEventPending = true; this.#renderRequested = false; From 20d19e80020d79ad3bc2a26f88859602560f5a8d Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:09:04 +0200 Subject: [PATCH 048/181] test: replaced blind sleeps with shared fixtures and condition polling - Shared immutable model registries and auth storage via beforeAll/afterAll. - Swapped fixed-delay settle sleeps for predicate polling and signals. - Stubbed network/timers to drop wall-clock waits in registry and history tests. - Added resetDisplay invalidation tests and startup-timing breakdown lines. --- .../src/eval/__tests__/agent-bridge.test.ts | 56 ++-- .../test/agent-session-concurrent.test.ts | 23 +- .../agent-session-context-promotion.test.ts | 30 +- .../test/agent-session-handoff.test.ts | 93 ++++-- .../agent-session-model-persistence.test.ts | 45 +-- ...nt-session-openai-responses-replay.test.ts | 91 +++--- .../test/agent-session-python-cleanup.test.ts | 3 + .../test/agent-session-retry-fallback.test.ts | 26 +- .../test/autoresearch-tools.test.ts | 34 +-- .../coding-agent/test/bash-executor.test.ts | 44 ++- .../test/extensions-runner.test.ts | 23 +- .../test/goals/goal-mode-integration.test.ts | 45 ++- .../test/interactive-mode-plan-review.test.ts | 2 - .../keybindings-selector-navigation.test.ts | 13 +- .../test/mcp-reconnect-storm.test.ts | 17 +- .../model-registry-runtime-provider.test.ts | 12 +- .../coding-agent/test/model-registry.test.ts | 5 +- .../components/transcript-container.test.ts | 19 ++ .../input-controller-tool-expansion.test.ts | 11 +- .../test/plan-mode-thinking-level.test.ts | 7 +- .../sdk-async-job-manager-singleton.test.ts | 25 +- .../sdk-credential-disabled-bridge.test.ts | 13 +- .../test/sdk-mcp-discovery.test.ts | 28 +- .../test/sdk-model-selection.test.ts | 23 +- .../test/sdk-session-isolation.test.ts | 25 +- packages/coding-agent/test/sdk-skills.test.ts | 28 +- .../test/sdk-tool-activation.test.ts | 152 +++------- packages/coding-agent/test/tools.test.ts | 31 +- .../test/tools/approval-mode.test.ts | 279 +++++++----------- .../test/tools/conflict-integration.test.ts | 6 +- .../test/tools/fetch-jina-stall.test.ts | 17 +- packages/coding-agent/test/tools/gh.test.ts | 51 +++- .../test/tools/lsp-regressions.test.ts | 4 + packages/tui/test/render-regressions.test.ts | 54 ++++ packages/utils/CHANGELOG.md | 4 + packages/utils/src/logger.ts | 15 +- 36 files changed, 854 insertions(+), 500 deletions(-) diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index dd66f44cc..838ed6c1a 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -377,18 +377,6 @@ describe("agent() through eval runtimes", () => { singleResult(options, { output: "hello from python" }), ); - const probe = await executePython('print("probe")', { - cwd: tempDir.path(), - sessionId: `${sessionId}:probe`, - sessionFile, - kernelMode: "per-call", - }); - if (probe.exitCode === undefined && probe.cancelled) { - expect(probe.output).toBe(""); - return; - } - expect(probe.exitCode).toBe(0); - const result = await executePython('print(agent("hi"))', { cwd: tempDir.path(), sessionId, @@ -396,6 +384,10 @@ describe("agent() through eval runtimes", () => { kernelMode: "per-call", toolSession: session, }); + if (result.exitCode === undefined && result.cancelled) { + expect(result.output).toBe(""); + return; // kernel unavailable in this environment + } expect(result.exitCode).toBe(0); expect(result.output.trim()).toBe("hello from python"); @@ -424,22 +416,14 @@ describe("agent() through eval runtimes", () => { } }); - const probe = await executePython('print("probe")', { - cwd: tempDir.path(), - sessionId: `${sessionId}:probe`, - sessionFile, - kernelMode: "per-call", - }); - if (probe.exitCode === undefined && probe.cancelled) { - expect(probe.output).toBe(""); - return; - } - expect(probe.exitCode).toBe(0); - const result = await executePython( 'import json\nprint(json.dumps(parallel([lambda n=n: agent(n) for n in ["a", "b", "c", "d"]])))', { cwd: tempDir.path(), sessionId, sessionFile, kernelMode: "per-call", toolSession: session }, ); + if (result.exitCode === undefined && result.cancelled) { + expect(result.output).toBe(""); + return; // kernel unavailable in this environment + } expect(result.exitCode).toBe(0); expect(JSON.parse(result.output.trim())).toEqual(["a", "b", "c", "d"]); @@ -463,7 +447,14 @@ describe("agent() through eval runtimes", () => { // The host must respond the instant the cell aborts so the kernel can // unwind via KeyboardInterrupt instead of being hard-killed (which used to // surface "[kernel] Python kernel shutdown" and lose all session state). + let inFlight = 0; + let markSaturated: (() => void) | undefined; + const saturated = new Promise(resolve => { + markSaturated = resolve; + }); vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => { + // task.maxConcurrency=6 → six bridge calls block at once; signal then. + if (++inFlight >= 6) markSaturated?.(); await Bun.sleep(9000); // deliberately ignores options.signal return singleResult(options, { output: options.assignment ?? "" }); }); @@ -483,8 +474,9 @@ describe("agent() through eval runtimes", () => { expect(seed.exitCode).toBe(0); const ac = new AbortController(); - // Abort ~1s in, after the worker threads are blocked in their bridge calls. - setTimeout(() => ac.abort(new Error("external interrupt")), 1000); + // Abort the instant all six worker threads are confirmed blocked in their + // bridge calls (condition-driven) instead of waiting a fixed wall second. + void saturated.then(() => ac.abort(new Error("external interrupt"))); const start = Date.now(); const result = await executePython( @@ -619,12 +611,12 @@ describe("agent() through eval runtimes", () => { // of its own. The bridge pause must make that delegated time invisible to // the watchdog. vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => { - await Bun.sleep(200); + await Bun.sleep(40); return singleResult(options, { output: "done" }); }); const ops: string[] = []; - using idle = new IdleTimeout(60); + using idle = new IdleTimeout(20); const result = await runEvalAgent( { prompt: "investigate" }, { @@ -642,7 +634,7 @@ describe("agent() through eval runtimes", () => { expect(ops).toEqual([EVAL_TIMEOUT_PAUSE_OP, EVAL_TIMEOUT_RESUME_OP]); expect(idle.signal.aborted).toBe(false); - await Bun.sleep(90); + await Bun.sleep(60); expect(idle.signal.aborted).toBe(true); }); @@ -655,7 +647,7 @@ describe("agent() through eval runtimes", () => { // They render as status, but timeout accounting is controlled only by the // bridge pause/resume events. vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => { - for (let i = 0; i < 40; i++) { + for (let i = 0; i < 20; i++) { options.onProgress?.({ index: options.index, id: options.id, @@ -672,13 +664,13 @@ describe("agent() through eval runtimes", () => { cost: 0, durationMs: i * 10, }); - await Bun.sleep(10); + await Bun.sleep(5); } return singleResult(options, { output: "done" }); }); const ops: string[] = []; - using idle = new IdleTimeout(80); + using idle = new IdleTimeout(40); const result = await runEvalAgent( { prompt: "investigate" }, { diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index f03cbf039..a5298263a 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -6,6 +6,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; +import { scheduler } from "node:timers/promises"; import { Agent, AgentBusyError, type AgentTool } from "@oh-my-pi/pi-agent-core"; import { type AssistantMessage, getBundledModel, type Message, type ToolCall } from "@oh-my-pi/pi-ai"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; @@ -25,6 +26,19 @@ import { createAssistantMessage } from "./helpers/agent-session-setup"; // Mock stream that mimics AssistantMessageEventStream +// AgentSession schedules its TTSR retry and context-promotion continuations +// through `scheduler.wait(delayMs, { signal })` (node:timers/promises), with +// blind 50ms/100ms "settle" delays. Tests that drive a continuation to +// completion would otherwise pay that wall-clock time on every run. This spy +// collapses the blind delay to a single macrotask hop (`scheduler.wait(0)`) +// while preserving the real abort-signal semantics, so the continuation still +// fires only after the aborted/overflowed turn has been recorded. Each test +// that opts in must run inside a block whose afterEach restores mocks. +const originalSchedulerWait = scheduler.wait.bind(scheduler); +function collapseSchedulerSettleDelays(): void { + vi.spyOn(scheduler, "wait").mockImplementation((_delayMs, options) => originalSchedulerWait(0, options)); +} + describe("AgentSession concurrent prompt guard", () => { let session: AgentSession; let tempDir: string; @@ -101,7 +115,7 @@ describe("AgentSession concurrent prompt guard", () => { const deadline = Date.now() + timeoutMs; while (Date.now() < deadline) { if (predicate()) return; - await Bun.sleep(10); + await Bun.sleep(1); } throw new Error("Timed out waiting for condition"); @@ -595,13 +609,14 @@ describe("AgentSession TTSR resume gate", () => { if (tempDir && fs.existsSync(tempDir)) { fs.rmSync(tempDir, { recursive: true }); } + vi.restoreAllMocks(); }); async function waitFor(predicate: () => boolean, timeoutMs = 500): Promise { const deadline = Date.now() + timeoutMs; while (Date.now() < deadline) { if (predicate()) return; - await Bun.sleep(10); + await Bun.sleep(1); } throw new Error("Timed out waiting for condition"); @@ -674,6 +689,7 @@ describe("AgentSession TTSR resume gate", () => { } it("prompt() blocks until TTSR interrupt continuation completes", async () => { + collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let streamCallCount = 0; let continuationCompleted = false; @@ -734,6 +750,7 @@ describe("AgentSession TTSR resume gate", () => { }); it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => { + collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let streamCallCount = 0; @@ -946,6 +963,7 @@ describe("AgentSession TTSR resume gate", () => { }); it("prompt() waits for TTSR continuation with tool calls to finish", async () => { + collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let streamCallCount = 0; let toolExecutionFinished = false; @@ -1306,6 +1324,7 @@ describe("AgentSession TTSR resume gate", () => { }); it("prompt() waits for context-promotion continuation to finish", async () => { + collapseSchedulerSettleDelays(); const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-promo.db")); authStorages.push(authStorage); authStorage.setRuntimeApiKey("openai-codex", "test-key"); diff --git a/packages/coding-agent/test/agent-session-context-promotion.test.ts b/packages/coding-agent/test/agent-session-context-promotion.test.ts index dcbe8af92..06347689b 100644 --- a/packages/coding-agent/test/agent-session-context-promotion.test.ts +++ b/packages/coding-agent/test/agent-session-context-promotion.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { Agent } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, Model, ProviderSessionState } from "@oh-my-pi/pi-ai"; @@ -15,19 +15,26 @@ describe("AgentSession context promotion", () => { let modelRegistry: ModelRegistry; let authStorage: AuthStorage; - beforeEach(async () => { + beforeAll(async () => { + // ModelRegistry eagerly loads the immutable bundled model catalog in its + // constructor (~100ms). The catalog and auth fixture never change between + // tests here (tests only read models and add benign extra runtime keys), + // so build them once instead of paying ~950ms across the 9 cases. tempDir = TempDir.createSync("@pi-context-promotion-"); authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); authStorage.setRuntimeApiKey("openai-codex", "test-key"); modelRegistry = new ModelRegistry(authStorage); }); + afterAll(() => { + authStorage.close(); + tempDir.removeSync(); + }); + afterEach(async () => { if (session) { await session.dispose(); } - authStorage.close(); - tempDir.removeSync(); }); function createOverflowMessage( @@ -113,6 +120,17 @@ describe("AgentSession context promotion", () => { throw new Error("Timed out waiting for condition"); } + // Deterministically drain the fire-and-forget `agent_end` handler that + // `emitExternalEvent` dispatches. The handler's terminal maintenance work + // (`#checkCompaction`) is microtask-based on the no-promotion paths, so a + // single macrotask turn fully flushes it; `waitForIdle` then settles any + // tracked continuation. Used by the negative tests, which assert that *no* + // promotion happened and therefore need the handler to have actually run. + async function settle(): Promise { + await new Promise(resolve => setTimeout(resolve, 0)); + await session.waitForIdle(); + } + it("promotes to a larger-context model on overflow and clears codex websocket session state", async () => { const sparkModel = modelRegistry.find("openai-codex", "gpt-5.3-codex-spark"); const codexModel = modelRegistry.find("openai-codex", "gpt-5.5"); @@ -390,7 +408,7 @@ describe("AgentSession context promotion", () => { session.agent.emitExternalEvent({ type: "message_end", message: overflowMessage }); session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowMessage] }); - await Bun.sleep(30); + await settle(); expect(session.model?.provider).toBe(sparkModel.provider); expect(session.model?.id).toBe(sparkModel.id); @@ -472,7 +490,7 @@ describe("AgentSession context promotion", () => { session.agent.emitExternalEvent({ type: "message_end", message: staleIncomplete }); session.agent.emitExternalEvent({ type: "agent_end", messages: [staleIncomplete] }); - await Bun.sleep(30); + await settle(); expect(session.model?.provider).toBe(codexModel.provider); expect(session.model?.id).toBe(codexModel.id); diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index e90506b33..3ecb612aa 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -1,8 +1,8 @@ -import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { Agent } from "@oh-my-pi/pi-agent-core"; import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction"; -import type { AssistantMessage, ToolCall } from "@oh-my-pi/pi-ai"; +import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai"; import { getBundledModel } from "@oh-my-pi/pi-ai/models"; import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -14,26 +14,67 @@ import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manage import { TempDir } from "@oh-my-pi/pi-utils"; describe("AgentSession handoff", () => { + // Immutable across the whole file: the model registry's synchronous bundled-model + // load dominates per-test setup (~100ms each), and the auth store + bundled model + // never change. Build them once. Per-test mutable state (session, session file, + // emitted events) is rebuilt in beforeEach. + let sharedDir: TempDir; + let authStorage: AuthStorage; + let modelRegistry: ModelRegistry; + let model: Model; + let tempDir: TempDir; let session: AgentSession; let sessionManager: SessionManager; - let authStorage: AuthStorage; - let modelRegistry: ModelRegistry; let events: AgentSessionEvent[]; + /** Poll `predicate` until it holds (returns as soon as the state is reached) or the + * deadline elapses. Replaces blind settle sleeps for tests with a positive signal. */ + async function waitFor(predicate: () => boolean, timeoutMs = 1_000): Promise { + const deadline = Date.now() + timeoutMs; + while (!predicate()) { + if (Date.now() >= deadline) { + throw new Error("Timed out waiting for condition"); + } + await Bun.sleep(1); + } + } + + /** Drain post-turn maintenance deterministically for negative tests (those proving + * maintenance did NOT run, where there is no positive signal to poll on). Post-turn + * work is scheduled fire-and-forget: a single event-loop turn lets the handler run to + * its decision and register any compaction pass as a tracked post-prompt task, then + * `waitForIdle()` drains that task to completion. */ + async function drainMaintenance(): Promise { + await Bun.sleep(0); + await session.waitForIdle(); + } + + beforeAll(async () => { + sharedDir = TempDir.createSync("@pi-handoff-shared-"); + authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + modelRegistry = new ModelRegistry(authStorage); + + const bundled = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!bundled) { + throw new Error("Expected built-in anthropic model to exist"); + } + model = bundled; + }); + + afterAll(async () => { + authStorage.close(); + try { + await sharedDir.remove(); + } catch {} + }); + beforeEach(async () => { tempDir = TempDir.createSync("@pi-handoff-"); - authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); - authStorage.setRuntimeApiKey("anthropic", "test-key"); - modelRegistry = new ModelRegistry(authStorage); sessionManager = SessionManager.create(tempDir.path(), tempDir.path()); events = []; - const model = getBundledModel("anthropic", "claude-sonnet-4-5"); - if (!model) { - throw new Error("Expected built-in anthropic model to exist"); - } - const agent = new Agent({ initialState: { model, @@ -85,7 +126,6 @@ describe("AgentSession handoff", () => { if (session) { await session.dispose(); } - authStorage.close(); try { await tempDir.remove(); } catch {} @@ -97,7 +137,7 @@ describe("AgentSession handoff", () => { const generateHandoffSpy = vi.spyOn(compactionModule, "generateHandoff").mockResolvedValue(handoffText); const result = await session.handoff(); - await Bun.sleep(20); + await drainMaintenance(); expect(generateHandoffSpy).toHaveBeenCalledTimes(1); expect(result?.document).toBe(handoffText); @@ -123,7 +163,11 @@ describe("AgentSession handoff", () => { }); await session.prompt("pending prompt ".repeat(120)); - await Bun.sleep(20); + await waitFor( + () => + compactSpy.mock.calls.length === 1 && + events.some(event => event.type === "auto_compaction_end" && event.aborted === false), + ); expect(compactSpy).toHaveBeenCalledTimes(1); expect(promptSpy).toHaveBeenCalledTimes(1); @@ -177,7 +221,7 @@ describe("AgentSession handoff", () => { isError: false, }); session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); - await Bun.sleep(20); + await drainMaintenance(); expect(handoffSpy).not.toHaveBeenCalled(); expect(events.filter(event => event.type === "auto_compaction_start")).toHaveLength(0); @@ -259,7 +303,7 @@ describe("AgentSession handoff", () => { const handoffSpy = vi.spyOn(session, "handoff"); session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); - await Bun.sleep(20); + await drainMaintenance(); expect(handoffSpy).not.toHaveBeenCalled(); expect(events.filter(event => event.type === "auto_compaction_start")).toHaveLength(0); @@ -307,7 +351,7 @@ describe("AgentSession handoff", () => { session.agent.emitExternalEvent({ type: "message_end", message: overflowAssistant }); session.agent.emitExternalEvent({ type: "agent_end", messages: [overflowAssistant] }); - await Bun.sleep(20); + await waitFor(() => events.filter(event => event.type === "auto_compaction_end").length === 1); expect(handoffSpy).not.toHaveBeenCalled(); const startEvents = events.filter(event => event.type === "auto_compaction_start"); @@ -352,7 +396,11 @@ describe("AgentSession handoff", () => { session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); - await Bun.sleep(20); + await waitFor( + () => + handoffSpy.mock.calls.length === 1 && + events.filter(event => event.type === "auto_compaction_end").length === 1, + ); expect(handoffSpy).toHaveBeenCalledTimes(1); expect(handoffSpy).toHaveBeenCalledWith(expect.stringContaining("Threshold-triggered maintenance"), { @@ -500,7 +548,8 @@ describe("AgentSession handoff", () => { session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); - await Bun.sleep(20); + await waitFor(() => handoffSpy.mock.calls.length === 1); + await session.waitForIdle(); expect(handoffSpy).toHaveBeenCalledTimes(1); // The bug surfaced as agent.continue() racing the deferred handoff. With the fix, @@ -554,7 +603,7 @@ describe("AgentSession handoff", () => { session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); // Let the deferred handoff post-prompt task enter the generateHandoff await. - await Bun.sleep(20); + await waitFor(() => session.isGeneratingHandoff); expect(generateHandoffSpy).toHaveBeenCalledTimes(1); expect(session.isGeneratingHandoff).toBe(true); @@ -601,7 +650,7 @@ describe("AgentSession handoff", () => { session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); - await Bun.sleep(20); + await waitFor(() => events.filter(event => event.type === "auto_compaction_end").length === 1); expect(handoffSpy).toHaveBeenCalledTimes(1); const endEvents = events.filter(event => event.type === "auto_compaction_end"); diff --git a/packages/coding-agent/test/agent-session-model-persistence.test.ts b/packages/coding-agent/test/agent-session-model-persistence.test.ts index 7b741027a..76f089355 100644 --- a/packages/coding-agent/test/agent-session-model-persistence.test.ts +++ b/packages/coding-agent/test/agent-session-model-persistence.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test"; import * as path from "node:path"; import { Agent } from "@oh-my-pi/pi-agent-core"; import { type Api, Effort, getBundledModel, type Model } from "@oh-my-pi/pi-ai"; @@ -18,7 +18,24 @@ describe("AgentSession model persistence", () => { let tempDir: TempDir; let session: AgentSession | undefined; let sessionSettings: Settings; - const authStorages: AuthStorage[] = []; + // Auth storage (SQLite DB) and the model registry are immutable across these tests: + // every test sets the same anthropic runtime key and only ever reads the bundled model + // list. Building them once avoids ~12 SQLite opens + registry constructions. + let sharedDir: TempDir; + let sharedAuthStorage: AuthStorage; + let sharedModelRegistry: ModelRegistry; + + beforeAll(async () => { + sharedDir = TempDir.createSync("@pi-model-persistence-shared-"); + sharedAuthStorage = await AuthStorage.create(path.join(sharedDir.path(), "auth.db")); + sharedAuthStorage.setRuntimeApiKey("anthropic", "test-key"); + sharedModelRegistry = new ModelRegistry(sharedAuthStorage, path.join(sharedDir.path(), "models.yml")); + }); + + afterAll(() => { + sharedAuthStorage.close(); + sharedDir.removeSync(); + }); beforeEach(() => { tempDir = TempDir.createSync("@pi-model-persistence-"); @@ -29,9 +46,6 @@ describe("AgentSession model persistence", () => { await session.dispose(); session = undefined; } - for (const authStorage of authStorages.splice(0)) { - authStorage.close(); - } tempDir.removeSync(); }); @@ -84,13 +98,7 @@ describe("AgentSession model persistence", () => { modelRoles?: Record; persist?: boolean; }): Promise<{ modelRegistry: ModelRegistry; settings: Settings; session: AgentSession }> { - const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${authStorages.length}.db`)); - authStorages.push(authStorage); - authStorage.setRuntimeApiKey("anthropic", "test-key"); - const modelRegistry = new ModelRegistry( - authStorage, - path.join(tempDir.path(), `models-${authStorages.length}.yml`), - ); + const modelRegistry = sharedModelRegistry; const model = options?.initialModel ?? options?.selectInitialModel?.(modelRegistry.getAvailable()) ?? @@ -118,7 +126,7 @@ describe("AgentSession model persistence", () => { session = new AgentSession({ agent, sessionManager: options?.persist - ? SessionManager.create(tempDir.path(), path.join(tempDir.path(), `active-${authStorages.length}`)) + ? SessionManager.create(tempDir.path(), path.join(tempDir.path(), "active")) : SessionManager.inMemory(), settings: sessionSettings, modelRegistry, @@ -131,19 +139,12 @@ describe("AgentSession model persistence", () => { targetSessionFile: string, settings: Settings = Settings.isolated(), ): Promise { - const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${authStorages.length}.db`)); - authStorages.push(authStorage); - authStorage.setRuntimeApiKey("anthropic", "test-key"); - const modelRegistry = new ModelRegistry( - authStorage, - path.join(tempDir.path(), `models-${authStorages.length}.yml`), - ); const sessionManager = await SessionManager.open(targetSessionFile, path.join(tempDir.path(), "startup")); const result = await createAgentSession({ cwd: tempDir.path(), agentDir: tempDir.path(), - authStorage, - modelRegistry, + authStorage: sharedAuthStorage, + modelRegistry: sharedModelRegistry, sessionManager, settings, disableExtensionDiscovery: true, diff --git a/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts b/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts index 7fa2c98ed..fd5452369 100644 --- a/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts +++ b/packages/coding-agent/test/agent-session-openai-responses-replay.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -12,8 +12,11 @@ import type { Usage, } from "@oh-my-pi/pi-ai/types"; import { createOpenAIResponsesHistoryPayload } from "@oh-my-pi/pi-ai/utils"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; -import type { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { type SessionEntry, SessionManager, @@ -209,20 +212,21 @@ async function createPersistedSession( return { sessionFile, treeTargetId: result?.treeTargetId }; } +// ModelRegistry construction loads the bundled model catalog plus the on-disk +// cache (~100ms) and dominated this file's runtime when rebuilt once per test. +// The registry and its pinned AuthStorage are immutable across these tests (none +// mutate the catalog or stored credentials), so a single instance is shared via +// createAgentSession's `modelRegistry`/`authStorage` seam. The registry pins +// itself to its AuthStorage, so both MUST be the same shared instances. +let sharedModelRegistry: ModelRegistry; +let sharedRegistryDir: string; + async function createSessionHarness( tempDir: string, sessionManager: SessionManager, options: { provider?: Parameters[0]; modelId?: string } = {}, -): Promise<{ session: AgentSession; authStorage: AuthStorage }> { +): Promise<{ session: AgentSession }> { const { provider = "openai", modelId = "gpt-5-mini" } = options; - const [{ createAgentSession }, { Settings }, { AuthStorage }] = await Promise.all([ - import("@oh-my-pi/pi-coding-agent/sdk"), - import("@oh-my-pi/pi-coding-agent/config/settings"), - import("@oh-my-pi/pi-coding-agent/session/auth-storage"), - ]); - const authStorage = await AuthStorage.create(path.join(tempDir, `testauth-${Snowflake.next()}.db`)); - authStorage.setRuntimeApiKey("openai", "test-key"); - authStorage.setRuntimeApiKey("openai-codex", "test-key"); const model = getBundledModel(provider, modelId); if (!model) { throw new Error(`Expected bundled test model ${provider}/${modelId}`); @@ -231,7 +235,8 @@ async function createSessionHarness( const { session } = await createAgentSession({ cwd: tempDir, agentDir: tempDir, - authStorage, + authStorage: sharedModelRegistry.authStorage, + modelRegistry: sharedModelRegistry, sessionManager, model, settings: Settings.isolated(), @@ -242,23 +247,42 @@ async function createSessionHarness( slashCommands: [], enableMCP: false, enableLsp: false, + // These tests exercise session reload/sanitization/provider-state, never tool + // execution, rule resolution, or the workspace-tree render. A minimal tool set + // plus empty rules and a prebuilt (empty) workspace tree skip the per-call + // startup scans (native listWorkspace + rule capability discovery) without + // touching any asserted behavior. + rules: [], + workspaceTree: { rootPath: tempDir, rendered: "", truncated: false, totalLines: 0, agentsMdFiles: [] }, + toolNames: ["read"], }); - return { session, authStorage }; + return { session }; } describe("AgentSession OpenAI Responses replay boundaries", () => { const sessions: AgentSession[] = []; - const authStorages: AuthStorage[] = []; const tempDirs: string[] = []; + beforeAll(async () => { + sharedRegistryDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-registry-${Snowflake.next()}-`)); + const authStorage = await AuthStorage.create(path.join(sharedRegistryDir, "auth.db")); + authStorage.setRuntimeApiKey("openai", "test-key"); + authStorage.setRuntimeApiKey("openai-codex", "test-key"); + sharedModelRegistry = new ModelRegistry(authStorage); + }); + + afterAll(() => { + sharedModelRegistry?.authStorage.close(); + if (sharedRegistryDir && fs.existsSync(sharedRegistryDir)) { + fs.rmSync(sharedRegistryDir, { recursive: true, force: true }); + } + }); + afterEach(async () => { while (sessions.length > 0) { await sessions.pop()?.dispose(); } - while (authStorages.length > 0) { - authStorages.pop()?.close(); - } while (tempDirs.length > 0) { const tempDir = tempDirs.pop(); if (tempDir && fs.existsSync(tempDir)) { @@ -285,9 +309,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { }); const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager); + const { session } = await createSessionHarness(tempDir, reloadedSessionManager); sessions.push(session); - authStorages.push(authStorage); const persistedUser = findPersistedMessageEntry(session.sessionManager, "user", "Preserved summary").message; if (persistedUser.role !== "user") { @@ -383,12 +406,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { }); const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, { + const { session } = await createSessionHarness(tempDir, reloadedSessionManager, { provider: "openai-codex", modelId: "gpt-5.2-codex", }); sessions.push(session); - authStorages.push(authStorage); const closeSpy = vi.fn(); session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState); @@ -421,12 +443,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { }); const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, { + const { session } = await createSessionHarness(tempDir, reloadedSessionManager, { provider: "openai-codex", modelId: "gpt-5.2-codex", }); sessions.push(session); - authStorages.push(authStorage); const closeSpy = vi.fn(); session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState); @@ -476,9 +497,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-reload-proxy-${Snowflake.next()}-`)); tempDirs.push(tempDir); const sessionManager = SessionManager.create(tempDir, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, sessionManager); + const { session } = await createSessionHarness(tempDir, sessionManager); sessions.push(session); - authStorages.push(authStorage); const proxyDetails = new Proxy({ ok: true, nested: { value: "preserved" } }, {}); await session.sendCustomMessage( @@ -514,12 +534,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { }); const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, { + const { session } = await createSessionHarness(tempDir, reloadedSessionManager, { provider: "openai-codex", modelId: "gpt-5.2-codex", }); sessions.push(session); - authStorages.push(authStorage); const closeSpy = vi.fn(); session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState); @@ -561,12 +580,11 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { }); const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager, { + const { session } = await createSessionHarness(tempDir, reloadedSessionManager, { provider: "openai-codex", modelId: "gpt-5.2-codex", }); sessions.push(session); - authStorages.push(authStorage); const closeSpy = vi.fn(); session.providerSessionState.set("openai-codex-responses", { close: closeSpy } satisfies ProviderSessionState); @@ -597,9 +615,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { }); const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager); + const { session } = await createSessionHarness(tempDir, reloadedSessionManager); sessions.push(session); - authStorages.push(authStorage); const closeSpy = vi.fn(); session.providerSessionState.set("openai-responses:openai", { close: closeSpy } satisfies ProviderSessionState); @@ -623,9 +640,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-switch-fail-${Snowflake.next()}-`)); tempDirs.push(tempDir); const currentSessionManager = SessionManager.create(tempDir, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, currentSessionManager); + const { session } = await createSessionHarness(tempDir, currentSessionManager); sessions.push(session); - authStorages.push(authStorage); const { sessionFile } = await createPersistedSession(tempDir, sessionManager => { appendStaleAssistantTurn(sessionManager, "Unreadable assistant snapshot"); @@ -657,9 +673,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { const assistantText = "Switched assistant response"; const currentSessionManager = SessionManager.create(tempDir, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, currentSessionManager); + const { session } = await createSessionHarness(tempDir, currentSessionManager); sessions.push(session); - authStorages.push(authStorage); const { sessionFile } = await createPersistedSession(tempDir, sessionManager => { sessionManager.appendMessage({ @@ -723,9 +738,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { } const reloadedSessionManager = await SessionManager.open(sessionFile, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, reloadedSessionManager); + const { session } = await createSessionHarness(tempDir, reloadedSessionManager); sessions.push(session); - authStorages.push(authStorage); const navigation = await session.navigateTree(treeTargetId, { summarize: false }); expect(navigation.cancelled).toBe(false); @@ -746,9 +760,8 @@ describe("AgentSession OpenAI Responses replay boundaries", () => { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-issue-505-new-${Snowflake.next()}-`)); tempDirs.push(tempDir); const sessionManager = SessionManager.create(tempDir, tempDir); - const { session, authStorage } = await createSessionHarness(tempDir, sessionManager); + const { session } = await createSessionHarness(tempDir, sessionManager); sessions.push(session); - authStorages.push(authStorage); const closeSpy = vi.fn(); session.providerSessionState.set("live-provider-session", { close: closeSpy } satisfies ProviderSessionState); diff --git a/packages/coding-agent/test/agent-session-python-cleanup.test.ts b/packages/coding-agent/test/agent-session-python-cleanup.test.ts index 478fc5fb7..c97f93ec7 100644 --- a/packages/coding-agent/test/agent-session-python-cleanup.test.ts +++ b/packages/coding-agent/test/agent-session-python-cleanup.test.ts @@ -116,6 +116,7 @@ const createSession = async ( disableExtensionDiscovery: true, extensions: options.extensions, skills: [], + rules: [], contextFiles: [], promptTemplates: [], workspaceTree: emptyWorkspaceTree(cwd), @@ -195,6 +196,7 @@ describe("AgentSession python cleanup", () => { disableExtensionDiscovery: true, extensions: [throwingExtension], skills: [], + rules: [], contextFiles: [], promptTemplates: [], slashCommands: [], @@ -261,6 +263,7 @@ describe("AgentSession python cleanup", () => { model: getModel(), disableExtensionDiscovery: true, skills: [], + rules: [], contextFiles: [], promptTemplates: [], slashCommands: [], diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index c9dc6d0ff..54c852f5f 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { scheduler } from "node:timers/promises"; import { Agent } from "@oh-my-pi/pi-agent-core"; @@ -66,16 +66,34 @@ function createFallbackAgent(primaryModel: Model, requestedModels: string[]): Ag describe("AgentSession retry fallback", () => { let tempDir: TempDir; let authStorage: AuthStorage; + let sharedRegistry: ModelRegistry; let modelRegistry: ModelRegistry; let session: AgentSession | undefined; - beforeEach(async () => { + // The model registry is an immutable fixture whose construction builds a + // canonical index over ~2.7k bundled models (~100ms). Build it (and the + // auth DB) once for the whole file instead of per-test; reset only the + // mutable retry-fallback cooldown state between tests. + beforeAll(async () => { tempDir = TempDir.createSync("@pi-retry-fallback-"); authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); authStorage.setRuntimeApiKey("anthropic", "anthropic-test-key"); authStorage.setRuntimeApiKey("openai", "openai-test-key"); authStorage.setRuntimeApiKey("google", "google-test-key"); - modelRegistry = new ModelRegistry(authStorage); + sharedRegistry = new ModelRegistry(authStorage); + }); + + afterAll(() => { + authStorage.close(); + tempDir.removeSync(); + }); + + beforeEach(() => { + // Reset to the shared registry (a few tests reassign it to a scoped + // instance) and clear cooldown suppressions left by fallback-path tests + // (default 5-minute suppression) so state never leaks between tests. + modelRegistry = sharedRegistry; + modelRegistry.clearSuppressedSelectors(); }); afterEach(async () => { @@ -83,8 +101,6 @@ describe("AgentSession retry fallback", () => { await session.dispose(); session = undefined; } - authStorage.close(); - tempDir.removeSync(); vi.restoreAllMocks(); }); diff --git a/packages/coding-agent/test/autoresearch-tools.test.ts b/packages/coding-agent/test/autoresearch-tools.test.ts index 3b2c173f0..8dbe03a28 100644 --- a/packages/coding-agent/test/autoresearch-tools.test.ts +++ b/packages/coding-agent/test/autoresearch-tools.test.ts @@ -68,16 +68,16 @@ function createPiHarness(initialTools: string[] = []): PiHarness { return { api, activeTools, appendEntries, setActiveToolsCalls }; } -async function initGitRepo(dir: string): Promise<{ baselineCommit: string; mainBranch: string }> { - await $`git init --initial-branch=main`.cwd(dir).quiet(); - await $`git config user.email tester@example.com`.cwd(dir).quiet(); - await $`git config user.name Tester`.cwd(dir).quiet(); +async function initGitRepo(dir: string): Promise<{ baselineCommit: string }> { await Bun.write(path.join(dir, "README.md"), "# baseline\n"); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m baseline`.cwd(dir).quiet(); + // One shell invocation instead of five: the git processes are unavoidable + // (config identity is read by the production tool's own commits), but + // chaining collapses the per-call Node↔shell spawn overhead. + await $`git init --initial-branch=main && git config user.email tester@example.com && git config user.name Tester && git add -A && git commit -m baseline` + .cwd(dir) + .quiet(); const sha = (await $`git rev-parse HEAD`.cwd(dir).text()).trim(); - const branch = (await $`git rev-parse --abbrev-ref HEAD`.cwd(dir).text()).trim(); - return { baselineCommit: sha, mainBranch: branch }; + return { baselineCommit: sha }; } async function checkoutBranch(dir: string, name: string): Promise { @@ -585,8 +585,7 @@ describe("log_experiment", () => { // Commit `src/edit-me.ts` to baseline so it is tracked, not in pre-run dirty paths. fs.mkdirSync(path.join(dir, "src"), { recursive: true }); await Bun.write(path.join(dir, "src", "edit-me.ts"), "export const v = 1;\n"); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m seed`.cwd(dir).quiet(); + await $`git add -A && git commit -m seed`.cwd(dir).quiet(); const runtime = createSessionRuntime(); const harness = createPiHarness(); const init = createInitExperimentTool({ @@ -638,8 +637,7 @@ describe("log_experiment", () => { await initGitRepo(dir); // Commit the harness on main so it is part of the autoresearch branch's baseline. await writeHarnessStub(dir); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m harness`.cwd(dir).quiet(); + await $`git add -A && git commit -m harness`.cwd(dir).quiet(); await checkoutBranch(dir, "autoresearch/test-20260501"); const runtime = createSessionRuntime(); const harness = createPiHarness(); @@ -651,8 +649,7 @@ describe("log_experiment", () => { await init.execute("i", { name: "x", primary_metric: "m" }, undefined, undefined, createCtx(dir)); // Simulate a previously kept iteration by committing it directly on the branch. await Bun.write(path.join(dir, "src", "kept.ts"), "export const v = 1;\n"); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m "kept iteration"`.cwd(dir).quiet(); + await $`git add -A && git commit -m "kept iteration"`.cwd(dir).quiet(); const headBeforeDiscard = (await $`git rev-parse HEAD`.cwd(dir).text()).trim(); const run = createRunExperimentTool({ @@ -691,13 +688,11 @@ describe("log_experiment", () => { const dir = makeTempDir(); await initGitRepo(dir); await writeHarnessStub(dir); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m harness`.cwd(dir).quiet(); + await $`git add -A && git commit -m harness`.cwd(dir).quiet(); // Seed a tracked file that the agent will edit during the iteration. fs.mkdirSync(path.join(dir, "src"), { recursive: true }); await Bun.write(path.join(dir, "src", "store.ts"), "export const v = 1;\n"); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m seed`.cwd(dir).quiet(); + await $`git add -A && git commit -m seed`.cwd(dir).quiet(); await checkoutBranch(dir, "autoresearch/keep-test"); const runtime = createSessionRuntime(); const harness = createPiHarness(); @@ -747,8 +742,7 @@ describe("log_experiment", () => { const dir = makeTempDir(); await initGitRepo(dir); await writeHarnessStub(dir); - await $`git add -A`.cwd(dir).quiet(); - await $`git commit -m harness`.cwd(dir).quiet(); + await $`git add -A && git commit -m harness`.cwd(dir).quiet(); await checkoutBranch(dir, "autoresearch/scope-test"); const runtime = createSessionRuntime(); const harness = createPiHarness(); diff --git a/packages/coding-agent/test/bash-executor.test.ts b/packages/coding-agent/test/bash-executor.test.ts index 96d92f5b8..3ff685827 100644 --- a/packages/coding-agent/test/bash-executor.test.ts +++ b/packages/coding-agent/test/bash-executor.test.ts @@ -14,12 +14,27 @@ import * as piNatives from "@oh-my-pi/pi-natives"; const ARTIFACT_HEAD_BYTES_DEFAULT = 20 * 1024; const BACKGROUND_COMPLETION_RACE_MS = 750; const KILL_MARKER_DELAY_SECONDS = "0.4"; -const KILL_MARKER_ASSERTION_WAIT_MS = 900; +const KILL_MARKER_DELAY_MS = 400; +// We prove a killed process never wrote its marker by observing until the +// wall-clock instant the marker WOULD have appeared (spawn + delay) plus a +// margin. Anchoring the deadline to a pre-spawn timestamp — instead of blindly +// sleeping a fixed amount after executeBash returns — keeps the wait bounded +// without shrinking the kill-propagation margin: the timeout/abort fires at +// ~100ms, well before the 400ms marker write, so the margin between kill and +// write is unchanged; only the redundant observation tail goes away. +const KILL_MARKER_OBSERVE_MARGIN_MS = 300; function makeTempDir(): string { return fs.mkdtempSync(path.join(os.tmpdir(), "omp-bash-exec-")); } +/** Spin-wait until the wall-clock deadline, polling rather than blind-sleeping. */ +async function waitUntil(deadlineMs: number): Promise { + while (Date.now() < deadlineMs) { + await Bun.sleep(20); + } +} + describe("executeBash", () => { let tempDir: string; @@ -532,6 +547,7 @@ describe("executeBash", () => { const markerEscaped = marker.replace(/'/g, "'\\''"); // Command creates marker after a short delay, but we timeout before then. + const start = Date.now(); const result = await executeBash(`sleep ${KILL_MARKER_DELAY_SECONDS} && echo done > '${markerEscaped}'`, { cwd: tempDir, timeout: 100, @@ -539,10 +555,9 @@ describe("executeBash", () => { expect(result.cancelled).toBe(true); - // Wait longer than the command would have needed to create the marker. - await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS); - - // If process was killed (not orphaned), marker should NOT exist + // Observe past the instant the marker would have been written had the + // process survived. If it was killed (not orphaned), it never appears. + await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS); expect(fs.existsSync(marker)).toBe(false); }); @@ -552,6 +567,7 @@ describe("executeBash", () => { const marker = path.join(tempDir, "marker-bg.txt"); const markerEscaped = marker.replace(/'/g, "'\\''"); + const start = Date.now(); const result = await executeBash( `{ sleep ${KILL_MARKER_DELAY_SECONDS}; echo done > '${markerEscaped}'; } & sleep 10`, { @@ -562,7 +578,7 @@ describe("executeBash", () => { expect(result.cancelled).toBe(true); - await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS); + await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS); expect(fs.existsSync(marker)).toBe(false); }); @@ -597,9 +613,13 @@ describe("executeBash", () => { expect(result.cancelled).toBe(true); expect(result.output).toContain("Command cancelled"); - await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS); + // The backgrounded subshell only writes its marker once `release` exists. + // If abort failed to kill the process group, the orphan is still polling + // for `release` every 50ms — touching it makes a survivor react within one + // poll. A short settle first lets the kill signal propagate before we probe. + await Bun.sleep(100); fs.writeFileSync(release, ""); - await Bun.sleep(150); + await Bun.sleep(200); expect(fs.existsSync(marker)).toBe(false); }); @@ -611,6 +631,7 @@ describe("executeBash", () => { const controller = new AbortController(); // Command creates marker after a short delay. + const start = Date.now(); const promise = executeBash(`sleep ${KILL_MARKER_DELAY_SECONDS} && echo done > '${markerEscaped}'`, { cwd: tempDir, timeout: 10000, @@ -625,10 +646,9 @@ describe("executeBash", () => { expect(result.cancelled).toBe(true); expect(result.output).toContain("Command cancelled"); - // Wait longer than the command would have needed to create the marker. - await Bun.sleep(KILL_MARKER_ASSERTION_WAIT_MS); - - // If process was killed (not orphaned), marker should NOT exist + // Observe past the instant the marker would have been written had the + // process survived. If it was killed (not orphaned), it never appears. + await waitUntil(start + KILL_MARKER_DELAY_MS + KILL_MARKER_OBSERVE_MARGIN_MS); expect(fs.existsSync(marker)).toBe(false); }); }); diff --git a/packages/coding-agent/test/extensions-runner.test.ts b/packages/coding-agent/test/extensions-runner.test.ts index 33e188b7f..7e0005a43 100644 --- a/packages/coding-agent/test/extensions-runner.test.ts +++ b/packages/coding-agent/test/extensions-runner.test.ts @@ -2,7 +2,7 @@ * Tests for ExtensionRunner - conflict detection, error handling, tool wrapping. */ -import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as path from "node:path"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -20,21 +20,34 @@ describe("ExtensionRunner", () => { let tempDir: TempDir; let extensionsDir: string; let sessionManager: SessionManager; + // Shared immutable fixtures. ModelRegistry's constructor synchronously loads + // every bundled model and rebuilds the canonical index (~100ms); these tests + // never mutate the registry or auth storage, so build them once per file + // instead of paying that cost in every beforeEach. + let sharedTempDir: TempDir; let modelRegistry: ModelRegistry; let authStorage: AuthStorage; - beforeEach(async () => { + beforeAll(async () => { + sharedTempDir = TempDir.createSync("@pi-runner-shared-"); + authStorage = await AuthStorage.create(path.join(sharedTempDir.path(), "testauth.db")); + modelRegistry = new ModelRegistry(authStorage); + }); + + afterAll(() => { + authStorage.close(); + sharedTempDir.removeSync(); + }); + + beforeEach(() => { tempDir = TempDir.createSync("@pi-runner-test-"); extensionsDir = path.join(getProjectAgentDir(tempDir.path()), "extensions"); fs.mkdirSync(extensionsDir, { recursive: true }); sessionManager = SessionManager.inMemory(); - authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); - modelRegistry = new ModelRegistry(authStorage); }); afterEach(() => { testSetExtensionHandlerTimeoutMs(EXTENSION_HANDLER_TIMEOUT_MS); - authStorage.close(); tempDir.removeSync(); }); diff --git a/packages/coding-agent/test/goals/goal-mode-integration.test.ts b/packages/coding-agent/test/goals/goal-mode-integration.test.ts index 4eafffe64..83d34f4cc 100644 --- a/packages/coding-agent/test/goals/goal-mode-integration.test.ts +++ b/packages/coding-agent/test/goals/goal-mode-integration.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; import * as path from "node:path"; import { Agent } from "@oh-my-pi/pi-agent-core"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -25,7 +25,6 @@ function createToolSession(cwd: string, settings: Settings, overrides: Partial Promise; }; -async function createGoalHarness(): Promise { - resetSettingsForTest(); - const tempDir = TempDir.createSync("@pi-goal-mode-"); - await Settings.init({ inMemory: true, cwd: tempDir.path() }); - const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); +// Immutable, expensive fixtures shared across every test. `new ModelRegistry` +// alone is ~110ms (loads + parses the bundled model catalog), which dominated +// this file's wall time when rebuilt per test. The registry, its auth storage, +// and the resolved model are never mutated by goal-mode flows, and +// AgentSession.dispose() never closes authStorage — so a single shared instance +// is safe and drops ~8×110ms of pure setup overhead. +type SharedFixture = { + authStorage: AuthStorage; + modelRegistry: ModelRegistry; + model: NonNullable>; + baseDir: TempDir; +}; + +async function createSharedFixture(): Promise { + const baseDir = TempDir.createSync("@pi-goal-mode-shared-"); + const authStorage = await AuthStorage.create(path.join(baseDir.path(), "testauth.db")); const modelRegistry = new ModelRegistry(authStorage); const model = modelRegistry.find("anthropic", "claude-sonnet-4-5"); if (!model) { throw new Error("Expected claude-sonnet-4-5 to exist in registry"); } + return { authStorage, modelRegistry, model, baseDir }; +} + +async function createGoalHarness(shared: SharedFixture): Promise { + resetSettingsForTest(); + const tempDir = TempDir.createSync("@pi-goal-mode-"); + await Settings.init({ inMemory: true, cwd: tempDir.path() }); + const { modelRegistry, model } = shared; const settings = Settings.isolated({ "compaction.enabled": false, @@ -77,7 +95,6 @@ async function createGoalHarness(): Promise { return { tempDir, - authStorage, settings, session, mode, @@ -85,7 +102,6 @@ async function createGoalHarness(): Promise { cleanup: async () => { mode.stop(); await session.dispose(); - authStorage.close(); tempDir.removeSync(); resetSettingsForTest(); }, @@ -98,13 +114,20 @@ async function toolNamesFor(harness: GoalHarness): Promise { describe("InteractiveMode goal mode integration", () => { let harness: GoalHarness; + let shared: SharedFixture; - beforeAll(() => { + beforeAll(async () => { initTheme(); + shared = await createSharedFixture(); + }); + + afterAll(() => { + shared.authStorage.close(); + shared.baseDir.removeSync(); }); beforeEach(async () => { - harness = await createGoalHarness(); + harness = await createGoalHarness(shared); }); afterEach(async () => { diff --git a/packages/coding-agent/test/interactive-mode-plan-review.test.ts b/packages/coding-agent/test/interactive-mode-plan-review.test.ts index 6fb7b28eb..85ae3f784 100644 --- a/packages/coding-agent/test/interactive-mode-plan-review.test.ts +++ b/packages/coding-agent/test/interactive-mode-plan-review.test.ts @@ -68,7 +68,6 @@ describe("InteractiveMode plan review rendering", () => { }); beforeEach(async () => { - Bun.gc(true); resetSettingsForTest(); tempDir = TempDir.createSync("@pi-plan-review-"); await Settings.init({ inMemory: true, cwd: tempDir.path() }); @@ -111,7 +110,6 @@ describe("InteractiveMode plan review rendering", () => { currentAuthStorage?.close(); currentTempDir?.removeSync(); resetSettingsForTest(); - Bun.gc(true); }); it("appends each submitted plan review preview to preserve scrollback", async () => { diff --git a/packages/coding-agent/test/keybindings-selector-navigation.test.ts b/packages/coding-agent/test/keybindings-selector-navigation.test.ts index e79557af5..78dd76e76 100644 --- a/packages/coding-agent/test/keybindings-selector-navigation.test.ts +++ b/packages/coding-agent/test/keybindings-selector-navigation.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeAll, describe, expect, it } from "bun:test"; +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; @@ -86,8 +86,15 @@ async function createHistoryStorage(prompts: string[]): Promise tempDirs.push(dir); HistoryStorage.resetInstance(); const storage = HistoryStorage.open(path.join(dir, "history.db")); - for (const prompt of prompts) { - await storage.add(prompt); + // add() batches writes behind a 100ms AsyncDrain timer. Drive that timer with + // fake timers so the flush is instant instead of waiting real wall-clock time. + vi.useFakeTimers(); + try { + const writes = prompts.map(prompt => storage.add(prompt)); + vi.advanceTimersByTime(100); + await Promise.all(writes); + } finally { + vi.useRealTimers(); } return storage; } diff --git a/packages/coding-agent/test/mcp-reconnect-storm.test.ts b/packages/coding-agent/test/mcp-reconnect-storm.test.ts index 6f75e8955..87a5ef49a 100644 --- a/packages/coding-agent/test/mcp-reconnect-storm.test.ts +++ b/packages/coding-agent/test/mcp-reconnect-storm.test.ts @@ -52,10 +52,19 @@ describe("MCP reconnect storm (issue #1592)", () => { try { await manager.connectServers({ crashy: config }, {}); - // Give the reconnect loop generous time to fire. With the bug this - // produced thousands of processes within a second; with the fix the - // circuit breaker caps the per-server spawn budget. - await Bun.sleep(3000); + // Wait for the circuit breaker to trip rather than blind-sleeping a + // fixed budget. During the storm `getConnectionStatus` is always + // "connected" or "connecting" (`#pendingReconnections` is set + // synchronously before any await in `#doReconnect`); it only reports + // "disconnected" once `#tripReconnectBreaker` opens, tears down the + // stale connection, and detaches `onClose` so no further spawns fire. + // That makes the terminal state a race-free signal: poll for it and + // return the instant the storm is capped instead of waiting out a + // fixed 3s. Generous deadline stays well under the 15s test timeout. + const deadline = Date.now() + 10_000; + while (manager.getConnectionStatus("crashy") !== "disconnected" && Date.now() < deadline) { + await Bun.sleep(5); + } const spawns = countSpawns(); // `RECONNECT_BURST_LIMIT` (5) is the per-server reconnect cap inside diff --git a/packages/coding-agent/test/model-registry-runtime-provider.test.ts b/packages/coding-agent/test/model-registry-runtime-provider.test.ts index 9538a0e34..30e40efbb 100644 --- a/packages/coding-agent/test/model-registry-runtime-provider.test.ts +++ b/packages/coding-agent/test/model-registry-runtime-provider.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import { afterEach, beforeEach, describe, expect, type Mock, spyOn, test } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -13,6 +13,9 @@ describe("ModelRegistry runtime provider registration", () => { let tempDir: string; let modelsJsonPath: string; let authStorage: AuthStorage; + // Neutralizes real network egress during "online" refresh tests so the merge + // path runs without wall-clock-bound DNS/socket latency. Restored in afterEach. + let fetchSpy: Mock | undefined; const sourceIds = ["ext://atomic", "ext://runtime", "ext://oauth"]; @@ -24,6 +27,8 @@ describe("ModelRegistry runtime provider registration", () => { }); afterEach(() => { + fetchSpy?.mockRestore(); + fetchSpy = undefined; clearCustomApis(); for (const sourceId of sourceIds) { unregisterOAuthProviders(sourceId); @@ -192,6 +197,11 @@ describe("ModelRegistry runtime provider registration", () => { }); test("extension-registered models survive refresh('online') cycle", async () => { + // The contract is overlay survival through the full online refresh path + // (static reload + discovery + merge), not discovery success. Stub fetch so + // the online branch runs identically to production-with-no-reachable-providers + // without paying real network latency (~400ms of DNS/socket time otherwise). + fetchSpy = spyOn(globalThis, "fetch").mockRejectedValue(new Error("network disabled in test")); const registry = new ModelRegistry(authStorage, modelsJsonPath); const config: ProviderConfigInput = { baseUrl: "https://runtime.example.com/v1", diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 0cdb092ca..218d98a22 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -29,7 +29,10 @@ describe("ModelRegistry", () => { fs.mkdirSync(tempDir, { recursive: true }); modelsJsonPath = path.join(tempDir, "models.json"); cacheDbPath = path.join(tempDir, "models.db"); - authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db")); + // In-memory auth DB: tests need a fresh, isolated credential store per case but + // never reopen it from disk, so :memory: avoids the WAL/chmod disk-open cost + // (~3ms/test) while preserving per-test isolation. + authStorage = await AuthStorage.create(":memory:"); }); afterEach(() => { diff --git a/packages/coding-agent/test/modes/components/transcript-container.test.ts b/packages/coding-agent/test/modes/components/transcript-container.test.ts index 6baf1bfac..a1b385eae 100644 --- a/packages/coding-agent/test/modes/components/transcript-container.test.ts +++ b/packages/coding-agent/test/modes/components/transcript-container.test.ts @@ -173,6 +173,25 @@ describe("TranscriptContainer", () => { expect(container.render(40)).toEqual(["a-final", "b1"]); // reconciled }); + it("invalidate() retires frozen snapshots so resetDisplay reflects current state", () => { + // resetDisplay() (Ctrl+L, and the Ctrl+O expand path) reflows by calling + // TUI.invalidate(), which propagates to this container. That must retire the + // frozen snapshots the same way thaw() does, or a forced full replay would + // still emit the pre-mutation (e.g. collapsed) render. + riskFlag.eagerEraseScrollbackRisk = true; + const container = new TranscriptContainer(); + const a = new MutableBlock(["a-collapsed"]); + const b = new MutableBlock(["b1"]); + container.addChild(a); + container.addChild(b); + container.render(40); + a.set(["a-expanded-1", "a-expanded-2"]); + expect(container.render(40)).toEqual(["a-collapsed", "b1"]); // frozen + + container.invalidate(); + expect(container.render(40)).toEqual(["a-expanded-1", "a-expanded-2", "b1"]); + }); + it("recomputes a frozen block on a width change", () => { riskFlag.eagerEraseScrollbackRisk = true; const container = new TranscriptContainer(); diff --git a/packages/coding-agent/test/modes/controllers/input-controller-tool-expansion.test.ts b/packages/coding-agent/test/modes/controllers/input-controller-tool-expansion.test.ts index bcf3900ea..38370dca3 100644 --- a/packages/coding-agent/test/modes/controllers/input-controller-tool-expansion.test.ts +++ b/packages/coding-agent/test/modes/controllers/input-controller-tool-expansion.test.ts @@ -3,20 +3,25 @@ import { InputController } from "../../../src/modes/controllers/input-controller import type { InteractiveModeContext } from "../../../src/modes/types"; describe("InputController tool output expansion", () => { - it("allows unknown viewport mutation when toggling tool output expansion", () => { + it("expands children and forces a full display reset to bypass frozen snapshots", () => { const expandable = { setExpanded: vi.fn() }; const inert = { render: vi.fn(() => []) }; const requestRender = vi.fn(); + const resetDisplay = vi.fn(); const ctx = { toolOutputExpanded: false, chatContainer: { children: [expandable, inert] }, - ui: { requestRender }, + ui: { requestRender, resetDisplay }, } as unknown as InteractiveModeContext; new InputController(ctx).toggleToolOutputExpansion(); expect(ctx.toolOutputExpanded).toBe(true); expect(expandable.setExpanded).toHaveBeenCalledWith(true); - expect(requestRender).toHaveBeenCalledWith(false, { allowUnknownViewportMutation: true }); + // resetDisplay() is the only path that retires the transcript's frozen + // block snapshots and re-emits the whole transcript at its new heights. + // A plain requestRender would replay the stale (collapsed) snapshots. + expect(resetDisplay).toHaveBeenCalledTimes(1); + expect(requestRender).not.toHaveBeenCalled(); }); }); diff --git a/packages/coding-agent/test/plan-mode-thinking-level.test.ts b/packages/coding-agent/test/plan-mode-thinking-level.test.ts index b5eee3bf3..548593336 100644 --- a/packages/coding-agent/test/plan-mode-thinking-level.test.ts +++ b/packages/coding-agent/test/plan-mode-thinking-level.test.ts @@ -6,7 +6,7 @@ * calls resolveModelRoleValue() but only returns .model, dropping the thinking level. * #applyPlanModeModel() therefore has no thinking level to apply. */ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test"; import * as path from "node:path"; import { Agent, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -22,7 +22,7 @@ describe("plan mode thinking level", () => { let modelRegistry: ModelRegistry; let authStorage: AuthStorage; - beforeEach(async () => { + beforeAll(async () => { tempDir = TempDir.createSync("@pi-plan-thinking-"); authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); authStorage.setRuntimeApiKey("anthropic", "test-key"); @@ -33,6 +33,9 @@ describe("plan mode thinking level", () => { if (session) { await session.dispose(); } + }); + + afterAll(() => { authStorage.close(); tempDir.removeSync(); }); diff --git a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts index eee3e1da3..903c00722 100644 --- a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts +++ b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts @@ -1,14 +1,35 @@ -import { afterEach, describe, expect, it } from "bun:test"; +import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async/job-manager"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { Snowflake } from "@oh-my-pi/pi-utils"; describe("AsyncJobManager singleton across concurrent top-level sessions", () => { const tempDirs: string[] = []; + // Building a ModelRegistry per session is the dominant cost here: createAgentSession + // otherwise runs discoverAuthStorage (a fresh AuthStorage DB create+reload) and a + // background online model refresh for every spawn (~450ms each). The singleton + // ownership behavior under test is independent of model resolution, so we hand every + // session one shared, network-free registry built once (~10ms/session instead). + let sharedTempDir: string; + let sharedAuthStorage: AuthStorage; + let sharedModelRegistry: ModelRegistry; + + beforeAll(async () => { + sharedTempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-sdk-async-singleton-shared-")); + sharedAuthStorage = await AuthStorage.create(path.join(sharedTempDir, "auth.db")); + sharedModelRegistry = new ModelRegistry(sharedAuthStorage, path.join(sharedTempDir, "models.yml")); + }); + + afterAll(() => { + sharedAuthStorage.close(); + fs.rmSync(sharedTempDir, { recursive: true, force: true }); + }); afterEach(async () => { for (const tempDir of tempDirs.splice(0)) { @@ -34,6 +55,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () => slashCommands: [], enableMCP: false, enableLsp: false, + modelRegistry: sharedModelRegistry, }); return session; } @@ -153,6 +175,7 @@ describe("AsyncJobManager singleton across concurrent top-level sessions", () => slashCommands: [], enableMCP: false, enableLsp: false, + modelRegistry: sharedModelRegistry, systemPrompt: () => { throw new Error("forced startup failure"); }, diff --git a/packages/coding-agent/test/sdk-credential-disabled-bridge.test.ts b/packages/coding-agent/test/sdk-credential-disabled-bridge.test.ts index 0001a780a..a01522c99 100644 --- a/packages/coding-agent/test/sdk-credential-disabled-bridge.test.ts +++ b/packages/coding-agent/test/sdk-credential-disabled-bridge.test.ts @@ -93,10 +93,17 @@ describe("createAgentSession credential_disabled subscription", () => { cwd: dirs.cwd, agentDir: dirs.agentDir, authStorage, + // Pin the model registry at a temp models.json. Without an explicit path, ModelRegistry + // loads the developer's real ~/.omp models config on every construction (~100ms each, + // and non-isolated). Pointing it at the (absent) temp file keeps construction at ~2ms and + // avoids leaking host config into the test. Providing the registry also skips the + // fire-and-forget background model discovery, which is irrelevant to credential_disabled. + modelRegistry: new ModelRegistry(authStorage, path.join(dirs.agentDir, "models.json")), settings: Settings.isolated(), disableExtensionDiscovery: true, extensions, skills: [], + rules: [], contextFiles: [], promptTemplates: [], workspaceTree: emptyWorkspaceTree(dirs.cwd), @@ -406,7 +413,7 @@ describe("createAgentSession credential_disabled subscription", () => { embedderEvents.push(event); }, }); - const modelRegistry = new ModelRegistry(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(dirs.agentDir, "models.json")); const ext = makeRecordingExtension(); const { session } = await createAgentSession({ @@ -448,7 +455,7 @@ describe("createAgentSession credential_disabled subscription", () => { const dirs = makeDirs("mismatch"); const registryStorage = await AuthStorage.create(path.join(dirs.agentDir, "agent-registry.db")); const otherStorage = await AuthStorage.create(path.join(dirs.agentDir, "agent-other.db")); - const modelRegistry = new ModelRegistry(registryStorage); + const modelRegistry = new ModelRegistry(registryStorage, path.join(dirs.agentDir, "models-registry.json")); await expect( createAgentSession({ @@ -477,7 +484,7 @@ describe("createAgentSession credential_disabled subscription", () => { // by one microtask so a sync onError() registration lands in time. const dirs = makeDirs("error-routing"); const authStorage = await AuthStorage.create(path.join(dirs.agentDir, "agent.db")); - const modelRegistry = new ModelRegistry(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(dirs.agentDir, "models.json")); try { const throwingExtension: Extension = { path: "test://throwing-credential-disabled", diff --git a/packages/coding-agent/test/sdk-mcp-discovery.test.ts b/packages/coding-agent/test/sdk-mcp-discovery.test.ts index f9d1082b5..ed866c6b2 100644 --- a/packages/coding-agent/test/sdk-mcp-discovery.test.ts +++ b/packages/coding-agent/test/sdk-mcp-discovery.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -47,18 +47,34 @@ const oldSessionMtime = new Date("2000-01-01T00:00:00.000Z"); describe("createAgentSession MCP discovery prompt gating", () => { let tempDir: string; + let registryDir: string; let authStorage: AuthStorage; let modelRegistry: ModelRegistry; - beforeEach(async () => { - tempDir = path.join(os.tmpdir(), `pi-sdk-mcp-discovery-${Snowflake.next()}`); - fs.mkdirSync(tempDir, { recursive: true }); - authStorage = await AuthStorage.create(path.join(tempDir, "auth.db")); + // Immutable across tests: ModelRegistry's constructor eagerly loads the bundled + // model catalog (~120ms). The tests pass models explicitly and never mutate the + // registry (refreshInBackground is skipped when modelRegistry is supplied, and + // extension source sync is empty under disableExtensionDiscovery), so build it once. + beforeAll(async () => { + registryDir = path.join(os.tmpdir(), `pi-sdk-mcp-discovery-registry-${Snowflake.next()}`); + fs.mkdirSync(registryDir, { recursive: true }); + authStorage = await AuthStorage.create(path.join(registryDir, "auth.db")); modelRegistry = new ModelRegistry(authStorage); }); - afterEach(() => { + afterAll(() => { authStorage.close(); + if (registryDir && fs.existsSync(registryDir)) { + fs.rmSync(registryDir, { recursive: true, force: true }); + } + }); + + beforeEach(() => { + tempDir = path.join(os.tmpdir(), `pi-sdk-mcp-discovery-${Snowflake.next()}`); + fs.mkdirSync(tempDir, { recursive: true }); + }); + + afterEach(() => { if (tempDir && fs.existsSync(tempDir)) { fs.rmSync(tempDir, { recursive: true, force: true }); } diff --git a/packages/coding-agent/test/sdk-model-selection.test.ts b/packages/coding-agent/test/sdk-model-selection.test.ts index c702edf29..89f072394 100644 --- a/packages/coding-agent/test/sdk-model-selection.test.ts +++ b/packages/coding-agent/test/sdk-model-selection.test.ts @@ -12,6 +12,7 @@ import { Snowflake } from "@oh-my-pi/pi-utils"; describe("createAgentSession deferred model pattern resolution", () => { let tempDir: string; + const authStoragesToClose: AuthStorage[] = []; beforeEach(() => { tempDir = path.join(os.tmpdir(), `pi-sdk-model-selection-${Snowflake.next()}`); @@ -19,6 +20,10 @@ describe("createAgentSession deferred model pattern resolution", () => { }); afterEach(() => { + for (const authStorage of authStoragesToClose) { + authStorage.close(); + } + authStoragesToClose.length = 0; if (tempDir && fs.existsSync(tempDir)) { fs.rmSync(tempDir, { recursive: true, force: true }); } @@ -52,10 +57,20 @@ describe("createAgentSession deferred model pattern resolution", () => { }); }; - function buildSessionOptions(modelPattern: string) { + async function buildSessionOptions(modelPattern: string) { + // Pass an explicit ModelRegistry so createAgentSession skips its implicit + // ModelRegistry.refreshInBackground() — a network model-discovery pass + // (~250ms/session) that contributes nothing here: the model resolves from + // the inline extension provider, never from network catalogs. Mirrors the + // explicit-registry pattern the resume tests below already rely on. + const authStorage = await AuthStorage.create(path.join(tempDir, "auth.db")); + authStoragesToClose.push(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); return { cwd: tempDir, agentDir: tempDir, + authStorage, + modelRegistry, sessionManager: SessionManager.inMemory(), disableExtensionDiscovery: true, extensions: [providerExtension], @@ -71,7 +86,7 @@ describe("createAgentSession deferred model pattern resolution", () => { test("resolves explicit modelPattern after extension providers register", async () => { const { session, modelFallbackMessage } = await createAgentSession( - buildSessionOptions("runtime-provider/runtime-model"), + await buildSessionOptions("runtime-provider/runtime-model"), ); expect(session.model).toBeDefined(); @@ -82,7 +97,7 @@ describe("createAgentSession deferred model pattern resolution", () => { test("does not silently fallback when explicit modelPattern is unresolved", async () => { const { session, modelFallbackMessage } = await createAgentSession( - buildSessionOptions("missing-provider/missing-model"), + await buildSessionOptions("missing-provider/missing-model"), ); expect(session.model).toBeUndefined(); @@ -95,7 +110,7 @@ describe("createAgentSession deferred model pattern resolution", () => { settings.setModelRole("default", "pi/smol:high"); const { session } = await createAgentSession({ - ...buildSessionOptions("runtime-provider/runtime-reasoning-model"), + ...(await buildSessionOptions("runtime-provider/runtime-reasoning-model")), settings, }); diff --git a/packages/coding-agent/test/sdk-session-isolation.test.ts b/packages/coding-agent/test/sdk-session-isolation.test.ts index 91355299c..7bac57a38 100644 --- a/packages/coding-agent/test/sdk-session-isolation.test.ts +++ b/packages/coding-agent/test/sdk-session-isolation.test.ts @@ -1,12 +1,14 @@ -import { afterEach, describe, expect, it } from "bun:test"; +import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { type AssistantMessage, getBundledModel } from "@oh-my-pi/pi-ai"; import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; import { SecretObfuscator } from "@oh-my-pi/pi-coding-agent/secrets"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { getSessionsDir, Snowflake } from "@oh-my-pi/pi-utils"; @@ -55,6 +57,23 @@ function getAssistantText(message: AssistantMessage | undefined): string { describe("createAgentSession session storage isolation", () => { const tempDirs: string[] = []; + // One shared, fully-populated (bundled models load synchronously in the + // constructor) registry for every case. Passing it via options skips the + // per-call discoverAuthStorage() SQLite open and the refreshInBackground() + // network model probe inside createAgentSession — the two real wall-clock + // sinks here. None of these cases assert on model discovery, so an + // ambient-credential-free in-memory auth store keeps them deterministic. + let sharedAuthStorage: AuthStorage; + let sharedModelRegistry: ModelRegistry; + + beforeAll(async () => { + sharedAuthStorage = await AuthStorage.create(":memory:"); + sharedModelRegistry = new ModelRegistry(sharedAuthStorage); + }); + + afterAll(() => { + sharedAuthStorage.close(); + }); afterEach(async () => { for (const tempDir of tempDirs.splice(0)) { @@ -72,6 +91,7 @@ describe("createAgentSession session storage isolation", () => { const { session } = await createAgentSession({ cwd, agentDir, + modelRegistry: sharedModelRegistry, settings: Settings.isolated(), disableExtensionDiscovery: true, skills: [], @@ -105,6 +125,7 @@ describe("createAgentSession session storage isolation", () => { const { session } = await createAgentSession({ cwd, agentDir, + modelRegistry: sharedModelRegistry, settings: Settings.isolated(), rules: [rule], disableExtensionDiscovery: true, @@ -137,6 +158,7 @@ describe("createAgentSession session storage isolation", () => { const commonOptions = { cwd, agentDir, + modelRegistry: sharedModelRegistry, settings: Settings.isolated({ "secrets.enabled": true }), disableExtensionDiscovery: true, skills: [], @@ -206,6 +228,7 @@ describe("createAgentSession session storage isolation", () => { const { session } = await createAgentSession({ cwd, agentDir, + modelRegistry: sharedModelRegistry, sessionManager: resumedManager, model, settings: Settings.isolated({ "secrets.enabled": true }), diff --git a/packages/coding-agent/test/sdk-skills.test.ts b/packages/coding-agent/test/sdk-skills.test.ts index 81ad42029..15bccd54a 100644 --- a/packages/coding-agent/test/sdk-skills.test.ts +++ b/packages/coding-agent/test/sdk-skills.test.ts @@ -1,10 +1,12 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { Skill } from "@oh-my-pi/pi-coding-agent/sdk"; import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { cleanupTempHome } from "./helpers/temp-home-cleanup"; @@ -24,6 +26,25 @@ describe("createAgentSession skills option", () => { let skillsDir: string; let tempHomeDir = ""; let originalHome: string | undefined; + // Auth storage (SQLite DB) and the model registry are immutable across these tests: skill + // discovery never touches models, and building them per test would make createAgentSession call + // modelRegistry.refreshInBackground(), whose online model discovery saturates the event loop and + // serializes the otherwise-parallel capability scans (~340ms/call). Supplying a prebuilt registry + // skips that refresh entirely (~24ms/call). + let sharedDir: string; + let sharedAuthStorage: AuthStorage; + let sharedModelRegistry: ModelRegistry; + + beforeAll(async () => { + sharedDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-sdk-skills-shared-")); + sharedAuthStorage = await AuthStorage.create(path.join(sharedDir, "auth.db")); + sharedModelRegistry = new ModelRegistry(sharedAuthStorage, path.join(sharedDir, "models.yml")); + }); + + afterAll(() => { + sharedAuthStorage.close(); + fs.rmSync(sharedDir, { recursive: true, force: true }); + }); beforeEach(() => { tempDir = path.join(os.tmpdir(), `pi-sdk-test-${Date.now()}-${Math.random().toString(36).slice(2)}`); @@ -74,6 +95,7 @@ Loaded via symbolic link. cwd: tempDir, agentDir: tempDir, sessionManager: SessionManager.inMemory(), + modelRegistry: sharedModelRegistry, settings: createIsolatedSkillsSettings(), }); @@ -87,6 +109,7 @@ Loaded via symbolic link. cwd: tempDir, agentDir: tempDir, sessionManager: SessionManager.inMemory(), + modelRegistry: sharedModelRegistry, settings: createIsolatedSkillsSettings(), }); @@ -102,6 +125,7 @@ Loaded via symbolic link. cwd: tempDir, agentDir: tempDir, sessionManager: SessionManager.inMemory(), + modelRegistry: sharedModelRegistry, settings: createIsolatedSkillsSettings(), }); @@ -112,6 +136,7 @@ Loaded via symbolic link. cwd: tempDir, agentDir: tempDir, sessionManager: SessionManager.inMemory(), + modelRegistry: sharedModelRegistry, skills: [], // Explicitly empty - like --no-skills settings: createIsolatedSkillsSettings(), }); @@ -135,6 +160,7 @@ Loaded via symbolic link. cwd: tempDir, agentDir: tempDir, sessionManager: SessionManager.inMemory(), + modelRegistry: sharedModelRegistry, skills: [customSkill], settings: createIsolatedSkillsSettings(), }); diff --git a/packages/coding-agent/test/sdk-tool-activation.test.ts b/packages/coding-agent/test/sdk-tool-activation.test.ts index 6758acde1..a4b8da987 100644 --- a/packages/coding-agent/test/sdk-tool-activation.test.ts +++ b/packages/coding-agent/test/sdk-tool-activation.test.ts @@ -4,7 +4,11 @@ import * as os from "node:os"; import * as path from "node:path"; import { getBundledModel } from "@oh-my-pi/pi-ai"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { createAgentSession, type ExtensionFactory } from "@oh-my-pi/pi-coding-agent/sdk"; +import { + type CreateAgentSessionOptions, + createAgentSession, + type ExtensionFactory, +} from "@oh-my-pi/pi-coding-agent/sdk"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { Snowflake } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; @@ -34,6 +38,35 @@ const toolActivationExtension: ExtensionFactory = pi => { describe("createAgentSession defaultInactive tool activation", () => { const tempDirs: string[] = []; + const makeTempDir = (): string => { + const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); + tempDirs.push(tempDir); + fs.mkdirSync(tempDir, { recursive: true }); + return tempDir; + }; + + // Shared options for every session. `rules: []` and `workspaceTree` short-circuit + // the two slow startup scans (rule discovery + native workspace walk, ~100ms each) + // that are irrelevant to tool activation: these tests assert only which tools are + // registered/active and that tool names appear in the system prompt. Each call + // returns fresh `settings`/`sessionManager` instances to keep tests isolated. + const baseOptions = (tempDir: string): CreateAgentSessionOptions => ({ + cwd: tempDir, + agentDir: tempDir, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated(), + model: getBundledModel("openai", "gpt-4o-mini"), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + rules: [], + workspaceTree: { rootPath: tempDir, rendered: "", truncated: false, totalLines: 0, agentsMdFiles: [] }, + }); + afterEach(() => { for (const tempDir of tempDirs.splice(0)) { fs.rmSync(tempDir, { recursive: true, force: true }); @@ -43,24 +76,11 @@ describe("createAgentSession defaultInactive tool activation", () => { }); it("excludes defaultInactive extension tools from the initial active set unless explicitly requested", async () => { - const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); - tempDirs.push(tempDir); - fs.mkdirSync(tempDir, { recursive: true }); + const tempDir = makeTempDir(); const { session } = await createAgentSession({ - cwd: tempDir, - agentDir: tempDir, - sessionManager: SessionManager.inMemory(), - settings: Settings.isolated(), - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, + ...baseOptions(tempDir), extensions: [toolActivationExtension], - skills: [], - contextFiles: [], - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, }); try { @@ -77,24 +97,11 @@ describe("createAgentSession defaultInactive tool activation", () => { }); it("allows explicitly requested defaultInactive extension tools into the initial active set", async () => { - const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); - tempDirs.push(tempDir); - fs.mkdirSync(tempDir, { recursive: true }); + const tempDir = makeTempDir(); const { session } = await createAgentSession({ - cwd: tempDir, - agentDir: tempDir, - sessionManager: SessionManager.inMemory(), - settings: Settings.isolated(), - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, + ...baseOptions(tempDir), extensions: [toolActivationExtension], - skills: [], - contextFiles: [], - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, toolNames: ["read", "default_inactive_tool"], }); @@ -113,23 +120,10 @@ describe("createAgentSession defaultInactive tool activation", () => { // (e.g. `["read", "search", "find", "lsp", "web_search"]`). Without this // invariant, `yield` ended up registered but not active, and the model // could not satisfy the idle-reminder contract that demands a `yield` call. - const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); - tempDirs.push(tempDir); - fs.mkdirSync(tempDir, { recursive: true }); + const tempDir = makeTempDir(); const { session } = await createAgentSession({ - cwd: tempDir, - agentDir: tempDir, - sessionManager: SessionManager.inMemory(), - settings: Settings.isolated(), - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, - skills: [], - contextFiles: [], - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, + ...baseOptions(tempDir), requireYieldTool: true, toolNames: ["read", "search", "find", "web_search"], }); @@ -149,23 +143,10 @@ describe("createAgentSession defaultInactive tool activation", () => { // the registry has no `deferrable` tool, so the previous gate dropped // `resolve` from the registry and plan mode silently activated without // it — leaving the agent stuck after drafting the plan. - const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); - tempDirs.push(tempDir); - fs.mkdirSync(tempDir, { recursive: true }); + const tempDir = makeTempDir(); const { session } = await createAgentSession({ - cwd: tempDir, - agentDir: tempDir, - sessionManager: SessionManager.inMemory(), - settings: Settings.isolated(), - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, - skills: [], - contextFiles: [], - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, + ...baseOptions(tempDir), toolNames: ["read", "search", "find", "web_search"], }); @@ -177,26 +158,14 @@ describe("createAgentSession defaultInactive tool activation", () => { }); it("drops the hidden resolve tool when neither a deferrable tool nor plan mode can use it", async () => { - const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); - tempDirs.push(tempDir); - fs.mkdirSync(tempDir, { recursive: true }); + const tempDir = makeTempDir(); const settings = Settings.isolated(); settings.set("plan.enabled", false); const { session } = await createAgentSession({ - cwd: tempDir, - agentDir: tempDir, - sessionManager: SessionManager.inMemory(), + ...baseOptions(tempDir), settings, - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, - skills: [], - contextFiles: [], - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, toolNames: ["read", "search", "find", "web_search"], }); @@ -208,23 +177,10 @@ describe("createAgentSession defaultInactive tool activation", () => { }); it("does not register the xAI TTS tool unless enabled", async () => { - const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); - tempDirs.push(tempDir); - fs.mkdirSync(tempDir, { recursive: true }); + const tempDir = makeTempDir(); const { session } = await createAgentSession({ - cwd: tempDir, - agentDir: tempDir, - sessionManager: SessionManager.inMemory(), - settings: Settings.isolated(), - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, - skills: [], - contextFiles: [], - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, + ...baseOptions(tempDir), }); try { @@ -237,23 +193,11 @@ describe("createAgentSession defaultInactive tool activation", () => { }); it("registers the xAI TTS tool when enabled", async () => { - const tempDir = path.join(os.tmpdir(), `pi-sdk-tool-activation-${Snowflake.next()}`); - tempDirs.push(tempDir); - fs.mkdirSync(tempDir, { recursive: true }); + const tempDir = makeTempDir(); const { session } = await createAgentSession({ - cwd: tempDir, - agentDir: tempDir, - sessionManager: SessionManager.inMemory(), + ...baseOptions(tempDir), settings: Settings.isolated({ "tts.enabled": true }), - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, - skills: [], - contextFiles: [], - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, }); try { diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index ff0825548..7493cbb1e 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -16,6 +16,7 @@ import { JobTool } from "@oh-my-pi/pi-coding-agent/tools/job"; import { wrapToolWithMetaNotice } from "@oh-my-pi/pi-coding-agent/tools/output-meta"; import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; import { DEFAULT_FILE_LIMIT, MULTI_FILE_PER_FILE_MATCHES, SearchTool } from "@oh-my-pi/pi-coding-agent/tools/search"; +import * as toolTimeouts from "@oh-my-pi/pi-coding-agent/tools/tool-timeouts"; import { WriteTool } from "@oh-my-pi/pi-coding-agent/tools/write"; import { $which, Snowflake } from "@oh-my-pi/pi-utils"; import { unzipSync } from "fflate"; @@ -1164,7 +1165,7 @@ function b() { const updates: string[] = []; const result = await bashTool.execute( "test-call-8-stream", - { command: "for i in 1 2 3; do echo $i; sleep 0.2; done" }, + { command: "for i in 1 2 3; do echo $i; sleep 0.1; done" }, undefined, update => { const text = update.content?.find(c => c.type === "text")?.text ?? ""; @@ -1310,13 +1311,19 @@ function b() { ), ), ); + // Drive the effective timeout via the production clamp seam so the + // backgrounded job times out in ~0.5s instead of a real wall-clock + // second. 0.5s still renders as "1 seconds" in the executor message + // (Math.round), so that delivery assertion is unchanged; the + // auto-background-on-timeout decision path is identical. + vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.5); const result = await autoBackgroundBashTool.execute("test-call-9-auto-timeout-background", { command: "printf 'start\\n'; sleep 1.2; printf 'done\\n'", timeout: 1, }); - expect(result.details?.timeoutSeconds).toBe(1); + expect(result.details?.timeoutSeconds).toBe(0.5); expect(result.details?.async?.state).toBe("running"); expect(getTextOutput(result)).toContain("Background job"); const jobId = result.details?.async?.jobId; @@ -1344,6 +1351,9 @@ function b() { }); it("should respect timeout", async () => { + // Reduce the effective timeout through the production clamp seam; the + // real subprocess kill-on-timeout path is still exercised, just faster. + vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.1); await expect(bashTool.execute("test-call-10", { command: "sleep 5", timeout: 1 })).rejects.toThrow( /timed out/i, ); @@ -1351,10 +1361,19 @@ function b() { it("should abort and recover for subsequent commands", async () => { const controller = new AbortController(); - const promise = bashTool.execute("test-call-10-abort", { command: "sleep 60" }, controller.signal); - // Give the native shell a beat to enter `sleep`; do not depend on chunk - // delivery timing, which is flaky on loaded CI runners. - await Bun.sleep(100); + const started = Promise.withResolvers(); + const promise = bashTool.execute( + "test-call-10-abort", + { command: "echo READY; sleep 60" }, + controller.signal, + update => { + const text = update.content?.find(c => c.type === "text")?.text ?? ""; + if (text.includes("READY")) started.resolve(); + }, + ); + // Abort as soon as the command has emitted output (proving the shell is + // live), instead of blindly waiting a fixed beat for it to enter `sleep`. + await started.promise; controller.abort("test abort"); await expect(promise).rejects.toThrow(/abort|cancel|timed out/i); diff --git a/packages/coding-agent/test/tools/approval-mode.test.ts b/packages/coding-agent/test/tools/approval-mode.test.ts index 9b0709881..f2fa648c2 100644 --- a/packages/coding-agent/test/tools/approval-mode.test.ts +++ b/packages/coding-agent/test/tools/approval-mode.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, it } from "bun:test"; +import { afterAll, beforeAll, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -19,31 +19,6 @@ function emptyWorkspaceTree(cwd: string) { return { rootPath: cwd, rendered: ".\n", truncated: false, totalLines: 1, agentsMdFiles: [] }; } -async function makeSession(extraSettings: Record = {}) { - const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-approval-mode-${Snowflake.next()}-`)); - const cwd = path.join(tempDir, "cwd"); - fs.mkdirSync(cwd, { recursive: true }); - const sessionManager = SessionManager.create(cwd, path.join(tempDir, "sessions")); - const settings = Settings.isolated({ ...BASE_SETTINGS, ...extraSettings }); - const { session } = await createAgentSession({ - cwd, - agentDir: tempDir, - sessionManager, - settings, - model: getBundledModel("openai", "gpt-4o-mini"), - disableExtensionDiscovery: true, - skills: [], - contextFiles: [], - workspaceTree: emptyWorkspaceTree(cwd), - promptTemplates: [], - slashCommands: [], - enableMCP: false, - enableLsp: false, - toolNames: ["bash"], - }); - return { tempDir, session, settings }; -} - function textOf(result: { content?: ReadonlyArray<{ type: string; text?: string }> }): string { const blocks = result.content ?? []; for (const block of blocks) { @@ -53,179 +28,155 @@ function textOf(result: { content?: ReadonlyArray<{ type: string; text?: string } describe("tools.approvalMode setting", () => { - const tempDirs: string[] = []; + // The per-tool approval gate (ExtensionToolWrapper) reads approvalMode / tools.approval / + // autoApprove exclusively from the execute-time AgentToolContext, never from the session's + // own settings. So a single shared session exercises every mode — we only vary the context + // settings per assertion. This avoids paying createAgentSession's cost (model registry, + // auth-storage discovery, settings init) nine times over. + let tempDir: string; + let session: Awaited>["session"]; - afterEach(async () => { - for (const tempDir of tempDirs.splice(0)) { - // Windows can briefly hold tempdir handles after session.dispose(); retry a few times. - for (let attempt = 0; attempt < 5; attempt++) { - try { - fs.rmSync(tempDir, { recursive: true, force: true }); - break; - } catch (err) { - const code = (err as NodeJS.ErrnoException).code; - if (code !== "EBUSY" && code !== "ENOTEMPTY" && code !== "EPERM") throw err; - if (attempt === 4) break; // best-effort: OS will reclaim - await Bun.sleep(50 * (attempt + 1)); - } + beforeAll(async () => { + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-approval-mode-${Snowflake.next()}-`)); + const cwd = path.join(tempDir, "cwd"); + fs.mkdirSync(cwd, { recursive: true }); + const sessionManager = SessionManager.create(cwd, path.join(tempDir, "sessions")); + const created = await createAgentSession({ + cwd, + agentDir: tempDir, + sessionManager, + settings: Settings.isolated(BASE_SETTINGS), + model: getBundledModel("openai", "gpt-4o-mini"), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + workspaceTree: emptyWorkspaceTree(cwd), + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + toolNames: ["bash"], + }); + session = created.session; + }); + + afterAll(async () => { + await session.dispose(); + // Windows can briefly hold tempdir handles after session.dispose(); retry a few times. + for (let attempt = 0; attempt < 5; attempt++) { + try { + fs.rmSync(tempDir, { recursive: true, force: true }); + break; + } catch (err) { + const code = (err as NodeJS.ErrnoException).code; + if (code !== "EBUSY" && code !== "ENOTEMPTY" && code !== "EPERM") throw err; + if (attempt === 4) break; // best-effort: OS will reclaim + await Bun.sleep(50 * (attempt + 1)); } } }); + function approvalSettings(extraSettings: Record = {}): Settings { + return Settings.isolated({ ...BASE_SETTINGS, ...extraSettings }); + } + + function bashTool() { + const bash = session.getToolByName("bash"); + if (!bash) throw new Error("Expected bash tool"); + return bash; + } + it("yolo mode (default) bypasses approval for non-overriding tool calls", async () => { - const { tempDir, session, settings } = await makeSession(); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - const result = await bash.execute("yolo", { command: "echo ok" }, undefined, undefined, { - settings, - } as AgentToolContext); - expect(textOf(result)).toContain("ok"); - } finally { - await session.dispose(); - } + const settings = approvalSettings(); + const result = await bashTool().execute("yolo", { command: "echo ok" }, undefined, undefined, { + settings, + } as AgentToolContext); + expect(textOf(result)).toContain("ok"); }); it("always-ask mode rejects exec tools when no UI is available", async () => { - const { tempDir, session, settings } = await makeSession({ - "tools.approvalMode": "always-ask", - }); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - await expect( - bash.execute("always-ask", { command: "echo blocked" }, undefined, undefined, { - settings, - } as AgentToolContext), - ).rejects.toThrow(/requires approval but no interactive UI available/); - } finally { - await session.dispose(); - } + const settings = approvalSettings({ "tools.approvalMode": "always-ask" }); + await expect( + bashTool().execute("always-ask", { command: "echo blocked" }, undefined, undefined, { + settings, + } as AgentToolContext), + ).rejects.toThrow(/requires approval but no interactive UI available/); }); it("per-tool allow overrides are honored in every mode", async () => { - const { tempDir, session, settings } = await makeSession({ + const settings = approvalSettings({ "tools.approvalMode": "always-ask", "tools.approval": { bash: "allow" }, }); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - const result = await bash.execute("always-ask-allow", { command: "echo allowed" }, undefined, undefined, { - settings, - } as AgentToolContext); - expect(textOf(result)).toContain("allowed"); - } finally { - await session.dispose(); - } + const result = await bashTool().execute("always-ask-allow", { command: "echo allowed" }, undefined, undefined, { + settings, + } as AgentToolContext); + expect(textOf(result)).toContain("allowed"); }); it("per-tool prompt overrides can tighten yolo mode", async () => { - const { tempDir, session, settings } = await makeSession({ + const settings = approvalSettings({ "tools.approvalMode": "yolo", "tools.approval": { bash: "prompt" }, }); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - await expect( - bash.execute("yolo-prompt", { command: "echo blocked" }, undefined, undefined, { - settings, - } as AgentToolContext), - ).rejects.toThrow(/requires approval but no interactive UI available/); - } finally { - await session.dispose(); - } + await expect( + bashTool().execute("yolo-prompt", { command: "echo blocked" }, undefined, undefined, { + settings, + } as AgentToolContext), + ).rejects.toThrow(/requires approval but no interactive UI available/); }); it("write mode still prompts exec-tier tools", async () => { - const { tempDir, session, settings } = await makeSession({ + const settings = approvalSettings({ "tools.approvalMode": "write", "tools.approval": {}, }); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - await expect( - bash.execute("write-mode", { command: "echo unconfigured" }, undefined, undefined, { - settings, - } as AgentToolContext), - ).rejects.toThrow(/requires approval but no interactive UI available/); - } finally { - await session.dispose(); - } + await expect( + bashTool().execute("write-mode", { command: "echo unconfigured" }, undefined, undefined, { + settings, + } as AgentToolContext), + ).rejects.toThrow(/requires approval but no interactive UI available/); }); it("critical bash patterns do not prompt in yolo mode with bash allowed", async () => { - const { tempDir, session, settings } = await makeSession({ + const settings = approvalSettings({ "tools.approvalMode": "yolo", "tools.approval": { bash: "allow" }, }); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - - const result = await bash.execute( - "critical", - { command: "rm -f /tmp/bun-fake-timer-probe.test.ts" }, - undefined, - undefined, - { - settings, - } as AgentToolContext, - ); - expect(textOf(result)).toContain("(no output)"); - } finally { - await session.dispose(); - } + const result = await bashTool().execute( + "critical", + { command: "rm -f /tmp/bun-fake-timer-probe.test.ts" }, + undefined, + undefined, + { + settings, + } as AgentToolContext, + ); + expect(textOf(result)).toContain("(no output)"); }); it("CLI --auto-approve forces yolo mode for non-overriding tool calls", async () => { - const { tempDir, session, settings } = await makeSession({ - "tools.approvalMode": "always-ask", - }); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - const result = await bash.execute("cli-override", { command: "echo override" }, undefined, undefined, { - settings, - autoApprove: true, - } as AgentToolContext); - expect(textOf(result)).toContain("override"); - } finally { - await session.dispose(); - } + const settings = approvalSettings({ "tools.approvalMode": "always-ask" }); + const result = await bashTool().execute("cli-override", { command: "echo override" }, undefined, undefined, { + settings, + autoApprove: true, + } as AgentToolContext); + expect(textOf(result)).toContain("override"); }); it("CLI --auto-approve also bypasses safety-override patterns", async () => { - const { tempDir, session, settings } = await makeSession({ - "tools.approvalMode": "always-ask", - }); - tempDirs.push(tempDir); - try { - const bash = session.getToolByName("bash"); - if (!bash) throw new Error("Expected bash tool"); - const result = await bash.execute( - "cli-critical", - { command: "rm -f /tmp/bun-fake-timer-probe.test.ts" }, - undefined, - undefined, - { - settings, - autoApprove: true, - } as AgentToolContext, - ); - expect(textOf(result)).toContain("(no output)"); - } finally { - await session.dispose(); - } + const settings = approvalSettings({ "tools.approvalMode": "always-ask" }); + const result = await bashTool().execute( + "cli-critical", + { command: "rm -f /tmp/bun-fake-timer-probe.test.ts" }, + undefined, + undefined, + { + settings, + autoApprove: true, + } as AgentToolContext, + ); + expect(textOf(result)).toContain("(no output)"); }); it("constructs an extensionRunner unconditionally so the approval gate is always installed", async () => { @@ -236,12 +187,6 @@ describe("tools.approvalMode setting", () => { // any non-yolo approval mode setting would be a no-op without feedback. The // fix is to construct the runner unconditionally; this test makes that contract explicit so // a future change to make the runner optional again cannot silently re-open the hole. - const { tempDir, session } = await makeSession(); - tempDirs.push(tempDir); - try { - expect(session.extensionRunner).toBeDefined(); - } finally { - await session.dispose(); - } + expect(session.extensionRunner).toBeDefined(); }); }); diff --git a/packages/coding-agent/test/tools/conflict-integration.test.ts b/packages/coding-agent/test/tools/conflict-integration.test.ts index 73c784d50..5d1a92a8b 100644 --- a/packages/coding-agent/test/tools/conflict-integration.test.ts +++ b/packages/coding-agent/test/tools/conflict-integration.test.ts @@ -26,7 +26,11 @@ function getText(result: { content: Array<{ type: string; text?: string }> }): s } async function getTool(session: ToolSession, name: "read" | "write") { - const tools = await createTools(session); + // Request only the tool under test: createTools(session) with no toolNames + // builds every builtin factory (LSP, MCP discovery, browser, eval preflight, + // …) on each call, which is pure overhead here. The conflict contract lives + // entirely in the read/write tools + session.conflictHistory. + const tools = await createTools(session, [name]); const tool = tools.find(entry => entry.name === name); if (!tool) throw new Error(`Missing ${name} tool`); return tool; diff --git a/packages/coding-agent/test/tools/fetch-jina-stall.test.ts b/packages/coding-agent/test/tools/fetch-jina-stall.test.ts index 03f33b1dc..7582f1cc2 100644 --- a/packages/coding-agent/test/tools/fetch-jina-stall.test.ts +++ b/packages/coding-agent/test/tools/fetch-jina-stall.test.ts @@ -45,9 +45,12 @@ describe("renderHtmlToText: jina stall does not starve local fallbacks (#1449)", }); const started = Date.now(); - // `timeout: 2` keeps the overall budget tight — the test must complete - // within ~2s even though Jina would otherwise hang for the full budget. - const result = await renderHtmlToText("https://example.com/article", html, 2, settings, undefined, null); + // Tight 300ms reader-mode budget. Jina would otherwise hang forever, but + // the remote sub-budget (min(timeout*1000, REMOTE_READER_MAX_MS)) aborts + // the stalled request so the local native renderer still runs. Kept small + // so the test exercises the same abort path without burning real + // wall-clock time waiting out the stall. + const result = await renderHtmlToText("https://example.com/article", html, 0.3, settings, undefined, null); const elapsedMs = Date.now() - started; expect(result.ok).toBe(true); @@ -56,10 +59,10 @@ describe("renderHtmlToText: jina stall does not starve local fallbacks (#1449)", // If trafilatura or lynx happened to succeed first, that's also a valid // non-aborted outcome. expect(["native", "trafilatura", "lynx"]).toContain(result.method); - // Must finish well before the overall budget elapses: the remote - // sub-budget caps Jina at min(timeout, REMOTE_READER_MAX_MS), so the - // remaining ~1s of the 2s budget is enough for the native renderer. - expect(elapsedMs).toBeLessThan(2_500); + // Must finish shortly after the 300ms budget aborts the stalled Jina + // request — never anywhere near an unbounded hang. The generous bound + // absorbs scheduler jitter under full-suite parallelism. + expect(elapsedMs).toBeLessThan(1_500); }); it("re-throws when the user signal is aborted, not when Jina sub-budget expires", async () => { diff --git a/packages/coding-agent/test/tools/gh.test.ts b/packages/coding-agent/test/tools/gh.test.ts index 2deb69989..b1776f807 100644 --- a/packages/coding-agent/test/tools/gh.test.ts +++ b/packages/coding-agent/test/tools/gh.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, it, vi } from "bun:test"; +import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; @@ -85,15 +85,24 @@ function runGit(cwd: string, args: string[]): string { return new TextDecoder().decode(result.stdout).trim(); } -async function createPrFixture(): Promise<{ +interface PrFixture { baseDir: string; repoRoot: string; originBare: string; forkBare: string; headRefName: string; headRefOid: string; -}> { - const baseDir = await fs.mkdtemp(path.join(os.tmpdir(), "gh-pr-tool-")); +} + +// Building the fixture costs ~16 real `git` subprocess spawns (~200ms). Six +// tests need it, so we build it ONCE as an immutable template in `beforeAll` +// and materialize per-test copies via `fs.cp` (~12ms). Each copy is a fully +// independent repo tree, so the mutating tests (worktree checkout, config +// writes, extra branches) can't contaminate each other. +let prFixtureTemplate: PrFixture | null = null; + +async function buildPrFixtureTemplate(): Promise { + const baseDir = await fs.mkdtemp(path.join(os.tmpdir(), "gh-pr-tool-template-")); const repoRoot = path.join(baseDir, "repo"); const originBare = path.join(baseDir, "origin.git"); const forkBare = path.join(baseDir, "fork.git"); @@ -119,13 +128,32 @@ async function createPrFixture(): Promise<{ runGit(repoRoot, ["push", "-u", "forksrc", headRefName]); runGit(repoRoot, ["checkout", "main"]); + return { baseDir, repoRoot, originBare, forkBare, headRefName, headRefOid }; +} + +async function createPrFixture(): Promise { + const template = prFixtureTemplate; + if (!template) throw new Error("PR fixture template was not built (missing beforeAll)"); + + const baseDir = await fs.mkdtemp(path.join(os.tmpdir(), "gh-pr-tool-")); + const repoRoot = path.join(baseDir, "repo"); + const originBare = path.join(baseDir, "origin.git"); + const forkBare = path.join(baseDir, "fork.git"); + + await fs.cp(template.baseDir, baseDir, { recursive: true }); + // Remote URLs in the copied repo still point at the template's absolute + // `origin.git`/`fork.git`. Repoint them at this copy so pushes/fetches stay + // isolated and `remote get-url` assertions match the returned paths. + runGit(repoRoot, ["remote", "set-url", "origin", originBare]); + runGit(repoRoot, ["remote", "set-url", "forksrc", forkBare]); + return { baseDir, repoRoot, originBare, forkBare, - headRefName, - headRefOid, + headRefName: template.headRefName, + headRefOid: template.headRefOid, }; } @@ -210,6 +238,17 @@ describe("parsePrUnifiedDiff", () => { }); describe("github tool", () => { + beforeAll(async () => { + prFixtureTemplate = await buildPrFixtureTemplate(); + }); + + afterAll(async () => { + if (prFixtureTemplate) { + await fs.rm(prFixtureTemplate.baseDir, { recursive: true, force: true }); + prFixtureTemplate = null; + } + }); + afterEach(() => { vi.useRealTimers(); vi.restoreAllMocks(); diff --git a/packages/coding-agent/test/tools/lsp-regressions.test.ts b/packages/coding-agent/test/tools/lsp-regressions.test.ts index f4e4aa2e6..f246f5545 100644 --- a/packages/coding-agent/test/tools/lsp-regressions.test.ts +++ b/packages/coding-agent/test/tools/lsp-regressions.test.ts @@ -373,6 +373,10 @@ for await (const chunk of Bun.stdin.stream()) { args: [serverPath, eventLogPath, statusCountPath, fileToUri(sourcePath)], fileTypes: ["rs"], rootMarkers: [], + // Shrink the workspace-ready polling window so the test exercises the + // timeout→retry→ready sequence without waiting out the 2s production settle. + // The status-request timeout stays generous to avoid racing the subprocess. + workspaceReadyTimings: { timeoutMs: 5_000, pollMs: 10, settleMs: 20, statusRequestTimeoutMs: 150 }, }; vi.spyOn(lspConfig, "loadConfig").mockReturnValue({ diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index 1693cc817..522f230b1 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -27,6 +27,34 @@ class MutableLinesComponent implements Component { } } +// Models a component that caches its rendered output and only refreshes it when +// `invalidate()` fires — like a transcript block that freezes a snapshot. A +// state change behind the cache is invisible until something invalidates it, +// which is exactly what `resetDisplay()` must do to surface a Ctrl+O expansion. +class CachedComponent implements Component { + #current: string[]; + #cache: string[] | undefined; + + constructor(lines: string[]) { + this.#current = [...lines]; + } + + setLines(lines: string[]): void { + this.#current = [...lines]; + } + + invalidate(): void { + this.#cache = undefined; + } + + render(width: number): string[] { + if (this.#cache === undefined) { + this.#cache = this.#current.map(line => line.slice(0, width)); + } + return this.#cache; + } +} + class WrappingLinesComponent implements Component { #lines: string[]; @@ -366,6 +394,32 @@ describe("TUI terminal-state regressions", () => { } }); + it("resetDisplay surfaces a state change hidden behind a component's render cache", async () => { + const term = new VirtualTerminal(20, 3); + const tui = new TUI(term); + const component = new CachedComponent(rows("L", 8)); + tui.addChild(component); + + try { + tui.start(); + await settle(term); + expect(visible(term)).toEqual(["L5", "L6", "L7"]); + + // The component's content changes, but its render stays cached (a + // frozen transcript snapshot). resetDisplay() must invalidate it so the + // forced replay reflects the new content rather than the stale cache — + // the Ctrl+O expansion path depends on this. + component.setLines(rows("M", 8)); + tui.resetDisplay(); + await settle(term); + + expect(term.getScrollBuffer().map(line => line.trimEnd())).toEqual(rows("M", 8)); + expect(visible(term)).toEqual(["M5", "M6", "M7"]); + } finally { + tui.stop(); + } + }); + it("keeps appended rows in scrollback when a forced render coalesces with content growth", async () => { const term = new VirtualTerminal(20, 3); const tui = new TUI(term); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 23eee711b..95442dc2a 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Changed + +- `logger.printTimings()` (the `PI_TIMING` startup tree) now surfaces two previously-invisible regions: a `(before instrumentation)` line for the runtime init + static module-graph load that elapses before the first marker (the dominant real-world startup cost, ~350ms — `startTiming()` only begins inside `runRootCommand`), and an `(unattributed self)` line for the root span's own untimed work so the gap between the visible top-level spans and `Total` is no longer silently swallowed. `Total` is now labelled `(since first marker)` to make the window explicit. + ## [15.9.2] - 2026-06-05 ### Added diff --git a/packages/utils/src/logger.ts b/packages/utils/src/logger.ts index ae0f7fd36..c682bbf8f 100644 --- a/packages/utils/src/logger.ts +++ b/packages/utils/src/logger.ts @@ -187,6 +187,13 @@ export function printTimings(): void { const lines: string[] = []; lines.push(""); lines.push("--- Startup timings (hierarchical) ---"); + // performance.now() shares the process-start origin, so the root span's start + // is the wall time spent before the first marker — runtime init plus the + // static module-graph evaluation (~the dominant cost). It is otherwise + // invisible because Total only spans startTiming()→printTimings(). + if (gRootSpan.start > LOGGED_TIMING_THRESHOLD_MS) { + lines.push(`(before instrumentation): ${fmtMs(gRootSpan.start)} [runtime init + module load]`); + } const work: Span[] = []; const loads: Span[] = []; for (const child of gRootSpan.children) { @@ -199,8 +206,14 @@ export function printTimings(): void { if (loads.length > 0) { printModuleLoadSummary(loads, 0, lines); } + // Surface the root's own unattributed time so the gap between the visible + // top-level spans and Total isn't silently swallowed. + const rootSelf = selfTimeOf(gRootSpan); + if (gRootSpan.children.length > 0 && rootSelf > LOGGED_TIMING_THRESHOLD_MS) { + lines.push(`(unattributed self): ${fmtMs(rootSelf)}`); + } const totalMs = (gRootSpan.end - gRootSpan.start).toFixed(1); - lines.push(`Total: ${totalMs}ms`); + lines.push(`Total: ${totalMs}ms (since first marker)`); lines.push("--------------------------------------"); lines.push(""); console.error(lines.join("\n")); From fde55bf927d059dfd638c8bcefca6ecbccb3f5f8 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:21:58 +0200 Subject: [PATCH 049/181] fix(model): added bracket-affix stripping and string-keyed resolution cache - Replaced WeakMap model cache with provider/id string keys for stable reuse. - Returned official model ids directly when matched, before heuristics. - Collapsed non-message token path to system prompt and tool schema totals. --- .../src/config/model-equivalence.ts | 31 ++++++--- .../src/config/model-id-affixes.ts | 61 ++++++++++------- .../src/modes/utils/context-usage.ts | 16 +++-- .../test/model-id-affixes.test.ts | 66 +++++++++++++++++++ .../test/status-line-context-cache.test.ts | 34 ++++++++++ 5 files changed, 171 insertions(+), 37 deletions(-) create mode 100644 packages/coding-agent/test/model-id-affixes.test.ts diff --git a/packages/coding-agent/src/config/model-equivalence.ts b/packages/coding-agent/src/config/model-equivalence.ts index 28ae3744a..e30755b2d 100644 --- a/packages/coding-agent/src/config/model-equivalence.ts +++ b/packages/coding-agent/src/config/model-equivalence.ts @@ -58,7 +58,7 @@ const EMPTY_COMPILED_EQUIVALENCE: CompiledEquivalenceConfig = { }; const kModelResolutionCache = Symbol("model-equivalence.resolutionCache"); interface CompiledEquivalenceConfigWithCache extends CompiledEquivalenceConfig { - [kModelResolutionCache]?: WeakMap, ResolvedCanonicalModel>; + [kModelResolutionCache]?: Map; } const FAMILY_EXTRACTION_PATTERNS = [ /(?:^|[/:._-])((?:claude|gemini|gpt|grok|glm|qwen|minimax|kimi|deepseek|llama|gemma|nova|mistral|ministral|pixtral|codestral|devstral|magistral|ernie|doubao|seed|aion|olmo|molmo|nemotron|palmyra|command|codex|coder|o[1345])[-a-z0-9.]+)(?::|$)/i, @@ -128,10 +128,18 @@ function normalizeCanonicalIdKey(canonicalId: string): string { return canonicalId.trim().toLowerCase(); } +function getCanonicalSuffixAliasKey(candidate: string): string { + return PENALTY_HAS_UPPERCASE.test(candidate) ? normalizeCanonicalIdKey(candidate) : candidate; +} + export function formatCanonicalVariantSelector(model: Model): string { return `${model.provider}/${model.id}`; } +function getModelResolutionCacheKey(model: Model): string { + return `${model.provider}\0${model.id}`; +} + function buildOverrideMap(overrides: Record | undefined): Map { const result = new Map(); if (!overrides) { @@ -728,10 +736,10 @@ function getPreferredFallbackCanonicalCandidate(modelId: string, candidates: rea function resolveCanonicalIdForModel( model: Model, + selector: string, equivalence: CompiledEquivalenceConfig, referenceData: CanonicalReferenceData, ): ResolvedCanonicalModel { - const selector = formatCanonicalVariantSelector(model); const normalizedSelector = normalizeSelectorKey(selector); if (equivalence.overrides.has(normalizedSelector)) { @@ -752,10 +760,14 @@ function resolveCanonicalIdForModel( return { id: claudeFamilyAlias, source: claudeFamilyAlias === model.id ? "bundled" : "heuristic" }; } + if (referenceData.officialIds.has(model.id) && !model.id.includes("/") && !model.id.includes(":")) { + return { id: model.id, source: "bundled" }; + } + const heuristicCandidates = getHeuristicCanonicalCandidates(model.id, referenceData.officialIds); const officialMatches = new Set(heuristicCandidates.filter(candidate => referenceData.officialIds.has(candidate))); for (const candidate of heuristicCandidates) { - const aliased = referenceData.suffixAliases.get(normalizeCanonicalIdKey(candidate)); + const aliased = referenceData.suffixAliases.get(getCanonicalSuffixAliasKey(candidate)); if (aliased) { officialMatches.add(aliased); } @@ -814,17 +826,18 @@ export function buildCanonicalModelIndex( const compiledWithCache = compiledEquivalence as CompiledEquivalenceConfigWithCache; let modelCache = compiledWithCache[kModelResolutionCache]; if (!modelCache) { - modelCache = new WeakMap, ResolvedCanonicalModel>(); + modelCache = new Map(); compiledWithCache[kModelResolutionCache] = modelCache; } for (const model of models) { - let canonical = modelCache.get(model); - if (!canonical) { - canonical = resolveCanonicalIdForModel(model, compiledEquivalence, referenceData); - modelCache.set(model, canonical); - } const selector = formatCanonicalVariantSelector(model); + const cacheKey = getModelResolutionCacheKey(model); + let canonical = modelCache.get(cacheKey); + if (!canonical) { + canonical = resolveCanonicalIdForModel(model, selector, compiledEquivalence, referenceData); + modelCache.set(cacheKey, canonical); + } const variant: CanonicalModelVariant = { canonicalId: canonical.id, selector, diff --git a/packages/coding-agent/src/config/model-id-affixes.ts b/packages/coding-agent/src/config/model-id-affixes.ts index b4aff136f..7cec31892 100644 --- a/packages/coding-agent/src/config/model-id-affixes.ts +++ b/packages/coding-agent/src/config/model-id-affixes.ts @@ -4,34 +4,49 @@ const MODEL_ID_SEGMENT_PATTERN = /[a-z0-9.:-]+/g; const MODEL_FAMILY_PREFIX_PATTERN = /^(claude|gemini|gpt|grok|glm|qwen|deepseek|kimi|mimo|doubao|ernie|gpt-oss|gemma|minimax|step|command|jamba|llama|o[1345])/i; -function hasDigit(value: string): boolean { - return /\d/.test(value); +function normalizeModelIdWhitespace(value: string): string { + return value.trim().replace(/\s+/g, " "); } +/** Ordering for model-like segments: longest first, ties broken lexicographically. */ function compareSegmentPreference(left: string, right: string): number { - if (left.length !== right.length) { - return right.length - left.length; - } - return left.localeCompare(right); + return left.length !== right.length ? right.length - left.length : left.localeCompare(right); } export function getModelLikeIdSegments(modelId: string): string[] { - const normalized = normalizeModelIdWhitespace(modelId).toLowerCase(); - if (!normalized) return []; - const segments = (normalized.match(MODEL_ID_SEGMENT_PATTERN) ?? []).filter( - segment => MODEL_FAMILY_PREFIX_PATTERN.test(segment) && hasDigit(segment), - ); - const unique = [...new Set(segments)]; - unique.sort(compareSegmentPreference); - return unique; + const matches = normalizeModelIdWhitespace(modelId).toLowerCase().match(MODEL_ID_SEGMENT_PATTERN); + if (!matches) return []; + const segments = new Set(); + for (const segment of matches) { + if (MODEL_FAMILY_PREFIX_PATTERN.test(segment) && /\d/.test(segment)) segments.add(segment); + } + return [...segments].sort(compareSegmentPreference); } export function getLongestModelLikeIdSegment(modelId: string): string | undefined { - return getModelLikeIdSegments(modelId)[0]; + const matches = normalizeModelIdWhitespace(modelId).toLowerCase().match(MODEL_ID_SEGMENT_PATTERN); + if (!matches) return undefined; + let best: string | undefined; + for (const segment of matches) { + if ( + MODEL_FAMILY_PREFIX_PATTERN.test(segment) && + /\d/.test(segment) && + (best === undefined || compareSegmentPreference(segment, best) < 0) + ) { + best = segment; + } + } + return best; } -function normalizeModelIdWhitespace(value: string): string { - return value.trim().replace(/\s+/g, " "); +function hasBracketAffixMarker(value: string): boolean { + for (let index = 0; index < value.length; index++) { + const code = value.charCodeAt(index); + if (code === 91 || code === 93 || code === 0x3010 || code === 0x3011) { + return true; + } + } + return false; } /** @@ -39,18 +54,20 @@ function normalizeModelIdWhitespace(value: string): string { * upstream model id, e.g. * "[Kiro] claude-opus-4-8" -> "claude-opus-4-8" * "[gcli转] gemini-3.1-pro-preview [假流]" -> "gemini-3.1-pro-preview" + * + * Candidates are returned most-stripped first: both ends, then leading-only, then trailing-only. */ export function getBracketStrippedModelIdCandidates(modelId: string): string[] { + if (!hasBracketAffixMarker(modelId)) return []; const normalized = normalizeModelIdWhitespace(modelId); if (!normalized) return []; - const candidates = new Set(); - const withoutLeading = normalizeModelIdWhitespace(normalized.replace(LEADING_BRACKETED_AFFIX_PATTERN, "")); + const strippedLeading = normalized.replace(LEADING_BRACKETED_AFFIX_PATTERN, ""); + const withoutLeading = normalizeModelIdWhitespace(strippedLeading); const withoutTrailing = normalizeModelIdWhitespace(normalized.replace(TRAILING_BRACKETED_AFFIX_PATTERN, "")); - const withoutBoth = normalizeModelIdWhitespace( - normalized.replace(LEADING_BRACKETED_AFFIX_PATTERN, "").replace(TRAILING_BRACKETED_AFFIX_PATTERN, ""), - ); + const withoutBoth = normalizeModelIdWhitespace(strippedLeading.replace(TRAILING_BRACKETED_AFFIX_PATTERN, "")); + const candidates = new Set(); for (const candidate of [withoutBoth, withoutLeading, withoutTrailing]) { if (candidate && candidate !== normalized) { candidates.add(candidate); diff --git a/packages/coding-agent/src/modes/utils/context-usage.ts b/packages/coding-agent/src/modes/utils/context-usage.ts index fd93070a8..d223d9a5e 100644 --- a/packages/coding-agent/src/modes/utils/context-usage.ts +++ b/packages/coding-agent/src/modes/utils/context-usage.ts @@ -37,6 +37,9 @@ export interface ContextBreakdown { freeTokens: number; } +const EMPTY_STRING_PARTS: readonly string[] = []; +const EMPTY_TOOLS: ReadonlyArray> = []; + export function estimateSkillsTokens(skills: readonly Skill[]): number { const fragments: string[] = []; for (const skill of skills) { @@ -75,15 +78,16 @@ export function estimateToolSchemaTokens( * messages walked incrementally as new entries append. */ export function computeNonMessageTokens(session: AgentSession): number { - const parts = computeNonMessageBreakdown(session); - return parts.systemPromptTokens + parts.systemContextTokens + parts.toolsTokens + parts.skillsTokens; + const systemPromptParts = session.systemPrompt ?? EMPTY_STRING_PARTS; + const tools = session.agent?.state?.tools ?? EMPTY_TOOLS; + return countTokens(systemPromptParts) + estimateToolSchemaTokens(tools); } /** - * Shared helper for the four non-message token totals. Single source of truth - * for both `computeNonMessageTokens` (status-line incremental cache) and - * `computeContextBreakdown` (/context panel). The split avoids drift between - * the two surfaces — they MUST report the same numbers. + * Shared helper for the four non-message token totals used by + * `computeContextBreakdown` (/context panel). Keep this category split stable: + * the status-line fast path intentionally uses the equivalent collapsed total + * in `computeNonMessageTokens`. */ function computeNonMessageBreakdown(session: AgentSession): { skillsTokens: number; diff --git a/packages/coding-agent/test/model-id-affixes.test.ts b/packages/coding-agent/test/model-id-affixes.test.ts new file mode 100644 index 000000000..b94a9bc48 --- /dev/null +++ b/packages/coding-agent/test/model-id-affixes.test.ts @@ -0,0 +1,66 @@ +import { describe, expect, test } from "bun:test"; +import { + getBracketStrippedModelIdCandidates, + getLongestModelLikeIdSegment, + getModelLikeIdSegments, + stripBracketedModelIdAffixes, +} from "../src/config/model-id-affixes"; + +describe("getModelLikeIdSegments", () => { + test("keeps only family-prefixed segments that carry a digit, deduped", () => { + expect(getModelLikeIdSegments("openrouter/anthropic/claude-3.5-sonnet")).toEqual(["claude-3.5-sonnet"]); + // `random-text` lacks a family prefix; `claude` (no digit) is dropped. + expect(getModelLikeIdSegments("random-text claude gemini-2")).toEqual(["gemini-2"]); + }); + + test("orders longest first with lexicographic tie-break", () => { + expect(getModelLikeIdSegments("claude-3 claude-3-5-haiku claude-2")).toEqual([ + "claude-3-5-haiku", + "claude-2", + "claude-3", + ]); + }); + + test("normalizes whitespace and case before matching", () => { + expect(getModelLikeIdSegments(" GLM-4.5-Air GEMINI-2 ")).toEqual(["glm-4.5-air", "gemini-2"]); + }); + + test("returns empty for ids with no model-like segment", () => { + expect(getModelLikeIdSegments("")).toEqual([]); + expect(getModelLikeIdSegments("just some words")).toEqual([]); + }); +}); + +describe("getLongestModelLikeIdSegment", () => { + test("matches getModelLikeIdSegments[0]", () => { + const id = "[Kiro] claude-3 claude-3-5-sonnet"; + expect(getLongestModelLikeIdSegment(id)).toBe(getModelLikeIdSegments(id)[0]); + expect(getLongestModelLikeIdSegment(id)).toBe("claude-3-5-sonnet"); + }); + + test("is undefined when nothing matches", () => { + expect(getLongestModelLikeIdSegment("vendor/unknown-tag")).toBeUndefined(); + }); +}); + +describe("getBracketStrippedModelIdCandidates", () => { + test("no brackets yields no candidates", () => { + expect(getBracketStrippedModelIdCandidates("claude-opus-4-8")).toEqual([]); + }); + + test("strips leading reseller tag", () => { + expect(getBracketStrippedModelIdCandidates("[Kiro] claude-opus-4-8")).toEqual(["claude-opus-4-8"]); + }); + + test("strips both ends first, then each side, in preference order", () => { + expect(getBracketStrippedModelIdCandidates("[gcli转] gemini-3.1-pro-preview [假流]")).toEqual([ + "gemini-3.1-pro-preview", + "gemini-3.1-pro-preview [假流]", + "[gcli转] gemini-3.1-pro-preview", + ]); + }); + + test("supports full-width brackets", () => { + expect(stripBracketedModelIdAffixes("【供应商】 deepseek-v3 【限时】")).toBe("deepseek-v3"); + }); +}); diff --git a/packages/coding-agent/test/status-line-context-cache.test.ts b/packages/coding-agent/test/status-line-context-cache.test.ts index 7abacdfdd..95ff25dc5 100644 --- a/packages/coding-agent/test/status-line-context-cache.test.ts +++ b/packages/coding-agent/test/status-line-context-cache.test.ts @@ -15,8 +15,10 @@ * (messages.length shrinks) resets the cache. */ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; +import { countTokens } from "@oh-my-pi/pi-natives"; import { resetSettingsForTest, Settings } from "../src/config/settings"; import { StatusLineComponent } from "../src/modes/components/status-line"; +import { computeNonMessageTokens, estimateToolSchemaTokens } from "../src/modes/utils/context-usage"; import { initTheme } from "../src/modes/theme/theme"; import type { AgentSession } from "../src/session/agent-session"; @@ -122,6 +124,38 @@ describe("StatusLineComponent incremental context breakdown cache", () => { expect(v3.usedTokens).toBeGreaterThan(v2.usedTokens); }); + it("non-message token shortcut matches previous category sum semantics", () => { + const session = makeSession({ + messages: [], + systemPrompt: [ + "You are an assistant.\n\n\n- code: Write code\n- review: Review code\n", + "Loaded context file", + "Runtime note", + ], + tools: [ + { + name: "bash", + description: "Run shell commands", + parameters: { type: "object", properties: { command: { type: "string" } } }, + }, + ], + skills: [ + { name: "code", description: "Write code" }, + { name: "review", description: "Review code" }, + ], + }); + + const skillsTokens = countTokens(["code", "Write code", "review", "Review code"]); + const previousCategorySum = + Math.max(0, countTokens(session.systemPrompt?.[0] ?? "") - skillsTokens) + + countTokens((session.systemPrompt ?? []).slice(1)) + + estimateToolSchemaTokens(session.agent?.state?.tools ?? []) + + skillsTokens; + + expect(new StatusLineComponent(session).getCachedContextBreakdown().usedTokens).toBe(previousCategorySum); + expect(computeNonMessageTokens(session)).toBe(previousCategorySum); + }); + it("zero messages: produces only non-message tokens, no crash", () => { const session = makeSession({ messages: [] }); const comp = new StatusLineComponent(session); From 67e6324aceb28e746f326002cffd08d0023ced1d Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 20:28:15 +0000 Subject: [PATCH 050/181] fix(tui): reduced row flicker while typing Changed text row repaints to overwrite first and clear only stale suffixes so non-synchronized WSL/Windows Terminal paints do not visibly blank already-rendered rows. Kept full-line pre-clears for image protocol rows and preserved exact-width row handling.\n\nFixes #2011 --- docs/tui-core-renderer.md | 2 +- packages/tui/CHANGELOG.md | 4 ++ packages/tui/src/tui.ts | 33 +++++++++------- packages/tui/test/render-regressions.test.ts | 41 ++++++++++++++++---- 4 files changed, 59 insertions(+), 21 deletions(-) diff --git a/docs/tui-core-renderer.md b/docs/tui-core-renderer.md index dd984a830..3cc17f8b7 100644 --- a/docs/tui-core-renderer.md +++ b/docs/tui-core-renderer.md @@ -78,7 +78,7 @@ the bytes written and the state update. All state flows through a single | `sessionReplace` | clear viewport **+ ED3** (outside multiplexers) | caller forced `{ clearScrollback: true }` (switch/branch/reload/resume) | | `historyRebuild` | clear viewport **+ ED3** (outside multiplexers) | geometry change rewrapped history, or a proven-at-tail rebuild | | `overlayRebuild` | rebuild viewport with overlay composite | overlay visibility changed | -| `liveRegionPinned` | relative moves + per-line `\x1b[2K` + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go | +| `liveRegionPinned` | relative moves + per-row rewrite/suffix-clear + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go | | `viewportRepaint` | rewrite the visible viewport in place (optional `appendFrom` tail first) | safe non-destructive repaint | | `deferredShrink` | padded viewport repaint, history left dirty | bottom-anchored shrink, viewport unobservable | | `deferredMutation` | **zero bytes**, history left dirty | row-reindexing edit while possibly scrolled | diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index f5151aefa..81cfd3272 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed WSL/Windows Terminal row flicker while typing by repainting changed text rows before clearing only their stale suffix ([#2011](https://github.com/can1357/oh-my-pi/issues/2011)). + ## [15.9.69] - 2026-06-06 ### Added diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index c568d09ea..4e2e78a49 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -44,6 +44,8 @@ const SEGMENT_RESET = "\x1b[0m"; * diffing so `#previousLines` mirrors what was actually written. */ const LINE_TERMINATOR = "\x1b[0m\x1b]8;;\x07"; +const ERASE_LINE = "\x1b[2K"; +const ERASE_TO_END_OF_LINE = "\x1b[K"; // Hide the hardware cursor before each paint/move write. Ghostty-style bar // cursors can otherwise leave visual afterimages while the TUI repaints the // row under a visible cursor. Paint writes also disable terminal autowrap: @@ -2276,8 +2278,8 @@ export class TUI extends Container { // Multiplexers (tmux/screen/zellij) cannot erase pane history with `\x1b[3J` // and cannot answer a viewport-position probe, so the destructive checkpoint // rebuild path is forever unavailable. The pinned emitter is built from the - // opposite primitives — relative cursor moves, per-line `\x1b[2K`, and - // `\r\n` to scroll sealed rows past the viewport bottom — which are exactly + // opposite primitives — relative cursor moves, per-row rewrite/suffix-clear, + // and `\r\n` to scroll sealed rows past the viewport bottom — which are exactly // what tmux pane history accepts. Without this commit-as-you-go path, the // streaming cap below clipped every frame to the visible tail and the // scrolled-off head was committed nowhere (issue #1974). @@ -2349,6 +2351,12 @@ export class TUI extends Container { return truncated + (truncated.includes("\x1b]8;") ? LINE_TERMINATOR : SEGMENT_RESET); } + #lineRewriteSequence(line: string, width: number): string { + const fitted = this.#fitLineToWidth(line, width); + if (TERMINAL.isImageLine(fitted)) return ERASE_LINE + fitted; + return visibleWidth(fitted) >= width ? fitted : fitted + ERASE_TO_END_OF_LINE; + } + /** * Single state-transition point. Every emitter calls this exactly once at * the end so cursor/viewport/scrollback accounting stays consistent. @@ -2515,8 +2523,7 @@ export class TUI extends Container { let buffer = `${this.#paintBeginSequence}\x1b[H`; for (let screenRow = 0; screenRow < height; screenRow++) { if (screenRow > 0) buffer += "\r\n"; - buffer += "\x1b[2K"; - buffer += texts[screenRow]; + buffer += this.#lineRewriteSequence(texts[screenRow], width); } // DECCARA rectangles paint the visible fills before cursor positioning; // the cleared cells written above are what the rectangles repaint. @@ -2554,8 +2561,8 @@ export class TUI extends Container { * leaving the transient live region out of saved lines. * * Uses only the no-scroll-snap vocabulary of {@link #emitDiff}: relative - * cursor moves, per-line `\x1b[2K`, and `\r\n` to push the sealed chunk into - * history. It deliberately avoids a full-screen erase (`\x1b[2J`) and absolute + * cursor moves, per-row rewrite/suffix-clear, and `\r\n` to push the sealed + * chunk into history. It deliberately avoids a full-screen erase (`\x1b[2J`) and absolute * cursor home (`\x1b[H`): on Ghostty those snap a reader scrolled into history * back to the bottom on every frame. */ @@ -2587,17 +2594,18 @@ export class TUI extends Container { // Write the sealed chunk followed by the full viewport from the top row. // The first (boundedAppendTo - boundedAppendFrom) rows scroll into native - // history; the trailing `height` rows fill the viewport. Each row clears - // itself with `\x1b[2K` instead of relying on a screen-wide erase. + // history; the trailing `height` rows fill the viewport. Text rows overwrite + // first and clear only the suffix so non-synchronized hosts do not visibly + // blank stable content before repainting it. let wroteLine = false; for (let i = boundedAppendFrom; i < boundedAppendTo; i++) { if (wroteLine) buffer += "\r\n"; - buffer += `\x1b[2K${this.#fitLineToWidth(lines[i] ?? "", width)}`; + buffer += this.#lineRewriteSequence(lines[i] ?? "", width); wroteLine = true; } for (let screenRow = 0; screenRow < height; screenRow++) { if (wroteLine) buffer += "\r\n"; - buffer += `\x1b[2K${this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width)}`; + buffer += this.#lineRewriteSequence(lines[viewportTop + screenRow] ?? "", width); wroteLine = true; } @@ -2676,7 +2684,7 @@ export class TUI extends Container { const currentScreenRow = Math.max(0, Math.min(height - 1, clampedCursor - prevViewportTop)); const moveDown = height - 1 - currentScreenRow; if (moveDown > 0) buffer += `\x1b[${moveDown}B`; - buffer += `\r\x1b[2K${this.#fitLineToWidth(line, width)}\x1b[?25l`; + buffer += `\r${this.#lineRewriteSequence(line, width)}\x1b[?25l`; buffer += this.#paintEndSequence; this.terminal.write(buffer); @@ -2835,8 +2843,7 @@ export class TUI extends Container { } for (let i = firstChanged; i <= renderEnd; i++) { if (i > firstChanged) buffer += "\r\n"; - buffer += "\x1b[2K"; - buffer += fillTexts && i >= fillStart ? fillTexts[i - fillStart] : this.#fitLineToWidth(lines[i], width); + buffer += this.#lineRewriteSequence(fillTexts && i >= fillStart ? fillTexts[i - fillStart] : lines[i], width); } // If the prior frame was taller, clear the trailing rows. diff --git a/packages/tui/test/render-regressions.test.ts b/packages/tui/test/render-regressions.test.ts index 1693cc817..bdd77ef2c 100644 --- a/packages/tui/test/render-regressions.test.ts +++ b/packages/tui/test/render-regressions.test.ts @@ -247,6 +247,33 @@ describe("TUI terminal-state regressions", () => { tui.stop(); } }); + it("rewrites changed rows before clearing suffixes for non-synchronized hosts", async () => { + const term = new VirtualTerminal(40, 8); + const tui = new TUI(term); + const component = new MutableLinesComponent([ + "assistant output already rendered", + "tool output already rendered", + "todos/status already rendered", + ]); + tui.addChild(component); + + try { + tui.start(); + await settle(term); + const writes = captureWrites(term); + + component.setLines(["assistant output already rendered", "tool", "todos/status already rendered"]); + tui.requestRender(); + await settle(term); + + const paint = writes.at(-1) ?? ""; + expect(paint).toContain("tool\x1b[0m\x1b[K"); + expect(paint).not.toContain("\x1b[2Ktool"); + expect(visible(term)[1]).toBe("tool"); + } finally { + tui.stop(); + } + }); it("clears removed tail lines after shrink", async () => { const term = new VirtualTerminal(40, 10); @@ -2142,7 +2169,7 @@ describe("TUI terminal-state regressions", () => { expect(viewport.at(-1)).toBe("spinner-b"); expect(term.getScrollBuffer().join("\n")).not.toContain("edited-0"); const paint = writes.at(-1) ?? ""; - expect(paint).toContain("\r\x1b[2Kspinner-b"); + expect(paint).toContain("\rspinner-b\x1b[0m\x1b[K"); expect(paint).not.toContain("\x1b[H"); expect(paint).not.toContain("\x1b[3J"); } finally { @@ -2170,7 +2197,7 @@ describe("TUI terminal-state regressions", () => { expect(visible(scrolledTerm).map(line => line.trim())).toEqual(beforeViewport); expect(scrolledTerm.getScrollBuffer().join("\n")).not.toContain("edited-0"); const paint = writes.at(-1) ?? ""; - expect(paint).toContain("\r\x1b[2Kspinner-b"); + expect(paint).toContain("\rspinner-b\x1b[0m\x1b[K"); expect(paint).not.toContain("\x1b[H"); expect(paint).not.toContain("\x1b[3J"); } finally { @@ -3426,8 +3453,8 @@ describe("TUI terminal-state regressions", () => { // Initial paint: only the styled row carries background cells. expect(backgroundRows(term, height)).toEqual([1]); - // Diff path: rewriting the row below starts with \x1b[2K — with leaked - // background, BCE would paint that whole row red. + // Diff path: rewriting the row below clears only after the row reset; + // with leaked background, BCE would otherwise paint that row red. component.setLines(["plain-0", UNRESET_BG_ROW, "EDITED-2"]); tui.requestRender(); await settle(term); @@ -3459,8 +3486,8 @@ describe("TUI terminal-state regressions", () => { expect(foregroundRows(term, height)).toEqual([1]); expect(underlineRows(term, height)).toEqual([1]); - // Rewriting the next row starts with an erase; leaked SGR would make - // the edited row green/underlined despite containing plain text. + // Rewriting the next row clears only after the row reset; leaked SGR + // would make the edited row green/underlined despite containing plain text. component.setLines(["plain-0", UNRESET_FG_UNDERLINE_ROW, "EDITED-2"]); tui.requestRender(); await settle(term); @@ -3495,7 +3522,7 @@ describe("TUI terminal-state regressions", () => { tui.start(); await settle(term); - // Force a full repaint (viewport rewrite path emits \x1b[2K per row). + // Force a full repaint (viewport rewrite path suffix-clears each text row). tui.requestRender(true); await settle(term); From 4bf9a92b28adf554e48c9efd2d9a74bafffa6efa Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:48:31 +0200 Subject: [PATCH 051/181] feat(utils): added module-load timing preload and DAG report - Added Bun preload that records inclusive per-module windows and resolved static import edges via plugin hooks. - Shared events through a dependency-free buffer so the preload and logger avoid importing each other. - Rendered module spans as a body/TLA-ranked dependency tree, separating graph wait from top-level work. - Back-folded captured load phase into the root window to shrink the opaque pre-instrumentation figure. --- packages/coding-agent/scripts/dev-launch | 4 + packages/utils/CHANGELOG.md | 2 +- packages/utils/src/logger.ts | 200 ++++++++++++++++++++--- packages/utils/src/module-timer.ts | 148 +++++++++++++++++ packages/utils/src/timing-buffer.ts | 47 ++++++ 5 files changed, 377 insertions(+), 24 deletions(-) create mode 100644 packages/utils/src/module-timer.ts create mode 100644 packages/utils/src/timing-buffer.ts diff --git a/packages/coding-agent/scripts/dev-launch b/packages/coding-agent/scripts/dev-launch index 519ccff0f..c187c6161 100755 --- a/packages/coding-agent/scripts/dev-launch +++ b/packages/coding-agent/scripts/dev-launch @@ -28,6 +28,7 @@ done scripts_dir=$(CDPATH= cd -- "$(dirname -- "$self")" && pwd -P) cli=$scripts_dir/../src/cli.ts preload=$scripts_dir/dev-launch-preload.ts +timing_preload=$scripts_dir/../../utils/src/module-timer.ts launch_dir=${OMP_DEV_LAUNCH_DIR:-${HOME}/.omp/.dev-cwd} mkdir -p "$launch_dir" @@ -35,4 +36,7 @@ mkdir -p "$launch_dir" OMP_LAUNCH_CWD=$PWD export OMP_LAUNCH_CWD cd "$launch_dir" +if [ -n "${PI_TIMING:-}" ]; then + exec bun --preload "$preload" --preload "$timing_preload" "$cli" "$@" +fi exec bun --preload "$preload" "$cli" "$@" diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 95442dc2a..ae4d73f15 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -4,7 +4,7 @@ ### Changed -- `logger.printTimings()` (the `PI_TIMING` startup tree) now surfaces two previously-invisible regions: a `(before instrumentation)` line for the runtime init + static module-graph load that elapses before the first marker (the dominant real-world startup cost, ~350ms — `startTiming()` only begins inside `runRootCommand`), and an `(unattributed self)` line for the root span's own untimed work so the gap between the visible top-level spans and `Total` is no longer silently swallowed. `Total` is now labelled `(since first marker)` to make the window explicit. +- `logger.printTimings()` (the `PI_TIMING` startup tree) now surfaces two previously-invisible regions: a `(before instrumentation)` line for runtime init / uncaptured pre-marker work, and an `(unattributed self)` line for the root span's own untimed work so the gap between visible top-level spans and `Total` is no longer swallowed. `Total` is now labelled `(since first marker)` to make the window explicit. The restored `module-timer.ts` preload can feed module spans into the report: each module records `onLoad` → final top-level marker as `total`, a prepended body marker → final marker as `body/TLA`, and resolved static imports as a bounded dependency tree so the report separates graph wait from actual top-level module work. ## [15.9.2] - 2026-06-05 diff --git a/packages/utils/src/logger.ts b/packages/utils/src/logger.ts index c682bbf8f..1591e1620 100644 --- a/packages/utils/src/logger.ts +++ b/packages/utils/src/logger.ts @@ -15,6 +15,7 @@ import { isPromise } from "node:util/types"; import winston from "winston"; import DailyRotateFile from "winston-daily-rotate-file"; import { getLogsDir } from "./dirs"; +import { drainModuleLoadEvents } from "./timing-buffer"; /** Ensure a logs directory exists; return the resolved path. */ function ensureDir(dir: string): string { @@ -166,12 +167,36 @@ interface Span { children: Span[]; /** Marker / point event without a duration. */ point?: boolean; + /** Absolute module path for module-load spans. */ + modulePath?: string; + /** Own top-level module body / TLA duration for module-load spans. */ + moduleBodyMs?: number; + /** Resolved static imports for module-load spans. */ + moduleImports?: string[]; } - const spanStorage = new AsyncLocalStorage(); let gRootSpan: Span | undefined; let gRecordTimings = false; +export function timingModeIncludes(option: "full" | "x"): boolean { + const value = process.env.PI_TIMING; + if (!value) return false; + if (value === option) return true; + let start = 0; + for (let i = 0; i <= value.length; i++) { + const code = i === value.length ? 44 : value.charCodeAt(i); + const separator = code === 44 || code === 58 || code === 59 || code === 43 || code <= 32; + if (!separator) continue; + if (i > start && value.slice(start, i) === option) return true; + start = i + 1; + } + return false; +} + +export function shouldExitAfterTimings(): boolean { + return timingModeIncludes("x") || timingModeIncludes("full"); +} + /** * Print collected timings as an indented tree. * Each span shows wall duration; parents with children also show "(self)" for unattributed time. @@ -184,13 +209,19 @@ export function printTimings(): void { } gRootSpan.end = performance.now(); + // Splice any preload-captured module-load events into the tree as root + // children and back-extend the root window over them, so the static-import + // phase that ran before the first explicit marker becomes visible (the + // `(modules)` summary below) instead of being lumped into the opaque + // `(before instrumentation)` figure. + spliceModuleLoadBuffer(); const lines: string[] = []; lines.push(""); lines.push("--- Startup timings (hierarchical) ---"); // performance.now() shares the process-start origin, so the root span's start - // is the wall time spent before the first marker — runtime init plus the - // static module-graph evaluation (~the dominant cost). It is otherwise - // invisible because Total only spans startTiming()→printTimings(). + // is the wall time before the first marker — runtime init plus any module + // loads not captured below. With the module-load preload active this shrinks + // to ~runtime init because the load phase is back-folded into the window. if (gRootSpan.start > LOGGED_TIMING_THRESHOLD_MS) { lines.push(`(before instrumentation): ${fmtMs(gRootSpan.start)} [runtime init + module load]`); } @@ -222,8 +253,8 @@ export function printTimings(): void { /** * Begin recording startup timings under a new root span. - * Idempotent: a second call while already recording is a no-op so that side-effect - * starters (see module-timer.ts) and explicit starters (main.ts) can coexist. + * Idempotent: a second call while already recording is a no-op, so an explicit + * starter (main.ts) and any future early starter can coexist. */ export function startTiming(): void { if (gRecordTimings) return; @@ -238,10 +269,16 @@ export function startTiming(): void { /** * Record an externally-measured span as a leaf child of the active span (or root - * when no span is active). Used by the module-load timing plugin to splice load - * events into the tree retroactively. + * when no span is active). Used by {@link spliceModuleLoadBuffer} to fold + * preload-captured module windows into the tree. */ -export function recordModuleLoadSpan(path: string, start: number, durationMs: number): void { +export function recordModuleLoadSpan( + path: string, + start: number, + durationMs: number, + bodyMs?: number, + imports: string[] = [], +): void { if (!gRecordTimings || !gRootSpan) return; const parent = spanStorage.getStore() ?? gRootSpan; const span: Span = { @@ -250,10 +287,32 @@ export function recordModuleLoadSpan(path: string, start: number, durationMs: nu end: start + durationMs, parent, children: [], + modulePath: path, + moduleBodyMs: bodyMs, + moduleImports: imports, }; parent.children.push(span); } +/** + * Drain the preload's module-load buffer (see module-timer.ts) into the tree as + * `load:` children of the root, then back-extend the root window to the earliest + * captured read so the pre-marker load phase is counted in Total rather than + * hidden as `(before instrumentation)`. No-op when nothing was captured (e.g. no + * `--preload`, or a compiled binary where module reads are not interceptable). + */ +function spliceModuleLoadBuffer(): void { + if (!gRootSpan) return; + const events = drainModuleLoadEvents(); + if (events.length === 0) return; + let earliest = gRootSpan.start; + for (const event of events) { + recordModuleLoadSpan(event.path, event.start, event.durationMs, event.bodyMs, event.imports); + if (event.start < earliest) earliest = event.start; + } + gRootSpan.start = earliest; +} + function shortenLoadPath(p: string): string { const cwd = process.cwd(); if (p.startsWith(`${cwd}/`)) return p.slice(cwd.length + 1); @@ -309,6 +368,16 @@ function fmtMs(ms: number): string { const MODULE_LOAD_PREFIX = "load:"; const MODULE_LOAD_VERBOSE_TOP = 10; +const MODULE_TREE_MAX_DEPTH = 5; +const MODULE_TREE_ROOT_TOP = 5; +const MODULE_TREE_CHILD_TOP = 8; + +interface ModuleTimingNode { + span: Span; + children: ModuleTimingNode[]; + parents: number; + body: number; +} function isModuleLoadSpan(span: Span): boolean { return span.op.startsWith(MODULE_LOAD_PREFIX); @@ -343,33 +412,118 @@ function printSpan(span: Span, depth: number, lines: string[]): void { } } -/** Collapse the (typically hundreds of) module-load spans into one summary line. */ +/** Render module-load spans as a dependency-aware DAG/tree. */ function printModuleLoadSummary(loads: Span[], depth: number, lines: string[]): void { const childIndent = " ".repeat(depth); const grandIndent = " ".repeat(depth + 1); let unionStart = Number.POSITIVE_INFINITY; let unionEnd = 0; - let totalSelf = 0; for (const span of loads) { if (span.end === undefined) continue; if (span.start < unionStart) unionStart = span.start; if (span.end > unionEnd) unionEnd = span.end; - totalSelf += span.end - span.start; } const wall = unionEnd > unionStart ? unionEnd - unionStart : 0; - lines.push(`${childIndent}(modules): ${loads.length} loaded, wall ${fmtMs(wall)}, sum ${fmtMs(totalSelf)}`); - const showAll = process.env.PI_TIMING === "full"; - const sorted = [...loads].sort((a, b) => durationOf(b) - durationOf(a)); - const visible = showAll ? sorted : sorted.slice(0, MODULE_LOAD_VERBOSE_TOP); - for (const span of visible) { - const dur = durationOf(span); - if (dur < LOGGED_TIMING_THRESHOLD_MS) break; - const tag = isParallel(span) ? " [parallel]" : ""; - lines.push(`${grandIndent}${span.op}: ${fmtMs(dur)}${tag}`); + const nodes = buildModuleTimingGraph(loads); + lines.push(`${childIndent}(modules): ${loads.length} loaded, wall ${fmtMs(wall)}`); + if (nodes.length === 0) return; + + const showAll = timingModeIncludes("full"); + const byBody = [...nodes].sort(compareModuleNodes); + const topBody = showAll ? byBody : byBody.slice(0, MODULE_LOAD_VERBOSE_TOP); + lines.push(`${grandIndent}top body/TLA:`); + for (const node of topBody) { + if (!showAll && node.body < LOGGED_TIMING_THRESHOLD_MS) break; + lines.push(`${grandIndent} ${node.span.op}: body ${fmtMs(node.body)} (total ${fmtMs(durationOf(node.span))})`); } - if (!showAll && sorted.length > MODULE_LOAD_VERBOSE_TOP) { - lines.push(`${grandIndent}… ${sorted.length - MODULE_LOAD_VERBOSE_TOP} more (PI_TIMING=full to show all)`); + if (!showAll && byBody.length > MODULE_LOAD_VERBOSE_TOP) { + lines.push(`${grandIndent} … ${byBody.length - MODULE_LOAD_VERBOSE_TOP} more (PI_TIMING=full to show all)`); } + + const roots = nodes.filter(node => node.parents === 0); + const treeRoots = (roots.length > 0 ? roots : nodes).sort((a, b) => durationOf(b.span) - durationOf(a.span)); + const visibleRoots = showAll ? treeRoots : treeRoots.slice(0, MODULE_TREE_ROOT_TOP); + lines.push(`${grandIndent}tree:`); + const rendered = new Set(); + for (const node of visibleRoots) { + renderModuleTimingNode(node, depth + 2, lines, rendered, new Set(), showAll); + } + if (!showAll && treeRoots.length > MODULE_TREE_ROOT_TOP) { + lines.push( + `${grandIndent} … ${treeRoots.length - MODULE_TREE_ROOT_TOP} more roots (PI_TIMING=full to show all)`, + ); + } +} + +function buildModuleTimingGraph(loads: Span[]): ModuleTimingNode[] { + const nodes = new Map(); + for (const span of loads) { + if (!span.modulePath || span.end === undefined) continue; + nodes.set(span.modulePath, { span, children: [], parents: 0, body: span.moduleBodyMs ?? 0 }); + } + for (const node of nodes.values()) { + for (const childPath of node.span.moduleImports ?? []) { + const child = nodes.get(childPath); + if (!child || child === node) continue; + node.children.push(child); + child.parents++; + } + } + for (const node of nodes.values()) { + node.children.sort(compareModuleNodes); + } + return [...nodes.values()]; +} + +function compareModuleNodes(a: ModuleTimingNode, b: ModuleTimingNode): number { + const bodyDiff = b.body - a.body; + if (Math.abs(bodyDiff) > 0.001) return bodyDiff; + return durationOf(b.span) - durationOf(a.span); +} + +function renderModuleTimingNode( + node: ModuleTimingNode, + depth: number, + lines: string[], + rendered: Set, + ancestors: Set, + showAll: boolean, +): void { + const path = node.span.modulePath; + if (!path) return; + const indent = " ".repeat(depth); + const total = durationOf(node.span); + if (!showAll && total < LOGGED_TIMING_THRESHOLD_MS && node.children.length === 0) return; + const wait = Math.max(0, total - node.body); + const shared = node.parents > 1 ? " [shared]" : ""; + const timing = + node.body > LOGGED_TIMING_THRESHOLD_MS || node.children.length > 0 + ? ` (body ${fmtMs(node.body)}, wait ${fmtMs(wait)})` + : ""; + const alreadyRendered = rendered.has(path); + const cycle = ancestors.has(path); + const suffix = cycle ? " [cycle]" : alreadyRendered ? " [already shown]" : ""; + lines.push(`${indent}${node.span.op}: ${fmtMs(total)}${timing}${shared}${suffix}`); + if (cycle || alreadyRendered) return; + rendered.add(path); + ancestors.add(path); + if (!showAll && ancestors.size >= MODULE_TREE_MAX_DEPTH) { + if (node.children.length > 0) { + lines.push(`${indent} … ${node.children.length} imports deeper (PI_TIMING=full to show all)`); + } + ancestors.delete(path); + return; + } + const visibleChildren = showAll ? node.children : node.children.slice(0, MODULE_TREE_CHILD_TOP); + for (const child of visibleChildren) { + renderModuleTimingNode(child, depth + 1, lines, rendered, ancestors, showAll); + } + if (!showAll && node.children.length > MODULE_TREE_CHILD_TOP) { + lines.push( + `${indent} … ${node.children.length - MODULE_TREE_CHILD_TOP} more imports (PI_TIMING=full to show all)`, + ); + } + ancestors.delete(path); } /** A span is parallel if it overlaps a sibling that started before it. */ diff --git a/packages/utils/src/module-timer.ts b/packages/utils/src/module-timer.ts new file mode 100644 index 000000000..b0ed91d39 --- /dev/null +++ b/packages/utils/src/module-timer.ts @@ -0,0 +1,148 @@ +/** + * Module-load timing preload. + * + * `bun --preload .../module-timer.ts ` installs Bun plugin hooks (only + * when `PI_TIMING` is set) that record an inclusive module window plus resolved + * static child edges: + * + * onLoad start → appended end marker after the module's top-level body + * + * Events are pushed into a process-global buffer that {@link logger.printTimings} + * drains and renders as a module DAG/tree. Each module row can therefore show + * both total time and `self` time after subtracting child module intervals. + * + * Why a preload (and not a normal import): Bun reads the *entire* statically + * reachable graph before evaluating any module, so hooks installed from inside + * that graph cannot observe its own loading — they only catch later dynamically + * loaded modules. A preload runs first, so it sees the static-import phase that + * dominates startup. + * + * Kept dependency-free on purpose: the sole import is Bun's `plugin`, so this is + * cheap to preload before pi-utils (and winston) exist. The buffer is shared with + * the logger via a registry Symbol so neither side needs to import the other. + * + * **What is measured:** an inclusive per-module window. `onLoad` stamps the + * start before reading source; the returned source has a tiny marker appended at + * the end of the module. That marker runs after Bun parses/transpiles the module + * and after any top-level await in that module completes, so the duration + * includes read + parse/transpile + dependency wait + top-level execution/TLA. + * If a module throws before its final statement, no end marker is recorded. + * + * **Tree shape:** `onResolve` observes importer → specifier edges and resolves + * them with `Bun.resolveSync` without taking over Bun's real resolution. The + * logger renders these edges as a DAG/tree and computes module `self` time by + * subtracting the union of child intervals, avoiding misleading flat inclusive + * totals. + * + * **Coverage limits:** + * - TS/TSX only — intercepting `node_modules` CJS `.js`/`.cjs` and forcing ESM + * breaks their default-export detection, so they are left to Bun's default path. + * - **Dev runs only.** In the compiled `omp` binary every module is pre-bundled + * into bunfs, so `onLoad` never fires; profile with a `bun --preload` dev run. + */ +import { plugin } from "bun"; +import { moduleLoadBuffer } from "./timing-buffer"; + +// Restrict to TS/TSX only. node_modules ships CommonJS `.js`/`.cjs` that Bun +// auto-detects when loaded via its default path; if we intercept and return +// `{ contents, loader: "js" }`, Bun forces ESM and CJS modules fail to load +// (e.g. `Missing 'default' export`). Our own source tree (where the interesting +// timing lives) is uniformly TypeScript, so a TS-only filter is both safe and +// sufficient. +const MODULE_LOADER_FILTER = /\.[mc]?tsx?$/; +const MODULE_COMPLETE_KEY: symbol = Symbol.for("omp.moduleLoadComplete"); +const MODULE_BODY_START_KEY: symbol = Symbol.for("omp.moduleBodyStart"); +const STATIC_IMPORT_PATTERN = + /\b(?:import|export)\s+(?:type\s+)?(?:[^"']*?\s+from\s+)?["']([^"']+)["']|\bimport\s*\(\s*["']([^"']+)["']\s*\)/g; + +type CompleteStore = Record void) | undefined>; + +function bodyStartMarker(path: string): string { + return `;globalThis[Symbol.for("omp.moduleBodyStart")]?.(${JSON.stringify(path)});\n`; +} + +function completionMarker(path: string): string { + return `\n;globalThis[Symbol.for("omp.moduleLoadComplete")]?.(${JSON.stringify(path)});\n`; +} + +function instrumentContents(path: string, contents: string): string { + const start = bodyStartMarker(path); + const end = completionMarker(path); + if (!contents.startsWith("#!")) return `${start}${contents}${end}`; + const newline = contents.indexOf("\n"); + if (newline === -1) return `${contents}\n${start}${end}`; + return `${contents.slice(0, newline + 1)}${start}${contents.slice(newline + 1)}${end}`; +} +function importerDir(importer: string): string { + const slash = importer.lastIndexOf("/"); + if (slash === -1) return "."; + return importer.slice(0, slash); +} + +function childSetFor(importsByPath: Map>, path: string): Set { + let children = importsByPath.get(path); + if (!children) { + children = new Set(); + importsByPath.set(path, children); + } + return children; +} + +function addImportEdges(importsByPath: Map>, importer: string, contents: string): void { + STATIC_IMPORT_PATTERN.lastIndex = 0; + for (const match of contents.matchAll(STATIC_IMPORT_PATTERN)) { + const specifier = match[1] ?? match[2]; + if (!specifier) continue; + try { + const resolved = Bun.resolveSync(specifier, importerDir(importer)); + if (MODULE_LOADER_FILTER.test(resolved) && resolved !== importer) { + childSetFor(importsByPath, importer).add(resolved); + } + } catch { + // Leave Bun's real resolver/runtime to surface any error. This scanner is only an observer. + } + } +} + +if (process.env.PI_TIMING) { + const buffer = moduleLoadBuffer(); + const starts = new Map(); + const bodyStarts = new Map(); + const importsByPath = new Map>(); + const store = globalThis as unknown as CompleteStore; + store[MODULE_BODY_START_KEY] = (path: string): void => { + bodyStarts.set(path, performance.now()); + }; + store[MODULE_COMPLETE_KEY] = (path: string): void => { + const start = starts.get(path); + if (start === undefined) return; + starts.delete(path); + const end = performance.now(); + const bodyStart = bodyStarts.get(path); + bodyStarts.delete(path); + const imports = importsByPath.get(path); + buffer.push({ + path, + start, + durationMs: end - start, + bodyMs: bodyStart === undefined ? undefined : end - bodyStart, + imports: imports ? [...imports] : [], + }); + }; + + plugin({ + name: "pi-module-load-timer", + setup(build) { + build.onLoad({ filter: MODULE_LOADER_FILTER }, async args => { + starts.set(args.path, performance.now()); + childSetFor(importsByPath, args.path); + const contents = await Bun.file(args.path).text(); + addImportEdges(importsByPath, args.path, contents); + return { + contents: instrumentContents(args.path, contents), + loader: args.path.endsWith(".tsx") ? "tsx" : "ts", + }; + }); + }, + }); +} diff --git a/packages/utils/src/timing-buffer.ts b/packages/utils/src/timing-buffer.ts new file mode 100644 index 000000000..4208f9298 --- /dev/null +++ b/packages/utils/src/timing-buffer.ts @@ -0,0 +1,47 @@ +/** + * Shared contract between the {@link module-timer} preload and {@link logger}'s + * timing tree. Kept in its own dependency-free module so the preload can import + * it without pulling in winston (via logger) and the logger can drain the buffer + * without importing the Bun-plugin preload. + */ + +export interface ModuleLoadEvent { + /** Absolute or Bun-resolved module path. */ + path: string; + /** `performance.now()` timestamp captured at Bun `onLoad` entry. */ + start: number; + /** Inclusive module window: `onLoad` entry → appended final marker. */ + durationMs: number; + /** Own top-level body / TLA time: prepended body marker → appended final marker. */ + bodyMs?: number; + /** Resolved static children imported by this module. */ + imports: string[]; +} + +/** + * Registry-global key under which the preload accumulates module-load events. + * `Symbol.for` so both modules resolve the same symbol independently. + */ +const KEY: symbol = Symbol.for("omp.moduleLoadBuffer"); + +type Store = Record; + +/** The append-only buffer the preload pushes into (created on first access). */ +export function moduleLoadBuffer(): ModuleLoadEvent[] { + const store = globalThis as unknown as Store; + let buffer = store[KEY]; + if (!buffer) { + buffer = []; + store[KEY] = buffer; + } + return buffer; +} + +/** Drain and return all buffered events, leaving the buffer empty. */ +export function drainModuleLoadEvents(): ModuleLoadEvent[] { + const store = globalThis as unknown as Store; + const buffer = store[KEY]; + if (!buffer || buffer.length === 0) return []; + store[KEY] = []; + return buffer; +} From ed55880f37842c45a8c283aecc543705622d7e6a Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:49:12 +0200 Subject: [PATCH 052/181] fix(ai): fixed orphaned tool-call handling for responses providers - Added `repairOrphanResponsesToolCalls` to append placeholder outputs for orphan calls. - Updated `openai-codex/request-transformer.ts` to repair unmatched tool-call/input pairs. - Wrapped `openai-responses.ts` and `azure-openai-responses.ts` with orphan call repair before request conversion. - Added regression coverage for orphan call repair in Codex and Responses tests. --- packages/ai/CHANGELOG.md | 1 + .../src/providers/azure-openai-responses.ts | 3 +- .../openai-codex/request-transformer.ts | 110 +++++++++++++----- .../src/providers/openai-responses-shared.ts | 53 +++++++++ packages/ai/src/providers/openai-responses.ts | 3 +- packages/ai/test/openai-codex.test.ts | 56 +++++++++ .../openai-responses-orphan-repair.test.ts | 69 +++++++++++ 7 files changed, 261 insertions(+), 34 deletions(-) create mode 100644 packages/ai/test/openai-responses-orphan-repair.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 6088cf64f..e3cd50803 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -9,6 +9,7 @@ - Fixed usage-report dedup ignoring `projectId` for Google Cloud providers, preventing duplicate credential entries from being recognized as the same account. - Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#2002](https://github.com/can1357/oh-my-pi/pull/2002)) +- Fixed OpenAI Responses-family providers (Codex, OpenAI Responses, Azure Responses) rejecting requests with `400 No tool output found for function call …` after the user branched/navigated the session tree to a node that ends on a tool call (the tool-result child is dropped from the reconstructed history) or after a turn was aborted/crashed between the call streaming and its result persisting. The converters now synthesize a placeholder `function_call_output`/`custom_tool_call_output` immediately after any unpaired `function_call`/`custom_tool_call`, symmetric to the existing orphan-output repair, so the model still sees the call and can recover instead of the whole request 400ing. ### Fixed - Fixed Anthropic-compatible reasoning endpoints losing prior-turn reasoning on continuation requests when they emit unsigned `thinking` blocks. `convertAnthropicMessages` treated unknown endpoints as signature-enforcing and demoted unsigned reasoning to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool. Official `api.anthropic.com` keeps the conservative text fallback; non-official `anthropic-messages` reasoning models now replay unsigned reasoning as native `type: "thinking"` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index 04027d02a..26b3f0a16 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -40,6 +40,7 @@ import { isOpenAIResponsesProgressEvent, normalizeResponsesToolCallIdForTransform, processResponsesStream, + repairOrphanResponsesToolCalls, } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; @@ -347,7 +348,7 @@ function convertMessages( msgIndex++; } - return messages; + return repairOrphanResponsesToolCalls(messages); } function convertTools(tools: Tool[]): OpenAITool[] { diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index f271bbc81..a12996ca6 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -76,6 +76,83 @@ function filterInput(input: InputItem[] | undefined): InputItem[] | undefined { }); } +const CODEX_ORPHAN_OUTPUT_LIMIT = 16_000; +/** Placeholder output for a tool call whose result never landed in the input. */ +const CODEX_INTERRUPTED_TOOL_OUTPUT = + "[No tool output recorded: the tool call was interrupted before it produced a result.]"; + +function orphanFunctionOutputToMessage(item: InputItem, callId: string): InputItem { + const itemRecord = item as unknown as Record; + const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool"; + let text = ""; + try { + const output = itemRecord.output; + text = typeof output === "string" ? output : JSON.stringify(output); + } catch { + text = String(itemRecord.output ?? ""); + } + if (text.length > CODEX_ORPHAN_OUTPUT_LIMIT) { + text = `${text.slice(0, CODEX_ORPHAN_OUTPUT_LIMIT)}\n...[truncated]`; + } + return { + type: "message", + role: "assistant", + content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`, + } as InputItem; +} + +/** + * Repair both halves of unpaired tool exchanges so the Responses input grammar + * stays valid — the API rejects either orphan with a 400: + * + * - `function_call_output` with no matching `function_call` → folded into an + * assistant message (`400 No tool call found for function call output …`). + * Regression of #472 / #1351. + * - `function_call` / `custom_tool_call` with no matching `*_output` → a + * placeholder output is synthesized immediately after the call + * (`400 No tool output found for function call …`). Hit when the user + * branches/navigates the session tree to a node that ends on a tool call (the + * tool-result child is dropped from the reconstructed history) or when a turn + * is aborted/crashes after the call streamed but before its result persisted. + */ +function repairToolCallPairs(input: InputItem[]): InputItem[] { + const callIds = new Set(); + const outputCallIds = new Set(); + for (const item of input) { + const callId = typeof item.call_id === "string" ? item.call_id : undefined; + if (callId === undefined) continue; + if (item.type === "function_call" || item.type === "custom_tool_call") callIds.add(callId); + else if (item.type === "function_call_output" || item.type === "custom_tool_call_output") { + outputCallIds.add(callId); + } + } + + const repaired: InputItem[] = []; + for (const item of input) { + const callId = typeof item.call_id === "string" ? item.call_id : undefined; + + if (item.type === "function_call_output" && callId !== undefined && !callIds.has(callId)) { + repaired.push(orphanFunctionOutputToMessage(item, callId)); + continue; + } + + repaired.push(item); + + if ( + (item.type === "function_call" || item.type === "custom_tool_call") && + callId !== undefined && + !outputCallIds.has(callId) + ) { + repaired.push({ + type: item.type === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output", + call_id: callId, + output: CODEX_INTERRUPTED_TOOL_OUTPUT, + } as InputItem); + } + } + return repaired; +} + export async function transformRequestBody( body: RequestBody, model: Model, @@ -87,39 +164,8 @@ export async function transformRequestBody( if (body.input && Array.isArray(body.input)) { body.input = filterInput(body.input); - if (body.input) { - const functionCallIds = new Set( - body.input - .filter(item => item.type === "function_call" && typeof item.call_id === "string") - .map(item => item.call_id as string), - ); - - body.input = body.input.map(item => { - if (item.type === "function_call_output" && typeof item.call_id === "string") { - const callId = item.call_id as string; - if (!functionCallIds.has(callId)) { - const itemRecord = item as unknown as Record; - const toolName = typeof itemRecord.name === "string" ? itemRecord.name : "tool"; - let text = ""; - try { - const output = itemRecord.output; - text = typeof output === "string" ? output : JSON.stringify(output); - } catch { - text = String(itemRecord.output ?? ""); - } - if (text.length > 16000) { - text = `${text.slice(0, 16000)}\n...[truncated]`; - } - return { - type: "message", - role: "assistant", - content: `[Previous ${toolName} result; call_id=${callId}]: ${text}`, - } as InputItem; - } - } - return item; - }); + body.input = repairToolCallPairs(body.input); } } diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 2b0f5f2b3..8fc389838 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -212,6 +212,59 @@ export function repairOrphanResponsesToolOutputs(input: ResponseInput): Response }); } +/** Placeholder output for a tool call whose result is absent from the input. */ +const ORPHAN_TOOL_CALL_PLACEHOLDER = + "[No tool output recorded: the tool call was interrupted before it produced a result.]"; + +/** + * Synthesize a placeholder `function_call_output` / `custom_tool_call_output` + * for every `function_call` / `custom_tool_call` whose `call_id` has no matching + * output later in the same input. The Responses API rejects an unpaired call + * with `400 No tool output found for function call …`. + * + * Orphan calls surface when the user branches/navigates the session tree to a + * node that ends on a tool call (the tool-result child is excluded from the + * reconstructed history) or when a turn is aborted/crashes after the call + * streamed but before its result persisted. Dropping the call would erase the + * assistant's action; a placeholder output keeps the call visible so the model + * can recover (e.g. re-issue the call). Symmetric to + * {@link repairOrphanResponsesToolOutputs}. + */ +export function repairOrphanResponsesToolCalls(input: ResponseInput): ResponseInput { + const outputCallIds = new Set(); + for (const item of input) { + const t = (item as { type?: string }).type; + if (t !== "function_call_output" && t !== "custom_tool_call_output") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId === "string") outputCallIds.add(callId); + } + let hasOrphan = false; + for (const item of input) { + const t = (item as { type?: string }).type; + if (t !== "function_call" && t !== "custom_tool_call") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId === "string" && !outputCallIds.has(callId)) { + hasOrphan = true; + break; + } + } + if (!hasOrphan) return input; + const repaired: ResponseInput = []; + for (const item of input) { + repaired.push(item); + const t = (item as { type?: string }).type; + if (t !== "function_call" && t !== "custom_tool_call") continue; + const callId = (item as { call_id?: unknown }).call_id; + if (typeof callId !== "string" || outputCallIds.has(callId)) continue; + repaired.push({ + type: t === "custom_tool_call" ? "custom_tool_call_output" : "function_call_output", + call_id: callId, + output: ORPHAN_TOOL_CALL_PLACEHOLDER, + } as ResponseInput[number]); + } + return repaired; +} + export function convertResponsesInputContent( content: string | Array, supportsImages: boolean, diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index ac1684b43..ed6de0481 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -62,6 +62,7 @@ import { isOpenAIResponsesProgressEvent, normalizeResponsesToolCallIdForTransform, processResponsesStream, + repairOrphanResponsesToolCalls, repairOrphanResponsesToolOutputs, } from "./openai-responses-shared"; import { transformMessages } from "./transform-messages"; @@ -614,7 +615,7 @@ function convertConversationMessages( msgIndex++; } - return repairOrphanResponsesToolOutputs(messages); + return repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(messages)); } /** diff --git a/packages/ai/test/openai-codex.test.ts b/packages/ai/test/openai-codex.test.ts index 4dbc616a7..ed81d7c78 100644 --- a/packages/ai/test/openai-codex.test.ts +++ b/packages/ai/test/openai-codex.test.ts @@ -71,6 +71,62 @@ describe("openai-codex request transformer", () => { }); }); +describe("openai-codex orphan tool-call repair", () => { + it("synthesizes a function_call_output for a function_call with no result", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [ + { type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] }, + { type: "function_call", call_id: "call_orphan", name: "read", arguments: "{}" }, + { type: "message", role: "user", content: [{ type: "input_text", text: "next" }] }, + ], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const callIndex = input.findIndex(item => item.type === "function_call" && item.call_id === "call_orphan"); + expect(callIndex).toBeGreaterThanOrEqual(0); + // The synthesized output sits immediately after the orphan call. + const output = input[callIndex + 1]; + expect(output?.type).toBe("function_call_output"); + expect(output?.call_id).toBe("call_orphan"); + expect(typeof output?.output).toBe("string"); + expect(output?.output as string).toMatch(/interrupted/i); + }); + + it("leaves a paired function_call untouched", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [ + { type: "function_call", call_id: "call_paired", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "call_paired", output: "real result" }, + ], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const outputs = input.filter(item => item.type === "function_call_output" && item.call_id === "call_paired"); + expect(outputs).toHaveLength(1); + expect(outputs[0]?.output).toBe("real result"); + }); + + it("synthesizes a custom_tool_call_output for an orphan custom_tool_call", async () => { + const body: RequestBody = { + model: "gpt-5.1-codex", + input: [{ type: "custom_tool_call", call_id: "call_custom", name: "apply_patch" }], + }; + + const transformed = await transformRequestBody(body, createCodexModel(body.model), {}); + const input = transformed.input || []; + + const output = input.find(item => item.type === "custom_tool_call_output" && item.call_id === "call_custom"); + expect(output).toBeDefined(); + expect(output?.output as string).toMatch(/interrupted/i); + }); +}); + describe("openai-codex reasoning effort validation", () => { it("rejects gpt-5.1 xhigh when metadata does not list it", async () => { const body: RequestBody = { model: "gpt-5.1", input: [] }; diff --git a/packages/ai/test/openai-responses-orphan-repair.test.ts b/packages/ai/test/openai-responses-orphan-repair.test.ts new file mode 100644 index 000000000..014d6988f --- /dev/null +++ b/packages/ai/test/openai-responses-orphan-repair.test.ts @@ -0,0 +1,69 @@ +import { describe, expect, it } from "bun:test"; +import { + repairOrphanResponsesToolCalls, + repairOrphanResponsesToolOutputs, +} from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; +import type { ResponseInput } from "openai/resources/responses/responses"; + +describe("repairOrphanResponsesToolCalls", () => { + it("appends a synthetic function_call_output after a call with no result", () => { + const input: ResponseInput = [ + { type: "function_call", call_id: "call_a", name: "read", arguments: "{}" }, + { role: "user", content: [{ type: "input_text", text: "continue" }] }, + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + const callIndex = repaired.findIndex( + item => + (item as { type?: string }).type === "function_call" && (item as { call_id?: string }).call_id === "call_a", + ); + const output = repaired[callIndex + 1] as { type?: string; call_id?: string; output?: unknown }; + expect(output.type).toBe("function_call_output"); + expect(output.call_id).toBe("call_a"); + expect(output.output).toMatch(/interrupted/i); + }); + + it("uses custom_tool_call_output for an orphan custom_tool_call", () => { + const input: ResponseInput = [ + { type: "custom_tool_call", call_id: "call_c", name: "apply_patch", input: "patch" } as ResponseInput[number], + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + const output = repaired.find(item => (item as { type?: string }).type === "custom_tool_call_output") as + | { call_id?: string } + | undefined; + expect(output?.call_id).toBe("call_c"); + }); + + it("returns the input unchanged when every call is paired", () => { + const input: ResponseInput = [ + { type: "function_call", call_id: "call_a", name: "read", arguments: "{}" }, + { type: "function_call_output", call_id: "call_a", output: "ok" } as ResponseInput[number], + ]; + + const repaired = repairOrphanResponsesToolCalls(input); + expect(repaired).toBe(input); + }); + + it("composes with output repair so a tree-branch snapshot stays API-valid", () => { + // Branching to a node that ends on a tool call drops the result child: + // the assistant turn keeps the call, but no matching output remains. + const input: ResponseInput = [ + { role: "user", content: [{ type: "input_text", text: "do it" }] }, + { type: "function_call", call_id: "call_x", name: "bash", arguments: "{}" }, + ]; + + const repaired = repairOrphanResponsesToolCalls(repairOrphanResponsesToolOutputs(input)); + const callIds = new Set( + repaired + .filter(i => (i as { type?: string }).type === "function_call") + .map(i => (i as { call_id: string }).call_id), + ); + const outputIds = new Set( + repaired + .filter(i => (i as { type?: string }).type === "function_call_output") + .map(i => (i as { call_id: string }).call_id), + ); + for (const id of callIds) expect(outputIds.has(id)).toBe(true); + }); +}); From 53b8f0f3fe7a08d9461123bbfab5380e53a31cd7 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 053/181] docs(packages/hashline): clarified patch hunk rules for #TAG-safe edits - Specified replace N..M ranges as inclusive to prevent accidental boundary truncation. - Clarified that edits must target only lines actually read, not merely covered by a #TAG. - Added guidance to keep pure insertions as insert hunks and avoid widening replace ranges. --- packages/hashline/src/prompt.md | 15 ++++++++++++++- 1 file changed, 14 insertions(+), 1 deletion(-) diff --git a/packages/hashline/src/prompt.md b/packages/hashline/src/prompt.md index 85396d8cf..6547ff2e6 100644 --- a/packages/hashline/src/prompt.md +++ b/packages/hashline/src/prompt.md @@ -5,7 +5,7 @@ Every file section starts with `[PATH#TAG]`. `TAG` is the 4-hex snapshot tag fro -replace N..M: replace original lines N..M with the body rows below. +replace N..M: replace original lines N..M with the body rows below. CAUTION, IT IS INCLUSIVE! MAKE SURE YOU INTEND TO DELETE BOTH ENDS! replace block N: replace the whole syntactic block that BEGINS on line N — its header line through its closing line — resolved with tree-sitter. Body rows below. Point N at the line that OPENS the construct (the `if`/`function`/`def`/`{`-bearing line), not a closing `}` or a blank line. delete N..M delete original lines N..M. No body. delete block N delete the whole syntactic block that BEGINS on line N. @@ -27,10 +27,13 @@ There is NO other body row kind. NEVER write `-old` or a bare/context line. To k - Numbers refer to the ORIGINAL file and stay valid for the whole patch — they do not shift as hunks apply. - Across calls they do NOT survive: each applied edit mints a fresh `#TAG` and renumbers the file, so the tag and line numbers you just used are dead. Anchor the next edit on the `[PATH#TAG]` and lines from the edit response (or re-`read`), never on pre-edit numbers. - A line number is an offset, not a structural boundary: never `insert after N` into a construct you have not read, and never start or end a `replace`/`delete` range mid-expression or mid-block. If unsure what is on those lines, `read` them first. +- A valid `#TAG` is NOT permission to patch the whole file — it certifies the snapshot, not your knowledge of it. Authority to touch a line comes from having literally seen that line as a `LINE:TEXT` row in a `read`/`search`, not from holding the tag. Every line in a hunk's range, and the lines bounding it, must be lines you actually saw. +- An elided or partial read is NOT a read of the gap. A `…` (or any collapsed/truncated region) between two excerpts means those lines are UNSEEN — treat them exactly like lines you never opened. Never place a hunk on, or span a range across, an elided region; `read` that range explicitly first. Reconstructing it from memory of "what the code probably looks like" is how ranges drift off-by-N and shred neighboring blocks. - On a stale-tag rejection — or any result you cannot fully account for — STOP and re-`read`. Never stack more line-numbered edits onto output you have not re-grounded; that compounds corruption. - One hunk per range; the body is the final content, never an old/new pair. - Keep every range as tight as the change: a range must cover ONLY lines whose content actually changes. Never widen it to swallow an unchanged signature, brace, or neighboring statement just to rewrite a few lines inside — change one line with `replace N..N`, not the whole block around it. (A range where every line genuinely changes is correctly long; tightness is about excluding unchanged lines, not about being short.) This bounds the blast radius if a number is off: a stale single-line replace corrupts one line, while a stale block replace shreds the whole block and its structure. - To change lines 2 and 5 while keeping 3–4, issue two hunks (`replace 2..2:` and `replace 5..5:`). Untouched lines are simply absent from every range. +- Pure additions use `insert`, never a widened `replace`. If the change only adds lines, `insert before/after` the spot and keep every existing line out of all ranges. Do NOT `replace` a span of keepers and retype them around the new line "to preserve" them — those retyped keepers are exactly what gets silently dropped when one is forgotten. A keeper that never enters your body cannot be lost. `replace` is only for lines whose own text changes. - NEVER use this tool to format code — reordering imports, re-indenting, aligning columns, or any mechanical restyling. That is the project formatter's job; run it instead of hand-editing layout here. @@ -99,6 +102,16 @@ replace 3..3: # RIGHT replace 3..3: + return msg + +# WRONG — a pure insertion done as a widened `replace`: you only want to add one line after 2, +# but you replace 2..4, retype the keepers in the body, and drop one (here line 4, `greet("world")`). +replace 2..4: ++ msg = "Hello, " + name ++ extra = compute(name) ++ print(msg) +# RIGHT — touch nothing you keep; the new line is the whole body. +insert after 2: ++ extra = compute(name) From 9a2e766c90a9d791b889769841d336db1ec3a75e Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 054/181] perf(packages/tui): optimized per-frame TUI line fitting cache - Added a per-frame cache for #fitLineToWidth results and cleared it at each #doRender. - Reused memoized line-fit outputs to skip repeated visibleWidth/truncate work. --- packages/tui/src/tui.ts | 36 ++++++++++++++++++++++++++++++------ 1 file changed, 30 insertions(+), 6 deletions(-) diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 732af473c..539144a33 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -340,7 +340,8 @@ export class Container implements Component { width = Math.max(1, width); const lines: string[] = []; for (const child of this.children) { - lines.push(...child.render(width)); + const childLines = child.render(width); + for (let i = 0; i < childLines.length; i++) lines.push(childLines[i]); } return lines; } @@ -395,6 +396,11 @@ type RenderIntent = export class TUI extends Container { terminal: Terminal; #previousLines: string[] = []; + // Per-frame cache of #fitLineToWidth results. Cleared at the top of every + // #doRender (where the frame width is fixed), so it only ever holds entries + // for one width. Eliminates the duplicate fit work between the compose pass + // and the emitters, plus repeated fits of identical blank padding rows. + #fitLineCache = new Map(); #previousWidth = 0; #previousHeight = 0; #focusedComponent: Component | null = null; @@ -507,7 +513,7 @@ export class TUI extends Container { this.#nativeScrollbackCommitSafeEnd = offset + boundedEnd; } } - lines.push(...childLines); + for (let i = 0; i < childLines.length; i++) lines.push(childLines[i]); } return lines; } @@ -1468,6 +1474,9 @@ export class TUI extends Container { if (this.#stopped) return; const width = this.terminal.columns; const height = this.terminal.rows; + // Reset the per-frame fit memo: width is fixed for this frame, so cached + // fit results stay valid across the compose pass and every emitter re-fit. + this.#fitLineCache.clear(); // 1. Compose the frame. Bracket the transcript render so the image budget // observes every inline image in display order (overlays carry none). @@ -2352,10 +2361,25 @@ export class TUI extends Container { } #fitLineToWidth(line: string, width: number): string { - if (TERMINAL.isImageLine(line)) return line; - if (visibleWidth(line) <= width) return line; - const truncated = truncateToWidth(line, width, Ellipsis.Omit); - return truncated + (truncated.includes("\x1b]8;") ? LINE_TERMINATOR : SEGMENT_RESET); + // Frame-scoped memo: #doRender clears this each frame after reading the + // terminal width, so within a frame `width` is constant and this map is + // keyed by line alone. The compose/fit pass (#fitLinesToWidth) and every + // emitter re-fit the same lines (and many repeated blank rows); the result + // is pure for a fixed width, so caching it is byte-identical and skips the + // redundant native visibleWidth/truncate work. + const cached = this.#fitLineCache.get(line); + if (cached !== undefined) return cached; + let result: string; + if (TERMINAL.isImageLine(line)) { + result = line; + } else if (visibleWidth(line) <= width) { + result = line; + } else { + const truncated = truncateToWidth(line, width, Ellipsis.Omit); + result = truncated + (truncated.includes("\x1b]8;") ? LINE_TERMINATOR : SEGMENT_RESET); + } + this.#fitLineCache.set(line, result); + return result; } /** From d5c1f6e3ab18439d8705475eea20e3c9781de167 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 055/181] perf(packages/coding-agent): optimized LSP frame parsing via chunk queue - Reworked message parsing to process complete frames from a pending chunk queue. - Added chunk-aware header scanning and range copy to avoid repeated buffer concatenation. - Persisted partially read bytes in client.messageBuffer during reader teardown. --- packages/coding-agent/src/lsp/client.ts | 143 +++++++++++++++--------- 1 file changed, 93 insertions(+), 50 deletions(-) diff --git a/packages/coding-agent/src/lsp/client.ts b/packages/coding-agent/src/lsp/client.ts index 3080fc083..8127cdfc4 100644 --- a/packages/coding-agent/src/lsp/client.ts +++ b/packages/coding-agent/src/lsp/client.ts @@ -174,49 +174,69 @@ const CLIENT_CAPABILITIES = { // LSP Message Protocol // ============================================================================= -/** - * Parse a single LSP message from a buffer. - * Returns the parsed message and remaining buffer, or null if incomplete. - */ -function parseMessage( - buffer: Buffer, -): { message: LspJsonRpcResponse | LspJsonRpcNotification; remaining: Buffer } | null { - // Only decode enough to find the header - const headerEndIndex = findHeaderEnd(buffer); - if (headerEndIndex === -1) return null; - - const headerText = new TextDecoder().decode(buffer.slice(0, headerEndIndex)); - const contentLengthMatch = headerText.match(/Content-Length: (\d+)/i); - if (!contentLengthMatch) return null; - - const contentLength = Number.parseInt(contentLengthMatch[1], 10); - const messageStart = headerEndIndex + 4; // Skip \r\n\r\n - const messageEnd = messageStart + contentLength; - - if (buffer.length < messageEnd) return null; - - const messageBytes = buffer.subarray(messageStart, messageEnd); - const messageText = new TextDecoder().decode(messageBytes); - const remaining = buffer.subarray(messageEnd); - - return { - message: JSON.parse(messageText), - remaining, - }; -} +// Reused for all full (non-streaming) decodes; each decode() resets state, so a +// single instance is safe and avoids per-message TextDecoder allocation. +const MESSAGE_DECODER = new TextDecoder("utf-8"); /** - * Find the end of the header section (before \r\n\r\n) + * Locate the `\r\n\r\n` header terminator across the pending chunk list. + * Returns the absolute byte index of the first `\r`, or -1 when not present. + * Equivalent to scanning the contiguous concatenation of the chunks. */ -function findHeaderEnd(buffer: Uint8Array): number { - for (let i = 0; i < buffer.length - 3; i++) { - if (buffer[i] === 13 && buffer[i + 1] === 10 && buffer[i + 2] === 13 && buffer[i + 3] === 10) { - return i; +function findHeaderEndInChunks(chunks: Buffer[]): number { + let global = 0; + let b0 = -1; + let b1 = -1; + let b2 = -1; + for (const chunk of chunks) { + for (let i = 0; i < chunk.length; i++) { + const b3 = chunk[i]; + if (b0 === 13 && b1 === 10 && b2 === 13 && b3 === 10) { + return global - 3; + } + b0 = b1; + b1 = b2; + b2 = b3; + global++; } } return -1; } +/** Copy the byte range [from, to) out of the pending chunk list into one Buffer. */ +function copyChunkRange(chunks: Buffer[], from: number, to: number): Buffer { + const out = Buffer.allocUnsafe(to - from); + let global = 0; + let written = 0; + for (const chunk of chunks) { + const chunkEnd = global + chunk.length; + if (chunkEnd > from && global < to) { + const start = Math.max(from, global) - global; + const end = Math.min(to, chunkEnd) - global; + chunk.copy(out, written, start, end); + written += end - start; + } + global = chunkEnd; + if (global >= to) break; + } + return out; +} + +/** Drop the first `count` bytes from the pending chunk list in place. */ +function dropChunkFront(chunks: Buffer[], count: number): void { + let removed = 0; + while (chunks.length > 0) { + const head = chunks[0]; + if (removed + head.length <= count) { + removed += head.length; + chunks.shift(); + } else { + chunks[0] = head.subarray(count - removed); + break; + } + } +} + async function writeMessage( sink: Bun.FileSink, message: LspJsonRpcRequest | LspJsonRpcNotification | LspJsonRpcResponse, @@ -249,22 +269,43 @@ async function startMessageReader(client: LspClient): Promise { const reader = (client.proc.stdout as ReadableStream).getReader(); + // Incoming bytes are buffered as a list of chunks and only joined when a full + // message is framed. Concatenating the accumulator on every read was O(n^2) + // for messages that span many reads (e.g. a large initial diagnostics burst). + const pendingChunks: Buffer[] = []; + let pendingLen = 0; + if (client.messageBuffer.length > 0) { + const seed = Buffer.from(client.messageBuffer); + pendingChunks.push(seed); + pendingLen = seed.length; + } + try { while (true) { const { done, value } = await reader.read(); if (done) break; - // Atomically update buffer before processing - const currentBuffer: Buffer = Buffer.concat([client.messageBuffer, value]); - client.messageBuffer = currentBuffer; + pendingChunks.push(Buffer.from(value)); + pendingLen += value.length; - // Process all complete messages in buffer - // Use local variable to avoid race with concurrent buffer updates - let workingBuffer = currentBuffer; - let parsed = parseMessage(workingBuffer); - while (parsed) { - const { message, remaining } = parsed; - workingBuffer = remaining; + // Drain every complete message currently buffered. + while (true) { + const headerEnd = findHeaderEndInChunks(pendingChunks); + if (headerEnd === -1) break; + + const headerText = MESSAGE_DECODER.decode(copyChunkRange(pendingChunks, 0, headerEnd)); + const contentLengthMatch = headerText.match(/Content-Length: (\d+)/i); + if (!contentLengthMatch) break; + + const contentLength = Number.parseInt(contentLengthMatch[1], 10); + const messageStart = headerEnd + 4; // Skip \r\n\r\n + const messageEnd = messageStart + contentLength; + if (pendingLen < messageEnd) break; + + const messageText = MESSAGE_DECODER.decode(copyChunkRange(pendingChunks, messageStart, messageEnd)); + const message: LspJsonRpcResponse | LspJsonRpcNotification = JSON.parse(messageText); + dropChunkFront(pendingChunks, messageEnd); + pendingLen -= messageEnd; // Route message if ("id" in message && message.id !== undefined) { @@ -301,12 +342,7 @@ async function startMessageReader(client: LspClient): Promise { } } } - - parsed = parseMessage(workingBuffer); } - - // Atomically commit processed buffer - client.messageBuffer = workingBuffer; } } catch (err) { // Connection closed or error - reject all pending requests @@ -315,6 +351,13 @@ async function startMessageReader(client: LspClient): Promise { } client.pendingRequests.clear(); } finally { + // Persist any unparsed remainder so a restarted reader resumes mid-message. + client.messageBuffer = + pendingChunks.length === 0 + ? new Uint8Array(0) + : pendingChunks.length === 1 + ? pendingChunks[0] + : Buffer.concat(pendingChunks, pendingLen); reader.releaseLock(); client.isReading = false; } From cc283cf50f3f0ec8971ca7a45fffb8c572d19a55 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 056/181] feat(packages/coding-agent): added streaming append-only preview behavior - Added `isStreamingPreviewAppendOnly` to `ToolRenderer` for per-tool streaming mode selection. - Updated `ToolExecutionComponent` to query append-only predicates only while a call preview is streaming. - Marked expanded write previews as append-only so over-tall streaming output can commit head rows. - Threaded resolved status-line `segmentOptions` into `#buildSegmentContext` construction. - Added regression tests for scrollback retention and append-only state transitions. --- .../src/modes/components/status-line.ts | 8 +- .../src/modes/components/tool-execution.ts | 28 ++++++- packages/coding-agent/src/tools/renderers.ts | 13 ++- packages/coding-agent/src/tools/write.ts | 10 +++ .../test/status-line-context-cache.test.ts | 2 +- .../test/tool-live-region-scrollback.test.ts | 82 +++++++++++++++++++ 6 files changed, 135 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/src/modes/components/status-line.ts b/packages/coding-agent/src/modes/components/status-line.ts index 26f6e10f4..6c8b338d4 100644 --- a/packages/coding-agent/src/modes/components/status-line.ts +++ b/packages/coding-agent/src/modes/components/status-line.ts @@ -546,7 +546,7 @@ export class StatusLineComponent implements Component { return `${modelId}|${sp.length}:${sp[0]?.length ?? 0}|${tools.length}|${skills.length}`; } - #buildSegmentContext(width: number): SegmentContext { + #buildSegmentContext(width: number, segmentOptions: StatusLineSettings["segmentOptions"]): SegmentContext { const state = this.session.state; // Trigger background fetch (5-min TTL); render uses cached value @@ -575,7 +575,7 @@ export class StatusLineComponent implements Component { return { session: this.session, width, - options: this.#resolveSettings().segmentOptions ?? {}, + options: segmentOptions ?? {}, planMode: this.#planModeStatus, loopMode: this.#loopModeStatus, goalMode: this.#goalModeStatus, @@ -632,8 +632,8 @@ export class StatusLineComponent implements Component { } #buildStatusLine(width: number): string { - const ctx = this.#buildSegmentContext(width); const effectiveSettings = this.#resolveSettings(); + const ctx = this.#buildSegmentContext(width, effectiveSettings.segmentOptions); const separatorDef = getSeparator(effectiveSettings.separator ?? "powerline-thin", theme); const bgAnsi = theme.getBgAnsi("statusLineBg"); @@ -759,8 +759,6 @@ export class StatusLineComponent implements Component { return leftGroup + (leftGroup && rightGroup ? " " : "") + rightGroup; } - leftWidth = groupWidth(left, leftCapWidth, leftSepWidth); - rightWidth = groupWidth(right, rightCapWidth, rightSepWidth); const gapWidth = Math.max(1, topFillWidth - leftWidth - rightWidth); const sessionName = effectiveSettings.sessionAccent !== false ? this.session.sessionManager?.getSessionName() : undefined; diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index 5b90a8a6b..328969a67 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -31,7 +31,7 @@ import { renderJsonTreeLines, } from "../../tools/json-tree"; import { formatExpandHint, replaceTabs, resolveImageOptions, truncateToWidth } from "../../tools/render-utils"; -import { toolRenderers } from "../../tools/renderers"; +import { type ToolRenderer, toolRenderers } from "../../tools/renderers"; import { TODO_STRIKE_TOTAL_FRAMES } from "../../tools/todo"; import { isFramedBlockComponent, renderStatusLine } from "../../tui"; import { sanitizeWithOptionalSixelPassthrough } from "../../utils/sixel"; @@ -530,6 +530,32 @@ export class ToolExecutionComponent extends Container { return (this.#result.details as { async?: { state?: string } } | undefined)?.async?.state === "running"; } + /** + * While streaming its call preview, a tool block whose preview is append-only + * (rows only grow at the bottom, never re-layout) lets the renderer commit the + * scrolled-off head of an over-tall preview to native scrollback instead of + * dropping it — the same anti-yank path a streaming assistant reply uses (see + * {@link TranscriptContainer} + `NativeScrollbackLiveRegion`). Gated on the + * call-preview phase (no result yet) so the boundary closes the instant the + * preview swaps to a result that may collapse; the renderer decides whether + * its current preview shape qualifies via `isStreamingPreviewAppendOnly`. + */ + isTranscriptBlockAppendOnly(): boolean { + // A result preview can collapse/re-layout; only the live call preview is a + // candidate. Sealed/aborted blocks are finalized, not streaming. + if (this.#sealed || this.#result !== undefined) return false; + const predicate = + (this.#tool as { isStreamingPreviewAppendOnly?: ToolRenderer["isStreamingPreviewAppendOnly"] } | undefined) + ?.isStreamingPreviewAppendOnly ?? toolRenderers[this.#toolName]?.isStreamingPreviewAppendOnly; + if (!predicate) return false; + try { + return predicate(this.#getCallArgsForRender(), this.#renderState); + } catch (err) { + logger.warn("Tool append-only predicate failed", { tool: this.#toolName, error: String(err) }); + return false; + } + } + /** * Mark the tool terminal even though no result arrived (the turn aborted or * abandoned it) and stop animating, so it can freeze and stops pinning the diff --git a/packages/coding-agent/src/tools/renderers.ts b/packages/coding-agent/src/tools/renderers.ts index eebbe57a1..9f8cbca29 100644 --- a/packages/coding-agent/src/tools/renderers.ts +++ b/packages/coding-agent/src/tools/renderers.ts @@ -31,7 +31,7 @@ import { sshToolRenderer } from "./ssh"; import { todoToolRenderer } from "./todo"; import { writeToolRenderer } from "./write"; -type ToolRenderer = { +export type ToolRenderer = { renderCall: (args: unknown, options: RenderResultOptions, theme: Theme) => Component; renderResult: ( result: { content: Array<{ type: string; text?: string }>; details?: unknown; isError?: boolean }, @@ -40,6 +40,17 @@ type ToolRenderer = { args?: unknown, ) => Component; mergeCallAndResult?: boolean; + /** + * While the call preview is streaming, report whether the currently-rendered + * preview is append-only: its rows only grow at the bottom and never + * re-layout (a full, top-anchored content preview). The transcript reports + * this up to the TUI so a streaming preview taller than the viewport commits + * its scrolled-off head to native scrollback instead of dropping it (see + * `ToolExecutionComponent.isTranscriptBlockAppendOnly`). Omit (or return + * `false`) for previews that slide a tail window or later collapse to a + * compact result — committing their head would strand stale rows. + */ + isStreamingPreviewAppendOnly?: (args: unknown, options: RenderResultOptions) => boolean; /** Render without background box, inline in the response flow */ inline?: boolean; }; diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index 4c1d92af4..35b0683f2 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -1021,6 +1021,16 @@ export const writeToolRenderer = { return new Text(text, 0, 0); }, + // Only the expanded (Ctrl+O) preview is append-only: it renders the whole + // content top-anchored, so streamed chunks only append rows at the bottom. + // The collapsed preview slides a bounded tail window (`formatStreamingContent` + // with `WRITE_STREAMING_PREVIEW_LINES`) whose visible rows re-layout as the + // window moves — not append-only, but it never overflows the viewport, so its + // head is never at risk of being dropped regardless. + isStreamingPreviewAppendOnly(args: WriteRenderArgs, options: RenderResultOptions): boolean { + return Boolean(options?.expanded && args.content); + }, + renderResult( result: { content: Array<{ type: string; text?: string }>; details?: WriteToolDetails; isError?: boolean }, options: RenderResultOptions, diff --git a/packages/coding-agent/test/status-line-context-cache.test.ts b/packages/coding-agent/test/status-line-context-cache.test.ts index 95ff25dc5..447dd825d 100644 --- a/packages/coding-agent/test/status-line-context-cache.test.ts +++ b/packages/coding-agent/test/status-line-context-cache.test.ts @@ -18,8 +18,8 @@ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; import { countTokens } from "@oh-my-pi/pi-natives"; import { resetSettingsForTest, Settings } from "../src/config/settings"; import { StatusLineComponent } from "../src/modes/components/status-line"; -import { computeNonMessageTokens, estimateToolSchemaTokens } from "../src/modes/utils/context-usage"; import { initTheme } from "../src/modes/theme/theme"; +import { computeNonMessageTokens, estimateToolSchemaTokens } from "../src/modes/utils/context-usage"; import type { AgentSession } from "../src/session/agent-session"; beforeAll(async () => { diff --git a/packages/coding-agent/test/tool-live-region-scrollback.test.ts b/packages/coding-agent/test/tool-live-region-scrollback.test.ts index e30e56e82..aae7130a7 100644 --- a/packages/coding-agent/test/tool-live-region-scrollback.test.ts +++ b/packages/coding-agent/test/tool-live-region-scrollback.test.ts @@ -139,6 +139,88 @@ describe("tool live-region scrollback", () => { } }); }); + + it("commits the scrolled-off head of an over-tall expanded streaming write to scrollback", async () => { + if (process.platform === "win32") return; + + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(120, 12); + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => + undefined; + const tui = new TUI(term); + const chat = new TranscriptContainer(); + const body = (n: number) => Array.from({ length: n }, (_unused, i) => `MARK-${i}`).join("\n"); + const filePath = "packages/coding-agent/test/probe.txt"; + // Expanded (Ctrl+O) lifts the tail-window cap, so the preview renders the + // whole content top-anchored — append-only growth as chunks stream in. + const component = new ToolExecutionComponent( + "write", + { file_path: filePath, content: body(4) }, + {}, + undefined, + tui, + process.cwd(), + ); + component.setExpanded(true); + + try { + chat.addChild(component); + tui.addChild(chat); + tui.start(); + tui.setEagerNativeScrollbackRebuild(true); + await term.waitForRender(); + + // A short preview that fits, then the full preview that alone overflows + // the 12-row viewport — the frame that scrolls the head above the top. + component.updateArgs({ file_path: filePath, content: body(4) }); + tui.requestRender(); + await term.waitForRender(); + + component.updateArgs({ file_path: filePath, content: body(40) }); + tui.requestRender(); + await term.waitForRender(); + + const strip = (rows: string[]) => rows.map(row => Bun.stripANSI(row).trimEnd()).join("\n"); + const scrollText = strip(term.getScrollBuffer()); + const viewportText = strip(term.getViewport()); + + // MARK-0 scrolled above the viewport: it must live in native scrollback + // (committed), not nowhere. Before the fix the tool block was not + // append-only, so its scrolled-off head was dropped — a yanked stream. + expect(viewportText).not.toContain("MARK-0"); + expect(scrollText).toContain("MARK-0"); + // The streaming tail stays on screen, and nothing went missing between. + expect(viewportText).toContain("MARK-39"); + expect(viewportText).toContain("(streaming)"); + expect(scrollText).toContain("MARK-20"); + } finally { + component.stopAnimation(); + tui.stop(); + await term.flush(); + } + }); + }); + + it("treats a tool block as append-only only while its expanded preview streams", async () => { + const filePath = "packages/coding-agent/test/probe.txt"; + const tui = new TUI(new VirtualTerminal(80, 24)); + const args = { file_path: filePath, content: "MARK-0\nMARK-1\nMARK-2" }; + const component = new ToolExecutionComponent("write", args, {}, undefined, tui, process.cwd()); + type AppendOnly = { isTranscriptBlockAppendOnly(): boolean }; + const probe = component as unknown as AppendOnly; + try { + // Collapsed: the preview slides a bounded tail window — not append-only. + expect(probe.isTranscriptBlockAppendOnly()).toBe(false); + // Expanded + streaming: append-only, eligible for head commit. + component.setExpanded(true); + expect(probe.isTranscriptBlockAppendOnly()).toBe(true); + // Once a final result lands the preview may collapse — boundary closes. + component.updateResult({ content: [{ type: "text", text: "" }], details: { path: filePath } }, false); + expect(probe.isTranscriptBlockAppendOnly()).toBe(false); + } finally { + component.stopAnimation(); + } + }); }); function makeAssistantMessage(text: string): AssistantMessage { From a3fb07428fe8ec4501fb987d685a74ae32563a60 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 057/181] perf(packages/coding-agent): optimized model equivalence namespace cache - Added a WeakMap cache in compileEquivalenceConfig to reuse compiled configs. - Raised QUALIFIED_NAMESPACE_SUFFIX_CACHE_CAP and HEURISTIC_CANDIDATES_CACHE_CAP from 256 to 4096. - Reworked resolveCanonicalIdForModel to build officialMatches during candidate iteration. --- .../src/config/model-equivalence.ts | 22 +++++++++++++++---- 1 file changed, 18 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/src/config/model-equivalence.ts b/packages/coding-agent/src/config/model-equivalence.ts index e30755b2d..8115f7d87 100644 --- a/packages/coding-agent/src/config/model-equivalence.ts +++ b/packages/coding-agent/src/config/model-equivalence.ts @@ -167,13 +167,24 @@ function buildExclusionSet(exclusions: readonly string[] | undefined): Set(); function compileEquivalenceConfig(config: ModelEquivalenceConfig | undefined): CompiledEquivalenceConfig { + if (config) { + const cached = compiledEquivalenceCache.get(config); + if (cached) { + return cached; + } + } const overrides = buildOverrideMap(config?.overrides); const exclude = buildExclusionSet(config?.exclude); if (overrides.size === 0 && exclude.size === 0) { return EMPTY_COMPILED_EQUIVALENCE; } - return { overrides, exclude }; + const compiled: CompiledEquivalenceConfig = { overrides, exclude }; + if (config) { + compiledEquivalenceCache.set(config, compiled); + } + return compiled; } function addCanonicalCandidate(candidates: Set, candidate: string): void { @@ -285,7 +296,7 @@ function expandCompactSeriesMinorVersions(candidate: string): string[] { // safely return the same instance. Cap keeps memory bounded under adversarial // model-id churn. const QUALIFIED_NAMESPACE_SUFFIX_CACHE = new Map(); -const QUALIFIED_NAMESPACE_SUFFIX_CACHE_CAP = 256; +const QUALIFIED_NAMESPACE_SUFFIX_CACHE_CAP = 4096; function getQualifiedNamespaceSuffixes(candidate: string): string[] { const cached = QUALIFIED_NAMESPACE_SUFFIX_CACHE.get(candidate); if (cached !== undefined) { @@ -678,7 +689,7 @@ function expandHeavyCanonicalCandidates(normalized: string, queue: string[]): vo // is unused — kept for signature stability). The returned array is consumed via // `.filter` at every callsite, so sharing the cached instance is safe. const HEURISTIC_CANDIDATES_CACHE = new Map(); -const HEURISTIC_CANDIDATES_CACHE_CAP = 256; +const HEURISTIC_CANDIDATES_CACHE_CAP = 4096; function getHeuristicCanonicalCandidates(modelId: string, _officialIds?: ReadonlySet): string[] { const cached = HEURISTIC_CANDIDATES_CACHE.get(modelId); if (cached !== undefined) { @@ -765,8 +776,11 @@ function resolveCanonicalIdForModel( } const heuristicCandidates = getHeuristicCanonicalCandidates(model.id, referenceData.officialIds); - const officialMatches = new Set(heuristicCandidates.filter(candidate => referenceData.officialIds.has(candidate))); + const officialMatches = new Set(); for (const candidate of heuristicCandidates) { + if (referenceData.officialIds.has(candidate)) { + officialMatches.add(candidate); + } const aliased = referenceData.suffixAliases.get(getCanonicalSuffixAliasKey(candidate)); if (aliased) { officialMatches.add(aliased); From e5e93ff762c080df19aefcd37c29e163263c9af9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 058/181] feat(packages/coding-agent): enabled setup-version-gated startup flow - Deferred setup wizard import until setup was forced or version stale. - Dynamically loaded ACP, RPC, and print mode runners only when used. - Added a marketplace auto-update scheduler with off-mode early exit and non-blocking errors. - Added setup-version assertions to keep CURRENT_SETUP_VERSION aligned with scenes. --- package.json | 1 + packages/coding-agent/CHANGELOG.md | 6 ++ .../plugins/marketplace-auto-update.ts | 49 ++++++++++ packages/coding-agent/src/main.ts | 89 +++++++++---------- packages/coding-agent/src/modes/index.ts | 9 +- .../coding-agent/src/modes/setup-version.ts | 11 +++ .../src/modes/setup-wizard/index.ts | 5 +- .../coding-agent/test/setup-wizard.test.ts | 9 ++ .../test/startup-import-graph.test.ts | 34 +++++++ 9 files changed, 160 insertions(+), 53 deletions(-) create mode 100644 packages/coding-agent/src/extensibility/plugins/marketplace-auto-update.ts create mode 100644 packages/coding-agent/src/modes/setup-version.ts create mode 100644 packages/coding-agent/test/startup-import-graph.test.ts diff --git a/package.json b/package.json index 78098d0fa..91848ac82 100644 --- a/package.json +++ b/package.json @@ -87,6 +87,7 @@ "scripts": { "install:dev": "bun install && bun --cwd=packages/coding-agent link && ln -sfn \"$(pwd)/packages/coding-agent/scripts/dev-launch\" \"$(bun pm -g bin)/omp\"", "dev": "bun --cwd=packages/coding-agent src/cli.ts", + "dev:timing": "PI_TIMING=x bun --cwd=packages/coding-agent --preload ../utils/src/module-timer.ts src/cli.ts", "stats": "bun --cwd=packages/coding-agent src/cli.ts stats", "claude:trace": "bun scripts/claude-trace.ts", "build": "bun run --workspaces --if-present build", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 5ae4c1bdd..aee58020f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -9,11 +9,17 @@ ### Changed - Changed eval `agent()` subagents so they are never subject to the `task.maxRuntimeMs` wall-clock cap. The parent cell's idle watchdog is already suspended for the entire bridge call (`withBridgeTimeoutPause`), so a long-running fan-out/recovery workflow must not be killed by a per-subagent runtime limit. `runEvalAgent` now passes `maxRuntimeMs: 0` to `runSubprocess`, which honors an explicit `ExecutorOptions.maxRuntimeMs` override over the inherited setting. +- Changed interactive timing behavior so `PI_TIMING=x pi` preloads the module timer before the CLI graph loads and includes the `(modules)` report. `PI_TIMING=full` now also exits after printing, matching `PI_TIMING=x`, so full module reports are usable for cold-start measurement without launching the TUI. Added the root `dev:timing` script for the same profiled startup path. +- Changed coding-agent startup imports so normal TUI launch imports `InteractiveMode` directly, keeps print/RPC/ACP runners on their branch-only paths, and moves marketplace auto-update work behind a lightweight deferred starter. +- Changed cold-launch setup gating so the full setup wizard (every scene plus the overlay and their TUI/OAuth/web-search/theme dependencies) is no longer statically imported by `main.ts`. The current setup version now lives in a tiny dependency-free `modes/setup-version` module, and the wizard barrel is lazy-loaded only when the stored setup version is stale or the wizard is forced — the common up-to-date launch skips loading it entirely. +- Changed cold-launch startup imports so the hot-path CLI files no longer pull the full `@oh-my-pi/pi-ai` barrel: `commands/launch.ts` and `cli/args.ts` import `THINKING_EFFORTS`/`Effort` from the tiny `@oh-my-pi/pi-ai/effort` module, and `config/model-registry.ts` now imports its ~20 symbols from narrow subpaths (`api-registry`, `model-cache`, `model-manager`, `model-thinking`, `models`, `provider-models`, `types`, `utils/event-stream`) instead of the barrel — so launching no longer eagerly loads every provider, auth, OAuth, and usage module re-exported by the barrel. + ### Fixed - Fixed eval `agent()` failures surfacing as an opaque `RuntimeError: bridge call '__agent__' failed` with no reason. When a subagent aborted, `runEvalAgent` built its failure message with `result.error ?? result.stderr ?? result.abortReason ?? …`, but `result.stderr` is the empty string on a clean abort (and `result.error` is gated on a non-empty `stderr`), so the nullish chain stopped at `""` and never reached `abortReason`. The empty string propagated through the loopback bridge and the Python prelude's `RuntimeError(msg or "bridge call … failed")`, discarding the real reason. The chain now uses `||` so an empty `stderr` falls through to `abortReason`. - Fixed subagent aborts being mislabeled as the generic "Cancelled by caller" when the abort originated inside the subagent's own turn (`stopReason: "aborted"` with no caller signal and no runtime-limit timer). `runSubprocess` now prefers the aborted assistant message's `errorMessage` (e.g. "Request was aborted" or a specific stream error) for that case, while a real caller signal or wall-clock abort still reports its precise reason. +- Fixed a long streaming tool preview that alone overflows the viewport dropping its scrolled-off head on ED3-risk terminals (ghostty/kitty/iTerm2/…). A streaming `write` preview expanded with `Ctrl+O` renders the whole content top-anchored and grows append-only, but the tool block never reported itself append-only to the transcript, so the renderer's commit-as-you-go boundary stopped at the block start and its earlier rows scrolled above the viewport without being committed to native scrollback — they vanished, leaving the preview looking like a viewport-tall circular buffer. `ToolExecutionComponent` now implements `isTranscriptBlockAppendOnly()`, delegating to a renderer-declared `isStreamingPreviewAppendOnly` predicate (gated to the live call-preview phase) so the expanded write stream commits its head exactly like a streamed assistant reply; collapsed previews (sliding tail window) and result previews (which can collapse) stay deferred. ## [15.9.69] - 2026-06-06 ### Fixed diff --git a/packages/coding-agent/src/extensibility/plugins/marketplace-auto-update.ts b/packages/coding-agent/src/extensibility/plugins/marketplace-auto-update.ts new file mode 100644 index 000000000..f81da80a2 --- /dev/null +++ b/packages/coding-agent/src/extensibility/plugins/marketplace-auto-update.ts @@ -0,0 +1,49 @@ +import { getProjectDir, logger } from "@oh-my-pi/pi-utils"; + +type MarketplaceAutoUpdateMode = "off" | "notify" | "auto"; + +interface MarketplaceAutoUpdateOptions { + autoUpdate: MarketplaceAutoUpdateMode; + resolveActiveProjectRegistryPath: (cwd: string) => Promise; + clearPluginRootsCache: () => void; +} + +export function scheduleMarketplaceAutoUpdate(options: MarketplaceAutoUpdateOptions): void { + if (options.autoUpdate === "off") { + return; + } + + void runMarketplaceAutoUpdate(options); +} + +async function runMarketplaceAutoUpdate(options: MarketplaceAutoUpdateOptions): Promise { + try { + // Startup perf: marketplace manager pulls scraper/fetch/cache code; keep it out of the initial TUI graph. + const { + MarketplaceManager, + getInstalledPluginsRegistryPath, + getMarketplacesCacheDir, + getMarketplacesRegistryPath, + getPluginsCacheDir, + } = await import("./marketplace"); + const mgr = new MarketplaceManager({ + marketplacesRegistryPath: getMarketplacesRegistryPath(), + installedRegistryPath: getInstalledPluginsRegistryPath(), + projectInstalledRegistryPath: (await options.resolveActiveProjectRegistryPath(getProjectDir())) ?? undefined, + marketplacesCacheDir: getMarketplacesCacheDir(), + pluginsCacheDir: getPluginsCacheDir(), + clearPluginRootsCache: options.clearPluginRootsCache, + }); + await mgr.refreshStaleMarketplaces(); + const updates = await mgr.checkForUpdates(); + if (updates.length === 0) return; + if (options.autoUpdate === "auto") { + await mgr.upgradeAllPlugins(); + logger.debug(`Auto-upgraded ${updates.length} marketplace plugin(s)`); + } else { + logger.debug(`${updates.length} marketplace plugin update(s) available — /marketplace upgrade`); + } + } catch { + // Silently ignore — network failure, corrupt data, offline. + } +} diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index 8354521b3..fa4c24011 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -43,16 +43,11 @@ import { injectOmpExtensionCliRoots } from "./discovery/omp-extension-roots"; import { exportFromFile } from "./export/html"; import { ExtensionRunner } from "./extensibility/extensions/runner"; import type { ExtensionUIContext } from "./extensibility/extensions/types"; -import { - getInstalledPluginsRegistryPath, - getMarketplacesCacheDir, - getMarketplacesRegistryPath, - getPluginsCacheDir, - MarketplaceManager, -} from "./extensibility/plugins/marketplace"; +import { scheduleMarketplaceAutoUpdate } from "./extensibility/plugins/marketplace-auto-update"; import type { MCPManager } from "./mcp"; -import { InteractiveMode, runAcpMode, runPrintMode, runRpcMode } from "./modes"; -import { ALL_SCENES, runSetupWizard, selectSetupScenes } from "./modes/setup-wizard"; +import { InteractiveMode } from "./modes/interactive-mode"; +import type { PrintModeOptions } from "./modes/print-mode"; +import { CURRENT_SETUP_VERSION } from "./modes/setup-version"; import { initTheme, stopThemeWatcher } from "./modes/theme/theme"; import type { SubmittedUserInput } from "./modes/types"; import { @@ -72,6 +67,13 @@ import type { LspStartupServerInfo } from "./tools"; import { getChangelogPath, getNewEntries, parseChangelog } from "./utils/changelog"; import { EventBus } from "./utils/event-bus"; +type RunAcpMode = (createSession: AcpSessionFactory) => Promise; +type RunPrintMode = (session: AgentSession, options: PrintModeOptions) => Promise; +type RunRpcMode = ( + session: AgentSession, + setToolUIContext?: (uiContext: ExtensionUIContext, hasUI: boolean) => void, +) => Promise; + async function checkForNewVersion(currentVersion: string): Promise { if (!settings.get("startup.checkUpdate")) { return; @@ -261,17 +263,26 @@ async function runInteractiveMode( eventBus, ); - const setupScenes = await selectSetupScenes(settings.get("setupVersion"), ALL_SCENES, mode, { - resuming, - isTTY: process.stdin.isTTY && process.stdout.isTTY, - setupWizardEnabled: settings.get("startup.setupWizard"), - force: forceSetupWizard, - }); + // Cold-launch gate: the full setup wizard (every scene + the overlay and + // their TUI/OAuth/search/theme deps) is heavy, yet the common case only needs + // to know whether the stored setup version is current. Lazy-load the wizard + // barrel only when setup is stale or forced; otherwise skip it entirely. + const storedSetupVersion = settings.get("setupVersion"); + const setupWizard = + forceSetupWizard || storedSetupVersion < CURRENT_SETUP_VERSION ? await import("./modes/setup-wizard") : undefined; + const setupScenes = setupWizard + ? await setupWizard.selectSetupScenes(storedSetupVersion, setupWizard.ALL_SCENES, mode, { + resuming, + isTTY: process.stdin.isTTY && process.stdout.isTTY, + setupWizardEnabled: settings.get("startup.setupWizard"), + force: forceSetupWizard, + }) + : []; await mode.init({ suppressWelcomeIntro: resuming || setupScenes.length > 0 }); - if (setupScenes.length > 0) { - await runSetupWizard(mode, setupScenes); + if (setupWizard && setupScenes.length > 0) { + await setupWizard.runSetupWizard(mode, setupScenes); } versionCheckPromise @@ -716,7 +727,7 @@ async function buildSessionOptions( interface RunRootCommandDependencies { createAgentSession?: typeof createAgentSession; discoverAuthStorage?: typeof discoverAuthStorage; - runAcpMode?: typeof runAcpMode; + runAcpMode?: RunAcpMode; settings?: Settings; forceSetupWizard?: boolean; } @@ -940,33 +951,11 @@ export async function runRootCommand( await pluginPreloadPromise; - // Background marketplace auto-update — never blocks startup. - const autoUpdate = settingsInstance.get("marketplace.autoUpdate"); - if (autoUpdate !== "off") { - void (async () => { - try { - const mgr = new MarketplaceManager({ - marketplacesRegistryPath: getMarketplacesRegistryPath(), - installedRegistryPath: getInstalledPluginsRegistryPath(), - projectInstalledRegistryPath: (await resolveActiveProjectRegistryPath(getProjectDir())) ?? undefined, - marketplacesCacheDir: getMarketplacesCacheDir(), - pluginsCacheDir: getPluginsCacheDir(), - clearPluginRootsCache: clearPluginRootsAndCaches, - }); - await mgr.refreshStaleMarketplaces(); - const updates = await mgr.checkForUpdates(); - if (updates.length === 0) return; - if (autoUpdate === "auto") { - await mgr.upgradeAllPlugins(); - logger.debug(`Auto-upgraded ${updates.length} marketplace plugin(s)`); - } else { - logger.debug(`${updates.length} marketplace plugin update(s) available — /marketplace upgrade`); - } - } catch { - // Silently ignore — network failure, corrupt data, offline. - } - })(); - } + scheduleMarketplaceAutoUpdate({ + autoUpdate: settingsInstance.get("marketplace.autoUpdate"), + resolveActiveProjectRegistryPath, + clearPluginRootsCache: clearPluginRootsAndCaches, + }); const { options: sessionOptions } = await logger.time( "buildSessionOptions", @@ -1027,7 +1016,9 @@ export async function runRootCommand( rawArgs, createSession, }); - await (deps.runAcpMode ?? runAcpMode)(createAcpSession); + // Branch-only protocol runner: keep ACP server code out of normal interactive startup. + const runAcpMode = deps.runAcpMode ?? (await import("./modes/acp/acp-mode")).runAcpMode; + await runAcpMode(createAcpSession); } else { // Resolve extension-registered CLI flags before creating the session so a // bad `@file` fails fast WITHOUT leaving a junk session/breadcrumb @@ -1091,6 +1082,8 @@ export async function runRootCommand( } if (mode === "rpc" || mode === "rpc-ui") { + // Branch-only protocol runner: keep RPC host code out of normal interactive startup. + const runRpcMode: RunRpcMode = (await import("./modes/rpc/rpc-mode")).runRpcMode; await runRpcMode(session, mode === "rpc-ui" ? setToolUIContext : undefined); } else if (isInteractive) { const versionCheckPromise = checkForNewVersion(VERSION).catch(() => undefined); @@ -1109,7 +1102,7 @@ export async function runRootCommand( if ($env.PI_TIMING) { logger.printTimings(); - if ($env.PI_TIMING === "x") { + if (logger.shouldExitAfterTimings()) { process.exit(0); } } @@ -1132,6 +1125,8 @@ export async function runRootCommand( initialImages, ); } else { + // Branch-only single-shot runner: keep print-mode code out of normal interactive startup. + const runPrintMode: RunPrintMode = (await import("./modes/print-mode")).runPrintMode; await runPrintMode(session, { mode, messages: initialArgs.messages, diff --git a/packages/coding-agent/src/modes/index.ts b/packages/coding-agent/src/modes/index.ts index 9b2726d68..ac9f12896 100644 --- a/packages/coding-agent/src/modes/index.ts +++ b/packages/coding-agent/src/modes/index.ts @@ -2,11 +2,13 @@ import { emergencyTerminalRestore } from "@oh-my-pi/pi-tui"; import { postmortem } from "@oh-my-pi/pi-utils"; /** - * Run modes for the coding agent. + * Interactive mode and embeddable RPC client exports for the coding agent. + * + * Branch-specific runners live in their concrete modules so importing this + * barrel does not pull print, RPC server, or ACP server mode into the normal + * TUI graph. */ -export { runAcpMode } from "./acp"; export { InteractiveMode, type InteractiveModeOptions } from "./interactive-mode"; -export { type PrintModeOptions, runPrintMode } from "./print-mode"; export { defineRpcClientTool, type ModelInfo, @@ -17,7 +19,6 @@ export { type RpcClientToolResult, type RpcEventListener, } from "./rpc/rpc-client"; -export { runRpcMode } from "./rpc/rpc-mode"; export type { RpcCommand, RpcHostToolCallRequest, diff --git a/packages/coding-agent/src/modes/setup-version.ts b/packages/coding-agent/src/modes/setup-version.ts new file mode 100644 index 000000000..33bed5051 --- /dev/null +++ b/packages/coding-agent/src/modes/setup-version.ts @@ -0,0 +1,11 @@ +/** + * Setup version the wizard advances a fresh install to. Bump it whenever a new + * setup scene lands (or an existing scene raises its `minVersion`). + * + * Kept in its own dependency-free module so the cold-launch gate in `main.ts` + * can answer "is the stored setup version stale?" without statically importing + * the full wizard — every scene (sign-in/OAuth, web search, theme previews) plus + * the overlay component and their TUI deps. MUST equal `max(scene.minVersion)` + * across `ALL_SCENES`; the `setup-wizard` barrel and test suite guard it. + */ +export const CURRENT_SETUP_VERSION = 1; diff --git a/packages/coding-agent/src/modes/setup-wizard/index.ts b/packages/coding-agent/src/modes/setup-wizard/index.ts index 29da5d873..5e5eea61d 100644 --- a/packages/coding-agent/src/modes/setup-wizard/index.ts +++ b/packages/coding-agent/src/modes/setup-wizard/index.ts @@ -1,4 +1,5 @@ import type { Settings } from "../../config/settings"; +import { CURRENT_SETUP_VERSION } from "../setup-version"; import type { InteractiveModeContext } from "../types"; import { glyphSetupScene } from "./scenes/glyph"; import { providersSetupScene } from "./scenes/providers"; @@ -8,14 +9,14 @@ import { SetupWizardComponent } from "./wizard-overlay"; export type { SetupScene, SetupSceneController, SetupSceneHost, SetupSceneResult } from "./scenes/types"; +export { CURRENT_SETUP_VERSION }; + export const ALL_SCENES = [ providersSetupScene, glyphSetupScene, themeSetupScene, ] as const satisfies readonly SetupScene[]; -export const CURRENT_SETUP_VERSION = ALL_SCENES.reduce((max, scene) => Math.max(max, scene.minVersion), 0); - export interface SetupSceneSelectionOptions { resuming?: boolean; isTTY?: boolean; diff --git a/packages/coding-agent/test/setup-wizard.test.ts b/packages/coding-agent/test/setup-wizard.test.ts index 7f01b01fe..7be0f62f0 100644 --- a/packages/coding-agent/test/setup-wizard.test.ts +++ b/packages/coding-agent/test/setup-wizard.test.ts @@ -49,6 +49,15 @@ describe("setup wizard scene selection", () => { expect(scenes.map(scene => scene.id)).toEqual(ALL_SCENES.map(scene => scene.id)); }); + it("keeps CURRENT_SETUP_VERSION in sync with the highest scene minVersion", () => { + // main.ts's cold-launch gate sources CURRENT_SETUP_VERSION from the tiny + // `setup-version` module to decide whether to load the wizard at all. If a + // new scene raises the bar but the constant is not bumped, stale installs + // would never see the scene. Guard the invariant the gate relies on. + const highestMinVersion = Math.max(...ALL_SCENES.map(scene => scene.minVersion)); + expect(CURRENT_SETUP_VERSION).toBe(highestMinVersion); + }); + it("runs only scenes newer than the stored setup version", async () => { const scenes = [testScene("v1-a", 1), testScene("v1-b", 1), testScene("v2", 2)]; const selected = await selectSetupScenes(1, scenes, fakeContextWithConfiguredModel(), { isTTY: true }); diff --git a/packages/coding-agent/test/startup-import-graph.test.ts b/packages/coding-agent/test/startup-import-graph.test.ts new file mode 100644 index 000000000..2a5d6d6a6 --- /dev/null +++ b/packages/coding-agent/test/startup-import-graph.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from "bun:test"; +import * as path from "node:path"; + +const sourceRoot = path.join(import.meta.dir, "..", "src"); + +describe("startup import graph", () => { + it("keeps normal startup off the aggregate modes barrel", async () => { + const mainSource = await Bun.file(path.join(sourceRoot, "main.ts")).text(); + + expect(mainSource).toContain('import { InteractiveMode } from "./modes/interactive-mode";'); + expect(mainSource).not.toContain('from "./modes"'); + }); + + it("keeps branch-only mode runners out of the modes barrel", async () => { + const modesBarrelSource = await Bun.file(path.join(sourceRoot, "modes/index.ts")).text(); + + expect(modesBarrelSource).toContain('from "./interactive-mode"'); + expect(modesBarrelSource).not.toContain("runAcpMode"); + expect(modesBarrelSource).not.toContain("runPrintMode"); + expect(modesBarrelSource).not.toContain("runRpcMode"); + expect(modesBarrelSource).not.toContain("./rpc/rpc-mode"); + }); + + it("keeps marketplace implementation behind the lightweight auto-update starter", async () => { + const mainSource = await Bun.file(path.join(sourceRoot, "main.ts")).text(); + const starterSource = await Bun.file( + path.join(sourceRoot, "extensibility/plugins/marketplace-auto-update.ts"), + ).text(); + + expect(mainSource).toContain('from "./extensibility/plugins/marketplace-auto-update"'); + expect(mainSource).not.toContain('from "./extensibility/plugins/marketplace"'); + expect(starterSource).toContain('await import("./marketplace")'); + }); +}); From ba6cc67f38728b7c258130732d91404e03064bbf Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 059/181] refactor(packages/ai): migrated effort types to dependency-free module - Added dependency-free `@oh-my-pi/pi-ai/effort` module and re-exported `Effort`/`THINKING_EFFORTS`. - Moved `Effort` and `THINKING_EFFORTS` from `model-thinking.ts` to `effort.ts`, then migrated runtime/tests imports. --- packages/ai/CHANGELOG.md | 4 ++++ packages/ai/src/auth-gateway/server.ts | 2 +- packages/ai/src/auth-gateway/types.ts | 2 +- packages/ai/src/effort.ts | 16 ++++++++++++++++ packages/ai/src/index.ts | 1 + packages/ai/src/model-thinking.ts | 18 +----------------- packages/ai/src/provider-models/ollama.ts | 2 +- .../ai/src/provider-models/openai-compat.ts | 2 +- packages/ai/src/providers/amazon-bedrock.ts | 2 +- .../openai-codex/request-transformer.ts | 2 +- .../ai/src/providers/openai-completions.ts | 3 ++- packages/ai/src/stream.ts | 2 +- packages/ai/src/types.ts | 2 +- .../test/auth-gateway-openai-responses.test.ts | 2 +- .../ai/test/auth-gateway-pi-native.test.ts | 2 +- .../test/github-copilot-model-limits.test.ts | 2 +- .../ai/test/github-copilot-reasoning.test.ts | 2 +- packages/ai/test/issue-1373-repro.test.ts | 2 +- packages/ai/test/issue-826-repro.test.ts | 2 +- packages/ai/test/issue-969-repro.test.ts | 3 ++- packages/ai/test/model-thinking.test.ts | 2 +- packages/ai/test/nanogpt-model-limits.test.ts | 2 +- packages/ai/test/ollama-provider.test.ts | 2 +- ...penai-completions-disable-reasoning.test.ts | 2 +- 24 files changed, 44 insertions(+), 37 deletions(-) create mode 100644 packages/ai/src/effort.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index e3cd50803..db832fa7b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added a dependency-free `@oh-my-pi/pi-ai/effort` module exporting the `Effort` enum and `THINKING_EFFORTS`, split out of `model-thinking` so hot-path consumers can import the thinking levels without pulling in `model-thinking` and its provider-compat dependency graph. The package barrel still re-exports both names, so existing imports are unaffected. + ### Fixed - Fixed Antigravity usage provider emitting one bar per model instead of deduplicating by tier — a single account's 15+ model entries now collapse to one bar per tier, matching the shared-quota reality of the upstream API. diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index dad2aa394..f5b0d9383 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -19,7 +19,7 @@ */ import { extractRetryHint, logger } from "@oh-my-pi/pi-utils"; import type { AuthStorage } from "../auth-storage"; -import { Effort } from "../model-thinking"; +import { Effort } from "../effort"; import * as anthropicMessages from "../providers/anthropic-messages-server"; import * as openaiChat from "../providers/openai-chat-server"; import * as openaiResponses from "../providers/openai-responses-server"; diff --git a/packages/ai/src/auth-gateway/types.ts b/packages/ai/src/auth-gateway/types.ts index 0390759c0..bdb563e3b 100644 --- a/packages/ai/src/auth-gateway/types.ts +++ b/packages/ai/src/auth-gateway/types.ts @@ -1,4 +1,4 @@ -import type { Effort } from "../model-thinking"; +import type { Effort } from "../effort"; import type { AssistantMessage, AssistantMessageEventStream, diff --git a/packages/ai/src/effort.ts b/packages/ai/src/effort.ts new file mode 100644 index 000000000..831a13ede --- /dev/null +++ b/packages/ai/src/effort.ts @@ -0,0 +1,16 @@ +/** User-facing thinking levels, ordered least to most intensive. */ +export const enum Effort { + Minimal = "minimal", + Low = "low", + Medium = "medium", + High = "high", + XHigh = "xhigh", +} + +export const THINKING_EFFORTS: readonly Effort[] = [ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, +]; diff --git a/packages/ai/src/index.ts b/packages/ai/src/index.ts index ecdce180e..3b1e855a3 100644 --- a/packages/ai/src/index.ts +++ b/packages/ai/src/index.ts @@ -4,6 +4,7 @@ export * from "./auth-broker"; export { type AuthGatewayBootOptions, type ModelResolver, startAuthGateway } from "./auth-gateway/server"; export * from "./auth-gateway/types"; export * from "./auth-storage"; +export * from "./effort"; export * from "./model-cache"; export * from "./model-manager"; export * from "./model-thinking"; diff --git a/packages/ai/src/model-thinking.ts b/packages/ai/src/model-thinking.ts index 75e930315..7c963bdbe 100644 --- a/packages/ai/src/model-thinking.ts +++ b/packages/ai/src/model-thinking.ts @@ -1,23 +1,7 @@ +import { Effort, THINKING_EFFORTS } from "./effort"; import { resolveOpenAICompat } from "./providers/openai-completions-compat"; import type { Api, Model as ApiModel, ThinkingConfig } from "./types"; -/** User-facing thinking levels, ordered least to most intensive. */ -export const enum Effort { - Minimal = "minimal", - Low = "low", - Medium = "medium", - High = "high", - XHigh = "xhigh", -} - -export const THINKING_EFFORTS: readonly Effort[] = [ - Effort.Minimal, - Effort.Low, - Effort.Medium, - Effort.High, - Effort.XHigh, -]; - const DEFAULT_REASONING_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]; const DEFAULT_REASONING_EFFORTS_WITH_XHIGH: readonly Effort[] = [ Effort.Minimal, diff --git a/packages/ai/src/provider-models/ollama.ts b/packages/ai/src/provider-models/ollama.ts index c71fbf617..ed539d924 100644 --- a/packages/ai/src/provider-models/ollama.ts +++ b/packages/ai/src/provider-models/ollama.ts @@ -1,6 +1,6 @@ import { fetchWithRetry } from "@oh-my-pi/pi-utils"; +import { Effort } from "../effort"; import type { ModelManagerOptions } from "../model-manager"; -import { Effort } from "../model-thinking"; import type { ThinkingConfig } from "../types"; import { createBundledReferenceMap, createReferenceResolver } from "./bundled-references"; diff --git a/packages/ai/src/provider-models/openai-compat.ts b/packages/ai/src/provider-models/openai-compat.ts index d774bfd9f..545662fa4 100644 --- a/packages/ai/src/provider-models/openai-compat.ts +++ b/packages/ai/src/provider-models/openai-compat.ts @@ -1,5 +1,5 @@ +import { Effort } from "../effort"; import type { ModelManagerOptions } from "../model-manager"; -import { Effort } from "../model-thinking"; import { getBundledModels } from "../models"; import type { Api, Model, Provider, ThinkingConfig } from "../types"; import { isAnthropicOAuthToken, isRecord, toBoolean, toNumber, toPositiveNumber } from "../utils"; diff --git a/packages/ai/src/providers/amazon-bedrock.ts b/packages/ai/src/providers/amazon-bedrock.ts index 7f161f32c..49b1523aa 100644 --- a/packages/ai/src/providers/amazon-bedrock.ts +++ b/packages/ai/src/providers/amazon-bedrock.ts @@ -8,7 +8,7 @@ */ import { $env, $flag, extractHttpStatusFromError, fetchWithRetry } from "@oh-my-pi/pi-utils"; -import type { Effort } from "../model-thinking"; +import type { Effort } from "../effort"; import { mapEffortToAnthropicAdaptiveEffort, requireSupportedEffort } from "../model-thinking"; import { calculateCost } from "../models"; import type { diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index a12996ca6..91abe9dc5 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -1,4 +1,4 @@ -import type { Effort } from "../../model-thinking"; +import type { Effort } from "../../effort"; import { requireSupportedEffort } from "../../model-thinking"; import type { Api, Model } from "../../types"; diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index b8b5e591c..67cae1e4a 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -10,7 +10,8 @@ import type { ChatCompletionToolMessageParam, } from "openai/resources/chat/completions"; import packageJson from "../../package.json" with { type: "json" }; -import { type Effort, getSupportedEfforts } from "../model-thinking"; +import type { Effort } from "../effort"; +import { getSupportedEfforts } from "../model-thinking"; import { calculateCost } from "../models"; import { getEnvApiKey } from "../stream"; import { diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 437a49344..a7d5fc739 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -3,7 +3,7 @@ import * as os from "node:os"; import * as path from "node:path"; import { $env, $pickenv, extractHttpStatusFromError } from "@oh-my-pi/pi-utils"; import { getCustomApi } from "./api-registry"; -import type { Effort } from "./model-thinking"; +import type { Effort } from "./effort"; import { mapEffortToAnthropicAdaptiveEffort, mapEffortToGoogleThinkingLevel, diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 9b03d99d8..4cf00265f 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -151,7 +151,7 @@ export type KnownProvider = | "lm-studio"; export type Provider = KnownProvider | string; -import type { Effort } from "./model-thinking"; +import type { Effort } from "./effort"; /** Token budgets for each thinking level (token-based providers only) */ export type ThinkingBudgets = { [key in Effort]?: number }; diff --git a/packages/ai/test/auth-gateway-openai-responses.test.ts b/packages/ai/test/auth-gateway-openai-responses.test.ts index fe3a21801..c6cbd8c9b 100644 --- a/packages/ai/test/auth-gateway-openai-responses.test.ts +++ b/packages/ai/test/auth-gateway-openai-responses.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { encodeResponse, encodeStream, parseRequest } from "../src/providers/openai-responses-server"; import type { AssistantMessage } from "../src/types"; import { AssistantMessageEventStream } from "../src/utils/event-stream"; diff --git a/packages/ai/test/auth-gateway-pi-native.test.ts b/packages/ai/test/auth-gateway-pi-native.test.ts index 7c10a65a7..5f3a77a76 100644 --- a/packages/ai/test/auth-gateway-pi-native.test.ts +++ b/packages/ai/test/auth-gateway-pi-native.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { encodeStream, formatError, parseRequest } from "../src/providers/pi-native-server"; import type { AssistantMessage, diff --git a/packages/ai/test/github-copilot-model-limits.test.ts b/packages/ai/test/github-copilot-model-limits.test.ts index c46f14487..07d76ee87 100644 --- a/packages/ai/test/github-copilot-model-limits.test.ts +++ b/packages/ai/test/github-copilot-model-limits.test.ts @@ -2,8 +2,8 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; +import { Effort } from "../src/effort"; import { createModelManager } from "../src/model-manager"; -import { Effort } from "../src/model-thinking"; import { getBundledModel } from "../src/models"; import { githubCopilotModelManagerOptions } from "../src/provider-models/openai-compat"; diff --git a/packages/ai/test/github-copilot-reasoning.test.ts b/packages/ai/test/github-copilot-reasoning.test.ts index 32a88c006..28746c85d 100644 --- a/packages/ai/test/github-copilot-reasoning.test.ts +++ b/packages/ai/test/github-copilot-reasoning.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { getBundledModel } from "../src/models"; import { streamAnthropic } from "../src/providers/anthropic"; import { streamOpenAIResponses } from "../src/providers/openai-responses"; diff --git a/packages/ai/test/issue-1373-repro.test.ts b/packages/ai/test/issue-1373-repro.test.ts index 986acbf3c..88e855f4e 100644 --- a/packages/ai/test/issue-1373-repro.test.ts +++ b/packages/ai/test/issue-1373-repro.test.ts @@ -1,5 +1,5 @@ import { afterAll, beforeAll, describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamBedrock } from "../src/providers/amazon-bedrock"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/issue-826-repro.test.ts b/packages/ai/test/issue-826-repro.test.ts index efcf9ccea..dd78aeda0 100644 --- a/packages/ai/test/issue-826-repro.test.ts +++ b/packages/ai/test/issue-826-repro.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamAnthropic } from "../src/providers/anthropic"; import type { Context, Model, Tool } from "../src/types"; diff --git a/packages/ai/test/issue-969-repro.test.ts b/packages/ai/test/issue-969-repro.test.ts index 1b347470e..9f42a85bb 100644 --- a/packages/ai/test/issue-969-repro.test.ts +++ b/packages/ai/test/issue-969-repro.test.ts @@ -1,5 +1,6 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { Effort, getSupportedEfforts } from "../src/model-thinking"; +import { Effort } from "../src/effort"; +import { getSupportedEfforts } from "../src/model-thinking"; import { streamOpenAICompletions } from "../src/providers/openai-completions"; import type { Context, Model } from "../src/types"; diff --git a/packages/ai/test/model-thinking.test.ts b/packages/ai/test/model-thinking.test.ts index 89b90da8e..e8475974c 100644 --- a/packages/ai/test/model-thinking.test.ts +++ b/packages/ai/test/model-thinking.test.ts @@ -1,8 +1,8 @@ import { describe, expect, it } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-ai/effort"; import { applyGeneratedModelPolicies, clampThinkingLevelForModel, - Effort, enrichModelThinking, linkOpenAIPromotionTargets, mapEffortToAnthropicAdaptiveEffort, diff --git a/packages/ai/test/nanogpt-model-limits.test.ts b/packages/ai/test/nanogpt-model-limits.test.ts index 03d7b8bab..0270b18ff 100644 --- a/packages/ai/test/nanogpt-model-limits.test.ts +++ b/packages/ai/test/nanogpt-model-limits.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { nanoGptModelManagerOptions } from "../src/provider-models/openai-compat"; const originalFetch = global.fetch; diff --git a/packages/ai/test/ollama-provider.test.ts b/packages/ai/test/ollama-provider.test.ts index 1a2e2aadb..4fa4866c2 100644 --- a/packages/ai/test/ollama-provider.test.ts +++ b/packages/ai/test/ollama-provider.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, test, vi } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { ollamaModelManagerOptions } from "../src/provider-models/openai-compat"; import { streamOllama } from "../src/providers/ollama"; import type { Context, Model, Tool } from "../src/types"; diff --git a/packages/ai/test/openai-completions-disable-reasoning.test.ts b/packages/ai/test/openai-completions-disable-reasoning.test.ts index f2d9107f6..629a74887 100644 --- a/packages/ai/test/openai-completions-disable-reasoning.test.ts +++ b/packages/ai/test/openai-completions-disable-reasoning.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it } from "bun:test"; -import { Effort } from "../src/model-thinking"; +import { Effort } from "../src/effort"; import { streamOpenAICompletions } from "../src/providers/openai-completions"; import type { Context, Model } from "../src/types"; From 76f08dd7d3d3537238c4e5cf7828987182224368 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 22:58:33 +0200 Subject: [PATCH 060/181] refactor(packages/coding-agent): migrated imports to pi-ai submodules - Migrated Effort and THINKING_EFFORTS imports to @oh-my-pi/pi-ai/effort in CLI args and launch command files. - Split model-registry dependencies across focused @oh-my-pi/pi-ai submodules instead of the root barrel export. --- packages/coding-agent/src/cli/args.ts | 2 +- packages/coding-agent/src/commands/launch.ts | 2 +- .../coding-agent/src/config/model-registry.ts | 24 +++++++------------ 3 files changed, 10 insertions(+), 18 deletions(-) diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 0f770c0ce..707d87c54 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -1,7 +1,7 @@ /** * CLI argument parsing and help display */ -import { type Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-ai"; +import { type Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-ai/effort"; import { APP_NAME, CONFIG_DIR_NAME, logger } from "@oh-my-pi/pi-utils"; import chalk from "chalk"; import { parseEffort } from "../thinking"; diff --git a/packages/coding-agent/src/commands/launch.ts b/packages/coding-agent/src/commands/launch.ts index a752cdf82..d26dc543a 100644 --- a/packages/coding-agent/src/commands/launch.ts +++ b/packages/coding-agent/src/commands/launch.ts @@ -2,7 +2,7 @@ * Root command for the coding agent CLI. */ -import { THINKING_EFFORTS } from "@oh-my-pi/pi-ai"; +import { THINKING_EFFORTS } from "@oh-my-pi/pi-ai/effort"; import { APP_NAME } from "@oh-my-pi/pi-utils"; import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli"; import { parseArgs } from "../cli/args"; diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index ac83a06ce..eb35413e1 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -1,27 +1,19 @@ import * as path from "node:path"; +import { registerCustomApi, unregisterCustomApis } from "@oh-my-pi/pi-ai/api-registry"; +import { readModelCache } from "@oh-my-pi/pi-ai/model-cache"; +import { createModelManager, type ModelManagerOptions, type ModelRefreshStrategy } from "@oh-my-pi/pi-ai/model-manager"; +import { enrichModelThinking } from "@oh-my-pi/pi-ai/model-thinking"; +import { getBundledModels, getBundledProviders } from "@oh-my-pi/pi-ai/models"; import { - type Api, - type AssistantMessageEventStream, - type Context, - createModelManager, - enrichModelThinking, - getBundledModels, - getBundledProviders, googleAntigravityModelManagerOptions, googleGeminiCliModelManagerOptions, - type Model, - type ModelManagerOptions, - type ModelRefreshStrategy, openaiCodexModelManagerOptions, PROVIDER_DESCRIPTORS, - readModelCache, - registerCustomApi, - type SimpleStreamOptions, - type ThinkingConfig, UNK_CONTEXT_WINDOW, UNK_MAX_TOKENS, - unregisterCustomApis, -} from "@oh-my-pi/pi-ai"; +} from "@oh-my-pi/pi-ai/provider-models"; +import type { Api, Context, Model, SimpleStreamOptions, ThinkingConfig } from "@oh-my-pi/pi-ai/types"; +import type { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; // Sentinel for local-only OAuth token (LM Studio, vLLM) — declared inline to avoid loading // any provider module at startup. Must match `DEFAULT_LOCAL_TOKEN` in oauth/lm-studio.ts. From 5ec6c0e7a7c37e586a5949780cba04ceaa01af17 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 23:07:57 +0200 Subject: [PATCH 061/181] fix(coding-agent): removed redundant official-id canonical shortcut - Let heuristic candidate matching handle official ids uniformly. --- packages/coding-agent/src/config/model-equivalence.ts | 3 --- 1 file changed, 3 deletions(-) diff --git a/packages/coding-agent/src/config/model-equivalence.ts b/packages/coding-agent/src/config/model-equivalence.ts index 8115f7d87..dd8185ab4 100644 --- a/packages/coding-agent/src/config/model-equivalence.ts +++ b/packages/coding-agent/src/config/model-equivalence.ts @@ -771,9 +771,6 @@ function resolveCanonicalIdForModel( return { id: claudeFamilyAlias, source: claudeFamilyAlias === model.id ? "bundled" : "heuristic" }; } - if (referenceData.officialIds.has(model.id) && !model.id.includes("/") && !model.id.includes(":")) { - return { id: model.id, source: "bundled" }; - } const heuristicCandidates = getHeuristicCanonicalCandidates(model.id, referenceData.officialIds); const officialMatches = new Set(); From f552ce4e6d098572bbfe4261a1f2f36566ca7d90 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 23:25:18 +0200 Subject: [PATCH 062/181] perf(coding-agent): deferred heavy module imports to startup paths - Lazy-loaded OTEL SDK, HTML export, TTSR, and autoresearch modules. - Made resolveMemoryBackend async to import backends on demand. - Replaced backend resolution with direct settings reads for rekey checks. --- .../src/config/model-equivalence.ts | 1 - packages/coding-agent/src/main.ts | 4 +- .../coding-agent/src/memory-backend/index.ts | 14 ++++++- .../src/memory-backend/resolve.ts | 8 ++-- .../coding-agent/src/memory-backend/types.ts | 2 +- .../modes/controllers/command-controller.ts | 4 +- .../modes/controllers/selector-controller.ts | 4 +- .../src/modes/interactive-mode.ts | 4 +- packages/coding-agent/src/modes/types.ts | 2 +- packages/coding-agent/src/sdk.ts | 42 +++++++++---------- .../coding-agent/src/session/agent-session.ts | 14 +++---- .../src/slash-commands/builtin-registry.ts | 2 +- packages/coding-agent/src/telemetry-export.ts | 32 ++++++++++---- .../test/memory-backend-resolve.test.ts | 6 +-- .../coding-agent/test/otel-export-probe.ts | 2 +- .../test/telemetry-export.test.ts | 24 +++++------ 16 files changed, 95 insertions(+), 70 deletions(-) diff --git a/packages/coding-agent/src/config/model-equivalence.ts b/packages/coding-agent/src/config/model-equivalence.ts index dd8185ab4..75fedfc2b 100644 --- a/packages/coding-agent/src/config/model-equivalence.ts +++ b/packages/coding-agent/src/config/model-equivalence.ts @@ -771,7 +771,6 @@ function resolveCanonicalIdForModel( return { id: claudeFamilyAlias, source: claudeFamilyAlias === model.id ? "bundled" : "heuristic" }; } - const heuristicCandidates = getHeuristicCanonicalCandidates(model.id, referenceData.officialIds); const officialMatches = new Set(); for (const candidate of heuristicCandidates) { diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index fa4c24011..a0f7fa1a0 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -40,7 +40,6 @@ import { resolveActiveProjectRegistryPath, } from "./discovery/helpers"; import { injectOmpExtensionCliRoots } from "./discovery/omp-extension-roots"; -import { exportFromFile } from "./export/html"; import { ExtensionRunner } from "./extensibility/extensions/runner"; import type { ExtensionUIContext } from "./extensibility/extensions/types"; import { scheduleMarketplaceAutoUpdate } from "./extensibility/plugins/marketplace-auto-update"; @@ -784,6 +783,7 @@ export async function runRootCommand( let result: string; try { const outputPath = parsedArgs.messages.length > 0 ? parsedArgs.messages[0] : undefined; + const { exportFromFile } = await import("./export/html"); result = await exportFromFile(parsedArgs.export, outputPath); } catch (error: unknown) { const message = error instanceof Error ? error.message : "Failed to export session"; @@ -977,7 +977,7 @@ export async function runRootCommand( // Both are no-ops when OTEL_EXPORTER_OTLP_ENDPOINT is unset. An empty config // is enough to enable telemetry — content capture is governed by the // standard OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT env var. - initTelemetryExport(); + await initTelemetryExport(); if (isTelemetryExportEnabled()) { sessionOptions.telemetry = {}; } diff --git a/packages/coding-agent/src/memory-backend/index.ts b/packages/coding-agent/src/memory-backend/index.ts index 2c2c678a3..62f504ee4 100644 --- a/packages/coding-agent/src/memory-backend/index.ts +++ b/packages/coding-agent/src/memory-backend/index.ts @@ -1,4 +1,16 @@ -export * from "../mnemopi"; +export type { + MnemopiBackendConfig, + MnemopiLlmMode, + MnemopiProviderOptions, + MnemopiScoping, +} from "../mnemopi/config"; +export type { + MnemopiMemoryEditOperation, + MnemopiMemoryEditOptions, + MnemopiMemoryEditResult, + MnemopiSessionState, + MnemopiSessionStateOptions, +} from "../mnemopi/state"; export * from "./local-backend"; export * from "./off-backend"; export * from "./resolve"; diff --git a/packages/coding-agent/src/memory-backend/resolve.ts b/packages/coding-agent/src/memory-backend/resolve.ts index a0066d7f9..638aabec1 100644 --- a/packages/coding-agent/src/memory-backend/resolve.ts +++ b/packages/coding-agent/src/memory-backend/resolve.ts @@ -1,6 +1,4 @@ import type { Settings } from "../config/settings"; -import { hindsightBackend } from "../hindsight"; -import { mnemopiBackend } from "../mnemopi"; import { localBackend } from "./local-backend"; import { offBackend } from "./off-backend"; import type { MemoryBackend } from "./types"; @@ -18,10 +16,10 @@ import type { MemoryBackend } from "./types"; * `memories.enabled` remains accepted only as a legacy migration input. Once * a config is loaded, `memory.backend` is the sole runtime selector. */ -export function resolveMemoryBackend(settings: Settings): MemoryBackend { +export async function resolveMemoryBackend(settings: Settings): Promise { const id = settings.get("memory.backend"); - if (id === "hindsight") return hindsightBackend; - if (id === "mnemopi") return mnemopiBackend; + if (id === "hindsight") return (await import("../hindsight/backend")).hindsightBackend; + if (id === "mnemopi") return (await import("../mnemopi/backend")).mnemopiBackend; if (id === "local") return localBackend; return offBackend; } diff --git a/packages/coding-agent/src/memory-backend/types.ts b/packages/coding-agent/src/memory-backend/types.ts index 3d72976ea..8b2e5cd15 100644 --- a/packages/coding-agent/src/memory-backend/types.ts +++ b/packages/coding-agent/src/memory-backend/types.ts @@ -1,7 +1,7 @@ /** * Memory backend abstraction. * - * Backends are mutually exclusive — `resolveMemoryBackend(settings)` returns + * Backends are mutually exclusive — `await resolveMemoryBackend(settings)` resolves * exactly one. Implementations MUST be self-contained: they own the per-session * state they create in `start()` and tear it down on `clear()`. */ diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index a64a254f1..23462c9a2 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -13,7 +13,6 @@ import { Loader, Markdown, padding, Spacer, Text, visibleWidth } from "@oh-my-pi import { formatDuration, Snowflake } from "@oh-my-pi/pi-utils"; import { $ } from "bun"; import { shouldEnableAppendOnlyContext } from "../../config/append-only-context-mode"; -import { loadCustomShare } from "../../export/custom-share"; import type { CompactOptions } from "../../extensibility/extensions/types"; import { diffMentalModelContent, @@ -131,6 +130,7 @@ export class CommandController { } try { + const { loadCustomShare } = await import("../../export/custom-share"); const customShare = await loadCustomShare(); if (customShare) { const loader = new BorderedLoader(this.ctx.ui, theme, "Sharing..."); @@ -465,7 +465,7 @@ export class CommandController { const argumentText = text.slice(7).trim(); const action = argumentText.split(/\s+/, 1)[0]?.toLowerCase() || "view"; const agentDir = this.ctx.settings.getAgentDir(); - const backend = resolveMemoryBackend(this.ctx.settings); + const backend = await resolveMemoryBackend(this.ctx.settings); if (action === "view") { const payload = await backend.buildDeveloperInstructions(agentDir, this.ctx.settings, this.ctx.session); diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index df8ae8d1b..b6c7cfffe 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -7,7 +7,6 @@ import { getAgentDbPath, getProjectDir, normalizePathForComparison } from "@oh-m import { getRoleInfo } from "../../config/model-registry"; import { formatModelSelectorValue } from "../../config/model-resolver"; import { settings } from "../../config/settings"; -import { DebugSelectorComponent } from "../../debug"; import { disableProvider, enableProvider } from "../../discovery"; import { clearPluginRootsAndCaches, resolveActiveProjectRegistryPath } from "../../discovery/helpers"; import { @@ -1080,7 +1079,8 @@ export class SelectorController { }); } - showDebugSelector(): void { + async showDebugSelector(): Promise { + const { DebugSelectorComponent } = await import("../../debug"); this.showSelector(done => { const selector = new DebugSelectorComponent(this.ctx, done); return { component: selector, focus: selector }; diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 009ef43e1..1166097df 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -2844,8 +2844,8 @@ export class InteractiveMode implements InteractiveModeContext { } } - showDebugSelector(): void { - this.#selectorController.showDebugSelector(); + async showDebugSelector(): Promise { + await this.#selectorController.showDebugSelector(); } showSessionObserver(): void { diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 2b38238a4..de2cfa880 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -269,7 +269,7 @@ export interface InteractiveModeContext { handleSessionDeleteCommand(): Promise; showOAuthSelector(mode: "login" | "logout", providerId?: string): Promise; showHookConfirm(title: string, message: string): Promise; - showDebugSelector(): void; + showDebugSelector(): Promise; showSessionObserver(): void; resetObserverRegistry(): void; diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 70bbcb58c..46ad4553e 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -36,7 +36,6 @@ import { } from "@oh-my-pi/pi-utils"; import chalk from "chalk"; import { type AsyncJob, AsyncJobManager, isBackgroundJobSupportEnabled } from "./async"; -import { createAutoresearchExtension } from "./autoresearch"; import { loadCapability } from "./capability"; import { type Rule, ruleCapability, setActiveRules } from "./capability/rule"; import { bucketRules } from "./capability/rule-buckets"; @@ -57,7 +56,6 @@ import { resolveConfigValue } from "./config/resolve-config-value"; import { initializeWithSettings } from "./discovery"; import { disposeAllKernelSessions, disposeKernelSessionsByOwner } from "./eval/py/executor"; import { defaultEvalSessionId } from "./eval/session-id"; -import { TtsrManager } from "./export/ttsr"; import { type CustomCommandsLoadResult, type LoadedCustomCommand, @@ -90,7 +88,7 @@ import { LocalProtocolHandler, type LocalProtocolOptions } from "./internal-urls import { LSP_STARTUP_EVENT_CHANNEL, type LspStartupEvent } from "./lsp/startup-events"; import { discoverAndLoadMCPTools, MCPManager, type MCPToolsLoadResult } from "./mcp"; import { resolveMemoryBackend } from "./memory-backend"; -import { getMnemopiSessionState, type MnemopiSessionState } from "./mnemopi/state"; +import type { MnemopiSessionState } from "./mnemopi/state"; import asyncResultTemplate from "./prompts/tools/async-result.md" with { type: "text" }; import { AgentRegistry, MAIN_AGENT_ID } from "./registry/agent-registry"; import { @@ -1150,6 +1148,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // Discover rules and bucket them in one pass to avoid repeated scans over large rule sets. const { ttsrManager, rulebookRules, alwaysApplyRules } = await logger.time("discoverTtsrRules", async () => { + const { TtsrManager } = await import("./export/ttsr"); const ttsrSettings = settings.getGroup("ttsr"); const ttsrManager = new TtsrManager(ttsrSettings); const rulesResult = @@ -1295,7 +1294,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} session ? session.trackEvalExecution(execution, abortController) : execution, getSessionId: () => sessionManager.getSessionId?.() ?? null, getHindsightSessionState: () => session?.getHindsightSessionState(), - getMnemopiSessionState: () => getMnemopiSessionState(session), + getMnemopiSessionState: () => session?.getMnemopiSessionState(), getAgentId: () => resolvedAgentId, getToolByName: name => session?.getToolByName(name), agentRegistry, @@ -1472,7 +1471,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } const inlineExtensions: ExtensionFactory[] = options.extensions ? [...options.extensions] : []; - inlineExtensions.push(createAutoresearchExtension); + inlineExtensions.push((await import("./autoresearch")).createAutoresearchExtension); if (customTools.length > 0) { inlineExtensions.push(createCustomToolsExtension(customTools)); } @@ -1607,9 +1606,9 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // `ExtensionToolWrapper` installed below is the only place the per-tool approval gate runs. // A conditional runner means the approval system silently disappears for users with no // extensions, contradicting non-yolo `tools.approvalMode` settings without feedback. - // (Today `createAutoresearchExtension` is unconditionally pushed below, so this scenario - // is unreachable; the unconditional construction makes that invariant explicit instead of - // implicit, so a future change to make autoresearch optional cannot silently re-open the hole.) + // (The builtin autoresearch extension is unconditionally loaded above, so this scenario + // is unreachable; unconditional runner construction keeps that invariant explicit and + // prevents future optional extensions from silently re-opening the hole.) const extensionRunner: ExtensionRunner = new ExtensionRunner( extensionsResult.extensions, extensionsResult.runtime, @@ -1749,7 +1748,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} const promptTools = buildSystemPromptToolMetadata(tools, { search_tool_bm25: { description: renderSearchToolBm25Description(discoverableToolsForDesc) }, }); - const memoryBackend = resolveMemoryBackend(settings); + const memoryBackend = await resolveMemoryBackend(settings); const memoryInstructions = await memoryBackend.buildDeveloperInstructions(agentDir, settings, session); // Build combined append prompt: memory instructions + MCP server instructions @@ -2267,19 +2266,18 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} } } - logger.time("startMemoryStartupTask", () => - Promise.resolve( - resolveMemoryBackend(settings).start({ - session, - settings, - modelRegistry, - agentDir, - taskDepth, - parentHindsightSessionState: options.parentHindsightSessionState, - parentMnemopiSessionState: options.parentMnemopiSessionState, - }), - ), - ); + logger.time("startMemoryStartupTask", async () => { + const memoryBackend = await resolveMemoryBackend(settings); + await memoryBackend.start({ + session, + settings, + modelRegistry, + agentDir, + taskDepth, + parentHindsightSessionState: options.parentHindsightSessionState, + parentMnemopiSessionState: options.parentMnemopiSessionState, + }); + }); // Wire MCP manager callbacks to session for reactive tool updates. // Skip when reusing a parent's manager — the parent owns the callbacks. diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index eca983671..746ef86e0 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -128,7 +128,6 @@ import { } from "../eval/py/executor"; import { defaultEvalSessionId } from "../eval/session-id"; import { type BashResult, executeBash as executeBashCommand } from "../exec/bash-executor"; -import { exportSessionToHtml } from "../export/html"; import type { TtsrManager, TtsrMatchContext } from "../export/ttsr"; import type { LoadedCustomCommand } from "../extensibility/custom-commands"; import type { CustomTool, CustomToolContext } from "../extensibility/custom-tools/types"; @@ -2967,14 +2966,14 @@ export class AgentSession { } #rekeyHindsightMemoryForCurrentSessionId(): void { - if (resolveMemoryBackend(this.settings).id !== "hindsight") return; + if (this.settings.get("memory.backend") !== "hindsight") return; const sid = this.agent.sessionId; if (!sid) return; this.getHindsightSessionState()?.setSessionId(sid); } #rekeyMnemopiMemoryForCurrentSessionId(): void { - if (resolveMemoryBackend(this.settings).id !== "mnemopi") return; + if (this.settings.get("memory.backend") !== "mnemopi") return; const sid = this.agent.sessionId; if (!sid) return; this.getMnemopiSessionState()?.setSessionId(sid); @@ -2982,14 +2981,14 @@ export class AgentSession { /** New session file: reset auto-recall / retain-threshold counters for the new transcript. */ #resetHindsightConversationTrackingIfHindsight(): void { - if (resolveMemoryBackend(this.settings).id !== "hindsight") return; + if (this.settings.get("memory.backend") !== "hindsight") return; const state = this.getHindsightSessionState(); if (!state || state.aliasOf) return; state.resetConversationTracking(); } #resetMnemopiConversationTrackingIfMnemopi(): void { - if (resolveMemoryBackend(this.settings).id !== "mnemopi") return; + if (this.settings.get("memory.backend") !== "mnemopi") return; const state = this.getMnemopiSessionState(); if (!state || state.aliasOf) return; state.resetConversationTracking(); @@ -3670,7 +3669,7 @@ export class AgentSession { } async #buildSystemPromptForAgentStart(promptText: string): Promise { - const backend = resolveMemoryBackend(this.settings); + const backend = await resolveMemoryBackend(this.settings); if (!backend.beforeAgentStartPrompt) return this.#baseSystemPrompt; try { @@ -6096,7 +6095,7 @@ export class AgentSession { messagesToSummarize: AgentMessage[]; turnPrefixMessages: AgentMessage[]; }): Promise { - const backend = resolveMemoryBackend(this.settings); + const backend = await resolveMemoryBackend(this.settings); if (!backend.preCompactionContext) return undefined; const messages = preparation.messagesToSummarize.concat(preparation.turnPrefixMessages); try { @@ -9588,6 +9587,7 @@ export class AgentSession { */ async exportToHtml(outputPath?: string): Promise { const themeName = getCurrentThemeName(); + const { exportSessionToHtml } = await import("../export/html"); return exportSessionToHtml(this.sessionManager, this.state, { outputPath, themeName }); } diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 394491f8b..bfb059a70 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -934,7 +934,7 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ allowArgs: true, handle: async (command, runtime) => { const verb = (command.args.trim().split(/\s+/)[0] ?? "").toLowerCase() || "view"; - const backend = resolveMemoryBackend(runtime.settings); + const backend = await resolveMemoryBackend(runtime.settings); switch (verb) { case "view": { const payload = await backend.buildDeveloperInstructions( diff --git a/packages/coding-agent/src/telemetry-export.ts b/packages/coding-agent/src/telemetry-export.ts index fbb5b3a62..234d43cf3 100644 --- a/packages/coding-agent/src/telemetry-export.ts +++ b/packages/coding-agent/src/telemetry-export.ts @@ -23,11 +23,7 @@ * `sdk-trace-base@2.7` exports cleanly on Bun. */ import { logger, postmortem } from "@oh-my-pi/pi-utils"; -import { AsyncLocalStorageContextManager } from "@opentelemetry/context-async-hooks"; -import { OTLPTraceExporter } from "@opentelemetry/exporter-trace-otlp-proto"; -import { resourceFromAttributes } from "@opentelemetry/resources"; -import { BatchSpanProcessor } from "@opentelemetry/sdk-trace-base"; -import { NodeTracerProvider } from "@opentelemetry/sdk-trace-node"; +import type * as TraceNode from "@opentelemetry/sdk-trace-node"; /** * Periodic flush interval. A long-lived `omp` process (the ACP server is @@ -36,7 +32,8 @@ import { NodeTracerProvider } from "@opentelemetry/sdk-trace-node"; */ const FLUSH_INTERVAL_MS = 30_000; -let provider: NodeTracerProvider | undefined; +let provider: TraceNode.NodeTracerProvider | undefined; +let initPromise: Promise | undefined; /** * Whether {@link initTelemetryExport} registered a real provider. The CLI uses @@ -53,8 +50,10 @@ export function isTelemetryExportEnabled(): boolean { * the OTEL kill-switches are engaged), so it is safe to call unconditionally at * startup. */ -export function initTelemetryExport(): void { +export async function initTelemetryExport(): Promise { if (provider) return; + if (initPromise) return initPromise; + // The OTEL env contract parses booleans and enum lists case-insensitively, so // OTEL_SDK_DISABLED=TRUE and OTEL_TRACES_EXPORTER=None must also disable export. if (process.env.OTEL_SDK_DISABLED?.trim().toLowerCase() === "true") return; @@ -77,6 +76,25 @@ export function initTelemetryExport(): void { return; } + initPromise = registerProvider(); + return initPromise; +} + +async function registerProvider(): Promise { + const [ + { AsyncLocalStorageContextManager }, + { OTLPTraceExporter }, + { resourceFromAttributes }, + { BatchSpanProcessor }, + { NodeTracerProvider }, + ] = await Promise.all([ + import("@opentelemetry/context-async-hooks"), + import("@opentelemetry/exporter-trace-otlp-proto"), + import("@opentelemetry/resources"), + import("@opentelemetry/sdk-trace-base"), + import("@opentelemetry/sdk-trace-node"), + ]); + // The exporter reads endpoint/headers/timeout from OTEL_EXPORTER_OTLP_* itself, // so there is nothing to thread through here. const exporter = new OTLPTraceExporter(); diff --git a/packages/coding-agent/test/memory-backend-resolve.test.ts b/packages/coding-agent/test/memory-backend-resolve.test.ts index 075f05e8f..46845e103 100644 --- a/packages/coding-agent/test/memory-backend-resolve.test.ts +++ b/packages/coding-agent/test/memory-backend-resolve.test.ts @@ -11,10 +11,10 @@ describe("resolveMemoryBackend", () => { resetSettingsForTest(); }); - it("returns the hindsight backend when memory.backend is hindsight, regardless of legacy memories.enabled", () => { + it("returns the hindsight backend when memory.backend is hindsight, regardless of legacy memories.enabled", async () => { const a = Settings.isolated({ "memory.backend": "hindsight", "memories.enabled": false }); const b = Settings.isolated({ "memory.backend": "hindsight", "memories.enabled": true }); - expect(resolveMemoryBackend(a).id).toBe("hindsight"); - expect(resolveMemoryBackend(b).id).toBe("hindsight"); + expect((await resolveMemoryBackend(a)).id).toBe("hindsight"); + expect((await resolveMemoryBackend(b)).id).toBe("hindsight"); }); }); diff --git a/packages/coding-agent/test/otel-export-probe.ts b/packages/coding-agent/test/otel-export-probe.ts index 2ada60550..2b285bcf8 100644 --- a/packages/coding-agent/test/otel-export-probe.ts +++ b/packages/coding-agent/test/otel-export-probe.ts @@ -35,7 +35,7 @@ const server = Bun.serve({ process.env.OTEL_EXPORTER_OTLP_TRACES_ENDPOINT = `http://localhost:${server.port}/v1/traces`; process.env.OTEL_SERVICE_NAME = "oh-my-pi-export-probe"; -initTelemetryExport(); +await initTelemetryExport(); if (!isTelemetryExportEnabled()) { console.error("PROBE: provider did not register"); await server.stop(true); diff --git a/packages/coding-agent/test/telemetry-export.test.ts b/packages/coding-agent/test/telemetry-export.test.ts index ecdaf86e8..fa0238b04 100644 --- a/packages/coding-agent/test/telemetry-export.test.ts +++ b/packages/coding-agent/test/telemetry-export.test.ts @@ -33,45 +33,45 @@ afterEach(() => { }); describe("initTelemetryExport gating", () => { - it("stays disabled when no OTLP endpoint is configured", () => { - initTelemetryExport(); + it("stays disabled when no OTLP endpoint is configured", async () => { + await initTelemetryExport(); expect(isTelemetryExportEnabled()).toBe(false); }); - it("stays disabled when OTEL_SDK_DISABLED=true even with an endpoint", () => { + it("stays disabled when OTEL_SDK_DISABLED=true even with an endpoint", async () => { process.env.OTEL_EXPORTER_OTLP_ENDPOINT = "http://localhost:4318"; process.env.OTEL_SDK_DISABLED = "true"; - initTelemetryExport(); + await initTelemetryExport(); expect(isTelemetryExportEnabled()).toBe(false); }); - it("stays disabled when OTEL_TRACES_EXPORTER=none even with an endpoint", () => { + it("stays disabled when OTEL_TRACES_EXPORTER=none even with an endpoint", async () => { process.env.OTEL_EXPORTER_OTLP_ENDPOINT = "http://localhost:4318"; process.env.OTEL_TRACES_EXPORTER = "none"; - initTelemetryExport(); + await initTelemetryExport(); expect(isTelemetryExportEnabled()).toBe(false); }); - it("declines unsupported OTLP protocols instead of misrouting spans", () => { + it("declines unsupported OTLP protocols instead of misrouting spans", async () => { process.env.OTEL_EXPORTER_OTLP_ENDPOINT = "http://localhost:4317"; process.env.OTEL_EXPORTER_OTLP_PROTOCOL = "grpc"; - initTelemetryExport(); + await initTelemetryExport(); expect(isTelemetryExportEnabled()).toBe(false); process.env.OTEL_EXPORTER_OTLP_TRACES_PROTOCOL = "http/json"; - initTelemetryExport(); + await initTelemetryExport(); expect(isTelemetryExportEnabled()).toBe(false); }); - it("honors the kill-switches case-insensitively per the OTEL env contract", () => { + it("honors the kill-switches case-insensitively per the OTEL env contract", async () => { process.env.OTEL_EXPORTER_OTLP_ENDPOINT = "http://localhost:4318"; process.env.OTEL_SDK_DISABLED = "TRUE"; - initTelemetryExport(); + await initTelemetryExport(); expect(isTelemetryExportEnabled()).toBe(false); delete process.env.OTEL_SDK_DISABLED; process.env.OTEL_TRACES_EXPORTER = "otlp,None"; - initTelemetryExport(); + await initTelemetryExport(); expect(isTelemetryExportEnabled()).toBe(false); }); }); From 485cc3fc0a6e717ea80929a805bfe7e0096c42c6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 23:26:33 +0200 Subject: [PATCH 063/181] fix(coding-agent): fixed streaming tool previews dropping scrolled-off output - ToolExecutionComponent was updated to skip append-only treatment for finalized blocks and pass result state into `isStreamingPreviewAppendOnly`. - Eval rendering was changed to render full code continuously and to report append-only status only once a result exists, avoiding commitment of stale pending previews. - Live-region tests were added for expanded eval output overflow and for append-only transitions from pending to finalized streaming states. --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/modes/components/tool-execution.ts | 31 ++++--- .../coding-agent/src/tools/eval-render.ts | 39 +++++---- packages/coding-agent/src/tools/renderers.ts | 22 +++-- packages/coding-agent/src/tools/write.ts | 5 +- .../test/tool-live-region-scrollback.test.ts | 87 +++++++++++++++++++ 6 files changed, 144 insertions(+), 42 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index aee58020f..bb0561565 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -19,7 +19,7 @@ - Fixed eval `agent()` failures surfacing as an opaque `RuntimeError: bridge call '__agent__' failed` with no reason. When a subagent aborted, `runEvalAgent` built its failure message with `result.error ?? result.stderr ?? result.abortReason ?? …`, but `result.stderr` is the empty string on a clean abort (and `result.error` is gated on a non-empty `stderr`), so the nullish chain stopped at `""` and never reached `abortReason`. The empty string propagated through the loopback bridge and the Python prelude's `RuntimeError(msg or "bridge call … failed")`, discarding the real reason. The chain now uses `||` so an empty `stderr` falls through to `abortReason`. - Fixed subagent aborts being mislabeled as the generic "Cancelled by caller" when the abort originated inside the subagent's own turn (`stopReason: "aborted"` with no caller signal and no runtime-limit timer). `runSubprocess` now prefers the aborted assistant message's `errorMessage` (e.g. "Request was aborted" or a specific stream error) for that case, while a real caller signal or wall-clock abort still reports its precise reason. -- Fixed a long streaming tool preview that alone overflows the viewport dropping its scrolled-off head on ED3-risk terminals (ghostty/kitty/iTerm2/…). A streaming `write` preview expanded with `Ctrl+O` renders the whole content top-anchored and grows append-only, but the tool block never reported itself append-only to the transcript, so the renderer's commit-as-you-go boundary stopped at the block start and its earlier rows scrolled above the viewport without being committed to native scrollback — they vanished, leaving the preview looking like a viewport-tall circular buffer. `ToolExecutionComponent` now implements `isTranscriptBlockAppendOnly()`, delegating to a renderer-declared `isStreamingPreviewAppendOnly` predicate (gated to the live call-preview phase) so the expanded write stream commits its head exactly like a streamed assistant reply; collapsed previews (sliding tail window) and result previews (which can collapse) stay deferred. +- Fixed a long streaming tool preview that alone overflows the viewport dropping its scrolled-off head on ED3-risk terminals (ghostty/kitty/iTerm2/…). When expanded with `Ctrl+O`, a streaming `write` (content streaming in) and a streaming `eval` (stdout streaming below its fixed code cell) render top-anchored and grow append-only, but the tool block never reported itself append-only to the transcript, so the renderer's commit-as-you-go boundary stopped at the block start and the earlier rows that scrolled above the viewport were committed nowhere — they vanished, leaving the preview looking like a viewport-tall circular buffer. `ToolExecutionComponent` now implements `isTranscriptBlockAppendOnly()` (gated on `isTranscriptBlockFinalized()`, so it also covers partial-result streams like `eval`), delegating to a renderer-declared `isStreamingPreviewAppendOnly` predicate so the expanded stream commits its head exactly like a streamed assistant reply. Collapsed previews (bounded sliding tail windows) and finalized/result previews (which can collapse to a capped view) stay deferred. ## [15.9.69] - 2026-06-06 ### Fixed diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index 328969a67..043fbe439 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -531,25 +531,32 @@ export class ToolExecutionComponent extends Container { } /** - * While streaming its call preview, a tool block whose preview is append-only - * (rows only grow at the bottom, never re-layout) lets the renderer commit the - * scrolled-off head of an over-tall preview to native scrollback instead of - * dropping it — the same anti-yank path a streaming assistant reply uses (see - * {@link TranscriptContainer} + `NativeScrollbackLiveRegion`). Gated on the - * call-preview phase (no result yet) so the boundary closes the instant the - * preview swaps to a result that may collapse; the renderer decides whether - * its current preview shape qualifies via `isStreamingPreviewAppendOnly`. + * While a tool's preview is still streaming, a block whose preview is + * append-only (rows only grow at the bottom, never re-layout) lets the + * renderer commit the scrolled-off head of an over-tall preview to native + * scrollback instead of dropping it — the same anti-yank path a streaming + * assistant reply uses (see {@link TranscriptContainer} + + * `NativeScrollbackLiveRegion`). Covers both phases: a pre-result call preview + * (a `write` whose content streams in) and a partial-result preview that + * streams output below fixed input (an `eval`/`bash` whose stdout grows under + * its code cell). Gated on {@link isTranscriptBlockFinalized} so the boundary + * closes the instant the block reaches a terminal state — a final result that + * may collapse to a compact view, a backgrounded async tool, or a seal — and + * the renderer decides whether its current preview shape qualifies via + * `isStreamingPreviewAppendOnly` (typically: only the expanded full view, + * which is top-anchored; the collapsed tail window re-layouts but is bounded + * so it never overflows anyway). */ isTranscriptBlockAppendOnly(): boolean { - // A result preview can collapse/re-layout; only the live call preview is a - // candidate. Sealed/aborted blocks are finalized, not streaming. - if (this.#sealed || this.#result !== undefined) return false; + // A finalized block's preview can collapse/re-layout; only a live, + // still-streaming block is a candidate. + if (this.isTranscriptBlockFinalized()) return false; const predicate = (this.#tool as { isStreamingPreviewAppendOnly?: ToolRenderer["isStreamingPreviewAppendOnly"] } | undefined) ?.isStreamingPreviewAppendOnly ?? toolRenderers[this.#toolName]?.isStreamingPreviewAppendOnly; if (!predicate) return false; try { - return predicate(this.#getCallArgsForRender(), this.#renderState); + return predicate(this.#getCallArgsForRender(), this.#renderState, this.#result); } catch (err) { logger.warn("Tool append-only predicate failed", { tool: this.#toolName, error: String(err) }); return false; diff --git a/packages/coding-agent/src/tools/eval-render.ts b/packages/coding-agent/src/tools/eval-render.ts index e751bf3ff..d9379bba2 100644 --- a/packages/coding-agent/src/tools/eval-render.ts +++ b/packages/coding-agent/src/tools/eval-render.ts @@ -40,14 +40,6 @@ import { wrapBrackets, } from "./render-utils"; export const EVAL_DEFAULT_PREVIEW_LINES = 10; -/** - * Rows of source kept in the *pending* eval preview. The window follows the - * streaming edge (newest lines pinned to the bottom) so you can watch the code - * being written, while staying bounded — a volatile tool block taller than the - * viewport would otherwise strand its scrolled-off head out of native scrollback - * on ED3-risk terminals. Matches the streaming windows used by edit/write. - */ -export const EVAL_STREAMING_PREVIEW_LINES = 12; function languageForHighlighter(language: EvalLanguage | undefined): "python" | "javascript" { return language === "js" ? "javascript" : "python"; @@ -517,15 +509,12 @@ export const evalToolRenderer = { title: cell.title, status: "pending", width, - codeMaxLines: EVAL_STREAMING_PREVIEW_LINES, - // Follow the streaming edge with a bounded tail window so the - // newest source stays visible as it is written, instead of - // rendering every line of a >100-line `code` — which would - // overflow the viewport and, because a tool block is volatile - // (it collapses to a capped result), strand its scrolled-off head - // out of native scrollback, cutting the box top. `Ctrl+O` lifts - // the window via `expanded` for a deliberate full view. - codeTail: true, + // Always render the full source: the code is fixed input, not the + // streaming part, so it is never compacted. While still pending + // (args streaming) the block is not yet committed to native + // scrollback — its head is only committed once a result exists and + // the code has finalized (see `isStreamingPreviewAppendOnly`). + codeMaxLines: Number.POSITIVE_INFINITY, expanded: options.expanded, animate, }, @@ -628,7 +617,9 @@ export const evalToolRenderer = { duration: cell.durationMs, output: outputLines.length > 0 ? outputLines.join("\n") : undefined, outputMaxLines: outputLines.length, - codeMaxLines: expanded ? Number.POSITIVE_INFINITY : EVAL_DEFAULT_PREVIEW_LINES, + // Code is fixed input — always shown in full, never compacted. + // Only `output` honors the collapsed preview cap above. + codeMaxLines: Number.POSITIVE_INFINITY, expanded, width, animate, @@ -760,6 +751,18 @@ export const evalToolRenderer = { }, }; }, + + // Append-only once a result exists (args complete → code finalized). The code + // is rendered in full as a fixed top-anchored prefix, and the streamed stdout + // below it only appends rows at the bottom, so the scrolled-off head commits + // to native scrollback instead of being yanked — collapsed or expanded, since + // the collapsed output cap keeps its sliding tail in the bottom live region. + // Returns false while still pending: the code is mid-stream (args incomplete) + // and its header still reads "pending", so committing it would strand a stale + // pending preview in history. + isStreamingPreviewAppendOnly(_args: EvalRenderArgs, _options: RenderResultOptions, result?: unknown): boolean { + return result != null; + }, mergeCallAndResult: true, inline: true, }; diff --git a/packages/coding-agent/src/tools/renderers.ts b/packages/coding-agent/src/tools/renderers.ts index 9f8cbca29..4f74bd090 100644 --- a/packages/coding-agent/src/tools/renderers.ts +++ b/packages/coding-agent/src/tools/renderers.ts @@ -41,16 +41,20 @@ export type ToolRenderer = { ) => Component; mergeCallAndResult?: boolean; /** - * While the call preview is streaming, report whether the currently-rendered - * preview is append-only: its rows only grow at the bottom and never - * re-layout (a full, top-anchored content preview). The transcript reports - * this up to the TUI so a streaming preview taller than the viewport commits - * its scrolled-off head to native scrollback instead of dropping it (see - * `ToolExecutionComponent.isTranscriptBlockAppendOnly`). Omit (or return - * `false`) for previews that slide a tail window or later collapse to a - * compact result — committing their head would strand stale rows. + * While a tool's preview is still streaming, report whether the + * currently-rendered preview is append-only: its rows only grow at the bottom + * and never re-layout above the bottom live region (a full, top-anchored + * content/code preview). The transcript reports this up to the TUI so a + * streaming preview taller than the viewport commits its scrolled-off head to + * native scrollback instead of dropping it (see + * `ToolExecutionComponent.isTranscriptBlockAppendOnly`). `result` is the + * latest (possibly partial) tool result, or `undefined` before one exists — + * `eval`/`bash` use its presence to defer committing until the streamed input + * (code) has finalized. Omit (or return `false`) for previews that slide a + * tail window or later collapse to a compact result — committing their head + * would strand stale rows. */ - isStreamingPreviewAppendOnly?: (args: unknown, options: RenderResultOptions) => boolean; + isStreamingPreviewAppendOnly?: (args: unknown, options: RenderResultOptions, result?: unknown) => boolean; /** Render without background box, inline in the response flow */ inline?: boolean; }; diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index 35b0683f2..fc3946fc0 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -1026,8 +1026,9 @@ export const writeToolRenderer = { // The collapsed preview slides a bounded tail window (`formatStreamingContent` // with `WRITE_STREAMING_PREVIEW_LINES`) whose visible rows re-layout as the // window moves — not append-only, but it never overflows the viewport, so its - // head is never at risk of being dropped regardless. - isStreamingPreviewAppendOnly(args: WriteRenderArgs, options: RenderResultOptions): boolean { + // head is never at risk of being dropped regardless. `write` has no partial + // result (content streams as args), so `result` is ignored here. + isStreamingPreviewAppendOnly(args: WriteRenderArgs, options: RenderResultOptions, _result?: unknown): boolean { return Boolean(options?.expanded && args.content); }, diff --git a/packages/coding-agent/test/tool-live-region-scrollback.test.ts b/packages/coding-agent/test/tool-live-region-scrollback.test.ts index aae7130a7..1c34da43d 100644 --- a/packages/coding-agent/test/tool-live-region-scrollback.test.ts +++ b/packages/coding-agent/test/tool-live-region-scrollback.test.ts @@ -221,6 +221,93 @@ describe("tool live-region scrollback", () => { component.stopAnimation(); } }); + + it("commits the scrolled-off head of an expanded eval whose output streams past the viewport", async () => { + if (process.platform === "win32") return; + + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(120, 12); + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => + undefined; + const tui = new TUI(term); + const chat = new TranscriptContainer(); + const title = "stream lots of output"; + const code = "for (let i = 0; i < 40; i++) console.log('MARK-' + i);"; + const args = { cells: [{ language: "js", title, code }] }; + const component = new ToolExecutionComponent("eval", args, {}, undefined, tui, process.cwd()); + component.setExpanded(true); + const out = (n: number) => Array.from({ length: n }, (_unused, i) => `MARK-${i}`).join("\n"); + const partial = (output: string) => + component.updateResult( + { + content: [{ type: "text", text: "" }], + details: { cells: [{ index: 0, title, code, language: "js", output, status: "running" }] }, + }, + true, + ); + + try { + chat.addChild(component); + tui.addChild(chat); + tui.start(); + tui.setEagerNativeScrollbackRebuild(true); + await term.waitForRender(); + + // A short output that fits, then the full stream that alone overflows the + // 12-row viewport — the frame that scrolls the output head above the top. + partial(out(4)); + tui.requestRender(); + await term.waitForRender(); + + partial(out(40)); + tui.requestRender(); + await term.waitForRender(); + + const strip = (rows: string[]) => rows.map(row => Bun.stripANSI(row).trimEnd()).join("\n"); + const scrollText = strip(term.getScrollBuffer()); + const viewportText = strip(term.getViewport()); + + // The streamed output head scrolled above the viewport: it must live in + // native scrollback (committed), not nowhere. The fixed code cell rides + // along as the stable prefix above it. + expect(viewportText).not.toContain("MARK-0"); + expect(scrollText).toContain("MARK-0"); + expect(scrollText).toContain("MARK-20"); + // The streaming tail stays on screen, and nothing went missing between. + expect(viewportText).toContain("MARK-39"); + } finally { + component.stopAnimation(); + tui.stop(); + await term.flush(); + } + }); + }); + + it("keeps a streaming eval append-only only while expanded and unfinalized", () => { + const tui = new TUI(new VirtualTerminal(80, 24)); + const title = "t"; + const code = "console.log('x')"; + const args = { cells: [{ language: "js", title, code }] }; + const component = new ToolExecutionComponent("eval", args, {}, undefined, tui, process.cwd()); + type AppendOnly = { isTranscriptBlockAppendOnly(): boolean }; + const probe = component as unknown as AppendOnly; + const details = { + cells: [{ index: 0, title, code, language: "js", output: "MARK-0\nMARK-1", status: "running" }], + }; + try { + // Collapsed: bounded sliding tail windows — not append-only. + expect(probe.isTranscriptBlockAppendOnly()).toBe(false); + component.setExpanded(true); + // Expanded + partial (streaming output): append-only. + component.updateResult({ content: [{ type: "text", text: "" }], details }, true); + expect(probe.isTranscriptBlockAppendOnly()).toBe(true); + // Final result may collapse to a capped view — boundary closes. + component.updateResult({ content: [{ type: "text", text: "" }], details }, false); + expect(probe.isTranscriptBlockAppendOnly()).toBe(false); + } finally { + component.stopAnimation(); + } + }); }); function makeAssistantMessage(text: string): AssistantMessage { From dcefc9e3d68c71be82e883e5ac502d0e3c9e39e3 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 21:34:33 +0000 Subject: [PATCH 064/181] fix(debug): waited for dlv unix socket Used fs.stat to confirm delayed dlv Unix socket creation before connecting so Linux socket-mode adapters do not race Bun.connect. Added a delayed socket adapter regression test covering the launch path.\n\nFixes #2013 --- packages/coding-agent/src/dap/client.ts | 26 +++++----- .../test/debug/dap-launch-failures.test.ts | 49 +++++++++++++++++++ 2 files changed, 63 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/src/dap/client.ts b/packages/coding-agent/src/dap/client.ts index 79b333a82..54ea4581d 100644 --- a/packages/coding-agent/src/dap/client.ts +++ b/packages/coding-agent/src/dap/client.ts @@ -1,4 +1,5 @@ -import { logger, ptree } from "@oh-my-pi/pi-utils"; +import * as fs from "node:fs/promises"; +import { isEnoent, logger, ptree } from "@oh-my-pi/pi-utils"; import { NON_INTERACTIVE_ENV } from "../exec/non-interactive-env"; import { ToolAbortError } from "../tools/tool-errors"; import type { @@ -165,16 +166,8 @@ export class DapClient { detached: true, }); - // Wait for the socket file to appear (dlv needs to start listening) await waitForCondition( - () => { - try { - Bun.file(socketPath).size; - return true; - } catch { - return false; - } - }, + () => isUnixSocketReady(socketPath), 10_000, proc, ); @@ -553,15 +546,24 @@ export class DapClient { } } +async function isUnixSocketReady(socketPath: string): Promise { + try { + return (await fs.stat(socketPath)).isSocket(); + } catch (error) { + if (isEnoent(error)) return false; + throw error; + } +} + /** Poll a condition until it returns true, or timeout/process exit. */ async function waitForCondition( - check: () => boolean, + check: () => boolean | Promise, timeoutMs: number, proc: { exitCode: number | null }, ): Promise { const deadline = Date.now() + timeoutMs; while (Date.now() < deadline) { - if (check()) return; + if (await check()) return; if (proc.exitCode !== null) { throw new Error("Adapter process exited before socket was ready"); } diff --git a/packages/coding-agent/test/debug/dap-launch-failures.test.ts b/packages/coding-agent/test/debug/dap-launch-failures.test.ts index 83a133e95..004332d59 100644 --- a/packages/coding-agent/test/debug/dap-launch-failures.test.ts +++ b/packages/coding-agent/test/debug/dap-launch-failures.test.ts @@ -22,6 +22,32 @@ const TEST_ADAPTER: DapResolvedAdapter = { connectMode: "stdio", }; +const DELAYED_UNIX_SOCKET_ADAPTER = ` +const listenPrefix = "--listen=unix:"; +const listenArg = process.argv.find(arg => arg.startsWith(listenPrefix)); +if (!listenArg) { + throw new Error("missing --listen=unix argument"); +} +const socketPath = listenArg.slice(listenPrefix.length); +let server; +process.on("SIGTERM", () => { + server?.stop(); + process.exit(0); +}); +await Bun.sleep(100); +server = Bun.listen({ + unix: socketPath, + socket: { + open() {}, + data() {}, + close() {}, + error() {}, + }, +}); +await Bun.sleep(2_000); +server.stop(); +`; + type DapEventHandler = (body: unknown, event: DapEventMessage) => void | Promise; class FakeDapClient { @@ -286,6 +312,29 @@ describe("DAP launch failure handling", () => { expect(message).toContain("launch: 'C:\\repo\\program' is not a valid executable"); expect(message).toContain("configurationDone: Expected process to be stopped."); }); + + it("waits for delayed Unix socket adapters before connecting on Linux", async () => { + if (process.platform !== "linux") return; + const cwd = await fs.mkdtemp(path.join(os.tmpdir(), "omp-debug-dlv-socket-")); + const adapterPath = path.join(cwd, "delayed-unix-socket-adapter.mjs"); + await fs.writeFile(adapterPath, DELAYED_UNIX_SOCKET_ADAPTER); + const adapter: DapResolvedAdapter = { + ...TEST_ADAPTER, + name: "dlv", + command: process.execPath, + args: [adapterPath], + resolvedCommand: process.execPath, + connectMode: "socket", + }; + let client: DapClient | undefined; + try { + client = await DapClient.spawn({ adapter, cwd }); + expect(client.isAlive()).toBe(true); + } finally { + await client?.dispose(); + await fs.rm(cwd, { recursive: true, force: true }); + } + }); }); describe("DebugTool launch validation", () => { From 9a5f5087df4ea318d3a38d3b61aa17e8b33b0a28 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sat, 6 Jun 2026 21:34:41 +0000 Subject: [PATCH 065/181] style: bun run fix --- packages/coding-agent/src/dap/client.ts | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/packages/coding-agent/src/dap/client.ts b/packages/coding-agent/src/dap/client.ts index 54ea4581d..a93e1df9c 100644 --- a/packages/coding-agent/src/dap/client.ts +++ b/packages/coding-agent/src/dap/client.ts @@ -166,11 +166,7 @@ export class DapClient { detached: true, }); - await waitForCondition( - () => isUnixSocketReady(socketPath), - 10_000, - proc, - ); + await waitForCondition(() => isUnixSocketReady(socketPath), 10_000, proc); const { readable, writeSink, socket } = await connectSocket({ unix: socketPath }); const client = new DapClient(adapter, cwd, proc, { readable, writeSink, socket }); From 7ba27b45280ed08c5807f990afc08905f3b26e70 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 23:41:40 +0200 Subject: [PATCH 066/181] fix(coding-agent): prevented long plan previews from clipping head - Added PlanReviewBlock reporting append-only so an over-tall plan plus selector commits the scrolled-off head to native scrollback. - Avoided top-clipping of long plans on ED3-risk terminals where a plain Container is deferred. --- .../coding-agent/src/modes/interactive-mode.ts | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 1166097df..8f2bfa1b1 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -250,6 +250,20 @@ export interface InteractiveModeOptions { initialMessages?: string[]; } +/** + * Plan-review preview block. Once rendered it is static (a one-shot Markdown of + * the plan file), so even while it sits as the live bottom block beneath the + * approval selector its scrolled-off head is safe to commit to native + * scrollback. Reporting append-only lets an over-tall plan + selector commit the + * plan's head instead of clipping it — without this a plain {@link Container} is + * deferred and a long plan is cut off the top on ED3-risk terminals. + */ +class PlanReviewBlock extends Container { + isTranscriptBlockAppendOnly(): boolean { + return true; + } +} + export class InteractiveMode implements InteractiveModeContext { session: AgentSession; sessionManager: SessionManager; @@ -1680,7 +1694,7 @@ export class InteractiveMode implements InteractiveModeContext { #renderPlanPreview(planContent: string, options?: { append?: boolean }): void { const existingContainer = this.#planReviewContainer; const replaceExisting = options?.append !== true && existingContainer !== undefined; - const planReviewContainer = replaceExisting ? existingContainer : new Container(); + const planReviewContainer = replaceExisting ? existingContainer : new PlanReviewBlock(); planReviewContainer.clear(); planReviewContainer.addChild(new Spacer(1)); planReviewContainer.addChild(new DynamicBorder()); From 22bb6b99272042a966c2e47ef1a68499aa1addb9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 23:52:33 +0200 Subject: [PATCH 067/181] feat(tui): widened sync-output defaults with runtime DECRQM upgrade - Enabled DEC 2026 for Alacritty/VS Code and via TERM_FEATURES Sy token. - Stopped blanket-disabling SSH for recognized direct terminals. - Made the DECRQM probe enable sync on a positive report, not just disable. - Extracted synchronizedOutputUserOverride so opt-out beats force-on. --- docs/tui-core-renderer.md | 13 +- .../test/streaming-preview-height.test.ts | 27 +-- .../test/tool-live-region-scrollback.test.ts | 2 +- packages/tui/CHANGELOG.md | 5 + packages/tui/src/terminal-capabilities.ts | 89 +++++++--- packages/tui/src/tui.ts | 25 +-- packages/tui/test/issue-1765-repro.test.ts | 160 ++++++++++++++++++ .../tui/test/terminal-capabilities.test.ts | 90 +++++++++- 8 files changed, 349 insertions(+), 62 deletions(-) diff --git a/docs/tui-core-renderer.md b/docs/tui-core-renderer.md index 3cc17f8b7..ba9705e8c 100644 --- a/docs/tui-core-renderer.md +++ b/docs/tui-core-renderer.md @@ -245,9 +245,16 @@ parameterized over `(env, platform)` so they are unit-testable: VTE, iTerm2, Apple Terminal, GNOME Terminal, Ptyxis, xfce4-terminal), Linux truecolor, **and every other unknown POSIX terminal**. The default is *risky* on purpose. -- `shouldEnableSynchronizedOutputByDefault(env, platform, id)` → DEC 2026 on by - default only for kitty/ghostty/wezterm/iterm2, off for win32 / SSH / - multiplexers / VTE-family. Layered with a runtime DECRQM auto-disable. +- `shouldEnableSynchronizedOutputByDefault(env, id)` → DEC 2026 default. Precedence: + user opt-out (`PI_NO_SYNC_OUTPUT`/`PI_TUI_SYNC_OUTPUT=0`) → user force-on + (`PI_FORCE_SYNC_OUTPUT=1`/`PI_TUI_SYNC_OUTPUT=1`) → `TERM_FEATURES` advertises + `Sy` → `WT_SESSION` (WT/WSL) → known direct terminals + (kitty/ghostty/wezterm/iterm2/alacritty/vscode; SSH passes through) → off for + risky multiplexers and everything else (VTE-family, GNU screen, Apple Terminal, + legacy conhost, unknown). Reconciled at runtime by the DECRQM mode-2026 report: + a positive report **enables** sync (upgrading default-off muxes like + zellij/tmux-master), a negative one disables it; a user override still wins. + `synchronizedOutputUserOverride(env)` is the shared opt-out/force resolver. - `detectRectangularSgrSupport(id, env)` → DECCARA fills: **kitty only** (ghostty does not implement the SGR-background extension), off in multiplexers and under `PI_NO_DECCARA`. diff --git a/packages/coding-agent/test/streaming-preview-height.test.ts b/packages/coding-agent/test/streaming-preview-height.test.ts index 5156d8382..b5c2d3f08 100644 --- a/packages/coding-agent/test/streaming-preview-height.test.ts +++ b/packages/coding-agent/test/streaming-preview-height.test.ts @@ -338,7 +338,7 @@ describe("streaming tool call preview height (bounded across renderers)", () => expect(visibleWidth(topBorder ?? "")).toBe(width); }); - test("eval/bash/ssh pending previews stay short even with very long multiline args", () => { + test("bash/ssh pending previews stay short even with very long multiline args", () => { const longLines = Array.from({ length: 80 }, (_, i) => `line-${i}`); const cases: Array<{ name: string; @@ -347,17 +347,6 @@ describe("streaming tool call preview height (bounded across renderers)", () => mustHide: string[]; marker: RegExp; }> = [ - { - // eval follows the streaming edge: a bounded tail window so the newest - // source stays visible while the box never overflows. - name: "eval", - args: { - cells: [{ language: "js", title: "big", code: longLines.map(line => `const ${line} = 1;`).join("\n") }], - }, - mustContain: ["const line-79 = 1;"], - mustHide: ["const line-0 = 1;"], - marker: /earlier lines/, - }, { // bash/ssh keep a bounded head+tail window: the start and the // latest are both visible, the middle is elided. @@ -403,4 +392,18 @@ describe("streaming tool call preview height (bounded across renderers)", () => expect(text).toContain("line-79"); expect(text).not.toMatch(/more lines/); }); + + test("eval pending preview preserves full code (never collapsed)", () => { + const longLines = Array.from({ length: 80 }, (_, i) => `line-${i}`); + const { lines, text } = renderPending("eval", { + cells: [{ language: "js", title: "big", code: longLines.map(line => `const ${line} = 1;`).join("\n") }], + }); + + expect(lines.length, "eval code preview should not be capped").toBeGreaterThan(80); + expect(text).toContain("const line-0 = 1;"); + expect(text).toContain("const line-40 = 1;"); + expect(text).toContain("const line-79 = 1;"); + expect(text).not.toMatch(/more lines/); + expect(text).not.toMatch(/earlier lines/); + }); }); diff --git a/packages/coding-agent/test/tool-live-region-scrollback.test.ts b/packages/coding-agent/test/tool-live-region-scrollback.test.ts index 1c34da43d..80fa86305 100644 --- a/packages/coding-agent/test/tool-live-region-scrollback.test.ts +++ b/packages/coding-agent/test/tool-live-region-scrollback.test.ts @@ -73,7 +73,7 @@ describe("tool live-region scrollback", () => { .join("\n"); expect(bufferText).not.toContain("pending [1/1]"); expect(bufferText).toContain("const line9 = 9;"); - expect(bufferText).toContain("… 10 more lines"); + expect(bufferText).toContain("const line19 = 19;"); } finally { component.stopAnimation(); tui.stop(); diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 81cfd3272..1cecbf5d5 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,9 +2,14 @@ ## [Unreleased] +### Changed + +- Reworked the DEC 2026 synchronized-output default policy: a positive DECRQM mode-2026 report now **enables** sync (previously a report could only disable it), so conservatively defaulted-off hosts that actually support it — current Zellij, tmux master, foot, contour, mintty — are upgraded at runtime. The static allowlist also covers Alacritty and the VS Code terminal, honors a `TERM_FEATURES` `Sy` advertisement and `WT_SESSION` (Windows Terminal / WSL), and no longer blanket-disables SSH (DEC 2026 passes through to the outer terminal). Risky multiplexers still start off and rely on the probe. Added `synchronizedOutputUserOverride()` as the shared opt-out/force resolver. + ### Fixed - Fixed WSL/Windows Terminal row flicker while typing by repainting changed text rows before clearing only their stale suffix ([#2011](https://github.com/can1357/oh-my-pi/issues/2011)). +- Fixed terminals that support DEC 2026 still tearing/flickering because the renderer ignored a positive DECRQM capability report and kept synchronized output off — most visibly WSL + Windows Terminal, Alacritty (≥0.13), and the VS Code terminal (≥1.108), which were detected yet refused sync. ## [15.9.69] - 2026-06-06 diff --git a/packages/tui/src/terminal-capabilities.ts b/packages/tui/src/terminal-capabilities.ts index bea4ed1f8..9324386a3 100644 --- a/packages/tui/src/terminal-capabilities.ts +++ b/packages/tui/src/terminal-capabilities.ts @@ -187,41 +187,76 @@ export function detectTerminalEagerEraseScrollbackRisk( return true; } -/** Whether DEC 2026 synchronized-output wrappers should be enabled by default. */ -export function shouldEnableSynchronizedOutputByDefault( - env: NodeJS.ProcessEnv = Bun.env, - platform: NodeJS.Platform = process.platform, - terminalId: TerminalId = TERMINAL_ID, -): boolean { +/** + * Resolve an explicit user override for DEC 2026 synchronized output. Returns + * `false` for an opt-out, `true` for a force-on, or `null` when the user has + * expressed no preference. Shared by the static default and the runtime DECRQM + * probe so both honor the same precedence — an opt-out beats a force-on. + */ +export function synchronizedOutputUserOverride(env: NodeJS.ProcessEnv = Bun.env): boolean | null { if (env.PI_NO_SYNC_OUTPUT || env.PI_TUI_SYNC_OUTPUT === "0") return false; if (env.PI_FORCE_SYNC_OUTPUT === "1" || env.PI_TUI_SYNC_OUTPUT === "1") return true; - if (platform === "win32") return false; + return null; +} +/** + * Whether `TERM_FEATURES` advertises DEC 2026 synchronized output via the `Sy` + * capability token. `TERM_FEATURES` is a run of capitalized two-letter codes + * (e.g. `…Sy…`), so a case-sensitive substring match is unambiguous: `Sy` + * cannot straddle a code boundary because those are always lowercase→uppercase. + */ +function advertisesSynchronizedOutput(termFeatures: string | undefined): boolean { + return termFeatures?.includes("Sy") ?? false; +} + +/** + * Whether DEC 2026 synchronized-output wrappers should be enabled by default. + * + * Policy (highest precedence first): + * 1. Explicit user override (`PI_NO_SYNC_OUTPUT`/`PI_TUI_SYNC_OUTPUT=0` off, + * `PI_FORCE_SYNC_OUTPUT=1`/`PI_TUI_SYNC_OUTPUT=1` on). + * 2. Positive `TERM_FEATURES` advertisement (`Sy`) — survives SSH/mux wrapping. + * 3. Windows Terminal (1.24+) via `WT_SESSION`, on native win32 and the + * WSL/SSH-fronted host alike. + * 4. Known direct terminals with confirmed support. SSH does *not* disable — + * DEC 2026 passes through SSH when the outer terminal honors it. + * 5. Everything else starts off, including risky multiplexers; the runtime + * DECRQM probe upgrades any of them when the terminal actually reports + * `?2026` supported (current zellij, tmux master, foot, contour, mintty…). + */ +export function shouldEnableSynchronizedOutputByDefault( + env: NodeJS.ProcessEnv = Bun.env, + terminalId: TerminalId = TERMINAL_ID, +): boolean { + const override = synchronizedOutputUserOverride(env); + if (override !== null) return override; + + if (advertisesSynchronizedOutput(env.TERM_FEATURES)) return true; + if (env.WT_SESSION) return true; + + // Risky multiplexers start off even when an inner terminal id leaks through: + // older tmux/screen synchronized-output handling is flaky and a mux may not + // pass DEC 2026 to the outer host. The DECRQM probe re-enables sync when the + // mux reports `?2026` supported. const term = env.TERM?.toLowerCase() ?? ""; - const termProgram = env.TERM_PROGRAM?.toLowerCase() ?? ""; - if ( - env.SSH_CONNECTION || - env.SSH_CLIENT || - env.SSH_TTY || - env.TMUX || - env.STY || - env.ZELLIJ || - term.startsWith("tmux") || - term.startsWith("screen") - ) { + if (env.TMUX || env.STY || env.ZELLIJ || term.startsWith("tmux") || term.startsWith("screen")) { return false; } - if (env.VTE_VERSION) return false; - switch (termProgram) { - case "gnome-terminal": - case "kgx": - case "ptyxis": - case "xfce4-terminal": - return false; + + switch (terminalId) { + case "kitty": + case "ghostty": + case "wezterm": + case "iterm2": + case "alacritty": + case "vscode": + return true; default: - break; + // VTE family, GNU screen, Apple Terminal, legacy native console host + // (no WT_SESSION), and bare/unknown xterm profiles stay off until the + // DECRQM probe proves support. + return false; } - return terminalId === "kitty" || terminalId === "ghostty" || terminalId === "wezterm" || terminalId === "iterm2"; } /** diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 3a0471d83..62db855a2 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -24,6 +24,7 @@ import { setCellDimensions, setTerminalImageProtocol, shouldEnableSynchronizedOutputByDefault, + synchronizedOutputUserOverride, TERMINAL, } from "./terminal-capabilities"; import { @@ -573,8 +574,9 @@ export class TUI extends Container { /** * Whether DEC 2026 synchronized-output wrappers are currently emitted around - * paints. Starts from conservative terminal/env detection and is force-disabled - * at runtime if the terminal reports mode 2026 unsupported via DECRQM. + * paints. Starts from conservative terminal/env detection and is reconciled at + * runtime against the terminal's DECRQM mode-2026 report — enabled on a + * positive report, disabled on a negative one. */ get synchronizedOutput(): boolean { return this.#synchronizedOutputEnabled; @@ -744,14 +746,15 @@ export class TUI extends Container { start(): void { this.#stopped = false; - // Disable synchronized output if the terminal reports DEC 2026 unsupported - // via DECRQM. PI_NO_SYNC_OUTPUT already forces it off at construction, so - // only react when the user has not already opted out. Future paints drop - // the begin/end markers; the autowrap guards stay (see #1765). + // A DECRQM report for mode 2026 is authoritative: enable synchronized + // output when the terminal reports support (upgrading conservatively + // defaulted-off hosts like zellij/tmux-master/foot) and disable it when + // the terminal reports it unsupported. An explicit user opt-out/force + // (resolved at construction) still wins, so skip the probe in that case. this.terminal.onPrivateModeReport?.((mode, supported) => { - if (mode === 2026 && !supported && !$flag("PI_NO_SYNC_OUTPUT")) { - this.#setSynchronizedOutput(false); - } + if (mode !== 2026) return; + if (synchronizedOutputUserOverride() !== null) return; + this.#setSynchronizedOutput(supported); }); this.terminal.start( data => this.#handleInput(data), @@ -914,8 +917,8 @@ export class TUI extends Container { /** * Toggle synchronized-output (DEC 2026) wrappers on paint/cursor writes and - * recompute the cached begin/end sequences. Honors a DECRQM report that the - * terminal does not support 2026 (#1765 covers the static env opt-out). + * recompute the cached begin/end sequences. Driven by the terminal's DECRQM + * mode-2026 report (#1765 covers the static env opt-out). */ #setSynchronizedOutput(enabled: boolean): void { if (this.#synchronizedOutputEnabled === enabled) return; diff --git a/packages/tui/test/issue-1765-repro.test.ts b/packages/tui/test/issue-1765-repro.test.ts index af64666d3..df73fda8c 100644 --- a/packages/tui/test/issue-1765-repro.test.ts +++ b/packages/tui/test/issue-1765-repro.test.ts @@ -33,6 +33,21 @@ class FocusedLine implements Component, Focusable { } } +// VirtualTerminal does not model DECRQM capability probing, so subclass it to +// register and replay the renderer's mode-2026 report callback on demand. This +// exercises the runtime probe path in `TUI.start()` end-to-end. +class ProbingTerminal extends VirtualTerminal { + #privateModeCallbacks: Array<(mode: number, supported: boolean) => void> = []; + + onPrivateModeReport(callback: (mode: number, supported: boolean) => void): void { + this.#privateModeCallbacks.push(callback); + } + + emitPrivateModeReport(mode: number, supported: boolean): void { + for (const callback of this.#privateModeCallbacks) callback(mode, supported); + } +} + const SYNC_BEGIN = "\x1b[?2026h"; const SYNC_END = "\x1b[?2026l"; const DISABLE_AUTOWRAP = "\x1b[?7l"; @@ -188,3 +203,148 @@ describe("issue #1765: synchronized-output opt-out", () => { }); }); }); + +describe("synchronized-output runtime DECRQM probe", () => { + it("enables synchronized output after a positive DEC 2026 report on a default-off host", async () => { + // TMUX forces the static default off; the positive probe must upgrade it. + await withEnvPatch( + { + TMUX: "1", + WT_SESSION: undefined, + TERM_FEATURES: undefined, + PI_NO_SYNC_OUTPUT: undefined, + PI_FORCE_SYNC_OUTPUT: undefined, + PI_TUI_SYNC_OUTPUT: undefined, + }, + async () => { + const term = new ProbingTerminal(32, 4, 100); + const writes = captureWrites(term); + const component = new MutableLines(["before probe"]); + const tui = new TUI(term); + tui.addChild(component); + + try { + tui.start(); + await term.waitForRender(); + expect(tui.synchronizedOutput).toBe(false); + expectNoSyncOutput(writes); + + const mark = writes.length; + term.emitPrivateModeReport(2026, true); + expect(tui.synchronizedOutput).toBe(true); + + component.lines = ["after probe"]; + tui.requestRender(); + await term.waitForRender(); + + const after = writes.slice(mark).join(""); + expect(after).toContain(SYNC_BEGIN); + expect(after).toContain(SYNC_END); + } finally { + tui.stop(); + } + }, + ); + }); + + it("disables synchronized output after a negative DEC 2026 report on a default-on host", async () => { + // WT_SESSION forces the static default on without a user override flag. + await withEnvPatch( + { + WT_SESSION: "abc", + PI_NO_SYNC_OUTPUT: undefined, + PI_FORCE_SYNC_OUTPUT: undefined, + PI_TUI_SYNC_OUTPUT: undefined, + }, + async () => { + const term = new ProbingTerminal(32, 4, 100); + const writes = captureWrites(term); + const component = new MutableLines(["before probe"]); + const tui = new TUI(term); + tui.addChild(component); + + try { + tui.start(); + await term.waitForRender(); + expect(tui.synchronizedOutput).toBe(true); + + const mark = writes.length; + term.emitPrivateModeReport(2026, false); + expect(tui.synchronizedOutput).toBe(false); + + component.lines = ["after probe"]; + tui.requestRender(); + await term.waitForRender(); + + expectNoSyncOutput(writes.slice(mark)); + } finally { + tui.stop(); + } + }, + ); + }); + + it("ignores a positive probe when the user opted out", async () => { + await withEnvPatch( + { PI_NO_SYNC_OUTPUT: "1", PI_FORCE_SYNC_OUTPUT: undefined, PI_TUI_SYNC_OUTPUT: undefined }, + async () => { + const term = new ProbingTerminal(32, 4, 100); + const writes = captureWrites(term); + const component = new MutableLines(["before probe"]); + const tui = new TUI(term); + tui.addChild(component); + + try { + tui.start(); + await term.waitForRender(); + expect(tui.synchronizedOutput).toBe(false); + + const mark = writes.length; + term.emitPrivateModeReport(2026, true); + expect(tui.synchronizedOutput).toBe(false); + + component.lines = ["after probe"]; + tui.requestRender(); + await term.waitForRender(); + + expectNoSyncOutput(writes.slice(mark)); + } finally { + tui.stop(); + } + }, + ); + }); + + it("ignores a negative probe when the user forced sync on", async () => { + await withEnvPatch( + { PI_FORCE_SYNC_OUTPUT: "1", PI_NO_SYNC_OUTPUT: undefined, PI_TUI_SYNC_OUTPUT: undefined }, + async () => { + const term = new ProbingTerminal(32, 4, 100); + const writes = captureWrites(term); + const component = new MutableLines(["before probe"]); + const tui = new TUI(term); + tui.addChild(component); + + try { + tui.start(); + await term.waitForRender(); + expect(tui.synchronizedOutput).toBe(true); + + const mark = writes.length; + term.emitPrivateModeReport(2026, false); + expect(tui.synchronizedOutput).toBe(true); + + component.lines = ["after probe"]; + tui.requestRender(); + await term.waitForRender(); + + const after = writes.slice(mark).join(""); + expect(after).toContain(SYNC_BEGIN); + expect(after).toContain(SYNC_END); + } finally { + tui.stop(); + } + }, + ); + }); +}); diff --git a/packages/tui/test/terminal-capabilities.test.ts b/packages/tui/test/terminal-capabilities.test.ts index 384b0a2cc..3d4d835f7 100644 --- a/packages/tui/test/terminal-capabilities.test.ts +++ b/packages/tui/test/terminal-capabilities.test.ts @@ -2,6 +2,7 @@ import { describe, expect, it } from "bun:test"; import { detectTerminalEagerEraseScrollbackRisk, shouldEnableSynchronizedOutputByDefault, + synchronizedOutputUserOverride, } from "@oh-my-pi/pi-tui/terminal-capabilities"; describe("terminal capability defaults", () => { @@ -22,20 +23,93 @@ describe("terminal capability defaults", () => { it("keeps native win32 on the dedicated ConPTY deferral path", () => { expect(detectTerminalEagerEraseScrollbackRisk({ WT_SESSION: "abc" }, "win32")).toBe(false); }); +}); - it("disables synchronized output by default for remote, VTE, and unknown profiles", () => { - expect(shouldEnableSynchronizedOutputByDefault({ SSH_CONNECTION: "1 2 3 4" }, "linux", "base")).toBe(false); - expect(shouldEnableSynchronizedOutputByDefault({ VTE_VERSION: "6800" }, "linux", "base")).toBe(false); - expect(shouldEnableSynchronizedOutputByDefault({ TERM: "xterm-256color" }, "linux", "base")).toBe(false); +describe("synchronizedOutputUserOverride", () => { + it("returns null when the user expresses no preference", () => { + expect(synchronizedOutputUserOverride({})).toBeNull(); + expect(synchronizedOutputUserOverride({ TERM: "xterm-256color" })).toBeNull(); }); - it("allows explicit synchronized-output force-on for diagnostics", () => { + it("returns false for either opt-out flag", () => { + expect(synchronizedOutputUserOverride({ PI_NO_SYNC_OUTPUT: "1" })).toBe(false); + expect(synchronizedOutputUserOverride({ PI_TUI_SYNC_OUTPUT: "0" })).toBe(false); + }); + + it("returns true for either force-on flag", () => { + expect(synchronizedOutputUserOverride({ PI_FORCE_SYNC_OUTPUT: "1" })).toBe(true); + expect(synchronizedOutputUserOverride({ PI_TUI_SYNC_OUTPUT: "1" })).toBe(true); + }); + + it("resolves opt-out ahead of force-on when both are set", () => { + expect(synchronizedOutputUserOverride({ PI_NO_SYNC_OUTPUT: "1", PI_FORCE_SYNC_OUTPUT: "1" })).toBe(false); + expect(synchronizedOutputUserOverride({ PI_TUI_SYNC_OUTPUT: "0", PI_FORCE_SYNC_OUTPUT: "1" })).toBe(false); + }); +}); + +describe("shouldEnableSynchronizedOutputByDefault", () => { + it("enables sync for every known direct terminal, including Alacritty and VS Code", () => { + for (const id of ["kitty", "ghostty", "wezterm", "iterm2", "alacritty", "vscode"] as const) { + expect(shouldEnableSynchronizedOutputByDefault({}, id)).toBe(true); + } + }); + + it("enables sync in Windows Terminal / WSL via WT_SESSION regardless of terminal id", () => { + expect(shouldEnableSynchronizedOutputByDefault({ WT_SESSION: "abc" }, "trueColor")).toBe(true); + // WSL shape: Linux + WT_SESSION + COLORTERM=truecolor collapses to trueColor id. + expect(shouldEnableSynchronizedOutputByDefault({ WT_SESSION: "abc", COLORTERM: "truecolor" }, "trueColor")).toBe( + true, + ); + }); + + it("enables sync when TERM_FEATURES advertises the Sy capability, even through SSH/mux", () => { + expect(shouldEnableSynchronizedOutputByDefault({ TERM_FEATURES: "ClSyTc" }, "base")).toBe(true); + expect( + shouldEnableSynchronizedOutputByDefault({ TERM_FEATURES: "ClSyTc", SSH_CONNECTION: "1 2 3 4" }, "base"), + ).toBe(true); + expect(shouldEnableSynchronizedOutputByDefault({ TERM_FEATURES: "ClSyTc", TMUX: "1" }, "base")).toBe(true); + }); + + it("does not treat a TERM_FEATURES list without the Sy token as advertising support", () => { + expect(shouldEnableSynchronizedOutputByDefault({ TERM_FEATURES: "ClTc" }, "base")).toBe(false); + }); + + it("no longer blanket-disables SSH for recognized terminals", () => { + expect(shouldEnableSynchronizedOutputByDefault({ SSH_CONNECTION: "1 2 3 4" }, "iterm2")).toBe(true); + expect(shouldEnableSynchronizedOutputByDefault({ SSH_TTY: "/dev/pts/3" }, "kitty")).toBe(true); + }); + + it("keeps risky multiplexers off by default even when an inner terminal id leaks", () => { + expect(shouldEnableSynchronizedOutputByDefault({ TMUX: "1" }, "kitty")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({ ZELLIJ: "0" }, "ghostty")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({ STY: "x" }, "wezterm")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({ TERM: "tmux-256color" }, "iterm2")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({ TERM: "screen-256color" }, "kitty")).toBe(false); + }); + + it("keeps known-unsupported and unknown profiles off", () => { + expect(shouldEnableSynchronizedOutputByDefault({ VTE_VERSION: "6800" }, "base")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({ TERM: "xterm-256color" }, "base")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({}, "base")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({}, "trueColor")).toBe(false); + }); + + it("lets a user opt-out beat every positive heuristic", () => { + expect(shouldEnableSynchronizedOutputByDefault({ PI_NO_SYNC_OUTPUT: "1" }, "kitty")).toBe(false); + expect(shouldEnableSynchronizedOutputByDefault({ PI_TUI_SYNC_OUTPUT: "0" }, "ghostty")).toBe(false); expect( shouldEnableSynchronizedOutputByDefault( - { PI_FORCE_SYNC_OUTPUT: "1", SSH_CONNECTION: "1 2 3 4" }, - "linux", - "base", + { PI_NO_SYNC_OUTPUT: "1", WT_SESSION: "abc", TERM_FEATURES: "Sy" }, + "kitty", ), + ).toBe(false); + }); + + it("lets a user force-on beat the conservative defaults", () => { + expect(shouldEnableSynchronizedOutputByDefault({ PI_FORCE_SYNC_OUTPUT: "1" }, "base")).toBe(true); + expect(shouldEnableSynchronizedOutputByDefault({ PI_TUI_SYNC_OUTPUT: "1", TMUX: "1" }, "base")).toBe(true); + expect( + shouldEnableSynchronizedOutputByDefault({ PI_FORCE_SYNC_OUTPUT: "1", SSH_CONNECTION: "1 2 3 4" }, "base"), ).toBe(true); }); }); From 54776365cda74ae59968019f91bcf600dd63292c Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 6 Jun 2026 23:58:02 +0200 Subject: [PATCH 068/181] feat(coding-agent/tools): added configurable fetch backend preference and fallback order - Replaced `providers.parallelFetch` with a new `providers.fetch` enum in settings and added migration cleanup for the legacy key. - Updated `renderHtmlToText` to follow configured reader preference with ordered fallback attempts and remote-reader timeout handling before local conversion. - Updated YouTube and fetch tests to use `providers.fetch` and cover Jina-first stall fallback behavior. --- packages/ai/CHANGELOG.md | 2 - packages/coding-agent/CHANGELOG.md | 20 +- packages/coding-agent/DEVELOPMENT.md | 2 +- .../src/config/settings-schema.ts | 23 ++- packages/coding-agent/src/config/settings.ts | 11 ++ packages/coding-agent/src/tools/fetch.ts | 176 +++++++++--------- .../coding-agent/src/web/scrapers/youtube.ts | 5 +- .../test/tools/fetch-jina-stall.test.ts | 6 +- .../test/tools/fetch-kagi-toggle.test.ts | 7 +- .../web-scrapers/youtube-parallel.test.ts | 2 +- 10 files changed, 147 insertions(+), 107 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index db832fa7b..35e9b0b6b 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -14,8 +14,6 @@ - Fixed Cloud Code Assist (Antigravity / Gemini CLI) rejecting the `github` tool with HTTP 400 when the `pr` parameter schema contained `anyOf: [string, array]`. The CCA mixed-type combiner collapse picked the first non-null type (`string`) but indiscriminately copied type-specific keys from variant branches — `items` from the array variant leaked onto the string-typed result, producing `{type: "string", items: {...}}` which Google's API rejects as invalid. The collapse now filters merged variant fields against the winning type's allowed key set. ([#2002](https://github.com/can1357/oh-my-pi/pull/2002)) - Fixed OpenAI Responses-family providers (Codex, OpenAI Responses, Azure Responses) rejecting requests with `400 No tool output found for function call …` after the user branched/navigated the session tree to a node that ends on a tool call (the tool-result child is dropped from the reconstructed history) or after a turn was aborted/crashed between the call streaming and its result persisting. The converters now synthesize a placeholder `function_call_output`/`custom_tool_call_output` immediately after any unpaired `function_call`/`custom_tool_call`, symmetric to the existing orphan-output repair, so the model still sees the call and can recover instead of the whole request 400ing. -### Fixed - - Fixed Anthropic-compatible reasoning endpoints losing prior-turn reasoning on continuation requests when they emit unsigned `thinking` blocks. `convertAnthropicMessages` treated unknown endpoints as signature-enforcing and demoted unsigned reasoning to `type: "text"`, which destabilized tool-call argument serialization on the next turn — the upstream symptom behind the `args?.ops?.map is not a function` crash reported against the `todo` tool. Official `api.anthropic.com` keeps the conservative text fallback; non-official `anthropic-messages` reasoning models now replay unsigned reasoning as native `type: "thinking"` ([#2005](https://github.com/can1357/oh-my-pi/issues/2005)). ## [15.9.67] - 2026-06-06 diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bb0561565..0297e1bbe 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Breaking Changes + +- Replaced the `providers.parallelFetch` boolean setting with the `providers.fetch` enum (`auto` / `native` / `trafilatura` / `lynx` / `parallel` / `jina`) that selects the URL reader-backend priority for the `read`/`fetch` tool, mirroring `providers.image`/`providers.webSearch`. Existing configs are migrated automatically: the legacy key is dropped and the new `auto` default applies. + ### Added - Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run). @@ -13,19 +17,20 @@ - Changed coding-agent startup imports so normal TUI launch imports `InteractiveMode` directly, keeps print/RPC/ACP runners on their branch-only paths, and moves marketplace auto-update work behind a lightweight deferred starter. - Changed cold-launch setup gating so the full setup wizard (every scene plus the overlay and their TUI/OAuth/web-search/theme dependencies) is no longer statically imported by `main.ts`. The current setup version now lives in a tiny dependency-free `modes/setup-version` module, and the wizard barrel is lazy-loaded only when the stored setup version is stale or the wizard is forced — the common up-to-date launch skips loading it entirely. - Changed cold-launch startup imports so the hot-path CLI files no longer pull the full `@oh-my-pi/pi-ai` barrel: `commands/launch.ts` and `cli/args.ts` import `THINKING_EFFORTS`/`Effort` from the tiny `@oh-my-pi/pi-ai/effort` module, and `config/model-registry.ts` now imports its ~20 symbols from narrow subpaths (`api-registry`, `model-cache`, `model-manager`, `model-thinking`, `models`, `provider-models`, `types`, `utils/event-stream`) instead of the barrel — so launching no longer eagerly loads every provider, auth, OAuth, and usage module re-exported by the barrel. - +- Changed the `read`/`fetch` HTML reader-backend priority to `native > trafilatura > lynx > parallel > jina` (was `parallel > jina > trafilatura > lynx > native`). The in-process native `htmlToMarkdown` runs first — instant, no network, full-fidelity — so the common case no longer depends on a remote service, and a stalled remote backend can no longer mask it. Selecting a specific backend via `providers.fetch` tries it first, then the rest fall back. The low-quality gate (`>100` chars and not `isLowQualityOutput`) now applies uniformly to every backend; when none clears it, the highest-priority substantial-but-low-quality output is still surfaced so the `llms.txt` / document-extraction fallbacks keep running. ### Fixed - Fixed eval `agent()` failures surfacing as an opaque `RuntimeError: bridge call '__agent__' failed` with no reason. When a subagent aborted, `runEvalAgent` built its failure message with `result.error ?? result.stderr ?? result.abortReason ?? …`, but `result.stderr` is the empty string on a clean abort (and `result.error` is gated on a non-empty `stderr`), so the nullish chain stopped at `""` and never reached `abortReason`. The empty string propagated through the loopback bridge and the Python prelude's `RuntimeError(msg or "bridge call … failed")`, discarding the real reason. The chain now uses `||` so an empty `stderr` falls through to `abortReason`. - Fixed subagent aborts being mislabeled as the generic "Cancelled by caller" when the abort originated inside the subagent's own turn (`stopReason: "aborted"` with no caller signal and no runtime-limit timer). `runSubprocess` now prefers the aborted assistant message's `errorMessage` (e.g. "Request was aborted" or a specific stream error) for that case, while a real caller signal or wall-clock abort still reports its precise reason. - Fixed a long streaming tool preview that alone overflows the viewport dropping its scrolled-off head on ED3-risk terminals (ghostty/kitty/iTerm2/…). When expanded with `Ctrl+O`, a streaming `write` (content streaming in) and a streaming `eval` (stdout streaming below its fixed code cell) render top-anchored and grow append-only, but the tool block never reported itself append-only to the transcript, so the renderer's commit-as-you-go boundary stopped at the block start and the earlier rows that scrolled above the viewport were committed nowhere — they vanished, leaving the preview looking like a viewport-tall circular buffer. `ToolExecutionComponent` now implements `isTranscriptBlockAppendOnly()` (gated on `isTranscriptBlockFinalized()`, so it also covers partial-result streams like `eval`), delegating to a renderer-declared `isStreamingPreviewAppendOnly` predicate so the expanded stream commits its head exactly like a streamed assistant reply. Collapsed previews (bounded sliding tail windows) and finalized/result previews (which can collapse to a capped view) stay deferred. - -## [15.9.69] - 2026-06-06 -### Fixed - +- Fixed `read`/`fetch` silently dropping whole list sections on pages with malformed list markup — stray ``, text, or `