From 396e91e101f7a584627816d484fff859532fa6aa Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:05:42 +0200 Subject: [PATCH 01/59] fix(ai): preserved streamed responses thinking on empty done summary Codex #handleOutputItemDone unconditionally rebuilt block.thinking from item.summary, clobbering already-streamed reasoning with "" when the done event carries no summary (GPT-5.x Codex renders in TUI). The generic processResponsesStream had the same clobber, and the codex handler also dropped response.reasoning_text.delta events entirely. Added shared finalizeReasoningThinking(): prefer done-item summary, then reasoning_text content, then the streamed accumulation; wired response.reasoning_text.delta into the codex processor. Adopted from PR #4935 minus unrelated prompt churn. Fixes #4918 --- packages/ai/CHANGELOG.md | 4 + .../src/providers/openai-codex-responses.ts | 20 ++- packages/ai/src/providers/openai-shared.ts | 18 +-- packages/ai/test/openai-codex-stream.test.ts | 123 ++++++++++++++++++ .../openai-responses-stream-terminal.test.ts | 43 ++++++ 5 files changed, 198 insertions(+), 10 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3b40e8bea..bf7f3af47 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -7,6 +7,10 @@ - Changed the xAI Grok OAuth (`xai-oauth`) provider to use manual code-paste login by default. `/login` now accepts a pasted authorization code or full `http://127.0.0.1:56121/callback?code=...` redirect URL without starting a local callback listener ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). - Renamed the xAI Grok OAuth provider in login and credential prompts to "xAI Grok OAuth (SuperGrok or X Premium+)" ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). +### Fixed + +- Fixed OpenAI Codex/Responses reasoning streams so streamed thinking content is preserved when the final `output_item.done` reconstructs to an empty summary ([#4918](https://github.com/can1357/oh-my-pi/issues/4918)). + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 80baabe8d..0bde541d0 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -95,6 +95,7 @@ import { encodeTextSignatureV1, finalizeCustomToolCallInputDone, finalizePendingResponsesToolCalls, + finalizeReasoningThinking, finalizeToolCallArgumentsDone, isOpenAIResponsesProgressEvent, mapOpenAIResponsesStopReason, @@ -1407,6 +1408,21 @@ class CodexStreamProcessor { return firstTokenTime; } + if (eventType === "response.reasoning_text.delta") { + const entry = this.runtime.openItemForEvent(rawEvent); + const delta = typeof rawEvent.delta === "string" ? rawEvent.delta : ""; + if (entry?.item.type === "reasoning" && entry.block?.type === "thinking") { + entry.block.thinking += delta; + stream.push({ + type: "thinking_delta", + contentIndex: entry.contentIndex, + delta, + partial: output, + }); + } + return firstTokenTime; + } + if (eventType === "response.reasoning_summary_part.done") { if (this.runtime.currentItem?.type === "reasoning" && this.runtime.currentBlock?.type === "thinking") { appendReasoningSummaryPartDone( @@ -1522,13 +1538,13 @@ class CodexStreamProcessor { // most-recently-added block may belong to a sibling (#2619). Some Codex // function/custom tool items omit `id`; in that case `output_index` still // routes `output_item.done` to the block that received `output_item.added`. - const itemId = typeof (item as { id?: string }).id === "string" ? (item as { id: string }).id : ""; + const itemId = "id" in item && typeof item.id === "string" ? item.id : ""; const entry = (itemId ? runtime.openItems.get(itemId) : null) ?? runtime.openItemForEvent(rawEvent); const block = entry?.block ?? null; const contentIndex = entry?.contentIndex ?? output.content.length - 1; if (item.type === "reasoning" && block?.type === "thinking") { - block.thinking = item.summary?.map(summary => summary.text).join("\n\n") || ""; + block.thinking = finalizeReasoningThinking(item, block.thinking); block.thinkingSignature = JSON.stringify(item); stream.push({ type: "thinking_end", diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index 1d40b0ac5..a0ac83994 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -1684,6 +1684,14 @@ export function appendReasoningSummaryPart( item.summary.push(part); } +/** Chooses the final reasoning text without discarding content already streamed into the block. */ +export function finalizeReasoningThinking(item: ResponseReasoningItem, streamedThinking: string): string { + const summaryThinking = item.summary?.map(part => part.text).join("\n\n") ?? ""; + if (summaryThinking) return summaryThinking; + const contentThinking = item.content?.[0]?.type === "reasoning_text" ? (item.content[0].text ?? "") : ""; + return contentThinking || streamedThinking || ""; +} + export function appendReasoningSummaryTextDelta( item: ResponseReasoningItem, block: ThinkingContent, @@ -2208,12 +2216,6 @@ export async function processResponsesStream( ? lookupOpenItem({ output_index: event.output_index, item_id: item.id ?? item.call_id }) : lookupOpenItem({ output_index: event.output_index, item_id: item.id }); if (item.type === "reasoning") { - const thinking = - item.summary?.length > 0 - ? item.summary.map(part => part.text).join("\n\n") - : item.content?.[0]?.type === "reasoning_text" - ? (item.content[0].text ?? "") - : ""; // Prefer the routed entry; the bare itemId find misroutes when ids are // absent (`undefined === undefined` matches the FIRST thinking block) and // misses entirely when the done-event id drifts from the added-event id. @@ -2224,12 +2226,12 @@ export async function processResponsesStream( | ThinkingContent | undefined); if (reasoningBlock) { - reasoningBlock.thinking = thinking; + reasoningBlock.thinking = finalizeReasoningThinking(item, reasoningBlock.thinking); reasoningBlock.thinkingSignature = JSON.stringify(item); stream.push({ type: "thinking_end", contentIndex: contentIndexOf(reasoningBlock), - content: thinking, + content: reasoningBlock.thinking, partial: output, }); } diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index e7fa92c2e..505039c5c 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -442,6 +442,129 @@ describe("openai-codex streaming", () => { expect(capturedText).toEqual({ verbosity: "low" }); }); + it("preserves streamed reasoning when the done item has no summary text", async () => { + const token = createCodexTestToken(); + const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false }; + const events = [ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { + type: "response.reasoning_summary_part.added", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + { + type: "response.reasoning_summary_text.delta", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + delta: "streamed thinking", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] }, + }, + { + type: "response.content_part.added", + output_index: 1, + item_id: "msg_1", + part: { type: "output_text", text: "" }, + }, + { type: "response.output_text.delta", output_index: 1, item_id: "msg_1", delta: "done" }, + { + type: "response.output_item.done", + output_index: 1, + item: { + type: "message", + id: "msg_1", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "done" }], + }, + }, + { + type: "response.completed", + response: { + id: "resp_1", + status: "completed", + usage: { + input_tokens: 5, + output_tokens: 3, + total_tokens: 8, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + const sse = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; + const fetchMock: FetchImpl = async () => + new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); + + const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock, + }).result(); + + expect(result.content.find(block => block.type === "thinking")?.thinking).toBe("streamed thinking"); + }); + + it("streams raw reasoning text deltas into the final thinking block", async () => { + const token = createCodexTestToken(); + const model = { ...createCodexTestModel("https://chatgpt.com/backend-api"), preferWebsockets: false }; + const events = [ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "rs_raw", summary: [] }, + }, + { + type: "response.reasoning_text.delta", + output_index: 0, + item_id: "rs_raw", + delta: "raw streamed thinking", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "reasoning", id: "rs_raw", summary: [] }, + }, + { + type: "response.completed", + response: { + id: "resp_raw", + status: "completed", + usage: { + input_tokens: 5, + output_tokens: 3, + total_tokens: 8, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }, + ]; + const sse = `${events.map(event => `data: ${JSON.stringify(event)}`).join("\n\n")}\n\n`; + const fetchMock: FetchImpl = async () => + new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }); + + const result = await streamOpenAICodexResponses(model, createCodexTestContext(), { + apiKey: token, + fetch: fetchMock, + }).result(); + + expect(result.content.find(block => block.type === "thinking")?.thinking).toBe("raw streamed thinking"); + }); + it("maps end_turn=false on the terminal event to a pause_turn stop", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); setAgentDir(tempDir.path()); diff --git a/packages/ai/test/openai-responses-stream-terminal.test.ts b/packages/ai/test/openai-responses-stream-terminal.test.ts index 95c501834..63bc5f331 100644 --- a/packages/ai/test/openai-responses-stream-terminal.test.ts +++ b/packages/ai/test/openai-responses-stream-terminal.test.ts @@ -311,6 +311,49 @@ describe("processResponsesStream: lost output_item.added recovery", () => { expect(second.thinkingSignature).toBeDefined(); }); + test("preserves streamed reasoning when the done item has no summary text", async () => { + const output = makeOutput(); + const stream = { push: () => {}, end: () => {} } as never; + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { + type: "response.reasoning_summary_part.added", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + part: { type: "summary_text", text: "" }, + }, + { + type: "response.reasoning_summary_text.delta", + output_index: 0, + item_id: "rs_1", + summary_index: 0, + delta: "streamed thinking", + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "reasoning", id: "rs_1", summary: [] }, + }, + { type: "response.completed", response: { id: "resp_reasoning", status: "completed" } }, + ]), + output, + stream, + makeModel(), + ); + + const block = output.content[0]; + if (block?.type !== "thinking") throw new Error("expected a thinking block"); + expect(block.thinking).toBe("streamed thinking"); + expect(block.thinkingSignature).toBeDefined(); + }); + test("treats content_filter incomplete responses as errors, not length", async () => { const output = makeOutput(); const stream = { push: () => {}, end: () => {} } as never; From c8196b6560bc20c374e68e2b93ed00534c6a5152 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:21:27 +0200 Subject: [PATCH 02/59] fix(ai): bounded anthropic ping keepalive extension of idle watchdog A wedged Anthropic stream that kept emitting SSE ping keepalives reset the idle deadline on every ping, so an active tool-call stream (notably long write calls on Opus 4.8 high/xhigh) could hang forever with no timeout, no error, and no retry path. Pings now count as liveness only within a bounded window (3x the idle timeout) since the last semantic stream event; past it the idle watchdog fires and the turn surfaces a terminal stream-stall error that session-level auto-retry can recover. Pings before message_start still never consume the first-event watchdog, and pings bridging legitimate generation gaps within the window still keep slow streams alive. Fixes #4900 --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/providers/anthropic.ts | 34 ++- .../ai/test/anthropic-ping-keepalive.test.ts | 233 ++++++++++++++++++ 3 files changed, 263 insertions(+), 5 deletions(-) create mode 100644 packages/ai/test/anthropic-ping-keepalive.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index bf7f3af47..45a58da7e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -10,6 +10,7 @@ ### Fixed - Fixed OpenAI Codex/Responses reasoning streams so streamed thinking content is preserved when the final `output_item.done` reconstructs to an empty summary ([#4918](https://github.com/can1357/oh-my-pi/issues/4918)). +- Fixed Anthropic streams hanging forever when generation wedges mid-stream (notably long `write` tool calls on Opus 4.8 high/xhigh) while the server keeps sending `ping` keepalives: pings now extend the idle watchdog only within a bounded window (3x the idle timeout) since the last real stream event, so a stalled tool-call stream times out and recovers instead of hanging with no retry path ([#4900](https://github.com/can1357/oh-my-pi/issues/4900)). ## [16.3.12] - 2026-07-08 diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 6b3570c23..a22302fd2 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -1462,6 +1462,16 @@ async function* observeDecodedAnthropicSdkEvents( const PROVIDER_MAX_RETRIES = 10; +/** + * How long `ping` keepalives may keep extending the idle deadline without any + * semantic stream progress, as a multiple of the idle timeout. Anthropic pings + * across legitimate generation gaps, so pings count as liveness — but a wedged + * upstream that pings forever while producing no events must eventually trip + * the idle watchdog instead of hanging an active tool-call stream without a + * recovery path (#4900). + */ +const PING_PROGRESS_MAX_IDLE_MULTIPLIER = 3; + /** * Log a malformed-stream-envelope anomaly without aborting the turn. The strict * parser would `throw new AnthropicStreamEnvelopeError(...)` here; we instead @@ -2007,11 +2017,20 @@ const streamAnthropicOnce = ( } >(); - // Pings keep the idle deadline alive once content is flowing, but a - // ping before message_start must not consume the first-event watchdog: - // it would flip the (retryable) pre-content stall classification into - // a terminal mid-stream idle timeout. + // Pings keep the idle deadline alive once content is flowing (Anthropic + // bridges legitimate generation gaps with keepalives), but only within a + // bounded window: a wedged upstream that pings forever while the model + // produces nothing must still trip the idle watchdog, otherwise an + // active tool-call stream hangs unrecoverably with no retry (#4900). + // A ping before message_start must not consume the first-event watchdog + // either: it would flip the (retryable) pre-content stall classification + // into a terminal mid-stream idle timeout. let sawNonPingEvent = false; + let lastNonPingProgressAtMs = 0; + const pingProgressCapMs = + idleTimeoutMs !== undefined && idleTimeoutMs > 0 + ? idleTimeoutMs * PING_PROGRESS_MAX_IDLE_MULTIPLIER + : undefined; const timedAnthropicStream = iterateWithIdleTimeout(anthropicStream, { idleTimeoutMs, firstItemTimeoutMs: firstEventTimeoutMs, @@ -2021,8 +2040,13 @@ const streamAnthropicOnce = ( onFirstItemTimeout: () => activeAbortTracker.abortLocally(firstEventTimeoutAbortError), abortSignal: options?.signal, isProgressItem: item => { - if ((item as AnthropicStreamEvent).type === "ping") return sawNonPingEvent; + if ((item as AnthropicStreamEvent).type === "ping") { + if (!sawNonPingEvent) return false; + if (pingProgressCapMs === undefined) return true; + return Date.now() - lastNonPingProgressAtMs < pingProgressCapMs; + } sawNonPingEvent = true; + lastNonPingProgressAtMs = Date.now(); return true; }, }); diff --git a/packages/ai/test/anthropic-ping-keepalive.test.ts b/packages/ai/test/anthropic-ping-keepalive.test.ts new file mode 100644 index 000000000..a7e309b9e --- /dev/null +++ b/packages/ai/test/anthropic-ping-keepalive.test.ts @@ -0,0 +1,233 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { streamAnthropic } from "../src/providers/anthropic"; +import type { AnthropicMessagesClientLike } from "../src/providers/anthropic-client"; +import type { Context, Model } from "../src/types"; +import { waitForDelayOrAbort } from "./helpers"; + +const model: Model<"anthropic-messages"> = buildModel({ + id: "claude-opus-4-8", + name: "Claude Opus 4.8", + api: "anthropic-messages", + provider: "anthropic", + baseUrl: "https://api.anthropic.com", + reasoning: true, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, +}); + +const context: Context = { + messages: [{ role: "user", content: "write a file", timestamp: Date.now() }], +}; + +type MockAnthropicEvent = Record; + +/** `{ waitMs, event }` script step; `waitMs` elapses (fake clock) before the event is yielded. */ +type ScriptStep = { waitMs: number; event: MockAnthropicEvent | "hang-with-pings" }; + +const writeToolCallOpening: MockAnthropicEvent[] = [ + { + type: "message_start", + message: { + id: "msg_ping_keepalive", + usage: { input_tokens: 10, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 }, + }, + }, + { + type: "content_block_start", + index: 0, + content_block: { type: "tool_use", id: "toolu_ping_keepalive", name: "write", input: {} }, + }, + { + type: "content_block_delta", + index: 0, + delta: { type: "input_json_delta", partial_json: '{"path":"notes.md",' }, + }, +]; + +const writeToolCallClosing: MockAnthropicEvent[] = [ + { + type: "content_block_delta", + index: 0, + delta: { type: "input_json_delta", partial_json: '"content":"hello world"}' }, + }, + { type: "content_block_stop", index: 0 }, + { + type: "message_delta", + delta: { stop_reason: "tool_use" }, + usage: { input_tokens: 10, output_tokens: 6, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 }, + }, + { type: "message_stop" }, +]; + +function createScriptedClient( + script: ScriptStep[], + counters: { pings: number }, + onIteratorStart: () => void, +): AnthropicMessagesClientLike { + const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => { + const signal = requestOptions?.signal; + const response = new Response(null, { status: 200, headers: { "request-id": "req_ping_keepalive" } }); + const stream = { + async *[Symbol.asyncIterator]() { + onIteratorStart(); + for (const step of script) { + if (step.event === "hang-with-pings") { + // Wedged upstream: no semantic events ever again, but the edge + // keeps the SSE connection alive with keepalive pings. + while (true) { + await waitForDelayOrAbort(step.waitMs, signal); + counters.pings += 1; + yield { type: "ping" }; + } + } + if (step.waitMs > 0) { + await waitForDelayOrAbort(step.waitMs, signal); + } + if (step.event.type === "ping") counters.pings += 1; + yield step.event; + } + }, + }; + return { + async withResponse() { + return { data: stream, response, request_id: "req_ping_keepalive" }; + }, + } as never; + }) as unknown as AnthropicMessagesClientLike["messages"]["create"]; + return { messages: { create } } as AnthropicMessagesClientLike; +} + +async function drainMicrotasks(count: number): Promise { + for (let i = 0; i < count; i++) { + await Promise.resolve(); + } +} + +async function drainMicrotasksUntil(predicate: () => boolean, errorMessage: string): Promise { + for (let i = 0; i < 1000; i++) { + if (predicate()) return; + await Promise.resolve(); + } + throw new Error(errorMessage); +} + +afterEach(() => { + vi.useRealTimers(); + vi.restoreAllMocks(); +}); + +describe("anthropic ping keepalive idle cap", () => { + it("times out a stalled tool-call stream instead of letting pings extend it forever", async () => { + vi.useFakeTimers(); + const counters = { pings: 0 }; + let iteratorStarted = false; + const script: ScriptStep[] = [ + ...writeToolCallOpening.map(event => ({ waitMs: 0, event })), + { waitMs: 500, event: "hang-with-pings" as const }, + ]; + const client = createScriptedClient(script, counters, () => { + iteratorStarted = true; + }); + const providerRetryWait = vi.fn(async () => {}); + + let settled = false; + const resultPromise = streamAnthropic(model, context, { + client, + streamFirstEventTimeoutMs: 1_000, + streamIdleTimeoutMs: 1_000, + providerRetryWait, + }) + .result() + .then(message => { + settled = true; + return message; + }); + + await drainMicrotasksUntil(() => iteratorStarted, "Anthropic mock stream never started"); + await drainMicrotasks(30); + + // Pings arrive every 500 fake-ms while generation is wedged. Drive far + // past the bounded keepalive window (3x idle = 3_000ms) plus one idle + // budget; without the cap the idle deadline is reset by every ping and + // this loop ends with the result still pending (issue #4900's hang). + let stepsRun = 0; + for (let step = 0; step < 40 && !settled; step++) { + vi.advanceTimersByTime(500); + await drainMicrotasks(30); + stepsRun = step + 1; + } + + expect(settled).toBe(true); + // Cap (3_000ms) + idle budget (1_000ms) = fires at 3_500-4_000 fake ms. + expect(stepsRun).toBeLessThanOrEqual(9); + // Keepalives within the window were honored before the watchdog fired. + expect(counters.pings).toBeGreaterThanOrEqual(5); + + const result = await resultPromise; + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toBe("Anthropic stream stalled while waiting for the next event"); + // Mid-stream idle stalls are terminal for the provider loop (session-level + // auto-retry owns recovery); the provider must not silently re-request. + expect(providerRetryWait).not.toHaveBeenCalled(); + }); + + it("keeps a slow-but-alive stream open across ping-bridged gaps within the cap", async () => { + vi.useFakeTimers(); + const counters = { pings: 0 }; + let iteratorStarted = false; + // Silent generation gap of 1_800ms (> 1_000ms idle budget) bridged by + // pings at t=600 and t=1200, then semantic progress resumes and the + // tool call completes. Pings within the cap must count as liveness. + const script: ScriptStep[] = [ + ...writeToolCallOpening.map(event => ({ waitMs: 0, event })), + { waitMs: 600, event: { type: "ping" } }, + { waitMs: 600, event: { type: "ping" } }, + { waitMs: 600, event: writeToolCallClosing[0]! }, + ...writeToolCallClosing.slice(1).map(event => ({ waitMs: 0, event })), + ]; + const client = createScriptedClient(script, counters, () => { + iteratorStarted = true; + }); + const providerRetryWait = vi.fn(async () => {}); + + let settled = false; + const resultPromise = streamAnthropic(model, context, { + client, + streamFirstEventTimeoutMs: 1_000, + streamIdleTimeoutMs: 1_000, + providerRetryWait, + }) + .result() + .then(message => { + settled = true; + return message; + }); + + await drainMicrotasksUntil(() => iteratorStarted, "Anthropic mock stream never started"); + await drainMicrotasks(30); + + for (let step = 0; step < 30 && !settled; step++) { + vi.advanceTimersByTime(200); + await drainMicrotasks(30); + } + + expect(settled).toBe(true); + expect(counters.pings).toBe(2); + + const result = await resultPromise; + expect(result.errorMessage).toBeUndefined(); + expect(result.stopReason).toBe("toolUse"); + expect(providerRetryWait).not.toHaveBeenCalled(); + expect(JSON.parse(JSON.stringify(result.content))).toEqual([ + { + type: "toolCall", + id: "toolu_ping_keepalive", + name: "write", + arguments: { path: "notes.md", content: "hello world" }, + }, + ]); + }); +}); From 2553495d0096a59d39880ddad8a591954f47ba32 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:09:43 +0200 Subject: [PATCH 03/59] fix(tools): treated empty read/grep selector fields as omitted Models emit optional string args as empty strings; since ff3b0c795 (#4622) read/grep rejected a present-but-empty selector as invalid instead of behaving like an omitted one. Normalize empty and whitespace-only selector params to undefined before validation. Adopted from PR #4881 minus unrelated prompt churn. Fixes #4879 --- packages/coding-agent/CHANGELOG.md | 4 ++ packages/coding-agent/src/tools/grep.ts | 8 +-- packages/coding-agent/src/tools/read.ts | 9 ++-- packages/coding-agent/test/tools.test.ts | 64 ++++++++++++++++++++++++ 4 files changed, 75 insertions(+), 10 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index f9b8bb0ef..3190b6d40 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `read` and `grep` treating empty optional `selector` fields emitted by models as invalid selectors instead of behaving like omitted selectors. ([#4879](https://github.com/can1357/oh-my-pi/issues/4879)) + ## [16.3.12] - 2026-07-08 ### Added diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index 393078133..6f698a1b0 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -158,11 +158,11 @@ async function parsePathSpecs( cwd: string, explicitSelector?: string, ): Promise { - const explicitRanges = - explicitSelector === undefined || explicitSelector.length === 0 ? undefined : parseLineRanges(explicitSelector); - if (explicitSelector !== undefined && !explicitRanges) { + const normalizedSelector = explicitSelector?.trim() || undefined; + const explicitRanges = normalizedSelector === undefined ? undefined : parseLineRanges(normalizedSelector); + if (normalizedSelector !== undefined && !explicitRanges) { throw new ToolError( - `selector "${explicitSelector}" is invalid — use line ranges like "50-100", "50+10", or "50-100,200-300" without a leading colon`, + `selector "${normalizedSelector}" is invalid — use line ranges like "50-100", "50+10", or "50-100,200-300" without a leading colon`, ); } const specs: GrepPathSpec[] = []; diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 14edfad2b..b8b3b42fe 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -2118,13 +2118,10 @@ export class ReadTool implements AgentTool { _toolContext?: AgentToolContext, ): Promise> { let { path: readPath } = params; - let explicitSelector = params.selector?.trim(); + let explicitSelector = params.selector?.trim() || undefined; let explicitParsedSelector = explicitSelector === undefined ? undefined : parseSel(explicitSelector); - if ( - params.selector !== undefined && - (explicitSelector === undefined || explicitSelector.length === 0 || explicitParsedSelector?.kind === "none") - ) { - throw invalidSelector(params.selector); + if (explicitSelector !== undefined && explicitParsedSelector?.kind === "none") { + throw invalidSelector(explicitSelector); } if (readPath.startsWith("file://")) { readPath = expandPath(readPath); diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index a3baf9bbd..90c001c02 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -330,6 +330,34 @@ describe("Coding Agent Tools", () => { expect(result.details?.truncation).toBeUndefined(); }); + it("treats empty optional selector as omitted for read", async () => { + const testFile = path.join(testDir, "read-empty-selector.txt"); + const content = "alpha\nselector target\nomega"; + fs.writeFileSync(testFile, content); + + const omitted = getTextOutput(await readTool.execute("test-read-empty-selector-omitted", { path: testFile })); + expect(omitted).toContain("alpha"); + expect(omitted).toContain("selector target"); + expect(omitted).toContain("omega"); + + for (const { name, selector } of [ + { name: "empty", selector: "" }, + { name: "whitespace", selector: " \t\n " }, + ]) { + const withOptionalSelector = getTextOutput( + await readTool.execute(`test-read-empty-selector-${name}`, { + path: testFile, + selector, + }), + ); + expect(withOptionalSelector).toBe(omitted); + } + + await expect( + readTool.execute("test-read-empty-selector-malformed", { path: testFile, selector: "-100" }), + ).rejects.toThrow(/Invalid selector/); + }); + it("truncates lines wider than the read column cap, leaving narrow lines untouched", async () => { const wideLine = "x".repeat(1500); const testFile = path.join(testDir, "wide.txt"); @@ -1649,6 +1677,42 @@ function b() { expect(output).toMatch(/\*2\|match line/); }); + it("treats empty optional selector as omitted for search", async () => { + const testFile = path.join(testDir, "grep-empty-selector.txt"); + fs.writeFileSync(testFile, "before\nneedle empty selector\nbetween\nneedle whitespace selector\nafter"); + + const omitted = getTextOutput( + await searchTool.execute("test-search-empty-selector-omitted", { + pattern: "needle", + path: testFile, + }), + ); + expect(omitted).toMatch(/\*2\|needle empty selector/); + expect(omitted).toMatch(/\*4\|needle whitespace selector/); + + for (const { name, selector } of [ + { name: "empty", selector: "" }, + { name: "whitespace", selector: " \t\n " }, + ]) { + const withOptionalSelector = getTextOutput( + await searchTool.execute(`test-search-empty-selector-${name}`, { + pattern: "needle", + path: testFile, + selector, + }), + ); + expect(withOptionalSelector).toBe(omitted); + } + + await expect( + searchTool.execute("test-search-empty-selector-malformed", { + pattern: "needle", + path: testFile, + selector: "not-a-range", + }), + ).rejects.toThrow(/selector "not-a-range" is invalid/); + }); + it("flags a zero-match search as contextually useless", async () => { fs.writeFileSync(path.join(testDir, "plain.txt"), "nothing interesting here\n"); From f9b425f4b7a3a2a452dd9bd5815ad2e914f1c0b5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 07:35:45 +0000 Subject: [PATCH 04/59] fix(tool): allowed grep directory line selectors - Applied explicit grep selectors as per-file line filters for directory and glob searches instead of pre-validating them as single files. - Clarified the grep selector prompt/schema language and added regression coverage for directory searches. Fixes #4898 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/prompts/system/workflow-notice.md | 10 ++-- .../coding-agent/src/prompts/tools/grep.md | 3 +- packages/coding-agent/src/tools/grep.ts | 20 +++++-- .../tools/grep-directory-selector.test.ts | 57 +++++++++++++++++++ 5 files changed, 81 insertions(+), 10 deletions(-) create mode 100644 packages/coding-agent/test/tools/grep-directory-selector.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 3190b6d40..4cae3243e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed `read` and `grep` treating empty optional `selector` fields emitted by models as invalid selectors instead of behaving like omitted selectors. ([#4879](https://github.com/can1357/oh-my-pi/issues/4879)) +- Fixed `grep` explicit line selectors on directory searches so they filter each matched file by line number instead of aborting with a single-file-only error ([#4898](https://github.com/can1357/oh-my-pi/issues/4898)). ## [16.3.12] - 2026-07-08 diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 74eddee21..3e62ca5b0 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -51,20 +51,20 @@ Decompose first, then {{#if taskBatch}}batch the independent leaves{{else}}issue {{#if taskBatch}} task( - context: "# Goal\nReview the auth diff...\n# Constraints\nRead-only...\n# Contract\nReturn findings as severity/file/line/fix...", + context: "# Goal\nReview the auth diff…\n# Constraints\nRead-only…\n# Contract\nReturn findings as severity/file/line/fix…", tasks: [ - { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection...\n# Acceptance\nReturn confirmed findings only..." }, - { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance...\n# Acceptance\nReturn mismatches and exact prompt lines..." }, + { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection…\n# Acceptance\nReturn confirmed findings only…" }, + { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance…\n# Acceptance\nReturn mismatches and exact prompt lines…" }, ] ) {{else}} task( role: "Auth Storage Reviewer", - assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only..." + assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only…" ) task( role: "Prompt Contract Reviewer", - assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only..." + assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only…" ) {{/if}} diff --git a/packages/coding-agent/src/prompts/tools/grep.md b/packages/coding-agent/src/prompts/tools/grep.md index eef17d10e..cc8c3f312 100644 --- a/packages/coding-agent/src/prompts/tools/grep.md +++ b/packages/coding-agent/src/prompts/tools/grep.md @@ -2,7 +2,8 @@ Greps files using regex. - Rust regex (RE2-style): alternation is `foo|bar`, not GNU BRE-style `foo\|bar`; Rust word boundaries like `\bword\b` are supported. Use line anchors or post-filters instead of lookaround/backreferences. -- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). Literal colon filename + line range? Use `selector` (e.g. `{"path":"test:1-2","selector":"1-2"}`), not recursive `path:"test:1-2:1-2"`. +- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). Use `selector` only for line-number filtering, never path/root selection (`"/"` belongs in `path`). +- Literal colon filename + line range? Use `selector` (e.g. `{"path":"test:1-2","selector":"1-2"}`), not recursive `path:"test:1-2:1-2"`. - Cross-line patterns detected from literal `\n` or `\\n` in `pattern`. diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index 6f698a1b0..c57410d97 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -80,7 +80,7 @@ const searchSchema = type({ 'file, directory, glob, internal URL, or ":" selector to search; pass several as a semicolon-delimited list ("src; tests"). Omitted -> searches the workspace root (".")', ), "selector?": type("string").describe( - 'line selector without a leading colon (e.g. "50-100", "50+10", "50-100,200-300"); keeps `path` literal when filenames contain colons', + 'line selector applied to every searched file (e.g. "50-100", "50+10", "50-100,200-300"); never a path like "/"', ), "case?": type("boolean").describe("case-sensitive search"), "gitignore?": type("boolean").describe("respect gitignore"), @@ -126,6 +126,7 @@ interface GrepPathSpec { clean: string; literalFilesystemMatch?: boolean; ranges?: [LineRange, ...LineRange[]]; + rangeSource?: "explicit" | "path"; } /** @@ -182,6 +183,7 @@ async function parsePathSpecs( clean: literalMatch && !rawPathHasScheme ? resolveReadPath(entry, cwd) : entry, literalFilesystemMatch: literalMatch, ranges: explicitRanges, + rangeSource: "explicit", }); continue; } @@ -210,6 +212,7 @@ async function parsePathSpecs( const literalFilesystemMatch = strictSplit.sel !== undefined && split.sel === undefined; let clean = literalFilesystemMatch ? resolveReadPath(entry, cwd) : entry; let ranges: [LineRange, ...LineRange[]] | undefined; + let rangeSource: "path" | undefined; if (!literalFilesystemMatch && split.sel) { const parsed = parseLineRanges(split.sel); if (!parsed) { @@ -222,8 +225,15 @@ async function parsePathSpecs( } clean = split.path; ranges = parsed; + rangeSource = "path"; } - specs.push({ original: entry, clean, literalFilesystemMatch, ranges }); + specs.push({ + original: entry, + clean, + literalFilesystemMatch, + ranges, + rangeSource: ranges ? rangeSource : undefined, + }); } return specs; } @@ -971,6 +981,7 @@ export class GrepTool implements AgentTool const searchablePaths = internalResolution.paths; const { virtualResources, virtualPathSet, virtualInputIndexes } = internalResolution; const rangesByAbsPath = new Map(); + const globalRanges = pathSpecs.find(spec => spec.rangeSource === "explicit")?.ranges; if ( archiveUnreadable.length > 0 && @@ -1030,6 +1041,7 @@ export class GrepTool implements AgentTool for (let idx = 0; idx < pathSpecs.length; idx++) { const spec = pathSpecs[idx]; if (!spec.ranges) continue; + if (spec.rangeSource === "explicit") continue; if (virtualInputIndexes.has(idx)) continue; const resolved = internalResolution.resolvedPathsByInput[idx]; if (!resolved) continue; @@ -1223,11 +1235,11 @@ export class GrepTool implements AgentTool throw err; } result = mergeGrepResults(result, virtualResult, INTERNAL_TOTAL_CAP); - if (rangesByAbsPath.size > 0) { + if (rangesByAbsPath.size > 0 || globalRanges) { const filteredMatches: GrepMatch[] = []; for (const match of result.matches) { const abs = matchAbsolutePath(match.path, searchPath); - const ranges = rangesByAbsPath.get(abs); + const ranges = rangesByAbsPath.get(abs) ?? globalRanges; if (!ranges) { // Path has no line-range constraint (e.g. a peer entry without `:N-M`). filteredMatches.push(match); diff --git a/packages/coding-agent/test/tools/grep-directory-selector.test.ts b/packages/coding-agent/test/tools/grep-directory-selector.test.ts new file mode 100644 index 000000000..d2bffb27c --- /dev/null +++ b/packages/coding-agent/test/tools/grep-directory-selector.test.ts @@ -0,0 +1,57 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { GrepTool } from "@oh-my-pi/pi-coding-agent/tools/grep"; +import { removeWithRetries } from "@oh-my-pi/pi-utils"; + +function resultText(result: { content: Array<{ type: string; text?: string }> }): string { + return result.content + .filter(entry => entry.type === "text") + .map(entry => entry.text ?? "") + .join("\n"); +} + +describe("grep explicit line selector on directory searches", () => { + let testDir: string; + + beforeEach(async () => { + testDir = await fs.mkdtemp(path.join(os.tmpdir(), "grep-directory-selector-")); + }); + + afterEach(async () => { + await removeWithRetries(testDir); + }); + + function createSession(): ToolSession { + return { + cwd: testDir, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: Settings.isolated({ "grep.contextBefore": 0, "grep.contextAfter": 0 }), + }; + } + + it("filters matches by per-file line number instead of rejecting the directory", async () => { + const appDir = path.join(testDir, "scripts", "app"); + await fs.mkdir(appDir, { recursive: true }); + await Bun.write(path.join(appDir, "one.ts"), "outside one\ninside one\noutside one again\n"); + await Bun.write(path.join(appDir, "two.ts"), "outside two\ninside two\noutside two again\n"); + + const result = await new GrepTool(createSession()).execute("grep-directory-selector", { + pattern: "inside|outside", + path: "scripts/app", + selector: "2-2", + }); + + const text = resultText(result); + expect(text).toContain("inside one"); + expect(text).toContain("inside two"); + expect(text).not.toContain("outside one"); + expect(text).not.toContain("outside two"); + expect(text).not.toContain("Line-range selector requires a single file"); + }); +}); From 81a7749836cff4d43f725e9254afa29e66888c1a Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 07:52:27 +0000 Subject: [PATCH 05/59] fix(tool): fetched ranged grep matches before filtering - Raised or removed native grep pre-filter caps when line selectors are present so later selected lines are available to the post-filter. - Added coverage for directory selectors beyond the normal multi-file per-file cap. --- packages/coding-agent/src/tools/grep.ts | 28 ++++++++++++++++--- .../tools/grep-directory-selector.test.ts | 21 ++++++++++++++ 2 files changed, 45 insertions(+), 4 deletions(-) diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index c57410d97..3cb0ce2df 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -410,6 +410,18 @@ function lineAllowed(lineNumber: number, ranges: readonly LineRange[] | undefine return !ranges || isLineInRanges(lineNumber, ranges); } +function lineRangeFetchEnd(pathSpecs: readonly GrepPathSpec[]): number | undefined { + let maxEnd = 0; + for (const spec of pathSpecs) { + if (!spec.ranges) continue; + for (const range of spec.ranges) { + if (range.endLine === undefined) return undefined; + maxEnd = Math.max(maxEnd, range.endLine); + } + } + return maxEnd; +} + /** Binary search for the index of the line containing byte `offset`. */ function findLineIndex(starts: readonly number[], offset: number): number { if (starts.length === 0) return -1; @@ -1107,6 +1119,14 @@ export class GrepTool implements AgentTool Boolean(multiTargets) || (virtualResources.length > 0 && (virtualResources.length > 1 || searchablePaths.length > 0)); const perFileMatchCap = isMultiScope ? MULTI_FILE_PER_FILE_MATCHES : SINGLE_FILE_MATCHES; + const hasLineRangeFilters = pathSpecs.some(spec => spec.ranges); + const lineRangeMatchFetchEnd = hasLineRangeFilters ? lineRangeFetchEnd(pathSpecs) : undefined; + const nativeMaxCount = hasLineRangeFilters ? undefined : INTERNAL_TOTAL_CAP; + const nativeMaxCountPerFile = hasLineRangeFilters + ? lineRangeMatchFetchEnd === undefined + ? undefined + : Math.max(perFileMatchCap + 1, lineRangeMatchFetchEnd) + : perFileMatchCap + 1; // Run grep let result: GrepResult = { @@ -1141,12 +1161,12 @@ export class GrepTool implements AgentTool multiline: effectiveMultiline, hidden: true, gitignore: useGitignore, - maxCount: INTERNAL_TOTAL_CAP, + maxCount: nativeMaxCount, contextBefore: normalizedContextBefore, contextAfter: normalizedContextAfter, maxColumns: DEFAULT_MAX_COLUMN, mode: effectiveOutputMode, - maxCountPerFile: perFileMatchCap + 1, + maxCountPerFile: nativeMaxCountPerFile, signal, timeoutMs: SEARCH_GREP_TIMEOUT_MS, }, @@ -1188,12 +1208,12 @@ export class GrepTool implements AgentTool multiline: effectiveMultiline, hidden: true, gitignore: useGitignore, - maxCount: INTERNAL_TOTAL_CAP, + maxCount: nativeMaxCount, contextBefore: normalizedContextBefore, contextAfter: normalizedContextAfter, maxColumns: DEFAULT_MAX_COLUMN, mode: effectiveOutputMode, - maxCountPerFile: perFileMatchCap + 1, + maxCountPerFile: nativeMaxCountPerFile, signal, timeoutMs: SEARCH_GREP_TIMEOUT_MS, }, diff --git a/packages/coding-agent/test/tools/grep-directory-selector.test.ts b/packages/coding-agent/test/tools/grep-directory-selector.test.ts index d2bffb27c..6ce7bcd07 100644 --- a/packages/coding-agent/test/tools/grep-directory-selector.test.ts +++ b/packages/coding-agent/test/tools/grep-directory-selector.test.ts @@ -54,4 +54,25 @@ describe("grep explicit line selector on directory searches", () => { expect(text).not.toContain("outside two"); expect(text).not.toContain("Line-range selector requires a single file"); }); + + it("fetches enough directory matches before applying an explicit later line selector", async () => { + const appDir = path.join(testDir, "scripts", "hot"); + await fs.mkdir(appDir, { recursive: true }); + const content = `${Array.from( + { length: 30 }, + (_, index) => `cap-needle line ${String(index + 1).padStart(2, "0")}`, + ).join("\n")}\n`; + await Bun.write(path.join(appDir, "many.ts"), content); + + const result = await new GrepTool(createSession()).execute("grep-directory-selector-cap", { + pattern: "cap-needle", + path: "scripts/hot", + selector: "25-25", + }); + + const text = resultText(result); + expect(text).toContain("cap-needle line 25"); + expect(text).not.toContain("cap-needle line 24"); + expect(text).not.toContain("cap-needle line 26"); + }); }); From 029720298946948426701ccbad704c6f914ff8da Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:17:01 +0200 Subject: [PATCH 06/59] fix(tool): bounded ranged grep native fetch budgets - Replaced the unbounded native fetch (no total cap; no per-file cap for open-ended ranges) adopted from PR #4903 with finite budgets: per-file fetch covers bounded ranges up to endLine and open-ended ranges up to startLine-1 plus the kept window, clamped to the native file-size ceiling; the global ceiling scales by the same amplification. - Threaded the scaled ceiling through mergeGrepResults so mixed native+virtual ranged searches are not re-truncated pre-filter. - Dropped the unrelated workflow-notice.md ellipsis churn from the PR. - Added open-ended directory selector coverage. Fixes #4898 --- .../src/prompts/system/workflow-notice.md | 10 +++--- packages/coding-agent/src/tools/grep.ts | 35 +++++++++++++------ .../tools/grep-directory-selector.test.ts | 21 +++++++++++ 3 files changed, 50 insertions(+), 16 deletions(-) diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 3e62ca5b0..74eddee21 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -51,20 +51,20 @@ Decompose first, then {{#if taskBatch}}batch the independent leaves{{else}}issue {{#if taskBatch}} task( - context: "# Goal\nReview the auth diff…\n# Constraints\nRead-only…\n# Contract\nReturn findings as severity/file/line/fix…", + context: "# Goal\nReview the auth diff...\n# Constraints\nRead-only...\n# Contract\nReturn findings as severity/file/line/fix...", tasks: [ - { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection…\n# Acceptance\nReturn confirmed findings only…" }, - { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance…\n# Acceptance\nReturn mismatches and exact prompt lines…" }, + { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection...\n# Acceptance\nReturn confirmed findings only..." }, + { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance...\n# Acceptance\nReturn mismatches and exact prompt lines..." }, ] ) {{else}} task( role: "Auth Storage Reviewer", - assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only…" + assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only..." ) task( role: "Prompt Contract Reviewer", - assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only…" + assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only..." ) {{/if}} diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index 3cb0ce2df..35c1d7368 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -410,16 +410,25 @@ function lineAllowed(lineNumber: number, ranges: readonly LineRange[] | undefine return !ranges || isLineInRanges(lineNumber, ranges); } -function lineRangeFetchEnd(pathSpecs: readonly GrepPathSpec[]): number | undefined { - let maxEnd = 0; +/** + * Per-file native fetch budget that guarantees the JS range filter can still + * surface `perFileKeep` in-range hits. Matches arrive one entry per matched + * line in line order, so a bounded range's hits all sit within the first + * `endLine` entries, and an open-ended range starting at S is preceded by at + * most S-1 out-of-range entries — S-1+perFileKeep entries cover the kept + * window or exhaust the file. Clamped to the native file-size ceiling (a + * ≤4 MiB file cannot have more matched lines than bytes), which also keeps + * the scaled global budget inside the native layer's u32 bounds. + */ +function lineRangeFetchCap(pathSpecs: readonly GrepPathSpec[], perFileKeep: number): number { + let cap = 0; for (const spec of pathSpecs) { if (!spec.ranges) continue; for (const range of spec.ranges) { - if (range.endLine === undefined) return undefined; - maxEnd = Math.max(maxEnd, range.endLine); + cap = Math.max(cap, range.endLine ?? range.startLine - 1 + perFileKeep); } } - return maxEnd; + return Math.min(cap, NATIVE_GREP_MAX_FILE_BYTES); } /** Binary search for the index of the line containing byte `offset`. */ @@ -1119,14 +1128,18 @@ export class GrepTool implements AgentTool Boolean(multiTargets) || (virtualResources.length > 0 && (virtualResources.length > 1 || searchablePaths.length > 0)); const perFileMatchCap = isMultiScope ? MULTI_FILE_PER_FILE_MATCHES : SINGLE_FILE_MATCHES; + // Range filtering happens in JS after the native fetch, so out-of-range + // matches consume fetch budget. Widen the per-file budget just enough + // that filtering can still yield `perFileMatchCap` in-range hits, and + // scale the global safety ceiling by the same amplification so ranged + // searches keep the baseline file coverage while staying finite. const hasLineRangeFilters = pathSpecs.some(spec => spec.ranges); - const lineRangeMatchFetchEnd = hasLineRangeFilters ? lineRangeFetchEnd(pathSpecs) : undefined; - const nativeMaxCount = hasLineRangeFilters ? undefined : INTERNAL_TOTAL_CAP; const nativeMaxCountPerFile = hasLineRangeFilters - ? lineRangeMatchFetchEnd === undefined - ? undefined - : Math.max(perFileMatchCap + 1, lineRangeMatchFetchEnd) + ? Math.max(perFileMatchCap + 1, lineRangeFetchCap(pathSpecs, perFileMatchCap + 1)) : perFileMatchCap + 1; + const nativeMaxCount = hasLineRangeFilters + ? Math.ceil(INTERNAL_TOTAL_CAP / (perFileMatchCap + 1)) * nativeMaxCountPerFile + : INTERNAL_TOTAL_CAP; // Run grep let result: GrepResult = { @@ -1254,7 +1267,7 @@ export class GrepTool implements AgentTool } throw err; } - result = mergeGrepResults(result, virtualResult, INTERNAL_TOTAL_CAP); + result = mergeGrepResults(result, virtualResult, nativeMaxCount); if (rangesByAbsPath.size > 0 || globalRanges) { const filteredMatches: GrepMatch[] = []; for (const match of result.matches) { diff --git a/packages/coding-agent/test/tools/grep-directory-selector.test.ts b/packages/coding-agent/test/tools/grep-directory-selector.test.ts index 6ce7bcd07..f164fc20b 100644 --- a/packages/coding-agent/test/tools/grep-directory-selector.test.ts +++ b/packages/coding-agent/test/tools/grep-directory-selector.test.ts @@ -75,4 +75,25 @@ describe("grep explicit line selector on directory searches", () => { expect(text).not.toContain("cap-needle line 24"); expect(text).not.toContain("cap-needle line 26"); }); + + it("supports open-ended selectors on directories with a finite fetch budget", async () => { + const appDir = path.join(testDir, "scripts", "tail"); + await fs.mkdir(appDir, { recursive: true }); + const content = `${Array.from( + { length: 60 }, + (_, index) => `open-needle line ${String(index + 1).padStart(2, "0")}`, + ).join("\n")}\n`; + await Bun.write(path.join(appDir, "long.ts"), content); + + const result = await new GrepTool(createSession()).execute("grep-directory-selector-open", { + pattern: "open-needle", + path: "scripts/tail", + selector: "35-", + }); + + const text = resultText(result); + expect(text).toContain("open-needle line 35"); + expect(text).not.toContain("open-needle line 34"); + expect(text).not.toContain("Line-range selector requires a single file"); + }); }); From ba91877b6ec59d57c97a83973cf782fe1b4e6d34 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:08:39 +0200 Subject: [PATCH 07/59] fix(tui): restored read selector previews for explicit selector args The v16.3.12 explicit `selector` field (ff3b0c795c) was consumed by the read tool but never threaded into the TUI renderers: ReadRenderArgs in both readToolRenderer (read.ts) and ReadToolGroupComponent only derived selectors from path-embedded `:sel` suffixes, so split-arg calls like { path, selector: "2-3" } rendered bare paths without line ranges or raw modifiers. Joined the explicit selector (trimmed, leading colons stripped, non-string guarded) back onto the display path in renderCall, renderResult error and success branches, and the grouped read summary, keeping hyperlinks on the base path only. Adopted from PR #4904 (both commits squashed), minus its unrelated workflow-notice.md prompt churn. Fixes #4899 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/components/read-tool-group.ts | 6 ++- packages/coding-agent/src/tools/read.ts | 37 +++++++++++++---- .../coding-agent/test/read-tool-group.test.ts | 39 ++++++++++++++++++ .../test/tools/read-renderer.test.ts | 40 +++++++++++++++++++ 5 files changed, 115 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 4cae3243e..bd86fa43c 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -6,6 +6,7 @@ - Fixed `read` and `grep` treating empty optional `selector` fields emitted by models as invalid selectors instead of behaving like omitted selectors. ([#4879](https://github.com/can1357/oh-my-pi/issues/4879)) - Fixed `grep` explicit line selectors on directory searches so they filter each matched file by line number instead of aborting with a single-file-only error ([#4898](https://github.com/can1357/oh-my-pi/issues/4898)). +- Fixed Read tool previews dropping explicit `selector` arguments, so line ranges and `raw` modifiers render in terminal read call titles again ([#4899](https://github.com/can1357/oh-my-pi/issues/4899)). ## [16.3.12] - 2026-07-08 diff --git a/packages/coding-agent/src/modes/components/read-tool-group.ts b/packages/coding-agent/src/modes/components/read-tool-group.ts index c2d1f1248..9d8c9f2f5 100644 --- a/packages/coding-agent/src/modes/components/read-tool-group.ts +++ b/packages/coding-agent/src/modes/components/read-tool-group.ts @@ -37,6 +37,7 @@ export function readArgsTargetInternalUrl(args: unknown): boolean { type ReadRenderArgs = { path?: string; file_path?: string; + selector?: string; // Legacy field from the old schema; tolerated for rebuilt transcripts. sel?: string; }; @@ -344,7 +345,10 @@ export class ReadToolGroupComponent extends Container implements ToolExecutionHa updateArgs(args: ReadRenderArgs, toolCallId?: string): void { if (!toolCallId) return; const basePath = args.file_path || args.path || ""; - const rawPath = args.sel ? `${basePath}:${args.sel}` : basePath; + const rawSelector = + typeof args.selector === "string" ? args.selector : typeof args.sel === "string" ? args.sel : undefined; + const selector = rawSelector?.trim().replace(/^:+/, ""); + const rawPath = selector && selector.length > 0 ? `${basePath}:${selector}` : basePath; const entry: ReadEntry = this.#entries.get(toolCallId) ?? { toolCallId, path: rawPath, diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index b8b3b42fe..511e59086 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -3311,6 +3311,7 @@ export class ReadTool implements AgentTool { interface ReadRenderArgs { path?: unknown; file_path?: unknown; + selector?: unknown; sel?: string; // Legacy fields from old schema — tolerated for in-flight tool calls during transition offset?: number; @@ -3375,10 +3376,20 @@ function formatReadPathLink( export const readToolRenderer = { renderCall(args: ReadRenderArgs, _options: RenderResultOptions, uiTheme: Theme): Component { - const rawPath = + const baseRawPath = typeof args.file_path === "string" ? args.file_path : typeof args.path === "string" ? args.path : ""; - if (isReadableUrlPath(rawPath)) { - return renderReadUrlCall({ path: rawPath, raw: args.raw }, _options, uiTheme); + const explicitSelector = + typeof args.selector === "string" + ? args.selector.trim().replace(/^:+/, "") + : args.sel?.trim().replace(/^:+/, ""); + const rawPath = + explicitSelector && explicitSelector.length > 0 ? `${baseRawPath}:${explicitSelector}` : baseRawPath; + if (isReadableUrlPath(baseRawPath)) { + return renderReadUrlCall( + { path: rawPath, raw: args.raw || explicitSelector?.toLowerCase() === "raw" }, + _options, + uiTheme, + ); } const offset = args.offset; @@ -3402,9 +3413,9 @@ export const readToolRenderer = { args?: ReadRenderArgs, ): Component { const urlDetails = result.details as ReadUrlToolDetails | undefined; - const rawPathForKind = + const baseRawPathForKind = typeof args?.file_path === "string" ? args.file_path : typeof args?.path === "string" ? args.path : ""; - if (urlDetails?.kind === "url" || isReadableUrlPath(rawPathForKind)) { + if (urlDetails?.kind === "url" || isReadableUrlPath(baseRawPathForKind)) { return renderReadUrlResult( result as { content: Array<{ type: string; text?: string }>; @@ -3419,8 +3430,14 @@ export const readToolRenderer = { if (result.isError) { const rawErrorText = result.content?.find(c => c.type === "text")?.text ?? ""; const errorText = (rawErrorText || "Unknown error").replace(/^Error:\s*/, ""); - const rawPath = + const baseRawPath = typeof args?.file_path === "string" ? args.file_path : typeof args?.path === "string" ? args.path : ""; + const explicitSelector = + typeof args?.selector === "string" + ? args.selector.trim().replace(/^:+/, "") + : args?.sel?.trim().replace(/^:+/, ""); + const rawPath = + explicitSelector && explicitSelector.length > 0 ? `${baseRawPath}:${explicitSelector}` : baseRawPath; const filePath = formatReadPathLink(rawPath, { offset: args?.offset, sourcePath: readSourceFsPath(result.details) }) || shortenPath(rawPath); @@ -3447,8 +3464,14 @@ export const readToolRenderer = { // echo next to the styled warning line below. const contentText = details?.displayContent?.text ?? stripOutputNotice(rawText, details?.meta); const imageContent = result.content?.find(c => c.type === "image"); - const rawPath = + const baseRawPath = typeof args?.file_path === "string" ? args.file_path : typeof args?.path === "string" ? args.path : ""; + const explicitSelector = + typeof args?.selector === "string" + ? args.selector.trim().replace(/^:+/, "") + : args?.sel?.trim().replace(/^:+/, ""); + const rawPath = + explicitSelector && explicitSelector.length > 0 ? `${baseRawPath}:${explicitSelector}` : baseRawPath; const renderPath = splitReadRenderPath(rawPath); const lang = getLanguageFromPath(renderPath.path); diff --git a/packages/coding-agent/test/read-tool-group.test.ts b/packages/coding-agent/test/read-tool-group.test.ts index e5b6c43da..f1d6da276 100644 --- a/packages/coding-agent/test/read-tool-group.test.ts +++ b/packages/coding-agent/test/read-tool-group.test.ts @@ -250,6 +250,45 @@ describe("ReadToolGroupComponent", () => { expect(extractLinkTexts(rendered)).not.toContain("src/example.ts:7-9"); }); + it("renders separate selector grouped summary paths while linking only the base path", () => { + settings.override("tui.hyperlinks", "always"); + const component = new ReadToolGroupComponent(); + const resolvedPath = path.resolve("/workspace/src/grouped.ts"); + component.updateArgs({ path: "src/grouped.ts", selector: "2-3" }, "read-split-selector"); + component.updateResult( + { + content: [{ type: "text", text: "line 2" }], + details: { meta: { source: { type: "path", value: resolvedPath } } }, + }, + false, + "read-split-selector", + ); + + const rendered = component.render(120).join("\n"); + + const groupedUri = new URL(url.pathToFileURL(path.resolve(resolvedPath)).href); + groupedUri.searchParams.set("line", "2"); + expect(Bun.stripANSI(rendered)).toContain("Read src/grouped.ts:2-3"); + expect(extractLinkUris(rendered)).toContain(groupedUri.href); + expect(extractLinkTexts(rendered)).toContain("src/grouped.ts"); + expect(extractLinkTexts(rendered)).not.toContain("src/grouped.ts:2-3"); + }); + + it("ignores non-string selectors from malformed runtime args", () => { + const component = new ReadToolGroupComponent(); + const malformedArgs = { path: "src/example.ts", selector: 10 } as unknown as { + path: string; + selector: string; + }; + + expect(() => component.updateArgs(malformedArgs, "read-malformed-selector")).not.toThrow(); + + const plain = Bun.stripANSI(component.render(120).join("\n")); + + expect(plain).toContain("Read src/example.ts"); + expect(plain).not.toContain("src/example.ts:10"); + }); + it("links inline preview titles when the summary row is suppressed", () => { settings.override("tui.hyperlinks", "always"); const component = new ReadToolGroupComponent({ showContentPreview: true }); diff --git a/packages/coding-agent/test/tools/read-renderer.test.ts b/packages/coding-agent/test/tools/read-renderer.test.ts index babab0ba0..2a128bd34 100644 --- a/packages/coding-agent/test/tools/read-renderer.test.ts +++ b/packages/coding-agent/test/tools/read-renderer.test.ts @@ -83,6 +83,46 @@ describe("readToolRenderer hyperlinks", () => { expect(extractLinkTexts(rendered)).not.toContain(`${examplePath}:10-12`); }); + it("renders separate selector read call paths while linking only the base path", async () => { + settings.override("tui.hyperlinks", "always"); + const theme = await getThemeByName("dark"); + expect(theme).toBeDefined(); + + const examplePath = path.resolve("/tmp/omp-read/separate-selector.ts"); + const component = readToolRenderer.renderCall( + { path: examplePath, selector: "10-12" }, + { expanded: false, isPartial: false }, + theme!, + ); + + const rendered = component.render(200).join("\n"); + expect(Bun.stripANSI(rendered)).toContain(`${examplePath}:10-12`); + const exampleUri = new URL(url.pathToFileURL(path.resolve(examplePath)).href); + exampleUri.searchParams.set("line", "10"); + expect(extractLinkUris(rendered)).toContain(exampleUri.href); + expect(extractLinkTexts(rendered)).toContain(examplePath); + expect(extractLinkTexts(rendered)).not.toContain(`${examplePath}:10-12`); + }); + + it("renders separate raw read selectors while linking only the base path", async () => { + settings.override("tui.hyperlinks", "always"); + const theme = await getThemeByName("dark"); + expect(theme).toBeDefined(); + + const examplePath = path.resolve("/tmp/omp-read/raw-selector.ts"); + const component = readToolRenderer.renderCall( + { path: examplePath, selector: "raw" }, + { expanded: false, isPartial: false }, + theme!, + ); + + const rendered = component.render(200).join("\n"); + expect(Bun.stripANSI(rendered)).toContain(`${examplePath}:raw`); + expect(extractLinkUris(rendered)).toContain(url.pathToFileURL(path.resolve(examplePath)).href); + expect(extractLinkTexts(rendered)).toContain(examplePath); + expect(extractLinkTexts(rendered)).not.toContain(`${examplePath}:raw`); + }); + it("links HTTP read result headers to the final URL", async () => { settings.override("tui.hyperlinks", "always"); const theme = await getThemeByName("dark"); From 58336f5886a6be96a6b1cd601201de3a2e214811 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 02:24:11 +0000 Subject: [PATCH 08/59] fix(tui): stopped slash tab from reopening files Prevented no-argument slash commands from falling through to empty-prefix file suggestions immediately after Tab completion. Added an editor regression covering /quit Tab completion with files present in the project root. Fixes #4808 --- packages/tui/CHANGELOG.md | 3 +++ packages/tui/src/autocomplete.ts | 3 +++ packages/tui/test/editor.test.ts | 45 +++++++++++++++++++++++++++++++- 3 files changed, 50 insertions(+), 1 deletion(-) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index e93024ea9..aeffd88b0 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -11,6 +11,9 @@ - Fixed mid-prompt skill autocomplete so Tab and Enter accept the highlighted `/skill:` suggestion and Backspace dismisses the popup immediately after removing the triggering slash ([#4619](https://github.com/can1357/oh-my-pi/issues/4619)). - Fixed submitted slash-command arguments treating `@` file-reference tokens as prompt-composer autocomplete triggers when the command does not define argument completions. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) - Fixed box-drawing tree lines (`├── item` — directory layouts, decision trees) in prose shearing apart when they wrap: continuation rows now hang under the node text with ancestor rails carried through (`├` → `│`, `└` → blank) instead of restarting at column 0. Applies to prose paragraphs (including inside blockquotes) only when a line with a branch-connector prefix (`├──`, `└─`, …) actually overflows; fitting lines, non-tree prose, and code blocks render byte-for-byte as before. +### Fixed + +- Fixed slash command Tab completion reopening the file autocomplete drawer after accepting no-argument commands ([#4808](https://github.com/can1357/oh-my-pi/issues/4808)). ## [16.3.10] - 2026-07-06 diff --git a/packages/tui/src/autocomplete.ts b/packages/tui/src/autocomplete.ts index df6b34b1e..9afdf8d9b 100644 --- a/packages/tui/src/autocomplete.ts +++ b/packages/tui/src/autocomplete.ts @@ -437,6 +437,9 @@ export class CombinedAutocompleteProvider implements AutocompleteProvider { const argumentText = commandText.slice(spaceIndex + 1); // Text after space const command = this.#commands.find(cmd => commandMatchesNameOrAlias(cmd, commandName)); + if (command && "allowArgs" in command && command.allowArgs === false && argumentText === "") { + return null; + } if (command && (!("allowArgs" in command) || command.allowArgs !== false)) { if (!("getArgumentCompletions" in command) || !command.getArgumentCompletions) { return null; // No argument completion for this command diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index 1e0b27dfb..a4eb03ded 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -1,4 +1,7 @@ -import { afterEach, describe, expect, it } from "bun:test"; +import { afterEach, describe, expect, it, vi } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { stripVTControlCharacters } from "node:util"; import { CURSOR_MARKER } from "@oh-my-pi/pi-tui"; import { CombinedAutocompleteProvider } from "@oh-my-pi/pi-tui/autocomplete"; @@ -424,6 +427,46 @@ describe("Editor component", () => { expect(editor.getText()).toBe("/help "); expect(editor.isShowingAutocomplete()).toBe(false); }); + + it("does not open file autocomplete after tab-completing no-arg slash commands", async () => { + vi.useFakeTimers(); + const baseDir = await fs.mkdtemp(path.join(os.tmpdir(), "slash-tab-no-arg-")); + try { + await Bun.write(path.join(baseDir, "visible-file.ts"), "export {};\n"); + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider([{ name: "quit", description: "Quit", allowArgs: false }], baseDir), + ); + + let nextUpdate = Promise.withResolvers(); + editor.onAutocompleteUpdate = () => nextUpdate.resolve(); + editor.handleInput("/"); + await nextUpdate.promise; + + nextUpdate = Promise.withResolvers(); + editor.onAutocompleteUpdate = () => nextUpdate.resolve(); + editor.handleInput("q"); + vi.advanceTimersByTime(100); + await nextUpdate.promise; + + const chainedUpdates = Promise.withResolvers(); + let updateCount = 0; + editor.onAutocompleteUpdate = () => { + updateCount += 1; + if (updateCount === 2) { + chainedUpdates.resolve(); + } + }; + editor.handleInput(" "); + await chainedUpdates.promise; + + expect(editor.getText()).toBe("/quit "); + expect(editor.isShowingAutocomplete()).toBe(false); + } finally { + vi.useRealTimers(); + await fs.rm(baseDir, { recursive: true, force: true }); + } + }); }); describe("Unicode text editing behavior", () => { From 80b186cd8c751ba32adedfa265ab17c0f0899645 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 02:34:15 +0000 Subject: [PATCH 09/59] fix(tui): handled whitespace after no-arg commands Treated whitespace-only slash command arguments like empty arguments so no-arg commands stay closed after repeated spaces. Added provider coverage for the whitespace-only no-arg command path. --- packages/tui/src/autocomplete.ts | 2 +- packages/tui/test/autocomplete.test.ts | 17 +++++++++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/packages/tui/src/autocomplete.ts b/packages/tui/src/autocomplete.ts index 9afdf8d9b..54298a60d 100644 --- a/packages/tui/src/autocomplete.ts +++ b/packages/tui/src/autocomplete.ts @@ -437,7 +437,7 @@ export class CombinedAutocompleteProvider implements AutocompleteProvider { const argumentText = commandText.slice(spaceIndex + 1); // Text after space const command = this.#commands.find(cmd => commandMatchesNameOrAlias(cmd, commandName)); - if (command && "allowArgs" in command && command.allowArgs === false && argumentText === "") { + if (command && "allowArgs" in command && command.allowArgs === false && !/\S/.test(argumentText)) { return null; } if (command && (!("allowArgs" in command) || command.allowArgs !== false)) { diff --git a/packages/tui/test/autocomplete.test.ts b/packages/tui/test/autocomplete.test.ts index 5f2733d07..f3f21ce4c 100644 --- a/packages/tui/test/autocomplete.test.ts +++ b/packages/tui/test/autocomplete.test.ts @@ -119,6 +119,23 @@ describe("CombinedAutocompleteProvider", () => { expect(result?.items.map(item => item.value)).toContain("/tmp/"); }); + it("does not treat whitespace-only no-arg slash command arguments as file prefixes", async () => { + const baseDir = fs.mkdtempSync(path.join(os.tmpdir(), "autocomplete-quit-whitespace-")); + try { + fs.writeFileSync(path.join(baseDir, "copy-target.ts"), "export {};\n"); + const provider = new CombinedAutocompleteProvider( + [{ name: "quit", description: "Quit", allowArgs: false }], + baseDir, + ); + const line = "/quit "; + const result = await provider.getSuggestions([line], 0, line.length); + + expect(result).toBeNull(); + } finally { + fs.rmSync(baseDir, { recursive: true, force: true }); + } + }); + it("treats @ file-reference tokens as literal text inside slash command arguments without completions", async () => { const baseDir = fs.mkdtempSync(path.join(os.tmpdir(), "autocomplete-rename-args-")); try { From e07909fdaf7bb1c8010db13df1678a2118bcd995 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:10:53 +0200 Subject: [PATCH 10/59] fix(tui): replayed detected terminal appearance to late subscribers ProcessTerminal.start() can parse the startup OSC 11 response before InteractiveMode.init() registers its onAppearanceChange callback (init awaits hooks/mode/draft restore between ui.start() and subscribing). The dedup in #handleOsc11Response then suppresses the value forever and theme auto-detection stays on the dark fallback despite a light terminal. Replay the already-detected appearance to subscribers that register after detection. Adopted from PR #4883; hardened the test to stop the terminal before asserting so a failure cannot leak a live terminal into later tests. Fixes #4731 --- packages/tui/CHANGELOG.md | 4 ++++ packages/tui/src/terminal.ts | 13 +++++++++++++ packages/tui/test/terminal-appearance.test.ts | 18 ++++++++++++++++++ 3 files changed, 35 insertions(+) diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index aeffd88b0..318668942 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed late terminal appearance subscribers missing the already-detected OSC 11 light/dark result, so theme auto-detection picks up the terminal appearance even when the response arrives before the UI subscribes ([#4731](https://github.com/can1357/oh-my-pi/issues/4731)). + ## [16.3.12] - 2026-07-08 ### Fixed diff --git a/packages/tui/src/terminal.ts b/packages/tui/src/terminal.ts index ed9210af9..5764c3b66 100644 --- a/packages/tui/src/terminal.ts +++ b/packages/tui/src/terminal.ts @@ -382,6 +382,8 @@ export interface Terminal { * Register a callback for terminal appearance (dark/light) changes. * Detection uses OSC 11 background color query with Mode 2031 as a change trigger. * Fires when the detected appearance changes, including the initial detection. + * Subscribers registered after detection are invoked immediately with the + * already-detected appearance so late subscribers never miss it. */ onAppearanceChange(callback: (appearance: TerminalAppearance) => void): void; /** The last detected terminal appearance, or undefined if not yet known. */ @@ -516,6 +518,17 @@ export class ProcessTerminal implements Terminal { onAppearanceChange(callback: (appearance: TerminalAppearance) => void): void { this.#appearanceCallbacks.push(callback); + // Replay an already-detected appearance: the startup OSC 11 response can + // arrive before consumers (e.g. the theme bridge) subscribe, and the + // dedup in #handleOsc11Response would otherwise suppress the value for + // them forever (#4731). + if (this.#appearance) { + try { + callback(this.#appearance); + } catch { + /* ignore callback errors */ + } + } } onPrivateModeReport(callback: (mode: number, supported: boolean) => void): void { diff --git a/packages/tui/test/terminal-appearance.test.ts b/packages/tui/test/terminal-appearance.test.ts index 88438c841..18ca4ad46 100644 --- a/packages/tui/test/terminal-appearance.test.ts +++ b/packages/tui/test/terminal-appearance.test.ts @@ -144,6 +144,24 @@ describe("ProcessTerminal OSC 11 appearance detection", () => { terminal.stop(); }); + it("replays already detected OSC 11 appearance to late subscribers", () => { + const { terminal } = setupTerminal(); + + process.stdin.emit("data", "\x1b]11;rgb:ffff/ffff/ffff\x07"); + process.stdin.emit("data", "\x1b[?1;2c"); + + const appearances: string[] = []; + terminal.onAppearanceChange(a => appearances.push(a)); + const detected = terminal.appearance; + + // Stop before asserting: a failing expect must not leak a live terminal + // (stdin listeners, kitty push) into subsequent tests. + terminal.stop(); + + expect(detected).toBe("light"); + expect(appearances).toEqual(["light"]); + }); + it("2-digit hex OSC 11 response is correctly normalized", () => { const { terminal } = setupTerminal(); From 2aee4bd3f7c4421888a60f43fa192d99b7d0ad54 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:18:12 +0200 Subject: [PATCH 11/59] fix(tui): applied DEC 2048 resize reports with colon subparameters The mode 2048 in-band resize spec permits `:`-separated subparameters on any field and requires clients to ignore them. The parser rejected such reports outright (and the split-reassembly prefix pattern dropped fragmented ones as garbage), so the grow-back report after an iOS soft keyboard dismissal under tmux-over-SSH never applied: rows stayed pinned at the keyboard-present height, with no accompanying OS resize event to reconcile the cached in-band geometry. Capture the leading digits of each field and skip the subparameter tail; accept `:` in the reassembly prefix so split reports complete instead of leaking their tails into the editor as keystrokes. Fixes #4748 --- packages/tui/src/terminal.ts | 13 +++-- .../tui/test/process-terminal-render.test.ts | 17 +++++++ packages/tui/test/terminal-appearance.test.ts | 47 +++++++++++++++++++ 3 files changed, 72 insertions(+), 5 deletions(-) diff --git a/packages/tui/src/terminal.ts b/packages/tui/src/terminal.ts index 5764c3b66..b35dfa1f5 100644 --- a/packages/tui/src/terminal.ts +++ b/packages/tui/src/terminal.ts @@ -717,7 +717,10 @@ export class ProcessTerminal implements Terminal { const decrpmResponsePattern = /^\x1b\[\?(\d+);(\d+)\$y$/; // In-band resize report (DEC mode 2048): \x1b[48;rows;cols;yPixels;xPixels t - const inBandResizePattern = /^\x1b\[48;(\d+);(\d+);(\d+);(\d+)t$/; + // Any field may carry `:`-separated subparameters, which clients MUST + // ignore per spec (#4748): capture the leading digits of each field and + // skip the subparameter tail instead of dropping the whole report. + const inBandResizePattern = /^\x1b\[48;(\d+)(?::[\d:]*)?;(\d+)(?::[\d:]*)?;(\d+)(?::[\d:]*)?;(\d+)(?::[\d:]*)?t$/; this.#stdinBuffer.on("data", (sequence: string) => { // Fast path for plain-text bytes: every escape-probe regex below @@ -789,7 +792,7 @@ export class ProcessTerminal implements Terminal { // reassembled sequence that turns out not to be a resize report (e.g. a // split kitty `\x1b[48;…u` for a digit key) is forwarded to the input // handler rather than dropped. - const inBandResizePartialPattern = /^\x1b\[4[\d;]*$/; + const inBandResizePartialPattern = /^\x1b\[4[\d;:]*$/; const isInBandResizePartial = this.#inBandResizeActive && inBandResizePartialPattern.test(sequence); if (this.#inBandResizeBuffer && sequence.startsWith("\x1b")) { // A new escape interrupted the partial; the stale partial is @@ -1198,9 +1201,9 @@ export class ProcessTerminal implements Terminal { * `rows` before the `resize` event fires, so they are authoritative for the * new cell geometry. A cached DEC 2048 report can be stale: the matching * post-resize report may be dropped (split across stdin reads past the flush - * window) or carry `:`-subparameters the parser skips, leaving the getters - * pinned to the old size — which freezes the rendered width because the - * renderer reflows against {@link columns}/{@link rows}, not the live OS + * window, or interrupted by another escape mid-reassembly), leaving the + * getters pinned to the old size — which freezes the rendered width because + * the renderer reflows against {@link columns}/{@link rows}, not the live OS * value. Drop a cached dimension that disagrees with the live OS value; the * terminal's next valid in-band report re-seeds pixel sizing. */ diff --git a/packages/tui/test/process-terminal-render.test.ts b/packages/tui/test/process-terminal-render.test.ts index d95bf6243..d9af2d757 100644 --- a/packages/tui/test/process-terminal-render.test.ts +++ b/packages/tui/test/process-terminal-render.test.ts @@ -71,4 +71,21 @@ describe("ProcessTerminal geometry reflow through the renderer", () => { expect(harness.terminal.columns).toBe(160); expect(harness.probe.last).toBe(160); }); + + it("recovers the full height when the grow-back report carries colon subparameters (#4748)", async () => { + // iOS soft keyboard under tmux-over-SSH: an in-band shrink lands (keyboard + // up), then the keyboard is dismissed and the grow-back report arrives with + // a spec-permitted `:`-subparameter and no accompanying OS resize. The + // parser must ignore the subparameter — dropping the report leaves the + // viewport pinned at the keyboard-present height. + harness = createProcessTerminalRenderHarness(100, 30); + await harness.feed("\x1b[?2048;1$y"); + await harness.inBand(15, 100, 300, 1000); // keyboard appears: 30 -> 15 rows + expect(harness.terminal.rows).toBe(15); + + await harness.feed("\x1b[48;30;100;600;1000:0t"); // keyboard dismissed: grow back + + expect(harness.terminal.rows).toBe(30); + expect(harness.terminal.columns).toBe(100); + }); }); diff --git a/packages/tui/test/terminal-appearance.test.ts b/packages/tui/test/terminal-appearance.test.ts index 18ca4ad46..ea1067c1b 100644 --- a/packages/tui/test/terminal-appearance.test.ts +++ b/packages/tui/test/terminal-appearance.test.ts @@ -600,6 +600,30 @@ describe("ProcessTerminal DECRQM + in-band resize (DEC 2026/2048)", () => { terminal.stop(); }); + it("applies a grow-back report whose fields carry colon subparameters (#4748)", () => { + // iOS soft keyboard dismissed under tmux-over-SSH: the pane grows back and + // the terminal reports the restored geometry in-band with a spec-permitted + // `:`-subparameter appended to a field. Mode 2048 allows subparameters on + // any field and requires clients to IGNORE them — dropping the whole + // report instead pins `rows` at the keyboard-present height, because no + // OS resize event accompanies the report to reconcile cached geometry. + Object.defineProperty(process.stdout, "columns", { value: 100, configurable: true }); + Object.defineProperty(process.stdout, "rows", { value: 40, configurable: true }); + const { terminal, received, resizeCount } = setup(); + process.stdin.emit("data", "\x1b[?2048;1$y"); // in-band active + process.stdin.emit("data", "\x1b[48;20;100;400;1000t"); // keyboard appears: shrink + expect(terminal.rows).toBe(20); + expect(resizeCount()).toBe(1); + + process.stdin.emit("data", "\x1b[48;40;100;800;1000:0t"); // keyboard dismissed: grow back + + expect(terminal.rows).toBe(40); + expect(terminal.columns).toBe(100); + expect(resizeCount()).toBe(2); + expect(received).toEqual([]); + terminal.stop(); + }); + it("tracks OS geometry on resize when the post-resize in-band report is missed", () => { // Real terminals always fire SIGWINCH (process.stdout dims refresh first), // but the matching DEC 2048 report can be dropped or arrive malformed. The @@ -676,6 +700,29 @@ describe("ProcessTerminal DECRQM + in-band resize (DEC 2026/2048)", () => { terminal.stop(); }); + it("reassembles a split grow-back report with colon subparameters without dropping or leaking it", () => { + // Same grow-back report, fragmented by the StdinBuffer flush window right + // after the subparameter colon. The partial pattern must accept `:`, or + // the prefix is rejected as garbage, the report never applies, and the + // `0t` tail leaks into the editor as literal keystrokes. + vi.useFakeTimers(); + Object.defineProperty(process.stdout, "columns", { value: 100, configurable: true }); + Object.defineProperty(process.stdout, "rows", { value: 40, configurable: true }); + const { terminal, received, resizeCount } = setup(); + process.stdin.emit("data", "\x1b[?2048;1$y"); // in-band active + process.stdin.emit("data", "\x1b[48;20;100;400;1000t"); // keyboard appears: shrink + expect(terminal.rows).toBe(20); + + process.stdin.emit("data", "\x1b[48;40;100;800;1000:"); + vi.advanceTimersByTime(50); // flush window elapses mid-report + process.stdin.emit("data", "0t"); + + expect(received).toEqual([]); + expect(terminal.rows).toBe(40); + expect(resizeCount()).toBe(2); + terminal.stop(); + }); + it("forwards a split report fragment as one escape sequence instead of leaking bare characters", () => { // The reported symptom: a fragment like `8;125;1156;1125t` (the tail of // `\x1b[48;125;1156;1125t`, missing a field) appeared as literal text in the From 9ce8b69b34d4d5ce2b83297a946d9f92dfdad600 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:07:14 +0200 Subject: [PATCH 12/59] fix(config): respected preseeded yaml settings First-run settings load now discovers an existing config.yaml next to config.yml, loads it as the main settings file, and keeps writing back to the discovered path instead of creating a stub config.yml beside it. Refs #4914 --- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/src/config/settings.ts | 69 +++++++++++++------ .../test/settings-manager.test.ts | 30 ++++++++ 3 files changed, 81 insertions(+), 20 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bd86fa43c..002892c13 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -20,6 +20,8 @@ ### Fixed +- Fixed first-run setup ignoring a pre-seeded `config.yaml`: the settings loader now treats `config.yml` and `config.yaml` as equivalent existing main config files, writes back to the existing extension, and only creates canonical `config.yml` for fresh installs. ([#4914](https://github.com/can1357/oh-my-pi/issues/4914)) + - Improved handling of unawaited promises in JS eval cells to prevent process crashes - Added warning logs for unhandled rejections originating from finished eval cells - Improved advisor robustness by blocking exhausted accounts during consecutive turn failures diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index c8aca9a30..23ae1d0c6 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -60,7 +60,7 @@ export interface RawSettings { export interface SettingsOptions { /** Current working directory for project settings discovery */ cwd?: string; - /** Agent directory for config.yml storage */ + /** Agent directory for config.yml/config.yaml storage */ agentDir?: string; /** Don't persist to disk (for tests) */ inMemory?: boolean; @@ -234,7 +234,7 @@ export class Settings { #storage: AgentStorage | null = null; #configFiles: string[] = []; - /** Global settings from config.yml */ + /** Global settings from config.yml/config.yaml */ #global: RawSettings = {}; /** Project settings from .claude/settings.yml etc */ #project: RawSettings = {}; @@ -676,16 +676,22 @@ export class Settings { async #load(): Promise { // Project settings load (loadCapability scans cwd) is independent of the - // persist chain (storage open → legacy migration → global config.yml read), - // so kick it off first and await after the persist chain completes. The - // persist steps remain sequential: migration may write config.yml, which - // #loadYaml then reads; migration's db fallback needs #storage opened. + // persist chain (storage open → legacy migration → global config read), so + // kick it off first and await after the persist chain completes. The + // persist steps remain sequential: existing config discovery decides + // whether migration may write config.yml before the global config is read; + // migration's db fallback needs #storage opened. const projectPromise = this.#loadProjectSettings(); if (this.#persist) { this.#storage = await AgentStorage.open(getAgentDbPath(this.#agentDir)); - await this.#migrateFromLegacy(); - this.#global = await this.#loadYaml(this.#configPath!); + const existingConfig = await this.#loadExistingMainYaml(); + if (existingConfig) { + this.#global = existingConfig; + } else { + await this.#migrateFromLegacy(); + this.#global = await this.#loadYaml(this.#configPath!); + } await this.#seedLastChangelogVersionMarker(); } @@ -701,8 +707,9 @@ export class Settings { async #loadReadOnly(): Promise { const projectPromise = this.#loadProjectSettings(); - if (this.#configPath) { - this.#global = await this.#loadYaml(this.#configPath); + const existingConfig = await this.#loadExistingMainYaml(); + if (existingConfig) { + this.#global = existingConfig; } this.#project = await projectPromise; @@ -712,20 +719,50 @@ export class Settings { } async #loadYaml(filePath: string): Promise { + const loaded = await this.#loadYamlIfPresent(filePath); + return loaded ?? {}; + } + + async #loadYamlIfPresent(filePath: string): Promise { + let content: string; + try { + content = await Bun.file(filePath).text(); + } catch (error) { + if (isEnoent(error)) return null; + logger.warn("Settings: failed to load", { path: filePath, error: String(error) }); + return {}; + } + try { - const content = await Bun.file(filePath).text(); const parsed = YAML.parse(content); if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) { return {}; } return this.#migrateRawSettings(parsed as RawSettings); } catch (error) { - if (isEnoent(error)) return {}; logger.warn("Settings: failed to load", { path: filePath, error: String(error) }); return {}; } } + async #loadExistingMainYaml(): Promise { + if (!this.#configPath) return null; + const ymlPath = path.join(this.#agentDir, "config.yml"); + const yml = await this.#loadYamlIfPresent(ymlPath); + if (yml) { + this.#configPath = ymlPath; + return yml; + } + const yamlPath = path.join(this.#agentDir, "config.yaml"); + const yaml = await this.#loadYamlIfPresent(yamlPath); + if (yaml) { + this.#configPath = yamlPath; + return yaml; + } + this.#configPath = ymlPath; + return null; + } + async #loadProjectSettings(): Promise { try { const result = await loadCapability(settingsCapability.id, { cwd: this.#cwd }); @@ -781,14 +818,6 @@ export class Settings { async #migrateFromLegacy(): Promise { if (!this.#configPath) return; - // Check if config.yml already exists - try { - await Bun.file(this.#configPath).text(); - return; // Already exists, no migration needed - } catch (err) { - if (!isEnoent(err)) return; - } - let settings: RawSettings = {}; let migrated = false; diff --git a/packages/coding-agent/test/settings-manager.test.ts b/packages/coding-agent/test/settings-manager.test.ts index a8bf84f85..f721c5640 100644 --- a/packages/coding-agent/test/settings-manager.test.ts +++ b/packages/coding-agent/test/settings-manager.test.ts @@ -70,6 +70,36 @@ describe("Settings", () => { await Bun.sleep(0); await tempDir?.remove(); }); + + describe("main config file selection", () => { + it("loads and updates an existing config.yaml without creating config.yml", async () => { + const yamlConfigPath = path.join(agentDir, "config.yaml"); + await Bun.write(yamlConfigPath, YAML.stringify({ setupVersion: 1 }, null, 2)); + + const settings = await Settings.init({ cwd: projectDir, agentDir }); + expect(settings.get("setupVersion")).toBe(1); + + settings.set("setupVersion", 2); + await settings.flush(); + + const savedSettings = YAML.parse(await Bun.file(yamlConfigPath).text()) as Record; + expect(savedSettings.setupVersion).toBe(2); + expect(await Bun.file(getConfigPath()).exists()).toBe(false); + }); + + it("creates config.yml for new persisted settings when no main config exists", async () => { + const yamlConfigPath = path.join(agentDir, "config.yaml"); + + const settings = await Settings.init({ cwd: projectDir, agentDir }); + settings.set("setupVersion", 1); + await settings.flush(); + + expect(await Bun.file(getConfigPath()).exists()).toBe(true); + expect(await Bun.file(yamlConfigPath).exists()).toBe(false); + expect((await readSettings()).setupVersion).toBe(1); + }); + }); + describe("defaults", () => { it("keeps eight inline images live by default", async () => { const settings = await Settings.init({ cwd: projectDir, agentDir }); From 668741b40c9177c19f4b964d4eb89714816b5bcb Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 11:06:35 +0000 Subject: [PATCH 13/59] fix(config): preserved selected path in clones Copied the selected main config path into per-CWD settings clones so config.yaml-backed sessions keep writing to config.yaml. Added regression coverage for cloneForCwd updates against preseeded config.yaml. --- packages/coding-agent/src/config/settings.ts | 1 + .../coding-agent/test/settings-manager.test.ts | 15 +++++++++++++++ 2 files changed, 16 insertions(+) diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index 23ae1d0c6..cab46a635 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -458,6 +458,7 @@ export class Settings { inMemory: !this.#persist, }); cloned.#storage = this.#storage; + cloned.#configPath = this.#configPath; cloned.#global = structuredClone(this.#global); cloned.#project = this.#persist ? await cloned.#loadProjectSettings() : structuredClone(this.#project); cloned.#configFiles = [...this.#configFiles]; diff --git a/packages/coding-agent/test/settings-manager.test.ts b/packages/coding-agent/test/settings-manager.test.ts index f721c5640..daf6e15a0 100644 --- a/packages/coding-agent/test/settings-manager.test.ts +++ b/packages/coding-agent/test/settings-manager.test.ts @@ -87,6 +87,21 @@ describe("Settings", () => { expect(await Bun.file(getConfigPath()).exists()).toBe(false); }); + it("clones the selected config.yaml path for persisted settings", async () => { + const yamlConfigPath = path.join(agentDir, "config.yaml"); + await Bun.write(yamlConfigPath, YAML.stringify({ setupVersion: 1 }, null, 2)); + + const settings = await Settings.init({ cwd: projectDir, agentDir }); + const cloned = await settings.cloneForCwd(tempDir.join("other-project")); + + cloned.set("setupVersion", 2); + await cloned.flush(); + + const savedSettings = YAML.parse(await Bun.file(yamlConfigPath).text()) as Record; + expect(savedSettings.setupVersion).toBe(2); + expect(await Bun.file(getConfigPath()).exists()).toBe(false); + }); + it("creates config.yml for new persisted settings when no main config exists", async () => { const yamlConfigPath = path.join(agentDir, "config.yaml"); From 314b9a7c9d1aba372dc76cdb3ad1af1344a1ca51 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:07:24 +0200 Subject: [PATCH 14/59] fix(config): shared yaml config discovery Moved the config.yml/config.yaml filename order into pi-utils as MAIN_CONFIG_FILENAMES and taught the auth-broker config reader to probe both extensions with the same precedence as the settings loader. Fixes #4914 --- packages/ai/src/auth-broker/discover.ts | 36 ++++++----- .../test/auth-broker-config-discovery.test.ts | 59 +++++++++++++++++++ packages/coding-agent/src/config/settings.ts | 23 ++++---- packages/utils/src/dirs.ts | 3 + 4 files changed, 92 insertions(+), 29 deletions(-) create mode 100644 packages/ai/test/auth-broker-config-discovery.test.ts diff --git a/packages/ai/src/auth-broker/discover.ts b/packages/ai/src/auth-broker/discover.ts index c66ccdd20..71ef2c5fe 100644 --- a/packages/ai/src/auth-broker/discover.ts +++ b/packages/ai/src/auth-broker/discover.ts @@ -1,6 +1,6 @@ /** * Broker-aware auth-storage discovery used by both the coding-agent runtime and - * the catalog model generator. Keeps the precedence logic (env → config.yml → + * the catalog model generator. Keeps the precedence logic (env → config.yml/config.yaml → * token file → local SQLite) in one place so build-time tooling sees the same * credentials as the TUI. */ @@ -12,6 +12,7 @@ import { getConfigRootDir, isEnoent, logger, + MAIN_CONFIG_FILENAMES, } from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; import { AuthStorage } from "../auth-storage"; @@ -72,21 +73,24 @@ interface ConfigSnapshot { } async function readConfigYaml(agentDir: string): Promise { - const configPath = path.join(agentDir, "config.yml"); - try { - const raw = await Bun.file(configPath).text(); - const parsed = YAML.parse(raw); - if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return {}; - const record = parsed as Record; - const url = typeof record["auth.broker.url"] === "string" ? (record["auth.broker.url"] as string) : undefined; - const token = - typeof record["auth.broker.token"] === "string" ? (record["auth.broker.token"] as string) : undefined; - return { url, token }; - } catch (err) { - if (isEnoent(err)) return {}; - logger.warn("auth-broker config.yml unreadable", { error: String(err) }); - return {}; + for (const filename of MAIN_CONFIG_FILENAMES) { + const configPath = path.join(agentDir, filename); + try { + const raw = await Bun.file(configPath).text(); + const parsed = YAML.parse(raw); + if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) return {}; + const record = parsed as Record; + const url = typeof record["auth.broker.url"] === "string" ? (record["auth.broker.url"] as string) : undefined; + const token = + typeof record["auth.broker.token"] === "string" ? (record["auth.broker.token"] as string) : undefined; + return { url, token }; + } catch (err) { + if (isEnoent(err)) continue; + logger.warn("auth-broker config unreadable", { path: configPath, error: String(err) }); + return {}; + } } + return {}; } function resolveSnapshotTtlMs(): number { @@ -104,7 +108,7 @@ function resolveSnapshotTtlMs(): number { * Resolve broker connection configuration using the same precedence as the TUI: * * 1. `OMP_AUTH_BROKER_URL` / `OMP_AUTH_BROKER_TOKEN` env vars. - * 2. `auth.broker.url` / `auth.broker.token` in `/config.yml`. + * 2. `auth.broker.url` / `auth.broker.token` in `/config.yml` or `/config.yaml`. * 3. `/auth-broker.token` file (paired with a URL from env/config). * * Returns `null` when no broker URL is configured — callers should fall back to diff --git a/packages/ai/test/auth-broker-config-discovery.test.ts b/packages/ai/test/auth-broker-config-discovery.test.ts new file mode 100644 index 000000000..67d4033aa --- /dev/null +++ b/packages/ai/test/auth-broker-config-discovery.test.ts @@ -0,0 +1,59 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { resolveAuthBrokerConfig } from "@oh-my-pi/pi-ai/auth-broker"; +import { removeWithRetries } from "../../utils/src/temp"; +import { withEnv } from "./helpers"; + +const SUPPRESS_AUTH_BROKER_ENV = { + OMP_AUTH_BROKER_URL: undefined, + OMP_AUTH_BROKER_TOKEN: undefined, +} as const; + +describe("resolveAuthBrokerConfig config discovery", () => { + let agentDir = ""; + + beforeEach(async () => { + agentDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-broker-config-")); + }); + + afterEach(async () => { + if (agentDir) { + await removeWithRetries(agentDir); + agentDir = ""; + } + }); + + test("resolves broker URL and token from config.yaml when config.yml is absent", async () => { + await Bun.write( + path.join(agentDir, "config.yaml"), + "auth.broker.url: https://yaml-broker.example/v1\nauth.broker.token: yaml-token\n", + ); + + await withEnv(SUPPRESS_AUTH_BROKER_ENV, async () => { + await expect(resolveAuthBrokerConfig({ agentDir })).resolves.toEqual({ + url: "https://yaml-broker.example/v1", + token: "yaml-token", + }); + }); + }); + + test("prefers config.yml over config.yaml when both exist", async () => { + await Bun.write( + path.join(agentDir, "config.yaml"), + "auth.broker.url: https://yaml-broker.example/v1\nauth.broker.token: yaml-token\n", + ); + await Bun.write( + path.join(agentDir, "config.yml"), + "auth.broker.url: https://yml-broker.example/v1\nauth.broker.token: yml-token\n", + ); + + await withEnv(SUPPRESS_AUTH_BROKER_ENV, async () => { + await expect(resolveAuthBrokerConfig({ agentDir })).resolves.toEqual({ + url: "https://yml-broker.example/v1", + token: "yml-token", + }); + }); + }); +}); diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index cab46a635..58b256ba5 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -22,6 +22,7 @@ import { getProjectDir, isEnoent, logger, + MAIN_CONFIG_FILENAMES, procmgr, setWorktreesDir, } from "@oh-my-pi/pi-utils"; @@ -264,7 +265,7 @@ export class Settings { private constructor(options: SettingsOptions = {}) { this.#cwd = path.normalize(options.cwd ?? getProjectDir()); this.#agentDir = path.normalize(options.agentDir ?? getAgentDir()); - this.#configPath = options.inMemory ? null : path.join(this.#agentDir, "config.yml"); + this.#configPath = options.inMemory ? null : path.join(this.#agentDir, MAIN_CONFIG_FILENAMES[0]); this.#configFiles = options.configFiles?.map(file => path.resolve(this.#cwd, expandTilde(file))) ?? []; this.#persist = !options.inMemory && options.readOnly !== true; @@ -748,19 +749,15 @@ export class Settings { async #loadExistingMainYaml(): Promise { if (!this.#configPath) return null; - const ymlPath = path.join(this.#agentDir, "config.yml"); - const yml = await this.#loadYamlIfPresent(ymlPath); - if (yml) { - this.#configPath = ymlPath; - return yml; + for (const filename of MAIN_CONFIG_FILENAMES) { + const configPath = path.join(this.#agentDir, filename); + const loaded = await this.#loadYamlIfPresent(configPath); + if (loaded) { + this.#configPath = configPath; + return loaded; + } } - const yamlPath = path.join(this.#agentDir, "config.yaml"); - const yaml = await this.#loadYamlIfPresent(yamlPath); - if (yaml) { - this.#configPath = yamlPath; - return yaml; - } - this.#configPath = ymlPath; + this.#configPath = path.join(this.#agentDir, MAIN_CONFIG_FILENAMES[0]); return null; } diff --git a/packages/utils/src/dirs.ts b/packages/utils/src/dirs.ts index c6fb9d9df..d0b63af69 100644 --- a/packages/utils/src/dirs.ts +++ b/packages/utils/src/dirs.ts @@ -22,6 +22,9 @@ export const APP_NAME: string = "omp"; /** Config directory name (e.g. ".omp") */ export const CONFIG_DIR_NAME: string = ".omp"; +/** Ordered main settings filenames: canonical write target first, legacy-compatible YAML fallback second. */ +export const MAIN_CONFIG_FILENAMES = ["config.yml", "config.yaml"] as const; + /** Version (e.g. "1.0.0") */ export const VERSION: string = version; From 48877de0846c0c4d80c34b03cb7e6870dd5ecd09 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:12:32 +0200 Subject: [PATCH 15/59] fix(coding-agent): restored profile keybinding inheritance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Named profiles loaded keybindings only from their own agent dir (~/.omp/profiles//agent), silently dropping user-level bindings from ~/.omp/agent/keybindings.* — e.g. Backspace remaps under tmux/QTerminal. KeybindingsManager.create now merges the default profile's keybindings under the active profile's, with the profile file overriding per binding. The inherited file is loaded read-only so a named-profile run never writes migration output into the default profile's dir. Documented the exception in docs/config-usage.md. Adopted from PR #4869 (dropped its unrelated workflow-notice.md churn, added the read-only inherited load and its regression test). Fixes #4867 Co-authored-by: roboomp --- docs/config-usage.md | 2 + packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/config/keybindings.ts | 72 ++++++++-- .../test/keybindings-migration.test.ts | 134 +++++++++++++++++- 4 files changed, 197 insertions(+), 12 deletions(-) diff --git a/docs/config-usage.md b/docs/config-usage.md index 6606618c2..984c08ee7 100644 --- a/docs/config-usage.md +++ b/docs/config-usage.md @@ -79,6 +79,8 @@ A named profile (`omp --profile `, the `--alias` shortcut, or `OMP_PROFILE The relocation is uniform across the native provider (`builtin.ts`) and the generic `config.ts` helpers, so it covers slash commands, rules, prompts, instructions, hooks, tools, extensions, settings, skills, and MCP, plus the top-level `SYSTEM.md` / `RULES.md` / `AGENTS.md` files and runtime state (sessions, blobs, `agent.db`). A profile sees only its own OMP config, never the default profile's `~/.omp/agent`. +Keybindings are the one exception: a named profile merges the default profile's `~/.omp/agent/keybindings.*` under its own `~/.omp/profiles//agent/keybindings.*`, with the profile file overriding per binding ([#4867](https://github.com/can1357/oh-my-pi/issues/4867)). Keybindings describe the terminal/keyboard in front of the user, which doesn't change with the active profile, so user-level remaps keep working in every profile unless the profile explicitly overrides them. The inherited file is read-only for the profile process — legacy-format migration of the default profile's file only happens when the default profile itself runs. + The other source bases are not profile-scoped and load identically under every profile: the external-tool bases (`~/.claude`, `~/.codex`, `~/.gemini`) belong to those tools, and the project-level bases (`/.omp`, `/.claude`, ...) are keyed to the working directory. Throughout this document, read `~/.omp/agent` as shorthand for the active profile's agent directory. ## Important constraint diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 002892c13..bc0beb837 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -7,6 +7,7 @@ - Fixed `read` and `grep` treating empty optional `selector` fields emitted by models as invalid selectors instead of behaving like omitted selectors. ([#4879](https://github.com/can1357/oh-my-pi/issues/4879)) - Fixed `grep` explicit line selectors on directory searches so they filter each matched file by line number instead of aborting with a single-file-only error ([#4898](https://github.com/can1357/oh-my-pi/issues/4898)). - Fixed Read tool previews dropping explicit `selector` arguments, so line ranges and `raw` modifiers render in terminal read call titles again ([#4899](https://github.com/can1357/oh-my-pi/issues/4899)). +- Fixed named profiles dropping default user keybindings from `~/.omp/agent/keybindings.*`; profile keybindings now inherit those defaults and override only the keys they define ([#4867](https://github.com/can1357/oh-my-pi/issues/4867)). ## [16.3.12] - 2026-07-08 diff --git a/packages/coding-agent/src/config/keybindings.ts b/packages/coding-agent/src/config/keybindings.ts index 385fa2258..c23449341 100644 --- a/packages/coding-agent/src/config/keybindings.ts +++ b/packages/coding-agent/src/config/keybindings.ts @@ -9,7 +9,7 @@ import { TUI_KEYBINDINGS, KeybindingsManager as TuiKeybindingsManager, } from "@oh-my-pi/pi-tui"; -import { getAgentDir, isEnoent, logger } from "@oh-my-pi/pi-utils"; +import { getActiveProfile, getAgentDir, getProfileRootDir, isEnoent, logger } from "@oh-my-pi/pi-utils"; import { JSONC, YAML } from "bun"; /** @@ -375,6 +375,12 @@ interface KeybindingsConfigPaths { writeBackPath: string; } +/** Controls inherited keybinding lookup when creating a manager for a named profile. */ +export interface KeybindingsCreateOptions { + /** Default-profile agent directory whose keybindings are merged before profile-specific bindings. */ + inheritedAgentDir?: string; +} + /** * Load raw config from a file synchronously. * Returns parsed JSON/YAML or null if file doesn't exist or is invalid. @@ -428,6 +434,48 @@ function resolveKeybindingsConfigPaths(agentDir: string): KeybindingsConfigPaths return { readPath: ymlPath, writeBackPath: ymlPath }; } +function mergeKeybindingsConfig( + inheritedConfig: KeybindingsConfig, + profileConfig: KeybindingsConfig, +): KeybindingsConfig { + return { ...inheritedConfig, ...profileConfig }; +} + +function resolveInheritedAgentDir(agentDir: string, options: KeybindingsCreateOptions): string | undefined { + const inheritedAgentDir = + options.inheritedAgentDir ?? (getActiveProfile() ? path.join(getProfileRootDir(undefined), "agent") : undefined); + if (!inheritedAgentDir) return undefined; + if (path.resolve(inheritedAgentDir) === path.resolve(agentDir)) return undefined; + return inheritedAgentDir; +} + +function loadMergedKeybindingsConfig( + agentDir: string, + options: KeybindingsCreateOptions, +): { + config: KeybindingsConfig; + profilePath: string; + inheritedPath: string | undefined; +} { + const profilePaths = resolveKeybindingsConfigPaths(agentDir); + const profile = loadKeybindingsConfig(profilePaths.readPath, profilePaths.writeBackPath); + const inheritedAgentDir = resolveInheritedAgentDir(agentDir, options); + if (!inheritedAgentDir) { + return { config: profile.config, profilePath: profile.persistedPath, inheritedPath: undefined }; + } + + const inheritedPaths = resolveKeybindingsConfigPaths(inheritedAgentDir); + // Read-only: a named-profile process must never write migration output into + // the default profile's agent dir. Name migration still applies in-memory; + // the on-disk migration happens when the default profile itself launches. + const inherited = loadKeybindingsConfig(inheritedPaths.readPath, undefined); + return { + config: mergeKeybindingsConfig(inherited.config, profile.config), + profilePath: profile.persistedPath, + inheritedPath: inherited.persistedPath, + }; +} + /** * Load and migrate keybindings config. * Legacy JSON is read for compatibility, but successful write-back goes to YAML. @@ -499,22 +547,23 @@ function keyConfigValue(keys: KeyId[]): KeyId | KeyId[] { */ export class KeybindingsManager extends TuiKeybindingsManager { #configPath: string | undefined; + #inheritedConfigPath: string | undefined; #userBindings: KeybindingsConfig; - constructor(userBindings: KeybindingsConfig = {}, configPath?: string) { + constructor(userBindings: KeybindingsConfig = {}, configPath?: string, inheritedConfigPath?: string) { super(KEYBINDINGS, userBindings); this.#configPath = configPath; + this.#inheritedConfigPath = inheritedConfigPath; this.#userBindings = userBindings; } /** - * Create from config file at agentDir/keybindings.yml. + * Create from config files at agentDir/keybindings.yml and the default profile. * Legacy keybindings.json is migrated to keybindings.yml on load. */ - static create(agentDir: string = getAgentDir()): KeybindingsManager { - const { readPath, writeBackPath } = resolveKeybindingsConfigPaths(agentDir); - const { config: userBindings, persistedPath } = KeybindingsManager.#loadFromFile(readPath, writeBackPath); - const manager = new KeybindingsManager(userBindings, persistedPath); + static create(agentDir: string = getAgentDir(), options: KeybindingsCreateOptions = {}): KeybindingsManager { + const { config: userBindings, profilePath, inheritedPath } = loadMergedKeybindingsConfig(agentDir, options); + const manager = new KeybindingsManager(userBindings, profilePath, inheritedPath); // Set globally so getKeybindings() returns this manager setKeybindings(manager); return manager; @@ -528,12 +577,15 @@ export class KeybindingsManager extends TuiKeybindingsManager { } /** - * Reload keybindings from the config file. + * Reload keybindings from the config files. */ reload(): void { if (!this.#configPath) return; - const { config } = KeybindingsManager.#loadFromFile(this.#configPath); - this.setUserBindings(config); + const { config: inheritedConfig } = this.#inheritedConfigPath + ? KeybindingsManager.#loadFromFile(this.#inheritedConfigPath) + : { config: {} }; + const { config: profileConfig } = KeybindingsManager.#loadFromFile(this.#configPath); + this.setUserBindings(mergeKeybindingsConfig(inheritedConfig, profileConfig)); } setUserBindings(userBindings: KeybindingsConfig): void { diff --git a/packages/coding-agent/test/keybindings-migration.test.ts b/packages/coding-agent/test/keybindings-migration.test.ts index ec113449f..bd54ff5a2 100644 --- a/packages/coding-agent/test/keybindings-migration.test.ts +++ b/packages/coding-agent/test/keybindings-migration.test.ts @@ -4,13 +4,32 @@ import * as os from "node:os"; import * as path from "node:path"; import { KeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings"; import { matchesAppFollowUp } from "@oh-my-pi/pi-coding-agent/modes/utils/keybinding-matchers"; -import { setKeybindings } from "@oh-my-pi/pi-tui"; -import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import { type KeybindingsConfig, setKeybindings } from "@oh-my-pi/pi-tui"; +import { + __resetDirsFromEnvForTests, + getAgentDir, + getProfileRootDir, + removeWithRetries, + setProfile, +} from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; function ctrl(key: string): string { return String.fromCharCode(key.toLowerCase().charCodeAt(0) & 31); } + +async function writeKeybindingsYaml(agentDir: string, config: KeybindingsConfig): Promise { + await fs.mkdir(agentDir, { recursive: true }); + await Bun.write(path.join(agentDir, "keybindings.yml"), YAML.stringify(config, null, 2)); +} + +function restoreEnvValue(key: string, value: string | undefined): void { + if (value === undefined) { + delete process.env[key]; + } else { + process.env[key] = value; + } +} describe("KeybindingsManager.create", () => { beforeEach(() => { setKeybindings(KeybindingsManager.inMemory()); @@ -149,6 +168,117 @@ describe("KeybindingsManager.create", () => { } }); + it("inherits default user keybindings for a named profile without a profile keybindings file (#4867)", async () => { + const rootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-profile-")); + const defaultAgentDir = path.join(rootDir, "default", "agent"); + const profileAgentDir = path.join(rootDir, "profiles", "work", "agent"); + + await writeKeybindingsYaml(defaultAgentDir, { + "app.session.fork": "ctrl+f", + "tui.editor.deleteCharBackward": ["backspace", "ctrl+h"], + }); + + try { + const manager = KeybindingsManager.create(profileAgentDir, { inheritedAgentDir: defaultAgentDir }); + + expect(manager.getKeys("app.session.fork")).toEqual(["ctrl+f"]); + expect(manager.getKeys("tui.editor.deleteCharBackward")).toEqual(["backspace", "ctrl+h"]); + } finally { + await removeWithRetries(rootDir); + } + }); + + it("merges default user keybindings with profile overrides for a named profile (#4867)", async () => { + const rootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-profile-")); + const defaultAgentDir = path.join(rootDir, "default", "agent"); + const profileAgentDir = path.join(rootDir, "profiles", "work", "agent"); + + await writeKeybindingsYaml(defaultAgentDir, { + "app.session.fork": "ctrl+f", + "app.session.new": "ctrl+n", + }); + await writeKeybindingsYaml(profileAgentDir, { + "app.session.fork": "alt+f", + "app.clipboard.copyLine": "alt+l", + }); + + try { + const manager = KeybindingsManager.create(profileAgentDir, { inheritedAgentDir: defaultAgentDir }); + + expect(manager.getKeys("app.session.new")).toEqual(["ctrl+n"]); + expect(manager.getKeys("app.session.fork")).toEqual(["alt+f"]); + expect(manager.getKeys("app.clipboard.copyLine")).toEqual(["alt+l"]); + } finally { + await removeWithRetries(rootDir); + } + }); + + it("never writes migration output into the inherited default agent dir (#4867)", async () => { + const rootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-profile-")); + const defaultAgentDir = path.join(rootDir, "default", "agent"); + const profileAgentDir = path.join(rootDir, "profiles", "work", "agent"); + + // Legacy JSON in the default dir: loading it with a write-back path would + // materialize keybindings.yml there. The inherited load must stay read-only. + await fs.mkdir(defaultAgentDir, { recursive: true }); + await Bun.write( + path.join(defaultAgentDir, "keybindings.json"), + JSON.stringify({ "app.session.fork": "ctrl+f" }, null, 2), + ); + + try { + const manager = KeybindingsManager.create(profileAgentDir, { inheritedAgentDir: defaultAgentDir }); + + expect(manager.getKeys("app.session.fork")).toEqual(["ctrl+f"]); + expect(await Bun.file(path.join(defaultAgentDir, "keybindings.yml")).exists()).toBe(false); + } finally { + await removeWithRetries(rootDir); + } + }); + + it("merges default user keybindings when create uses the active profile with no arguments (#4867)", async () => { + const originalConfigDir = process.env.PI_CONFIG_DIR; + const originalAgentDirEnv = process.env.PI_CODING_AGENT_DIR; + const originalOmpProfile = process.env.OMP_PROFILE; + const originalPiProfile = process.env.PI_PROFILE; + const configRootDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-keybindings-active-profile-")); + + try { + process.env.PI_CONFIG_DIR = path.relative(os.homedir(), configRootDir); + restoreEnvValue("PI_CODING_AGENT_DIR", originalAgentDirEnv); + restoreEnvValue("OMP_PROFILE", originalOmpProfile); + restoreEnvValue("PI_PROFILE", originalPiProfile); + __resetDirsFromEnvForTests(); + + const defaultAgentDir = path.join(getProfileRootDir(undefined), "agent"); + const profileAgentDir = path.join(getProfileRootDir("work"), "agent"); + await writeKeybindingsYaml(defaultAgentDir, { + "app.session.fork": "ctrl+f", + "app.session.new": "ctrl+n", + }); + await writeKeybindingsYaml(profileAgentDir, { + "app.session.fork": "alt+f", + "app.clipboard.copyLine": "alt+l", + }); + + setProfile("work"); + + expect(getAgentDir()).toBe(profileAgentDir); + const manager = KeybindingsManager.create(); + + expect(manager.getKeys("app.session.new")).toEqual(["ctrl+n"]); + expect(manager.getKeys("app.session.fork")).toEqual(["alt+f"]); + expect(manager.getKeys("app.clipboard.copyLine")).toEqual(["alt+l"]); + } finally { + restoreEnvValue("PI_CONFIG_DIR", originalConfigDir); + restoreEnvValue("PI_CODING_AGENT_DIR", originalAgentDirEnv); + restoreEnvValue("OMP_PROFILE", originalOmpProfile); + restoreEnvValue("PI_PROFILE", originalPiProfile); + __resetDirsFromEnvForTests(); + await removeWithRetries(configRootDir); + } + }); + it("defaults model selection to Alt+M and display reset to Ctrl+L", () => { const manager = KeybindingsManager.inMemory(); From e4291666735075505dfec0a5ee1f0747c430b4fb Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:12:18 +0200 Subject: [PATCH 16/59] fix(coding-agent): queued extension sendUserMessage as steer while streaming Extension sendUserMessage() without deliverAs fell through to prompt(), which throws AgentBusyError during an active stream; the message was dropped and surfaced as 'Extension sendUserMessage failed'. Route the omitted-deliverAs path through prompt() with streamingBehavior 'steer' so streaming queues a steer with normal prompt-flow side effects (keyword notices, advisor auto-resume reset) and idle still starts a turn. ACP skill-command prompts now pass streamingBehavior 'steer'; the RPC skill fast-path honors the prompt command's streamingBehavior field (default steer) like the plain-prompt path already did. Documented the extension-facing delivery semantics. Synthesized from PR #4942 (prompt-flow steer routing, docs, tests) and PR #4922 (RPC streamingBehavior threading, steer regression test); dropped PR #4942's unrelated workflow-notice.md ellipsis churn. Fixes #4923 Co-authored-by: roboomp Co-authored-by: metaphorics --- docs/extensions.md | 2 +- docs/sdk.md | 5 +- packages/coding-agent/CHANGELOG.md | 1 + .../src/extensibility/extensions/types.ts | 2 +- .../coding-agent/src/modes/acp/acp-agent.ts | 17 ++-- .../coding-agent/src/modes/rpc/rpc-mode.ts | 20 +++-- .../coding-agent/src/session/agent-session.ts | 12 +-- packages/coding-agent/test/acp-agent.test.ts | 8 +- .../test/agent-session-concurrent.test.ts | 86 ++++++++++++++++++- .../test/rpc-skill-command.test.ts | 37 +++++++- 10 files changed, 161 insertions(+), 29 deletions(-) diff --git a/docs/extensions.md b/docs/extensions.md index 337a1d65c..a3ac610f9 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -142,7 +142,7 @@ Also exposed: - `deliverAs: "nextTurn"` — stored and injected on the next user prompt - `triggerTurn: true` — starts a turn when idle (also honored with `deliverAs: "nextTurn"`: idle prompts immediately; while streaming the queued message schedules an internal continuation) -`pi.sendUserMessage(content, { deliverAs })` always goes through prompt flow; while streaming it queues as steer/follow-up. +`pi.sendUserMessage(content, { deliverAs })` always goes through prompt flow. Omit `deliverAs` to start a normal prompt when idle; while streaming, omitted `deliverAs` queues the message as a steer. Set `deliverAs: "followUp"` to wait until the current run finishes. ## 2) Handler context (`ExtensionContext`) diff --git a/docs/sdk.md b/docs/sdk.md index a0ab6b403..c06dec3a4 100644 --- a/docs/sdk.md +++ b/docs/sdk.md @@ -215,8 +215,9 @@ Behavior: 1. optional command/template expansion (`/` commands, custom commands, file slash commands, prompt templates) 2. if currently streaming: - - requires `streamingBehavior: "steer" | "followUp"` - - queues instead of throwing work away + - `streamingBehavior: "steer" | "followUp"` chooses how `prompt()` queues + - extension `sendUserMessage(content)` defaults to steer when `deliverAs` is omitted + - queued messages are preserved instead of throwing work away 3. if idle: - validates model + API key - appends user message diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index bc0beb837..a24b42882 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -23,6 +23,7 @@ - Fixed first-run setup ignoring a pre-seeded `config.yaml`: the settings loader now treats `config.yml` and `config.yaml` as equivalent existing main config files, writes back to the existing extension, and only creates canonical `config.yml` for fresh installs. ([#4914](https://github.com/can1357/oh-my-pi/issues/4914)) +- Fixed extension `sendUserMessage()` without `deliverAs` surfacing `AgentBusyError` during active streams; omitted `deliverAs` now queues a steer through the normal prompt flow, and ACP/RPC skill-command prompts queue while streaming (RPC honors the prompt command's `streamingBehavior`, defaulting to steer) ([#4923](https://github.com/can1357/oh-my-pi/issues/4923)). - Improved handling of unawaited promises in JS eval cells to prevent process crashes - Added warning logs for unhandled rejections originating from finished eval cells - Improved advisor robustness by blocking exhausted accounts during consecutive turn failures diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 82d513257..658b349f2 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -1107,7 +1107,7 @@ export interface ExtensionAPI { options?: { triggerTurn?: boolean; deliverAs?: "steer" | "followUp" | "nextTurn" }, ): void; - /** Send a user message to the agent, or queue it when deliverAs is set. */ + /** Send a user prompt: idle starts a turn; streaming queues as steer unless deliverAs is set. */ sendUserMessage( content: string | (TextContent | ImageContent)[], options?: { deliverAs?: "steer" | "followUp" }, diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index 8d4083c6c..b1d56128f 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -843,13 +843,16 @@ export class AcpAgent implements Agent { return false; } const built = await buildSkillPromptMessage(skill, parsed.args, "user"); - await record.session.promptCustomMessage({ - customType: SKILL_PROMPT_MESSAGE_TYPE, - content: built.message, - display: true, - details: built.details, - attribution: "user", - }); + await record.session.promptCustomMessage( + { + customType: SKILL_PROMPT_MESSAGE_TYPE, + content: built.message, + display: true, + details: built.details, + attribution: "user", + }, + { streamingBehavior: "steer" }, + ); return true; } diff --git a/packages/coding-agent/src/modes/rpc/rpc-mode.ts b/packages/coding-agent/src/modes/rpc/rpc-mode.ts index 56497b224..7dabb5849 100644 --- a/packages/coding-agent/src/modes/rpc/rpc-mode.ts +++ b/packages/coding-agent/src/modes/rpc/rpc-mode.ts @@ -88,6 +88,7 @@ export type RpcSkillCommandResult = { agentInvoked: true }; export async function tryRunRpcSkillCommand( session: RpcSkillCommandSession, text: string, + streamingBehavior: "steer" | "followUp" = "steer", ): Promise { if (!session.skillsSettings?.enableSkillCommands) return false; const parsed = parseSkillInvocation(text); @@ -95,13 +96,16 @@ export async function tryRunRpcSkillCommand( const skill = session.skills.find(candidate => candidate.name === parsed.name); if (!skill) return false; const built = await buildSkillPromptMessage(skill, parsed.args, "user"); - await session.promptCustomMessage({ - customType: SKILL_PROMPT_MESSAGE_TYPE, - content: built.message, - display: true, - details: built.details, - attribution: "user", - }); + await session.promptCustomMessage( + { + customType: SKILL_PROMPT_MESSAGE_TYPE, + content: built.message, + display: true, + details: built.details, + attribution: "user", + }, + { streamingBehavior }, + ); return { agentInvoked: true }; } @@ -842,7 +846,7 @@ export async function runRpcMode( // ================================================================= case "prompt": { - const skillResult = await tryRunRpcSkillCommand(session, command.message); + const skillResult = await tryRunRpcSkillCommand(session, command.message, command.streamingBehavior); if (skillResult) { return success(id, "prompt", skillResult); } diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 85e544ca1..992f5bb6d 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -8309,11 +8309,10 @@ export class AgentSession { } /** - * Send a user message to the agent. - * When deliverAs is set, queue the message instead of starting a new turn. + * Send a user message through the prompt flow. * - * @param content User message content (string or content array) - * @param options.deliverAs Delivery mode: "steer" or "followUp" + * Omitted `deliverAs` starts a turn when idle and queues as a steer while streaming. + * Explicit `deliverAs` queues without starting a turn in either state. */ async sendUserMessage( content: string | (TextContent | ImageContent)[], @@ -8348,10 +8347,13 @@ export class AgentSession { return; } - // Use prompt() with expandPromptTemplates: false to skip command handling and template expansion + // Use prompt() with expandPromptTemplates: false to skip command handling and template expansion. + // `streamingBehavior: "steer"` preserves prompt-flow side effects during streaming while + // covering the narrow race where a stream starts before prompt() acquires the turn. await this.prompt(text, { expandPromptTemplates: false, images, + streamingBehavior: "steer", }); } diff --git a/packages/coding-agent/test/acp-agent.test.ts b/packages/coding-agent/test/acp-agent.test.ts index 556e32813..04c4b44b2 100644 --- a/packages/coding-agent/test/acp-agent.test.ts +++ b/packages/coding-agent/test/acp-agent.test.ts @@ -124,6 +124,7 @@ class FakeAgentSession { } promptCalls: string[] = []; customMessages: Array<{ customType: string; content: string; details?: unknown }> = []; + customMessageOptions: Array<{ streamingBehavior?: "steer" | "followUp"; queueChipText?: string } | undefined> = []; skillsSettings = { enableSkillCommands: true }; skills: Array<{ name: string; description: string; filePath: string; baseDir: string; source: string }> = []; planModeState: PlanModeState | undefined; @@ -235,8 +236,12 @@ class FakeAgentSession { this.isStreaming = false; } - async promptCustomMessage(message: { customType: string; content: string; details?: unknown }): Promise { + async promptCustomMessage( + message: { customType: string; content: string; details?: unknown }, + options?: { streamingBehavior?: "steer" | "followUp"; queueChipText?: string }, + ): Promise { this.customMessages.push(message); + this.customMessageOptions.push(options); this.isStreaming = true; const assistantMessage = makeAssistantMessage("skill pong"); for (const listener of this.#listeners) { @@ -1443,6 +1448,7 @@ describe("ACP agent", () => { expect(customMessage.content).toContain(`[Skill directory: ${skillDir}]`); expect(customMessage.content).toMatch(/[Rr]esolve any relative paths/); expect(customMessage.content).toContain("User: extra context"); + expect(session.customMessageOptions[0]).toEqual({ streamingBehavior: "steer" }); harness.abortController.abort(); await Bun.sleep(0); diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index 7b4b045d6..5d01d6b80 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -15,7 +15,7 @@ import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async"; import type { Rule } from "@oh-my-pi/pi-coding-agent/capability/rule"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { type SettingPath, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { TtsrManager } from "@oh-my-pi/pi-coding-agent/export/ttsr"; import type { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; @@ -68,7 +68,7 @@ describe("AgentSession concurrent prompt guard", () => { AsyncJobManager.resetForTests(); }); - async function createSession() { + async function createSession(settingsOverrides?: Partial>) { const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let abortSignal: AbortSignal | undefined; @@ -100,7 +100,7 @@ describe("AgentSession concurrent prompt guard", () => { }); const sessionManager = SessionManager.inMemory(); - const settings = Settings.isolated(); + const settings = Settings.isolated(settingsOverrides); const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db")); authStorages.push(authStorage); const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); @@ -176,6 +176,86 @@ describe("AgentSession concurrent prompt guard", () => { await firstPrompt.catch(() => {}); }); + it("queues sendUserMessage as steer while streaming without AgentBusyError", async () => { + await createSession(); + + const firstPrompt = session.prompt("First message"); + await waitFor(() => session.isStreaming); + + // The first agent loop may dequeue a steer before the assertion runs, so + // observe agent.steer itself rather than the residual queue length. + const steered: AgentMessage[] = []; + const originalSteer = session.agent.steer.bind(session.agent); + session.agent.steer = (message: AgentMessage) => { + steered.push(message); + originalSteer(message); + }; + + // Extension path: no deliverAs while busy must queue, not throw. + await expect(session.sendUserMessage("hello from extension")).resolves.toBeUndefined(); + expect(steered).toHaveLength(1); + const queued = steered[0]; + expect(queued?.role).toBe("user"); + if (queued?.role === "user") { + expect(queued.content).toEqual([{ type: "text", text: "hello from extension" }]); + expect(queued.steering).toBe(true); + } + + session.agent.clearAllQueues(); + await session.abort(); + await firstPrompt.catch(() => {}); + }); + + it("sendUserMessage without deliverAs preserves prompt-flow keyword notices while streaming", async () => { + await createSession({ "magicKeywords.enabled": true, "magicKeywords.ultrathink": true }); + + const firstPrompt = session.prompt("First message"); + await waitFor(() => session.isStreaming); + + try { + await session.sendUserMessage("ultrathink fix via extension"); + const queuedShape = session.agent + .peekSteeringQueue() + .map(message => (message.role === "custom" ? message.customType : message.role)); + expect(queuedShape).toEqual(["ultrathink-notice", "user"]); + expect(session.getQueuedMessages()).toEqual({ + steering: ["ultrathink fix via extension"], + followUp: [], + }); + } finally { + session.agent.clearAllQueues(); + await session.abort(); + await firstPrompt.catch(() => {}); + } + }); + + it("sendUserMessage without deliverAs starts a normal prompt when idle", async () => { + await createSession(); + + let rejected: unknown; + let settled = false; + const turn = session + .sendUserMessage("Idle extension message") + .catch(error => { + rejected = error; + }) + .finally(() => { + settled = true; + }); + + try { + await waitFor(() => session.isStreaming || settled); + if (rejected) throw rejected; + + expect(session.isStreaming).toBe(true); + expect(settled).toBe(false); + expect(session.getQueuedMessages()).toEqual({ steering: [], followUp: [] }); + } finally { + await session.abort(); + await turn; + } + }); + it("delivers hidden nextTurn stop reactions through the next LLM call without exposing them in the visible queue", async () => { const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let firstStream: AssistantMessageEventStream | undefined; diff --git a/packages/coding-agent/test/rpc-skill-command.test.ts b/packages/coding-agent/test/rpc-skill-command.test.ts index 068b55bfb..02795389b 100644 --- a/packages/coding-agent/test/rpc-skill-command.test.ts +++ b/packages/coding-agent/test/rpc-skill-command.test.ts @@ -16,6 +16,7 @@ describe("tryRunRpcSkillCommand", () => { ); let message: Pick | undefined; + let options: { streamingBehavior?: "steer" | "followUp" } | undefined; const handled = await tryRunRpcSkillCommand( { @@ -23,8 +24,9 @@ describe("tryRunRpcSkillCommand", () => { skills: [ { name: "reviewer", description: "Review code", filePath: skillPath, baseDir: dir, source: "project" }, ], - async promptCustomMessage(nextMessage: typeof message) { + async promptCustomMessage(nextMessage: typeof message, nextOptions?: typeof options) { message = nextMessage; + options = nextOptions; }, }, "/skill:reviewer focus on risks", @@ -39,10 +41,43 @@ describe("tryRunRpcSkillCommand", () => { expect(message?.content).toContain("User: focus on risks"); expect(message?.display).toBe(true); expect(message?.attribution).toBe("user"); + expect(options).toEqual({ streamingBehavior: "steer" }); await removeWithRetries(dir); }); + test("honors the RPC prompt streaming behavior for registered /skill commands", async () => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), `omp-rpc-skill-${Snowflake.next()}-`)); + const skillPath = path.join(dir, "SKILL.md"); + await Bun.write( + skillPath, + "---\nname: reviewer\ndescription: Review code\n---\n\nReview the supplied code carefully.\n", + ); + + let options: { streamingBehavior?: "steer" | "followUp" } | undefined; + try { + const handled = await tryRunRpcSkillCommand( + { + skillsSettings: { enableSkillCommands: true }, + skills: [ + { name: "reviewer", description: "Review code", filePath: skillPath, baseDir: dir, source: "project" }, + ], + async promptCustomMessage(nextMessage, nextOptions) { + expect(nextMessage.customType).toBe(SKILL_PROMPT_MESSAGE_TYPE); + options = nextOptions; + }, + }, + "/skill:reviewer wait for the current turn", + "followUp", + ); + + expect(handled).toEqual({ agentInvoked: true }); + expect(options?.streamingBehavior).toBe("followUp"); + } finally { + await removeWithRetries(dir); + } + }); + test("ignores unknown skill commands so normal prompt handling can continue", async () => { const handled = await tryRunRpcSkillCommand( { From c94487056637dad4ab5b9a4af17cfa484a1b18bd Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:20:08 +0200 Subject: [PATCH 17/59] fix(coding-agent): implemented pi's ui.addAutocompleteProvider API Extensions calling ctx.ui.addAutocompleteProvider (e.g. @ff-labs/pi-fff) crashed at load with 'TypeError: ... is not a function' because omp's ExtensionAPI.ui omitted pi's autocomplete-provider API; the throw also aborted the rest of a try/catch-guarded session_start init. ExtensionUIContext now declares addAutocompleteProvider(factory). Interactive mode stacks each factory on the built-in editor provider in registration order, re-applies the stack on every slash-command refresh, and skips throwing/malformed factories; RPC, ACP, and headless contexts accept the factory as a no-op, matching upstream pi's RPC behavior. Fixes #4919 --- docs/extensions.md | 5 +- docs/porting-from-pi-mono.md | 1 + packages/coding-agent/CHANGELOG.md | 1 + .../src/extensibility/extensions/runner.ts | 1 + .../src/extensibility/extensions/types.ts | 13 +- .../coding-agent/src/modes/acp/acp-agent.ts | 1 + .../extension-ui-controller.test.ts | 16 ++ .../controllers/extension-ui-controller.ts | 1 + .../src/modes/interactive-mode.ts | 45 ++- .../coding-agent/src/modes/rpc/rpc-mode.ts | 4 + packages/coding-agent/src/modes/types.ts | 3 + .../coding-agent/src/session/agent-session.ts | 1 + .../test/extensions-runner.test.ts | 1 + ...19-extension-autocomplete-provider.test.ts | 256 ++++++++++++++++++ 14 files changed, 344 insertions(+), 5 deletions(-) create mode 100644 packages/coding-agent/test/issue-4919-extension-autocomplete-provider.test.ts diff --git a/docs/extensions.md b/docs/extensions.md index a3ac610f9..7702c1767 100644 --- a/docs/extensions.md +++ b/docs/extensions.md @@ -311,6 +311,7 @@ Supported: - dialogs: `select`, `confirm`, `input`, `editor` - input editing: `setEditorText`, `getEditorText`, `pasteToEditor`, `editor` +- autocomplete stacking: `addAutocompleteProvider(factory)` wraps the built-in editor provider (factories apply in registration order and re-apply on every slash-command refresh) - terminal title and working message (`setTitle`, `setWorkingMessage`) - notifications/status/editor text/terminal input/custom overlays - theme listing/loading by name (`setTheme` supports string names) @@ -334,7 +335,7 @@ Unsupported/no-op in RPC implementation: - `onTerminalInput` - `custom` -- `setFooter`, `setHeader`, `setEditorComponent` +- `setFooter`, `setHeader`, `setEditorComponent`, `addAutocompleteProvider` - `setWorkingMessage` - theme switching/loading (`setTheme` returns failure) - tool expansion controls are inert @@ -345,7 +346,7 @@ When no UI context is supplied to runner init, `ctx.hasUI` is `false` and method ### ACP mode -ACP installs an elicitation-bridged UI context (`createAcpExtensionUiContext` in `acp-agent.ts`). `ctx.hasUI` is `true` while only `select`/`confirm`/`input` round-trip (as ACP elicitations; defaults are returned when the client lacks the `elicitation.form` capability). The non-elicitation surface (widgets, editor, theming, terminal input) is stubbed no-op. +ACP installs an elicitation-bridged UI context (`createAcpExtensionUiContext` in `acp-agent.ts`). `ctx.hasUI` is `true` while only `select`/`confirm`/`input` round-trip (as ACP elicitations; defaults are returned when the client lacks the `elicitation.form` capability). The non-elicitation surface (widgets, editor, theming, terminal input, autocomplete stacking) is stubbed no-op. ## Session and state patterns diff --git a/docs/porting-from-pi-mono.md b/docs/porting-from-pi-mono.md index 427d47394..6234a940d 100644 --- a/docs/porting-from-pi-mono.md +++ b/docs/porting-from-pi-mono.md @@ -307,6 +307,7 @@ Our fork has architectural decisions that differ from upstream. **Do not port th | `FooterDataProvider` class | `StatusLineComponent` | Simpler, integrated status line | | `ctx.ui.setHeader()` / `ctx.ui.setFooter()` | No-op stubs in current extension contexts | Not currently wired to replace the TUI status/header UI | | `ctx.ui.setEditorComponent()` | Wired in interactive mode; no-op stubs in ACP/RPC/headless contexts | Custom editor replacement works in the interactive TUI; non-TUI runtimes keep stubs | +| `ctx.ui.addAutocompleteProvider()` | Wired in interactive mode; no-op stubs in ACP/RPC/headless contexts | Factory wrapping matches upstream; omp's editor has no custom `triggerCharacters`, so wrapped providers surface at the built-in trigger points | | `InteractiveModeOptions` options object | Positional constructor args (options type still exported) | Keep constructor signature; update the type when upstream adds fields | ### Component Naming diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a24b42882..45371a000 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -8,6 +8,7 @@ - Fixed `grep` explicit line selectors on directory searches so they filter each matched file by line number instead of aborting with a single-file-only error ([#4898](https://github.com/can1357/oh-my-pi/issues/4898)). - Fixed Read tool previews dropping explicit `selector` arguments, so line ranges and `raw` modifiers render in terminal read call titles again ([#4899](https://github.com/can1357/oh-my-pi/issues/4899)). - Fixed named profiles dropping default user keybindings from `~/.omp/agent/keybindings.*`; profile keybindings now inherit those defaults and override only the keys they define ([#4867](https://github.com/can1357/oh-my-pi/issues/4867)). +- Fixed pi extensions calling `ctx.ui.addAutocompleteProvider(...)` crashing at load with `TypeError: ... is not a function` — a failure that, for extensions guarding init in one `try/catch` (e.g. `@ff-labs/pi-fff`), aborted the extension's entire initialization. `ExtensionUIContext` now implements pi's autocomplete-provider API: interactive mode stacks each registered factory on top of the built-in editor provider (re-applied on every slash-command refresh, with throwing or malformed factories skipped), while RPC/ACP/headless contexts accept the factory as a no-op. ([#4919](https://github.com/can1357/oh-my-pi/issues/4919)) ## [16.3.12] - 2026-07-08 diff --git a/packages/coding-agent/src/extensibility/extensions/runner.ts b/packages/coding-agent/src/extensibility/extensions/runner.ts index e0226e0b9..bec6b662c 100644 --- a/packages/coding-agent/src/extensibility/extensions/runner.ts +++ b/packages/coding-agent/src/extensibility/extensions/runner.ts @@ -207,6 +207,7 @@ const noOpUIContext: ExtensionUIContext = { pasteToEditor: () => {}, getEditorText: () => "", editor: async () => undefined, + addAutocompleteProvider: () => {}, setEditorComponent: () => {}, get theme() { return theme; diff --git a/packages/coding-agent/src/extensibility/extensions/types.ts b/packages/coding-agent/src/extensibility/extensions/types.ts index 658b349f2..cfff0ed01 100644 --- a/packages/coding-agent/src/extensibility/extensions/types.ts +++ b/packages/coding-agent/src/extensibility/extensions/types.ts @@ -30,7 +30,7 @@ import type { TSchema, } from "@oh-my-pi/pi-ai"; import type { OAuthCredentials, OAuthLoginCallbacks } from "@oh-my-pi/pi-ai/oauth/types"; -import type { AutocompleteItem, Component, EditorTheme, KeyId, TUI } from "@oh-my-pi/pi-tui"; +import type { AutocompleteItem, AutocompleteProvider, Component, EditorTheme, KeyId, TUI } from "@oh-my-pi/pi-tui"; import type { logger as PiLogger } from "@oh-my-pi/pi-utils"; import type { Type as arktype } from "arktype"; import type * as zod from "zod/v4"; @@ -163,6 +163,9 @@ export type ExtensionUiComponent = Component & { dispose?(): void }; export type ExtensionUiComponentFactory = (tui: TUI, theme: Theme) => ExtensionUiComponent; export type ExtensionWidgetContent = string[] | ExtensionUiComponentFactory | undefined; +/** Wrap the current autocomplete provider with additional behavior (pi-compatible). */ +export type AutocompleteProviderFactory = (current: AutocompleteProvider) => AutocompleteProvider; + /** * UI context for extensions to request interactive UI. * Each mode (interactive, RPC, print) provides its own implementation. @@ -243,6 +246,14 @@ export interface ExtensionUIContext { editorOptions?: { promptStyle?: boolean }, ): Promise; + /** + * Stack additional autocomplete behavior on top of the built-in provider + * (pi-compatible). Interactive mode rebuilds the editor's provider through + * every registered factory, in registration order; headless modes (print, + * RPC, ACP, subagents) accept and ignore the factory. + */ + addAutocompleteProvider(factory: AutocompleteProviderFactory): void; + /** * Set a custom editor component via factory function, or `undefined` to restore the default editor. * diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index b1d56128f..efdfaa45b 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -425,6 +425,7 @@ export function createAcpExtensionUiContext( setEditorText: () => {}, getEditorText: () => "", editor: async () => undefined, + addAutocompleteProvider: () => {}, setEditorComponent: () => {}, get theme() { return theme; diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.test.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.test.ts index de29d689b..1dc3e51f2 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.test.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.test.ts @@ -8,6 +8,7 @@ import { ExtensionUiController } from "./extension-ui-controller"; function makeHarness() { const editor = new CustomEditor(getEditorTheme()); const requestRender = vi.fn(); + const addAutocompleteProvider = vi.fn(); let uiContext: ExtensionUIContext | undefined; const ctx = { editor, @@ -21,11 +22,13 @@ function makeHarness() { expect(hasUI).toBe(true); uiContext = context; }, + addAutocompleteProvider, } as unknown as InteractiveModeContext; return { editor, requestRender, + addAutocompleteProvider, async init(): Promise { await new ExtensionUiController(ctx).initHooksAndCustomTools(); expect(uiContext).toBeDefined(); @@ -55,4 +58,17 @@ describe("ExtensionUiController editor UI", () => { expect(harness.editor.getText()).toBe("hello"); expect(harness.requestRender).toHaveBeenCalledTimes(1); }); + + it("bridges addAutocompleteProvider factories to the interactive mode context (#4919)", async () => { + const harness = makeHarness(); + const ui = await harness.init(); + + expect(typeof ui.addAutocompleteProvider).toBe("function"); + + const factory = (current: unknown) => current as never; + ui.addAutocompleteProvider(factory); + + expect(harness.addAutocompleteProvider).toHaveBeenCalledTimes(1); + expect(harness.addAutocompleteProvider).toHaveBeenCalledWith(factory); + }); }); diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index baea76ffd..684642b12 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -82,6 +82,7 @@ export class ExtensionUiController { getEditorText: () => this.ctx.editor.getText(), editor: (title, prefill, dialogOptions, editorOptions) => this.showCollabAwareEditor(title, prefill, dialogOptions, editorOptions), + addAutocompleteProvider: factory => this.ctx.addAutocompleteProvider(factory), get theme() { return theme; }, diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 78bec222b..5bd64e097 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -16,6 +16,7 @@ import type { CompactionOutcome } from "@oh-my-pi/pi-agent-core/compaction"; import type { AssistantMessage, ImageContent, Message, Model, Usage, UsageReport } from "@oh-my-pi/pi-ai"; import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models"; import type { + AutocompleteProvider, Component, EditorTheme, LoaderMessageColorFn, @@ -59,6 +60,7 @@ import { applyProviderGlobalsFromSettings } from "../config/provider-globals"; import { isSettingsInitialized, onStatusLineSessionAccentChanged, Settings, settings } from "../config/settings"; import { clearClaudePluginRootsCache } from "../discovery/helpers"; import type { + AutocompleteProviderFactory, ContextUsage, ExtensionUIContext, ExtensionUIDialogOptions, @@ -516,6 +518,10 @@ export class InteractiveMode implements InteractiveModeContext { collabGuest?: CollabGuestLink; #pendingSlashCommands: SlashCommand[] = []; + /** Built-in editor autocomplete provider, before extension wrapping. */ + #baseAutocompleteProvider: AutocompleteProvider | undefined; + /** Extension-registered provider factories, applied in registration order (#4919). */ + #autocompleteProviderFactories: AutocompleteProviderFactory[] = []; #cleanupUnsubscribe?: () => void; #signalTeardown?: SessionTeardown; readonly #version: string; @@ -1094,14 +1100,49 @@ export class InteractiveMode implements InteractiveModeContext { // source suffix (e.g. "Review code (project)"), so pass it through verbatim. description: template.description, })); - const autocompleteProvider = this.#inputController.createAutocompleteProvider( + this.#baseAutocompleteProvider = this.#inputController.createAutocompleteProvider( [...this.#pendingSlashCommands, ...fileSlashCommands, ...promptTemplateCommands], basePath, ); - this.editor.setAutocompleteProvider(autocompleteProvider); + this.#applyAutocompleteProvider(); this.session.setSlashCommands(fileCommands); } + /** + * Rebuild the editor's autocomplete provider: the built-in provider wrapped + * by every extension-registered factory, in registration order. A factory + * that throws or returns a malformed provider is skipped so one broken + * extension cannot take down core autocomplete. + */ + #applyAutocompleteProvider(): void { + const base = this.#baseAutocompleteProvider; + if (!base) return; + let provider = base; + for (const factory of this.#autocompleteProviderFactories) { + try { + const wrapped = factory(provider); + if ( + wrapped && + typeof wrapped.getSuggestions === "function" && + typeof wrapped.applyCompletion === "function" + ) { + provider = wrapped; + } else { + logger.warn("Extension autocomplete provider factory returned an invalid provider; skipping it"); + } + } catch (error) { + logger.warn("Extension autocomplete provider factory threw; skipping it", { error: String(error) }); + } + } + this.editor.setAutocompleteProvider(provider); + } + + /** Stack extension autocomplete behavior on top of the built-in editor provider (#4919). */ + addAutocompleteProvider(factory: AutocompleteProviderFactory): void { + this.#autocompleteProviderFactories.push(factory); + this.#applyAutocompleteProvider(); + } + /** * Re-point the process and every cwd-derived cache at `newCwd` after the * active session's working directory changed (`/move` relocation or resuming diff --git a/packages/coding-agent/src/modes/rpc/rpc-mode.ts b/packages/coding-agent/src/modes/rpc/rpc-mode.ts index 7dabb5849..16d1cf1b3 100644 --- a/packages/coding-agent/src/modes/rpc/rpc-mode.ts +++ b/packages/coding-agent/src/modes/rpc/rpc-mode.ts @@ -759,6 +759,10 @@ export async function runRpcMode( return requestRpcEditor(this.pendingRequests, this.output, title, prefill, dialogOptions, editorOptions); } + addAutocompleteProvider(): void { + // Autocomplete provider composition is not supported in RPC mode + } + get theme(): Theme { return theme; } diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 6eb2c0f4a..8c66eb8b7 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -7,6 +7,7 @@ import type { CollabHost } from "../collab/host"; import type { KeybindingsManager } from "../config/keybindings"; import type { Settings } from "../config/settings"; import type { + AutocompleteProviderFactory, ExtensionUIContext, ExtensionUIDialogOptions, ExtensionUISelectItem, @@ -218,6 +219,8 @@ export interface InteractiveModeContext { // Extension UI integration setToolUIContext(uiContext: ExtensionUIContext, hasUI: boolean): void; initializeHookRunner(uiContext: ExtensionUIContext, hasUI: boolean): void; + /** Stack extension autocomplete behavior on top of the built-in editor provider. */ + addAutocompleteProvider(factory: AutocompleteProviderFactory): void; setEditorComponent( factory: ((tui: TUI, theme: EditorTheme, keybindings: KeybindingsManager) => CustomEditor) | undefined, ): void; diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 992f5bb6d..3b4274e5f 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1180,6 +1180,7 @@ const noOpUIContext: ExtensionUIContext = { pasteToEditor: () => {}, getEditorText: () => "", editor: async () => undefined, + addAutocompleteProvider: () => {}, get theme() { return theme; }, diff --git a/packages/coding-agent/test/extensions-runner.test.ts b/packages/coding-agent/test/extensions-runner.test.ts index d24518737..0a2680515 100644 --- a/packages/coding-agent/test/extensions-runner.test.ts +++ b/packages/coding-agent/test/extensions-runner.test.ts @@ -1169,6 +1169,7 @@ describe("ExtensionRunner", () => { setEditorText: () => {}, getEditorText: () => "", editor: async () => undefined, + addAutocompleteProvider: () => {}, setEditorComponent: () => {}, get theme() { return {} as never; diff --git a/packages/coding-agent/test/issue-4919-extension-autocomplete-provider.test.ts b/packages/coding-agent/test/issue-4919-extension-autocomplete-provider.test.ts new file mode 100644 index 000000000..31fb9c42d --- /dev/null +++ b/packages/coding-agent/test/issue-4919-extension-autocomplete-provider.test.ts @@ -0,0 +1,256 @@ +/** + * Issue #4919: a pi extension calling `ctx.ui.addAutocompleteProvider(...)` in its + * `session_start` handler crashed at load under omp — the method was absent from + * `ExtensionUIContext`, so the call threw `TypeError: ... is not a function` and + * (for extensions that wrap init in try/catch, e.g. @ff-labs/pi-fff) aborted the + * extension's entire initialization. + * + * These tests pin the pi-compatible contract: + * - headless contexts accept the factory as a no-op instead of throwing, and + * - interactive mode stacks each factory on top of the built-in editor provider. + * + * NOTE: imports are relative (`../src/...`) so the tests exercise this checkout + * even when `node_modules/@oh-my-pi/pi-coding-agent` resolves elsewhere. + */ + +import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import { type Api, Effort, type Model } from "@oh-my-pi/pi-ai"; +import type { AutocompleteProvider } from "@oh-my-pi/pi-tui"; +import { logger, TempDir } from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; +import { ModelRegistry } from "../src/config/model-registry"; +import { resetSettingsForTest, Settings } from "../src/config/settings"; +import { loadExtensions } from "../src/extensibility/extensions/loader"; +import { ExtensionRunner } from "../src/extensibility/extensions/runner"; +import { InteractiveMode } from "../src/modes/interactive-mode"; +import { initTheme } from "../src/modes/theme/theme"; +import { AgentSession } from "../src/session/agent-session"; +import { AuthStorage } from "../src/session/auth-storage"; +import { SessionManager } from "../src/session/session-manager"; + +function makeTool(name: string): AgentTool { + return { + name, + label: name, + description: `Fake ${name}`, + parameters: type({}), + async execute() { + return { content: [{ type: "text" as const, text: "ok" }] }; + }, + }; +} + +/** + * Wrap `current` the way a well-behaved pi extension does: contribute items for + * its own trigger prefix, delegate everything else to the wrapped provider. + */ +function makeWrappingFactory(tag: string): (current: AutocompleteProvider) => AutocompleteProvider { + return current => ({ + async getSuggestions(lines, cursorLine, cursorCol) { + const line = lines[cursorLine] ?? ""; + if (line.startsWith("##")) { + const base = await current.getSuggestions(lines, cursorLine, cursorCol); + return { + items: [...(base?.items ?? []), { value: tag, label: tag }], + prefix: base?.prefix ?? line.slice(0, cursorCol), + }; + } + return current.getSuggestions(lines, cursorLine, cursorCol); + }, + applyCompletion(lines, cursorLine, cursorCol, item, prefix) { + return current.applyCompletion(lines, cursorLine, cursorCol, item, prefix); + }, + }); +} + +describe("extension autocomplete provider API (#4919)", () => { + let tempDir: TempDir; + let authStorage: AuthStorage; + let registry: ModelRegistry; + let model: Model; + let tools: AgentTool[]; + let originalHome: string | undefined; + let mode: InteractiveMode | undefined; + let session: AgentSession | undefined; + + beforeAll(async () => { + initTheme(); + resetSettingsForTest(); + // One empty temp dir doubles as the project cwd and the (isolated) home + // directory, keeping `refreshSlashCommandState`'s capability scan off the + // real home dir (mirrors the prompt-template autocomplete harness). + tempDir = TempDir.createSync("@pi-ext-autocomplete-"); + originalHome = process.env.HOME; + process.env.HOME = tempDir.path(); + await Settings.init({ inMemory: true, cwd: tempDir.path() }); + Settings.instance.set("startup.quiet", true); + authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + registry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml")); + const resolved = registry.find("anthropic", "claude-sonnet-4-5"); + if (!resolved) throw new Error("Expected anthropic model claude-sonnet-4-5 to exist"); + model = resolved; + tools = [makeTool("read")]; + }); + + beforeEach(() => { + vi.spyOn(os, "homedir").mockReturnValue(tempDir.path()); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + mode?.stop(); + await session?.dispose(); + mode = undefined; + session = undefined; + }); + + afterAll(() => { + authStorage?.close(); + if (originalHome === undefined) delete process.env.HOME; + else process.env.HOME = originalHome; + tempDir?.removeSync(); + resetSettingsForTest(); + }); + + function createHarness(): { mode: InteractiveMode; session: AgentSession } { + const manager = SessionManager.create(tempDir.path(), path.join(tempDir.path(), `active-${Bun.nanoseconds()}`)); + const created = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools, + messages: [], + thinkingLevel: Effort.Medium, + }, + }), + sessionManager: manager, + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry: registry, + toolRegistry: new Map(tools.map(tool => [tool.name, tool])), + promptTemplates: [], + }); + const createdMode = new InteractiveMode(created, "test"); + session = created; + mode = createdMode; + return { mode: createdMode, session: created }; + } + + function captureAutocompleteProvider(target: InteractiveMode): { current: AutocompleteProvider | undefined } { + const slot: { current: AutocompleteProvider | undefined } = { current: undefined }; + vi.spyOn(target.editor, "setAutocompleteProvider").mockImplementation(provider => { + slot.current = provider; + }); + return slot; + } + + it("does not abort a session_start handler that registers a provider without UI", async () => { + // Mimics @ff-labs/pi-fff: registerAutocompleteProvider(ctx) runs first and + // unconditionally inside the try/catch that guards the whole init routine. + const extensionsDir = path.join(tempDir.path(), "runner-extensions"); + fs.mkdirSync(extensionsDir, { recursive: true }); + const markerPath = path.join(extensionsDir, "init-marker.txt"); + const extPath = path.join(extensionsDir, "fff-like.ts"); + fs.writeFileSync( + extPath, + `import * as fs from "node:fs"; +export default function (pi) { + pi.on("session_start", async (_event, ctx) => { + try { + ctx.ui.addAutocompleteProvider((current) => current); + // "Rest of init" — on baseline the call above throws and this never runs. + fs.writeFileSync(${JSON.stringify(markerPath)}, "initialized"); + } catch (error) { + fs.writeFileSync( + ${JSON.stringify(markerPath)}, + "failed: " + (error instanceof Error ? error.message : String(error)), + ); + } + }); +} +`, + ); + + const result = await loadExtensions([extPath], tempDir.path()); + expect(result.errors).toEqual([]); + const runner = new ExtensionRunner( + result.extensions, + result.runtime, + tempDir.path(), + SessionManager.inMemory(), + registry, + ); + const surfaced: string[] = []; + runner.onError(error => { + surfaced.push(error.error); + }); + + await runner.emit({ type: "session_start" }); + + expect(surfaced).toEqual([]); + expect(fs.readFileSync(markerPath, "utf8")).toBe("initialized"); + }); + + it("stacks extension factories on top of the built-in editor provider", async () => { + const created = createHarness(); + const slot = captureAutocompleteProvider(created.mode); + + // Registration before the first refresh (session_start fires before init's + // refreshSlashCommandState) must land once the base provider exists. + created.mode.addAutocompleteProvider(makeWrappingFactory("##fff-first")); + await created.mode.refreshSlashCommandState(tempDir.path()); + + const provider = slot.current; + expect(provider).toBeDefined(); + + // The extension's trigger prefix surfaces its items... + const extension = await provider!.getSuggestions(["##"], 0, 2); + expect(extension?.items.map(item => item.value)).toContain("##fff-first"); + + // ...while built-in slash completion still flows through the wrapper. + const slash = await provider!.getSuggestions(["/"], 0, 1); + expect(slash?.items.map(item => item.value)).toContain("model"); + + // Registration after the refresh re-applies immediately, preserving the chain. + created.mode.addAutocompleteProvider(makeWrappingFactory("##fff-second")); + const restacked = slot.current; + expect(restacked).toBeDefined(); + expect(restacked).not.toBe(provider); + + const chained = await restacked!.getSuggestions(["##"], 0, 2); + const values = chained?.items.map(item => item.value) ?? []; + expect(values).toContain("##fff-first"); + expect(values).toContain("##fff-second"); + }); + + it("skips broken factories without losing core autocomplete or healthy wrappers", async () => { + const created = createHarness(); + const slot = captureAutocompleteProvider(created.mode); + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + created.mode.addAutocompleteProvider(() => { + throw new Error("boom"); + }); + created.mode.addAutocompleteProvider(() => ({}) as AutocompleteProvider); + created.mode.addAutocompleteProvider(makeWrappingFactory("##healthy")); + await created.mode.refreshSlashCommandState(tempDir.path()); + + const provider = slot.current; + expect(provider).toBeDefined(); + + const slash = await provider!.getSuggestions(["/"], 0, 1); + expect(slash?.items.map(item => item.value)).toContain("model"); + + const extension = await provider!.getSuggestions(["##"], 0, 2); + expect(extension?.items.map(item => item.value)).toContain("##healthy"); + + expect(warnSpy.mock.calls.some(([message]) => String(message).includes("autocomplete provider factory"))).toBe( + true, + ); + }); +}); From 3a9bcc9ed46a2b534e8038ef6d054ca3965e08b1 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 03:56:24 +0000 Subject: [PATCH 18/59] fix(agent): let role agents inherit configured effort - Removed bundled reviewer and plan thinking-level hard pins so their model roles can supply configured effort. - Added regression coverage for bundled reviewer and plan parsing. - Updated the coding-agent changelog. Fixes #4761 --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/prompts/agents/plan.md | 1 - .../src/prompts/agents/reviewer.md | 1 - .../test/bundled-agent-parsing.test.ts | 66 +++++++++++++++++++ 4 files changed, 67 insertions(+), 2 deletions(-) create mode 100644 packages/coding-agent/test/bundled-agent-parsing.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 45371a000..4e14e530b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -9,6 +9,7 @@ - Fixed Read tool previews dropping explicit `selector` arguments, so line ranges and `raw` modifiers render in terminal read call titles again ([#4899](https://github.com/can1357/oh-my-pi/issues/4899)). - Fixed named profiles dropping default user keybindings from `~/.omp/agent/keybindings.*`; profile keybindings now inherit those defaults and override only the keys they define ([#4867](https://github.com/can1357/oh-my-pi/issues/4867)). - Fixed pi extensions calling `ctx.ui.addAutocompleteProvider(...)` crashing at load with `TypeError: ... is not a function` — a failure that, for extensions guarding init in one `try/catch` (e.g. `@ff-labs/pi-fff`), aborted the extension's entire initialization. `ExtensionUIContext` now implements pi's autocomplete-provider API: interactive mode stacks each registered factory on top of the built-in editor provider (re-applied on every slash-command refresh, with throwing or malformed factories skipped), while RPC/ACP/headless contexts accept the factory as a no-op. ([#4919](https://github.com/can1357/oh-my-pi/issues/4919)) +- Fixed bundled reviewer and plan subagents to inherit their model roles' explicit thinking effort suffixes instead of pinning `high` ([#4761](https://github.com/can1357/oh-my-pi/issues/4761)). ## [16.3.12] - 2026-07-08 diff --git a/packages/coding-agent/src/prompts/agents/plan.md b/packages/coding-agent/src/prompts/agents/plan.md index 09c9fbb4e..528a38d42 100644 --- a/packages/coding-agent/src/prompts/agents/plan.md +++ b/packages/coding-agent/src/prompts/agents/plan.md @@ -4,7 +4,6 @@ description: Software architect for complex multi-file architectural decisions. tools: read, grep, glob, bash, lsp, web_search, ast_grep spawns: explore model: pi/plan, pi/slow -thinking-level: high --- Analyze the codebase and the user's request. Produce a detailed implementation plan. diff --git a/packages/coding-agent/src/prompts/agents/reviewer.md b/packages/coding-agent/src/prompts/agents/reviewer.md index 3d9c40078..262d72dde 100644 --- a/packages/coding-agent/src/prompts/agents/reviewer.md +++ b/packages/coding-agent/src/prompts/agents/reviewer.md @@ -4,7 +4,6 @@ description: "Code review specialist for quality/security analysis" tools: read, grep, glob, bash, lsp, web_search, ast_grep spawns: explore model: pi/slow -thinking-level: high output: properties: overall_correctness: diff --git a/packages/coding-agent/test/bundled-agent-parsing.test.ts b/packages/coding-agent/test/bundled-agent-parsing.test.ts new file mode 100644 index 000000000..e66f41f0e --- /dev/null +++ b/packages/coding-agent/test/bundled-agent-parsing.test.ts @@ -0,0 +1,66 @@ +import { describe, expect, it } from "bun:test"; +import { Effort } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { + resolveAgentModelPatterns, + resolveModelOverride, +} from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { getBundledAgent } from "@oh-my-pi/pi-coding-agent/task/agents"; + +describe("bundled agent parsing", () => { + it("lets reviewer inherit thinking effort from its model role", () => { + const reviewer = getBundledAgent("reviewer"); + + expect(reviewer).toBeDefined(); + expect(reviewer?.source).toBe("bundled"); + expect(reviewer?.model).toEqual(["pi/slow"]); + expect(reviewer?.thinkingLevel).toBeUndefined(); + }); + + it("lets plan inherit thinking effort from its model role", () => { + const plan = getBundledAgent("plan"); + + expect(plan).toBeDefined(); + expect(plan?.source).toBe("bundled"); + expect(plan?.model).toEqual(["pi/plan", "pi/slow"]); + expect(plan?.thinkingLevel).toBeUndefined(); + }); + + // Issue #4761: with `modelRoles.slow: ...:xhigh`, the role's explicit effort + // suffix must survive agent-pattern expansion and model resolution for the + // bundled agents routed at that role. The executor picks + // `agent.thinkingLevel ?? resolvedThinkingLevel` (task/executor.ts), so a + // bundled frontmatter pin would mask the suffix — reviewer/plan declare none + // (asserted above) and the resolved level below is what the subagent runs at. + it("resolves the configured slow-role effort suffix for reviewer and plan", () => { + const gpt55 = buildModel({ + id: "gpt-5.5", + name: "GPT-5.5 Codex", + api: "openai-codex-responses", + provider: "openai-codex", + baseUrl: "https://chatgpt.com/backend-api/codex", + reasoning: true, + thinking: { mode: "effort", efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh] }, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 272000, + maxTokens: 128000, + }); + const settings = Settings.isolated({ + modelRoles: { slow: "openai-codex/gpt-5.5:xhigh", plan: "openai-codex/gpt-5.5:xhigh" }, + }); + const registry = { getAvailable: () => [gpt55] } as Parameters[1]; + + for (const name of ["reviewer", "plan"]) { + const agent = getBundledAgent(name); + expect(agent?.thinkingLevel).toBeUndefined(); + const patterns = resolveAgentModelPatterns({ agentModel: agent?.model, settings }); + const resolved = resolveModelOverride(patterns, registry, settings); + expect(resolved.model?.provider).toBe("openai-codex"); + expect(resolved.model?.id).toBe("gpt-5.5"); + expect(resolved.thinkingLevel).toBe(Effort.XHigh); + expect(resolved.explicitThinkingLevel).toBe(true); + } + }); +}); From 898643f9a9f8af72d42375a5f20d129e16da8e02 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:20:09 +0200 Subject: [PATCH 19/59] fix(coding-agent): refreshed expired OAuth in built-in discovery MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Built-in model discovery admitted providers via peekApiKey, which deliberately never refreshes OAuth rows, so a provider whose only stored credential was an expired OAuth token was silently dropped from online discovery and its token was never rotated (model selector 'refresh' stayed empty for logged-in users). Resolve built-in discovery keys through an online-only preflight that refreshes an expired stored OAuth credential, applying the disabled/ configured/targeted provider filters before the side-effecting resolution so refreshProvider(x) cannot rotate unrelated credentials. Offline discovery stays peek-only. Under online-if-uncached the preflight consults the same cache freshness the model manager uses (2h default TTL, 5min non-authoritative retry) so tokens refresh exactly when the manager will fetch — a fresh cache never triggers a token-endpoint call. Adopted from PR #4896 with two amendments: dropped an unrelated workflow-notice.md prompt edit, and aligned the preflight cache TTL with the manager's real 2h default (was 24h, which skipped the refresh on the common startup path for caches aged 2-24h; regression covered by the new online-if-uncached tests). Also corrected the stale 'Default: 24h' doc on cacheTtlMs in the catalog. Fixes #4893 Co-authored-by: roboomp --- packages/catalog/src/model-manager.ts | 2 +- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/config/model-registry.ts | 100 +++++++-- .../coding-agent/test/model-discovery.test.ts | 200 ++++++++++++++++++ .../model-registry-runtime-provider.test.ts | 46 ++++ 5 files changed, 328 insertions(+), 21 deletions(-) diff --git a/packages/catalog/src/model-manager.ts b/packages/catalog/src/model-manager.ts index eccec451f..33fc8cf88 100644 --- a/packages/catalog/src/model-manager.ts +++ b/packages/catalog/src/model-manager.ts @@ -35,7 +35,7 @@ export interface ModelManagerOptions(["llama-cpp-local", "lm-stud * so a successful fast path does not leave an armed timeout signal for concurrent GC. */ const RUNTIME_DYNAMIC_MODEL_FETCH_TIMEOUT_MS = 15_000; +// Built-in discovery preflight mirror of the catalog model-manager's private +// cache timings (model-manager.ts: DEFAULT_CACHE_TTL_MS / NON_AUTHORITATIVE_RETRY_MS). +// Built-in descriptors never override cacheTtlMs, so agreeing with these values +// makes the OAuth-refresh preflight fire exactly when the manager will fetch. +const BUILT_IN_DISCOVERY_CACHE_TTL_MS = 2 * 60 * 60 * 1000; +const BUILT_IN_DISCOVERY_NON_AUTHORITATIVE_RETRY_MS = 5 * 60 * 1000; import type { ApiKeyResolver, FetchImpl } from "@oh-my-pi/pi-ai"; import { registerOAuthProvider, unregisterOAuthProviders } from "@oh-my-pi/pi-ai/oauth"; @@ -1560,12 +1566,11 @@ export class ModelRegistry { ): Promise { // Skip providers already handled by configured discovery (e.g. user-configured ollama with discovery.type) const configuredDiscoveryProviders = new Set(this.#discoverableProviders.map(p => p.provider)); - const managerOptions = (await this.#collectBuiltInModelManagerOptions()).filter(opts => { - if (configuredDiscoveryProviders.has(opts.providerId)) { - return false; - } - return providerFilter ? providerFilter.has(opts.providerId) : true; - }); + const managerOptions = await this.#collectBuiltInModelManagerOptions( + strategy, + providerFilter, + configuredDiscoveryProviders, + ); if (managerOptions.length === 0) { return { models: [], authoritativeProviders: new Set() }; } @@ -1583,7 +1588,44 @@ export class ModelRegistry { return { models, authoritativeProviders }; } - async #collectBuiltInModelManagerOptions(): Promise[]> { + async #resolveBuiltInDiscoveryApiKey( + providerId: string, + strategy: ModelRefreshStrategy, + cacheProviderId: string, + ): Promise { + const peekedKey = await this.#peekApiKeyForProvider(providerId); + if (isAuthenticated(peekedKey) || strategy === "offline") { + return peekedKey; + } + const oauthCredentials = getOAuthCredentialsForProvider(this.authStorage, providerId); + if (oauthCredentials.length === 0) { + return peekedKey; + } + if (strategy === "online-if-uncached") { + // Mirror shouldFetchRemoteSources: built-in managers use the catalog's + // default TTL, so only refresh when the manager will actually fetch. + const cache = readModelCache(cacheProviderId, BUILT_IN_DISCOVERY_CACHE_TTL_MS, Date.now, this.#cacheDbPath); + const cacheAgeMs = cache ? Date.now() - cache.updatedAt : Number.POSITIVE_INFINITY; + if (cache?.fresh && (cache.authoritative || cacheAgeMs < BUILT_IN_DISCOVERY_NON_AUTHORITATIVE_RETRY_MS)) { + return peekedKey; + } + } + try { + return await this.getApiKeyForProvider(providerId); + } catch (error) { + logger.debug("OAuth refresh failed during model discovery preflight", { + provider: providerId, + error: error instanceof Error ? error.message : String(error), + }); + return peekedKey; + } + } + + async #collectBuiltInModelManagerOptions( + strategy: ModelRefreshStrategy, + providerFilter: ReadonlySet | undefined, + configuredDiscoveryProviders: ReadonlySet, + ): Promise[]> { const specialProviderDescriptors: Array<{ providerId: string; resolveKey: (value: string | undefined) => string | undefined; @@ -1622,20 +1664,33 @@ export class ModelRegistry { }, ]; const disabledProviders = getDisabledProviderIdsFromSettings(); - const standardProviderDescriptors = PROVIDER_DESCRIPTORS.filter( - descriptor => !disabledProviders.has(descriptor.providerId), + const standardProviderDescriptors = PROVIDER_DESCRIPTORS.filter(descriptor => { + if (disabledProviders.has(descriptor.providerId)) return false; + if (configuredDiscoveryProviders.has(descriptor.providerId)) return false; + return providerFilter ? providerFilter.has(descriptor.providerId) : true; + }); + const enabledSpecialProviderDescriptors = specialProviderDescriptors.filter(descriptor => { + if (disabledProviders.has(descriptor.providerId)) return false; + if (configuredDiscoveryProviders.has(descriptor.providerId)) return false; + return providerFilter ? providerFilter.has(descriptor.providerId) : true; + }); + const standardProviderKeys = await Promise.all( + standardProviderDescriptors.map(descriptor => { + const discoveryBaseUrl = + this.#runtimeProviderOverrides.get(descriptor.providerId)?.baseUrl ?? + this.#providerOverrides.get(descriptor.providerId)?.baseUrl ?? + this.getProviderBaseUrl(descriptor.providerId); + const cacheProviderId = + descriptor.createModelManagerOptions({ baseUrl: discoveryBaseUrl, fetch: this.#fetch }) + .cacheProviderId ?? descriptor.providerId; + return this.#resolveBuiltInDiscoveryApiKey(descriptor.providerId, strategy, cacheProviderId); + }), ); - const enabledSpecialProviderDescriptors = specialProviderDescriptors.filter( - descriptor => !disabledProviders.has(descriptor.providerId), + const specialKeys = await Promise.all( + enabledSpecialProviderDescriptors.map(descriptor => + this.#resolveBuiltInDiscoveryApiKey(descriptor.providerId, strategy, descriptor.providerId), + ), ); - // Use peekApiKey to avoid OAuth token refresh during discovery. - // The token is only needed if the dynamic fetch fires (cache miss), - // and failures there are handled gracefully. - const peekKey = (descriptor: { providerId: string }) => this.#peekApiKeyForProvider(descriptor.providerId); - const [standardProviderKeys, specialKeys] = await Promise.all([ - Promise.all(standardProviderDescriptors.map(peekKey)), - Promise.all(enabledSpecialProviderDescriptors.map(peekKey)), - ]); const options: ModelManagerOptions[] = []; for (let i = 0; i < standardProviderDescriptors.length; i++) { const descriptor = standardProviderDescriptors[i]; @@ -1670,7 +1725,12 @@ export class ModelRegistry { } // Append runtime model managers registered by extensions via fetchDynamicModels. for (const { options: managerOpts } of this.#runtimeModelManagers.values()) { - options.push(managerOpts); + if ( + !configuredDiscoveryProviders.has(managerOpts.providerId) && + (!providerFilter || providerFilter.has(managerOpts.providerId)) + ) { + options.push(managerOpts); + } } return options; } diff --git a/packages/coding-agent/test/model-discovery.test.ts b/packages/coding-agent/test/model-discovery.test.ts index dd7d89913..393e60645 100644 --- a/packages/coding-agent/test/model-discovery.test.ts +++ b/packages/coding-agent/test/model-discovery.test.ts @@ -3,6 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { Effort, type FetchImpl, type Model } from "@oh-my-pi/pi-ai"; +import type { OAuthCredentials } from "@oh-my-pi/pi-ai/oauth/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache"; import type { OpenAICompat } from "@oh-my-pi/pi-catalog/types"; @@ -19,15 +20,18 @@ describe("ModelRegistry runtime discovery", () => { let originalOllamaBaseUrl: string | undefined; let originalOllamaHost: string | undefined; let originalOllamaContextLength: string | undefined; + let originalAnthropicApiKey: string | undefined; beforeEach(async () => { resetSettingsForTest(); originalOllamaBaseUrl = Bun.env.OLLAMA_BASE_URL; originalOllamaHost = Bun.env.OLLAMA_HOST; originalOllamaContextLength = Bun.env.OLLAMA_CONTEXT_LENGTH; + originalAnthropicApiKey = Bun.env.ANTHROPIC_API_KEY; delete Bun.env.OLLAMA_BASE_URL; delete Bun.env.OLLAMA_HOST; delete Bun.env.OLLAMA_CONTEXT_LENGTH; + delete Bun.env.ANTHROPIC_API_KEY; tempDir = path.join(os.tmpdir(), `pi-test-model-registry-${Snowflake.next()}`); fs.mkdirSync(tempDir, { recursive: true }); modelsJsonPath = path.join(tempDir, "models.json"); @@ -55,6 +59,11 @@ describe("ModelRegistry runtime discovery", () => { } else { Bun.env.OLLAMA_CONTEXT_LENGTH = originalOllamaContextLength; } + if (originalAnthropicApiKey === undefined) { + delete Bun.env.ANTHROPIC_API_KEY; + } else { + Bun.env.ANTHROPIC_API_KEY = originalAnthropicApiKey; + } authStorage.close(); if (tempDir && fs.existsSync(tempDir)) { removeSyncWithRetries(tempDir); @@ -115,6 +124,197 @@ describe("ModelRegistry runtime discovery", () => { }; } + async function useAuthStorageWithRefreshTracker() { + authStorage.close(); + const refreshCalls: string[] = []; + authStorage = await AuthStorage.create(":memory:", { + refreshOAuthCredential: async (provider, _credentialId, credential): Promise => { + refreshCalls.push(provider); + return { + ...credential, + access: provider === "anthropic" ? "sk-ant-oat-fresh-anthropic" : `fresh-${provider}`, + expires: Date.now() + 3_600_000, + }; + }, + }); + return { refreshCalls }; + } + + type AnthropicDiscoveryCapture = { + modelListAuthorization?: string | null; + modelListXApiKey?: string | null; + modelListCalls: number; + }; + + function mockAnthropicModelsDiscovery(capture: AnthropicDiscoveryCapture): FetchImpl { + const endpointPrefix = "https://api.anthropic.com/"; + return async (input, init) => { + const url = String(input); + if (url === "https://models.dev/api.json") { + return Response.json({}); + } + if (url.startsWith(endpointPrefix) && url.endsWith("/models")) { + const headers = new Headers(init?.headers); + capture.modelListAuthorization = headers.get("authorization"); + capture.modelListXApiKey = headers.get("x-api-key"); + capture.modelListCalls++; + return Response.json({ + data: [{ id: "claude-regression-4893", display_name: "Claude Regression 4893" }], + }); + } + throw new Error(`Unexpected URL: ${url}`); + }; + } + + test("refreshProvider online refreshes expired anthropic OAuth before model discovery", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online"); + + expect(refreshCalls).toEqual(["anthropic"]); + expect(capture.modelListCalls).toBe(1); + expect(capture.modelListAuthorization).toBe("Bearer sk-ant-oat-fresh-anthropic"); + expect(capture.modelListXApiKey).toBeNull(); + expect(registry.find("anthropic", "claude-regression-4893")).toBeDefined(); + }); + + test("refreshProvider online does not refresh unrelated expired OAuth credentials", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + await authStorage.set("openai", { + type: "oauth", + access: "expired-openai", + refresh: "refresh-openai", + expires: Date.now() - 60_000, + }); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online"); + + expect(refreshCalls).toEqual(["anthropic"]); + expect(authStorage.getOAuthCredential("openai")?.access).toBe("expired-openai"); + expect(capture.modelListCalls).toBe(1); + }); + + test("refreshProvider offline does not touch expired OAuth credentials", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: async input => { + throw new Error(`Offline discovery should not fetch ${String(input)}`); + }, + }); + + await registry.refreshProvider("anthropic", "offline"); + + expect(refreshCalls).toEqual([]); + expect(authStorage.getOAuthCredential("anthropic")?.access).toBe("sk-ant-oat-expired-anthropic"); + }); + test("online-if-uncached refreshes expired OAuth when the discovery cache is stale for the model manager", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + // Older than the model manager's 2h default TTL: the manager WILL fetch, + // so the preflight must mint a fresh bearer first. + writeModelCache("anthropic", Date.now() - 3 * 60 * 60 * 1000, [], true, "", cacheDbPath); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online-if-uncached"); + + expect(refreshCalls).toEqual(["anthropic"]); + expect(capture.modelListCalls).toBe(1); + expect(capture.modelListAuthorization).toBe("Bearer sk-ant-oat-fresh-anthropic"); + }); + + test("online-if-uncached leaves expired OAuth untouched when the discovery cache is fresh", async () => { + const { refreshCalls } = await useAuthStorageWithRefreshTracker(); + await authStorage.set("anthropic", { + type: "oauth", + access: "sk-ant-oat-expired-anthropic", + refresh: "refresh-anthropic", + expires: Date.now() - 60_000, + }); + // Fresh authoritative cache: the manager will not fetch, so opening a + // cached model selector must not rotate (or risk disabling) credentials. + writeModelCache("anthropic", Date.now() - 60_000, [], true, "", cacheDbPath); + const capture: AnthropicDiscoveryCapture = { modelListCalls: 0 }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { + fetch: mockAnthropicModelsDiscovery(capture), + }); + + await registry.refreshProvider("anthropic", "online-if-uncached"); + + expect(refreshCalls).toEqual([]); + expect(capture.modelListCalls).toBe(0); + expect(authStorage.getOAuthCredential("anthropic")?.access).toBe("sk-ant-oat-expired-anthropic"); + }); + + test("configured discovery suppresses built-in special OAuth discovery", async () => { + await authStorage.set("google-gemini-cli", { + type: "oauth", + access: "fresh-google-gemini-cli", + refresh: "refresh-google-gemini-cli", + expires: Date.now() + 3_600_000, + }); + writeRawModelsJson({ + "google-gemini-cli": { + baseUrl: "http://127.0.0.1:4893", + api: "openai-completions", + auth: "none", + discovery: { type: "openai-models-list" }, + }, + }); + const unexpectedUrls: string[] = []; + const fetchMock: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:4893/v1/models") { + return Response.json({ + data: [{ id: "configured-gemini-cli-model", context_length: 65_536 }], + }); + } + unexpectedUrls.push(url); + throw new Error(`Unexpected URL: ${url}`); + }; + const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); + + await registry.refreshProvider("google-gemini-cli", "online"); + + expect(unexpectedUrls).toEqual([]); + const configuredModel = registry.find("google-gemini-cli", "configured-gemini-cli-model"); + expect(configuredModel?.baseUrl).toBe("http://127.0.0.1:4893"); + expect(configuredModel?.contextWindow).toBe(65_536); + }); + test("auto-discovers ollama models without provider config", async () => { const fetchMock = mockOllamaDiscovery(["phi4-mini"]); const registry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: fetchMock }); diff --git a/packages/coding-agent/test/model-registry-runtime-provider.test.ts b/packages/coding-agent/test/model-registry-runtime-provider.test.ts index 467cc0ec8..977e8a665 100644 --- a/packages/coding-agent/test/model-registry-runtime-provider.test.ts +++ b/packages/coding-agent/test/model-registry-runtime-provider.test.ts @@ -261,6 +261,52 @@ describe("ModelRegistry runtime provider registration", () => { }); }); + test("configured discovery suppresses extension fetchDynamicModels for the same provider", async () => { + const providerName = "runtime-configured-provider"; + fs.writeFileSync( + modelsJsonPath, + JSON.stringify({ + providers: { + [providerName]: { + baseUrl: "http://127.0.0.1:4893", + api: "openai-completions", + auth: "none", + discovery: { type: "openai-models-list" }, + }, + }, + }), + ); + const configuredFetch: FetchImpl = async input => { + const url = String(input); + if (url === "http://127.0.0.1:4893/v1/models") { + return Response.json({ + data: [{ id: "shared-runtime-model", context_length: 32_768 }], + }); + } + throw new Error(`Unexpected URL: ${url}`); + }; + const configuredRegistry = new ModelRegistry(authStorage, modelsJsonPath, { fetch: configuredFetch }); + let runtimeFetchCalls = 0; + configuredRegistry.registerProvider( + providerName, + { + baseUrl: "https://runtime.example.com/v1", + apiKey: "RUNTIME_KEY", + api: "openai-completions", + fetchDynamicModels: async () => { + runtimeFetchCalls++; + return [{ ...baseModel, id: "shared-runtime-model", contextWindow: 999_999 }]; + }, + }, + "ext://runtime", + ); + + await configuredRegistry.refreshProvider(providerName, "online"); + + expect(runtimeFetchCalls).toBe(0); + expect(configuredRegistry.find(providerName, "shared-runtime-model")?.contextWindow).toBe(32_768); + }); + test("refreshRuntimeProviders times out extension fetchDynamicModels that never resolves", async () => { vi.useFakeTimers(); const hangingFetch = Promise.withResolvers[number][]>(); From 6e209d3ecc5a5f707bd1e11ec284d570d6dbe946 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:18:22 +0200 Subject: [PATCH 20/59] fix(catalog): inferred image input for reference-less Cursor models Cursor GetUsableModels carries no per-model modality metadata; the reference-less fallback in normalizeCursorModel hardcoded input: ["text"], classifying multimodal families (claude/gpt/codex/gemini) as vision-blind, so attached images were silently replaced by text descriptions. Infer modalities from the model family instead, mirroring inferInputFromGeminiId in discovery/gemini.ts. Bundled references stay authoritative and text-only families (composer-*, grok-code-*) keep ["text"]. Fixes #4726 --- packages/catalog/src/discovery/cursor.ts | 24 ++++- .../catalog/test/cursor-discovery.test.ts | 96 +++++++++++++++++++ 2 files changed, 119 insertions(+), 1 deletion(-) create mode 100644 packages/catalog/test/cursor-discovery.test.ts diff --git a/packages/catalog/src/discovery/cursor.ts b/packages/catalog/src/discovery/cursor.ts index 880aefec5..65d6ca645 100644 --- a/packages/catalog/src/discovery/cursor.ts +++ b/packages/catalog/src/discovery/cursor.ts @@ -13,6 +13,13 @@ const CURSOR_GET_USABLE_MODELS_PATH = "/agent.v1.AgentService/GetUsableModels"; const DEFAULT_CONTEXT_WINDOW = 200_000; const DEFAULT_MAX_TOKENS = 64_000; +/** + * Model-id families whose native catalogs (anthropic, openai/openai-codex, + * google) are multimodal. Cursor-only or text-only families (`composer-*`, + * `grok-code-*`) intentionally stay outside this pattern. + */ +const CURSOR_MULTIMODAL_ID_PATTERN = /claude|gemini|gpt-|codex/; + const OptionalDisplayNameSchema = type("unknown").pipe(raw => (typeof raw === "string" ? raw : undefined)); const CursorAliasesSchema = type("unknown").pipe(raw => { if (Array.isArray(raw)) { @@ -292,7 +299,7 @@ function normalizeCursorModel( provider: "cursor", baseUrl: baseUrlOverride ?? CURSOR_DEFAULT_BASE_URL, reasoning, - input: ["text"], + input: inferInputFromCursorId(id), cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, contextWindow: DEFAULT_CONTEXT_WINDOW, maxTokens: DEFAULT_MAX_TOKENS, @@ -312,3 +319,18 @@ function pickModelDisplayName(model: CursorModelDetailsValue, fallbackId: string } return fallbackId; } + +/** + * Infers input modalities for Cursor models without a bundled reference. + * + * `GetUsableModels` carries no per-model modality metadata, so classification + * falls back to the model family: families that are multimodal in OMP's own + * native catalogs accept images, everything else stays text-only. Mirrors + * `inferInputFromGeminiId` in ./gemini.ts. + */ +function inferInputFromCursorId(id: string): ("text" | "image")[] { + if (CURSOR_MULTIMODAL_ID_PATTERN.test(id.toLowerCase())) { + return ["text", "image"]; + } + return ["text"]; +} diff --git a/packages/catalog/test/cursor-discovery.test.ts b/packages/catalog/test/cursor-discovery.test.ts new file mode 100644 index 000000000..f62b7c37c --- /dev/null +++ b/packages/catalog/test/cursor-discovery.test.ts @@ -0,0 +1,96 @@ +import { afterAll, beforeAll, describe, expect, it } from "bun:test"; +import * as http2 from "node:http2"; +import { create, toBinary } from "@bufbuild/protobuf"; +// Import from source, not the package specifier: the workspace `node_modules` +// copy resolves to the primary checkout, not this worktree. +import { fetchCursorUsableModels } from "../src/discovery/cursor"; +import { GetUsableModelsResponseSchema, ModelDetailsSchema } from "../src/discovery/cursor-gen/agent_pb"; +import type { ModelSpec } from "../src/types"; + +const FIXTURE_MODEL_IDS = [ + // Reference-less ids from families whose native catalogs are multimodal. + "claude-opus-4-8-99999999", + "gpt-5.5-codex-20991231", + "gemini-4-pro-exp", + // Reference-less ids from text-only families. + "composer-3", + "grok-code-fast-2", + // Bundled-reference ids: the reference stays authoritative. + "claude-4.5-opus-high", + "claude-4.6-opus-high", + "composer-1", +]; + +let server: http2.Http2Server; +let baseUrl: string; + +beforeAll(async () => { + const response = create(GetUsableModelsResponseSchema, { + models: FIXTURE_MODEL_IDS.map(modelId => create(ModelDetailsSchema, { modelId })), + }); + const payload = Buffer.from(toBinary(GetUsableModelsResponseSchema, response)); + + server = http2.createServer(); + server.on("stream", (stream: http2.ServerHttp2Stream, headers: http2.IncomingHttpHeaders) => { + stream.on("data", () => {}); + stream.on("end", () => { + if (headers[":path"] !== "/agent.v1.AgentService/GetUsableModels") { + stream.respond({ ":status": 404 }); + stream.end(); + return; + } + stream.respond({ ":status": 200, "content-type": "application/proto" }); + stream.end(payload); + }); + }); + await new Promise(resolve => server.listen(0, "127.0.0.1", resolve)); + const address = server.address(); + if (!address || typeof address === "string") { + throw new Error("expected http2 fixture server to bind a tcp port"); + } + baseUrl = `http://127.0.0.1:${address.port}`; +}); + +afterAll(() => { + server?.close(); +}); + +async function discover(): Promise>> { + const models = await fetchCursorUsableModels({ apiKey: "test-key", baseUrl }); + expect(models).not.toBeNull(); + return new Map((models ?? []).map(model => [model.id, model])); +} + +describe("cursor discovery input modalities (issue #4726)", () => { + it("classifies reference-less multimodal-family models as text+image", async () => { + const byId = await discover(); + expect(byId.get("claude-opus-4-8-99999999")?.input).toEqual(["text", "image"]); + expect(byId.get("gpt-5.5-codex-20991231")?.input).toEqual(["text", "image"]); + expect(byId.get("gemini-4-pro-exp")?.input).toEqual(["text", "image"]); + }); + + it("keeps reference-less text-only families text-only", async () => { + const byId = await discover(); + expect(byId.get("composer-3")?.input).toEqual(["text"]); + expect(byId.get("grok-code-fast-2")?.input).toEqual(["text"]); + }); + + it("keeps bundled references authoritative for input modalities", async () => { + const byId = await discover(); + // Bundled cursor references carry their own input classification; the + // id-based inference must not override it in either direction. + expect(byId.get("claude-4.5-opus-high")?.input).toEqual(["text", "image"]); + expect(byId.get("claude-4.6-opus-high")?.input).toEqual(["text"]); + expect(byId.get("composer-1")?.input).toEqual(["text"]); + }); + + it("preserves fallback defaults for reference-less models", async () => { + const byId = await discover(); + const spec = byId.get("claude-opus-4-8-99999999"); + expect(spec?.provider).toBe("cursor"); + expect(spec?.api).toBe("cursor-agent"); + expect(spec?.contextWindow).toBe(200_000); + expect(spec?.maxTokens).toBe(64_000); + expect(spec?.cost).toEqual({ input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }); + }); +}); From ca68daa81c2a2fb1dda7d09be1a339a2166f70f4 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:25:00 +0200 Subject: [PATCH 21/59] fix(mnemopi): made recall fact ids resolvable via memory reads recall (includeFacts) surfaces facts.fact_id as a result id, but store.get only searched working_memory + episodic_memory, so every surfaced fact id was a dead end for 'read memory://' and memory_edit ('not found in any scoped bank'). - store.get now falls back to the facts table (visibility mirrors factRecall: same-session or scope='global'), returning a read-only row with memory_store 'fact' and the full triple as content. - coding-agent labels the store honestly ('fact') in memory:// reads and reports not_editable (instead of not_found) for memory_edit ops on fact ids; the facts table stays immutable. Fixes #4725 --- packages/coding-agent/src/mnemopi/state.ts | 24 +++++-- .../src/prompts/tools/memory-edit.md | 2 + .../coding-agent/src/tools/memory-edit.ts | 4 +- .../internal-urls/memory-protocol.test.ts | 45 ++++++++++++ packages/mnemopi/src/core/beam/store.ts | 41 ++++++++++- packages/mnemopi/test/beam-store.test.ts | 68 +++++++++++++++++++ 6 files changed, 177 insertions(+), 7 deletions(-) diff --git a/packages/coding-agent/src/mnemopi/state.ts b/packages/coding-agent/src/mnemopi/state.ts index e4c403380..4034592f0 100644 --- a/packages/coding-agent/src/mnemopi/state.ts +++ b/packages/coding-agent/src/mnemopi/state.ts @@ -108,11 +108,16 @@ export interface MnemopiMemoryEditOptions { } export interface MnemopiMemoryEditResult { - status: "updated" | "deleted" | "invalidated" | "not_found"; + status: "updated" | "deleted" | "invalidated" | "not_found" | "not_editable"; bank?: string; - store?: "working" | "episodic"; + store?: MnemopiMemoryStore; } +/** Which mnemopi table a resolved memory id lives in. `fact` rows are + * read-only projections of fact extraction (issue #4725): resolvable for + * reads, never editable. */ +export type MnemopiMemoryStore = "working" | "episodic" | "fact"; + interface MnemopiStoredMemoryRow { id?: unknown; content?: unknown; @@ -136,7 +141,7 @@ interface MnemopiStoredMemoryRow { */ export interface MnemopiScopedMemoryHit { bank: string; - store: "working" | "episodic"; + store: MnemopiMemoryStore; row: { id: string; content: string; @@ -256,7 +261,8 @@ export class MnemopiSessionState { for (const target of targets) { const raw = target.memory.get(id) as MnemopiStoredMemoryRow | null; if (!raw) continue; - const store: MnemopiScopedMemoryHit["store"] = raw.memory_store === "episodic" ? "episodic" : "working"; + const store: MnemopiMemoryStore = + raw.memory_store === "episodic" || raw.memory_store === "fact" ? raw.memory_store : "working"; return { bank: target.bank, store, @@ -291,8 +297,16 @@ export class MnemopiSessionState { for (const target of targets) { const row = target.memory.get(id) as MnemopiStoredMemoryRow | null; if (!row) continue; - const store: MnemopiMemoryEditResult["store"] = row.memory_store === "episodic" ? "episodic" : "working"; + const store: MnemopiMemoryStore = + row.memory_store === "episodic" || row.memory_store === "fact" ? row.memory_store : "working"; const resultContext: Pick = { bank: target.bank, store }; + if (store === "fact") { + // Facts are read-only: no memory_edit op mutates the facts + // table, so report that precisely instead of `not_found` + // (the id DID resolve — issue #4725). + ineligible ??= { status: "not_editable", ...resultContext }; + continue; + } if ((op === "update" || op === "forget") && store !== "working") { ineligible ??= { status: "not_found", ...resultContext }; continue; diff --git a/packages/coding-agent/src/prompts/tools/memory-edit.md b/packages/coding-agent/src/prompts/tools/memory-edit.md index 641c482c2..0cb3314ff 100644 --- a/packages/coding-agent/src/prompts/tools/memory-edit.md +++ b/packages/coding-agent/src/prompts/tools/memory-edit.md @@ -5,6 +5,8 @@ Use only with ids returned by the `recall` tool. Operations: - `forget`: permanently delete a working memory. - `invalidate`: softly supersede a working or episodic memory, optionally pointing at `replacement_id`. +Fact ids (recall results marked `[facts]`) are read-only: inspect them with `read memory://`; every edit op on a fact id returns `not_editable`. + Prefer `invalidate` when a memory became stale but its history may still be useful. Use `forget` only for content that should be hard-deleted. **Always read the full memory before `update`.** Recall results are clipped previews (the trailing `…` marks a truncation and `full_length` reports the original size); `update` replaces content wholesale, so overwriting the preview would delete the unseen tail. Fetch the row first with `read memory://`, then pass the merged content in `content`. diff --git a/packages/coding-agent/src/tools/memory-edit.ts b/packages/coding-agent/src/tools/memory-edit.ts index 45b3ab020..1ac505355 100644 --- a/packages/coding-agent/src/tools/memory-edit.ts +++ b/packages/coding-agent/src/tools/memory-edit.ts @@ -50,7 +50,9 @@ export class MemoryEditTool implements AgentTool { const text = result.status === "not_found" ? `Memory ${params.id} was not found${location}.` - : `Memory ${params.id} ${result.status}${location}.`; + : result.status === "not_editable" + ? `Memory ${params.id} is a read-only fact${location}; it cannot be edited. Read it with memory://${params.id}.` + : `Memory ${params.id} ${result.status}${location}.`; return { content: [{ type: "text", text }], details: result, diff --git a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts index b88eee52d..331ad66ce 100644 --- a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts +++ b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts @@ -238,6 +238,51 @@ describe("MemoryProtocolHandler — mnemopi bridge (issue #4443)", () => { }); }); + it("resolves memory:// to a read-only fact row (issue #4725)", async () => { + await withMnemopiSession(async ({ state }) => { + const beam = state.memory.beam; + beam.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, timestamp, confidence) VALUES (?, ?, ?, ?, ?, ?, ?)", + ) + .run("0473bbdb8da6df92", beam.sessionId, "Glab", "works-without", "mise prefix", "2026-07-01T00:00:00.000Z", 0.9); + + const router = InternalUrlRouter.instance(); + const resource = await router.resolve("memory://0473bbdb8da6df92"); + + expect(resource.content).toContain("id: 0473bbdb8da6df92"); + expect(resource.content).toContain("store: fact"); + expect(resource.content).toContain("Glab works-without mise prefix"); + }); + }); + + it("reports not_editable (not not_found) for memory_edit ops on a fact id (issue #4725)", async () => { + await withMnemopiSession(async ({ state }) => { + const beam = state.memory.beam; + beam.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, timestamp, confidence) VALUES (?, ?, ?, ?, ?, ?, ?)", + ) + .run("fact-readonly", beam.sessionId, "service", "uses", "postgres", "2026-07-01T00:00:00.000Z", 0.9); + + expect(state.editScopedMemory("update", "fact-readonly", { content: "x" })).toMatchObject({ + status: "not_editable", + store: "fact", + }); + expect(state.editScopedMemory("forget", "fact-readonly")).toMatchObject({ + status: "not_editable", + store: "fact", + }); + expect(state.editScopedMemory("invalidate", "fact-readonly")).toMatchObject({ + status: "not_editable", + store: "fact", + }); + + // The fact row itself is untouched by the rejected edits. + expect(beam.db.prepare("SELECT fact_id FROM facts WHERE fact_id = ?").get("fact-readonly")).not.toBeNull(); + }); + }); + it("routes memory://root to the file-backed summary even when mnemopi is active", async () => { await withMnemopiSession(async () => { const router = InternalUrlRouter.instance(); diff --git a/packages/mnemopi/src/core/beam/store.ts b/packages/mnemopi/src/core/beam/store.ts index 5d0d2f9dd..225452fc5 100644 --- a/packages/mnemopi/src/core/beam/store.ts +++ b/packages/mnemopi/src/core/beam/store.ts @@ -665,7 +665,46 @@ export function get(beam: BeamMemoryState, memoryId: string): Row | null { WHERE id = ? AND (session_id = ? OR scope = 'global') `) .get(memoryId, beam.sessionId) as Row | null | undefined; - return episodic == null ? null : { ...episodic, metadata: episodic.metadata_json, memory_store: "episodic" }; + if (episodic != null) return { ...episodic, metadata: episodic.metadata_json, memory_store: "episodic" }; + + return getFact(beam, memoryId); +} + +/** + * Read-only resolution for ids minted from the `facts` table. `recall` + * surfaces `facts.fact_id` as a result id (`factRecall`), so `get` must + * resolve those ids too — otherwise every surfaced fact id is a dead end + * for the read path (issue #4725). Visibility mirrors `factRecall`: + * same-session facts plus explicitly global ones (`scope` is an optional + * column on `facts`; `SELECT *` tolerates banks without it, in which case + * only same-session facts resolve). The row is shaped like the + * working/episodic hits with the full triple as content; + * `memory_store: "fact"` marks it read-only — no update/forget/invalidate + * path mutates `facts`. + */ +function getFact(beam: BeamMemoryState, memoryId: string): Row | null { + const fact = beam.db.prepare("SELECT * FROM facts WHERE fact_id = ?").get(memoryId) as Row | null | undefined; + if (fact == null) return null; + if (fact.session_id !== beam.sessionId && fact.scope !== "global") return null; + const subject = typeof fact.subject === "string" ? fact.subject : ""; + const predicate = typeof fact.predicate === "string" ? fact.predicate : ""; + const object = typeof fact.object === "string" ? fact.object : ""; + return { + id: fact.fact_id, + content: [subject, predicate, object].filter(part => part.length > 0).join(" "), + source: "facts", + timestamp: fact.timestamp ?? null, + session_id: fact.session_id ?? null, + importance: fact.confidence ?? null, + metadata: JSON.stringify({ + subject, + predicate, + object, + source_msg_id: fact.source_msg_id ?? null, + }), + created_at: fact.created_at ?? null, + memory_store: "fact", + }; } export function forgetWorking(beam: BeamMemoryState, memoryId: string): boolean { diff --git a/packages/mnemopi/test/beam-store.test.ts b/packages/mnemopi/test/beam-store.test.ts index f9af3a8b5..fb09c7a62 100644 --- a/packages/mnemopi/test/beam-store.test.ts +++ b/packages/mnemopi/test/beam-store.test.ts @@ -1,4 +1,5 @@ import { afterEach, describe, expect, it } from "bun:test"; +import { recallEnhanced } from "@oh-my-pi/pi-mnemopi/core/beam/recall"; import { initBeam } from "@oh-my-pi/pi-mnemopi/core/beam/schema"; import { exportToDict, @@ -217,3 +218,70 @@ describe("beam store free functions", () => { expect(scratchpadRead(dest).map(row => row.content)).toEqual([]); }); }); + +describe("fact-id read path (issue #4725)", () => { + function insertFact( + beam: BeamMemoryState, + factId: string, + sessionId: string, + subject: string, + predicate: string, + object: string, + confidence = 0.9, + ): void { + beam.db + .prepare( + "INSERT INTO facts (fact_id, session_id, subject, predicate, object, timestamp, confidence) VALUES (?, ?, ?, ?, ?, ?, ?)", + ) + .run(factId, sessionId, subject, predicate, object, "2026-05-30T00:00:00.000Z", confidence); + } + + it("resolves an id surfaced by fact recall to a read-only fact row", async () => { + const beam = makeState(); + insertFact(beam, "fact-postgres", beam.sessionId, "service", "uses", "postgres database", 0.91); + + const results = await recallEnhanced(beam, "postgres", 5, { includeFacts: true }); + const surfaced = results.find(result => result.source === "facts"); + expect(surfaced?.id).toBe("fact-postgres"); + + // memory:// reads and memory_edit both resolve ids via get(); a + // surfaced fact id must not be a dead end. + const row = get(beam, "fact-postgres"); + expect(row).toMatchObject({ + id: "fact-postgres", + content: "service uses postgres database", + source: "facts", + importance: 0.91, + session_id: beam.sessionId, + memory_store: "fact", + }); + expect(JSON.parse(String(row?.metadata))).toMatchObject({ + subject: "service", + predicate: "uses", + object: "postgres database", + }); + }); + + it("keeps fact reads session-scoped like fact recall, honoring explicit global scope", () => { + const beam = makeState(); + insertFact(beam, "fact-other", "session-other", "service", "uses", "postgres database"); + expect(get(beam, "fact-other")).toBeNull(); + + beam.db.run("ALTER TABLE facts ADD COLUMN scope TEXT DEFAULT 'session'"); + beam.db.run("UPDATE facts SET scope = 'global' WHERE fact_id = 'fact-other'"); + expect(get(beam, "fact-other")?.memory_store).toBe("fact"); + }); + + it("keeps working rows first on id collision and never deletes facts via forgetWorking", () => { + const beam = makeState(); + insertFact(beam, "shared-id", beam.sessionId, "service", "uses", "postgres database"); + const workingId = remember(beam, "working row shadowing a fact id"); + beam.db.prepare("UPDATE working_memory SET id = ? WHERE id = ?").run("shared-id", workingId); + + expect(get(beam, "shared-id")?.memory_store).toBe("working"); + + expect(forgetWorking(beam, "fact-missing")).toBe(false); + expect(forgetWorking(beam, "shared-id")).toBe(true); + expect(get(beam, "shared-id")?.memory_store).toBe("fact"); + }); +}); From 894cf489ff4dc5f45f0e5797efe59655346e3e57 Mon Sep 17 00:00:00 2001 From: roboomp Date: Thu, 9 Jul 2026 18:17:20 +0200 Subject: [PATCH 22/59] fix(tui): canceled streaming prompts on first escape MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Esc during an active streaming turn required a second press within 2s (two-step arm from #3493). In the no-input-waiter submit path the turn starts with isStreaming=true but no working loader, so Esc fell into the two-step branch and the agent_start subscription then wiped the arm — repeated presses kept re-arming and never aborted. The loader-up path already aborted on a single press, so the confirmation guarded no coherent state. First Esc now aborts the streaming turn directly. Adopted from PR #4938 (test + input-controller + changelog hunks only; unrelated workflow-notice.md churn dropped). Fixes #4921 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/controllers/input-controller.ts | 60 +------- .../test/input-controller-escape.test.ts | 128 +++++------------- 3 files changed, 38 insertions(+), 151 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index f8cc87c48..19987cf2e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -11,6 +11,7 @@ - Fixed pi extensions calling `ctx.ui.addAutocompleteProvider(...)` crashing at load with `TypeError: ... is not a function` — a failure that, for extensions guarding init in one `try/catch` (e.g. `@ff-labs/pi-fff`), aborted the extension's entire initialization. `ExtensionUIContext` now implements pi's autocomplete-provider API: interactive mode stacks each registered factory on top of the built-in editor provider (re-applied on every slash-command refresh, with throwing or malformed factories skipped), while RPC/ACP/headless contexts accept the factory as a no-op. ([#4919](https://github.com/can1357/oh-my-pi/issues/4919)) - Fixed bundled reviewer and plan subagents to inherit their model roles' explicit thinking effort suffixes instead of pinning `high` ([#4761](https://github.com/can1357/oh-my-pi/issues/4761)). - Built-in provider model discovery now refreshes an expired stored OAuth credential before an online refresh needs it, instead of silently skipping the provider. The refresh is scoped to the providers actually being discovered (`refreshProvider` cannot rotate unrelated credentials), fires under `online-if-uncached` only when the model manager will actually fetch, and offline discovery stays peek-only ([#4893](https://github.com/can1357/oh-my-pi/issues/4893)). +- Fixed Escape during an active TUI prompt requiring a second press before canceling; the first Escape now aborts the streaming turn immediately. ([#4921](https://github.com/can1357/oh-my-pi/issues/4921)) ## [16.3.12] - 2026-07-08 diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index aa70e59ca..bb19390b4 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -154,7 +154,6 @@ const TINY_TITLE_PROGRESS_REVEAL_DELAY_MS = 1_000; // deliberate human double-tap is always tens of milliseconds apart. const LEFT_DOUBLE_TAP_MIN_GAP_MS = 40; const LEFT_DOUBLE_TAP_MAX_GAP_MS = 500; -const STREAMING_ESCAPE_CANCEL_WINDOW_MS = 2_000; export class InputController { constructor( @@ -179,16 +178,6 @@ export class InputController { // (>= LEFT_DOUBLE_TAP_MAX_GAP_MS) starts a fresh sequence. See // #detectLeftDoubleTap. #leftTapCount = 0; - // Streaming turns use a two-step Esc: first press arms this token, second press - // within the window aborts the same live assistant turn. The token is a per-turn - // sentinel minted lazily on demand and reset on every `agent_start`/`agent_end` - // (see setupKeyHandlers), so it survives `message_start`/`message_update` - // transitions inside a single turn but cannot leak across turn boundaries. - #streamingEscapeTurnSentinel: object | undefined; - #streamingEscapeArmedToken: object | undefined; - #streamingEscapeArmedUntil = 0; - #streamingEscapeTimer: NodeJS.Timeout | undefined; - #streamingEscapeSessionSubscribed = false; // Sequential index for `local://attachment-N` references created by large-paste and // pasted-file attachments. Seeded from 0 and bumped past existing attachment files. #attachmentCounter = 0; @@ -238,50 +227,12 @@ export class InputController { const unsubscribe = tinyTitleClient.onProgress(update); } - #clearStreamingEscapeArm(): void { - this.#streamingEscapeArmedToken = undefined; - this.#streamingEscapeArmedUntil = 0; - if (this.#streamingEscapeTimer) { - clearTimeout(this.#streamingEscapeTimer); - this.#streamingEscapeTimer = undefined; - } - } - - #handleStreamingEscape(): void { - if (!this.#streamingEscapeTurnSentinel) { - this.#streamingEscapeTurnSentinel = {}; - } - const token = this.#streamingEscapeTurnSentinel; - const now = Date.now(); - if (this.#streamingEscapeArmedToken === token && now <= this.#streamingEscapeArmedUntil) { - this.#clearStreamingEscapeArm(); - void this.ctx.session.abort({ reason: USER_INTERRUPT_LABEL }); - return; - } - - this.#clearStreamingEscapeArm(); - this.#streamingEscapeArmedToken = token; - this.#streamingEscapeArmedUntil = now + STREAMING_ESCAPE_CANCEL_WINDOW_MS; - this.#streamingEscapeTimer = setTimeout(() => { - if (this.#streamingEscapeArmedToken === token && Date.now() >= this.#streamingEscapeArmedUntil) { - this.#clearStreamingEscapeArm(); - } - }, STREAMING_ESCAPE_CANCEL_WINDOW_MS); - this.#streamingEscapeTimer.unref?.(); - this.ctx.showStatus("Press Esc again within 2s to cancel streaming."); + #abortStreamingTurn(): void { + void this.ctx.session.abort({ reason: USER_INTERRUPT_LABEL }); } setupKeyHandlers(): void { this.ctx.editor.setActionKeys("app.interrupt", this.ctx.keybindings.getKeys("app.interrupt")); - if (!this.#streamingEscapeSessionSubscribed && typeof this.ctx.session.subscribe === "function") { - this.#streamingEscapeSessionSubscribed = true; - this.ctx.session.subscribe(event => { - if (event.type === "agent_start" || event.type === "agent_end") { - this.#streamingEscapeTurnSentinel = undefined; - this.#clearStreamingEscapeArm(); - } - }); - } if (!this.#focusedLeftTapListenerInstalled) { this.#focusedLeftTapListenerInstalled = true; this.ctx.ui.addInputListener(data => { @@ -351,7 +302,7 @@ export class InputController { if (this.ctx.loopModeEnabled) { this.ctx.pauseLoop(); if (this.ctx.session.isStreaming) { - this.#handleStreamingEscape(); + this.#abortStreamingTurn(); } else { this.ctx.cancelPendingSubmission(); } @@ -402,11 +353,10 @@ export class InputController { this.ctx.isPythonMode = false; this.ctx.updateEditorBorderColor(); } else if (this.ctx.session.isStreaming) { - this.#handleStreamingEscape(); + this.#abortStreamingTurn(); } else if (this.ctx.editor.getText().trim()) { - // Esc must not destroy an in-progress draft; it only disarms a previous empty-editor Esc. + // Esc must not destroy an in-progress draft. this.ctx.lastEscapeTime = 0; - this.#clearStreamingEscapeArm(); } else if (vocalizer.isSpeaking()) { // TTS buffers seconds of PCM past the streaming abort, so an Esc // arriving after the model stopped would otherwise fall through to diff --git a/packages/coding-agent/test/input-controller-escape.test.ts b/packages/coding-agent/test/input-controller-escape.test.ts index e7ae3af47..d2828106d 100644 --- a/packages/coding-agent/test/input-controller-escape.test.ts +++ b/packages/coding-agent/test/input-controller-escape.test.ts @@ -269,6 +269,16 @@ function abortViewSession(ctx: InteractiveModeContext): AbortViewSession { // so property access is explicit. return ctx.viewSession as unknown as AbortViewSession; } + +type MutableSessionState = InteractiveModeContext["session"] & { + isStreaming: boolean; +}; + +function mutableSessionState(ctx: InteractiveModeContext): MutableSessionState { + // Test harness installs a mutable fake AgentSession; keep the unchecked cast named + // so state mutations are explicit. + return ctx.session as MutableSessionState; +} beforeEach(async () => { await Settings.init({ inMemory: true }); }); @@ -438,111 +448,37 @@ describe("InputController escape behavior", () => { expect(spies.abort).not.toHaveBeenCalled(); }); - it("requires a second Esc within two seconds to abort streaming", () => { - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); + it("aborts an active streaming turn on the first Esc without asking for confirmation", () => { const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; + mutableSessionState(ctx).isStreaming = true; const controller = new InputController(ctx); controller.setupKeyHandlers(); editor.onEscape?.(); + expect(spies.abort).toHaveBeenCalledTimes(1); + expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); + expect(spies.showStatus).not.toHaveBeenCalledWith("Press Esc again within 2s to cancel streaming."); + }); + + it("aborts the submitted turn on the first Esc once the main session starts streaming", async () => { + const { ctx, editor, spies } = createContext(); + const submission = createSubmission({ text: "fix issue #4921" }); + spies.startPendingSubmission.mockReturnValue(submission); + const controller = new InputController(ctx); + + controller.setupKeyHandlers(); + controller.setupEditorSubmitHandler(); + await editor.onSubmit?.("fix issue #4921"); + mutableSessionState(ctx).isStreaming = true; + ctx.loadingAnimation = undefined; + + editor.onEscape?.(); + expect(spies.cancelPendingSubmission).not.toHaveBeenCalled(); - expect(spies.clearQueue).not.toHaveBeenCalled(); - expect(spies.abort).not.toHaveBeenCalled(); - expect(spies.showStatus).toHaveBeenCalledWith("Press Esc again within 2s to cancel streaming."); - - now.mockReturnValue(2_500); - editor.onEscape?.(); - expect(spies.abort).toHaveBeenCalledTimes(1); expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); - }); - - it("expires the streaming Esc arm instead of aborting on a late second press", () => { - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - editor.onEscape?.(); - now.mockReturnValue(3_001); - editor.onEscape?.(); - - expect(spies.abort).not.toHaveBeenCalled(); - expect(spies.showStatus).toHaveBeenCalledTimes(2); - }); - - it("preserves the streaming Esc arm when streamingComponent appears between presses", () => { - // Pre-`message_start`: first Esc arms on the per-turn sentinel. `message_start` - // then publishes `ctx.streamingComponent`; the second Esc must still abort the - // same live turn instead of re-arming on the new component reference. - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - editor.onEscape?.(); - (ctx as unknown as { streamingComponent: object }).streamingComponent = {}; - now.mockReturnValue(1_500); - editor.onEscape?.(); - - expect(spies.abort).toHaveBeenCalledTimes(1); - expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); - }); - - it("aborts on the second Esc even when ctx.streamingMessage was replaced by a delta in between", () => { - // `EventController` replaces `ctx.streamingMessage` with a fresh immutable - // snapshot on every `message_update`; the per-turn sentinel is unaffected so - // swapping the message must not invalidate the armed token. - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - (ctx as unknown as { streamingComponent: object }).streamingComponent = {}; - (ctx as unknown as { streamingMessage: object }).streamingMessage = { content: [] }; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - editor.onEscape?.(); - (ctx as unknown as { streamingMessage: object }).streamingMessage = { content: ["delta"] }; - now.mockReturnValue(1_500); - editor.onEscape?.(); - - expect(spies.abort).toHaveBeenCalledTimes(1); - expect(spies.abort).toHaveBeenCalledWith({ reason: USER_INTERRUPT_LABEL }); - }); - - it("clears the streaming Esc arm when the current turn ends", () => { - const now = vi.spyOn(Date, "now"); - now.mockReturnValue(1_000); - const { ctx, editor, spies, sessionListeners } = createContext(); - (ctx.session as { isStreaming: boolean }).isStreaming = true; - const controller = new InputController(ctx); - - controller.setupKeyHandlers(); - // Fallback arm (no streamingMessage/streamingComponent yet — pre-message_start). - editor.onEscape?.(); - expect(sessionListeners).toHaveLength(1); - - // Turn 1 ends; a new turn starts. session.subscribe receives both transitions, - // either of which must invalidate the still-armed fallback token so it cannot - // fast-abort the new turn's first Esc. - for (const listener of sessionListeners) { - listener({ type: "agent_end" }); - listener({ type: "agent_start" }); - } - - now.mockReturnValue(1_500); - editor.onEscape?.(); - - expect(spies.abort).not.toHaveBeenCalled(); - expect(spies.showStatus).toHaveBeenCalledTimes(2); + expect(spies.showStatus).not.toHaveBeenCalledWith("Press Esc again within 2s to cancel streaming."); }); it("returns focused subagent view to main on Esc instead of aborting", () => { From cde9ee750107fc581304bdc31cc43262fa1e504e Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:25:37 +0200 Subject: [PATCH 23/59] fix(tui): repainted write first partial result over pending tail preview The first-result viewport-repaint gate assumed only streamed __partialJson placeholder shapes (SSH) could re-anchor; the write renderer's collapsed pending preview paints a tail window from decoded content, so its first partial result re-anchored to the top of the file and left the committed tail rows stale above the new frame. Resolve forceFirstResultViewportRepaint per renderer as a boolean or an (args, options) predicate evaluated at paint time: write opts in when a collapsed preview outgrew the streaming tail window, SSH stays scoped to the streamed-placeholder shape it always covered. Adopted from PR #4478 (roboomp) with an allocation-free line-count scan and terminal-buffer regression coverage. Fixes #4477 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/components/tool-execution.ts | 51 ++++--- packages/coding-agent/src/tools/renderers.ts | 18 ++- packages/coding-agent/src/tools/ssh.ts | 13 +- packages/coding-agent/src/tools/write.ts | 26 ++++ .../test/tool-execution-write-repaint.test.ts | 140 ++++++++++++++++++ 6 files changed, 217 insertions(+), 32 deletions(-) create mode 100644 packages/coding-agent/test/tool-execution-write-repaint.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 19987cf2e..ce7239af2 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -12,6 +12,7 @@ - Fixed bundled reviewer and plan subagents to inherit their model roles' explicit thinking effort suffixes instead of pinning `high` ([#4761](https://github.com/can1357/oh-my-pi/issues/4761)). - Built-in provider model discovery now refreshes an expired stored OAuth credential before an online refresh needs it, instead of silently skipping the provider. The refresh is scoped to the providers actually being discovered (`refreshProvider` cannot rotate unrelated credentials), fires under `online-if-uncached` only when the model manager will actually fetch, and offline discovery stays peek-only ([#4893](https://github.com/can1357/oh-my-pi/issues/4893)). - Fixed Escape during an active TUI prompt requiring a second press before canceling; the first Escape now aborts the streaming turn immediately. ([#4921](https://github.com/can1357/oh-my-pi/issues/4921)) +- Fixed the streamed `write` tool's collapsed pending tail preview leaving stale rows above the first partial-result frame in the TUI; the first result now replays the viewport like the SSH placeholder seam already did ([#4477](https://github.com/can1357/oh-my-pi/issues/4477)) ## [16.3.12] - 2026-07-08 diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index 4b86a68a6..a958a233c 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -38,7 +38,7 @@ import { resolveImageOptions, truncateToWidth, } from "../../tools/render-utils"; -import { toolRenderers } from "../../tools/renderers"; +import { type FirstResultViewportRepaint, toolRenderers } from "../../tools/renderers"; import { TODO_STRIKE_TOTAL_FRAMES, type TodoToolDetails } from "../../tools/todo"; import { isFramedBlockComponent, renderStatusLine, WidthAwareText } from "../../tui"; import { sanitizeWithOptionalSixelPassthrough } from "../../utils/sixel"; @@ -283,13 +283,13 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac // history, so progress renders static gray and further partial snapshots are // dropped (see #maybeFreezeBackgroundTask). #backgroundTaskFrozen = false; - // Set on each `render()` when the last painted shape carried the streamed - // SSH-style placeholder / partial-result chrome. Reset gates key off these - // so a topology-changing update that lands before the shape reaches the - // terminal never triggers a full-viewport replay (which on direct terminals - // wipes native scrollback and flashes the user's history — reviewer note on - // PR #4315). - #placeholderShapePainted = false; + // Set on each `render()` when the last painted pending shape must be + // replayed wholesale when the first result arrives. Reset gates key off + // these so a topology-changing update that lands before the shape reaches + // the terminal never triggers a full-viewport replay (which on direct + // terminals wipes native scrollback and flashes the user's history — + // reviewer note on PR #4315). + #firstResultViewportRepaintShapePainted = false; #partialResultShapePainted = false; #renderState: { spinnerFrame?: number; @@ -497,9 +497,9 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac } const hadNoResult = this.#result === undefined; const wasPartialResult = this.#result !== undefined && this.#isPartial; - const placeholderPainted = this.#placeholderShapePainted; + const firstResultRepaintShapePainted = this.#firstResultViewportRepaintShapePainted; const partialResultPainted = this.#partialResultShapePainted; - this.#placeholderShapePainted = false; + this.#firstResultViewportRepaintShapePainted = false; this.#partialResultShapePainted = false; this.#result = result; this.#resultVersion++; @@ -513,7 +513,7 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac this.#updateTodoStrikeAnimation(); this.#updateDisplay(); this.#resetDisplayForResultTopologyChange( - hadNoResult && placeholderPainted, + hadNoResult && firstResultRepaintShapePainted, wasPartialResult && partialResultPainted, isPartial, ); @@ -810,34 +810,37 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac this.#displayBuilt = true; } - #rendererFlag(name: "forceFirstResultViewportRepaint" | "forceResultViewportRepaintOnSettle"): boolean { + #rendererFlag(name: "forceResultViewportRepaintOnSettle"): boolean { const toolValue = (this.#tool as Record | undefined)?.[name]; const rendererValue = toolRenderers[this.#toolName]?.[name]; return toolValue === true || (toolValue === undefined && rendererValue === true); } /** - * True while the last painted shape uses the streamed placeholder path - * (`⏳ SSH: […]` / `$ …`) — the render call ran with `__partialJson` args - * and no result. Kept as a per-paint fact so a topology-changing update - * that lands before the placeholder reaches the terminal skips the reset. + * True while the last painted pending-call shape opted into a full viewport + * repaint at the first result (`forceFirstResultViewportRepaint`) — e.g. the + * streamed SSH placeholder (`⏳ SSH: […]` / `$ …`) or a collapsed write tail + * window, both of which the first result render re-anchors instead of + * preserving. Kept as a per-paint fact so a topology-changing update that + * lands before the pending rows reach the terminal skips the reset. */ - #isPlaceholderShapeAtRender(): boolean { + #needsFirstResultViewportRepaintAtRender(): boolean { if (this.#result !== undefined) return false; - if (!this.#rendererFlag("forceFirstResultViewportRepaint")) return false; - return partialJsonOf(this.#args) !== undefined; + const toolValue = (this.#tool as { forceFirstResultViewportRepaint?: FirstResultViewportRepaint } | undefined) + ?.forceFirstResultViewportRepaint; + const value = toolValue !== undefined ? toolValue : toolRenderers[this.#toolName]?.forceFirstResultViewportRepaint; + if (typeof value === "function") return value(this.#args, this.#renderState); + return value === true; } #resetDisplayForResultTopologyChange( - firstResultAfterPlaceholderPaint: boolean, + firstResultAfterRepaintShapePaint: boolean, partialResultPaintedBeforeSettle: boolean, isPartial: boolean, ): void { - const firstResultReplacesStreamedPlaceholder = - firstResultAfterPlaceholderPaint && this.#rendererFlag("forceFirstResultViewportRepaint"); const provisionalResultSettled = partialResultPaintedBeforeSettle && !isPartial && this.#rendererFlag("forceResultViewportRepaintOnSettle"); - if (firstResultReplacesStreamedPlaceholder || provisionalResultSettled) { + if (firstResultAfterRepaintShapePaint || provisionalResultSettled) { this.#ui.resetDisplay(); } } @@ -848,7 +851,7 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac // override runs on every compose the parent Container performs, so a // frame that never gets composed leaves the flags false and prevents a // spurious `resetDisplay()`. - this.#placeholderShapePainted = this.#isPlaceholderShapeAtRender(); + this.#firstResultViewportRepaintShapePainted = this.#needsFirstResultViewportRepaintAtRender(); this.#partialResultShapePainted = this.#result !== undefined && this.#isPartial; return lines; } diff --git a/packages/coding-agent/src/tools/renderers.ts b/packages/coding-agent/src/tools/renderers.ts index bc5043f33..3e6baf824 100644 --- a/packages/coding-agent/src/tools/renderers.ts +++ b/packages/coding-agent/src/tools/renderers.ts @@ -32,6 +32,15 @@ import { sshToolRenderer } from "./ssh"; import { todoToolRenderer } from "./todo"; import { writeToolRenderer } from "./write"; +/** + * Per-renderer opt-in for a full viewport replay when the first result + * replaces a painted pending-call render. A predicate receives the painted + * call args and render options so the repaint stays scoped to the pending + * shapes that actually re-anchor (an over-eager replay wipes native + * scrollback on direct terminals). + */ +export type FirstResultViewportRepaint = boolean | ((args: unknown, options: RenderResultOptions) => boolean); + export type ToolRenderer = { renderCall: (args: unknown, options: RenderResultOptions, theme: Theme) => Component; renderResult: ( @@ -55,12 +64,11 @@ export type ToolRenderer = { */ animatedPartialResult?: boolean | ((args: unknown) => boolean); /** - * Whether replacing a streamed pending placeholder with the first result - * requires a full viewport repaint. Use for merged renderers whose pending - * streamed args may have committed placeholder rows that the result render - * re-anchors instead of preserving. + * Whether replacing a pending call render with the first result requires a + * full viewport repaint. Use for merged renderers whose pending rows can be + * re-anchored instead of preserved by the result render. */ - forceFirstResultViewportRepaint?: boolean; + forceFirstResultViewportRepaint?: FirstResultViewportRepaint; /** * Whether settling a provisional partial result into the final render requires * a full viewport repaint. Use when the result renderer changes chrome or diff --git a/packages/coding-agent/src/tools/ssh.ts b/packages/coding-agent/src/tools/ssh.ts index a59f0da71..5540d4a65 100644 --- a/packages/coding-agent/src/tools/ssh.ts +++ b/packages/coding-agent/src/tools/ssh.ts @@ -244,6 +244,13 @@ interface SshRenderArgs { timeout?: number; } +/** Whether the painted call args still carry the streamed raw-JSON buffer — + * the shape that renders the `⏳ SSH: […]` / `$ …` placeholder. */ +function hasStreamedRenderArgs(args: unknown): boolean { + if (args == null || typeof args !== "object" || !("__partialJson" in args)) return false; + return typeof args.__partialJson === "string"; +} + interface SshRenderContext { /** Visual lines for truncated output (pre-computed by tool-execution) */ visualLines?: string[]; @@ -388,9 +395,9 @@ export const sshToolRenderer = { mergeCallAndResult: true, // Streamed args can initially render the SSH placeholder (`⏳ SSH: […]` / // `$ …`), then the first partial result inserts the `Output` section and - // re-anchors the frame. Force a full repaint at that seam so placeholder rows - // do not survive in viewport/native scrollback. - forceFirstResultViewportRepaint: true, + // re-anchors the frame. Force a full repaint only at that streamed-placeholder + // seam so placeholder rows do not survive in viewport/native scrollback. + forceFirstResultViewportRepaint: hasStreamedRenderArgs, // The provisional pending-result frame settles into the final `⇄ SSH: [host]` // frame, so clear/replay the viewport at that topology flip too. forceResultViewportRepaintOnSettle: true, diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index 3ec53b838..a7356d166 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -985,6 +985,24 @@ function countLines(text: string): number { return text.split("\n").length; } +/** Bounded newline scan: whether `text` spans more than `maxLines` lines. + * Runs on every live compose (the repaint predicate below), so it must not + * materialize the split the way `countLines` does. */ +function exceedsLineCount(text: string, maxLines: number): boolean { + if (!text) return false; + let lines = 1; + for (let index = text.indexOf("\n"); index !== -1; index = text.indexOf("\n", index + 1)) { + if (++lines > maxLines) return true; + } + return false; +} + +function writeContentOf(args: unknown): string { + if (args == null || typeof args !== "object" || !("content" in args)) return ""; + const content = args.content; + return typeof content === "string" ? content : ""; +} + function formatLineCountSuffix(lineCount: number, uiTheme: Theme): string { if (lineCount <= 0) return ""; return uiTheme.fg("dim", ` · ${lineCount} line${lineCount === 1 ? "" : "s"}`); @@ -1218,4 +1236,12 @@ export const writeToolRenderer = { }); }, mergeCallAndResult: true, + // The collapsed pending preview follows the streaming edge with a tail + // window once the content outgrows it (`… (N earlier lines)` + last rows); + // the first partial result re-anchors the frame to the top of the file, so + // tail rows already committed to viewport/native scrollback would survive + // as stale content above the new frame without a full replay. Expanded and + // short previews stay top-anchored and skip the (scrollback-wiping) reset. + forceFirstResultViewportRepaint: (args: unknown, options: RenderResultOptions) => + !options.expanded && exceedsLineCount(writeContentOf(args), WRITE_STREAMING_PREVIEW_LINES), }; diff --git a/packages/coding-agent/test/tool-execution-write-repaint.test.ts b/packages/coding-agent/test/tool-execution-write-repaint.test.ts new file mode 100644 index 000000000..d011987e5 --- /dev/null +++ b/packages/coding-agent/test/tool-execution-write-repaint.test.ts @@ -0,0 +1,140 @@ +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; +import { ToolExecutionComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tool-execution"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { type Component, TUI } from "@oh-my-pi/pi-tui"; +import { StressRenderScheduler } from "../../tui/test/render-stress-scheduler"; +import { VirtualTerminal } from "../../tui/test/virtual-terminal"; + +function writeArgs(lineCount: number) { + return { + path: "notes.txt", + content: Array.from({ length: lineCount }, (_, i) => `line ${i + 1}`).join("\n"), + }; +} + +function partialWriteResult(text = "Writing notes.txt...") { + return { content: [{ type: "text", text }] }; +} + +class Footer implements Component { + constructor(readonly rows: number) {} + invalidate(): void {} + render(_width: number): string[] { + return Array.from({ length: this.rows }, (_, i) => `editor-${i}`); + } +} + +function plainBuffer(term: VirtualTerminal): string[] { + return term + .getScrollBuffer() + .map(row => Bun.stripANSI(row).trimEnd()) + .filter(Boolean); +} + +describe("ToolExecutionComponent write repaint seam", () => { + const components: ToolExecutionComponent[] = []; + + beforeAll(async () => { + await initTheme(); + }); + + afterEach(() => { + for (const component of components) component.stopAnimation(); + components.length = 0; + vi.restoreAllMocks(); + }); + + function makeComponent(args: unknown) { + const resetDisplay = vi.fn(); + const ui = { requestRender() {}, requestComponentRender() {}, resetDisplay } as unknown as TUI; + const component = new ToolExecutionComponent("write", args, {}, undefined, ui); + components.push(component); + resetDisplay.mockClear(); + return { component, resetDisplay }; + } + + it("forces a viewport repaint when a painted collapsed tail window receives its first result", () => { + // 20 lines > WRITE_STREAMING_PREVIEW_LINES (12): the pending preview is a + // tail window the first-result render re-anchors to the top of the file. + const { component, resetDisplay } = makeComponent(writeArgs(20)); + component.render(80); + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).toHaveBeenCalledTimes(1); + }); + + it("does not repaint when the pending tail window never reaches the terminal", () => { + const { component, resetDisplay } = makeComponent(writeArgs(20)); + // No render() before the result: a resetDisplay here would wipe native + // scrollback for a shape the user never saw. + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).not.toHaveBeenCalled(); + }); + + it("does not repaint a collapsed preview that fits the streaming window", () => { + // 12 lines render top-anchored without a tail window, so the first result + // does not re-anchor the frame; wiping scrollback would be gratuitous. + const { component, resetDisplay } = makeComponent(writeArgs(12)); + component.render(80); + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).not.toHaveBeenCalled(); + }); + + it("does not repaint an expanded pending preview", () => { + // Expanded previews show the whole file top-anchored — no tail window to + // re-anchor. + const { component, resetDisplay } = makeComponent(writeArgs(20)); + component.setExpanded(true); + component.render(80); + + component.updateResult(partialWriteResult(), true); + + expect(resetDisplay).not.toHaveBeenCalled(); + }); + + it("removes stale pending tail rows from the terminal buffer when the first partial result arrives", async () => { + const term = new VirtualTerminal(80, 8, 1_000); + const scheduler = new StressRenderScheduler(); + const tui = new TUI(term, undefined, { renderScheduler: scheduler }); + const component = new ToolExecutionComponent("write", writeArgs(20), {}, undefined, tui); + components.push(component); + tui.addChild(component); + tui.addChild(new Footer(5)); + + try { + tui.start(); + await scheduler.drain(term); + const pendingRows = plainBuffer(term); + expect(pendingRows.some(row => row.includes("… (8 earlier lines)"))).toBe(true); + expect(pendingRows.some(row => row.includes("… (streaming)"))).toBe(true); + expect(pendingRows.some(row => row.includes("20 line 20"))).toBe(true); + + component.setArgsComplete(); + tui.requestRender(); + await scheduler.drain(term); + + component.updateResult(partialWriteResult(), true); + tui.requestRender(); + await scheduler.drain(term); + + const rows = plainBuffer(term); + // The stale pending tail window must not survive above the new frame. + expect(rows.some(row => row.includes("… (streaming)"))).toBe(false); + expect(rows.some(row => row.includes("earlier lines"))).toBe(false); + expect(rows.some(row => row.includes("20 line 20"))).toBe(false); + // The first partial-result frame is what remains: progress line plus the + // top-anchored preview. + expect(rows.some(row => row.includes("Writing notes.txt..."))).toBe(true); + expect(rows.some(row => row.includes(" 1 line 1"))).toBe(true); + expect(rows.some(row => row.includes("… 14 more lines"))).toBe(true); + } finally { + tui.stop(); + await term.flush(); + } + }); +}); From 7f771049290842a00381e15b83821e6480b62c1f Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:24:11 +0200 Subject: [PATCH 24/59] fix(ai): deferred stream idle watchdog while cursor exec tools run locally The lazy stream wrapper registered streamCursor without provider-handled timeouts, so iterateWithIdleTimeout treated every gap between AssistantMessageEvents as potential provider death. During a Cursor exec-channel round-trip the server is waiting on OUR local tool result (shell/read/grep/write/MCP/...) and legitimately sends nothing, so any local tool outliving the idle budget (120s default) tripped onIdle -> abortLocally(StreamTimeoutError) and killed a healthy stream mid-task. Fix at the watchdog seam instead of faking progress events: - EventStream tracks consumer-side local work in flight (trackLocalWork / hasPendingLocalWork). - iterateWithIdleTimeout accepts a hasPendingLocalWork probe; an expired idle or first-event deadline slides forward while it reports true and the pending iterator.next() is persisted across the extension so no item is dropped. Once local work finishes the watchdog re-arms with a full budget, so genuinely silent streams still abort. - forwardStream wires the probe for any provider stream instance. - cursor.ts marks the exec-server dispatch as local work, covering every exec case including shellStream and MCP. Unlike synthesizing empty toolcall_delta keepalives (PR #4594), no synthetic events reach consumers: post-toolcall_end deltas would clobber reconstructed tool arguments in proxy adapters and re-trigger TTSR argument checks. Adopted from PR #4594: the setCursorProviderModule test seam. Fixes #4593 --- packages/ai/CHANGELOG.md | 4 + packages/ai/src/providers/cursor.ts | 27 ++- .../ai/src/providers/register-builtins.ts | 15 ++ packages/ai/src/utils/event-stream.ts | 26 +++ packages/ai/src/utils/idle-iterator.ts | 74 ++++++- packages/ai/test/cursor-exec-handlers.test.ts | 200 +++++++++++++++++- packages/ai/test/issue-4593-repro.test.ts | 190 +++++++++++++++++ 7 files changed, 515 insertions(+), 21 deletions(-) create mode 100644 packages/ai/test/issue-4593-repro.test.ts diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 45a58da7e..3eee5357a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed the generic lazy-stream idle watchdog aborting healthy `cursor-agent` streams with "Provider stream stalled while waiting for the next event" while a Cursor exec-channel local tool (shell/read/grep/write/MCP/…) legitimately ran longer than the idle budget. Provider streams now advertise consumer-side local work in flight and the watchdog slides its deadline instead of aborting; genuinely silent streams still time out. ([#4593](https://github.com/can1357/oh-my-pi/issues/4593)) + ### Changed - Changed the xAI Grok OAuth (`xai-oauth`) provider to use manual code-paste login by default. `/login` now accepts a pasted authorization code or full `http://127.0.0.1:56121/callback?code=...` redirect URL without starting a local callback listener ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). diff --git a/packages/ai/src/providers/cursor.ts b/packages/ai/src/providers/cursor.ts index b7cb3e3de..d2034de31 100644 --- a/packages/ai/src/providers/cursor.ts +++ b/packages/ai/src/providers/cursor.ts @@ -653,7 +653,8 @@ export interface UsageState { sawTokenDelta: boolean; } -async function handleServerMessage( +/** Exported for tests: drives one Cursor server message through the stream (exec waits mark the stream busy). */ +export async function handleServerMessage( msg: AgentServerMessage, output: AssistantMessage, stream: AssistantMessageEventStream, @@ -675,15 +676,21 @@ async function handleServerMessage( } else if (msgCase === "kvServerMessage") { handleKvServerMessage(msg.message.value as KvServerMessage, blobStore, h2Request); } else if (msgCase === "execServerMessage") { - await handleExecServerMessage( - msg.message.value as ExecServerMessage, - h2Request, - execHandlers, - onToolResult, - requestContextTools, - output, - stream, - state, + // The server is waiting on OUR local tool result during this window — no + // AssistantMessageEvent flows until the handler finishes. Mark the wait + // as local work so the lazy stream idle watchdog attributes the silence + // to the tool run instead of aborting a healthy stream (issue #4593). + await stream.trackLocalWork( + handleExecServerMessage( + msg.message.value as ExecServerMessage, + h2Request, + execHandlers, + onToolResult, + requestContextTools, + output, + stream, + state, + ), ); } else if (msgCase === "conversationCheckpointUpdate") { handleConversationCheckpointUpdate(msg.message.value, output, usageState, onConversationCheckpoint); diff --git a/packages/ai/src/providers/register-builtins.ts b/packages/ai/src/providers/register-builtins.ts index efd6b8dd6..8bb3a20f0 100644 --- a/packages/ai/src/providers/register-builtins.ts +++ b/packages/ai/src/providers/register-builtins.ts @@ -157,6 +157,7 @@ let openAICompletionsProviderModulePromise: Promise> | undefined; let ollamaProviderModulePromise: Promise> | undefined; let cursorProviderModulePromise: Promise> | undefined; +let cursorProviderModuleOverride: LazyProviderModule<"cursor-agent"> | undefined; let devinProviderModulePromise: Promise> | undefined; let bedrockProviderModuleOverride: LazyProviderModule<"bedrock-converse-stream"> | undefined; let bedrockProviderModulePromise: Promise> | undefined; @@ -167,6 +168,12 @@ export function setBedrockProviderModule(module: BedrockProviderModule): void { }; } +export function setCursorProviderModule(module: CursorProviderModule): void { + cursorProviderModuleOverride = { + stream: module.streamCursor, + }; +} + // --------------------------------------------------------------------------- // Stream forwarding / error helpers // --------------------------------------------------------------------------- @@ -245,6 +252,10 @@ function forwardStream( (limits?.openAIIdleEnvFloorsFirstEvent ? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs, limits.defaultFirstEventTimeoutMs) : getStreamFirstEventTimeoutMs(idleTimeoutMs, limits?.defaultFirstEventTimeoutMs))); + // Providers with a server-driven local tool bridge (e.g. the Cursor + // exec channel) mark their stream busy while a local tool runs; the + // watchdog must not read that silence as a provider stall (#4593). + const localWorkSource = source instanceof EventStreamImpl ? source : undefined; const watchedSource = iterateWithIdleTimeout(source, { idleTimeoutMs, firstItemTimeoutMs, @@ -260,6 +271,7 @@ function forwardStream( // `idleTimeoutMs` while we're still legitimately waiting on the model's // first response (slow first-token from reasoning models, cold proxies, etc.). isProgressItem: event => (event as AssistantMessageEvent).type !== "start", + hasPendingLocalWork: localWorkSource ? () => localWorkSource.hasPendingLocalWork : undefined, }); for await (const event of watchedSource) { @@ -411,6 +423,9 @@ function loadOllamaProviderModule(): Promise> } function loadCursorProviderModule(): Promise> { + if (cursorProviderModuleOverride) { + return Promise.resolve(cursorProviderModuleOverride); + } cursorProviderModulePromise ||= import("./cursor").then(module => { const provider = module as CursorProviderModule; return { stream: provider.streamCursor }; diff --git a/packages/ai/src/utils/event-stream.ts b/packages/ai/src/utils/event-stream.ts index d7b75e863..6d1255e39 100644 --- a/packages/ai/src/utils/event-stream.ts +++ b/packages/ai/src/utils/event-stream.ts @@ -10,6 +10,14 @@ export class EventStream implements AsyncIterable { resultSettled = false; #failed = false; #error: unknown = undefined; + /** + * Consumer-side local operations currently in flight for this stream — a + * provider transport waiting on a server-requested local tool bridge + * (e.g. the Cursor exec channel) before it can send the result upstream. + * While non-zero, event silence is attributable to our own pending work, + * not a provider stall; idle watchdogs consult {@link hasPendingLocalWork}. + */ + #pendingLocalWork = 0; finalResultPromise: Promise; resolveFinalResult!: (result: R) => void; rejectFinalResult!: (err: unknown) => void; @@ -116,6 +124,24 @@ export class EventStream implements AsyncIterable { result(): Promise { return this.finalResultPromise; } + + /** True while local work tracked via {@link trackLocalWork} is pending. */ + get hasPendingLocalWork(): boolean { + return this.#pendingLocalWork > 0; + } + + /** + * Track a local-work promise so idle watchdogs on this stream do not treat + * the event silence while it is pending as a provider stall. + */ + async trackLocalWork(work: Promise): Promise { + this.#pendingLocalWork++; + try { + return await work; + } finally { + this.#pendingLocalWork--; + } + } } export class AssistantMessageEventStream extends EventStream { diff --git a/packages/ai/src/utils/idle-iterator.ts b/packages/ai/src/utils/idle-iterator.ts index 7cebaf64e..3accaf3c2 100644 --- a/packages/ai/src/utils/idle-iterator.ts +++ b/packages/ai/src/utils/idle-iterator.ts @@ -135,6 +135,16 @@ export interface IdleTimeoutIteratorOptions { * keepalive/no-op events from keeping a stalled tool call alive forever. */ isProgressItem?: (item: unknown) => boolean; + /** + * Reports consumer-side local work in flight for the stream: the provider + * transport is waiting on a server-requested local tool bridge (e.g. the + * Cursor exec channel) before anything can flow upstream again. While it + * returns true, an expired idle / first-item deadline slides forward + * instead of aborting — the silence is ours, not a provider stall. The + * watchdog re-arms with a full budget once the local work completes, so a + * provider that stalls afterwards is still caught. + */ + hasPendingLocalWork?: () => boolean; /** * Cancel iteration as soon as this signal aborts. Required for caller-driven * cancellation (ESC) when the underlying transport does not surface signal @@ -157,7 +167,7 @@ export async function* iterateWithIdleTimeout( options: IdleTimeoutIteratorOptions, ): AsyncGenerator { const firstItemTimeoutMs = options.firstItemTimeoutMs ?? options.idleTimeoutMs; - const firstItemDeadlineMs = + let firstItemDeadlineMs = firstItemTimeoutMs !== undefined && firstItemTimeoutMs > 0 ? Date.now() + firstItemTimeoutMs : undefined; const abortSignal = options.abortSignal; const iterator = iterable[Symbol.asyncIterator](); @@ -197,6 +207,28 @@ export async function* iterateWithIdleTimeout( }; let lastProgressAt = Date.now(); + const hasPendingLocalWork = (): boolean => { + if (!options.hasPendingLocalWork) return false; + try { + return options.hasPendingLocalWork(); + } catch { + return false; + } + }; + // Local work means the current gap is attributable to the consumer side, + // not the provider: slide the active deadline a full budget past now + // instead of aborting. Once the work completes the watchdog resumes from + // the last extension, so a provider that stalls afterwards is still caught. + const extendDeadlineForLocalWork = (): void => { + if (awaitingFirstItem) { + if (firstItemDeadlineMs !== undefined && firstItemTimeoutMs !== undefined) { + firstItemDeadlineMs = Date.now() + firstItemTimeoutMs; + } + } else { + lastProgressAt = Date.now(); + } + }; + const noTimeoutEnforced = (firstItemTimeoutMs === undefined || firstItemTimeoutMs <= 0) && (options.idleTimeoutMs === undefined || options.idleTimeoutMs <= 0); @@ -271,6 +303,12 @@ export async function* iterateWithIdleTimeout( timer = setTimeout(onTimerFire, Math.max(0, deadlineMs - Date.now())); }; + // The in-flight iterator.next() promise, persisted across loop iterations: + // a deadline extension for pending local work loops without consuming it, + // and issuing a second next() while one is outstanding would drop an item. + let pendingNext: + | Promise<{ kind: "next"; result: IteratorResult } | { kind: "error"; error: unknown }> + | undefined; try { let raceCount = 0; while (true) { @@ -291,21 +329,29 @@ export async function* iterateWithIdleTimeout( if (firstItemDeadlineMs !== undefined) { activeTimeoutMs = firstItemDeadlineMs - Date.now(); if (activeTimeoutMs <= 0) { - options.onFirstItemTimeout?.(); - closeIterator(); - throw new AIError.StreamTimeoutError(options.firstItemErrorMessage ?? options.errorMessage); + if (!hasPendingLocalWork()) { + options.onFirstItemTimeout?.(); + closeIterator(); + throw new AIError.StreamTimeoutError(options.firstItemErrorMessage ?? options.errorMessage); + } + extendDeadlineForLocalWork(); + activeTimeoutMs = firstItemDeadlineMs! - Date.now(); } } } else if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) { activeTimeoutMs = options.idleTimeoutMs - (Date.now() - lastProgressAt); if (activeTimeoutMs <= 0) { - options.onIdle?.(); - closeIterator(); - throw new AIError.StreamTimeoutError(options.errorMessage); + if (!hasPendingLocalWork()) { + options.onIdle?.(); + closeIterator(); + throw new AIError.StreamTimeoutError(options.errorMessage); + } + extendDeadlineForLocalWork(); + activeTimeoutMs = options.idleTimeoutMs; } } - const nextResultPromise = withRacy(iterator.next()); + pendingNext ??= withRacy(iterator.next()); const racers: Array< Promise< @@ -314,7 +360,7 @@ export async function* iterateWithIdleTimeout( | { kind: "timeout" } | { kind: "abort" } > - > = [nextResultPromise]; + > = [pendingNext]; const enforceTimeout = !noTimeoutEnforced && activeTimeoutMs !== undefined && activeTimeoutMs > 0; if (enforceTimeout) { @@ -333,11 +379,21 @@ export async function* iterateWithIdleTimeout( let continuing = false; try { const outcome = await Promise.race(racers); + if (outcome.kind === "next" || outcome.kind === "error") { + pendingNext = undefined; + } if (outcome.kind === "abort") { closeIterator(); throw abortReason(abortSignal!); } if (outcome.kind === "timeout") { + if (hasPendingLocalWork()) { + // A local tool is still running; the provider cannot make + // progress until we hand its result back. Keep waiting. + extendDeadlineForLocalWork(); + continuing = true; + continue; + } if (!awaitingFirstItem) { options.onIdle?.(); } else { diff --git a/packages/ai/test/cursor-exec-handlers.test.ts b/packages/ai/test/cursor-exec-handlers.test.ts index e77bc484f..c79b0e12d 100644 --- a/packages/ai/test/cursor-exec-handlers.test.ts +++ b/packages/ai/test/cursor-exec-handlers.test.ts @@ -1,14 +1,25 @@ +import { create } from "@bufbuild/protobuf"; import { describe, expect, it } from "bun:test"; import { + type BlockState, buildCursorHistoryForTest, buildCursorSystemPromptJsons, emptyGrepPatternRejection, + handleServerMessage, resolveExecHandler, streamCursor, + type ToolCallState, } from "@oh-my-pi/pi-ai/providers/cursor"; -import type { Context, Model } from "@oh-my-pi/pi-ai/types"; +import { setCursorProviderModule, streamCursor as lazyStreamCursor } from "@oh-my-pi/pi-ai/providers/register-builtins"; +import type { AssistantMessage, Context, CursorExecHandlers, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; -import type { AgentRunRequest } from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb"; +import { + type AgentRunRequest, + AgentServerMessageSchema, + ExecServerMessageSchema, + ReadArgsSchema, +} from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb"; const cursorModel: Model<"cursor-agent"> = buildModel({ id: "cursor-composer-2.5", @@ -361,3 +372,188 @@ describe("Cursor grepArgs empty-pattern guard (issue #4574)", () => { expect(emptyGrepPatternRejection("\t\n", "src/**/*.ts")).toContain('"src/**/*.ts"'); }); }); + +function cursorAssistantMessage(): AssistantMessage { + return { + role: "assistant", + content: [], + api: "cursor-agent", + provider: "cursor", + model: "cursor-composer-2.5", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: 0, + }; +} + +function newBlockState(): BlockState { + let textBlock: BlockState["currentTextBlock"] = null; + let thinkingBlock: BlockState["currentThinkingBlock"] = null; + let toolCall: ToolCallState | null = null; + return { + get currentTextBlock() { + return textBlock; + }, + get currentThinkingBlock() { + return thinkingBlock; + }, + get currentToolCall() { + return toolCall; + }, + firstTokenTime: undefined, + setTextBlock: b => { + textBlock = b; + }, + setThinkingBlock: b => { + thinkingBlock = b; + }, + setToolCall: t => { + toolCall = t; + }, + setFirstTokenTime: () => {}, + }; +} + +describe("Cursor exec local-work tracking (issue #4593)", () => { + it("marks the stream busy for the duration of a local exec handler", async () => { + const output = cursorAssistantMessage(); + const stream = new AssistantMessageEventStream(); + const state = newBlockState(); + const written: unknown[] = []; + const h2Request = { + write: (chunk: unknown) => { + written.push(chunk); + return true; + }, + } as unknown as Parameters[5]; + const handlerGate = Promise.withResolvers(); + const execHandlers: CursorExecHandlers = { + async read(args) { + await handlerGate.promise; + return { + role: "toolResult", + toolCallId: args.toolCallId, + toolName: "read", + content: [{ type: "text", text: "file contents" }], + isError: false, + timestamp: 1, + } satisfies ToolResultMessage; + }, + }; + const serverMsg = create(AgentServerMessageSchema, { + message: { + case: "execServerMessage", + value: create(ExecServerMessageSchema, { + id: 1, + execId: "exec-1", + message: { + case: "readArgs", + value: create(ReadArgsSchema, { path: "/tmp/slow-file", toolCallId: "call-read-1" }), + }, + }), + }, + }); + + expect(stream.hasPendingLocalWork).toBe(false); + const dispatch = handleServerMessage( + serverMsg, + output, + stream, + state, + new Map(), + h2Request, + execHandlers, + undefined, + { sawTokenDelta: false }, + [], + ); + + // The exec round-trip is in flight: the stream must advertise local + // work so the lazy idle watchdog defers instead of aborting. + expect(stream.hasPendingLocalWork).toBe(true); + + handlerGate.resolve(); + await dispatch; + + expect(stream.hasPendingLocalWork).toBe(false); + // The read result went back out on the exec channel. + expect(written.length).toBe(1); + }); + + it("survives a local exec tool outliving the lazy idle budget end to end", async () => { + const workDone = Promise.withResolvers(); + // The tracked work completes only once the lazy watchdog has consulted + // the stream's local-work state at two expired deadlines, proving the + // idle budget was truly exceeded while the exec tool ran. + class ProbedStream extends AssistantMessageEventStream { + probeCalls = 0; + override get hasPendingLocalWork(): boolean { + this.probeCalls++; + if (this.probeCalls >= 2) workDone.resolve(); + return super.hasPendingLocalWork; + } + } + const source = new ProbedStream(); + let providerSignal: AbortSignal | undefined; + setCursorProviderModule({ + streamCursor: (_model, _context, options) => { + providerSignal = options.signal; + void (async () => { + const partial = cursorAssistantMessage(); + source.push({ type: "start", partial }); + source.push({ type: "text_delta", contentIndex: 0, delta: "spawning local tool", partial }); + await source.trackLocalWork(workDone.promise); + const message = cursorAssistantMessage(); + source.push({ type: "done", reason: "stop", message }); + })(); + return source; + }, + }); + + const stream = lazyStreamCursor(cursorModel, { messages: [] }, { apiKey: "test", streamIdleTimeoutMs: 5 }); + const result = await stream.result(); + + expect(providerSignal?.aborted).toBe(false); + expect(source.probeCalls).toBeGreaterThanOrEqual(2); + expect(result.stopReason).toBe("stop"); + }); + + it("still aborts a silent cursor stream with no local work in flight", async () => { + const partial = cursorAssistantMessage(); + let providerSignal: AbortSignal | undefined; + const source = { + async *[Symbol.asyncIterator]() { + yield { type: "start", partial } as const; + yield { type: "text_delta", contentIndex: 0, delta: "hello", partial } as const; + const stalled = Promise.withResolvers(); + if (providerSignal?.aborted) { + stalled.reject(new Error("Request was aborted")); + } + providerSignal?.addEventListener("abort", () => stalled.reject(new Error("Request was aborted")), { + once: true, + }); + await stalled.promise; + }, + } as unknown as AssistantMessageEventStream; + setCursorProviderModule({ + streamCursor: (_model, _context, options) => { + providerSignal = options.signal; + return source; + }, + }); + + const stream = lazyStreamCursor(cursorModel, { messages: [] }, { apiKey: "test", streamIdleTimeoutMs: 10 }); + const result = await stream.result(); + + expect(providerSignal?.aborted).toBe(true); + expect(result.stopReason).toBe("error"); + expect(result.errorMessage).toBe("Provider stream stalled while waiting for the next event"); + }); +}); diff --git a/packages/ai/test/issue-4593-repro.test.ts b/packages/ai/test/issue-4593-repro.test.ts new file mode 100644 index 000000000..e79f4f1e1 --- /dev/null +++ b/packages/ai/test/issue-4593-repro.test.ts @@ -0,0 +1,190 @@ +import { describe, expect, it } from "bun:test"; +import { setBedrockProviderModule, streamBedrock } from "@oh-my-pi/pi-ai/providers/register-builtins"; +import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { iterateWithIdleTimeout } from "@oh-my-pi/pi-ai/utils/idle-iterator"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +// Issue #4593: the generic lazy stream watchdog treats "no AssistantMessageEvent" +// as "provider stalled". During a Cursor exec-channel round-trip the server is +// waiting on OUR local tool result and legitimately sends nothing, so a local +// tool outliving the idle budget aborted a healthy stream with "Provider stream +// stalled while waiting for the next event". Provider streams now advertise +// pending local work and the watchdog slides its deadline instead of aborting. +// +// These tests exercise the real watchdog timer against the platform clock (that +// timer IS the unit under test), but never guess durations: the simulated local +// work completes only once the watchdog has demonstrably reached an expired +// deadline and consulted the local-work probe, so the tests stay causal on a +// loaded machine. Budgets are a few milliseconds. + +function createModel(): Model<"bedrock-converse-stream"> { + return buildModel({ + id: "mock-bedrock", + name: "Mock Bedrock", + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + baseUrl: "https://example.invalid", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 8192, + maxTokens: 2048, + }); +} + +function createAssistantMessage(): AssistantMessage { + return { + role: "assistant", + content: [{ type: "text", text: "ok" }], + api: "bedrock-converse-stream", + provider: "amazon-bedrock", + model: "mock-bedrock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + timestamp: Date.now(), + }; +} + +const baseContext: Context = { messages: [] }; + +describe("idle watchdog local-work deferral (issue #4593)", () => { + it("slides the idle deadline while consumer-side local work is pending", async () => { + const workDone = Promise.withResolvers(); + let probeCalls = 0; + let busy = true; + async function* source() { + yield "first"; + // The "local tool": finishes only after the watchdog has hit an + // expired deadline twice and deferred both times. + await workDone.promise; + busy = false; + yield "second"; + } + let idleFired = false; + const items: string[] = []; + for await (const item of iterateWithIdleTimeout(source(), { + idleTimeoutMs: 5, + errorMessage: "stalled", + onIdle: () => { + idleFired = true; + }, + hasPendingLocalWork: () => { + probeCalls++; + if (probeCalls >= 2) workDone.resolve(); + return busy; + }, + })) { + items.push(item); + } + expect(items).toEqual(["first", "second"]); + expect(probeCalls).toBeGreaterThanOrEqual(2); + expect(idleFired).toBe(false); + }); + + it("still aborts a silent stream once local work has finished", async () => { + const workDone = Promise.withResolvers(); + let busy = true; + async function* source() { + yield "first"; + await workDone.promise; + busy = false; + // The provider genuinely stalls after the local work completed. + await new Promise(() => {}); + yield "never"; + } + const items: string[] = []; + let error: Error | undefined; + try { + for await (const item of iterateWithIdleTimeout(source(), { + idleTimeoutMs: 5, + errorMessage: "stalled", + hasPendingLocalWork: () => { + workDone.resolve(); + return busy; + }, + })) { + items.push(item); + } + } catch (err) { + error = err as Error; + } + expect(items).toEqual(["first"]); + expect(error?.message).toBe("stalled"); + }); + + it("slides the first-event deadline while local work is pending", async () => { + const workDone = Promise.withResolvers(); + let probeCalls = 0; + let busy = true; + async function* source() { + // Local bridge work before the model has produced any event. + await workDone.promise; + yield "first"; + } + const items: string[] = []; + for await (const item of iterateWithIdleTimeout(source(), { + idleTimeoutMs: 5, + firstItemTimeoutMs: 5, + errorMessage: "stalled", + firstItemErrorMessage: "first event timed out", + hasPendingLocalWork: () => { + probeCalls++; + if (probeCalls >= 2) workDone.resolve(); + return busy; + }, + })) { + items.push(item); + busy = false; + } + expect(items).toEqual(["first"]); + expect(probeCalls).toBeGreaterThanOrEqual(2); + }); + + it("does not abort a lazy provider stream while tracked local work outlives the idle budget", async () => { + const workDone = Promise.withResolvers(); + // Counts how often the lazy wrapper's watchdog consults the stream's + // local-work state at an expired deadline; the tracked work completes + // only after two deferrals, proving the budget was truly exceeded. + class ProbedStream extends AssistantMessageEventStream { + probeCalls = 0; + override get hasPendingLocalWork(): boolean { + this.probeCalls++; + if (this.probeCalls >= 2) workDone.resolve(); + return super.hasPendingLocalWork; + } + } + const source = new ProbedStream(); + let providerSignal: AbortSignal | undefined; + setBedrockProviderModule({ + streamBedrock: (_model, _context, options) => { + providerSignal = options.signal; + void (async () => { + const partial = createAssistantMessage(); + source.push({ type: "start", partial }); + source.push({ type: "text_delta", contentIndex: 0, delta: "running a local tool", partial }); + // Server-driven local tool run: no events flow while the + // tracked work is pending. + await source.trackLocalWork(workDone.promise); + source.push({ type: "done", reason: "stop", message: createAssistantMessage() }); + })(); + return source; + }, + }); + + const stream = streamBedrock(createModel(), baseContext, { streamIdleTimeoutMs: 5 }); + const result = await stream.result(); + + expect(providerSignal?.aborted).toBe(false); + expect(source.probeCalls).toBeGreaterThanOrEqual(2); + expect(result.stopReason).toBe("stop"); + expect(result.errorMessage).toBeUndefined(); + }); +}); From 5d1e25f8342a16921e28ab64249c3203d039db42 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:23:19 +0200 Subject: [PATCH 25/59] fix(shell): bounded backpressured native bash output bridge MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The non-PTY bash streaming bridge queued every decoded chunk into flume::unbounded and fired ThreadsafeFunction callbacks NonBlocking with no budget, so a producer outrunning the JS event loop grew the native queue (and the napi queue behind it) without bound — measured 33.5 MB queued for a 32 MiB stream with a stalled consumer, and multi-GB RSS on longer runs. The downstream OutputSink caps sit after the N-API boundary and cannot bound either queue. Bound the pipeline end to end without dropping data: - pi-natives: bridge_chunks now creates flume::bounded(64) and the drain task (extracted as pump_chunks) awaits on_chunk.call_async per coalesced <=64 KiB batch, so at most one batch sits in the napi queue and the JS event loop's real consumption rate backpressures the whole pipeline. If the JS side is gone, the pump exits and drops the receiver so senders fail fast. - pi-shell: emit_chunk sends with send_async().await — a full bridge queue parks the pipe reader, which parks the child on its stdout/stderr pipe (ordinary pipe backpressure) instead of buffering; a disconnected receiver fails immediately so child pipes always keep draining. Unlike a drop-after-cap design, every byte still reaches JS: the rolling tail view, lossless [raw output: artifact://…] capture, and totalBytes accounting keep working for outputs past the display cap. E2E (darwin-arm64 addon): 32 MiB through a JS callback stalling 1 ms per call — lossless, 472 coalesced callbacks, peak RSS +21.8 MiB. Fixes #4078 --- crates/pi-natives/src/shell.rs | 163 ++++++++++++++++++++++++++------- crates/pi-shell/src/shell.rs | 66 ++++++++++--- packages/natives/CHANGELOG.md | 4 + 3 files changed, 188 insertions(+), 45 deletions(-) diff --git a/crates/pi-natives/src/shell.rs b/crates/pi-natives/src/shell.rs index 47402b5fc..f095abec0 100644 --- a/crates/pi-natives/src/shell.rs +++ b/crates/pi-natives/src/shell.rs @@ -5,7 +5,7 @@ use std::{collections::HashMap, sync::Arc}; use napi::{ Env, Result, bindgen_prelude::*, - threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode}, + threadsafe_function::{ThreadsafeFunction, UnknownReturnValue}, }; use napi_derive::napi; use pi_shell::{ @@ -216,7 +216,7 @@ impl Shell { env: &'env Env, options: ShellRunOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] - on_chunk: Option>, + on_chunk: Option>, ) -> Result> { let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal); let inner = Arc::clone(&self.inner); @@ -269,7 +269,7 @@ pub fn execute_shell<'env>( env: &'env Env, options: ShellExecuteOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] - on_chunk: Option>, + on_chunk: Option>, ) -> Result> { let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal); let exec_options = CoreShellExecuteOptions { @@ -294,42 +294,66 @@ pub fn execute_shell<'env>( }) } +/// Capacity (in chunks) of the queue between the pipe readers and the JS +/// forwarding pump. One queued chunk is at most one pipe read (≤64 KiB), so +/// the Rust side of the bridge holds ~4 MiB worst case before the readers' +/// `send_async` parks — which in turn parks the child on its stdout/stderr +/// pipe (ordinary pipe backpressure) instead of buffering the surplus in +/// process memory (#4078). +const BRIDGE_QUEUE_CHUNKS: usize = 64; + fn bridge_chunks( - on_chunk: Option>, + on_chunk: Option>, ) -> (Option>, Option>) { let Some(on_chunk) = on_chunk else { return (None, None); }; - let (tx, rx) = flume::unbounded::(); - let handle = napi::tokio::spawn(async move { - // Hard cap on one coalesced batch so the JS main thread never sees a - // multi-MB napi callback (a giant single string would stall sanitize + - // tail-buffer maintenance for the whole copy). - const MAX_BATCH_BYTES: usize = 64 * 1024; - // Initial capacity sized for typical bursty pipe output. Re-allocated - // each batch because `String` ownership is moved into the napi call. - const INITIAL_BATCH_CAP: usize = 8 * 1024; - let mut batch = String::with_capacity(INITIAL_BATCH_CAP); - while let Ok(first) = rx.recv_async().await { - batch.push_str(&first); - // Greedily drain everything already queued. Child processes that - // write byte-at-a-time (printf-style progress, llama-cli token - // streams) otherwise produce one napi callback per `write(2)`, - // saturating the JS main thread (~200% CPU observed) and leaving - // the queue draining long after the child exits. - while batch.len() < MAX_BATCH_BYTES { - match rx.try_recv() { - Ok(more) => batch.push_str(&more), - Err(_) => break, - } - } - let payload = std::mem::replace(&mut batch, String::with_capacity(INITIAL_BATCH_CAP)); - on_chunk.call(Ok(payload), ThreadsafeFunctionCallMode::NonBlocking); - } - }); + let (tx, rx) = flume::bounded::(BRIDGE_QUEUE_CHUNKS); + let handle = napi::tokio::spawn(pump_chunks(rx, async move |payload: String| { + // `call_async` resolves only after the JS callback ran, so at most + // one batch sits in the napi queue at a time and the JS event loop's + // actual consumption rate backpressures the whole pipeline. An error + // means the JS side is gone (env teardown) — stop forwarding. + on_chunk.call_async(Ok(payload)).await.is_ok() + })); (Some(tx), Some(handle)) } +/// Drain `rx`, greedily coalescing queued chunks into ≤64 KiB batches, and +/// feed each batch to `forward`, awaiting its completion before pulling more. +/// Returns when `rx` disconnects (all senders dropped) or `forward` reports +/// the consumer is gone; dropping `rx` then disconnects the channel so +/// parked/future senders fail fast and the pipe readers keep draining the +/// child instead of wedging it. +async fn pump_chunks(rx: flume::Receiver, mut forward: impl AsyncFnMut(String) -> bool) { + // Hard cap on one coalesced batch so the JS main thread never sees a + // multi-MB napi callback (a giant single string would stall sanitize + + // tail-buffer maintenance for the whole copy). + const MAX_BATCH_BYTES: usize = 64 * 1024; + // Initial capacity sized for typical bursty pipe output. Re-allocated + // each batch because `String` ownership is moved into the napi call. + const INITIAL_BATCH_CAP: usize = 8 * 1024; + let mut batch = String::with_capacity(INITIAL_BATCH_CAP); + while let Ok(first) = rx.recv_async().await { + batch.push_str(&first); + // Greedily drain everything already queued. Child processes that + // write byte-at-a-time (printf-style progress, llama-cli token + // streams) otherwise produce one napi callback per `write(2)`, + // saturating the JS main thread (~200% CPU observed) and leaving + // the queue draining long after the child exits. + while batch.len() < MAX_BATCH_BYTES { + match rx.try_recv() { + Ok(more) => batch.push_str(&more), + Err(_) => break, + } + } + let payload = std::mem::replace(&mut batch, String::with_capacity(INITIAL_BATCH_CAP)); + if !forward(payload).await { + return; + } + } +} + /// Result of [`apply_bash_fixups`]: a possibly-rewritten command plus the /// substrings that were removed (in source order). #[napi(object)] @@ -360,7 +384,6 @@ pub fn apply_bash_fixups(command: String) -> BashFixupResult { mod tests { use std::time::Duration; - #[cfg(unix)] use flume; use pi_shell::{ ShellRunOptions as CoreShellRunOptions, @@ -368,7 +391,81 @@ mod tests { }; use tokio::time; - use super::CoreShell; + use super::{BRIDGE_QUEUE_CHUNKS, CoreShell, pump_chunks}; + + /// Regression for #4078: the reader→JS bridge queue must stay bounded when + /// the JS side (here: a deliberately slow `forward`) cannot keep up with a + /// fast producer, and backpressure must never drop or reorder chunks. On + /// the pre-fix bridge (`flume::unbounded` + fire-and-forget + /// `ThreadsafeFunctionCallMode::NonBlocking`) the same harness accumulates + /// the producer's entire surplus in the queue (measured: a 32 MiB stream + /// queued all 33_554_432 bytes while the consumer stalled). + #[tokio::test(flavor = "multi_thread")] + async fn bridge_pump_bounds_queue_and_delivers_all_bytes() { + const CHUNKS: usize = 512; + const CHUNK_BYTES: usize = 4096; + let (tx, rx) = flume::bounded::(BRIDGE_QUEUE_CHUNKS); + let producer = tokio::spawn(async move { + let mut expected = String::with_capacity(CHUNKS * CHUNK_BYTES); + let mut max_queued = 0usize; + for i in 0..CHUNKS { + let chunk = format!("[{i:06}]{}", "x".repeat(CHUNK_BYTES - 8)); + expected.push_str(&chunk); + tx.send_async(chunk).await.expect("pump should outlive the producer"); + max_queued = max_queued.max(tx.len()); + } + (expected, max_queued) + }); + + let mut received = String::with_capacity(CHUNKS * CHUNK_BYTES); + time::timeout( + Duration::from_secs(30), + pump_chunks(rx, async |payload: String| { + received.push_str(&payload); + // Emulate a busy JS event loop: each napi callback takes a while. + time::sleep(Duration::from_micros(500)).await; + true + }), + ) + .await + .expect("pump should finish once the producer hangs up"); + + let (expected, max_queued) = producer.await.expect("producer task"); + assert!( + max_queued <= BRIDGE_QUEUE_CHUNKS, + "bridge queue grew past its bound: {max_queued} chunks", + ); + assert_eq!(received.len(), expected.len(), "bytes were dropped or duplicated"); + assert_eq!(received, expected, "chunks must arrive losslessly and in order"); + } + + /// When the JS side dies (`forward` fails: threadsafe function aborted on + /// env teardown), the pump must drop its receiver so parked and future + /// sends fail fast — the pipe readers keep draining the child instead of + /// wedging it on a full bridge queue. + #[tokio::test(flavor = "multi_thread")] + async fn bridge_pump_death_disconnects_channel_without_blocking_senders() { + let (tx, rx) = flume::bounded::(4); + let pump = tokio::spawn(pump_chunks(rx, async |_payload: String| false)); + let producer = tokio::spawn(async move { + let mut disconnected = 0usize; + for _ in 0..64 { + if tx.send_async("x".repeat(1024)).await.is_err() { + disconnected += 1; + } + } + disconnected + }); + let disconnected = time::timeout(Duration::from_secs(5), producer) + .await + .expect("sends must not park once the consumer died") + .expect("producer task"); + assert!(disconnected > 0, "channel should disconnect after the pump stops"); + time::timeout(Duration::from_secs(5), pump) + .await + .expect("pump should exit after forward fails") + .expect("pump task"); + } mod child_session_action_tests { use pi_shell::{ChildSessionAction, child_session_action}; diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index 80634bf91..b3ebe92b8 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -1649,7 +1649,7 @@ async fn read_output( let pending = &buf[..it]; match str::from_utf8(pending) { Ok(text) => { - emit_chunk(text, on_chunk.as_ref()); + emit_chunk(text, on_chunk.as_ref()).await; it = 0; break; }, @@ -1658,7 +1658,7 @@ async fn read_output( if p > 0 { // SAFETY: [..p] is guaranteed valid UTF-8 by valid_up_to(). let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; - emit_chunk(text, on_chunk.as_ref()); + emit_chunk(text, on_chunk.as_ref()).await; // copy p..it to the beginning of the buffer buf.copy_within(p..it, 0); it -= p; @@ -1667,7 +1667,7 @@ async fn read_output( match err.error_len() { Some(p) => { // Invalid byte sequence: emit replacement and drop those bytes. - emit_chunk(REPLACEMENT, on_chunk.as_ref()); + emit_chunk(REPLACEMENT, on_chunk.as_ref()).await; // copy p..it to the beginning of the buffer buf.copy_within(p..it, 0); it -= p; @@ -1688,10 +1688,10 @@ async fn read_output( for chunk in buf[..it].utf8_chunks() { let valid = chunk.valid(); if !valid.is_empty() { - emit_chunk(valid, on_chunk.as_ref()); + emit_chunk(valid, on_chunk.as_ref()).await; } if !chunk.invalid().is_empty() { - emit_chunk(REPLACEMENT, on_chunk.as_ref()); + emit_chunk(REPLACEMENT, on_chunk.as_ref()).await; } } } @@ -1777,7 +1777,7 @@ async fn read_output_buffered( while !pending.is_empty() { match str::from_utf8(&pending) { Ok(text) => { - emit_chunk(text, Some(cb)); + emit_chunk(text, Some(cb)).await; pending.clear(); break; }, @@ -1786,12 +1786,12 @@ async fn read_output_buffered( if p > 0 { // SAFETY: [..p] is valid UTF-8 per valid_up_to(). let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; - emit_chunk(text, Some(cb)); + emit_chunk(text, Some(cb)).await; pending.drain(..p); } match err.error_len() { Some(skip) => { - emit_chunk(REPLACEMENT, Some(cb)); + emit_chunk(REPLACEMENT, Some(cb)).await; pending.drain(..skip); }, None => break, @@ -1807,10 +1807,10 @@ async fn read_output_buffered( for chunk in pending.utf8_chunks() { let valid = chunk.valid(); if !valid.is_empty() { - emit_chunk(valid, Some(cb)); + emit_chunk(valid, Some(cb)).await; } if !chunk.invalid().is_empty() { - emit_chunk(REPLACEMENT, Some(cb)); + emit_chunk(REPLACEMENT, Some(cb)).await; } } } @@ -1858,9 +1858,16 @@ fn read_nonblocking(file: &T, buf: &mut [u8]) -> io::Re } } -fn emit_chunk(text: &str, callback: Option<&Sender>) { +/// Forward one decoded chunk to the streaming callback, honouring channel +/// backpressure: on a bounded channel (the pi-natives JS bridge) the send +/// parks until the consumer frees a slot — which parks the pipe reader and, +/// transitively, the child on its stdout/stderr pipe — so a fast producer +/// can never buffer unbounded output in memory (#4078). A disconnected +/// receiver (consumer gone) fails immediately, so the pipe keeps draining +/// and the child never wedges on a full pipe. +async fn emit_chunk(text: &str, callback: Option<&Sender>) { if let Some(callback) = callback { - let _ = callback.send(text.to_string()); + let _ = callback.send_async(text.to_string()).await; } } @@ -4140,4 +4147,39 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }] "builtin nohup masked SIGHUP like the external tool (output: {out:?})", ); } + + /// Regression for #4078: the JS bridge hands the pipe readers a *bounded* + /// chunk channel. With a consumer slower than the producer the readers + /// must park on `send_async` (backpressuring the child through its pipe) + /// rather than buffer unboundedly — and, unlike a drop-on-full design, + /// every produced byte must still reach the consumer. + #[cfg(unix)] + #[tokio::test(flavor = "multi_thread")] + async fn streaming_output_backpressures_on_bounded_channel_without_loss() { + const TOTAL_BYTES: usize = 1_048_576; + let (tx, rx) = flume::bounded::(4); + let options = ShellExecuteOptions { + command: format!("yes x | head -c {TOTAL_BYTES}"), + ..Default::default() + }; + let run = tokio::spawn(execute_shell(options, Some(tx), CancelToken::default())); + + let mut received = 0usize; + while let Ok(chunk) = rx.recv_async().await { + received += chunk.len(); + // Slow consumer: forces the bounded queue to fill and the readers + // to park between chunks. + time::sleep(Duration::from_micros(50)).await; + } + + let result = time::timeout(Duration::from_secs(30), run) + .await + .expect("command should finish despite backpressure") + .expect("run task should not panic") + .expect("execute should succeed"); + assert_eq!(result.exit_code, Some(0)); + assert!(!result.cancelled); + assert!(!result.timed_out); + assert_eq!(received, TOTAL_BYTES, "streamed bytes were dropped under backpressure"); + } } diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index f16aff66a..58027bbd9 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed unbounded memory growth in the native bash output bridge when a command produces output faster than the JS event loop consumes it: the shell streaming path now uses a bounded chunk queue with real backpressure (pipe readers park until the JS callback catches up, parking the child on its pipe) instead of buffering the entire surplus in memory. No output is dropped — the rolling tail view, `[raw output: artifact://…]` lossless capture, and byte accounting are unaffected ([#4078](https://github.com/can1357/oh-my-pi/issues/4078)). + ## [16.3.12] - 2026-07-08 ### Fixed From 606c51a7b7e5b81d86f14f234599d38e9b0d3f3f Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:31:36 +0200 Subject: [PATCH 26/59] fix(natives): decoded CF_DIB directly when arboard rejects an image arboard's Windows reader feeds Qt-style CF_DIBV5 payloads (BI_RGB plus alpha mask, rewritten to BI_BITFIELDS by its header tweak) to a header-less BMP decode that mis-places the pixel offset for V4/V5 bitfield headers, so PixPin/Snipaste screenshots failed with ConversionFailure. read_image_from_clipboard now falls back to reading the raw CF_DIB clipboard bytes and decoding them through the BMP file path with an explicit bfOffBits, keeping native Windows image paste off the PowerShell bridge. Fixes #3426 --- Cargo.lock | 1 + Cargo.toml | 1 + crates/pi-natives/Cargo.toml | 3 +- crates/pi-natives/src/clipboard.rs | 239 ++++++++++++++++++++++++++++- packages/natives/CHANGELOG.md | 1 + 5 files changed, 242 insertions(+), 3 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 1b29c8dce..87f1ebd9f 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2969,6 +2969,7 @@ dependencies = [ "ast-grep-core", "base64", "clap", + "clipboard-win", "flume", "fontdue", "globset", diff --git a/Cargo.toml b/Cargo.toml index 88c64391b..ecb8d8d92 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -269,6 +269,7 @@ napi-derive = "3" # Terminal & PTY # ────────────────────────────────────────────────────────────────────────────── arboard = { version = "3.6.1", features = ["wayland-data-control"] } +clipboard-win = "5.4" icy_sixel = "0.5" portable-pty = "0.9" diff --git a/crates/pi-natives/Cargo.toml b/crates/pi-natives/Cargo.toml index 6495a02c8..0f230bcd6 100644 --- a/crates/pi-natives/Cargo.toml +++ b/crates/pi-natives/Cargo.toml @@ -26,7 +26,7 @@ grep-searcher.workspace = true html-to-markdown-rs.workspace = true icy_sixel.workspace = true ignore.workspace = true -image.workspace = true +image = { workspace = true, features = ["bmp"] } inferno.workspace = true memmap2.workspace = true napi.workspace = true @@ -61,6 +61,7 @@ libc.workspace = true [target.'cfg(windows)'.dependencies] windows-sys = { workspace = true, features = ["Wdk_Storage_FileSystem", "Win32_Security"] } +clipboard-win.workspace = true winreg.workspace = true [build-dependencies] diff --git a/crates/pi-natives/src/clipboard.rs b/crates/pi-natives/src/clipboard.rs index 060cae6c7..74ba55c89 100644 --- a/crates/pi-natives/src/clipboard.rs +++ b/crates/pi-natives/src/clipboard.rs @@ -30,7 +30,11 @@ fn encode_png(image: ImageData<'_>) -> Result> { let bytes = image.bytes.into_owned(); let buffer = RgbaImage::from_raw(width, height, bytes) .ok_or_else(|| Error::from_reason("Clipboard image buffer size mismatch"))?; - let capacity = width.saturating_mul(height).saturating_mul(4) as usize; + rgba_to_png(buffer) +} + +fn rgba_to_png(buffer: RgbaImage) -> Result> { + let capacity = (buffer.width().saturating_mul(buffer.height()).saturating_mul(4)) as usize; let mut output = Vec::with_capacity(capacity); DynamicImage::ImageRgba8(buffer) .write_to(&mut Cursor::new(&mut output), ImageFormat::Png) @@ -38,6 +42,84 @@ fn encode_png(image: ImageData<'_>) -> Result> { Ok(output) } +/// Decode a packed DIB clipboard payload (`CF_DIB`: a `BITMAPINFOHEADER`-family +/// header, optional bitfield masks and palette, then the pixel array) into PNG +/// bytes. +/// +/// The payload is wrapped in a synthesized `BITMAPFILEHEADER` and decoded +/// through the BMP *file* path so the explicit `bfOffBits` pins the pixel +/// offset. This matters: the header-less decode path arboard uses mis-places +/// the pixel offset for V4/V5 headers with `BI_BITFIELDS` compression (it +/// skips 12 trailing mask bytes that those headers embed instead), which is +/// why Qt-based screenshot tools (PixPin, Snipaste, ...) fail through arboard +/// in the first place (#3426). +#[cfg_attr( + not(windows), + allow( + dead_code, + reason = "reached only by the Windows clipboard fallback; kept target-independent so unit tests cover it on every host" + ) +)] +fn dib_to_png(dib: &[u8]) -> Result> { + const FILE_HEADER_SIZE: u64 = 14; + const INFO_HEADER_SIZE: u64 = 40; + const BI_BITFIELDS: u32 = 3; + + if dib.len() < INFO_HEADER_SIZE as usize { + return Err(Error::from_reason("Clipboard DIB shorter than BITMAPINFOHEADER")); + } + let u32_at = + |at: usize| u32::from_le_bytes(dib[at..at + 4].try_into().expect("bounds checked above")); + let header_size = u64::from(u32_at(0)); + if header_size < INFO_HEADER_SIZE || header_size > dib.len() as u64 { + return Err(Error::from_reason("Clipboard DIB header size out of range")); + } + let bit_count = u16::from_le_bytes([dib[14], dib[15]]); + let compression = u32_at(16); + let colors_used = u64::from(u32_at(32)); + + // A plain BITMAPINFOHEADER with BI_BITFIELDS is trailed by three DWORD + // masks; larger (V2..V5) headers embed the masks in the header itself. + let mask_bytes: u64 = + if header_size == INFO_HEADER_SIZE && compression == BI_BITFIELDS { 12 } else { 0 }; + let palette_entries: u64 = if colors_used != 0 { + colors_used + } else if bit_count <= 8 { + 1u64 << bit_count + } else { + 0 + }; + let pixel_offset = + u32::try_from(FILE_HEADER_SIZE + header_size + mask_bytes + palette_entries * 4) + .map_err(|_| Error::from_reason("Clipboard DIB layout overflow"))?; + let file_size = u32::try_from(FILE_HEADER_SIZE + dib.len() as u64) + .map_err(|_| Error::from_reason("Clipboard DIB too large"))?; + + let mut bmp = Vec::with_capacity(FILE_HEADER_SIZE as usize + dib.len()); + bmp.extend_from_slice(b"BM"); + bmp.extend_from_slice(&file_size.to_le_bytes()); + bmp.extend_from_slice(&0u32.to_le_bytes()); + bmp.extend_from_slice(&pixel_offset.to_le_bytes()); + bmp.extend_from_slice(dib); + + let decoded = image::load_from_memory_with_format(&bmp, ImageFormat::Bmp) + .map_err(|err| Error::from_reason(format!("Failed to decode clipboard DIB: {err}")))?; + rgba_to_png(decoded.into_rgba8()) +} + +/// Read the raw `CF_DIB` bytes from the Windows clipboard. +/// +/// Windows synthesizes `CF_DIB` from whatever bitmap formats are present, so +/// it is available whenever the clipboard holds any image at all. +#[cfg(windows)] +fn read_raw_cf_dib() -> Option> { + let clip = clipboard_win::Clipboard::new_attempts(10).ok()?; + let mut dib = Vec::new(); + clipboard_win::raw::get_vec(clipboard_win::formats::CF_DIB, &mut dib).ok()?; + drop(clip); + (!dib.is_empty()).then_some(dib) +} + /// Copy plain text to the system clipboard. /// /// # Parameters @@ -120,7 +202,160 @@ pub fn read_image_from_clipboard() -> task::Promise> { })) }, Err(ClipboardError::ContentNotAvailable) => Ok(None), - Err(err) => Err(Error::from_reason(format!("Failed to read clipboard image: {err}"))), + Err(err) => { + // arboard rejects the CF_DIBV5 payloads Qt-based screenshot + // tools (PixPin, Snipaste, ...) produce; decode the raw CF_DIB + // ourselves before surfacing the error (#3426). A fallback + // decode failure keeps the original arboard error. + #[cfg(windows)] + if let Some(bytes) = read_raw_cf_dib().and_then(|dib| dib_to_png(&dib).ok()) { + return Ok(Some(ClipboardImage { + data: Uint8Array::from(bytes), + mime_type: "image/png".to_string(), + })); + } + Err(Error::from_reason(format!("Failed to read clipboard image: {err}"))) + }, } }) } + +#[cfg(test)] +mod tests { + use super::dib_to_png; + + fn push32(v: u32, out: &mut Vec) { + out.extend_from_slice(&v.to_le_bytes()); + } + + fn push16(v: u16, out: &mut Vec) { + out.extend_from_slice(&v.to_le_bytes()); + } + + /// 2x2 bottom-up BGRA pixel array: memory rows are [red, green] (bottom) + /// then [blue, white] (top), all with alpha 0xff. + const PIXELS_2X2: [u8; 16] = [ + 0x00, 0x00, 0xff, 0xff, // (0,1) red + 0x00, 0xff, 0x00, 0xff, // (1,1) green + 0xff, 0x00, 0x00, 0xff, // (0,0) blue + 0xff, 0xff, 0xff, 0xff, // (1,0) white + ]; + + /// `CF_DIB` as Qt's clipboard writer emits it for 32-bit content: a plain + /// `BITMAPINFOHEADER` with `BI_BITFIELDS` compression and three DWORD + /// masks between header and pixels. + fn qt_cf_dib(width: u32, height: u32, pixels_bgra: &[u8], compression: u32) -> Vec { + let mut d = Vec::with_capacity(52 + pixels_bgra.len()); + push32(40, &mut d); // biSize + push32(width, &mut d); + push32(height, &mut d); // positive: bottom-up + push16(1, &mut d); // biPlanes + push16(32, &mut d); // biBitCount + push32(compression, &mut d); + push32(pixels_bgra.len() as u32, &mut d); // biSizeImage + push32(0, &mut d); // biXPelsPerMeter + push32(0, &mut d); // biYPelsPerMeter + push32(0, &mut d); // biClrUsed + push32(0, &mut d); // biClrImportant + if compression == 3 { + push32(0x00ff_0000, &mut d); // red mask + push32(0x0000_ff00, &mut d); // green mask + push32(0x0000_00ff, &mut d); // blue mask + } + d.extend_from_slice(pixels_bgra); + d + } + + /// `CF_DIBV5` as PixPin (Qt) places it, after arboard's + /// `maybe_tweak_header` rewrite: a 124-byte `BITMAPV5HEADER` carrying + /// `BI_BITFIELDS` compression with the BGRA masks embedded in the header + /// and pixels immediately after it. This is the exact buffer shape that + /// arboard's header-less BMP decode rejects with `ConversionFailure` + /// (issue #3426); the file-header wrap must decode it. + fn pixpin_dibv5_tweaked(width: u32, height: u32, pixels_bgra: &[u8]) -> Vec { + let mut d = Vec::with_capacity(124 + pixels_bgra.len()); + push32(124, &mut d); // bV5Size + push32(width, &mut d); + push32(height, &mut d); + push16(1, &mut d); // bV5Planes + push16(32, &mut d); // bV5BitCount + push32(3, &mut d); // bV5Compression = BI_BITFIELDS (arboard-tweaked) + push32(0, &mut d); // bV5SizeImage + push32(0, &mut d); // bV5XPelsPerMeter + push32(0, &mut d); // bV5YPelsPerMeter + push32(0, &mut d); // bV5ClrUsed + push32(0, &mut d); // bV5ClrImportant + push32(0x00ff_0000, &mut d); // bV5RedMask + push32(0x0000_ff00, &mut d); // bV5GreenMask + push32(0x0000_00ff, &mut d); // bV5BlueMask + push32(0xff00_0000, &mut d); // bV5AlphaMask + push32(0x7352_4742, &mut d); // bV5CSType = LCS_sRGB + d.extend_from_slice(&[0u8; 36]); // bV5Endpoints + push32(0, &mut d); // bV5GammaRed + push32(0, &mut d); // bV5GammaGreen + push32(0, &mut d); // bV5GammaBlue + push32(4, &mut d); // bV5Intent = LCS_GM_IMAGES + push32(0, &mut d); // bV5ProfileData + push32(0, &mut d); // bV5ProfileSize + push32(0, &mut d); // bV5Reserved + assert_eq!(d.len(), 124); + d.extend_from_slice(pixels_bgra); + d + } + + fn decode_pixels(png: &[u8]) -> (u32, u32, Vec<[u8; 4]>) { + let img = image::load_from_memory(png).expect("fallback output must be valid PNG"); + let rgba = img.into_rgba8(); + let (w, h) = rgba.dimensions(); + let px = rgba.pixels().map(|p| p.0).collect(); + (w, h, px) + } + + const RED: [u8; 4] = [255, 0, 0, 255]; + const GREEN: [u8; 4] = [0, 255, 0, 255]; + const BLUE: [u8; 4] = [0, 0, 255, 255]; + const WHITE: [u8; 4] = [255, 255, 255, 255]; + + #[test] + fn decodes_qt_cf_dib_with_bitfields_masks() { + let dib = qt_cf_dib(2, 2, &PIXELS_2X2, 3); + let png = dib_to_png(&dib).expect("BI_BITFIELDS CF_DIB must decode"); + let (w, h, px) = decode_pixels(&png); + assert_eq!((w, h), (2, 2)); + // Row order flipped versus the bottom-up pixel array; BGRA -> RGBA. + assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]); + } + + #[test] + fn decodes_pixpin_dibv5_payload_that_arboard_rejects() { + let dib = pixpin_dibv5_tweaked(2, 2, &PIXELS_2X2); + let png = dib_to_png(&dib).expect("V5 BI_BITFIELDS DIB must decode"); + let (w, h, px) = decode_pixels(&png); + assert_eq!((w, h), (2, 2)); + assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]); + } + + #[test] + fn decodes_plain_bi_rgb_dib() { + // The common "copy image" payload: BI_RGB, 32-bit, no masks. The + // fourth byte is unused per the DIB contract — zero it to prove the + // decode still yields opaque pixels. + let mut pixels = PIXELS_2X2; + for alpha in pixels.iter_mut().skip(3).step_by(4) { + *alpha = 0; + } + let dib = qt_cf_dib(2, 2, &pixels, 0); + let png = dib_to_png(&dib).expect("BI_RGB CF_DIB must decode"); + let (w, h, px) = decode_pixels(&png); + assert_eq!((w, h), (2, 2)); + assert_eq!(px, vec![BLUE, WHITE, RED, GREEN]); + } + + #[test] + fn rejects_malformed_dib() { + assert!(dib_to_png(&[0u8; 12]).is_err(), "short buffer must not decode"); + let mut oversized_header = qt_cf_dib(2, 2, &PIXELS_2X2, 3); + oversized_header[0..4].copy_from_slice(&0xffff_ffffu32.to_le_bytes()); + assert!(dib_to_png(&oversized_header).is_err(), "header size beyond buffer must not decode"); + } +} diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index 58027bbd9..d65f4e4cc 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed unbounded memory growth in the native bash output bridge when a command produces output faster than the JS event loop consumes it: the shell streaming path now uses a bounded chunk queue with real backpressure (pipe readers park until the JS callback catches up, parking the child on its pipe) instead of buffering the entire surplus in memory. No output is dropped — the rolling tail view, `[raw output: artifact://…]` lossless capture, and byte accounting are unaffected ([#4078](https://github.com/can1357/oh-my-pi/issues/4078)). +- Fixed `readImageFromClipboard` on Windows failing with "could not be converted to the appropriate format" for screenshots taken by Qt-based tools such as PixPin and Snipaste. arboard hands their `CF_DIBV5` payload (`BI_RGB` plus an alpha mask, rewritten to `BI_BITFIELDS`) to a header-less BMP decode that mis-places the pixel offset for V4/V5 bitfield headers; the native reader now falls back to decoding the raw `CF_DIB` clipboard bytes directly, so image paste no longer depends on the PowerShell bridge. ([#3426](https://github.com/can1357/oh-my-pi/issues/3426)) ## [16.3.12] - 2026-07-08 From cab5c4b62aaf7594599030eaae6725e124f5075a Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:31:36 +0200 Subject: [PATCH 27/59] test(coding-agent): covered PowerShell fallback when native read throws Adopted the dispatch regression test from PR #3427: a native Windows image conversion failure must fall through to the PowerShell GetImage() bridge. The dispatch behavior itself already landed in d718d54a33. Refs #3426 --- .../coding-agent/test/utils/clipboard.test.ts | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/packages/coding-agent/test/utils/clipboard.test.ts b/packages/coding-agent/test/utils/clipboard.test.ts index 5e7b5f53d..b9598bce0 100644 --- a/packages/coding-agent/test/utils/clipboard.test.ts +++ b/packages/coding-agent/test/utils/clipboard.test.ts @@ -166,6 +166,23 @@ describe("readImageFromClipboard dispatch", () => { expect(calls[0]?.cmd).toContain("-Sta"); }); + it("falls back to PowerShell when native Windows image conversion fails", async () => { + setPlatform("win32"); + const calls: SpawnCall[] = []; + spyPowershell(calls, RED_1X1_PNG_BASE64); + vi.spyOn(native, "readImageFromClipboard").mockRejectedValue( + new Error("The clipboard image could not be converted to the appropriate format."), + ); + + const image = await readImageFromClipboard(); + + expect(calls).toHaveLength(1); + expect(calls[0]?.cmd[0]).toBe("powershell.exe"); + expect(calls[0]?.cmd).toContain("-Sta"); + expect(image?.mimeType).toBe("image/png"); + expect(Array.from(image!.data.subarray(0, 8))).toEqual([0x89, 0x50, 0x4e, 0x47, 0x0d, 0x0a, 0x1a, 0x0a]); + }); + it("delegates straight to the native bridge on non-WSL linux with a display", async () => { setPlatform("linux"); process.env.DISPLAY = ":0"; From efc90d26b816b7ba55532bb7a45a371676cc4527 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:34:06 +0200 Subject: [PATCH 28/59] style: applied biome fixes to integrated issue fixes --- packages/ai/test/cursor-exec-handlers.test.ts | 4 ++-- packages/coding-agent/src/config/model-registry.ts | 7 ++++++- .../src/modes/components/tool-execution.ts | 3 ++- .../coding-agent/test/bundled-agent-parsing.test.ts | 5 +---- .../test/internal-urls/memory-protocol.test.ts | 10 +++++++++- packages/coding-agent/test/rpc-skill-command.test.ts | 8 +++++++- 6 files changed, 27 insertions(+), 10 deletions(-) diff --git a/packages/ai/test/cursor-exec-handlers.test.ts b/packages/ai/test/cursor-exec-handlers.test.ts index c79b0e12d..5063e81ec 100644 --- a/packages/ai/test/cursor-exec-handlers.test.ts +++ b/packages/ai/test/cursor-exec-handlers.test.ts @@ -1,5 +1,5 @@ -import { create } from "@bufbuild/protobuf"; import { describe, expect, it } from "bun:test"; +import { create } from "@bufbuild/protobuf"; import { type BlockState, buildCursorHistoryForTest, @@ -10,7 +10,7 @@ import { streamCursor, type ToolCallState, } from "@oh-my-pi/pi-ai/providers/cursor"; -import { setCursorProviderModule, streamCursor as lazyStreamCursor } from "@oh-my-pi/pi-ai/providers/register-builtins"; +import { streamCursor as lazyStreamCursor, setCursorProviderModule } from "@oh-my-pi/pi-ai/providers/register-builtins"; import type { AssistantMessage, Context, CursorExecHandlers, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 44d11ffcf..82c4e0545 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -1604,7 +1604,12 @@ export class ModelRegistry { if (strategy === "online-if-uncached") { // Mirror shouldFetchRemoteSources: built-in managers use the catalog's // default TTL, so only refresh when the manager will actually fetch. - const cache = readModelCache(cacheProviderId, BUILT_IN_DISCOVERY_CACHE_TTL_MS, Date.now, this.#cacheDbPath); + const cache = readModelCache( + cacheProviderId, + BUILT_IN_DISCOVERY_CACHE_TTL_MS, + Date.now, + this.#cacheDbPath, + ); const cacheAgeMs = cache ? Date.now() - cache.updatedAt : Number.POSITIVE_INFINITY; if (cache?.fresh && (cache.authoritative || cacheAgeMs < BUILT_IN_DISCOVERY_NON_AUTHORITATIVE_RETRY_MS)) { return peekedKey; diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index a958a233c..de9f9c483 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -828,7 +828,8 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac if (this.#result !== undefined) return false; const toolValue = (this.#tool as { forceFirstResultViewportRepaint?: FirstResultViewportRepaint } | undefined) ?.forceFirstResultViewportRepaint; - const value = toolValue !== undefined ? toolValue : toolRenderers[this.#toolName]?.forceFirstResultViewportRepaint; + const value = + toolValue !== undefined ? toolValue : toolRenderers[this.#toolName]?.forceFirstResultViewportRepaint; if (typeof value === "function") return value(this.#args, this.#renderState); return value === true; } diff --git a/packages/coding-agent/test/bundled-agent-parsing.test.ts b/packages/coding-agent/test/bundled-agent-parsing.test.ts index e66f41f0e..a4da8e57d 100644 --- a/packages/coding-agent/test/bundled-agent-parsing.test.ts +++ b/packages/coding-agent/test/bundled-agent-parsing.test.ts @@ -1,10 +1,7 @@ import { describe, expect, it } from "bun:test"; import { Effort } from "@oh-my-pi/pi-ai"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; -import { - resolveAgentModelPatterns, - resolveModelOverride, -} from "@oh-my-pi/pi-coding-agent/config/model-resolver"; +import { resolveAgentModelPatterns, resolveModelOverride } from "@oh-my-pi/pi-coding-agent/config/model-resolver"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { getBundledAgent } from "@oh-my-pi/pi-coding-agent/task/agents"; diff --git a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts index 331ad66ce..a0efc7bee 100644 --- a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts +++ b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts @@ -245,7 +245,15 @@ describe("MemoryProtocolHandler — mnemopi bridge (issue #4443)", () => { .prepare( "INSERT INTO facts (fact_id, session_id, subject, predicate, object, timestamp, confidence) VALUES (?, ?, ?, ?, ?, ?, ?)", ) - .run("0473bbdb8da6df92", beam.sessionId, "Glab", "works-without", "mise prefix", "2026-07-01T00:00:00.000Z", 0.9); + .run( + "0473bbdb8da6df92", + beam.sessionId, + "Glab", + "works-without", + "mise prefix", + "2026-07-01T00:00:00.000Z", + 0.9, + ); const router = InternalUrlRouter.instance(); const resource = await router.resolve("memory://0473bbdb8da6df92"); diff --git a/packages/coding-agent/test/rpc-skill-command.test.ts b/packages/coding-agent/test/rpc-skill-command.test.ts index 02795389b..1100ace66 100644 --- a/packages/coding-agent/test/rpc-skill-command.test.ts +++ b/packages/coding-agent/test/rpc-skill-command.test.ts @@ -60,7 +60,13 @@ describe("tryRunRpcSkillCommand", () => { { skillsSettings: { enableSkillCommands: true }, skills: [ - { name: "reviewer", description: "Review code", filePath: skillPath, baseDir: dir, source: "project" }, + { + name: "reviewer", + description: "Review code", + filePath: skillPath, + baseDir: dir, + source: "project", + }, ], async promptCustomMessage(nextMessage, nextOptions) { expect(nextMessage.customType).toBe(SKILL_PROMPT_MESSAGE_TYPE); From 7c560c71511d7707ddb918136ec3d0f67b6e8553 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:35:07 +0200 Subject: [PATCH 29/59] fix(acp): flushed final assistant text lost to agent_end race The assistant message_end fan-out is fire-and-forget in the session layer and can be parked on extension delivery while agent_end is flushed through #endInFlight, so agent_end can overtake it. #finishPrompt then unsubscribes the prompt turn and the mapAssistantMessageEnd fallback never runs: an ACP client that only received agent_thought_chunk updates (thinking streamed, text arrived only on the trailing message) stays stuck on the thinking block with no visible answer. On agent_end, emit the last assistant message's text before resolving the prompt when live-message progress shows no text was ever delivered, and defer the live-state reset past that flush so a late message_end cannot resurrect fresh progress and double-emit. Fixes #4902 --- .../coding-agent/src/modes/acp/acp-agent.ts | 59 +++++++++- .../src/modes/acp/acp-event-mapper.ts | 2 +- packages/coding-agent/test/acp-agent.test.ts | 109 ++++++++++++++++++ 3 files changed, 168 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/src/modes/acp/acp-agent.ts b/packages/coding-agent/src/modes/acp/acp-agent.ts index efdfaa45b..6d76f2721 100644 --- a/packages/coding-agent/src/modes/acp/acp-agent.ts +++ b/packages/coding-agent/src/modes/acp/acp-agent.ts @@ -84,6 +84,7 @@ import { canonicalizeMessage } from "../../utils/thinking-display"; import { createAcpClientBridge } from "./acp-client-bridge"; import { buildToolCallStartUpdate, + extractAssistantMessageText, mapAgentSessionEventToAcpSessionUpdates, normalizeReplayToolArguments, } from "./acp-event-mapper"; @@ -1214,8 +1215,11 @@ export class AcpAgent implements Agent { this.#clearLiveAssistantMessageAfterEvent(record, event); if (event.type === "agent_end") { + await this.#flushMissedFinalAssistantText(record, event); await this.#emitEndOfTurnUpdates(record); await this.#waitForAcpPromptIdle(record); + record.liveMessageId = undefined; + record.liveMessageProgress = undefined; this.#finishPrompt(record, { stopReason: this.#resolveStopReason(event, promptTurn.cancelRequested), usage: this.#buildTurnUsage(promptTurn.usageBaseline, record.session.sessionManager.getUsageStatistics()), @@ -1223,6 +1227,51 @@ export class AcpAgent implements Agent { } } + /** + * Deliver the final visible answer when the assistant `message_end` never + * reached this prompt turn's subscription. Session event handlers are + * fire-and-forget (`Agent#emit` does not await async listeners), and + * `agent_end` is flushed through the session's `#endInFlight` path while the + * assistant `message_end` fan-out can still be parked on extension delivery — + * so `agent_end` can overtake `message_end`. Once the turn finishes, + * `#finishPrompt` unsubscribes and the fallback text emission in + * `mapAssistantMessageEnd` is lost for good: a client that only received + * `agent_thought_chunk`s stays stuck on the thinking block (#4902). The live + * message progress records whether visible text ever reached the client; if + * it has not, emit the last assistant message's text before the prompt + * resolves. A `message_end` that lands during the end-of-turn waits still + * takes the normal mapper path and sees `textEmitted` already set, so the + * answer is delivered exactly once. + */ + async #flushMissedFinalAssistantText( + record: ManagedSessionRecord, + event: Extract, + ): Promise { + const progress = record.liveMessageProgress; + if (!progress || progress.textEmitted) { + return; + } + const lastAssistant = [...event.messages] + .reverse() + .find((message): message is AssistantMessage => message.role === "assistant"); + if (!lastAssistant) { + return; + } + const text = extractAssistantMessageText(lastAssistant); + if (text.length === 0) { + return; + } + progress.textEmitted = true; + await this.#connection.sessionUpdate({ + sessionId: record.session.sessionId, + update: { + sessionUpdate: "agent_message_chunk", + content: { type: "text", text }, + messageId: record.liveMessageId, + }, + }); + } + async #waitForAcpPromptIdle(record: ManagedSessionRecord): Promise { for (let pass = 0; pass < ACP_ASYNC_DELIVERY_DRAIN_MAX_PASSES; pass++) { await record.session.waitForIdle(); @@ -1248,8 +1297,16 @@ export class AcpAgent implements Agent { } } + /** + * Reset live-message tracking once the assistant `message_end` is handled. + * The `agent_end` reset happens inside the `agent_end` branch of + * `#handlePromptEvent` — after `#flushMissedFinalAssistantText` — so a + * `message_end` that arrives during the end-of-turn waits maps against the + * real progress instead of resurrecting a fresh one (which would double-emit + * the final answer). + */ #clearLiveAssistantMessageAfterEvent(record: ManagedSessionRecord, event: AgentSessionEvent): void { - if ((event.type === "message_end" && event.message.role === "assistant") || event.type === "agent_end") { + if (event.type === "message_end" && event.message.role === "assistant") { record.liveMessageId = undefined; record.liveMessageProgress = undefined; } diff --git a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts index bde460d11..57f0cd388 100644 --- a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts +++ b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts @@ -922,7 +922,7 @@ function isTerminalOnlyDetails(value: unknown): boolean { return content === undefined || (Array.isArray(content) && content.length === 0); } -function extractAssistantMessageText(value: unknown): string { +export function extractAssistantMessageText(value: unknown): string { if (typeof value !== "object" || value === null || !("content" in value)) { return ""; } diff --git a/packages/coding-agent/test/acp-agent.test.ts b/packages/coding-agent/test/acp-agent.test.ts index 04c4b44b2..7e9554119 100644 --- a/packages/coding-agent/test/acp-agent.test.ts +++ b/packages/coding-agent/test/acp-agent.test.ts @@ -1018,6 +1018,115 @@ describe("ACP agent", () => { await Bun.sleep(0); }); + it("delivers the final visible answer when agent_end overtakes the assistant message_end (#4902)", async () => { + const harness = await createHarness(); + const created = await harness.agent.newSession({ cwd: harness.cwdA, mcpServers: [] }); + const session = harness.findSession(created.sessionId); + if (!session) throw new Error("session not registered"); + + // Live turn as observed through the prompt subscription when the + // fire-and-forget assistant message_end handler loses the race against + // the agent_end flush: thinking streams, then the turn ends. No + // text_delta and no message_end ever reach this subscriber — the final + // text exists only on the agent_end payload. + const assistantMessage = makeAssistantMessage("Final visible answer.", "Considering the greeting."); + session.prompt = async (text: string): Promise => { + session.promptCalls.push(text); + session.isStreaming = true; + for (const listener of session.listeners()) { + listener({ + type: "message_update", + message: assistantMessage, + assistantMessageEvent: { type: "thinking_delta", delta: "Considering the greeting." }, + } as AgentSessionEvent); + } + session.sessionManager.appendMessage(assistantMessage); + for (const listener of session.listeners()) { + listener({ type: "agent_end", messages: [assistantMessage] } as AgentSessionEvent); + } + session.isStreaming = false; + return true; + }; + + const response = await harness.agent.prompt({ + sessionId: created.sessionId, + prompt: [{ type: "text", text: "Say hello" }], + }); + expectAcpStructure(zPromptResponse, response); + expect(response.stopReason).toBe("end_turn"); + + const chunks = harness.updates.filter(update => update.sessionId === created.sessionId); + const thoughtChunks = chunks.filter(update => update.update.sessionUpdate === "agent_thought_chunk"); + const messageChunks = chunks.filter(update => update.update.sessionUpdate === "agent_message_chunk"); + expect(thoughtChunks).toHaveLength(1); + // The visible answer must reach the client exactly once even though the + // assistant message_end never arrived on this subscription. + expect(messageChunks).toHaveLength(1); + expect(messageChunks[0]?.update).toEqual( + expect.objectContaining({ + sessionUpdate: "agent_message_chunk", + content: { type: "text", text: "Final visible answer." }, + }), + ); + // Flushed answer belongs to the same live message as the thought chunk. + expect(getChunkMessageId(messageChunks[0]!)).toBe(getChunkMessageId(thoughtChunks[0]!)!); + expectAcpNotifications(harness.updates); + + harness.abortController.abort(); + await Bun.sleep(0); + }); + + it("does not duplicate the final answer when the assistant message_end arrives before agent_end", async () => { + // Companion to the #4902 regression: when message_end IS delivered, its + // fallback emission wins and the agent_end flush must stay silent. + const harness = await createHarness(); + const created = await harness.agent.newSession({ cwd: harness.cwdA, mcpServers: [] }); + const session = harness.findSession(created.sessionId); + if (!session) throw new Error("session not registered"); + + const assistantMessage = makeAssistantMessage("Composed offline.", "quiet planning"); + session.prompt = async (text: string): Promise => { + session.promptCalls.push(text); + session.isStreaming = true; + for (const listener of session.listeners()) { + listener({ + type: "message_update", + message: assistantMessage, + assistantMessageEvent: { type: "thinking_delta", delta: "quiet planning" }, + } as AgentSessionEvent); + } + for (const listener of session.listeners()) { + listener({ type: "message_end", message: assistantMessage } as AgentSessionEvent); + } + session.sessionManager.appendMessage(assistantMessage); + for (const listener of session.listeners()) { + listener({ type: "agent_end", messages: [assistantMessage] } as AgentSessionEvent); + } + session.isStreaming = false; + return true; + }; + + const response = await harness.agent.prompt({ + sessionId: created.sessionId, + prompt: [{ type: "text", text: "Say hello" }], + }); + expectAcpStructure(zPromptResponse, response); + + const messageChunks = harness.updates.filter( + update => update.sessionId === created.sessionId && update.update.sessionUpdate === "agent_message_chunk", + ); + expect(messageChunks).toHaveLength(1); + expect(messageChunks[0]?.update).toEqual( + expect.objectContaining({ + content: { type: "text", text: "Composed offline." }, + }), + ); + expectAcpNotifications(harness.updates); + + harness.abortController.abort(); + await Bun.sleep(0); + }); + it("replays assistant tool calls and matching results without duplicating the start", async () => { const harness = await createHarness(); const stored = new FakeAgentSession(harness.cwdA); From b04cb7059214d965cc887e4f727384e74ef7757d Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:36:50 +0200 Subject: [PATCH 30/59] style: applied rustfmt to integrated native fixes --- crates/pi-natives/src/clipboard.rs | 15 +++++++++++---- crates/pi-natives/src/shell.rs | 4 +++- 2 files changed, 14 insertions(+), 5 deletions(-) diff --git a/crates/pi-natives/src/clipboard.rs b/crates/pi-natives/src/clipboard.rs index 74ba55c89..c60a6f44e 100644 --- a/crates/pi-natives/src/clipboard.rs +++ b/crates/pi-natives/src/clipboard.rs @@ -34,7 +34,10 @@ fn encode_png(image: ImageData<'_>) -> Result> { } fn rgba_to_png(buffer: RgbaImage) -> Result> { - let capacity = (buffer.width().saturating_mul(buffer.height()).saturating_mul(4)) as usize; + let capacity = (buffer + .width() + .saturating_mul(buffer.height()) + .saturating_mul(4)) as usize; let mut output = Vec::with_capacity(capacity); DynamicImage::ImageRgba8(buffer) .write_to(&mut Cursor::new(&mut output), ImageFormat::Png) @@ -57,7 +60,8 @@ fn rgba_to_png(buffer: RgbaImage) -> Result> { not(windows), allow( dead_code, - reason = "reached only by the Windows clipboard fallback; kept target-independent so unit tests cover it on every host" + reason = "reached only by the Windows clipboard fallback; kept target-independent so unit \ + tests cover it on every host" ) )] fn dib_to_png(dib: &[u8]) -> Result> { @@ -80,8 +84,11 @@ fn dib_to_png(dib: &[u8]) -> Result> { // A plain BITMAPINFOHEADER with BI_BITFIELDS is trailed by three DWORD // masks; larger (V2..V5) headers embed the masks in the header itself. - let mask_bytes: u64 = - if header_size == INFO_HEADER_SIZE && compression == BI_BITFIELDS { 12 } else { 0 }; + let mask_bytes: u64 = if header_size == INFO_HEADER_SIZE && compression == BI_BITFIELDS { + 12 + } else { + 0 + }; let palette_entries: u64 = if colors_used != 0 { colors_used } else if bit_count <= 8 { diff --git a/crates/pi-natives/src/shell.rs b/crates/pi-natives/src/shell.rs index f095abec0..e159ba6ed 100644 --- a/crates/pi-natives/src/shell.rs +++ b/crates/pi-natives/src/shell.rs @@ -411,7 +411,9 @@ mod tests { for i in 0..CHUNKS { let chunk = format!("[{i:06}]{}", "x".repeat(CHUNK_BYTES - 8)); expected.push_str(&chunk); - tx.send_async(chunk).await.expect("pump should outlive the producer"); + tx.send_async(chunk) + .await + .expect("pump should outlive the producer"); max_queued = max_queued.max(tx.len()); } (expected, max_queued) From f27596d287772708f859fc77954db4a0ef2e464a Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:37:46 +0200 Subject: [PATCH 31/59] style: satisfied clippy doc-markdown in clipboard docs --- crates/pi-natives/src/clipboard.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/pi-natives/src/clipboard.rs b/crates/pi-natives/src/clipboard.rs index c60a6f44e..167f167ff 100644 --- a/crates/pi-natives/src/clipboard.rs +++ b/crates/pi-natives/src/clipboard.rs @@ -54,7 +54,7 @@ fn rgba_to_png(buffer: RgbaImage) -> Result> { /// offset. This matters: the header-less decode path arboard uses mis-places /// the pixel offset for V4/V5 headers with `BI_BITFIELDS` compression (it /// skips 12 trailing mask bytes that those headers embed instead), which is -/// why Qt-based screenshot tools (PixPin, Snipaste, ...) fail through arboard +/// why Qt-based screenshot tools (`PixPin`, `Snipaste`, ...) fail through arboard /// in the first place (#3426). #[cfg_attr( not(windows), From 4f8c3417454a2af4236e65d599bed0eeafaa1817 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:36:02 +0200 Subject: [PATCH 32/59] test(natives): pinned bash timeout output-bridge crash contract Regression test for the WSL crash where a timed-out bash command got OMP OOM-killed: an output-heavy command (in-process 'yes | cat') with a short timeoutMs must resolve near its deadline with bounded RSS, on both executeShell and Shell.run. On the pre-fix bridge the same harness measured ~4 GiB RSS growth and minutes-late resolution (JS event loop starved by the unbounded callback flood); with the bounded backpressured bridge it resolves at the deadline with flat memory. Also added the user-facing changelog entry for the crash symptom. Fixes #4866 --- packages/natives/CHANGELOG.md | 1 + .../natives/test/issue-4866-repro.test.ts | 109 ++++++++++++++++++ 2 files changed, 110 insertions(+) create mode 100644 packages/natives/test/issue-4866-repro.test.ts diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index d65f4e4cc..2e1f8df66 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -6,6 +6,7 @@ - Fixed unbounded memory growth in the native bash output bridge when a command produces output faster than the JS event loop consumes it: the shell streaming path now uses a bounded chunk queue with real backpressure (pipe readers park until the JS callback catches up, parking the child on its pipe) instead of buffering the entire surplus in memory. No output is dropped — the rolling tail view, `[raw output: artifact://…]` lossless capture, and byte accounting are unaffected ([#4078](https://github.com/can1357/oh-my-pi/issues/4078)). - Fixed `readImageFromClipboard` on Windows failing with "could not be converted to the appropriate format" for screenshots taken by Qt-based tools such as PixPin and Snipaste. arboard hands their `CF_DIBV5` payload (`BI_RGB` plus an alpha mask, rewritten to `BI_BITFIELDS`) to a header-less BMP decode that mis-places the pixel offset for V4/V5 bitfield headers; the native reader now falls back to decoding the raw `CF_DIB` clipboard bytes directly, so image paste no longer depends on the PowerShell bridge. ([#3426](https://github.com/can1357/oh-my-pi/issues/3426)) +- Fixed OMP being killed outright (OOM on memory-capped hosts such as WSL) when an output-heavy bash command hit its timeout: the unbounded output-bridge backlog could grow by gigabytes before cancellation and starve the JS event loop far past the deadline; with the bounded backpressured bridge the run resolves at its deadline with flat memory ([#4866](https://github.com/can1357/oh-my-pi/issues/4866)). ## [16.3.12] - 2026-07-08 diff --git a/packages/natives/test/issue-4866-repro.test.ts b/packages/natives/test/issue-4866-repro.test.ts new file mode 100644 index 000000000..b26e97fdb --- /dev/null +++ b/packages/natives/test/issue-4866-repro.test.ts @@ -0,0 +1,109 @@ +/** + * Regression for https://github.com/can1357/oh-my-pi/issues/4866. + * + * "When bash command times out, it exits/crashes OMP as a whole" (WSL). + * + * Root cause: the native shell output bridge (`bridge_chunks` in + * `crates/pi-natives/src/shell.rs` + `emit_chunk` in + * `crates/pi-shell/src/shell.rs`) queued decoded output chunks into an + * unbounded cross-thread channel and fired the JS threadsafe function + * non-blocking, with no backpressure. A producer outrunning the JS consumer + * (`yes | cat` runs as in-process uutils builtins at memory speed; any + * output-heavy long task qualifies) ballooned the native queue by gigabytes + * before the timeout fired, and the callback flood then kept the JS event + * loop saturated so the deadline machinery ran tens of seconds late. + * Measured on the pre-fix baseline (macOS arm64): a `timeoutMs: 1500` run + * through the bash executor resolved after ~30-36 s having forwarded ~6.9 GB, + * with process RSS pinned at ~7 GB. On WSL's memory-capped VM that backlog + * trips the Linux OOM killer, which SIGKILLs the whole OMP process — the + * reported "crashes OMP as a whole". + * + * This test models the real consumer (OutputSink sanitize/tail/render work) + * with a deliberately slow `onChunk` (~1 ms per callback) and pins the fixed + * contract for both the one-shot (`executeShell`) and persistent-session + * (`Shell.run`) paths: + * 1. The run resolves near its deadline (raced against a generous window) + * instead of being dragged out by an unbounded backlog drain. On the + * pre-fix bridge the drain alone needs minutes (tens of thousands of + * queued 64 KiB batches through a ~1 ms consumer). + * 2. Native memory stays bounded: RSS growth over the run stays far under + * the gigabytes the unbounded queue accumulated (bounded(64) queue × + * 64 KiB batches plus JS churn). + * 3. The run still reports `timedOut`, so timeout annotation and session + * quarantine behave as before. + * + * Bounds carry >4x headroom on both sides of every threshold (fixed path + * measured: resolve ≈1 s, RSS delta ≈60 MiB; baseline: unresolved at 6 s, + * RSS delta ≥2 GiB), so the test stays robust on slow CI hosts while the + * failure mode overshoots by orders of magnitude. + */ +import { describe, expect, it } from "bun:test"; +import { executeShell, Shell, type ShellRunResult } from "../native/index.js"; + +/** `yes` and `cat` are in-process uutils builtins: output is produced at + * memory speed, which is what made the unbounded bridge lethal. */ +const FAST_PRODUCER = "yes issue-4866-crash-line | cat"; +const TIMEOUT_MS = 800; +/** Window the timed-out run must resolve within (fixed path: ~1 s; pre-fix + * baseline is still draining its multi-GB backlog minutes later). */ +const RESOLVE_WINDOW_MS = 8_000; +/** RSS growth budget. Fixed path: tens of MiB. Pre-fix: multiple GiB. */ +const MAX_RSS_DELTA_BYTES = 512 * 1024 * 1024; +/** Per-callback consumer cost emulating OutputSink/TUI work. */ +const CONSUMER_STALL_MS = 1; +const TEST_BUDGET_MS = 60_000; + +const posixIt = process.platform === "win32" ? it.skip : it; + +// Real-clock integration test (ts-no-test-timers exception): the run under +// test is a native tokio shell execution behind the N-API boundary — fake JS +// timers cannot advance the native runtime's clock, and the defect being +// pinned is precisely a real-time liveness failure (the JS event loop and +// deadline machinery starved by the callback flood). The stall emulates +// synchronous per-callback consumer cost (CPU work, not scheduling), and the +// resolve window is a liveness bound, not a synchronization guess. +async function runTimedOutFastProducer( + run: (onChunk: (err: Error | null, chunk: string) => void) => Promise, +): Promise { + const rssBefore = process.memoryUsage.rss(); + const slowConsumer = (_err: Error | null, chunk: string) => { + if (chunk) Bun.sleepSync(CONSUMER_STALL_MS); + }; + + const settled = run(slowConsumer).then(result => ({ done: true as const, result })); + const raced = await Promise.race([settled, Bun.sleep(RESOLVE_WINDOW_MS).then(() => ({ done: false as const }))]); + const rssDelta = process.memoryUsage.rss() - rssBefore; + + // (2) Bounded native memory — the unbounded bridge queued gigabytes here. + expect(rssDelta).toBeLessThan(MAX_RSS_DELTA_BYTES); + // (1) Timely resolution — the unbounded bridge dragged the run out for + // minutes past its deadline. + expect(raced.done).toBe(true); + if (raced.done) { + // (3) Timeout is still reported as such. + expect(raced.result.timedOut).toBe(true); + } +} + +describe("issue 4866: bash timeout must not flood the output bridge", () => { + posixIt( + "one-shot executeShell: fast producer with slow consumer times out near its deadline with bounded memory", + async () => { + await runTimedOutFastProducer(onChunk => + executeShell({ command: FAST_PRODUCER, timeoutMs: TIMEOUT_MS }, onChunk), + ); + }, + TEST_BUDGET_MS, + ); + + posixIt( + "persistent Shell.run: fast producer with slow consumer times out near its deadline with bounded memory", + async () => { + const shell = new Shell(); + await runTimedOutFastProducer(onChunk => + shell.run({ command: FAST_PRODUCER, timeoutMs: TIMEOUT_MS }, onChunk), + ); + }, + TEST_BUDGET_MS, + ); +}); From 4ff9dfc12457e8a844b57cc4a04ab1cf0f6a06b6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:39:56 +0200 Subject: [PATCH 33/59] style: rewrapped clipboard tool-name docs --- crates/pi-natives/src/clipboard.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/crates/pi-natives/src/clipboard.rs b/crates/pi-natives/src/clipboard.rs index 167f167ff..c7b2251b2 100644 --- a/crates/pi-natives/src/clipboard.rs +++ b/crates/pi-natives/src/clipboard.rs @@ -54,8 +54,8 @@ fn rgba_to_png(buffer: RgbaImage) -> Result> { /// offset. This matters: the header-less decode path arboard uses mis-places /// the pixel offset for V4/V5 headers with `BI_BITFIELDS` compression (it /// skips 12 trailing mask bytes that those headers embed instead), which is -/// why Qt-based screenshot tools (`PixPin`, `Snipaste`, ...) fail through arboard -/// in the first place (#3426). +/// why Qt-based screenshot tools (`PixPin`, `Snipaste`, ...) fail through +/// arboard in the first place (#3426). #[cfg_attr( not(windows), allow( From 72d4e03620b559d5e6c1a3e14338c78d0e8d864c Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 18:46:27 +0200 Subject: [PATCH 34/59] docs: normalized unreleased changelog entries for integrated fixes --- packages/ai/CHANGELOG.md | 5 +---- packages/coding-agent/CHANGELOG.md | 5 ++--- packages/tui/CHANGELOG.md | 4 +--- 3 files changed, 4 insertions(+), 10 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 3eee5357a..851cfbaad 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,10 +2,6 @@ ## [Unreleased] -### Fixed - -- Fixed the generic lazy-stream idle watchdog aborting healthy `cursor-agent` streams with "Provider stream stalled while waiting for the next event" while a Cursor exec-channel local tool (shell/read/grep/write/MCP/…) legitimately ran longer than the idle budget. Provider streams now advertise consumer-side local work in flight and the watchdog slides its deadline instead of aborting; genuinely silent streams still time out. ([#4593](https://github.com/can1357/oh-my-pi/issues/4593)) - ### Changed - Changed the xAI Grok OAuth (`xai-oauth`) provider to use manual code-paste login by default. `/login` now accepts a pasted authorization code or full `http://127.0.0.1:56121/callback?code=...` redirect URL without starting a local callback listener ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). @@ -13,6 +9,7 @@ ### Fixed +- Fixed the generic lazy-stream idle watchdog aborting healthy `cursor-agent` streams with "Provider stream stalled while waiting for the next event" while a Cursor exec-channel local tool (shell/read/grep/write/MCP/…) legitimately ran longer than the idle budget. Provider streams now advertise consumer-side local work in flight and the watchdog slides its deadline instead of aborting; genuinely silent streams still time out. ([#4593](https://github.com/can1357/oh-my-pi/issues/4593)) - Fixed OpenAI Codex/Responses reasoning streams so streamed thinking content is preserved when the final `output_item.done` reconstructs to an empty summary ([#4918](https://github.com/can1357/oh-my-pi/issues/4918)). - Fixed Anthropic streams hanging forever when generation wedges mid-stream (notably long `write` tool calls on Opus 4.8 high/xhigh) while the server keeps sending `ping` keepalives: pings now extend the idle watchdog only within a bounded window (3x the idle timeout) since the last real stream event, so a stalled tool-call stream times out and recovers instead of hanging with no retry path ([#4900](https://github.com/can1357/oh-my-pi/issues/4900)). diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ce7239af2..7b6d25997 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,6 +13,8 @@ - Built-in provider model discovery now refreshes an expired stored OAuth credential before an online refresh needs it, instead of silently skipping the provider. The refresh is scoped to the providers actually being discovered (`refreshProvider` cannot rotate unrelated credentials), fires under `online-if-uncached` only when the model manager will actually fetch, and offline discovery stays peek-only ([#4893](https://github.com/can1357/oh-my-pi/issues/4893)). - Fixed Escape during an active TUI prompt requiring a second press before canceling; the first Escape now aborts the streaming turn immediately. ([#4921](https://github.com/can1357/oh-my-pi/issues/4921)) - Fixed the streamed `write` tool's collapsed pending tail preview leaving stale rows above the first partial-result frame in the TUI; the first result now replays the viewport like the SSH placeholder seam already did ([#4477](https://github.com/can1357/oh-my-pi/issues/4477)) +- Fixed first-run setup ignoring a pre-seeded `config.yaml`: the settings loader now treats `config.yml` and `config.yaml` as equivalent existing main config files, writes back to the existing extension, and only creates canonical `config.yml` for fresh installs. ([#4914](https://github.com/can1357/oh-my-pi/issues/4914)) +- Fixed extension `sendUserMessage()` without `deliverAs` surfacing `AgentBusyError` during active streams; omitted `deliverAs` now queues a steer through the normal prompt flow, and ACP/RPC skill-command prompts queue while streaming (RPC honors the prompt command's `streamingBehavior`, defaulting to steer) ([#4923](https://github.com/can1357/oh-my-pi/issues/4923)). ## [16.3.12] - 2026-07-08 @@ -26,9 +28,6 @@ ### Fixed -- Fixed first-run setup ignoring a pre-seeded `config.yaml`: the settings loader now treats `config.yml` and `config.yaml` as equivalent existing main config files, writes back to the existing extension, and only creates canonical `config.yml` for fresh installs. ([#4914](https://github.com/can1357/oh-my-pi/issues/4914)) - -- Fixed extension `sendUserMessage()` without `deliverAs` surfacing `AgentBusyError` during active streams; omitted `deliverAs` now queues a steer through the normal prompt flow, and ACP/RPC skill-command prompts queue while streaming (RPC honors the prompt command's `streamingBehavior`, defaulting to steer) ([#4923](https://github.com/can1357/oh-my-pi/issues/4923)). - Improved handling of unawaited promises in JS eval cells to prevent process crashes - Added warning logs for unhandled rejections originating from finished eval cells - Improved advisor robustness by blocking exhausted accounts during consecutive turn failures diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 318668942..34d67e489 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed late terminal appearance subscribers missing the already-detected OSC 11 light/dark result, so theme auto-detection picks up the terminal appearance even when the response arrives before the UI subscribes ([#4731](https://github.com/can1357/oh-my-pi/issues/4731)). +- Fixed slash command Tab completion reopening the file autocomplete drawer after accepting no-argument commands ([#4808](https://github.com/can1357/oh-my-pi/issues/4808)). ## [16.3.12] - 2026-07-08 @@ -15,9 +16,6 @@ - Fixed mid-prompt skill autocomplete so Tab and Enter accept the highlighted `/skill:` suggestion and Backspace dismisses the popup immediately after removing the triggering slash ([#4619](https://github.com/can1357/oh-my-pi/issues/4619)). - Fixed submitted slash-command arguments treating `@` file-reference tokens as prompt-composer autocomplete triggers when the command does not define argument completions. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) - Fixed box-drawing tree lines (`├── item` — directory layouts, decision trees) in prose shearing apart when they wrap: continuation rows now hang under the node text with ancestor rails carried through (`├` → `│`, `└` → blank) instead of restarting at column 0. Applies to prose paragraphs (including inside blockquotes) only when a line with a branch-connector prefix (`├──`, `└─`, …) actually overflows; fitting lines, non-tree prose, and code blocks render byte-for-byte as before. -### Fixed - -- Fixed slash command Tab completion reopening the file autocomplete drawer after accepting no-argument commands ([#4808](https://github.com/can1357/oh-my-pi/issues/4808)). ## [16.3.10] - 2026-07-06 From cde5d75804a3b6d445366b6b4f76c82eeaeb725d Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 19:38:43 +0200 Subject: [PATCH 35/59] chore: bumped models --- packages/catalog/CHANGELOG.md | 14 + packages/catalog/src/models.json | 2399 +++++++++++++++++++++++++++--- 2 files changed, 2229 insertions(+), 184 deletions(-) diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 8bf50c3a7..4efa8d8bb 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,20 @@ ## [Unreleased] +### Added + +- Added support for Grok 4.5 across multiple providers +- Added support for GPT-5.6 series models (Luna, Sol, Terra) +- Added Aion 3.0 and 3.0 Mini models +- Added Kuaishou KAT-Coder v2.5 models +- Added Nex-N2-Mini and SWE-1.7 series models +- Added Hy3 models and free variants + +### Changed + +- Updated cost and token configurations for various models across providers +- Renamed several models for consistency (e.g., MiniMax M3, Gemma 4 31B, Qwen variants) + ## [16.3.12] - 2026-07-08 ### Fixed diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 6493e482b..0d39e022c 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -4972,7 +4972,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "openai-completions", "provider": "aimlapi", "baseUrl": "https://api.aimlapi.com/v1", @@ -14269,7 +14269,7 @@ "cost": { "input": 0.55, "output": 1.65, - "cacheRead": 0, + "cacheRead": 0.55, "cacheWrite": 0 }, "contextWindow": 161000, @@ -14286,13 +14286,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.14, + "output": 0.28, + "cacheRead": 0.07, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 384000, + "maxTokens": 1048576, "thinking": { "mode": "effort", "efforts": [ @@ -14322,13 +14322,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.74, + "output": 3.48, + "cacheRead": 0.14, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 393216, + "maxTokens": 1048576, "thinking": { "mode": "effort", "efforts": [ @@ -14349,7 +14349,7 @@ }, "google/gemma-4-31B-it": { "id": "google/gemma-4-31B-it", - "name": "Gemma 4 31B Instruct", + "name": "Gemma 4 31B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14359,13 +14359,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.12, + "output": 0.35, + "cacheRead": 0.09, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 131072, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14388,9 +14388,9 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.05, + "output": 0.1, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 131072, @@ -14398,7 +14398,7 @@ }, "JetBrains/Mellum2-12B-A2.5B-Instruct": { "id": "JetBrains/Mellum2-12B-A2.5B-Instruct", - "name": "JetBrains/Mellum2-12B-A2.5B-Instruct", + "name": "Mellum2 12B A2.5B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14407,13 +14407,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.05, + "output": 0.1, + "cacheRead": 0.05, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 131072, + "maxTokens": 131072 }, "meta-llama/Llama-3.1-70B-Instruct": { "id": "meta-llama/Llama-3.1-70B-Instruct", @@ -14428,7 +14428,7 @@ "cost": { "input": 0.8, "output": 0.8, - "cacheRead": 0, + "cacheRead": 0.8, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14436,7 +14436,7 @@ }, "meta-llama/Llama-3.1-8B-Instruct": { "id": "meta-llama/Llama-3.1-8B-Instruct", - "name": "Meta-Llama-3.1-8B-Instruct", + "name": "Llama 3.1 8B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14447,7 +14447,7 @@ "cost": { "input": 0.22, "output": 0.22, - "cacheRead": 0, + "cacheRead": 0.22, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14455,7 +14455,7 @@ }, "meta-llama/Llama-3.3-70B-Instruct": { "id": "meta-llama/Llama-3.3-70B-Instruct", - "name": "Llama-3.3-70B-Instruct", + "name": "Llama 3.3 70B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14466,7 +14466,7 @@ "cost": { "input": 0.71, "output": 0.71, - "cacheRead": 0, + "cacheRead": 0.71, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14494,7 +14494,7 @@ }, "microsoft/Phi-4-mini-instruct": { "id": "microsoft/Phi-4-mini-instruct", - "name": "Phi-4-mini-instruct", + "name": "Phi 4 Mini 3.8B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14505,7 +14505,7 @@ "cost": { "input": 0.08, "output": 0.35, - "cacheRead": 0, + "cacheRead": 0.08, "cacheWrite": 0 }, "contextWindow": 128000, @@ -14524,7 +14524,7 @@ "cost": { "input": 0.3, "output": 1.2, - "cacheRead": 0, + "cacheRead": 0.3, "cacheWrite": 0 }, "contextWindow": 196608, @@ -14551,9 +14551,9 @@ "image" ], "cost": { - "input": 0.5, - "output": 2.85, - "cacheRead": 0, + "input": 0.6, + "output": 3, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14571,7 +14571,7 @@ }, "moonshotai/Kimi-K2.6": { "id": "moonshotai/Kimi-K2.6", - "name": "Kimi-K2.6", + "name": "Kimi K2.6", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14581,9 +14581,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.95, + "output": 4, + "cacheRead": 0.16, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14611,9 +14611,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.94, + "output": 4, + "cacheRead": 0.19, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14631,7 +14631,7 @@ }, "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8": { "id": "nvidia/NVIDIA-Nemotron-3-Super-120B-A12B-FP8", - "name": "NVIDIA Nemotron 3 Super 120B", + "name": "Nemotron 3 Super", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14642,7 +14642,7 @@ "cost": { "input": 0.2, "output": 0.8, - "cacheRead": 0, + "cacheRead": 0.2, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14660,22 +14660,32 @@ }, "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B": { "id": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", - "name": "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B", + "name": "Nemotron 3 Ultra", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.75, + "output": 2.75, + "cacheRead": 0.15, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", @@ -14688,9 +14698,9 @@ "text" ], "cost": { - "input": 0.15, - "output": 0.6, - "cacheRead": 0, + "input": 0.04, + "output": 0.14, + "cacheRead": 0.04, "cacheWrite": 0 }, "contextWindow": 131072, @@ -14715,9 +14725,9 @@ "text" ], "cost": { - "input": 0.05, - "output": 0.2, - "cacheRead": 0, + "input": 0.03, + "output": 0.13, + "cacheRead": 0.03, "cacheWrite": 0 }, "contextWindow": 131072, @@ -14733,7 +14743,7 @@ }, "OpenPipe/Qwen3-14B-Instruct": { "id": "OpenPipe/Qwen3-14B-Instruct", - "name": "OpenPipe Qwen3 14B Instruct", + "name": "Qwen3 14B Instruct", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14744,7 +14754,7 @@ "cost": { "input": 0.05, "output": 0.22, - "cacheRead": 0, + "cacheRead": 0.05, "cacheWrite": 0 }, "contextWindow": 32768, @@ -14752,7 +14762,7 @@ }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", - "name": "Qwen3 235B A22B Instruct 2507", + "name": "Qwen3 235B A22B-2507", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14763,7 +14773,7 @@ "cost": { "input": 0.1, "output": 0.1, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14771,7 +14781,7 @@ }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", - "name": "Qwen3-235B-A22B-Thinking-2507", + "name": "Qwen3 235B A22B Thinking-2507", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14782,7 +14792,7 @@ "cost": { "input": 0.1, "output": 0.1, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14811,7 +14821,7 @@ "cost": { "input": 0.1, "output": 0.3, - "cacheRead": 0, + "cacheRead": 0.1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14819,7 +14829,7 @@ }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", - "name": "Qwen3-Coder-480B-A35B-Instruct", + "name": "Qwen3 Coder 480B A35B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14830,7 +14840,7 @@ "cost": { "input": 1, "output": 1.5, - "cacheRead": 0, + "cacheRead": 1, "cacheWrite": 0 }, "contextWindow": 262144, @@ -14838,7 +14848,7 @@ }, "Qwen/Qwen3.5-27B": { "id": "Qwen/Qwen3.5-27B", - "name": "Qwen3.5 27B", + "name": "Qwen3.5-27B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14848,13 +14858,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.39, + "output": 3.12, + "cacheRead": 0.08, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14867,7 +14877,7 @@ }, "Qwen/Qwen3.5-35B-A3B": { "id": "Qwen/Qwen3.5-35B-A3B", - "name": "Qwen3.5 35B-A3B", + "name": "Qwen3.5-35B-A3B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14877,13 +14887,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.25, + "output": 1.25, + "cacheRead": 0.25, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14906,13 +14916,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.6, + "output": 3.6, + "cacheRead": 0.12, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14925,7 +14935,7 @@ }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", - "name": "Qwen3.6 35B-A3B", + "name": "Qwen3.6 35B A3B", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14935,13 +14945,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.25, + "output": 1.25, + "cacheRead": 0.25, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -14973,7 +14983,7 @@ }, "zai-org/GLM-5.1": { "id": "zai-org/GLM-5.1", - "name": "GLM-5.1", + "name": "GLM 5.1", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -14987,8 +14997,8 @@ "cacheRead": 0.26, "cacheWrite": 0 }, - "contextWindow": 200000, - "maxTokens": 131072, + "contextWindow": 202752, + "maxTokens": 202752, "thinking": { "mode": "effort", "efforts": [ @@ -15002,7 +15012,7 @@ }, "zai-org/GLM-5.2": { "id": "zai-org/GLM-5.2", - "name": "GLM-5.2", + "name": "GLM 5.2", "api": "openai-completions", "provider": "coreweave", "baseUrl": "https://api.inference.wandb.ai/v1", @@ -15011,13 +15021,13 @@ "text" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 1.39, + "output": 4.4, + "cacheRead": 0.26, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 164000, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -16764,26 +16774,6 @@ }, "requestModelId": "MODEL_GOOGLE_GEMINI_3_0_FLASH_MINIMAL" }, - "glm-5-1": { - "id": "glm-5-1", - "name": "GLM-5.1", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 200000, - "maxTokens": 64000 - }, "glm-5-2": { "id": "glm-5-2", "name": "GLM-5.2 High", @@ -17224,6 +17214,699 @@ }, "requestModelId": "gpt-5-5-none-priority" }, + "gpt-5-6-luna-high": { + "id": "gpt-5-6-luna-high", + "name": "GPT-5.6 Luna High Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-high-priority": { + "id": "gpt-5-6-luna-high-priority", + "name": "GPT-5.6 Luna High Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-low": { + "id": "gpt-5-6-luna-low", + "name": "GPT-5.6 Luna Low Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-low-priority": { + "id": "gpt-5-6-luna-low-priority", + "name": "GPT-5.6 Luna Low Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-max": { + "id": "gpt-5-6-luna-max", + "name": "GPT-5.6 Luna Max Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-medium": { + "id": "gpt-5-6-luna-medium", + "name": "GPT-5.6 Luna Medium Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-medium-priority": { + "id": "gpt-5-6-luna-medium-priority", + "name": "GPT-5.6 Luna Medium Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-none": { + "id": "gpt-5-6-luna-none", + "name": "GPT-5.6 Luna No Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": false, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-none-priority": { + "id": "gpt-5-6-luna-none-priority", + "name": "GPT-5.6 Luna No Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": false, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-xhigh": { + "id": "gpt-5-6-luna-xhigh", + "name": "GPT-5.6 Luna XHigh Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-luna-xhigh-priority": { + "id": "gpt-5-6-luna-xhigh-priority", + "name": "GPT-5.6 Luna XHigh Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-high": { + "id": "gpt-5-6-sol-high", + "name": "GPT-5.6 Sol High Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-high-priority": { + "id": "gpt-5-6-sol-high-priority", + "name": "GPT-5.6 Sol High Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-low": { + "id": "gpt-5-6-sol-low", + "name": "GPT-5.6 Sol Low Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-low-priority": { + "id": "gpt-5-6-sol-low-priority", + "name": "GPT-5.6 Sol Low Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-max": { + "id": "gpt-5-6-sol-max", + "name": "GPT-5.6 Sol Max Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-medium": { + "id": "gpt-5-6-sol-medium", + "name": "GPT-5.6 Sol Medium Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-medium-priority": { + "id": "gpt-5-6-sol-medium-priority", + "name": "GPT-5.6 Sol Medium Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-none": { + "id": "gpt-5-6-sol-none", + "name": "GPT-5.6 Sol No Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": false, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-none-priority": { + "id": "gpt-5-6-sol-none-priority", + "name": "GPT-5.6 Sol No Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": false, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-xhigh": { + "id": "gpt-5-6-sol-xhigh", + "name": "GPT-5.6 Sol XHigh Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-sol-xhigh-priority": { + "id": "gpt-5-6-sol-xhigh-priority", + "name": "GPT-5.6 Sol XHigh Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-high": { + "id": "gpt-5-6-terra-high", + "name": "GPT-5.6 Terra High Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-high-priority": { + "id": "gpt-5-6-terra-high-priority", + "name": "GPT-5.6 Terra High Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-low": { + "id": "gpt-5-6-terra-low", + "name": "GPT-5.6 Terra Low Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-low-priority": { + "id": "gpt-5-6-terra-low-priority", + "name": "GPT-5.6 Terra Low Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-max": { + "id": "gpt-5-6-terra-max", + "name": "GPT-5.6 Terra Max Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-medium": { + "id": "gpt-5-6-terra-medium", + "name": "GPT-5.6 Terra Medium Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-medium-priority": { + "id": "gpt-5-6-terra-medium-priority", + "name": "GPT-5.6 Terra Medium Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-none": { + "id": "gpt-5-6-terra-none", + "name": "GPT-5.6 Terra No Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": false, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-none-priority": { + "id": "gpt-5-6-terra-none-priority", + "name": "GPT-5.6 Terra No Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": false, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-xhigh": { + "id": "gpt-5-6-terra-xhigh", + "name": "GPT-5.6 Terra XHigh Thinking", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, + "gpt-5-6-terra-xhigh-priority": { + "id": "gpt-5-6-terra-xhigh-priority", + "name": "GPT-5.6 Terra XHigh Thinking Fast", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 64000 + }, "kimi-k2-6": { "id": "kimi-k2-6", "name": "Kimi K2.6", @@ -17454,6 +18137,46 @@ }, "contextWindow": 200000, "maxTokens": 64000 + }, + "swe-1-7": { + "id": "swe-1-7", + "name": "SWE-1.7", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262000, + "maxTokens": 64000 + }, + "swe-1-7-lightning": { + "id": "swe-1-7-lightning", + "name": "SWE-1.7 Lightning", + "api": "devin-agent", + "provider": "devin", + "baseUrl": "https://server.codeium.com", + "reasoning": true, + "input": [ + "text" + ], + "supportsTools": true, + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 202752, + "maxTokens": 64000 } }, "firepass": { @@ -23447,6 +24170,33 @@ ] } }, + "openai/gpt-oss-20b": { + "id": "openai/gpt-oss-20b", + "name": "GPT OSS 20B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.1, + "output": 0.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ] + } + }, "Qwen/Qwen3-235B-A22B": { "id": "Qwen/Qwen3-235B-A22B", "name": "Qwen3 235B-A22B", @@ -24542,6 +25292,44 @@ "contextWindow": null, "maxTokens": null }, + "aion-labs/aion-3.0": { + "id": "aion-labs/aion-3.0", + "name": "Aion-3.0", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, + "aion-labs/aion-3.0-mini": { + "id": "aion-labs/aion-3.0-mini", + "name": "Aion-3.0-Mini", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "aion-labs/aion-rp-llama-3.1-8b": { "id": "aion-labs/aion-rp-llama-3.1-8b", "name": "Aion-RP 1.0 (8B)", @@ -28532,7 +29320,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -28542,13 +29330,13 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.3, + "output": 1.2, + "cacheRead": 0.06, "cacheWrite": 0 }, - "contextWindow": 512000, - "maxTokens": 128000, + "contextWindow": 1048576, + "maxTokens": 512000, "thinking": { "mode": "effort", "efforts": [ @@ -29470,6 +30258,25 @@ "contextWindow": 131072, "maxTokens": 163840 }, + "nex-agi/nex-n2-mini": { + "id": "nex-agi/nex-n2-mini", + "name": "Nex-N2-Mini", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144 + }, "nex-agi/nex-n2-pro": { "id": "nex-agi/nex-n2-pro", "name": "Nex-N2-Pro", @@ -29486,8 +30293,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144 }, "nex-agi/nex-n2-pro:free": { "id": "nex-agi/nex-n2-pro:free", @@ -29505,8 +30312,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144 }, "nousresearch/hermes-2-pro-llama-3-8b": { "id": "nousresearch/hermes-2-pro-llama-3-8b", @@ -31048,6 +31855,120 @@ }, "contextPromotionTarget": "kilo/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 372000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-luna-pro": { + "id": "openai/gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 372000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-sol-pro": { + "id": "openai/gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 372000, + "maxTokens": 128000 + }, + "openai/gpt-5.6-terra-pro": { + "id": "openai/gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000 + }, "openai/gpt-audio": { "id": "openai/gpt-audio", "name": "GPT Audio", @@ -33859,6 +34780,25 @@ "contextWindow": null, "maxTokens": null }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": null + }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Hy3 Preview", @@ -33907,6 +34847,25 @@ "contextWindow": 262144, "maxTokens": 64000 }, + "tencent/hy3:free": { + "id": "tencent/hy3:free", + "name": "Hy3 (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": null + }, "thedrummer/cydonia-24b-v4.1": { "id": "thedrummer/cydonia-24b-v4.1", "name": "Cydonia 24B V4.1", @@ -37363,6 +38322,44 @@ "contextWindow": null, "maxTokens": null }, + "aion-labs/aion-3.0": { + "id": "aion-labs/aion-3.0", + "name": "aion-labs/aion-3.0", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, + "aion-labs/aion-3.0-mini": { + "id": "aion-labs/aion-3.0-mini", + "name": "aion-labs/aion-3.0-mini", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "aion-labs/aion-rp-llama-3.1-8b": { "id": "aion-labs/aion-rp-llama-3.1-8b", "name": "aion-labs/aion-rp-llama-3.1-8b", @@ -44952,6 +45949,25 @@ "contextWindow": null, "maxTokens": null }, + "mellum2-12b-a2-5b-instruct": { + "id": "mellum2-12b-a2-5b-instruct", + "name": "mellum2-12b-a2-5b-instruct", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": null, + "maxTokens": null + }, "mercury-2": { "id": "mercury-2", "name": "Mercury 2", @@ -46617,6 +47633,25 @@ "contextWindow": 131072, "maxTokens": 163840 }, + "nex-agi/nex-n2-mini": { + "id": "nex-agi/nex-n2-mini", + "name": "nex-agi/nex-n2-mini", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144 + }, "nex-agi/nex-n2-pro": { "id": "nex-agi/nex-n2-pro", "name": "nex-agi/nex-n2-pro", @@ -46633,8 +47668,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": null, - "maxTokens": null + "contextWindow": 262144, + "maxTokens": 262144 }, "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16": { "id": "nothingiisreal/L3.1-70B-Celeste-V0.1-BF16", @@ -48807,7 +49842,7 @@ }, "Qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "Qwen/Qwen3-235B-A22B-Instruct-2507", - "name": "Qwen3 235B A22B Instruct 2507", + "name": "Qwen3 235B A22B-2507", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -48864,7 +49899,7 @@ }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", - "name": "Qwen3-235B-A22B-Thinking-2507", + "name": "Qwen3 235B A22B Thinking-2507", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -49250,7 +50285,7 @@ }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", - "name": "Qwen3.6 35B-A3B", + "name": "Qwen3.6 35B A3B", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", @@ -52792,6 +53827,25 @@ ] } }, + "x-ai/grok-4.5": { + "id": "x-ai/grok-4.5", + "name": "x-ai/grok-4.5", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000 + }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", @@ -60261,7 +61315,7 @@ "contextWindow": 272000, "maxTokens": 128000, "preferWebsockets": true, - "priority": 7, + "priority": 0, "applyPatchToolType": "freeform", "thinking": { "mode": "effort", @@ -60273,6 +61327,117 @@ ] }, "contextPromotionTarget": "openai-codex/gpt-5.4" + }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6-Luna", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 3, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.6-sol": { + "id": "gpt-5.6-sol", + "name": "GPT-5.6-Sol", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 1, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "gpt-5.6-terra": { + "id": "gpt-5.6-terra", + "name": "GPT-5.6-Terra", + "api": "openai-codex-responses", + "provider": "openai-codex", + "baseUrl": "https://chatgpt.com/backend-api", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "remoteCompaction": { + "enabled": true, + "api": "openai-codex-responses", + "v2StreamingEnabled": true + }, + "contextWindow": 372000, + "maxTokens": 128000, + "preferWebsockets": true, + "priority": 2, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } } }, "opencode": { @@ -62311,6 +63476,36 @@ }, "contextPromotionTarget": "opencode-zen/gpt-5.4" }, + "grok-4.5": { + "id": "grok-4.5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "grok-build-0.1": { "id": "grok-build-0.1", "name": "Grok Build 0.1", @@ -62341,6 +63536,35 @@ ] } }, + "hy3-free": { + "id": "hy3-free", + "name": "Hy3 Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "hy3-preview-free": { "id": "hy3-preview-free", "name": "Hy3 preview Free", @@ -63214,7 +64438,7 @@ "image" ], "cost": { - "input": 0.66, + "input": 0.65, "output": 3.41, "cacheRead": 0.14, "cacheWrite": 0 @@ -63289,6 +64513,35 @@ ] } }, + "~x-ai/grok-latest": { + "id": "~x-ai/grok-latest", + "name": "Grok Latest", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "ai21/jamba-large-1.7": { "id": "ai21/jamba-large-1.7", "name": "Jamba Large 1.7", @@ -63308,6 +64561,90 @@ "contextWindow": 256000, "maxTokens": 4096 }, + "aion-labs/aion-2.0": { + "id": "aion-labs/aion-2.0", + "name": "Aion-2.0", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7999999999999999, + "output": 1.5999999999999999, + "cacheRead": 0.19999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "aion-labs/aion-3.0": { + "id": "aion-labs/aion-3.0", + "name": "Aion-3.0", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3, + "output": 6, + "cacheRead": 0.75, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "aion-labs/aion-3.0-mini": { + "id": "aion-labs/aion-3.0-mini", + "name": "Aion-3.0-Mini", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.7, + "output": 1.4, + "cacheRead": 0.18, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "alibaba/tongyi-deepresearch-30b-a3b": { "id": "alibaba/tongyi-deepresearch-30b-a3b", "name": "Tongyi DeepResearch 30B A3B", @@ -64846,7 +66183,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 16384, + "maxTokens": 65536, "thinking": { "mode": "effort", "efforts": [ @@ -66146,8 +67483,8 @@ "text" ], "cost": { - "input": 0.12, - "output": 0.48, + "input": 0.15, + "output": 0.8999999999999999, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, @@ -66205,8 +67542,8 @@ "text" ], "cost": { - "input": 0.18, - "output": 0.72, + "input": 0.24, + "output": 0.96, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, @@ -66224,7 +67561,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "openrouter", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", @@ -66875,7 +68212,7 @@ "cost": { "input": 0.375, "output": 2.025, - "cacheRead": 0.09, + "cacheRead": 0.203, "cacheWrite": 0 }, "contextWindow": 262144, @@ -66902,7 +68239,7 @@ "image" ], "cost": { - "input": 0.66, + "input": 0.65, "output": 3.41, "cacheRead": 0.14, "cacheWrite": 0 @@ -66960,13 +68297,13 @@ "image" ], "cost": { - "input": 0.74, + "input": 0.72, "output": 3.5, "cacheRead": 0.15, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 16384, + "maxTokens": 262144, "thinking": { "mode": "effort", "efforts": [ @@ -66996,6 +68333,64 @@ "contextWindow": 131072, "maxTokens": 163840 }, + "nex-agi/nex-n2-mini": { + "id": "nex-agi/nex-n2-mini", + "name": "Nex-N2-Mini", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.024999999999999998, + "output": 0.09999999999999999, + "cacheRead": 0.0025, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "nex-agi/nex-n2-pro": { + "id": "nex-agi/nex-n2-pro", + "name": "Nex-N2-Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 1, + "cacheRead": 0.024999999999999998, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "nex-agi/nex-n2-pro:free": { "id": "nex-agi/nex-n2-pro:free", "name": "Nex-N2-Pro (free)", @@ -68452,6 +69847,180 @@ }, "contextPromotionTarget": "openrouter/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-luna-pro": { + "id": "openai/gpt-5.6-luna-pro", + "name": "GPT-5.6 Luna Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-sol-pro": { + "id": "openai/gpt-5.6-sol-pro", + "name": "GPT-5.6 Sol Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-terra-pro": { + "id": "openai/gpt-5.6-terra-pro", + "name": "GPT-5.6 Terra Pro", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 0 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, "openai/gpt-audio": { "id": "openai/gpt-audio", "name": "GPT Audio", @@ -68530,8 +70099,8 @@ "text" ], "cost": { - "input": 0.03, - "output": 0.15, + "input": 0.036, + "output": 0.18, "cacheRead": 0, "cacheWrite": 0 }, @@ -70301,7 +71870,7 @@ "cost": { "input": 0.385, "output": 2.4499999999999997, - "cacheRead": 0.195, + "cacheRead": 0.111, "cacheWrite": 0 }, "contextWindow": 256000, @@ -70938,6 +72507,34 @@ "supportsToolChoice": false } }, + "tencent/hy3": { + "id": "tencent/hy3", + "name": "Hy3", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.58, + "cacheRead": 0.035, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": null, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "tencent/hy3-preview": { "id": "tencent/hy3-preview", "name": "Hy3 Preview", @@ -70994,6 +72591,34 @@ ] } }, + "tencent/hy3:free": { + "id": "tencent/hy3:free", + "name": "Hy3 (free)", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "thedrummer/rocinante-12b": { "id": "thedrummer/rocinante-12b", "name": "Rocinante 12B", @@ -71418,6 +73043,35 @@ ] } }, + "x-ai/grok-4.5": { + "id": "x-ai/grok-4.5", + "name": "Grok 4.5", + "api": "openrouter", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", @@ -71571,7 +73225,7 @@ "cost": { "input": 0.105, "output": 0.28, - "cacheRead": 0.0028, + "cacheRead": 0.028, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -71971,13 +73625,13 @@ "text" ], "cost": { - "input": 0.9086, - "output": 2.8556, - "cacheRead": 0.16874, + "input": 0.546, + "output": 1.716, + "cacheRead": 0.10139999999999999, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 128000, "thinking": { "mode": "effort", "efforts": [ @@ -73401,38 +75055,6 @@ "escapeBuiltinToolNames": true } }, - "umans-glm-5.2-nvfp4": { - "id": "umans-glm-5.2-nvfp4", - "name": "Umans GLM 5.2 NVFP4 (experimental, short test from Jun 29)", - "api": "anthropic-messages", - "provider": "umans", - "baseUrl": "https://api.code.umans.ai", - "reasoning": true, - "thinking": { - "mode": "anthropic-budget-effort", - "efforts": [ - "high", - "xhigh" - ], - "effortMap": { - "xhigh": "max" - } - }, - "input": [ - "text" - ], - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 405504, - "maxTokens": 131071, - "compat": { - "escapeBuiltinToolNames": true - } - }, "umans-kimi-k2.7": { "id": "umans-kimi-k2.7", "name": "Umans Kimi K2.7 Code", @@ -73524,6 +75146,64 @@ "supportsUsageInStreaming": false } }, + "aion-labs-aion-3-0": { + "id": "aion-labs-aion-3-0", + "name": "Aion 3.0", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3.75, + "output": 7.5, + "cacheRead": 0.9375, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "aion-labs-aion-3-0-mini": { + "id": "aion-labs-aion-3-0-mini", + "name": "Aion 3.0 Mini", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.875, + "output": 1.75, + "cacheRead": 0.225, + "cacheWrite": 0 + }, + "contextWindow": 128000, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "aion-labs.aion-2-0": { "id": "aion-labs.aion-2-0", "name": "aion-labs.aion-2-0", @@ -74080,6 +75760,28 @@ } } }, + "e2ee-deepseek-v4-flash": { + "id": "e2ee-deepseek-v4-flash", + "name": "e2ee-deepseek-v4-flash", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 1048576, + "compat": { + "supportsUsageInStreaming": false + } + }, "e2ee-gemma-3-27b-p": { "id": "e2ee-gemma-3-27b-p", "name": "e2ee-gemma-3-27b-p", @@ -74366,6 +76068,28 @@ "supportsUsageInStreaming": false } }, + "e2ee-qwen3-6-27b": { + "id": "e2ee-qwen3-6-27b", + "name": "e2ee-qwen3-6-27b", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 65536, + "compat": { + "supportsUsageInStreaming": false + } + }, "e2ee-qwen3-6-35b-a3b": { "id": "e2ee-qwen3-6-35b-a3b", "name": "e2ee-qwen3-6-35b-a3b", @@ -74845,6 +76569,36 @@ ] } }, + "grok-4-5": { + "id": "grok-4-5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "venice", + "baseUrl": "https://api.venice.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.27, + "output": 6.8, + "cacheRead": 0.57, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 32000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "grok-41-fast": { "id": "grok-41-fast", "name": "Grok 4.1 Fast", @@ -79099,7 +80853,7 @@ }, "minimax/minimax-m3": { "id": "minimax/minimax-m3", - "name": "MiniMax-M3", + "name": "MiniMax M3", "api": "anthropic-messages", "provider": "vercel-ai-gateway", "baseUrl": "https://ai-gateway.vercel.sh", @@ -80667,6 +82421,93 @@ }, "contextPromotionTarget": "vercel-ai-gateway/openai/gpt-5.4" }, + "openai/gpt-5.6-luna": { + "id": "openai/gpt-5.6-luna", + "name": "GPT 5.6 Luna", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.09999999999999999, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-sol": { + "id": "openai/gpt-5.6-sol", + "name": "GPT 5.6 Sol", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "openai/gpt-5.6-terra": { + "id": "openai/gpt-5.6-terra", + "name": "GPT 5.6 Terra", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "thinking": { + "mode": "budget", + "efforts": [ + "low", + "medium", + "high", + "xhigh" + ] + } + }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", @@ -81557,6 +83398,36 @@ ] } }, + "xai/grok-4.5": { + "id": "xai/grok-4.5", + "name": "Grok 4.5", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "xai/grok-build-0.1": { "id": "xai/grok-build-0.1", "name": "Grok Build 0.1", @@ -82313,6 +84184,39 @@ "supportsDeveloperRole": false } }, + "GLM5.2-Turbo": { + "id": "GLM5.2-Turbo", + "name": "GLM5.2-Turbo", + "api": "openai-completions", + "provider": "wafer-serverless", + "baseUrl": "https://pass.wafer.ai/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 3.75, + "output": 12.8125, + "cacheRead": 0.625, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + }, + "compat": { + "reasoningContentField": "reasoning_content", + "supportsDeveloperRole": false + } + }, "Kimi-K2.6": { "id": "Kimi-K2.6", "name": "Kimi-K2.6", @@ -83094,6 +84998,35 @@ ] } }, + "grok-4.5": { + "id": "grok-4.5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "xai", + "baseUrl": "https://api.x.ai/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "grok-beta": { "id": "grok-beta", "name": "Grok Beta", @@ -86308,6 +88241,25 @@ ] } }, + "kuaishou/kat-coder-air-v2.5": { + "id": "kuaishou/kat-coder-air-v2.5", + "name": "KAT-Coder-Air-V2.5", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.135, + "output": 0.54, + "cacheRead": 0.027, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": null + }, "kuaishou/kat-coder-pro-v1": { "id": "kuaishou/kat-coder-pro-v1", "name": "KAT-Coder-Pro-V1", @@ -86365,6 +88317,25 @@ "contextWindow": 256000, "maxTokens": 80000 }, + "kuaishou/kat-coder-pro-v2.5": { + "id": "kuaishou/kat-coder-pro-v2.5", + "name": "KAT-Coder-Pro-V2.5", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.444, + "output": 1.776, + "cacheRead": 0.09, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": null + }, "meituan/longcat-2.0": { "id": "meituan/longcat-2.0", "name": "LongCat-2.0", @@ -88482,9 +90453,9 @@ "text" ], "cost": { - "input": 0.134561595, - "output": 0.539161765, - "cacheRead": 0.033869245, + "input": 0.1323, + "output": 0.5301, + "cacheRead": 0.0333, "cacheWrite": 0 }, "contextWindow": 262144, @@ -88938,6 +90909,66 @@ ] } }, + "x-ai/grok-4.5": { + "id": "x-ai/grok-4.5", + "name": "Grok 4.5", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2, + "output": 6, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "x-ai/grok-4.5-free": { + "id": "x-ai/grok-4.5-free", + "name": "Grok 4.5 (Free)", + "api": "openai-completions", + "provider": "zenmux", + "baseUrl": "https://zenmux.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 500000, + "maxTokens": 500000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "x-ai/grok-build-0.1": { "id": "x-ai/grok-build-0.1", "name": "Grok Build 0.1", From dd67447a0325de62db9146991bab492e260b38ff Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 19:39:55 +0200 Subject: [PATCH 36/59] chore: bump version to 16.3.13 --- Cargo.lock | 42 ++++++++++----------- Cargo.toml | 2 +- bun.lock | 54 +++++++++++++-------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 24 ++++++------ packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 + packages/ai/package.json | 2 +- packages/catalog/CHANGELOG.md | 2 + packages/catalog/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/package.json | 2 +- packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/CHANGELOG.md | 2 + packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/snapcompact/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/CHANGELOG.md | 2 + packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- packages/wire/package.json | 2 +- 25 files changed, 87 insertions(+), 77 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 87f1ebd9f..e3c6b33b8 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1951,9 +1951,9 @@ dependencies = [ [[package]] name = "inotify" -version = "0.11.3" +version = "0.11.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dd854a95a4ac672fed8c054136039fd32c22cf039ff09ead7280afe920486483" +checksum = "153be1941a183ec9ccd095ddbe17a8b8d435ef6c76e9e02451b933c3999af2c8" dependencies = [ "bitflags 2.13.0", "inotify-sys", @@ -2026,9 +2026,9 @@ checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" [[package]] name = "jiff" -version = "0.2.31" +version = "0.2.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ccfe6121cbe750cf81efa362d85c0bde7ea298ec43092d3a193baca59cdbd634" +checksum = "961d16382652bfdd8c6f68b223b26a8c93e0d475c672f414411db31c6c5c900e" dependencies = [ "defmt", "jiff-static", @@ -2053,9 +2053,9 @@ dependencies = [ [[package]] name = "jiff-static" -version = "0.2.31" +version = "0.2.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e165e897f662d428f3cd3828a919dbe067c2d42bb1031eede74ef9d27ecdedd2" +checksum = "d0879bd39df99c4c5e2c6615ccc026391a423dde10532c573e6086eb94a802cc" dependencies = [ "proc-macro2", "quote", @@ -2064,9 +2064,9 @@ dependencies = [ [[package]] name = "jiff-tzdb" -version = "0.1.7" +version = "0.1.8" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6142247df1a93c2b3587402a19710be3e6e942f1581a1702e76408f2c21d6590" +checksum = "142bd39932ad231f10513df9ab62661fead8719872150b7ad02a2df79f4e141e" [[package]] name = "jiff-tzdb-platform" @@ -2881,7 +2881,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.3.12" +version = "16.3.13" dependencies = [ "anyhow", "ast-grep-core", @@ -2950,7 +2950,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.3.12" +version = "16.3.13" dependencies = [ "async-trait", "libc", @@ -2962,7 +2962,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.3.12" +version = "16.3.13" dependencies = [ "anyhow", "arboard", @@ -3015,7 +3015,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.3.12" +version = "16.3.13" dependencies = [ "anyhow", "brush-builtins", @@ -3064,7 +3064,7 @@ dependencies = [ [[package]] name = "pi-walker" -version = "16.3.12" +version = "16.3.13" dependencies = [ "dashmap", "globset", @@ -3404,9 +3404,9 @@ dependencies = [ [[package]] name = "regex" -version = "1.12.4" +version = "1.13.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f1292b7759ae1cb9ec195452d1390a074f0cd8541ab7a5a8c31cd6db45d4a6ba" +checksum = "2a0e75113e14dc5acb068cd0786884f214f1312650a3d36d269f5c4f3cdee8a2" dependencies = [ "aho-corasick", "memchr", @@ -3416,9 +3416,9 @@ dependencies = [ [[package]] name = "regex-automata" -version = "0.4.14" +version = "0.4.15" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "6e1dd4122fc1595e8162618945476892eefca7b88c52820e74af6262213cae8f" +checksum = "1f388202e4b80542a0921078cc23b6333bcf1409c1e3f86404cae4766a6131db" dependencies = [ "aho-corasick", "memchr", @@ -5766,18 +5766,18 @@ dependencies = [ [[package]] name = "zerocopy" -version = "0.8.53" +version = "0.8.54" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "75726053136156d419e285b9b7eddaaea9e3fea6ce32eed44a89901f0bd98de1" +checksum = "b7cbbc0a705a0fd05cc3676525980d2bf5a9bc4adac6d6475209a7887cf59d19" dependencies = [ "zerocopy-derive", ] [[package]] name = "zerocopy-derive" -version = "0.8.53" +version = "0.8.54" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4714fd92cf900833d49538023a9b3915155210801d1c1169eba513b2addefd71" +checksum = "e2e817b7b52d0c7358d3246da9d69935ebb18116b2b102b4230dac079b4862f5" dependencies = [ "proc-macro2", "quote", diff --git a/Cargo.toml b/Cargo.toml index ecb8d8d92..52c4f4015 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/vendor/brush-core", "crates/vendor/brush-builtins"] resolver = "3" [workspace.package] -version = "16.3.12" +version = "16.3.13" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index f97a68579..ae493b35d 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.12", + "version": "16.3.13", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.3.12", + "version": "16.3.13", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.3.12", + "version": "16.3.13", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.3.12", + "version": "16.3.13", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.3.12", + "version": "16.3.13", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.12", + "version": "16.3.13", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.3.12", + "version": "16.3.13", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.3.12", + "version": "16.3.13", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.3.12", + "version": "16.3.13", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.3.12", + "version": "16.3.13", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.3.12", + "version": "16.3.13", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.3.12", + "version": "16.3.13", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.3.12", + "version": "16.3.13", "devDependencies": { "@types/bun": "catalog:", }, @@ -338,18 +338,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.12", - "@oh-my-pi/omp-stats": "16.3.12", - "@oh-my-pi/pi-agent-core": "16.3.12", - "@oh-my-pi/pi-ai": "16.3.12", - "@oh-my-pi/pi-catalog": "16.3.12", - "@oh-my-pi/pi-coding-agent": "16.3.12", - "@oh-my-pi/pi-mnemopi": "16.3.12", - "@oh-my-pi/pi-natives": "16.3.12", - "@oh-my-pi/pi-tui": "16.3.12", - "@oh-my-pi/pi-utils": "16.3.12", - "@oh-my-pi/pi-wire": "16.3.12", - "@oh-my-pi/snapcompact": "16.3.12", + "@oh-my-pi/hashline": "16.3.13", + "@oh-my-pi/omp-stats": "16.3.13", + "@oh-my-pi/pi-agent-core": "16.3.13", + "@oh-my-pi/pi-ai": "16.3.13", + "@oh-my-pi/pi-catalog": "16.3.13", + "@oh-my-pi/pi-coding-agent": "16.3.13", + "@oh-my-pi/pi-mnemopi": "16.3.13", + "@oh-my-pi/pi-natives": "16.3.13", + "@oh-my-pi/pi-tui": "16.3.13", + "@oh-my-pi/pi-utils": "16.3.13", + "@oh-my-pi/pi-wire": "16.3.13", + "@oh-my-pi/snapcompact": "16.3.13", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -811,7 +811,7 @@ "@protobufjs/pool": ["@protobufjs/pool@1.1.0", "", {}, "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw=="], - "@protobufjs/utf8": ["@protobufjs/utf8@1.1.1", "", {}, "sha512-oOAWABowe8EAbMyWKM0tYDKi8Yaox52D+HWZhAIJqQXbqe0xI/GV7FhLWqlEKreMkfDjshR5FKgi3mnle0h6Eg=="], + "@protobufjs/utf8": ["@protobufjs/utf8@1.1.2", "", {}, "sha512-b1UQwcEZ4yCnMCD8DAL1VlbvBJE9/IX4FTIp7BG1xYpf29SLazLSrqUkj4w7Y5y7cCVP6E5tcqqcI0xemPkHug=="], "@puppeteer/browsers": ["@puppeteer/browsers@3.0.6", "", { "dependencies": { "modern-tar": "^0.7.6", "yargs": "^18.0.0" }, "peerDependencies": { "proxy-agent": ">=8.0.1", "yauzl": "^2.10.0 || ^3.4.0" }, "optionalPeers": ["proxy-agent", "yauzl"], "bin": { "browsers": "lib/main-cli.js" } }, "sha512-B/gKoqlFkzhvzsI6jo9K1cZz9o5ypviVv/xu8CwA4grZzyVwN+XfkT+tu8T1zrauuEXv6VhS2oGX+6NL95WcKA=="], @@ -963,7 +963,7 @@ "brace-expansion": ["brace-expansion@5.0.7", "", { "dependencies": { "balanced-match": "^4.0.2" } }, "sha512-7oFy703dxfY3/NLxC1fh2SUCQ0H9rmAY+5EpDVfXjUTTs+HEwR2nYaqLv+GWcTsumwxPfiz6CzCNkwXwBUwqCA=="], - "browserslist": ["browserslist@4.28.4", "", { "dependencies": { "baseline-browser-mapping": "^2.10.38", "caniuse-lite": "^1.0.30001799", "electron-to-chromium": "^1.5.376", "node-releases": "^2.0.48", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-MTc8i/x9jBQd1iMw2CFGS+rwMa07eYjLR0CCTLDACl9xhxy+nIs3KeML/biicXtk9JrZ6dnnTatmc7ErPXIxqw=="], + "browserslist": ["browserslist@4.28.5", "", { "dependencies": { "baseline-browser-mapping": "^2.10.42", "caniuse-lite": "^1.0.30001800", "electron-to-chromium": "^1.5.387", "node-releases": "^2.0.50", "update-browserslist-db": "^1.2.3" }, "bin": { "browserslist": "cli.js" } }, "sha512-Cu2E6QejHWzuDMTkuwgpABFgDfZrXLQq5V13YOACZx4mFAG4IwGTbTfHPMr4WtxlHoXSM8FIuRwYYCz5XiabaQ=="], "bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 7699d1fb1..d142ebce9 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -248,7 +248,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_3_12")] +#[napi(js_name = "__piNativesV16_3_13")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/package.json b/package.json index aaee74002..786ffe1f0 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.3.12", - "@oh-my-pi/omp-stats": "16.3.12", - "@oh-my-pi/pi-agent-core": "16.3.12", - "@oh-my-pi/pi-ai": "16.3.12", - "@oh-my-pi/pi-catalog": "16.3.12", - "@oh-my-pi/pi-coding-agent": "16.3.12", - "@oh-my-pi/pi-mnemopi": "16.3.12", - "@oh-my-pi/pi-natives": "16.3.12", - "@oh-my-pi/pi-tui": "16.3.12", - "@oh-my-pi/pi-utils": "16.3.12", - "@oh-my-pi/pi-wire": "16.3.12", - "@oh-my-pi/snapcompact": "16.3.12", + "@oh-my-pi/hashline": "16.3.13", + "@oh-my-pi/omp-stats": "16.3.13", + "@oh-my-pi/pi-agent-core": "16.3.13", + "@oh-my-pi/pi-ai": "16.3.13", + "@oh-my-pi/pi-catalog": "16.3.13", + "@oh-my-pi/pi-coding-agent": "16.3.13", + "@oh-my-pi/pi-mnemopi": "16.3.13", + "@oh-my-pi/pi-natives": "16.3.13", + "@oh-my-pi/pi-tui": "16.3.13", + "@oh-my-pi/pi-utils": "16.3.13", + "@oh-my-pi/pi-wire": "16.3.13", + "@oh-my-pi/snapcompact": "16.3.13", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index 5d945c29a..098ed870c 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.3.12", + "version": "16.3.13", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 851cfbaad..1c9a41476 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.3.13] - 2026-07-09 + ### Changed - Changed the xAI Grok OAuth (`xai-oauth`) provider to use manual code-paste login by default. `/login` now accepts a pasted authorization code or full `http://127.0.0.1:56121/callback?code=...` redirect URL without starting a local callback listener ([#3277](https://github.com/can1357/oh-my-pi/pull/3277) by [@Jaaneek](https://github.com/Jaaneek)). diff --git a/packages/ai/package.json b/packages/ai/package.json index 814352b52..47844c32f 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.3.12", + "version": "16.3.13", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 4efa8d8bb..b6f938072 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.3.13] - 2026-07-09 + ### Added - Added support for Grok 4.5 across multiple providers diff --git a/packages/catalog/package.json b/packages/catalog/package.json index e2272c3d0..bfdab0fd0 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.3.12", + "version": "16.3.13", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7b6d25997..d1ad895e4 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.3.13] - 2026-07-09 + ### Fixed - Fixed `read` and `grep` treating empty optional `selector` fields emitted by models as invalid selectors instead of behaving like omitted selectors. ([#4879](https://github.com/can1357/oh-my-pi/issues/4879)) diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 5ab0c2c33..4a9fe0f42 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.3.12", + "version": "16.3.13", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 742cc1670..d1f6d86a2 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.3.12", + "version": "16.3.13", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 2f37b76b6..edaf8eb06 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.3.12", + "version": "16.3.13", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index 2e1f8df66..fb1e08bea 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.3.13] - 2026-07-09 + ### Fixed - Fixed unbounded memory growth in the native bash output bridge when a command produces output faster than the JS event loop consumes it: the shell streaming path now uses a bounded chunk queue with real backpressure (pipe readers park until the JS callback catches up, parking the child on its pipe) instead of buffering the entire surplus in memory. No output is dropped — the rolling tail view, `[raw output: artifact://…]` lossless capture, and byte accounting are unaffected ([#4078](https://github.com/can1357/oh-my-pi/issues/4078)). diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 6649dc486..3f5f608c0 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -170,7 +170,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_3_12(): void +export declare function __piNativesV16_3_13(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index d7314c43d..e7567f430 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_3_12 = nativeBindings.__piNativesV16_3_12; +export const __piNativesV16_3_13 = nativeBindings.__piNativesV16_3_13; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index c0f0e2add..8cb891985 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.3.12", + "version": "16.3.13", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index 3e52a7a2c..d05db1604 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.3.12", + "version": "16.3.13", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/package.json b/packages/stats/package.json index 10c8e2e97..6613aac38 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.3.12", + "version": "16.3.13", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index ccc809a1b..8f99c2391 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.3.12", + "version": "16.3.13", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 34d67e489..7bae35203 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.3.13] - 2026-07-09 + ### Fixed - Fixed late terminal appearance subscribers missing the already-detected OSC 11 light/dark result, so theme auto-detection picks up the terminal appearance even when the response arrives before the UI subscribes ([#4731](https://github.com/can1357/oh-my-pi/issues/4731)). diff --git a/packages/tui/package.json b/packages/tui/package.json index 682e2b532..4c155b35f 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.3.12", + "version": "16.3.13", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index 9a805d384..b4f006250 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.3.12", + "version": "16.3.13", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/wire/package.json b/packages/wire/package.json index 94fda90cf..64ecb824c 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.3.12", + "version": "16.3.13", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From 10846c255add38f75269633f5eac4593b38fb0dd Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 20:01:37 +0200 Subject: [PATCH 37/59] ci(release): pinned npm 11 for publishing --- .github/workflows/ci.yml | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 7dbe06010..00df0a797 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -622,10 +622,11 @@ jobs: with: node-version: "24" registry-url: "https://registry.npmjs.org" - # Keep npm aligned with trusted publishing setup (>= 11.16.0). + # npm runs under Bun when invoked by the release script; npm 12 + # requires a newer emulated Node version than Bun 1.3 provides. - name: Ensure npm supports trusted publishing if: ${{ !inputs.skip_npm }} - run: npm install -g npm@latest + run: npm install -g npm@11.17.0 - name: Cache bun dependencies uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 with: @@ -775,9 +776,10 @@ jobs: with: node-version: "24" registry-url: "https://registry.npmjs.org" - # Keep npm aligned with trusted publishing setup (>= 11.16.0). + # npm runs under Bun when invoked by the release script; npm 12 + # requires a newer emulated Node version than Bun 1.3 provides. - name: Ensure npm supports trusted publishing - run: npm install -g npm@latest + run: npm install -g npm@11.17.0 - name: Cache bun dependencies uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0 with: From 0e74c85c0816b1c9366db9befa88ba32248703a9 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 20:09:15 +0200 Subject: [PATCH 38/59] fix(coding-agent/utils): filtered gpt-5 reasoning comment noise - Removed literal HTML comment sentinels (``) from thinking block displays. - Added logic to hide blocks that consist entirely of reasoning noise and updated display validation to omit empty formatted output. - Refactored the memoization cache to maintain separate slots for prose and raw modes. --- packages/coding-agent/CHANGELOG.md | 7 ++ .../src/utils/thinking-display.ts | 102 +++++++++++++----- 2 files changed, 81 insertions(+), 28 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d1ad895e4..636167fe5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,13 @@ ## [Unreleased] +### Fixed + +- Improved rendering of raw thinking blocks by stripping empty HTML comment noise +- Fixed display of thinking blocks consisting entirely of hidden comment noise + +- Fixed gpt-5.6 reasoning summaries rendering literal `` sentinel lines in thinking blocks; empty HTML comments (and the unterminated ``), streamed as a ``. Comments with actual content are left untouched. +const EMPTY_COMMENT_RE = /^$/; +const OPEN_COMMENT_RE = /^` + * sentinel lines outside code fences (see {@link isCommentNoise}); prose-only + * mode additionally elides fenced code down to a trailing ellipsis. + */ export function formatThinkingForDisplay(text: string, proseOnly: boolean): string { - if (!proseOnly || !text) return text; - if (text === formatCacheKey) return formatCacheValue; + if (!text) return text; + const hasComment = text.includes("` leaves no blank tail. + if (hasComment && isCommentNoise(line, i === lines.length - 1)) continue; + + const open = FENCE.exec(line); + if (open) { const marker = open[2]!; const ch = marker[0]!; // A backtick fence's info string may not contain a backtick. @@ -79,18 +116,25 @@ export function formatThinkingForDisplay(text: string, proseOnly: boolean): stri inFence = true; fenceChar = ch; fenceLen = marker.length; - appendEllipsis(); - } else { - resultLines.push(line); + if (proseOnly) { + appendEllipsis(); + } else { + resultLines.push(line); + } + continue; } - } else { - resultLines.push(line); } + resultLines.push(line); } const formatted = resultLines.join("\n"); - formatCacheKey = text; - formatCacheValue = formatted; + if (proseOnly) { + proseCacheKey = text; + proseCacheValue = formatted; + } else { + rawCacheKey = text; + rawCacheValue = formatted; + } return formatted; } @@ -99,9 +143,11 @@ export function hasDisplayableThinking( text: string | null | undefined, formattedText: string | null | undefined, ): boolean { - if (!text) return false; - if (!formattedText) return false; - return formattedText.length > 0 && canonicalizeMessage(text).length > 0; + if (!text || !formattedText) return false; + // Visibility keys off the formatted text: a block whose raw text is only + // comment noise (`\n`) formats to whitespace and stays hidden. The + // raw canonicalize check still hides dot/ellipsis-only placeholder blocks. + return formattedText.trim().length > 0 && canonicalizeMessage(text).length > 0; } /** Whether an assistant message contains thinking content the TUI can reveal. */ From 9d5207ea36357e7e84df7da640be1cae376f8894 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 20:32:30 +0200 Subject: [PATCH 39/59] feat: integrated gpt-5.6 models and unified logical model resolution - Added support for GPT-5.6 (Luna, Sol, Terra) models including configuration updates and context window values. - Implemented automatic effort tier remapping for wire-effort models to ensure proper translation between user-facing tiers and provider requirements. - Updated Codex request transformers to handle effort shifting and added validation for reasoning configurations. - Collapsed Devin-specific model variants to unify logical model handling and added comprehensive test coverage for effort resolution. --- packages/ai/CHANGELOG.md | 8 + .../openai-codex/request-transformer.ts | 62 +- packages/ai/src/providers/openai-shared.ts | 12 +- packages/ai/test/openai-codex.test.ts | 30 + packages/catalog/CHANGELOG.md | 7 + packages/catalog/src/model-thinking.ts | 52 +- packages/catalog/src/models.json | 974 +++++++----------- packages/catalog/src/variant-collapse.ts | 73 +- packages/catalog/test/model-thinking.test.ts | 90 ++ 9 files changed, 658 insertions(+), 650 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 1c9a41476..b3dcc30ab 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +### Changed + +- Updated Codex reasoning effort mapping to support shifted wire tiers for newer models + +### Fixed + +- Fixed the Codex Responses request transformer bypassing catalog/compat reasoning effort maps: the clamped user effort is now remapped to the provider wire tier (GPT-5.6's shifted five-tier scale sends `max` for user `xhigh` and `xhigh` for `high`), failing loudly if a map produces a value outside the Codex wire vocabulary. + ## [16.3.13] - 2026-07-09 ### Changed diff --git a/packages/ai/src/providers/openai-codex/request-transformer.ts b/packages/ai/src/providers/openai-codex/request-transformer.ts index d603130c3..67af8f232 100644 --- a/packages/ai/src/providers/openai-codex/request-transformer.ts +++ b/packages/ai/src/providers/openai-codex/request-transformer.ts @@ -1,19 +1,33 @@ -import type { Effort } from "@oh-my-pi/pi-catalog/effort"; +import { Effort } from "@oh-my-pi/pi-catalog/effort"; import { supportsAllTurnsReasoningContext, supportsCodexReasoningSummary } from "@oh-my-pi/pi-catalog/identity"; import { requireSupportedEffort } from "@oh-my-pi/pi-catalog/model-thinking"; -import type { Api, Model } from "../../types"; +import type { Model } from "../../types"; +import { mapOpenAIReasoningEffort } from "../openai-shared"; /** Reasoning replay scope for the Codex Responses API (`reasoning.context`). */ export type CodexReasoningContext = "auto" | "current_turn" | "all_turns"; +/** User-facing effort levels accepted by Codex request options. */ +type CodexCallerEffort = "minimal" | "low" | "medium" | "high" | "xhigh"; + +/** Caller literal → catalog `Effort` bridge (the enum is nominal). */ +const EFFORT_BY_NAME: Record = { + minimal: Effort.Minimal, + low: Effort.Low, + medium: Effort.Medium, + high: Effort.High, + xhigh: Effort.XHigh, +}; + export interface ReasoningConfig { - effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh"; + effort: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | "max"; summary?: "auto" | "concise" | "detailed"; context?: CodexReasoningContext; } export interface CodexRequestOptions { - reasoningEffort?: ReasoningConfig["effort"]; + /** User-facing effort; the wire-only `max` tier is reached via the model's effort map. */ + reasoningEffort?: CodexCallerEffort | "none"; reasoningSummary?: ReasoningConfig["summary"] | null; /** Explicit `reasoning.context` override; defaults to `all_turns` when unset. The `all_turns` value is gated to gpt-5.4+ Codex models — older ids reject it, so it is suppressed and `context` omitted. */ reasoningContext?: CodexReasoningContext; @@ -80,10 +94,40 @@ export function shouldUseCodexResponsesLite(body: RequestBody, requested: boolea return requested === true && !containsInputImage(body.input); } -function getReasoningConfig(model: Model, options: CodexRequestOptions): ReasoningConfig { +/** + * Clamp a user-facing effort to the model's ladder, then remap to the wire + * tier (e.g. GPT-5.6's shifted five-tier scale sends `max` for user `xhigh`). + * A mapped value outside the Codex wire vocabulary is a broken compat/model + * effort map — fail loudly rather than silently sending a different tier. + */ +function mapCodexWireEffort( + model: Model<"openai-codex-responses">, + effort: CodexCallerEffort, +): ReasoningConfig["effort"] { + const mapped = mapOpenAIReasoningEffort(model, model.compat, requireSupportedEffort(model, EFFORT_BY_NAME[effort])); + switch (mapped) { + case "none": + case "minimal": + case "low": + case "medium": + case "high": + case "xhigh": + case "max": + return mapped; + default: + throw new Error( + `Effort map for ${model.provider}/${model.id} produced invalid Codex reasoning effort "${mapped}"`, + ); + } +} + +function getReasoningConfig( + model: Model<"openai-codex-responses">, + effort: NonNullable, + options: CodexRequestOptions, +): ReasoningConfig { const config: ReasoningConfig = { - effort: - options.reasoningEffort === "none" ? "none" : requireSupportedEffort(model, options.reasoningEffort as Effort), + effort: effort === "none" ? "none" : mapCodexWireEffort(model, effort), }; // `reasoning.summary` is accepted only from gpt-5.4 onward; earlier Codex ids // (gpt-5.1-codex, gpt-5.3-codex, gpt-5.3-codex-spark) reject it with @@ -216,7 +260,7 @@ function stripImageDetails(input: InputItem[]): void { export async function transformRequestBody( body: RequestBody, - model: Model, + model: Model<"openai-codex-responses">, options: CodexRequestOptions = {}, prompt?: { developerMessages: string[] }, ): Promise { @@ -300,7 +344,7 @@ export async function transformRequestBody( } if (options.reasoningEffort !== undefined) { - const reasoningConfig = getReasoningConfig(model, options); + const reasoningConfig = getReasoningConfig(model, options.reasoningEffort, options); body.reasoning = { ...body.reasoning, ...reasoningConfig, diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index a0ac83994..bacf6d8af 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -695,13 +695,19 @@ export interface OpenAICompatPolicy { }; } -function mapOpenAIReasoningEffort( +/** + * Map a user-facing effort to the provider wire value: explicit compat + * override first, then the model's baked `thinking.effortMap`, else identity. + * Shared by the chat-completions/Responses policy resolver and the Codex + * request transformer. + */ +export function mapOpenAIReasoningEffort( model: Pick, - compat: OpenAICompatPolicyCompat, + compat: { reasoningEffortMap?: Partial> } | undefined, effort: string, ): string { const level = effort as Effort; - return compat.reasoningEffortMap?.[level] ?? model.thinking?.effortMap?.[level] ?? effort; + return compat?.reasoningEffortMap?.[level] ?? model.thinking?.effortMap?.[level] ?? effort; } function isImplicitDisableWhenNotRequested(disableMode: OpenAIReasoningDisableMode): boolean { diff --git a/packages/ai/test/openai-codex.test.ts b/packages/ai/test/openai-codex.test.ts index e28633fd1..1c7492afc 100644 --- a/packages/ai/test/openai-codex.test.ts +++ b/packages/ai/test/openai-codex.test.ts @@ -314,6 +314,36 @@ describe("openai-codex reasoning effort validation", () => { }); }); +describe("openai-codex reasoning effort wire mapping", () => { + it("shifts gpt-5.6 user efforts one wire tier up via the baked effort map", async () => { + const model = createCodexModel("gpt-5.6-sol"); + const shifted = [ + ["minimal", "low"], + ["low", "medium"], + ["medium", "high"], + ["high", "xhigh"], + ["xhigh", "max"], + ] as const; + + for (const [requested, wire] of shifted) { + const transformed = await transformRequestBody({ model: model.id }, model, { + reasoningEffort: requested, + }); + expect(transformed.reasoning?.effort).toBe(wire); + } + }); + + it("keeps pre-5.6 efforts unshifted and passes none through unmapped", async () => { + const gpt55 = createCodexModel("gpt-5.5"); + const unshifted = await transformRequestBody({ model: gpt55.id }, gpt55, { reasoningEffort: "xhigh" }); + expect(unshifted.reasoning?.effort).toBe("xhigh"); + + const gpt56 = createCodexModel("gpt-5.6-sol"); + const none = await transformRequestBody({ model: gpt56.id }, gpt56, { reasoningEffort: "none" }); + expect(none.reasoning?.effort).toBe("none"); + }); +}); + describe("openai-codex error parsing", () => { it("produces friendly usage-limit messages and rate limits", async () => { const resetAt = Math.floor(Date.now() / 1000) + 600; diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index b6f938072..bbd58a0ca 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,13 @@ ## [Unreleased] +### Added + +- Added support for GPT-5.6 (Luna, Sol, Terra) model variants +- Enabled expanded five-tier reasoning effort scale (minimal to xhigh) for GPT-5.6 models + +- Added GPT-5.6 (Terra/Luna/Sol) support for the new `max` reasoning tier: on wire-effort APIs (OpenAI Responses, Codex, Azure, openai-compat/OpenRouter models that advertise reasoning) user efforts shift up one notch — `xhigh` sends `max`, `high` sends `xhigh` — mirroring the Claude Fable/Opus 4.7+ five-tier mapping, and the exposed ladder becomes `minimal..xhigh` with `minimal` reaching the native `low` tier. Devin's per-tier GPT-5.6 sibling rows now collapse into `gpt-5-6-{luna,sol,terra}` logical models with the same shifted routing (`xhigh` → `-max`), plus `-fast` families that keep the direct `low..xhigh` `-priority` scale since Devin serves no `-max-priority` tier. + ## [16.3.13] - 2026-07-09 ### Added diff --git a/packages/catalog/src/model-thinking.ts b/packages/catalog/src/model-thinking.ts index b0e01af59..611eb1f60 100644 --- a/packages/catalog/src/model-thinking.ts +++ b/packages/catalog/src/model-thinking.ts @@ -17,6 +17,7 @@ import { type ParsedModel, parseAnthropicModel, parseKnownModel, + parseOpenAIModel, semverEqual, semverGte, } from "./identity/classify"; @@ -101,12 +102,14 @@ const MIMO_REASONING_EFFORT_MAP: Readonly = { }; /** - * Effort → wire-value map for the 5-tier adaptive scale (Opus 4.7+ and - * Fable/Mythos 5 on the Messages API). User-facing efforts shift up one notch - * so the top tier reaches the genuine "max" and "high" lands on Anthropic's - * recommended "xhigh" coding/agentic default. + * Effort → wire-value map for a shifted five-tier scale (`low..max`): + * user-facing efforts shift up one notch so the top tier reaches the genuine + * "max" and "high" lands on the recommended "xhigh" coding/agentic default. + * Used by Anthropic adaptive models with a real xhigh tier (Opus 4.7+ and + * Fable/Mythos 5 on the Messages API) and by GPT-5.6+ wire-effort models, + * which expose the same genuine `max` tier above `xhigh`. */ -export const ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER: Readonly>> = { +export const SHIFTED_FIVE_TIER_EFFORT_MAP: Readonly>> = { [Effort.Minimal]: "low", [Effort.Low]: "medium", [Effort.Medium]: "high", @@ -295,6 +298,27 @@ function isOpenAICompatReasoningApi(api: Api): boolean { return api === "openai-completions" || api === "openrouter"; } +/** + * GPT-5.6+ addressed through a wire `reasoning.effort`/`reasoning_effort` + * field, where the shifted five-tier map applies. Devin (`devin-agent`) + * selects effort by routing to per-tier sibling model ids instead and must + * stay unmapped. + */ +function isGpt56PlusWireEffortModel(spec: ModelSpec): boolean { + switch (spec.api) { + case "openai-responses": + case "openai-codex-responses": + case "azure-openai-responses": + case "openai-completions": + case "openrouter": + break; + default: + return false; + } + const parsed = parseOpenAIModel(bareModelId(spec.id)); + return parsed !== null && semverGte(parsed.version, "5.6"); +} + function getModelDefinedEfforts( spec: ModelSpec, compat: CompatOf, @@ -313,6 +337,12 @@ function getModelDefinedEfforts( if (isSakanaFuguReasoningModel(spec)) { return FUGU_REASONING_EFFORTS; } + if (isGpt56PlusWireEffortModel(spec)) { + // Normalize stale baked/discovered `low..xhigh` surfaces to the full + // five-tier ladder so the shifted map keeps the native `low` tier + // reachable (user `minimal`). + return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + } return isOpenAICompatReasoningApi(spec.api) && (isMinimaxM2FamilyModelId(spec.id) || isOpenAIGptOssModelId(spec.id) || @@ -373,7 +403,7 @@ function inferDetectedEffortMap( return MINIMAX_ANTHROPIC_ADAPTIVE_EFFORT_MAP; } return anthropicModelHasRealXHighEffort(spec, parsedModel) - ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER + ? SHIFTED_FIVE_TIER_EFFORT_MAP : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; } // GLM-5.2 coding SKUs accept `reasoning_effort`, but the effort dialect is @@ -397,6 +427,9 @@ function inferDetectedEffortMap( if (isSakanaFuguReasoningModel(spec)) { return FUGU_REASONING_EFFORT_MAP; } + if (isGpt56PlusWireEffortModel(spec)) { + return SHIFTED_FIVE_TIER_EFFORT_MAP; + } if (!isOpenAICompatReasoningApi(spec.api)) { return undefined; } @@ -446,7 +479,7 @@ function getOpenRouterAnthropicReasoningEffortMap(modelId: string): EffortMap | if (!isAnthropicAdaptiveGenAtLeast(parsed, "4.6")) return undefined; const hasRealXHigh = isAnthropicAdaptiveGenAtLeast(parsed, "4.7"); - return hasRealXHigh ? ANTHROPIC_ADAPTIVE_EFFORT_MAP_5_TIER : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; + return hasRealXHigh ? SHIFTED_FIVE_TIER_EFFORT_MAP : ANTHROPIC_ADAPTIVE_EFFORT_MAP_4_TIER; } function inferSupportedEfforts( @@ -474,6 +507,11 @@ function inferOpenAISupportedEfforts(model: OpenAIModel): readonly Effort[] { if (model.variant === "codex-mini" && semverEqual(model.version, "5.1")) { return GPT_5_1_CODEX_MINI_EFFORTS; } + // 5.6+ exposes the full five-tier ladder: the shifted wire map spans + // low..max, with user `minimal` reaching the native `low` tier. + if (semverGte(model.version, "5.6")) { + return DEFAULT_REASONING_EFFORTS_WITH_XHIGH; + } if (semverGte(model.version, "5.2")) { return GPT_5_2_PLUS_EFFORTS; } diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index 0d39e022c..55d87779c 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -17214,9 +17214,9 @@ }, "requestModelId": "gpt-5-5-none-priority" }, - "gpt-5-6-luna-high": { - "id": "gpt-5-6-luna-high", - "name": "GPT-5.6 Luna High Thinking", + "gpt-5-6-luna": { + "id": "gpt-5-6-luna", + "name": "GPT-5.6 Luna", "api": "devin-agent", "provider": "devin", "baseUrl": "https://server.codeium.com", @@ -17233,11 +17233,30 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 64000 + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-luna-none", + "minimal": "gpt-5-6-luna-low", + "low": "gpt-5-6-luna-medium", + "medium": "gpt-5-6-luna-high", + "high": "gpt-5-6-luna-xhigh", + "xhigh": "gpt-5-6-luna-max" + } + }, + "requestModelId": "gpt-5-6-luna-none" }, - "gpt-5-6-luna-high-priority": { - "id": "gpt-5-6-luna-high-priority", - "name": "GPT-5.6 Luna High Thinking Fast", + "gpt-5-6-luna-fast": { + "id": "gpt-5-6-luna-fast", + "name": "GPT-5.6 Luna Fast", "api": "devin-agent", "provider": "devin", "baseUrl": "https://server.codeium.com", @@ -17254,11 +17273,30 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 64000 + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-luna-none-priority", + "minimal": "gpt-5-6-luna-low-priority", + "low": "gpt-5-6-luna-low-priority", + "medium": "gpt-5-6-luna-medium-priority", + "high": "gpt-5-6-luna-high-priority", + "xhigh": "gpt-5-6-luna-xhigh-priority" + } + }, + "requestModelId": "gpt-5-6-luna-none-priority" }, - "gpt-5-6-luna-low": { - "id": "gpt-5-6-luna-low", - "name": "GPT-5.6 Luna Low Thinking", + "gpt-5-6-sol": { + "id": "gpt-5-6-sol", + "name": "GPT-5.6 Sol", "api": "devin-agent", "provider": "devin", "baseUrl": "https://server.codeium.com", @@ -17275,11 +17313,30 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 64000 + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-sol-none", + "minimal": "gpt-5-6-sol-low", + "low": "gpt-5-6-sol-medium", + "medium": "gpt-5-6-sol-high", + "high": "gpt-5-6-sol-xhigh", + "xhigh": "gpt-5-6-sol-max" + } + }, + "requestModelId": "gpt-5-6-sol-none" }, - "gpt-5-6-luna-low-priority": { - "id": "gpt-5-6-luna-low-priority", - "name": "GPT-5.6 Luna Low Thinking Fast", + "gpt-5-6-sol-fast": { + "id": "gpt-5-6-sol-fast", + "name": "GPT-5.6 Sol Fast", "api": "devin-agent", "provider": "devin", "baseUrl": "https://server.codeium.com", @@ -17296,11 +17353,30 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 64000 + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-sol-none-priority", + "minimal": "gpt-5-6-sol-low-priority", + "low": "gpt-5-6-sol-low-priority", + "medium": "gpt-5-6-sol-medium-priority", + "high": "gpt-5-6-sol-high-priority", + "xhigh": "gpt-5-6-sol-xhigh-priority" + } + }, + "requestModelId": "gpt-5-6-sol-none-priority" }, - "gpt-5-6-luna-max": { - "id": "gpt-5-6-luna-max", - "name": "GPT-5.6 Luna Max Thinking", + "gpt-5-6-terra": { + "id": "gpt-5-6-terra", + "name": "GPT-5.6 Terra", "api": "devin-agent", "provider": "devin", "baseUrl": "https://server.codeium.com", @@ -17317,11 +17393,30 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 64000 + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-terra-none", + "minimal": "gpt-5-6-terra-low", + "low": "gpt-5-6-terra-medium", + "medium": "gpt-5-6-terra-high", + "high": "gpt-5-6-terra-xhigh", + "xhigh": "gpt-5-6-terra-max" + } + }, + "requestModelId": "gpt-5-6-terra-none" }, - "gpt-5-6-luna-medium": { - "id": "gpt-5-6-luna-medium", - "name": "GPT-5.6 Luna Medium Thinking", + "gpt-5-6-terra-fast": { + "id": "gpt-5-6-terra-fast", + "name": "GPT-5.6 Terra Fast", "api": "devin-agent", "provider": "devin", "baseUrl": "https://server.codeium.com", @@ -17338,574 +17433,26 @@ "cacheWrite": 0 }, "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-luna-medium-priority": { - "id": "gpt-5-6-luna-medium-priority", - "name": "GPT-5.6 Luna Medium Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortRouting": { + "off": "gpt-5-6-terra-none-priority", + "minimal": "gpt-5-6-terra-low-priority", + "low": "gpt-5-6-terra-low-priority", + "medium": "gpt-5-6-terra-medium-priority", + "high": "gpt-5-6-terra-high-priority", + "xhigh": "gpt-5-6-terra-xhigh-priority" + } }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-luna-none": { - "id": "gpt-5-6-luna-none", - "name": "GPT-5.6 Luna No Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-luna-none-priority": { - "id": "gpt-5-6-luna-none-priority", - "name": "GPT-5.6 Luna No Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-luna-xhigh": { - "id": "gpt-5-6-luna-xhigh", - "name": "GPT-5.6 Luna XHigh Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-luna-xhigh-priority": { - "id": "gpt-5-6-luna-xhigh-priority", - "name": "GPT-5.6 Luna XHigh Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-high": { - "id": "gpt-5-6-sol-high", - "name": "GPT-5.6 Sol High Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-high-priority": { - "id": "gpt-5-6-sol-high-priority", - "name": "GPT-5.6 Sol High Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-low": { - "id": "gpt-5-6-sol-low", - "name": "GPT-5.6 Sol Low Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-low-priority": { - "id": "gpt-5-6-sol-low-priority", - "name": "GPT-5.6 Sol Low Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-max": { - "id": "gpt-5-6-sol-max", - "name": "GPT-5.6 Sol Max Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-medium": { - "id": "gpt-5-6-sol-medium", - "name": "GPT-5.6 Sol Medium Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-medium-priority": { - "id": "gpt-5-6-sol-medium-priority", - "name": "GPT-5.6 Sol Medium Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-none": { - "id": "gpt-5-6-sol-none", - "name": "GPT-5.6 Sol No Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-none-priority": { - "id": "gpt-5-6-sol-none-priority", - "name": "GPT-5.6 Sol No Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-xhigh": { - "id": "gpt-5-6-sol-xhigh", - "name": "GPT-5.6 Sol XHigh Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-sol-xhigh-priority": { - "id": "gpt-5-6-sol-xhigh-priority", - "name": "GPT-5.6 Sol XHigh Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-high": { - "id": "gpt-5-6-terra-high", - "name": "GPT-5.6 Terra High Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-high-priority": { - "id": "gpt-5-6-terra-high-priority", - "name": "GPT-5.6 Terra High Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-low": { - "id": "gpt-5-6-terra-low", - "name": "GPT-5.6 Terra Low Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-low-priority": { - "id": "gpt-5-6-terra-low-priority", - "name": "GPT-5.6 Terra Low Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-max": { - "id": "gpt-5-6-terra-max", - "name": "GPT-5.6 Terra Max Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-medium": { - "id": "gpt-5-6-terra-medium", - "name": "GPT-5.6 Terra Medium Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-medium-priority": { - "id": "gpt-5-6-terra-medium-priority", - "name": "GPT-5.6 Terra Medium Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-none": { - "id": "gpt-5-6-terra-none", - "name": "GPT-5.6 Terra No Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-none-priority": { - "id": "gpt-5-6-terra-none-priority", - "name": "GPT-5.6 Terra No Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": false, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-xhigh": { - "id": "gpt-5-6-terra-xhigh", - "name": "GPT-5.6 Terra XHigh Thinking", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 - }, - "gpt-5-6-terra-xhigh-priority": { - "id": "gpt-5-6-terra-xhigh-priority", - "name": "GPT-5.6 Terra XHigh Thinking Fast", - "api": "devin-agent", - "provider": "devin", - "baseUrl": "https://server.codeium.com", - "reasoning": true, - "input": [ - "text", - "image" - ], - "supportsTools": true, - "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 - }, - "contextWindow": 1000000, - "maxTokens": 64000 + "requestModelId": "gpt-5-6-terra-none-priority" }, "kimi-k2-6": { "id": "kimi-k2-6", @@ -60549,6 +60096,120 @@ }, "contextPromotionTarget": "openai/gpt-5.4" }, + "gpt-5.6-luna": { + "id": "gpt-5.6-luna", + "name": "GPT-5.6 Luna", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-sol": { + "id": "gpt-5.6-sol", + "name": "GPT-5.6 Sol", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, + "gpt-5.6-terra": { + "id": "gpt-5.6-terra", + "name": "GPT-5.6 Terra", + "api": "openai-responses", + "provider": "openai", + "baseUrl": "https://api.openai.com/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 + }, + "contextWindow": 1050000, + "maxTokens": 128000, + "applyPatchToolType": "freeform", + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } + } + }, "o1": { "id": "o1", "name": "o1", @@ -61330,7 +60991,7 @@ }, "gpt-5.6-luna": { "id": "gpt-5.6-luna", - "name": "GPT-5.6-Luna", + "name": "GPT-5.6 Luna", "api": "openai-codex-responses", "provider": "openai-codex", "baseUrl": "https://chatgpt.com/backend-api", @@ -61340,10 +61001,10 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 + "input": 1, + "output": 6, + "cacheRead": 0.1, + "cacheWrite": 1.25 }, "remoteCompaction": { "enabled": true, @@ -61358,16 +61019,24 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "gpt-5.6-sol": { "id": "gpt-5.6-sol", - "name": "GPT-5.6-Sol", + "name": "GPT-5.6 Sol", "api": "openai-codex-responses", "provider": "openai-codex", "baseUrl": "https://chatgpt.com/backend-api", @@ -61377,10 +61046,10 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 6.25 }, "remoteCompaction": { "enabled": true, @@ -61395,16 +61064,24 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "gpt-5.6-terra": { "id": "gpt-5.6-terra", - "name": "GPT-5.6-Terra", + "name": "GPT-5.6 Terra", "api": "openai-codex-responses", "provider": "openai-codex", "baseUrl": "https://chatgpt.com/backend-api", @@ -61414,10 +61091,10 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, - "cacheWrite": 0 + "input": 2.5, + "output": 15, + "cacheRead": 0.25, + "cacheWrite": 3.125 }, "remoteCompaction": { "enabled": true, @@ -61432,11 +61109,19 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } } }, @@ -69869,11 +69554,19 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "openai/gpt-5.6-luna-pro": { @@ -69898,11 +69591,19 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "openai/gpt-5.6-sol": { @@ -69927,11 +69628,19 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "openai/gpt-5.6-sol-pro": { @@ -69956,11 +69665,19 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "openai/gpt-5.6-terra": { @@ -69985,11 +69702,19 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "openai/gpt-5.6-terra-pro": { @@ -70014,11 +69739,19 @@ "thinking": { "mode": "effort", "efforts": [ + "minimal", "low", "medium", "high", "xhigh" - ] + ], + "effortMap": { + "minimal": "low", + "low": "medium", + "medium": "high", + "high": "xhigh", + "xhigh": "max" + } } }, "openai/gpt-audio": { @@ -73625,13 +73358,13 @@ "text" ], "cost": { - "input": 0.546, - "output": 1.716, - "cacheRead": 0.10139999999999999, + "input": 0.54, + "output": 1.76, + "cacheRead": 0.09999999999999999, "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 128000, + "maxTokens": 101376, "thinking": { "mode": "effort", "efforts": [ @@ -75776,7 +75509,7 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 1048576, + "contextWindow": 1000000, "maxTokens": 1048576, "compat": { "supportsUsageInStreaming": false @@ -82443,6 +82176,7 @@ "thinking": { "mode": "budget", "efforts": [ + "minimal", "low", "medium", "high", @@ -82472,6 +82206,7 @@ "thinking": { "mode": "budget", "efforts": [ + "minimal", "low", "medium", "high", @@ -82501,6 +82236,7 @@ "thinking": { "mode": "budget", "efforts": [ + "minimal", "low", "medium", "high", diff --git a/packages/catalog/src/variant-collapse.ts b/packages/catalog/src/variant-collapse.ts index b3042503d..46d1f9d52 100644 --- a/packages/catalog/src/variant-collapse.ts +++ b/packages/catalog/src/variant-collapse.ts @@ -113,6 +113,14 @@ function thinkingPair(baseId: string, name: string): EffortVariantFamily { type DevinTierRoutes = Partial>; +const DEVIN_FIVE_TIER_EFFORTS: readonly Effort[] = [ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, +]; + function devinTierFamily( id: string, name: string, @@ -160,6 +168,44 @@ function devinTierFamily( }; } +/** + * GPT-5.6 (Luna/Sol/Terra) adds a genuine `max` tier above `xhigh`, so the + * standard family shifts every user effort up one notch (`minimal` → `-low` + * … `xhigh` → `-max`), mirroring the Opus 4.7+ five-tier mapping. Devin + * serves no `-max-priority` sibling, so the fast family keeps the direct + * `low..xhigh` `-priority` scale. + */ +function devinGpt56Families(variant: "luna" | "sol" | "terra", name: string): readonly EffortVariantFamily[] { + const base = `gpt-5-6-${variant}`; + return [ + devinTierFamily( + base, + name, + { + off: `${base}-none`, + minimal: `${base}-low`, + low: `${base}-medium`, + medium: `${base}-high`, + high: `${base}-xhigh`, + xhigh: `${base}-max`, + }, + DEVIN_FIVE_TIER_EFFORTS, + ), + devinTierFamily( + `${base}-fast`, + `${name} Fast`, + { + off: `${base}-none-priority`, + low: `${base}-low-priority`, + medium: `${base}-medium-priority`, + high: `${base}-high-priority`, + xhigh: `${base}-xhigh-priority`, + }, + DEVIN_FIVE_TIER_EFFORTS, + ), + ]; +} + const GEMINI_3_FLASH_FAMILY_EFFORTS: readonly Effort[] = [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High]; const GEMINI_3_PRO_FAMILY_EFFORTS: readonly Effort[] = [Effort.Low, Effort.High]; @@ -330,7 +376,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -353,7 +399,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -376,7 +422,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -399,7 +445,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { }, thinking: { mode: "effort", - efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + efforts: DEVIN_FIVE_TIER_EFFORTS, requiresEffort: true, }, }, @@ -413,7 +459,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "MODEL_GPT_5_2_HIGH", xhigh: "MODEL_GPT_5_2_XHIGH", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex", @@ -424,7 +470,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high", xhigh: "gpt-5-3-codex-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-3-codex-fast", @@ -435,7 +481,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-3-codex-high-priority", xhigh: "gpt-5-3-codex-xhigh-priority", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4", @@ -447,7 +493,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high", xhigh: "gpt-5-4-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-fast", @@ -459,7 +505,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-high-priority", xhigh: "gpt-5-4-xhigh-priority", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-4-mini", @@ -470,7 +516,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-4-mini-high", xhigh: "gpt-5-4-mini-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5", @@ -482,7 +528,7 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high", xhigh: "gpt-5-5-xhigh", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), devinTierFamily( "gpt-5-5-fast", @@ -494,8 +540,11 @@ export const DEVIN_VARIANT_COLLAPSE_TABLE: VariantCollapseTable = { high: "gpt-5-5-high-priority", xhigh: "gpt-5-5-xhigh-priority", }, - [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + DEVIN_FIVE_TIER_EFFORTS, ), + ...devinGpt56Families("luna", "GPT-5.6 Luna"), + ...devinGpt56Families("sol", "GPT-5.6 Sol"), + ...devinGpt56Families("terra", "GPT-5.6 Terra"), devinTierFamily( "gemini-3-1-pro", "Gemini 3.1 Pro", diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index 6d6019fe1..719d8c59e 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -567,6 +567,96 @@ describe("model thinking derivation", () => { expect(getSupportedEfforts(model)).toEqual([]); expect(clampThinkingLevelForModel(model, Effort.High)).toBeUndefined(); }); + + it("bakes the GPT-5.6 shifted five-tier effort map on wire-effort APIs", () => { + const codex = createModel({ + id: "gpt-5.6-sol", + api: "openai-codex-responses", + provider: "openai-codex", + }); + + expect(codex.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "low", + low: "medium", + medium: "high", + high: "xhigh", + xhigh: "max", + }, + }); + + // Stale baked four-tier metadata (caches/discovery) normalizes back to + // the five-tier ladder with the map attached — the wire-defaults + // backfill path — and namespaced OpenRouter ids parse. + const staleOpenRouter = createModel({ + id: "openai/gpt-5.6-terra", + api: "openrouter", + provider: "openrouter", + baseUrl: "https://openrouter.ai/api/v1", + thinking: { + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + }, + }); + + expect(staleOpenRouter.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortMap: { + minimal: "low", + low: "medium", + medium: "high", + high: "xhigh", + xhigh: "max", + }, + }); + }); + + it("keeps pre-5.6 and Devin-routed GPT models off the shifted effort map", () => { + const gpt55 = createModel({ + id: "gpt-5.5", + api: "openai-responses", + provider: "openai", + }); + + expect(gpt55.thinking).toEqual({ + mode: "effort", + efforts: [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + }); + expect(gpt55.thinking?.effortMap).toBeUndefined(); + + // Devin selects effort by routing to per-tier sibling model ids, never + // via a wire reasoning.effort field — the shifted map must not attach. + const devin = createModel({ + id: "gpt-5-6-sol", + api: "devin-agent", + provider: "devin", + baseUrl: "https://server.codeium.com", + thinking: { + mode: "effort", + efforts: [Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh], + effortRouting: { + off: "gpt-5-6-sol-none", + minimal: "gpt-5-6-sol-low", + low: "gpt-5-6-sol-medium", + medium: "gpt-5-6-sol-high", + high: "gpt-5-6-sol-xhigh", + xhigh: "gpt-5-6-sol-max", + }, + }, + }); + + expect(devin.thinking?.effortMap).toBeUndefined(); + expect(devin.thinking?.efforts).toEqual([ + Effort.Minimal, + Effort.Low, + Effort.Medium, + Effort.High, + Effort.XHigh, + ]); + }); }); describe("model thinking runtime helpers", () => { From 4a20b51ca8024868a187e32a347f5269b134cc3c Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 9 Jul 2026 20:35:43 +0200 Subject: [PATCH 40/59] feat: implemented auto-sealing for transcript blocks and TUI row emission - Added auto-sealing logic to `FinalizableBlock` to finalize displaceable snapshots when they enter the scrollback area. - Updated TUI frame emission to publish committed rows and clamp them to segment bounds, ensuring accurate component updates. - Introduced component tracking and cleanup in event controller tests to prevent resource leaks during finalization. - Validated state transitions and post-emit synchronization through comprehensive new test suites for transcript and TUI components. --- packages/catalog/test/model-thinking.test.ts | 8 +- packages/coding-agent/CHANGELOG.md | 3 +- .../modes/components/transcript-container.ts | 33 ++++ .../src/prompts/system/workflow-notice.md | 10 +- .../components/transcript-container.test.ts | 160 ++++++++++++++++++ ...event-controller-toolcall-finalize.test.ts | 9 +- packages/tui/CHANGELOG.md | 4 + packages/tui/src/tui.ts | 31 +++- .../test/streaming-scrollback-defer.test.ts | 141 +++++++++++++++ 9 files changed, 382 insertions(+), 17 deletions(-) diff --git a/packages/catalog/test/model-thinking.test.ts b/packages/catalog/test/model-thinking.test.ts index 719d8c59e..a21a323fc 100644 --- a/packages/catalog/test/model-thinking.test.ts +++ b/packages/catalog/test/model-thinking.test.ts @@ -649,13 +649,7 @@ describe("model thinking derivation", () => { }); expect(devin.thinking?.effortMap).toBeUndefined(); - expect(devin.thinking?.efforts).toEqual([ - Effort.Minimal, - Effort.Low, - Effort.Medium, - Effort.High, - Effort.XHigh, - ]); + expect(devin.thinking?.efforts).toEqual([Effort.Minimal, Effort.Low, Effort.Medium, Effort.High, Effort.XHigh]); }); }); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 636167fe5..896d13193 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,9 +4,10 @@ ### Fixed +- Fixed issue where unfinalized tool blocks could incorrectly pin the live-region scroll seam + - Improved rendering of raw thinking blocks by stripping empty HTML comment noise - Fixed display of thinking blocks consisting entirely of hidden comment noise - - Fixed gpt-5.6 reasoning summaries rendering literal `` sentinel lines in thinking blocks; empty HTML comments (and the unterminated `` sentinel lines in thinking blocks; empty HTML comments (and the unterminated `