From b74f7bb720a89b1bf1f6a86d48d3f9e63bf89280 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 21 Jun 2026 07:21:22 +0000 Subject: [PATCH 01/43] fix(compaction): trigger goal-mode threshold on billed context, not post-prune estimate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pruning frees bytes for the NEXT prompt — it does not change the size of the prompt the LLM just billed for. Subtracting the per-turn `#pruneStaleToolResults` / `#pruneToolOutputs` savings from the threshold input let a long-running `/goal` session sit above `compaction.thresholdTokens` indefinitely: the visible context (anchored to the same provider billing) showed >threshold, but `shouldCompact` no-op'd because the subtraction dropped the input below the trigger. The `compactionContextTokens` floor against the post-prune local estimate is still applied, so a payload-compression hook still can't deflate the trigger. Regression test seeds one large `useless` tool result whose suffix sits inside the 8k cache-warm window so `#pruneStaleToolResults` actually returns ≥20k savings, then asserts compaction fires when the final turn bills 91k tokens against the reporter's `thresholdTokens: 76384`. Fixes #3174 --- packages/coding-agent/CHANGELOG.md | 4 + .../coding-agent/src/session/agent-session.ts | 31 ++-- ...gent-session-auto-compaction-queue.test.ts | 145 ++++++++++++++++++ 3 files changed, 165 insertions(+), 15 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 001c32bcd..8bfa5e333 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `/goal` threshold auto-compaction never firing whenever the per-turn supersede/drop-useless prune saved ≥`compaction.thresholdTokens - calculateContextTokens(usage)` tokens. The pre-fix code subtracted prune savings from the threshold input, so a long-running goal session whose visible context (anchored to the same provider billing) sat above `compaction.thresholdTokens` could keep growing past it indefinitely. Threshold maintenance now triggers from the actual last-turn billed context, with the post-prune local estimate kept as a payload-compression floor. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) + ## [16.1.10] - 2026-06-21 ### Added diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index fa5146cb8..7902a02c3 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -8323,7 +8323,7 @@ export class AgentSession { // Stale-result pass runs every turn, before any threshold gating: it is // cheap (bails when no candidate) and independent of the compaction // setting. - const supersedeResult = await this.#pruneStaleToolResults(); + await this.#pruneStaleToolResults(); const compactionSettings = this.settings.getGroup("compaction"); if (!compactionSettings.enabled || compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; @@ -8331,20 +8331,21 @@ export class AgentSession { // Case 4: Threshold - turn succeeded but context is getting large // Skip if this was an error (non-overflow errors don't have usage data) if (assistantMessage.stopReason === "error") return COMPACTION_CHECK_NONE; - const pruneResult = await this.#pruneToolOutputs(); - let contextTokens = calculateContextTokens(assistantMessage.usage); - if (supersedeResult) { - contextTokens = Math.max(0, contextTokens - supersedeResult.tokensSaved); - } - if (pruneResult) { - contextTokens = Math.max(0, contextTokens - pruneResult.tokensSaved); - } - // Floor by the real stored-conversation estimate so a payload-shrinking - // before_provider_request hook (e.g. a compression extension such as - // Headroom) can't deflate the provider-reported usage below the true - // history size and skip the threshold. The estimate runs after the prune - // passes above, so it reflects the post-prune message set. - contextTokens = compactionContextTokens(contextTokens, this.#estimateStoredContextTokens()); + await this.#pruneToolOutputs(); + // Pruning frees bytes for the NEXT prompt; it does not change the size of + // the prompt the LLM just billed for. Earlier revisions subtracted the + // per-turn supersede/prune `tokensSaved` from the threshold input, which + // let a long-running `/goal` session sit above `compaction.thresholdTokens` + // indefinitely whenever per-turn pruning saved enough to drop the + // post-prune estimate below the user-configured trigger — the visible + // context (anchored to the same provider billing) still showed >threshold, + // but `shouldCompact` no-op'd (#3174). Anchor on the last turn's billed + // context tokens, floored by the post-prune stored-conversation estimate + // so a payload-compression hook still can't deflate the trigger. + const contextTokens = compactionContextTokens( + calculateContextTokens(assistantMessage.usage), + this.#estimateStoredContextTokens(), + ); if (shouldCompact(contextTokens, contextWindow, compactionSettings)) { // Try promotion first — if a larger model is available, switch instead of compacting const promoted = await this.#tryContextPromotion(assistantMessage); diff --git a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts index e7a15265e..cb86914e8 100644 --- a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts +++ b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts @@ -273,6 +273,151 @@ describe("AgentSession auto-compaction queue resume", () => { expect(runtimeSignals.some(signal => signal.startsWith("compaction:end:"))).toBe(true); }); + it("triggers threshold compaction in active goals even when per-turn pruning shaves the post-prune estimate below threshold", async () => { + // Regression for #3174. Goal mode is the most common scenario: the agent + // runs many tool-result-heavy turns and the per-turn "useless" / + // "supersede" passes shave tokens off every check. Pre-fix + // `#checkCompaction` subtracted those savings from the threshold input, so + // with the reporter's fixed `compaction.thresholdTokens: 76384`, the + // threshold input fell below the trigger even when the provider-billed + // prompt (and the visible context anchored to it) sat above 90k tokens — + // auto-compaction silently no-op'd indefinitely while the loop kept + // running. + // + // This seeds one large `useless` tool result whose suffix sits inside the + // 8k cache-warm window so `#pruneStaleToolResults` actually returns ≥20k + // savings (well above the buggy code's mis-subtraction needed to drop + // 91000 below 76384). Compaction MUST still fire because the last turn's + // billed context tokens (91k) are above the configured threshold. + const now = Date.now(); + + // Seed: small user, small toolCall, ONE big useless tool result, then a + // handful of small turns that keep the suffix after the big result under + // the 8000-token cache-warm cutoff. The big result is the only viable + // prune candidate, and it alone saves well over 20k tokens — enough to + // drag the pre-fix threshold input from 91k well below 76384. + sessionManager.appendMessage({ + role: "user", + content: "Investigate every module of the project.", + timestamp: now - 200, + }); + const bigCallId = "call-big-useless"; + sessionManager.appendMessage({ + role: "assistant", + content: [{ type: "toolCall", id: bigCallId, name: "search", arguments: { pattern: "TODO" } }], + api: "anthropic-messages", + provider: "anthropic", + model: "claude-sonnet-4-5", + stopReason: "toolUse", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: now - 180, + }); + sessionManager.appendMessage({ + role: "toolResult", + toolCallId: bigCallId, + toolName: "search", + content: [{ type: "text", text: "match line\n".repeat(20000) }], // ~40k+ tokens + isError: false, + useless: true, + timestamp: now - 170, + }); + // A few small follow-up turns so the big result's suffix stays inside the + // 8000-token cache-warm window. Each pair is well under a hundred tokens. + for (let i = 0; i < 4; i++) { + const smallId = `call-small-${i}`; + const ts = now - 160 + i * 2; + sessionManager.appendMessage({ + role: "assistant", + content: [{ type: "toolCall", id: smallId, name: "read", arguments: { path: `note-${i}.md` } }], + api: "anthropic-messages", + provider: "anthropic", + model: "claude-sonnet-4-5", + stopReason: "toolUse", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: ts, + }); + sessionManager.appendMessage({ + role: "toolResult", + toolCallId: smallId, + toolName: "read", + content: [{ type: "text", text: `tiny note ${i}` }], + isError: false, + timestamp: ts + 1, + }); + } + session.agent.replaceMessages(session.buildDisplaySessionContext().messages); + + session.setGoalModeState({ + enabled: true, + mode: "active", + goal: { + id: "goal-threshold-pruneable", + objective: "continue until compacted", + status: "active", + tokensUsed: 0, + timeUsedSeconds: 0, + createdAt: now, + updatedAt: now, + }, + }); + + vi.spyOn(session.agent, "continue").mockImplementation(async () => { + session.agent.clearAllQueues(); + }); + + session.settings.set("compaction.thresholdTokens", 76384); + session.settings.set("compaction.thresholdPercent", -1); + session.settings.set("compaction.strategy", "context-full"); + session.settings.set("compaction.dropUseless", true); + session.settings.set("compaction.supersedeReads", true); + session.settings.set("compaction.keepRecentTokens", 10000); + session.settings.set("compaction.reserveTokens", 16384); + + // Final assistant turn: billed at ~91k context tokens, just over the + // reporter's threshold. The pre-fix code would have subtracted ≥20k of + // prune savings and dropped the threshold input below 76384, skipping + // compaction. Post-fix it must trigger. + const finalAssistant = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "Investigated module-7; continuing." }], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + stopReason: "stop" as const, + usage: { + input: 5000, + output: 1000, + cacheRead: 85000, + cacheWrite: 0, + totalTokens: 91000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: now, + }; + + session.agent.emitExternalEvent({ type: "message_end", message: finalAssistant }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [finalAssistant] }); + + await session.waitForIdle(); + + const runtimeSignals = getRuntimeSignals(); + expect(runtimeSignals).toContain("compaction:start:threshold"); + expect(runtimeSignals.some(signal => signal.startsWith("compaction:end:"))).toBe(true); + }); it("has isCompacting true when the auto_compaction_start event fires", async () => { // Defect 1: the compaction AbortController (which backs isCompacting) must be // installed before auto_compaction_start is emitted. If it is installed after, From 8ab754f63601ac66e460d85ea9480a53f2125c56 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 21 Jun 2026 08:35:58 +0000 Subject: [PATCH 02/43] fix(compaction): run goal threshold maintenance before retry continuations Active goal turns that stopped with text could hit the empty/unexpected-stop continuation guards before threshold maintenance. When those guards scheduled another goal turn, #checkCompaction never ran, so no auto_compaction_start was emitted even while visible context stayed above thresholdTokens. Run threshold maintenance once before those active-goal self-continuations and log the threshold decision inputs: billed context, stored estimate, resolved trigger tokens, post-maintenance tokens, strategy, threshold, promotion state, and shouldCompact. Also pass post-prune maintenance tokens into the shake recovery-band check so supersede/drop-useless savings are preserved when deciding whether shake still needs to fall back to context-full compaction. Fixes #3174 --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/session/agent-session.ts | 92 ++++++++++++++----- ...gent-session-auto-compaction-queue.test.ts | 54 +++++++++++ packages/coding-agent/test/shake.test.ts | 64 +++++++++++++ 4 files changed, 188 insertions(+), 24 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 8bfa5e333..c52a42930 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `/goal` threshold auto-compaction never firing whenever the per-turn supersede/drop-useless prune saved ≥`compaction.thresholdTokens - calculateContextTokens(usage)` tokens. The pre-fix code subtracted prune savings from the threshold input, so a long-running goal session whose visible context (anchored to the same provider billing) sat above `compaction.thresholdTokens` could keep growing past it indefinitely. Threshold maintenance now triggers from the actual last-turn billed context, with the post-prune local estimate kept as a payload-compression floor. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) +- Fixed `/goal` threshold auto-compaction skipping real sessions through two separate paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context, and active-goal text stops now attempt threshold maintenance before empty/unexpected-stop retry continuations can return from post-turn handling. The threshold decision also logs the billed, stored, resolved, and post-maintenance token counts so future no-start reports identify the exact skip path. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) ## [16.1.10] - 2026-06-21 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 7902a02c3..36dac06fa 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2790,9 +2790,10 @@ export class AgentSession { return; } + const activeGoal = this.#goalModeState?.enabled === true && this.#goalModeState.goal.status === "active"; if (this.#assistantEndedWithSuccessfulYield(msg)) { this.#lastSuccessfulYieldToolCallId = undefined; - if (this.#goalModeState?.enabled && this.#goalModeState.goal.status === "active") { + if (activeGoal) { const compactionTask = this.#checkCompaction(msg); this.#trackPostPromptTask(compactionTask); await compactionTask; @@ -2802,6 +2803,19 @@ export class AgentSession { } this.#lastSuccessfulYieldToolCallId = undefined; + let compactionResult = COMPACTION_CHECK_NONE; + let checkedCompaction = false; + if (activeGoal) { + const compactionTask = this.#checkCompaction(msg); + this.#trackPostPromptTask(compactionTask); + compactionResult = await compactionTask; + checkedCompaction = true; + if (compactionResult.deferredHandoff || compactionResult.continuationScheduled) { + await emitAgentEndNotification(); + return; + } + } + if (await this.#handleEmptyAssistantStop(msg)) { await emitAgentEndNotification(); return; @@ -2846,9 +2860,11 @@ export class AgentSession { } this.#resolveRetry(); - const compactionTask = this.#checkCompaction(msg); - this.#trackPostPromptTask(compactionTask); - const compactionResult = await compactionTask; + if (!checkedCompaction) { + const compactionTask = this.#checkCompaction(msg); + this.#trackPostPromptTask(compactionTask); + compactionResult = await compactionTask; + } // Check for incomplete todos only after a final assistant stop, not intermediate tool-use turns. const hasToolCalls = msg.content.some(content => content.type === "toolCall"); if (hasToolCalls) { @@ -8323,7 +8339,7 @@ export class AgentSession { // Stale-result pass runs every turn, before any threshold gating: it is // cheap (bails when no candidate) and independent of the compaction // setting. - await this.#pruneStaleToolResults(); + const supersedeResult = await this.#pruneStaleToolResults(); const compactionSettings = this.settings.getGroup("compaction"); if (!compactionSettings.enabled || compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; @@ -8331,7 +8347,10 @@ export class AgentSession { // Case 4: Threshold - turn succeeded but context is getting large // Skip if this was an error (non-overflow errors don't have usage data) if (assistantMessage.stopReason === "error") return COMPACTION_CHECK_NONE; - await this.#pruneToolOutputs(); + const pruneResult = await this.#pruneToolOutputs(); + const maintenanceTokensFreed = (supersedeResult?.tokensSaved ?? 0) + (pruneResult?.tokensSaved ?? 0); + const assistantUsageContextTokens = calculateContextTokens(assistantMessage.usage); + const storedContextTokens = this.#estimateStoredContextTokens(); // Pruning frees bytes for the NEXT prompt; it does not change the size of // the prompt the LLM just billed for. Earlier revisions subtracted the // per-turn supersede/prune `tokensSaved` from the threshold input, which @@ -8339,22 +8358,48 @@ export class AgentSession { // indefinitely whenever per-turn pruning saved enough to drop the // post-prune estimate below the user-configured trigger — the visible // context (anchored to the same provider billing) still showed >threshold, - // but `shouldCompact` no-op'd (#3174). Anchor on the last turn's billed - // context tokens, floored by the post-prune stored-conversation estimate - // so a payload-compression hook still can't deflate the trigger. - const contextTokens = compactionContextTokens( - calculateContextTokens(assistantMessage.usage), - this.#estimateStoredContextTokens(), + // but `shouldCompact` no-op'd (#3174). Anchor the initial trigger on the + // last turn's billed context tokens, floored by the post-prune + // stored-conversation estimate so a payload-compression hook still can't + // deflate the trigger. + const contextTokens = compactionContextTokens(assistantUsageContextTokens, storedContextTokens); + const postMaintenanceContextTokens = compactionContextTokens( + Math.max(0, assistantUsageContextTokens - maintenanceTokensFreed), + storedContextTokens, ); - if (shouldCompact(contextTokens, contextWindow, compactionSettings)) { + const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings); + const shouldThresholdCompact = shouldCompact(contextTokens, contextWindow, compactionSettings); + logger.debug("Auto-compaction threshold decision", { + phase: "post-agent-end", + goalModeEnabled: this.#goalModeState?.enabled === true, + goalStatus: this.#goalModeState?.goal.status, + stopReason: assistantMessage.stopReason, + sameModel: sameModel === true, + contextWindow, + strategy: compactionSettings.strategy, + thresholdTokens, + assistantUsageContextTokens, + storedContextTokens, + resolvedContextTokens: contextTokens, + postMaintenanceContextTokens, + maintenanceTokensFreed, + shouldCompact: shouldThresholdCompact, + contextPromotionEnabled: this.settings.get("contextPromotion.enabled") === true, + }); + if (shouldThresholdCompact) { // Try promotion first — if a larger model is available, switch instead of compacting const promoted = await this.#tryContextPromotion(assistantMessage); if (!promoted) { return await this.#runAutoCompaction("threshold", false, false, allowDefer, { autoContinue, - triggerContextTokens: contextTokens, + triggerContextTokens: postMaintenanceContextTokens, }); } + logger.debug("Auto-compaction threshold satisfied but context promotion took over", { + contextTokens, + contextWindow, + model: `${assistantMessage.provider}/${assistantMessage.model}`, + }); } return COMPACTION_CHECK_NONE; } @@ -10008,15 +10053,16 @@ export class AgentSession { // situation actually resolves; "idle" is exempt because its 60s+ timer // re-checks usage before re-firing and cannot dead-loop on its own. // - // #2275: the post-shake check MUST be anchored on the same metric that - // triggered compaction. The local estimator (`#estimatePendingPromptTokens`) - // undercounts thinking-signature payloads, so on thinking-heavy sessions it - // reads well below the provider-reported usage that fired the threshold. - // When that estimate slips under the threshold, the fallback never fires - // and the auto-continue prompt re-injects every turn. Prefer the trigger's - // own `contextTokens` (provider-anchored) when the caller supplies it, and - // add hysteresis (80% recovery band) so we don't oscillate at the boundary - // while shake keeps reclaiming a trickle of the previous turn's output. + // #2275: the post-shake check MUST stay provider-anchored when caller + // usage and local estimates diverge. The local estimator undercounts + // thinking-signature payloads, so thinking-heavy sessions can read well + // below the provider usage that fired the threshold. Prefer the caller's + // context figure when supplied, then subtract shake's own savings and add + // hysteresis (80% recovery band) so we don't oscillate at the boundary. + // Threshold callers pass the provider-billed trigger after accounting for + // any supersede/drop-useless pruning that already rewrote the next prompt; + // without that pre-shake savings, shake can fall through to context-full + // even though the post-prune history is already inside the recovery band. const contextWindow = this.model?.contextWindow ?? 0; const compactionSettings = this.settings.getGroup("compaction"); let stillOverThreshold = false; diff --git a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts index cb86914e8..95eb5699d 100644 --- a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts +++ b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts @@ -8,6 +8,7 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { loadExtensions } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/loader"; import { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/runner"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import * as unexpectedStopClassifier from "@oh-my-pi/pi-coding-agent/session/unexpected-stop-classifier"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { getProjectAgentDir, TempDir, withTimeout } from "@oh-my-pi/pi-utils"; @@ -418,6 +419,59 @@ describe("AgentSession auto-compaction queue resume", () => { expect(runtimeSignals).toContain("compaction:start:threshold"); expect(runtimeSignals.some(signal => signal.startsWith("compaction:end:"))).toBe(true); }); + it("runs active-goal threshold compaction before unexpected-stop retry continuation", async () => { + const now = Date.now(); + session.setGoalModeState({ + enabled: true, + mode: "active", + goal: { + id: "goal-unexpected-stop-threshold", + objective: "continue until compacted", + status: "active", + tokensUsed: 0, + timeUsedSeconds: 0, + createdAt: now, + updatedAt: now, + }, + }); + session.settings.set("compaction.thresholdTokens", 76384); + session.settings.set("compaction.thresholdPercent", -1); + session.settings.set("compaction.autoContinue", true); + session.settings.set("contextPromotion.enabled", false); + session.settings.set("features.unexpectedStopDetection", true); + session.settings.set("providers.unexpectedStopModel", "online"); + + vi.spyOn(unexpectedStopClassifier, "classifyUnexpectedStop").mockResolvedValue(true); + vi.spyOn(session.agent, "continue").mockImplementation(async () => { + session.agent.clearAllQueues(); + }); + + const assistantMsg = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "I should continue investigating another module." }], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + stopReason: "stop" as const, + usage: { + input: 5000, + output: 1000, + cacheRead: 85000, + cacheWrite: 0, + totalTokens: 91000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: now, + }; + + session.agent.emitExternalEvent({ type: "message_end", message: assistantMsg }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMsg] }); + + await session.waitForIdle(); + + expect(getRuntimeSignals()).toContain("compaction:start:threshold"); + }); + it("has isCompacting true when the auto_compaction_start event fires", async () => { // Defect 1: the compaction AbortController (which backs isCompacting) must be // installed before auto_compaction_start is emitted. If it is installed after, diff --git a/packages/coding-agent/test/shake.test.ts b/packages/coding-agent/test/shake.test.ts index 14cc47744..01c7aff07 100644 --- a/packages/coding-agent/test/shake.test.ts +++ b/packages/coding-agent/test/shake.test.ts @@ -347,6 +347,70 @@ describe("AgentSession shake", () => { expect(fullStart).toBeDefined(); }); + it("counts pre-shake prune savings when deciding whether to fall back to context-full", async () => { + session.settings.set("compaction.strategy", "shake"); + session.settings.set("compaction.thresholdTokens", 76384); + session.settings.set("compaction.thresholdPercent", -1); + session.settings.set("compaction.dropUseless", true); + session.settings.set("contextPromotion.enabled", false); + + const now = Date.now(); + sessionManager.appendMessage({ + role: "user", + content: "Investigate every module of the project.", + timestamp: now - 200, + }); + const bigCallId = "call-big-useless-for-shake"; + sessionManager.appendMessage({ + role: "assistant", + content: [{ type: "toolCall", id: bigCallId, name: "search", arguments: { pattern: "TODO" } }], + ...apiInfo, + stopReason: "toolUse", + usage, + timestamp: now - 180, + }); + sessionManager.appendMessage({ + role: "toolResult", + toolCallId: bigCallId, + toolName: "search", + content: [{ type: "text", text: "match line\n".repeat(20000) }], + isError: false, + useless: true, + timestamp: now - 170, + }); + session.agent.replaceMessages(session.buildDisplaySessionContext().messages); + + const shakeSpy = vi + .spyOn(session, "shake") + .mockResolvedValue({ mode: "elide", toolResultsDropped: 1, blocksDropped: 0, tokensFreed: 100 }); + + const assistantMessage: AssistantMessage = { + role: "assistant", + content: [{ type: "text", text: "trigger" }], + ...apiInfo, + stopReason: "stop", + usage: { + input: 5000, + output: 1000, + cacheRead: 85000, + cacheWrite: 0, + totalTokens: 91000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: now, + }; + + session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); + await Bun.sleep(50); + + expect(shakeSpy).toHaveBeenCalledTimes(1); + const fullStart = events.find( + event => event.type === "auto_compaction_start" && (event as { action?: string }).action === "context-full", + ); + expect(fullStart).toBeUndefined(); + }); + it("falls back after pre-prompt shake when the floored stored conversation remains over threshold", async () => { session.settings.set("compaction.strategy", "shake"); session.settings.set("compaction.thresholdTokens", 8_000); From d444b3c47975410a132bfe8b53d7b464855d6829 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 21 Jun 2026 08:36:05 +0000 Subject: [PATCH 03/43] style: bun run fix --- .../test/agent-session-auto-compaction-queue.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts index 95eb5699d..f7b6aaaaa 100644 --- a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts +++ b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts @@ -8,9 +8,9 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { loadExtensions } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/loader"; import { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/runner"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; -import * as unexpectedStopClassifier from "@oh-my-pi/pi-coding-agent/session/unexpected-stop-classifier"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import * as unexpectedStopClassifier from "@oh-my-pi/pi-coding-agent/session/unexpected-stop-classifier"; import { getProjectAgentDir, TempDir, withTimeout } from "@oh-my-pi/pi-utils"; const runtimeSignalStoreKey = "__ompRuntimeSignals"; From 00d14accfbf66d0a555848d0cd57cbd09938fd87 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 21 Jun 2026 08:49:28 +0000 Subject: [PATCH 04/43] fix(compaction): keep empty-stop cleanup before active-goal compaction continuation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Codex review on #3175: the active-goal compaction pre-empt I added in 8ab754f636 short-circuited #handleEmptyAssistantStop. That handler is the only path that strips an orphan toolUse assistant (stopReason "toolUse" with no toolCall block) from both active context and the session branch via #removeEmptyStopFromActiveContext. With the pre-empt ordering, an over-threshold goal turn that returned an empty toolUse left the orphan as the session leaf, and the compaction auto-continue prompt fed it back into the next Anthropic turn as a tool_use with no matching tool_result — the exact history-corruption pattern the existing cleanup comment defends against. Move #handleEmptyAssistantStop back ahead of the active-goal compaction probe. Empty stops still self-retry and never reach the threshold pre-empt; non-empty stops (the reporter's failure in #3174) still hit threshold maintenance before the unexpected-stop classifier. Regression test seeds a goal-mode empty toolUse stop billed at 91k against thresholdTokens 76384 and asserts the threshold compaction never starts and the orphan is no longer in the session branch. Fixes #3174 --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/session/agent-session.ts | 16 +++- ...gent-session-auto-compaction-queue.test.ts | 79 +++++++++++++++++++ 3 files changed, 92 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index c52a42930..fe332b447 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `/goal` threshold auto-compaction skipping real sessions through two separate paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context, and active-goal text stops now attempt threshold maintenance before empty/unexpected-stop retry continuations can return from post-turn handling. The threshold decision also logs the billed, stored, resolved, and post-maintenance token counts so future no-start reports identify the exact skip path. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) +- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. The threshold decision also logs the billed, stored, resolved, and post-maintenance token counts so future no-start reports identify the exact skip path. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) ## [16.1.10] - 2026-06-21 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 36dac06fa..fd2a9a265 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2803,6 +2803,18 @@ export class AgentSession { } this.#lastSuccessfulYieldToolCallId = undefined; + // Empty-stop cleanup MUST run before any compaction continuation: an + // empty toolUse stop must be stripped from active context + session + // history before we schedule another turn, otherwise the next + // Anthropic turn carries a tool_use block with no matching + // tool_result and corrupts message history. The handler also + // schedules its own retry, so a real empty stop never needs the + // active-goal threshold pre-empt below. + if (await this.#handleEmptyAssistantStop(msg)) { + await emitAgentEndNotification(); + return; + } + let compactionResult = COMPACTION_CHECK_NONE; let checkedCompaction = false; if (activeGoal) { @@ -2816,10 +2828,6 @@ export class AgentSession { } } - if (await this.#handleEmptyAssistantStop(msg)) { - await emitAgentEndNotification(); - return; - } if (await this.#handleUnexpectedAssistantStop(msg)) { await emitAgentEndNotification(); return; diff --git a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts index f7b6aaaaa..b8b7b1fdc 100644 --- a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts +++ b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts @@ -472,6 +472,85 @@ describe("AgentSession auto-compaction queue resume", () => { expect(getRuntimeSignals()).toContain("compaction:start:threshold"); }); + it("removes orphan toolUse assistant before active-goal threshold compaction continuation", async () => { + // Codex review on #3175: when an active goal turn is over threshold AND + // stops with an empty `toolUse` (no tool call), the new ordering must NOT + // skip `#handleEmptyAssistantStop` — that handler is the only path that + // strips the orphan assistant from active context + session history. If a + // compaction continuation runs with the orphan still in place, the next + // Anthropic turn carries a `tool_use` block with no matching + // `tool_result` and corrupts the message history. + const now = Date.now(); + session.setGoalModeState({ + enabled: true, + mode: "active", + goal: { + id: "goal-orphan-toolUse-threshold", + objective: "continue until compacted", + status: "active", + tokensUsed: 0, + timeUsedSeconds: 0, + createdAt: now, + updatedAt: now, + }, + }); + session.settings.set("compaction.thresholdTokens", 76384); + session.settings.set("compaction.thresholdPercent", -1); + session.settings.set("compaction.autoContinue", true); + session.settings.set("contextPromotion.enabled", false); + + vi.spyOn(session.agent, "continue").mockImplementation(async () => { + session.agent.clearAllQueues(); + }); + + const orphanToolUse = { + role: "assistant" as const, + // Empty toolUse stop: stopReason says a tool was requested but the + // content block is empty (no toolCall). This is the case the empty-stop + // cleanup defends against. + content: [] as never[], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + stopReason: "toolUse" as const, + usage: { + input: 5000, + output: 1000, + cacheRead: 85000, + cacheWrite: 0, + totalTokens: 91000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: now, + }; + session.agent.emitExternalEvent({ type: "message_end", message: orphanToolUse }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [orphanToolUse] }); + + await session.waitForIdle(); + + // Empty-stop cleanup short-circuits before any compaction continuation, so + // the threshold compaction MUST NOT fire on this turn — the next turn + // starts from the cleaned-up branch with the retry-reminder developer + // message instead. The pre-fix ordering let compaction reach + // `auto_compaction_start` first, scheduling a continuation while the + // orphan `toolUse` entry was still the session leaf. + const signals = getRuntimeSignals(); + expect(signals).not.toContain("compaction:start:threshold"); + + // `#removeEmptyStopFromActiveContext` rewinds the session leaf past the + // orphan via `sessionManager.branch(parentId)` / `resetLeaf()`. If the + // cleanup is skipped, the orphan is still the leaf when the compaction + // continuation runs and the next Anthropic turn sends a `tool_use` block + // with no matching `tool_result`. + const branch = sessionManager.getBranch(); + const orphanInBranch = branch.some(entry => { + if (entry.type !== "message") return false; + const message = entry.message as { role: string; stopReason?: string }; + return message.role === "assistant" && message.stopReason === "toolUse"; + }); + expect(orphanInBranch).toBe(false); + }); + it("has isCompacting true when the auto_compaction_start event fires", async () => { // Defect 1: the compaction AbortController (which backs isCompacting) must be // installed before auto_compaction_start is emitted. If it is installed after, From e70c71077e60e73a6d419e58e97c1087dc3a165e Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 21 Jun 2026 08:53:54 +0000 Subject: [PATCH 05/43] fix(compaction): log agent_end maintenance routing for goal turns MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reporter on #3174 still sees no auto-compaction with thresholdTokens lowered to 32768 against a 70k+ visible context, and reports the existing `Auto-compaction threshold decision` log never appears. Either the goal turn never reaches `#checkCompaction`, or the log fires but is filtered out of their view (winston is at debug level so it writes to ~/.omp/logs/omp..log, not the TUI). Add an `agent_end maintenance routing` debug log at every branch of the `agent_end` handler — entered/no-message, skip-post-turn-maintenance, successful-yield (active goal vs not), empty-stop-handled, active-goal pre-empt (and whether it scheduled a continuation), unexpected-stop-handled, and bottom checkCompaction — together with stopReason, provider/model, content shape, goal state, and `successfulYield`. Combined with the existing `Auto-compaction threshold decision` log, the next no-start report identifies the exact early-return branch and the inputs that fed `shouldCompact`. Refs #3174 --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/session/agent-session.ts | 34 +++++++++++++++++++ 2 files changed, 35 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index fe332b447..4dac863e1 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. The threshold decision also logs the billed, stored, resolved, and post-maintenance token counts so future no-start reports identify the exact skip path. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) +- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) ## [16.1.10] - 2026-06-21 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index fd2a9a265..8498ff38b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2769,10 +2769,32 @@ export class AgentSession { this.#lastAssistantMessage = undefined; if (!msg) { this.#lastSuccessfulYieldToolCallId = undefined; + logger.debug("agent_end maintenance routing", { + reason: "no-assistant-message", + goalModeEnabled: this.#goalModeState?.enabled === true, + goalStatus: this.#goalModeState?.goal.status, + }); await emitAgentEndNotification(); return; } + const maintenanceRoute = (route: string, extra?: Record) => { + logger.debug("agent_end maintenance routing", { + route, + stopReason: msg.stopReason, + provider: msg.provider, + model: msg.model, + contentBlocks: msg.content.length, + hasToolCalls: msg.content.some(content => content.type === "toolCall"), + hasText: msg.content.some(content => content.type === "text"), + goalModeEnabled: this.#goalModeState?.enabled === true, + goalStatus: this.#goalModeState?.goal.status, + successfulYield: this.#assistantEndedWithSuccessfulYield(msg), + ...extra, + }); + }; + maintenanceRoute("entered"); + // Invalidate GitHub Copilot credentials on auth failure so stale tokens // aren't reused on the next request if ( @@ -2786,6 +2808,7 @@ export class AgentSession { if (this.#skipPostTurnMaintenanceAssistantTimestamp === msg.timestamp) { this.#skipPostTurnMaintenanceAssistantTimestamp = undefined; this.#lastSuccessfulYieldToolCallId = undefined; + maintenanceRoute("skip-post-turn-maintenance"); await emitAgentEndNotification(); return; } @@ -2794,9 +2817,12 @@ export class AgentSession { if (this.#assistantEndedWithSuccessfulYield(msg)) { this.#lastSuccessfulYieldToolCallId = undefined; if (activeGoal) { + maintenanceRoute("successful-yield-active-goal-checkCompaction"); const compactionTask = this.#checkCompaction(msg); this.#trackPostPromptTask(compactionTask); await compactionTask; + } else { + maintenanceRoute("successful-yield-no-active-goal"); } await emitAgentEndNotification(); return; @@ -2811,6 +2837,7 @@ export class AgentSession { // schedules its own retry, so a real empty stop never needs the // active-goal threshold pre-empt below. if (await this.#handleEmptyAssistantStop(msg)) { + maintenanceRoute("empty-stop-handled"); await emitAgentEndNotification(); return; } @@ -2818,17 +2845,23 @@ export class AgentSession { let compactionResult = COMPACTION_CHECK_NONE; let checkedCompaction = false; if (activeGoal) { + maintenanceRoute("active-goal-pre-empt-checkCompaction"); const compactionTask = this.#checkCompaction(msg); this.#trackPostPromptTask(compactionTask); compactionResult = await compactionTask; checkedCompaction = true; if (compactionResult.deferredHandoff || compactionResult.continuationScheduled) { + maintenanceRoute("active-goal-pre-empt-continuation-scheduled", { + deferredHandoff: compactionResult.deferredHandoff, + continuationScheduled: compactionResult.continuationScheduled, + }); await emitAgentEndNotification(); return; } } if (await this.#handleUnexpectedAssistantStop(msg)) { + maintenanceRoute("unexpected-stop-handled"); await emitAgentEndNotification(); return; } @@ -2869,6 +2902,7 @@ export class AgentSession { this.#resolveRetry(); if (!checkedCompaction) { + maintenanceRoute("bottom-checkCompaction"); const compactionTask = this.#checkCompaction(msg); this.#trackPostPromptTask(compactionTask); compactionResult = await compactionTask; From eef2f7b7e98f42e4f377472e6000b37d5601e778 Mon Sep 17 00:00:00 2001 From: roboomp Date: Sun, 21 Jun 2026 09:08:59 +0000 Subject: [PATCH 06/43] fix(compaction): resolve retry before goal continuation return Active-goal threshold compaction can pre-empt the normal post-turn tail and return once it schedules a deferred handoff or auto-continue. When that turn is the successful response from an auto-retry, returning there skips the later retry-gate cleanup and leaves isRetrying stuck. Resolve the completed retry gate before the compaction-continuation return, and cover the retry-success-over-threshold path so future changes cannot strand prompt()/waitForIdle() behind a stale retry state. Refs #3174 --- packages/coding-agent/CHANGELOG.md | 2 +- .../coding-agent/src/session/agent-session.ts | 1 + ...gent-session-auto-compaction-queue.test.ts | 100 ++++++++++++++++++ 3 files changed, 102 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 4dac863e1..205813f94 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) +- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Active-goal compaction continuations now also resolve completed retry gates before returning, preventing `isRetrying` from staying stuck after a retry succeeds over the threshold. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) ## [16.1.10] - 2026-06-21 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 8498ff38b..38db5736b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2855,6 +2855,7 @@ export class AgentSession { deferredHandoff: compactionResult.deferredHandoff, continuationScheduled: compactionResult.continuationScheduled, }); + this.#resolveRetry(); await emitAgentEndNotification(); return; } diff --git a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts index b8b7b1fdc..fa24ca3e6 100644 --- a/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts +++ b/packages/coding-agent/test/agent-session-auto-compaction-queue.test.ts @@ -1,6 +1,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; import * as path from "node:path"; +import { scheduler } from "node:timers/promises"; import { Agent } from "@oh-my-pi/pi-agent-core"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; @@ -472,6 +473,105 @@ describe("AgentSession auto-compaction queue resume", () => { expect(getRuntimeSignals()).toContain("compaction:start:threshold"); }); + it("resolves a pending retry before active-goal compaction continuation returns", async () => { + // Codex review on #3175: a retry can succeed with a non-empty text stop + // that is already over the active-goal compaction threshold. If the + // compaction pre-empt schedules its own continuation before the normal + // bottom-of-handler `#resolveRetry()` call runs, the session stays + // `isRetrying` and later prompt/idle gates remain blocked. + vi.useRealTimers(); + const now = Date.now(); + session.setGoalModeState({ + enabled: true, + mode: "active", + goal: { + id: "goal-retry-threshold", + objective: "recover from retry and compact", + status: "active", + tokensUsed: 0, + timeUsedSeconds: 0, + createdAt: now, + updatedAt: now, + }, + }); + session.settings.set("compaction.thresholdTokens", 76384); + session.settings.set("compaction.thresholdPercent", -1); + session.settings.set("compaction.autoContinue", true); + session.settings.set("contextPromotion.enabled", false); + session.settings.set("retry.enabled", true); + session.settings.set("retry.baseDelayMs", 5); + session.settings.set("retry.maxDelayMs", 5_000); + session.settings.set("retry.maxRetries", 1); + session.settings.set("retry.modelFallback", false); + + vi.spyOn(scheduler, "wait").mockResolvedValue(undefined); + vi.spyOn(session.agent, "continue").mockImplementation(async () => { + session.agent.clearAllQueues(); + }); + + const { promise: retryStarted, resolve: onRetryStarted } = Promise.withResolvers(); + const { promise: retryEnded, resolve: onRetryEnded } = Promise.withResolvers(); + const { promise: compactionDone, resolve: onCompactionDone } = Promise.withResolvers(); + session.subscribe(event => { + if (event.type === "auto_retry_start") onRetryStarted(); + if (event.type === "auto_retry_end") onRetryEnded(); + if (event.type === "auto_compaction_end") onCompactionDone(); + }); + + const retryableError = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "Transient provider failure." }], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + stopReason: "error" as const, + errorMessage: "503 service unavailable: overloaded_error retry-after-ms=50", + usage: { + input: 100, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 100, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: now - 1, + }; + session.agent.emitExternalEvent({ type: "message_end", message: retryableError }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [retryableError] }); + + await withTimeout(retryStarted, 1000, "Retry start timed out"); + expect(session.isRetrying).toBe(true); + + const recoveredOverThreshold = { + role: "assistant" as const, + content: [{ type: "text" as const, text: "Recovered; continuing the active goal." }], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + stopReason: "stop" as const, + usage: { + input: 5000, + output: 1000, + cacheRead: 85000, + cacheWrite: 0, + totalTokens: 91000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: now, + }; + session.agent.emitExternalEvent({ type: "message_end", message: recoveredOverThreshold }); + await withTimeout(retryEnded, 1000, "Retry end timed out"); + expect(session.isRetrying).toBe(true); + + session.agent.emitExternalEvent({ type: "agent_end", messages: [recoveredOverThreshold] }); + + await withTimeout(compactionDone, 1000, "Compaction end timed out"); + await session.waitForIdle(); + + expect(getRuntimeSignals()).toContain("compaction:start:threshold"); + expect(session.isRetrying).toBe(false); + }); + it("removes orphan toolUse assistant before active-goal threshold compaction continuation", async () => { // Codex review on #3175: when an active goal turn is over threshold AND // stops with an empty `toolUse` (no tool call), the new ordering must NOT From 12aedd738884b31ee9729b6191d9cacdb49a6b92 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 17:10:49 +0000 Subject: [PATCH 07/43] fix(coding-agent): returned to ask options after custom escape Keep the ask tool on the current question when the custom answer editor is dismissed, so Escape from Other returns to the option selector instead of aborting or recording an empty answer. Fixes #3269 --- packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/tools/ask.ts | 86 +++++++++++--------- packages/coding-agent/test/tools/ask.test.ts | 42 +++++++--- 3 files changed, 79 insertions(+), 53 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1c47a3a82..828ee1129 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed `ask` returning `(cancelled)` or aborting the tool when Escape dismissed `Other (type your own)` custom input; it now returns to the option selector so the user can pick a listed answer instead. ([#3269](https://github.com/can1357/oh-my-pi/issues/3269)) + ## [16.1.15] - 2026-06-22 ### Added diff --git a/packages/coding-agent/src/tools/ask.ts b/packages/coding-agent/src/tools/ask.ts index 404ab7d6a..86dfa6dd0 100644 --- a/packages/coding-agent/src/tools/ask.ts +++ b/packages/coding-agent/src/tools/ask.ts @@ -308,7 +308,7 @@ async function askSingleQuestion( } const customResult = await promptForCustomInput(); if (customResult.input === undefined) { - break; + continue; } customInput = customResult.input; break; @@ -332,51 +332,57 @@ async function askSingleQuestion( } selectedOptions = Array.from(selected); } else { - const displayOptions = addRecommendedSuffix(questionOptions, recommended); - const optionsWithNavigation: ExtensionUISelectItem[] = [...displayOptions, OTHER_OPTION]; + while (true) { + const displayOptions = addRecommendedSuffix(questionOptions, recommended); + const optionsWithNavigation: ExtensionUISelectItem[] = [...displayOptions, OTHER_OPTION]; - let initialIndex = recommended; - const previouslySelected = selectedOptions[0]; - if (previouslySelected) { - const selectedIndex = questionOptions.findIndex(option => option.label === previouslySelected); - if (selectedIndex >= 0) initialIndex = selectedIndex; - } else if (customInput !== undefined) { - initialIndex = displayOptions.length; - } - if (initialIndex !== undefined) { - const maxIndex = Math.max(optionsWithNavigation.length - 1, 0); - initialIndex = Math.max(0, Math.min(initialIndex, maxIndex)); - } - - const { - choice, - timedOut: selectTimedOut, - navigation: arrowNavigation, - } = await selectOption(promptWithProgress, optionsWithNavigation, initialIndex, { - selectionMarker: "radio", - markableCount: displayOptions.length, - }); - timedOut = selectTimedOut; - - if (arrowNavigation) { - return { selectedOptions, customInput, timedOut, navigation: arrowNavigation }; - } - if (choice === undefined) { - if (!timedOut) { - return { selectedOptions, customInput, timedOut, cancelled: true }; + let initialIndex = recommended; + const previouslySelected = selectedOptions[0]; + if (previouslySelected) { + const selectedIndex = questionOptions.findIndex(option => option.label === previouslySelected); + if (selectedIndex >= 0) initialIndex = selectedIndex; + } else if (customInput !== undefined) { + initialIndex = displayOptions.length; } - } else if (choice === OTHER_OPTION) { - if (!selectTimedOut) { - const customResult = await promptForCustomInput(); - if (customResult.input !== undefined) { - customInput = customResult.input; - selectedOptions = []; + if (initialIndex !== undefined) { + const maxIndex = Math.max(optionsWithNavigation.length - 1, 0); + initialIndex = Math.max(0, Math.min(initialIndex, maxIndex)); + } + + const { + choice, + timedOut: selectTimedOut, + navigation: arrowNavigation, + } = await selectOption(promptWithProgress, optionsWithNavigation, initialIndex, { + selectionMarker: "radio", + markableCount: displayOptions.length, + }); + timedOut = selectTimedOut; + + if (arrowNavigation) { + return { selectedOptions, customInput, timedOut, navigation: arrowNavigation }; + } + if (choice === undefined) { + if (!timedOut) { + return { selectedOptions, customInput, timedOut, cancelled: true }; } - // If editor was dismissed (undefined), keep prior selectedOptions/customInput intact + break; + } + if (choice === OTHER_OPTION) { + if (selectTimedOut) { + break; + } + const customResult = await promptForCustomInput(); + if (customResult.input === undefined) { + continue; + } + customInput = customResult.input; + selectedOptions = []; + break; } - } else { selectedOptions = [stripRecommendedSuffix(choice)]; customInput = undefined; + break; } if (navigation?.allowForward) { return { selectedOptions, customInput, timedOut, navigation: "forward" }; diff --git a/packages/coding-agent/test/tools/ask.test.ts b/packages/coding-agent/test/tools/ask.test.ts index b09f79ce3..50348e332 100644 --- a/packages/coding-agent/test/tools/ask.test.ts +++ b/packages/coding-agent/test/tools/ask.test.ts @@ -496,7 +496,7 @@ describe("AskTool option descriptions", () => { step += 1; return selectItemLabel(options.find(o => selectItemLabel(o)?.endsWith("beta"))); } - return "Other (type your own)"; + return selectItemLabel(options.find(o => selectItemLabel(o)?.includes("Done selecting"))); }, editor, }); @@ -564,10 +564,11 @@ describe("AskTool custom input", () => { expect(abort).not.toHaveBeenCalled(); }); - it("aborts when editor is cancelled in single-question flow", async () => { + it("returns to the option selector when custom input is dismissed in single-question flow", async () => { const tool = new AskTool(createSession()); const abort = vi.fn(); const editor = vi.fn(async () => undefined); + let selectCalls = 0; const questions = [ { id: "details", @@ -576,22 +577,27 @@ describe("AskTool custom input", () => { }, ]; const context = createContext({ - select: async () => "Other (type your own)", + select: async () => { + selectCalls += 1; + return selectCalls === 1 ? "Other (type your own)" : "yes"; + }, editor, abort, }); - await expect( - tool.execute("call-editor-cancel", { questions }, undefined, undefined, context), - ).rejects.toBeInstanceOf(ToolAbortError); + const result = await tool.execute("call-editor-cancel", { questions }, undefined, undefined, context); + expect(result.details?.selectedOptions).toEqual(["yes"]); + expect(result.details?.customInput).toBeUndefined(); + expect(selectCalls).toBe(2); expect(editor).toHaveBeenCalledTimes(1); - expect(abort).toHaveBeenCalledTimes(1); + expect(abort).not.toHaveBeenCalled(); }); - it("continues multi-question flow when editor is dismissed on a fresh question", async () => { + it("returns to the option selector when custom input is dismissed in multi-question flow", async () => { const tool = new AskTool(createSession()); const abort = vi.fn(); const editor = vi.fn(async () => undefined); + let detailsVisits = 0; const questions = [ { id: "first", @@ -607,7 +613,10 @@ describe("AskTool custom input", () => { const context = createContext({ select: async prompt => { if (prompt.includes("First?")) return "one"; - if (prompt.includes("Details?")) return "Other (type your own)"; + if (prompt.includes("Details?")) { + detailsVisits += 1; + return detailsVisits === 1 ? "Other (type your own)" : "short"; + } return undefined; }, editor, @@ -616,10 +625,10 @@ describe("AskTool custom input", () => { const result = await tool.execute("call-editor-multi-dismiss", { questions }, undefined, undefined, context); - // Editor dismissed on "Details?" — flow continues with empty answer, not abort expect(result.details?.results?.[0]?.selectedOptions).toEqual(["one"]); - expect(result.details?.results?.[1]?.selectedOptions).toEqual([]); + expect(result.details?.results?.[1]?.selectedOptions).toEqual(["short"]); expect(result.details?.results?.[1]?.customInput).toBeUndefined(); + expect(detailsVisits).toBe(2); expect(editor).toHaveBeenCalledTimes(1); expect(abort).not.toHaveBeenCalled(); }); @@ -745,7 +754,7 @@ describe("AskTool custom input", () => { expect(renderedText).toContain("custom detail"); }); - it("preserves prior multi-select answers when custom editor is dismissed", async () => { + it("returns to the option selector when multi-select custom input is dismissed", async () => { const tool = new AskTool(createSession()); let step = 0; const editor = vi.fn(async () => undefined); @@ -757,7 +766,13 @@ describe("AskTool custom input", () => { if (!alphaOption) throw new Error("Missing alpha option"); return selectItemLabel(alphaOption); } - return "Other (type your own)"; + if (step === 1) { + step += 1; + return "Other (type your own)"; + } + const doneOption = options.find(option => selectItemLabel(option)?.includes("Done selecting")); + if (!doneOption) throw new Error("Missing done option"); + return selectItemLabel(doneOption); }, editor, }); @@ -786,6 +801,7 @@ describe("AskTool custom input", () => { throw new Error("Expected text result"); } expect(result.content[0].text).toContain("User selected: alpha"); + expect(step).toBe(2); expect(editor).toHaveBeenCalledTimes(1); }); }); From 9e32c197934670b3f05766cc2b3d5b19fee38ff9 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 17:47:28 +0000 Subject: [PATCH 08/43] fix(web-search): paced exa search requests Added configurable Exa search request pacing via exa.searchDelayMs so repeated web_search calls no longer burst directly into Exa rate limits. Covered the provider contract with a focused Exa test and recorded the back-to-back request repro. Fixes #3271 --- packages/coding-agent/CHANGELOG.md | 4 ++ .../src/config/settings-schema.ts | 12 ++++++ .../src/web/search/providers/exa.ts | 41 ++++++++++++++++++- .../test/tools/web-search-exa.test.ts | 34 ++++++++++++++- 4 files changed, 89 insertions(+), 2 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1c47a3a82..b0bb7dc83 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Exa web search requests firing back-to-back with no client-side pacing by adding a configurable `exa.searchDelayMs` delay (default 1000ms) between Exa search requests. ([#3271](https://github.com/can1357/oh-my-pi/issues/3271)) + ## [16.1.15] - 2026-06-22 ### Added diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 636dbb4a5..5f1311e2c 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -4479,6 +4479,17 @@ export const SETTINGS_SCHEMA = { }, }, + "exa.searchDelayMs": { + type: "number", + default: 1_000, + ui: { + tab: "providers", + group: "Services", + label: "Exa Search Delay", + description: "Minimum delay between Exa web search requests in milliseconds; set 0 to disable pacing", + }, + }, + "exa.enableResearcher": { type: "boolean", default: false, @@ -4785,6 +4796,7 @@ export interface TtsrSettings { export interface ExaSettings { enabled: boolean; enableSearch: boolean; + searchDelayMs: number; enableResearcher: boolean; enableWebsets: boolean; } diff --git a/packages/coding-agent/src/web/search/providers/exa.ts b/packages/coding-agent/src/web/search/providers/exa.ts index a82e20ec8..c25291eb2 100644 --- a/packages/coding-agent/src/web/search/providers/exa.ts +++ b/packages/coding-agent/src/web/search/providers/exa.ts @@ -7,7 +7,7 @@ * them into a combined `answer` string on the SearchResponse. */ import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai"; -import { settings } from "../../../config/settings"; +import { getDefault, settings } from "../../../config/settings"; import { findApiKey, isSearchResponse } from "../../../exa/mcp-client"; import { parseSSE } from "../../../mcp/json-rpc"; import type { SearchResponse, SearchSource } from "../../../web/search/types"; @@ -18,6 +18,43 @@ import { SearchProvider } from "./base"; import { classifyProviderHttpError, withHardTimeout } from "./utils"; const EXA_API_URL = "https://api.exa.ai/search"; +const DEFAULT_EXA_SEARCH_DELAY_MS = getDefault("exa.searchDelayMs"); + +let nextExaSearchRequestAt = 0; +let exaSearchThrottle = Promise.resolve(); + +function configuredExaSearchDelayMs(): number { + try { + const delayMs = settings.get("exa.searchDelayMs"); + return Number.isFinite(delayMs) && delayMs > 0 ? Math.floor(delayMs) : 0; + } catch { + return DEFAULT_EXA_SEARCH_DELAY_MS; + } +} + +async function waitForExaSearchSlot(signal: AbortSignal | undefined): Promise { + const delayMs = configuredExaSearchDelayMs(); + if (delayMs <= 0) return; + + const prior = exaSearchThrottle.catch(() => {}); + const current = prior.then(async () => { + signal?.throwIfAborted(); + const waitMs = Math.max(0, nextExaSearchRequestAt - Date.now()); + if (waitMs > 0) { + await Bun.sleep(waitMs); + } + signal?.throwIfAborted(); + nextExaSearchRequestAt = Date.now() + delayMs; + }); + exaSearchThrottle = current.catch(() => {}); + await current; +} + +/** Reset Exa request pacing state for isolated provider tests. */ +export function resetExaSearchThrottleForTest(): void { + nextExaSearchRequestAt = 0; + exaSearchThrottle = Promise.resolve(); +} type ExaSearchType = "neural" | "fast" | "auto" | "deep"; @@ -224,6 +261,7 @@ async function callExaSearch(apiKey: string, params: ExaSearchParams): Promise = {}) { describe("searchExa", () => { let capturedRequestBody: Record | null = null; - beforeEach(() => { + beforeEach(async () => { + resetSettingsForTest(); + resetExaSearchThrottleForTest(); + await Settings.init({ inMemory: true, overrides: { "exa.searchDelayMs": 0 } }); capturedRequestBody = null; process.env.EXA_API_KEY = "test-key-123"; }); afterEach(() => { vi.restoreAllMocks(); + resetExaSearchThrottleForTest(); + resetSettingsForTest(); delete process.env.EXA_API_KEY; }); @@ -290,6 +297,31 @@ describe("searchExa", () => { }); }); + it("paces consecutive Exa API requests by the configured delay", async () => { + resetSettingsForTest(); + resetExaSearchThrottleForTest(); + await Settings.init({ inMemory: true, overrides: { "exa.searchDelayMs": 25 } }); + const requestTimes: number[] = []; + const fetchMock: FetchImpl = (_url, init) => { + requestTimes.push(Date.now()); + if (init?.body) { + capturedRequestBody = JSON.parse(init.body as string); + } + return Promise.resolve( + new Response(JSON.stringify(makeMockExaResponse()), { + status: 200, + headers: { "Content-Type": "application/json" }, + }), + ); + }; + + await searchExa({ query: "first paced request", fetch: fetchMock }); + await searchExa({ query: "second paced request", fetch: fetchMock }); + + expect(requestTimes).toHaveLength(2); + expect(requestTimes[1] - requestTimes[0]).toBeGreaterThanOrEqual(20); + }); + it("prefers summary over text for snippet field", async () => { const result = await searchExa({ query: "snippet test", From 022010ad97808eb067b3bcabbd574df73d44e9c5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 17:54:15 +0000 Subject: [PATCH 09/43] fix(web-search): aborted exa throttle waits Made Exa request pacing observe cancellation during the configured delay instead of waiting for the full delay before checking the signal. Added regression coverage for a queued Exa request cancelled while throttled. Fixes #3271 --- .../src/web/search/providers/exa.ts | 36 ++++++++++++++++++- .../test/tools/web-search-exa.test.ts | 27 ++++++++++++++ 2 files changed, 62 insertions(+), 1 deletion(-) diff --git a/packages/coding-agent/src/web/search/providers/exa.ts b/packages/coding-agent/src/web/search/providers/exa.ts index c25291eb2..2c58df630 100644 --- a/packages/coding-agent/src/web/search/providers/exa.ts +++ b/packages/coding-agent/src/web/search/providers/exa.ts @@ -32,6 +32,40 @@ function configuredExaSearchDelayMs(): number { } } +function rejectWithAbortReason(reject: (reason?: unknown) => void, signal: AbortSignal): void { + try { + signal.throwIfAborted(); + reject(new DOMException("The operation was aborted.", "AbortError")); + } catch (error) { + reject(error); + } +} + +function abortableSleep(ms: number, signal: AbortSignal | undefined): Promise { + if (ms <= 0) return Promise.resolve(); + signal?.throwIfAborted(); + const { promise, resolve, reject } = Promise.withResolvers(); + let timer: NodeJS.Timeout | undefined; + const cleanup = (): void => { + if (timer) { + clearTimeout(timer); + timer = undefined; + } + signal?.removeEventListener("abort", onAbort); + }; + const onAbort = (): void => { + cleanup(); + if (signal) rejectWithAbortReason(reject, signal); + }; + timer = setTimeout(() => { + cleanup(); + resolve(); + }, ms); + signal?.addEventListener("abort", onAbort, { once: true }); + if (signal?.aborted) onAbort(); + return promise; +} + async function waitForExaSearchSlot(signal: AbortSignal | undefined): Promise { const delayMs = configuredExaSearchDelayMs(); if (delayMs <= 0) return; @@ -41,7 +75,7 @@ async function waitForExaSearchSlot(signal: AbortSignal | undefined): Promise 0) { - await Bun.sleep(waitMs); + await abortableSleep(waitMs, signal); } signal?.throwIfAborted(); nextExaSearchRequestAt = Date.now() + delayMs; diff --git a/packages/coding-agent/test/tools/web-search-exa.test.ts b/packages/coding-agent/test/tools/web-search-exa.test.ts index 1148cb280..6df9fba7f 100644 --- a/packages/coding-agent/test/tools/web-search-exa.test.ts +++ b/packages/coding-agent/test/tools/web-search-exa.test.ts @@ -322,6 +322,33 @@ describe("searchExa", () => { expect(requestTimes[1] - requestTimes[0]).toBeGreaterThanOrEqual(20); }); + it("aborts while waiting for the configured Exa request delay", async () => { + resetSettingsForTest(); + resetExaSearchThrottleForTest(); + await Settings.init({ inMemory: true, overrides: { "exa.searchDelayMs": 1_000 } }); + let fetchCount = 0; + const fetchMock: FetchImpl = () => { + fetchCount += 1; + return Promise.resolve( + new Response(JSON.stringify(makeMockExaResponse()), { + status: 200, + headers: { "Content-Type": "application/json" }, + }), + ); + }; + + await searchExa({ query: "first request", fetch: fetchMock }); + const controller = new AbortController(); + const startedAt = Date.now(); + const pending = searchExa({ query: "cancelled request", fetch: fetchMock, signal: controller.signal }); + await Bun.sleep(0); + controller.abort(new Error("cancelled Exa throttle wait")); + + await expect(pending).rejects.toThrow("cancelled Exa throttle wait"); + expect(Date.now() - startedAt).toBeLessThan(250); + expect(fetchCount).toBe(1); + }); + it("prefers summary over text for snippet field", async () => { const result = await searchExa({ query: "snippet test", From 921904b3deaf653c4e2bd63adc32b5ed47d20eb0 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 17:59:17 +0000 Subject: [PATCH 10/43] fix(web-search): cancelled queued exa waits Raced queued Exa throttle waits against the caller abort signal so requests cancelled behind an earlier throttle wait reject immediately without breaking the serialized throttle chain. Added regression coverage for cancelling a third Exa request queued behind another delayed request. Fixes #3271 --- .../src/web/search/providers/exa.ts | 17 ++++++++-- .../test/tools/web-search-exa.test.ts | 32 +++++++++++++++++++ 2 files changed, 46 insertions(+), 3 deletions(-) diff --git a/packages/coding-agent/src/web/search/providers/exa.ts b/packages/coding-agent/src/web/search/providers/exa.ts index 2c58df630..a26143d67 100644 --- a/packages/coding-agent/src/web/search/providers/exa.ts +++ b/packages/coding-agent/src/web/search/providers/exa.ts @@ -66,12 +66,23 @@ function abortableSleep(ms: number, signal: AbortSignal | undefined): Promise(promise: Promise, signal: AbortSignal | undefined): Promise { + if (!signal) return promise; + signal.throwIfAborted(); + const { promise: aborted, reject } = Promise.withResolvers(); + const onAbort = (): void => rejectWithAbortReason(reject, signal); + signal.addEventListener("abort", onAbort, { once: true }); + return Promise.race([promise, aborted]).finally(() => { + signal.removeEventListener("abort", onAbort); + }); +} + async function waitForExaSearchSlot(signal: AbortSignal | undefined): Promise { const delayMs = configuredExaSearchDelayMs(); if (delayMs <= 0) return; const prior = exaSearchThrottle.catch(() => {}); - const current = prior.then(async () => { + const queued = prior.then(async () => { signal?.throwIfAborted(); const waitMs = Math.max(0, nextExaSearchRequestAt - Date.now()); if (waitMs > 0) { @@ -80,8 +91,8 @@ async function waitForExaSearchSlot(signal: AbortSignal | undefined): Promise {}); - await current; + exaSearchThrottle = queued.catch(() => {}); + await waitUntilDoneOrAborted(queued, signal); } /** Reset Exa request pacing state for isolated provider tests. */ diff --git a/packages/coding-agent/test/tools/web-search-exa.test.ts b/packages/coding-agent/test/tools/web-search-exa.test.ts index 6df9fba7f..34065411c 100644 --- a/packages/coding-agent/test/tools/web-search-exa.test.ts +++ b/packages/coding-agent/test/tools/web-search-exa.test.ts @@ -349,6 +349,38 @@ describe("searchExa", () => { expect(fetchCount).toBe(1); }); + it("aborts while queued behind another Exa throttle wait", async () => { + resetSettingsForTest(); + resetExaSearchThrottleForTest(); + await Settings.init({ inMemory: true, overrides: { "exa.searchDelayMs": 1_000 } }); + let fetchCount = 0; + const fetchMock: FetchImpl = () => { + fetchCount += 1; + return Promise.resolve( + new Response(JSON.stringify(makeMockExaResponse()), { + status: 200, + headers: { "Content-Type": "application/json" }, + }), + ); + }; + + await searchExa({ query: "first request", fetch: fetchMock }); + const secondController = new AbortController(); + const thirdController = new AbortController(); + const second = searchExa({ query: "second request", fetch: fetchMock, signal: secondController.signal }); + const startedAt = Date.now(); + const third = searchExa({ query: "third request", fetch: fetchMock, signal: thirdController.signal }); + await Bun.sleep(0); + thirdController.abort(new Error("cancelled queued Exa throttle wait")); + + await expect(third).rejects.toThrow("cancelled queued Exa throttle wait"); + expect(Date.now() - startedAt).toBeLessThan(250); + expect(fetchCount).toBe(1); + + secondController.abort(new Error("cleanup second Exa throttle wait")); + await expect(second).rejects.toThrow("cleanup second Exa throttle wait"); + }); + it("prefers summary over text for snippet field", async () => { const result = await searchExa({ query: "snippet test", From a305e68a53e06f783207d33c2680d1193b3fcfe8 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 20:02:33 +0000 Subject: [PATCH 11/43] fix(compaction): compact goal runs between tool turns Active goal loops can stay inside one agent run while the model keeps emitting tool calls, so the normal agent_end threshold maintenance never runs. That lets context grow past the soft threshold until provider overflow or user abort. Run threshold maintenance from the per-turn onTurnEnd hook for active goals, splice the compacted agent state back into the live loop message array, and suppress queued continuations because the current run is already continuing. Cover the mid-run tool-call path and the non-goal control case. Refs #3174 --- packages/coding-agent/CHANGELOG.md | 1 + .../coding-agent/src/session/agent-session.ts | 66 ++++++- ...ent-session-goal-midrun-compaction.test.ts | 181 ++++++++++++++++++ 3 files changed, 244 insertions(+), 4 deletions(-) create mode 100644 packages/coding-agent/test/agent-session-goal-midrun-compaction.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 205813f94..712d49eda 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Active-goal compaction continuations now also resolve completed retry gates before returning, preventing `isRetrying` from staying stuck after a retry succeeds over the threshold. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) +- Fixed active `/goal` runs that never reached `agent_end` because the model kept emitting tool calls inside one agent run. Threshold maintenance now runs between tool-call turns, compacts the live loop context in place, and suppresses queued continuations that would race the still-running goal loop. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) ## [16.1.10] - 2026-06-21 diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 38db5736b..873dac1ad 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -1612,6 +1612,7 @@ export class AgentSession { await this.#advisorRuntime.waitForCatchup(30000, threshold, signal); } } + await this.#maintainContextMidRun(messages, signal); }); this.yieldQueue = new YieldQueue({ isStreaming: () => this.isStreaming, @@ -8266,6 +8267,59 @@ export class AgentSession { }); } + + /** + * Compact active `/goal` runs that never settle to `agent_end`. + * + * Long autonomous goals can keep producing tool calls inside one agent run. + * The post-turn `agent_end` threshold check never fires in that shape, so + * context can grow until provider overflow. `onTurnEnd` is the safe boundary: + * tool results for the just-finished turn are already paired in + * `activeMessages`, the live array the agent loop reads before its next + * model call. Run maintenance here and splice the compacted state back into + * that array, mirroring [`AgentSession.#applyRewind`]. + */ + async #maintainContextMidRun(activeMessages: AgentMessage[], signal?: AbortSignal): Promise { + if (signal?.aborted || this.#isDisposed || this.isCompacting || this.isGeneratingHandoff) return; + if (!(this.#goalModeState?.enabled === true && this.#goalModeState.goal.status === "active")) return; + + const model = this.model; + const contextWindow = model?.contextWindow ?? 0; + if (contextWindow <= 0) return; + + const compactionSettings = this.settings.getGroup("compaction"); + if (!compactionSettings.enabled || compactionSettings.strategy === "off") return; + + const lastAssistant = [...activeMessages] + .reverse() + .find((message): message is AssistantMessage => message.role === "assistant"); + if (!lastAssistant || lastAssistant.stopReason === "aborted" || lastAssistant.stopReason === "error") return; + + const billedContextTokens = calculateContextTokens(lastAssistant.usage); + const storedContextTokens = this.#estimateStoredContextTokens(); + const contextTokens = compactionContextTokens(billedContextTokens, storedContextTokens); + if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) return; + + const messagesBefore = activeMessages.length; + await this.#runAutoCompaction("threshold", false, false, false, { + autoContinue: false, + suppressContinuation: true, + triggerContextTokens: contextTokens, + }); + + if (signal?.aborted) return; + const compactedMessages = this.agent.state.messages; + if (compactedMessages !== activeMessages) { + activeMessages.splice(0, activeMessages.length, ...compactedMessages); + } + logger.debug("Mid-run goal compaction ran between tool-call turns", { + contextTokens, + contextWindow, + strategy: compactionSettings.strategy, + messagesBefore, + messagesAfter: activeMessages.length, + }); + } /** * Check if context maintenance or promotion is needed and run it. * Called after agent_end and before prompt submission. @@ -9536,13 +9590,15 @@ export class AgentSession { willRetry: boolean, deferred = false, allowDefer = true, - options: { autoContinue?: boolean; triggerContextTokens?: number } = {}, + options: { autoContinue?: boolean; triggerContextTokens?: number; suppressContinuation?: boolean } = {}, ): Promise { const compactionSettings = this.settings.getGroup("compaction"); if (compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE; if (reason !== "idle" && !compactionSettings.enabled) return COMPACTION_CHECK_NONE; const generation = this.#promptGeneration; - const shouldAutoContinue = options.autoContinue !== false && compactionSettings.autoContinue !== false; + const suppressContinuation = options.suppressContinuation === true; + const shouldAutoContinue = + !suppressContinuation && options.autoContinue !== false && compactionSettings.autoContinue !== false; // Shake runs inline (cheap, no remote LLM). On overflow recovery, if shake // reclaims nothing we fall through to the summary-compaction body below so // the oversized input still gets resolved. @@ -9553,6 +9609,7 @@ export class AgentSession { generation, shouldAutoContinue, options.triggerContextTokens, + suppressContinuation, ); if (outcome !== "fallback") return outcome; } @@ -10006,7 +10063,7 @@ export class AgentSession { this.#scheduleAgentContinue({ delayMs: 100, generation }); continuationScheduled = true; - } else if (this.agent.hasQueuedMessages()) { + } else if (!suppressContinuation && this.agent.hasQueuedMessages()) { // Auto-compaction can complete while follow-up/steering/custom messages are waiting. // Kick the loop so queued messages are actually delivered. this.#scheduleAgentContinue({ @@ -10066,6 +10123,7 @@ export class AgentSession { generation: number, autoContinue: boolean, triggerContextTokens?: number, + suppressContinuation = false, ): Promise { const action = "shake"; this.#autoCompactionAbortController?.abort(); @@ -10165,7 +10223,7 @@ export class AgentSession { } this.#scheduleAgentContinue({ delayMs: 100, generation }); continuationScheduled = true; - } else if (this.agent.hasQueuedMessages()) { + } else if (!suppressContinuation && this.agent.hasQueuedMessages()) { this.#scheduleAgentContinue({ delayMs: 100, generation, diff --git a/packages/coding-agent/test/agent-session-goal-midrun-compaction.test.ts b/packages/coding-agent/test/agent-session-goal-midrun-compaction.test.ts new file mode 100644 index 000000000..6bdba47d9 --- /dev/null +++ b/packages/coding-agent/test/agent-session-goal-midrun-compaction.test.ts @@ -0,0 +1,181 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as path from "node:path"; +import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { GoalModeState } from "@oh-my-pi/pi-coding-agent/goals/state"; +import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import { TempDir } from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; + +function activeGoalState(): GoalModeState { + const now = Date.now(); + return { + enabled: true, + mode: "active", + goal: { + id: "goal-midrun-compaction", + objective: "Ship the release", + status: "active", + tokensUsed: 0, + timeUsedSeconds: 0, + createdAt: now, + updatedAt: now, + }, + }; +} + +function highUsage(input: number) { + return { + input, + output: 100, + cacheRead: 0, + cacheWrite: 0, + totalTokens: input + 100, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }; +} + +describe("AgentSession mid-run goal compaction", () => { + let tempDir: TempDir; + const cleanups: Array<() => Promise> = []; + + beforeEach(() => { + tempDir = TempDir.createSync("@pi-agent-goal-midrun-compaction-"); + cleanups.length = 0; + }); + + afterEach(async () => { + for (const cleanup of cleanups) await cleanup(); + cleanups.length = 0; + tempDir.removeSync(); + vi.restoreAllMocks(); + }); + + async function createHarness(settingsOverride: Record = {}): Promise<{ + session: AgentSession; + observedContexts: string[][]; + }> { + const observedContexts: string[][] = []; + const model = getBundledModel("anthropic", "claude-sonnet-4-5"); + if (!model) throw new Error("Expected claude-sonnet-4-5 model to exist"); + + const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${cleanups.length}.db`)); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), `models-${cleanups.length}.yml`)); + const settings = Settings.isolated({ + "compaction.enabled": true, + "compaction.strategy": "context-full", + "compaction.autoContinue": true, + "compaction.thresholdTokens": 1000, + "compaction.thresholdPercent": -1, + "todo.enabled": false, + "todo.reminders": false, + ...settingsOverride, + }); + const sessionManager = SessionManager.inMemory(tempDir.path()); + + const mockBashTool: AgentTool = { + name: "bash", + label: "Bash", + description: "Mock bash tool", + parameters: type({}), + execute: async () => ({ content: [{ type: "text" as const, text: "tool output" }] }), + }; + + let call = 0; + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model, systemPrompt: ["Test"], tools: [mockBashTool], messages: [] }, + convertToLlm, + streamFn: (_model, context) => { + const index = call++; + observedContexts.push(context.messages.map(message => JSON.stringify(message))); + const stream = new AssistantMessageEventStream(); + const isToolTurn = index === 0; + const message = isToolTurn + ? { + role: "assistant" as const, + content: [ + { type: "toolCall" as const, id: `tc-${index}`, name: "bash", arguments: { cmd: "ls" } }, + ], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + usage: highUsage(50_000), + stopReason: "toolUse" as const, + timestamp: Date.now(), + } + : { + role: "assistant" as const, + content: [{ type: "text" as const, text: "All done." }], + api: "anthropic-messages" as const, + provider: "anthropic" as const, + model: "claude-sonnet-4-5", + usage: highUsage(200), + stopReason: "stop" as const, + timestamp: Date.now(), + }; + queueMicrotask(() => { + stream.push({ type: "start", partial: message }); + stream.push({ type: "done", reason: message.stopReason, message }); + }); + return stream; + }, + }); + + const session = new AgentSession({ + agent, + sessionManager, + settings, + modelRegistry, + toolRegistry: new Map([[mockBashTool.name, mockBashTool]]), + }); + + cleanups.push(async () => { + await session.dispose(); + authStorage.close(); + }); + return { session, observedContexts }; + } + + it("compacts in place between tool-call turns during an active goal run", async () => { + const { session, observedContexts } = await createHarness(); + session.setGoalModeState(activeGoalState()); + + const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async preparation => ({ + summary: "MID-RUN-COMPACTED", + shortSummary: undefined, + firstKeptEntryId: preparation.firstKeptEntryId, + tokensBefore: preparation.tokensBefore, + details: {}, + })); + + await session.prompt("work on the release"); + + expect(compactSpy).toHaveBeenCalledTimes(1); + expect(observedContexts.length).toBeGreaterThanOrEqual(2); + expect(observedContexts[1].join("\n")).toContain("MID-RUN-COMPACTED"); + }); + + it("does not compact mid-run when no goal is active", async () => { + const { session } = await createHarness(); + const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async preparation => ({ + summary: "SHOULD-NOT-RUN", + shortSummary: undefined, + firstKeptEntryId: preparation.firstKeptEntryId, + tokensBefore: preparation.tokensBefore, + details: {}, + })); + + await session.prompt("work on the release"); + + expect(compactSpy).not.toHaveBeenCalled(); + }); +}); From 29e69f53644d5d4329bfe50bf047485a9eb7cce4 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 20:02:46 +0000 Subject: [PATCH 12/43] style: bun run fix --- packages/coding-agent/src/session/agent-session.ts | 1 - 1 file changed, 1 deletion(-) diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 873dac1ad..f3c3ca8c7 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -8267,7 +8267,6 @@ export class AgentSession { }); } - /** * Compact active `/goal` runs that never settle to `agent_end`. * From 24d58033bba805a920446fd75d30433806adb1ff Mon Sep 17 00:00:00 2001 From: can1357 Date: Mon, 22 Jun 2026 22:12:24 +0200 Subject: [PATCH 13/43] =?UTF-8?q?refactor(eval):=20rename=20agent()=20para?= =?UTF-8?q?ms=20agent=5Ftype=E2=86=92agent,=20return=5Fhandle=E2=86=92hand?= =?UTF-8?q?le?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The eval agent() helper used `agent_type`/`return_handle` (snake_case) in Python/Ruby/Julia and `agentType`/`returnHandle` (camelCase) in JS, forcing the prelude docs to repeat every option twice ("JS same but camelcased"). Both are now single lowercase words identical across all four runtimes, and `agent` matches the `task` tool's existing agent-selection parameter. - Renamed across py/js/rb/jl preludes (signatures, forwarding, docstrings). - Renamed the `__agent__` bridge wire protocol + `EvalAgentArgs` (`agentType` → `agent`, `returnHandle` → `handle`) so no prelude-side remap is needed. - Updated prompt docs (workflow-notice.md, tools/eval.md), repo docs (docs/tools/eval.md, docs/python-repl.md), and all bridge/prelude tests. - CHANGELOG: Breaking Changes entry under [Unreleased]. --- docs/python-repl.md | 2 +- docs/tools/eval.md | 10 +++++----- packages/coding-agent/CHANGELOG.md | 4 ++++ .../src/eval/__tests__/agent-bridge.test.ts | 14 +++++++------- .../src/eval/__tests__/prelude-agent.test.ts | 18 +++++++++--------- .../coding-agent/src/eval/agent-bridge.ts | 13 ++++++------- packages/coding-agent/src/eval/jl/prelude.jl | 19 ++++++------------- .../src/eval/js/shared/prelude.txt | 10 +++++----- .../src/eval/py/__tests__/prelude.test.ts | 4 ++-- packages/coding-agent/src/eval/py/prelude.py | 18 +++++++++--------- packages/coding-agent/src/eval/rb/prelude.rb | 8 ++++---- .../src/prompts/system/workflow-notice.md | 2 +- .../coding-agent/src/prompts/tools/eval.md | 8 ++++---- .../test/eval/agent-bridge.test.ts | 2 +- 14 files changed, 64 insertions(+), 68 deletions(-) diff --git a/docs/python-repl.md b/docs/python-repl.md index 67582309b..51974819e 100644 --- a/docs/python-repl.md +++ b/docs/python-repl.md @@ -160,7 +160,7 @@ The runner additionally receives `PYTHONUNBUFFERED=1` and `PYTHONIOENCODING=utf- If Python preflight fails and `eval.js` is enabled, `eval` remains available for `js` cells; `py` cells fail with a Python-backend availability error. -Python prelude helpers include `agent(prompt, *, agent_type="task", model=None, label=None, schema=None, return_handle=False)`. It synchronously calls the host bridge, runs one subagent through the task executor, and returns the final text. When `schema` is supplied, the helper parses the subagent's JSON output and returns the object. When `return_handle=True`, it instead returns a DAG node dict (`{"text", "output", "handle", "id", "agent"}`) whose `handle` is the spawned agent's recoverable `agent://` URI (the parsed object lands under `"data"` when `schema` is also set), so a downstream `pipeline`/`parallel` stage can reference the transcript by handle instead of re-inlining it. +Python prelude helpers include `agent(prompt, *, agent="task", model=None, label=None, schema=None, handle=False)`. It synchronously calls the host bridge, runs one subagent through the task executor, and returns the final text. When `schema` is supplied, the helper parses the subagent's JSON output and returns the object. When `handle=True`, it instead returns a DAG node dict (`{"text", "output", "handle", "id", "agent"}`) whose `handle` is the spawned agent's recoverable `agent://` URI (the parsed object lands under `"data"` when `schema` is also set), so a downstream `pipeline`/`parallel` stage can reference the transcript by handle instead of re-inlining it. ## Execution flow and cancellation/timeout diff --git a/docs/tools/eval.md b/docs/tools/eval.md index cd01c4de1..edbb2f730 100644 --- a/docs/tools/eval.md +++ b/docs/tools/eval.md @@ -139,7 +139,7 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod - `await read(path, offset?, limit?)` or `await read(path, { offset?, limit? })` - `await tree(path = ".", maxDepth?, showHidden?)` or `await tree(path, { maxDepth?, showHidden? })` - `sort(text, reverse?, unique?)`, `uniq(text, count?)`, `counter(items, limit?, reverse?)` - - `await agent(prompt, agentType?, model?, label?, schema?)` or `await agent(prompt, { agentType?, model?, label?, schema?, returnHandle? })` + - `await agent(prompt, agent?, model?, label?, schema?)` or `await agent(prompt, { agent?, model?, label?, schema?, handle? })` - `await parallel([() => agent("a"), () => agent("b")])` - `await pipeline(items, stage1, stage2)` - `display(value)` behavior: @@ -193,13 +193,13 @@ Both runtimes expose `completion()` — a single stateless completion against a Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state. - Signatures: - - JS: `await agent(prompt, agentType?, model?, label?, schema?)` or `await agent(prompt, { agentType?, model?, label?, schema?, returnHandle? })` - - Python: `agent(prompt, *, agent_type="task", model=None, label=None, schema=None, return_handle=False)` -- `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work. + - JS: `await agent(prompt, agent?, model?, label?, schema?)` or `await agent(prompt, { agent?, model?, label?, schema?, handle? })` + - Python: `agent(prompt, *, agent="task", model=None, label=None, schema=None, handle=False)` +- `agent` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work. - `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply. - Shared background is passed via files: write a `local://` file and reference it in the prompt. `label` controls the `agent://` output label prefix. - `schema` passes a JSON Schema to the subagent structured-output path. When present, the helper parses the final JSON text and returns an object. -- `returnHandle` / `return_handle` (default off) returns a DAG node dict — `{ text, output, handle: "agent://", id, agent }`, plus a parsed `data` field when `schema` is set — instead of the bare output, so a downstream stage can reference the transcript by handle. +- `handle` (default off) returns a DAG node dict — `{ text, output, handle: "agent://", id, agent }`, plus a parsed `data` field when `schema` is set — instead of the bare output, so a downstream stage can reference the transcript by handle. - Spawn restrictions use `session.getSessionSpawns()` exactly like the `task` tool. Eval-driven subagent recursion is capped at depth 3. - JS and Python both expose `parallel(thunks)` and `pipeline(items, ...stages)`; both use a bounded async/threaded pool whose width tracks the `task.maxConcurrency` setting (the same ceiling the `task` tool uses; `0` = run every item at once), preserve item order, and propagate rejections. The width is fetched live from the host via the `__concurrency__` bridge, so the helpers no longer take a `concurrency` argument. - Errors surface as exceptions: unknown or disabled agent, disallowed spawn, recursion cap, subagent failure, or invalid structured output all fail the eval cell. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index cde28978f..abce94513 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Breaking Changes + +- Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. + ### Added - Added `isolated`, `apply`, and `merge` options to eval `agent()` across every workflow runtime (Python, JavaScript, Ruby, Julia) so `workflowz`-driven fan-outs can request the same copy-on-write worktree isolation the `task` tool offers (strict opt-in via `isolated: true`, matching the `task` tool; `apply: false` keeps captured patches/branches without merging back; `merge: false` forces patch mode). Extracted the task-isolation lifecycle into `task/isolation-runner.ts` so the eval bridge and `TaskTool` share one implementation ([#3196](https://github.com/can1357/oh-my-pi/issues/3196)) diff --git a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts index 24f4b2509..60bf6a0dd 100644 --- a/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts +++ b/packages/coding-agent/src/eval/__tests__/agent-bridge.test.ts @@ -156,7 +156,7 @@ describe("runEvalAgent", () => { vi.restoreAllMocks(); }); - it("resolves the default task agent and agentType overrides", async () => { + it("resolves the default task agent and agent overrides", async () => { mockAgents(); const runSpy = vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options, { @@ -166,7 +166,7 @@ describe("runEvalAgent", () => { const session = makeSession(); const defaultResult = await runEvalAgent({ prompt: "hello" }, { session }); - const overrideResult = await runEvalAgent({ prompt: "hello", agentType: "reviewer" }, { session }); + const overrideResult = await runEvalAgent({ prompt: "hello", agent: "reviewer" }, { session }); expect(defaultResult.text).toBe("task"); expect(overrideResult.text).toBe("reviewer"); @@ -178,7 +178,7 @@ describe("runEvalAgent", () => { mockAgents([taskAgent]); vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options)); - await expect(runEvalAgent({ prompt: "hello", agentType: "missing" }, { session: makeSession() })).rejects.toThrow( + await expect(runEvalAgent({ prompt: "hello", agent: "missing" }, { session: makeSession() })).rejects.toThrow( 'Unknown agent "missing"', ); }); @@ -844,12 +844,12 @@ describe("runEvalAgent isolation", () => { expect(mergeSpy).toHaveBeenCalledTimes(1); }); - it("preserves temp artifacts for non-isolated returnHandle outputs", async () => { + it("preserves temp artifacts for non-isolated handle outputs", async () => { mockAgents(); const rmSpy = vi.spyOn(fs, "rm").mockResolvedValue(undefined); vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options)); - await runEvalAgent({ prompt: "plain handle", returnHandle: true }, { session: makeSession() }); + await runEvalAgent({ prompt: "plain handle", handle: true }, { session: makeSession() }); const removedArtifactsDir = rmSpy.mock.calls.some( ([target]) => typeof target === "string" && target.includes("omp-eval-agent-"), @@ -1204,7 +1204,7 @@ describe("runEvalAgent isolation", () => { expect(removedArtifactsDir).toBe(true); }); - it("preserves the temp artifacts dir after a successful apply when returnHandle is requested", async () => { + it("preserves the temp artifacts dir after a successful apply when handle is requested", async () => { mockAgents(); mockIsolationContext(); const rmSpy = vi.spyOn(fs, "rm").mockResolvedValue(undefined); @@ -1218,7 +1218,7 @@ describe("runEvalAgent isolation", () => { mergedBranchForNestedPatches: false, }); - await runEvalAgent({ prompt: "scout", isolated: true, returnHandle: true }, { session: isolatedSession() }); + await runEvalAgent({ prompt: "scout", isolated: true, handle: true }, { session: isolatedSession() }); const removedArtifactsDir = rmSpy.mock.calls.some( ([target]) => typeof target === "string" && target.includes("omp-eval-agent-"), diff --git a/packages/coding-agent/src/eval/__tests__/prelude-agent.test.ts b/packages/coding-agent/src/eval/__tests__/prelude-agent.test.ts index 6106a168c..1ff2b836e 100644 --- a/packages/coding-agent/src/eval/__tests__/prelude-agent.test.ts +++ b/packages/coding-agent/src/eval/__tests__/prelude-agent.test.ts @@ -3,7 +3,7 @@ import * as vm from "node:vm"; import { JAVASCRIPT_PRELUDE_SOURCE } from "../js/shared/prelude"; /** - * The eval `agent()` helper grows a `returnHandle` option that turns its bare + * The eval `agent()` helper grows a `handle` option that turns its bare * text result into a DAG node dict carrying the spawned agent's recoverable * `agent://` handle, so a downstream `pipeline`/`parallel` stage can wire * the transcript by reference instead of re-inlining it. These lock the node @@ -23,8 +23,8 @@ function loadPrelude(callTool: (name: string, args: unknown) => Promise type AgentHelper = (prompt: string, opts?: Record) => Promise; -describe("eval js agent() returnHandle", () => { - it("returns a DAG node carrying the agent:// handle when returnHandle is set", async () => { +describe("eval js agent() handle", () => { + it("returns a DAG node carrying the agent:// handle when handle is set", async () => { let seenName: string | undefined; let seenArgs: Record | undefined; const sandbox = loadPrelude(async (name, args) => { @@ -32,9 +32,9 @@ describe("eval js agent() returnHandle", () => { seenArgs = args as Record; return { text: "hello world", details: { agent: "task", id: "abc123", model: "m", structured: false } }; }); - const node = await (sandbox.agent as AgentHelper)("say hi", { returnHandle: true }); + const node = await (sandbox.agent as AgentHelper)("say hi", { handle: true }); expect(seenName).toBe("__agent__"); - expect(seenArgs?.returnHandle).toBe(true); + expect(seenArgs?.handle).toBe(true); expect(node).toEqual({ text: "hello world", output: "hello world", @@ -53,7 +53,7 @@ describe("eval js agent() returnHandle", () => { expect(out).toBe("hello world"); }); - it("carries the parsed object under data when schema and returnHandle combine", async () => { + it("carries the parsed object under data when schema and handle combine", async () => { const payload = JSON.stringify({ k: 1 }); const sandbox = loadPrelude(async () => ({ text: payload, @@ -61,7 +61,7 @@ describe("eval js agent() returnHandle", () => { })); const node = (await (sandbox.agent as AgentHelper)("emit", { schema: { type: "object" }, - returnHandle: true, + handle: true, })) as Record; expect(node.handle).toBe("agent://id-9"); expect(node.data).toEqual({ k: 1 }); @@ -70,7 +70,7 @@ describe("eval js agent() returnHandle", () => { it("falls back to a null handle without throwing when the bridge omits details", async () => { const sandbox = loadPrelude(async () => ({ text: "lonely" })); - const node = await (sandbox.agent as AgentHelper)("x", { returnHandle: true }); + const node = await (sandbox.agent as AgentHelper)("x", { handle: true }); expect(node).toEqual({ text: "lonely", output: "lonely", handle: null, id: null, agent: null }); }); @@ -93,7 +93,7 @@ describe("eval js agent() returnHandle", () => { schema: { type: "object" }, isolated: true, apply: false, - returnHandle: true, + handle: true, })) as Record; expect(node.handle).toBe("agent://iso-1"); expect(node.data).toEqual({ ok: true }); diff --git a/packages/coding-agent/src/eval/agent-bridge.ts b/packages/coding-agent/src/eval/agent-bridge.ts index 8a9fbdfc4..7ed3547b9 100644 --- a/packages/coding-agent/src/eval/agent-bridge.ts +++ b/packages/coding-agent/src/eval/agent-bridge.ts @@ -43,19 +43,19 @@ const DEFAULT_AGENT_LABEL = "EvalAgent"; const agentArgsSchema = type({ prompt: "string>0", - "agentType?": "string>0", + "agent?": "string>0", "model?": "string>0|string>0[]", "label?": "string", "schema?": "unknown", "isolated?": "boolean", "apply?": "boolean", "merge?": "boolean", - "returnHandle?": "boolean", + "handle?": "boolean", }); interface EvalAgentArgs { prompt: string; - agentType?: string; + agent?: string; model?: string | string[]; label?: string; schema?: unknown; @@ -83,7 +83,7 @@ interface EvalAgentArgs { */ merge?: boolean; /** True when a runtime helper will return an `agent://` handle backed by the output artifacts. */ - returnHandle?: boolean; + handle?: boolean; } export interface EvalAgentBridgeOptions { @@ -276,7 +276,7 @@ function buildSubagentFailureMessage(agentName: string, result: SingleResult): s */ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOptions): Promise { const parsed = parseAgentArgs(args); - const agentName = parsed.agentType ?? DEFAULT_AGENT_TYPE; + const agentName = parsed.agent ?? DEFAULT_AGENT_TYPE; const structured = Object.hasOwn(parsed, "schema"); assertNotPlanMode(options.session); @@ -521,8 +521,7 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption // consumes `details.patchPath` / `details.branchName` / // `details.nestedPatches` out of band. Failed isolated applies throw // earlier with a recovery hint, so they never reach this gate. - const shouldCleanupTempArtifacts = - tempArtifactsDir && !parsed.returnHandle && (!isIsolated || changesApplied === true); + const shouldCleanupTempArtifacts = tempArtifactsDir && !parsed.handle && (!isIsolated || changesApplied === true); if (shouldCleanupTempArtifacts) { await fs.rm(artifactsDir, { recursive: true, force: true }); } diff --git a/packages/coding-agent/src/eval/jl/prelude.jl b/packages/coding-agent/src/eval/jl/prelude.jl index 72e8644f3..f199cf4f0 100644 --- a/packages/coding-agent/src/eval/jl/prelude.jl +++ b/packages/coding-agent/src/eval/jl/prelude.jl @@ -734,10 +734,10 @@ function completion(prompt::String; model="default", system=nothing, schema=noth return schema === nothing ? text : Main.json_parse(string(text)) end -function agent(prompt::String; agent_type="task", model=nothing, label=nothing, schema=nothing, isolated=nothing, apply=nothing, merge=nothing, return_handle=false, kwargs...) +function agent(prompt::String; agent="task", model=nothing, label=nothing, schema=nothing, isolated=nothing, apply=nothing, merge=nothing, handle=false, kwargs...) args_dict = Dict{String, Any}("prompt" => prompt) - if agent_type !== nothing - args_dict["agentType"] = agent_type + if agent !== nothing + args_dict["agent"] = agent end if model !== nothing args_dict["model"] = model @@ -759,20 +759,13 @@ function agent(prompt::String; agent_type="task", model=nothing, label=nothing, if merge !== nothing args_dict["merge"] = Bool(merge) end - handle_result = return_handle + handle_result = handle for (k, v) in kwargs - key = string(k) - if key == "agent_type" || key == "agentType" - args_dict["agentType"] = v - elseif key == "return_handle" || key == "returnHandle" - handle_result = Bool(v) - else - args_dict[key] = v - end + args_dict[string(k)] = v end # Tell the bridge a handle is wanted so it preserves the backing artifacts. if handle_result - args_dict["returnHandle"] = true + args_dict["handle"] = true end res = __omp_call_bridge("__agent__", args_dict) text = res isa AbstractDict ? get(res, "text", res) : res diff --git a/packages/coding-agent/src/eval/js/shared/prelude.txt b/packages/coding-agent/src/eval/js/shared/prelude.txt index 21e33d406..abf230123 100644 --- a/packages/coding-agent/src/eval/js/shared/prelude.txt +++ b/packages/coding-agent/src/eval/js/shared/prelude.txt @@ -121,14 +121,14 @@ if (!globalThis.__omp_js_prelude_loaded__) { "agent", opts, rest, - ["agentType", "model", "label", "schema", "isolated", "apply", "merge"], - "{ agentType, model, label, schema, isolated, apply, merge, returnHandle }", + ["agent", "model", "label", "schema", "isolated", "apply", "merge"], + "{ agent, model, label, schema, isolated, apply, merge, handle }", ); - const { returnHandle, ...callArgs } = o; - const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...callArgs, returnHandle: Boolean(returnHandle) }); + const { handle, ...callArgs } = o; + const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...callArgs, handle: Boolean(handle) }); const text = res && typeof res === "object" ? res.text : res; const parsed = hasOwn(callArgs, "schema") ? JSON.parse(text) : text; - if (!returnHandle) return parsed; + if (!handle) return parsed; const details = res && typeof res === "object" ? res.details : undefined; if (!details || typeof details !== "object" || details.id == null) { return { text, output: text, handle: null, id: null, agent: null }; diff --git a/packages/coding-agent/src/eval/py/__tests__/prelude.test.ts b/packages/coding-agent/src/eval/py/__tests__/prelude.test.ts index f9ef0e852..7ff55a61f 100644 --- a/packages/coding-agent/src/eval/py/__tests__/prelude.test.ts +++ b/packages/coding-agent/src/eval/py/__tests__/prelude.test.ts @@ -17,8 +17,8 @@ describe("python prelude", () => { expect(signature).toContain("limit"); }); - it("exposes isolation artifacts on the agent() return_handle node", () => { - // agent(..., return_handle=True) is the only escape hatch for + it("exposes isolation artifacts on the agent() handle node", () => { + // agent(..., handle=True) is the only escape hatch for // recovering apply=False patch/branch/nested artifacts (the bare // schema return is just the parsed object), so the helper MUST // translate the bridge's camelCase details onto the node — otherwise diff --git a/packages/coding-agent/src/eval/py/prelude.py b/packages/coding-agent/src/eval/py/prelude.py index 9d14eceee..6cb3bfb6d 100644 --- a/packages/coding-agent/src/eval/py/prelude.py +++ b/packages/coding-agent/src/eval/py/prelude.py @@ -520,10 +520,10 @@ if "__omp_prelude_loaded__" not in globals(): text = res.get("text") if isinstance(res, dict) else res return json.loads(text) if schema is not None else text - def agent(prompt, *, agent_type="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, return_handle=False): + def agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False): """Run a subagent and return its final output. - `agent_type` selects the subagent definition (default "task"). Pass + `agent` selects the subagent definition (default "task"). Pass `model` to override that agent's model, `label` for the output artifact id, and `schema` to request structured JSON output; when `schema` is supplied the parsed object is returned. Share background by writing a @@ -539,13 +539,13 @@ if "__omp_prelude_loaded__" not in globals(): When isolated, `apply=False` keeps captured changes inside the worktree and surfaces the root patch path, branch name, and nested repository patches through the DAG node dict (combine with - `return_handle=True` to receive them — see below; the bare return type + `handle=True` to receive them — see below; the bare return type stays bytes/string/parsed object and has nowhere to expose artifacts). `merge=False` forces patch mode even when `task.isolation.merge` is `"branch"`, avoiding the per-call git lock + repo mutation that branch mode performs. - Set `return_handle=True` to receive a DAG node dict instead of bare + Set `handle=True` to receive a DAG node dict instead of bare text: ``{"text", "output", "handle", "id", "agent"}`` where ``handle`` is the spawned agent's recoverable ``agent://`` URI. A downstream ``pipeline``/``parallel`` stage embeds that ``handle`` (or ``output``) @@ -560,8 +560,8 @@ if "__omp_prelude_loaded__" not in globals(): ``handle=None`` — the helper never throws. """ args = {"prompt": prompt} - if agent_type is not None: - args["agentType"] = agent_type + if agent is not None: + args["agent"] = agent if model is not None: args["model"] = model if label is not None: @@ -574,12 +574,12 @@ if "__omp_prelude_loaded__" not in globals(): args["apply"] = bool(apply) if merge is not None: args["merge"] = bool(merge) - if return_handle: - args["returnHandle"] = True + if handle: + args["handle"] = True res = _bridge_call("__agent__", args) text = res.get("text") if isinstance(res, dict) else res parsed = json.loads(text) if schema is not None else text - if not return_handle: + if not handle: return parsed details = res.get("details") if isinstance(res, dict) else None if not isinstance(details, dict) or details.get("id") is None: diff --git a/packages/coding-agent/src/eval/rb/prelude.rb b/packages/coding-agent/src/eval/rb/prelude.rb index 9136069c8..4615c185b 100644 --- a/packages/coding-agent/src/eval/rb/prelude.rb +++ b/packages/coding-agent/src/eval/rb/prelude.rb @@ -577,9 +577,9 @@ unless defined?($__omp_prelude_loaded) && $__omp_prelude_loaded schema.nil? ? text : JSON.parse(text) end - def agent(prompt, agent_type: "task", model: nil, label: nil, schema: nil, isolated: nil, apply: nil, merge: nil, return_handle: false) + def agent(prompt, agent: "task", model: nil, label: nil, schema: nil, isolated: nil, apply: nil, merge: nil, handle: false) args = { "prompt" => prompt } - args["agentType"] = agent_type unless agent_type.nil? + args["agent"] = agent unless agent.nil? args["model"] = model unless model.nil? args["label"] = label unless label.nil? args["schema"] = schema unless schema.nil? @@ -589,11 +589,11 @@ unless defined?($__omp_prelude_loaded) && $__omp_prelude_loaded args["apply"] = !!apply unless apply.nil? args["merge"] = !!merge unless merge.nil? # Tell the bridge a handle is wanted so it preserves the backing artifacts. - args["returnHandle"] = true if return_handle + args["handle"] = true if handle res = OmpBridge.call("__agent__", args) text = res.is_a?(Hash) ? res["text"] : res parsed = schema.nil? ? text : JSON.parse(text) - return parsed unless return_handle + return parsed unless handle details = res.is_a?(Hash) ? res["details"] : nil if !details.is_a?(Hash) || details["id"].nil? return { "text" => text, "output" => text, "handle" => nil, "id" => nil, "agent" => nil } diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 1fbfbc59c..f2c79d7b6 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -13,7 +13,7 @@ Worth it when the task benefits from decomposition + parallel coverage, or from State persists across cells, so scout in one cell and fan out in the next. Every cell has: -- `agent(prompt, *, agent_type="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, return_handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `return_handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts. +- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts. - `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one. - `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`. - `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out. diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 2ff30cdb6..c379579ef 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -43,9 +43,9 @@ tool.(args) → unknown Invoke any session tool; `args` = its parameter object. completion(prompt, model?="default", system?=None, schema?=None) → str | dict Oneshot, stateless (no history/tools). `model`: "smol" fast | "default" session | "slow" most capable. `schema` (JSON-Schema) → structured output, parsed object. -{{#if spawns}}agent(prompt, agent_type?="task", model?=None, label?=None, schema?=None, return_handle?=False) → str | dict - Run a subagent → final output. `agent_type`/`agentType` picks another discovered agent; `schema` as in completion(). Background via `local://` files named in the prompt. `return_handle`/`returnHandle` → DAG node dict { text, output, handle: "agent://", id, agent } (parsed under `data` when `schema` set). -{{#if js}} JS: options are ONE trailing object — agent(prompt, { agentType, schema, returnHandle }). +{{#if spawns}}agent(prompt, agent?="task", model?=None, label?=None, schema?=None, handle?=False) → str | dict + Run a subagent → final output. `agent` picks another discovered agent; `schema` as in completion(). Background via `local://` files named in the prompt. `handle` → DAG node dict { text, output, handle: "agent://", id, agent } (parsed under `data` when `schema` set). +{{#if js}} JS: options are ONE trailing object — agent(prompt, { agent, schema, handle }). {{/if}} {{/if}} parallel(thunks) → list @@ -63,7 +63,7 @@ budget → per-turn token budget {{#if spawns}} Pipe handles through stage helpers to build a dependency graph — acyclic waves: -- **Name nodes.** Capture each `agent(…, {{#if py}}return_handle=True{{/if}}{{#if js}}{ returnHandle: true }{{/if}}{{#if jl}}return_handle=true{{/if}})` result; carries `handle` (`agent://`) + `output`. +- **Name nodes.** Capture each `agent(…, {{#if py}}handle=True{{/if}}{{#if js}}{ handle: true }{{/if}}{{#if jl}}handle=true{{/if}})` result; carries `handle` (`agent://`) + `output`. - **Wire edges by reference.** Put an upstream node's `handle`/`output` in the dependent stage's prompt — large transcript never re-inlined. Bulk: `write("local://.md", …)`, pass the URI. - **`pipeline(items, *stages)` = staged waves**, barrier between stages (every item clears stage N before any enters N+1). **`parallel(thunks)` = one wave** of independent nodes. - **Isolate failure.** A raising node re-raises the lowest-index error, aborts its wave; wrap risky nodes in try/except so a failure degrades only its dependent subtree, independent branches finish. diff --git a/packages/coding-agent/test/eval/agent-bridge.test.ts b/packages/coding-agent/test/eval/agent-bridge.test.ts index 893303388..61a69dd4f 100644 --- a/packages/coding-agent/test/eval/agent-bridge.test.ts +++ b/packages/coding-agent/test/eval/agent-bridge.test.ts @@ -55,7 +55,7 @@ describe("runEvalAgent", () => { getAgentId: () => "BridgeParent", } as unknown as ToolSession; - await runEvalAgent({ prompt: "do work", agentType: "task" }, { session }); + await runEvalAgent({ prompt: "do work", agent: "task" }, { session }); expect(runSubprocessSpy).toHaveBeenCalledTimes(1); const options = runSubprocessSpy.mock.calls[0]?.[0]; From bda98c63ef2ede2cba8fc6ca312804843bf3a6fd Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:01:05 +0200 Subject: [PATCH 14/43] feat(coding-agent): elevated `eval` to an essential tool to ensure - Elevated `eval` to an essential tool to ensure availability across all discovery modes. - Updated system and tool prompts to mandate the use of `eval` for non-trivial shell operations like conditionals, loops, heredocs, and complex pipelines. - Restricted `bash` usage to simple binary invocations and single-fact computation to reduce shell-escaping and execution errors. --- packages/coding-agent/CHANGELOG.md | 5 +++++ .../coding-agent/src/config/settings-schema.ts | 2 +- .../src/prompts/system/system-prompt.md | 6 +++--- packages/coding-agent/src/prompts/tools/bash.md | 15 +++++++++++++++ packages/coding-agent/src/tools/eval.ts | 2 +- packages/coding-agent/src/tools/index.ts | 9 ++++++++- .../test/tool-discovery/initial-tools.test.ts | 15 +++++++++++++++ 7 files changed, 48 insertions(+), 6 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index abce94513..b447eb54e 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -10,6 +10,11 @@ - Added `isolated`, `apply`, and `merge` options to eval `agent()` across every workflow runtime (Python, JavaScript, Ruby, Julia) so `workflowz`-driven fan-outs can request the same copy-on-write worktree isolation the `task` tool offers (strict opt-in via `isolated: true`, matching the `task` tool; `apply: false` keeps captured patches/branches without merging back; `merge: false` forces patch mode). Extracted the task-isolation lifecycle into `task/isolation-runner.ts` so the eval bridge and `TaskTool` share one implementation ([#3196](https://github.com/can1357/oh-my-pi/issues/3196)) +### Changed + +- Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised). +- Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`. + ### Fixed - Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 636dbb4a5..c5fc811c8 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -3572,7 +3572,7 @@ export const SETTINGS_SCHEMA = { group: "Discovery & MCP", label: "Essential Tools Override", description: - "Override the always-loaded built-in tools (default: read, bash, edit). Leave empty to use defaults.", + "Override the always-loaded built-in tools (default: read, bash, edit, write, find, eval). Leave empty to use defaults.", }, }, diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 0bef2bca1..09e304ef8 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -109,9 +109,9 @@ You MUST use the specialized tool over its shell equivalent: {{#has tools "lsp"}}- Code intelligence → `{{toolRefs.lsp}}`.{{/has}} {{#has tools "search"}}- Regex search → `{{toolRefs.search}}`, not `grep`, `rg`, or `awk`.{{/has}} {{#has tools "find"}}- Globbing → `{{toolRefs.find}}`, not `ls **/*.ext` or `fd`.{{/has}} -{{#has tools "eval"}}- Quick compute → `{{toolRefs.eval}}`; you SHOULD go step by step.{{/has}} -{{#has tools "bash"}}- Use `{{toolRefs.bash}}` for terminal work—builds, tests, git, package managers—and pipelines that COMPUTE a fact: `wc -l`, `sort | uniq -c`, `comm`, `diff a b`, checksums. Commands shadowing the tools above are blocked. -- Litmus: produces a count, frequency, set difference, or checksum no tool returns → bash. Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}} +{{#has tools "eval"}}- Default for any compute: `{{toolRefs.eval}}` cells. Bash is the EXCEPTION — only single binary calls or short fact-computing pipelines (`wc -l`, `sort | uniq -c`, `diff`, checksums). The moment a command grows a loop, conditional, heredoc, `-e`/`-c` script, `$(...)` nesting, or >2 pipe stages, it's a program → `{{toolRefs.eval}}`. MUST NOT write multiline or inline-script bash.{{/has}} +{{#has tools "bash"}}- `{{toolRefs.bash}}`: real binaries and short fact pipelines only. Commands shadowing the specialized tools above are blocked.{{/has}} +{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash.{{#has tools "eval"}} Needs control flow, state, or fights shell quoting → `{{toolRefs.eval}}`.{{/has}} Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}} {{#has tools "report_tool_issue"}} diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 69ee52050..6a7e391a3 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -1,5 +1,19 @@ Runs bash in a shell session — terminal ops: git, bun, cargo, python. +# When to use bash — and when not to + +Bash invokes **real binaries** with simple args. It is NOT a scripting surface. + +Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a fact (`wc -l`, `sort | uniq -c`, `comm`, `diff`, a checksum, `git status`). + +Anything below → `eval` cell, not bash: +- Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language +- Heredocs (`< - `cwd` sets the working dir, not `cd dir && …` - `env: { NAME: "…" }` for multiline / quote-heavy / untrusted values; reference `$NAME` @@ -14,6 +28,7 @@ Runs bash in a shell session — terminal ops: git, bun, cargo, python. +- Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`node -e`, `python -c`), several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps. - NEVER shell out to search content or files: `grep/rg` → `search`. - Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`. diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index 19e0ba8f2..05bb79360 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -305,7 +305,7 @@ export class EvalTool implements AgentTool { get summary(): string { return summarizeEvalLanguages(this.#enabledLanguages()); } - readonly loadMode = "discoverable"; + readonly loadMode = "essential"; readonly label = "Eval"; get description(): string { if (!this.session) return getEvalToolDescription(); diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 2eb00921e..494b20df7 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -377,7 +377,14 @@ export type ToolFactory = (session: ToolSession) => Tool | null | Promise { } expect(missing).toEqual([]); }); + + it("marks eval essential so it survives tools.discoveryMode 'all'", async () => { + const metadata = await getToolMetadata(); + expect(metadata.get("eval")?.loadMode).toBe("essential"); + // Essential loadMode keeps eval active under discovery-all even when it is + // absent from the essential-names set — not relying on the names list. + const kept = filterInitialToolsForDiscoveryAll(["eval"], { + loadModeOf: name => metadata.get(name)?.loadMode as BuiltinToolLoadMode | undefined, + essentialNames: new Set(), + explicitlyRequested: new Set(), + restored: new Set(), + forceActive: new Set(), + }); + expect(kept).toEqual(["eval"]); + }); }); describe("computeEssentialBuiltinNames", () => { From 374c51122cda3d973e7dd0621ef950d86eff9d67 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:02:27 +0200 Subject: [PATCH 15/43] chore: update changelogs --- packages/agent/CHANGELOG.md | 3 ++- packages/coding-agent/CHANGELOG.md | 15 +++------------ .../src/prompts/system/system-prompt.md | 2 +- packages/coding-agent/src/prompts/tools/bash.md | 4 ++-- 4 files changed, 8 insertions(+), 16 deletions(-) diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 222bed6e2..3c2e82d06 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Added - Added `generateHandoffFromContext(context, model, options)` to `@oh-my-pi/pi-agent-core/compaction`: runs the handoff oneshot against a fully-built provider `Context` (system prompt, normalized tools, transformed history, trailing handoff prompt) with `streamOptions` mirroring the live turn's cache routing, so a host that owns the transform pipeline can make the handoff request share the prompt cache the main turn populated. `generateHandoff(messages, …)` is unchanged and now delegates to it. @@ -930,4 +931,4 @@ Initial release under @oh-my-pi scope. See previous releases at [badlogic/pi-mon ### Changed - `Agent` constructor now has all options optional (empty options use defaults). -- `queueMessage()` is now synchronous (no longer returns a Promise). \ No newline at end of file +- `queueMessage()` is now synchronous (no longer returns a Promise). diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 289d329a2..ba9c39bfb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -22,14 +22,10 @@ - Prevented `/handoff` from executing while a response is streaming to avoid session corruption - Fixed `/handoff` cold-missing the provider prompt cache. Handoff generation now builds its request through the same pipeline a live turn uses (`convertMessagesToLlm` + `Agent.buildSideRequestContext` + `prepareSimpleStreamOptions`, via the new `generateHandoffFromContext`), so it reuses the live system prompt, normalized tools, transformed/obfuscated message history, and — critically — a stable `promptCacheKey` with a unique side `sessionId`. Previously the oneshot sent no cache-routing key and skipped the `transformContext`/`transformProviderContext` and tool/message normalization the loop applies, so its prefix never matched what the turn populated and every handoff re-read the whole context uncached. Mirrors the cache-preserving path already used by `/btw` and `/omfg`. - Fixed `/handoff` (and the RPC `handoff` command) resetting the agent while a response was still streaming, which let the live turn keep emitting into the torn-down session. Manual handoff now refuses while a prompt is in flight (matching `/fork` and `/move`); the auto-handoff path is unaffected. - -### Fixed - - Fixed Exa web search requests firing back-to-back with no client-side pacing by adding a configurable `exa.searchDelayMs` delay (default 1000ms) between Exa search requests. ([#3271](https://github.com/can1357/oh-my-pi/issues/3271)) - -### Fixed - - Fixed `ask` returning `(cancelled)` or aborting the tool when Escape dismissed `Other (type your own)` custom input; it now returns to the option selector so the user can pick a listed answer instead. ([#3269](https://github.com/can1357/oh-my-pi/issues/3269)) +- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Active-goal compaction continuations now also resolve completed retry gates before returning, preventing `isRetrying` from staying stuck after a retry succeeds over the threshold. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) +- Fixed active `/goal` runs that never reached `agent_end` because the model kept emitting tool calls inside one agent run. Threshold maintenance now runs between tool-call turns, compacts the live loop context in place, and suppresses queued continuations that would race the still-running goal loop. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) ## [16.1.15] - 2026-06-22 @@ -124,11 +120,6 @@ - Fixed Hindsight retain (and the shared mnemopi/Hindsight recall paths) framing assistant turns whose only content was punctuation/whitespace — most commonly the lone `.` some providers emit for tool-call-only or thinking-only turns — into `[role: assistant]\n.\n[assistant:end]` blocks that polluted the bank, wasted retain tokens, and degraded recall. `prepareRetentionTranscript`, `extractMessages`, and `flattenMessagesForRecall` now require at least one letter or digit per message via a shared `hasSubstantiveContent` predicate ([#1806](https://github.com/can1357/oh-my-pi/issues/1806)). - Fixed `/resume` rendering forked child sessions without a fork tag, making them indistinguishable from their parent when titles match ([#1792](https://github.com/can1357/oh-my-pi/issues/1792)). -### Fixed - -- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Active-goal compaction continuations now also resolve completed retry gates before returning, preventing `isRetrying` from staying stuck after a retry succeeds over the threshold. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) -- Fixed active `/goal` runs that never reached `agent_end` because the model kept emitting tool calls inside one agent run. Threshold maintenance now runs between tool-call turns, compacts the live loop context in place, and suppresses queued continuations that would race the still-running goal loop. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) - ## [16.1.10] - 2026-06-21 ### Added @@ -12374,4 +12365,4 @@ Initial public release. ## [0.7.6] - 2025-11-13 -Previous releases did not maintain a changelog. \ No newline at end of file +Previous releases did not maintain a changelog. diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 09e304ef8..10ba2774d 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -109,7 +109,7 @@ You MUST use the specialized tool over its shell equivalent: {{#has tools "lsp"}}- Code intelligence → `{{toolRefs.lsp}}`.{{/has}} {{#has tools "search"}}- Regex search → `{{toolRefs.search}}`, not `grep`, `rg`, or `awk`.{{/has}} {{#has tools "find"}}- Globbing → `{{toolRefs.find}}`, not `ls **/*.ext` or `fd`.{{/has}} -{{#has tools "eval"}}- Default for any compute: `{{toolRefs.eval}}` cells. Bash is the EXCEPTION — only single binary calls or short fact-computing pipelines (`wc -l`, `sort | uniq -c`, `diff`, checksums). The moment a command grows a loop, conditional, heredoc, `-e`/`-c` script, `$(...)` nesting, or >2 pipe stages, it's a program → `{{toolRefs.eval}}`. MUST NOT write multiline or inline-script bash.{{/has}} +{{#has tools "eval"}}- Default for any compute: `{{toolRefs.eval}}` cells. Bash is the EXCEPTION — only single binary calls or short fact-computing pipelines (`wc -l`, `sort | uniq -c`, `diff`, checksums). The moment a command grows a loop, conditional, heredoc, `-e`/`-c` script, `$(…)` nesting, or >2 pipe stages, it's a program → `{{toolRefs.eval}}`. NEVER write multiline or inline-script bash.{{/has}} {{#has tools "bash"}}- `{{toolRefs.bash}}`: real binaries and short fact pipelines only. Commands shadowing the specialized tools above are blocked.{{/has}} {{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash.{{#has tools "eval"}} Needs control flow, state, or fights shell quoting → `{{toolRefs.eval}}`.{{/has}} Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}} diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 6a7e391a3..2b64f9948 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -9,7 +9,7 @@ Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a f Anything below → `eval` cell, not bash: - Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language - Heredocs (`< -- Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`node -e`, `python -c`), several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps. +- Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps. - NEVER shell out to search content or files: `grep/rg` → `search`. - Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`. From dc8dfa48c94f273003c8856779b382f25ef43dd7 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:13:58 +0200 Subject: [PATCH 16/43] feat(coding-agent/eval): expanded rich media support for cross-language evaluation - Added headless plot configuration for Julia to prevent GUI popup windows during execution. - Implemented robust mime-bundle serialization in Julia using `invokelatest` to handle runtime-loaded library methods. - Added IRuby-protocol and magic-byte image sniffing support to the Ruby runner to enable inline rendering for graphics gems like Gruff, ChunkyPNG, and RMagick. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/eval/jl/runner.jl | 50 ++++++-- packages/coding-agent/src/eval/rb/runner.rb | 125 ++++++++++++++++++-- 3 files changed, 155 insertions(+), 21 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ba9c39bfb..ad881301f 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -17,6 +17,7 @@ ### Fixed +- Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default. - Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners - Fixed streaming `bash`/`eval` tool output duplicating its `… (N earlier lines, showing 10 of M) (ctrl+o to expand)` preview into native scrollback. The collapsed output is a sliding tail window fixed at 10 lines, so when the box outgrew the live viewport (a tall command/output under a still-live predecessor such as a parallel tool) its mutating tail scrolled above the commit window and the renderer re-committed a fresh snapshot every frame, stacking dozens of stale preview banners and chunks. The output preview is now clamped to the viewport tail (`Math.min(10, previewWindowRows())`) and measured in visual rows at the box's inner content width (via the new `outputBlockContentWidth` helper), so on short terminals the volatile tail shrinks to stay on-screen and is never committed. Fixes the duplication introduced when scroll-off commits were made loss-free. - Prevented `/handoff` from executing while a response is streaming to avoid session corruption diff --git a/packages/coding-agent/src/eval/jl/runner.jl b/packages/coding-agent/src/eval/jl/runner.jl index 62c466e51..3e8ffc6a2 100644 --- a/packages/coding-agent/src/eval/jl/runner.jl +++ b/packages/coding-agent/src/eval/jl/runner.jl @@ -3,6 +3,12 @@ using Base64 +# Force GR (the default Plots.jl backend) into a headless workstation so a plot +# never pops up a native gksqt GUI window — the harness renders the inline PNG +# from `show(io, MIME"image/png", plt)` itself. `get!` keeps an explicit +# user-provided value, mirroring the Python runner's MPLBACKEND=Agg default. +get!(ENV, "GKSwstype", "100") + const ORIGINAL_STDOUT = stdout const ORIGINAL_STDERR = stderr const ORIGINAL_STDIN = stdin @@ -441,24 +447,38 @@ end function build_mime_bundle(value) bundle = Dict{String, Any}() - - # text/plain - io_plain = IOBuffer() - show(io_plain, MIME"text/plain"(), value) - bundle["text/plain"] = String(take!(io_plain)) - + + # text/plain — every mime probe below uses `Base.invokelatest` because this + # function runs from the frozen-world `main()` loop: `show`/`showable` + # methods that a package adds when it is `using`-ed *inside* a cell (e.g. + # Plots/Makie/GraphRecipes registering rich `show` for their plot types) are + # invisible to direct dispatch here and fall back to the default struct show, + # which can itself throw. Guard text/plain too so a failing repr never aborts + # the whole bundle before the image mime is reached. + try + io_plain = IOBuffer() + Base.invokelatest(show, io_plain, MIME"text/plain"(), value) + bundle["text/plain"] = String(take!(io_plain)) + catch + bundle["text/plain"] = try + summary(value) + catch + string(typeof(value)) + end + end + # rich mime types for mime_str in ["text/html", "text/markdown", "image/png", "image/jpeg"] m = MIME(Symbol(mime_str)) - if showable(m, value) + if Base.invokelatest(showable, m, value) try io = IOBuffer() if mime_str in ["image/png", "image/jpeg"] b64_io = Base64EncodePipe(io) - show(b64_io, m, value) + Base.invokelatest(show, b64_io, m, value) close(b64_io) else - show(io, m, value) + Base.invokelatest(show, io, m, value) end bundle[mime_str] = String(take!(io)) catch @@ -466,7 +486,7 @@ function build_mime_bundle(value) end end end - + if value isa AbstractDict || value isa AbstractVector try bundle["application/json"] = value @@ -474,7 +494,7 @@ function build_mime_bundle(value) # ignore end end - + return bundle end @@ -493,7 +513,13 @@ pushdisplay(OmpDisplay()) function emit_error(rid, err, bt) io = IOBuffer() - showerror(io, err) + # invokelatest + guard: custom error types from packages loaded inside the + # cell define `showerror` methods invisible to this frozen-world function. + try + Base.invokelatest(showerror, io, err) + catch + print(io, string(err)) + end err_str = String(take!(io)) tb = String[] diff --git a/packages/coding-agent/src/eval/rb/runner.rb b/packages/coding-agent/src/eval/rb/runner.rb index cc497d292..a09c71a8f 100644 --- a/packages/coding-agent/src/eval/rb/runner.rb +++ b/packages/coding-agent/src/eval/rb/runner.rb @@ -129,10 +129,122 @@ def __omp_emit_status(op, data = {}) __omp_emit_display({ "application/x-omp-status" => status }, "display") end +OMP_IMAGE_MIMES = %w[image/png image/jpeg].freeze + +# True when `str` already looks like base64 text (ASCII, base64 alphabet, length +# a multiple of 4). Raw image blobs (PNG/JPEG bytes) contain high bytes, so they +# fail the ASCII check and get encoded instead of passed through unchanged. +def __omp_base64?(str) + s = str.to_s + # ascii_only? is safe on any encoding (no regex over invalid bytes). Raw image + # blobs carry high bytes and fail here, so they get encoded rather than scanned. + return false unless s.ascii_only? + stripped = s.gsub(/\s+/, "") + return false if stripped.empty? || (stripped.bytesize % 4) != 0 + stripped.match?(%r{\A[A-Za-z0-9+/]*={0,2}\z}) +end + +# Coerce an image payload to the base64 ASCII the host renders. IRuby-style +# `to_iruby` hands back raw binary blobs (Gruff#to_blob, ChunkyPNG, RMagick), +# which would also break JSON.generate; strict-encode them unless already base64. +def __omp_image_payload(content) + require "base64" + s = content.to_s + return s.gsub(/\s+/, "") if __omp_base64?(s) + Base64.strict_encode64(s.b) +end + +# Detect a host-renderable image MIME from a binary blob's magic bytes. Lets us +# treat the generic `to_blob` (Gruff/RMagick/ChunkyPNG/Vips) as an image only +# when it really is one, avoiding false positives on unrelated `to_blob` methods. +def __omp_sniff_image_mime(bytes) + b = bytes.to_s.b + return "image/png" if b.start_with?("\x89PNG\r\n\x1a\n".b) + return "image/jpeg" if b.start_with?("\xFF\xD8\xFF".b) + nil +end + +# Stringify keys, base64-encode image payloads, and scrub text payloads so the +# bundle is always JSON-safe before it reaches __omp_emit. +def __omp_normalize_bundle(hash) + bundle = {} + hash.each do |key, val| + k = key.to_s + bundle[k] = + if OMP_IMAGE_MIMES.include?(k) + __omp_image_payload(val) + elsif val.is_a?(String) + __omp_scrub(val) + else + val + end + end + bundle +end + +# Guarantee a text/plain entry so the model always sees a textual hint, even for +# image-only bundles (mirrors the Python runner). +def __omp_finalize_bundle(bundle, value) + bundle["text/plain"] ||= __omp_scrub((value.inspect rescue value.class.name)) + bundle +end + +# Rich-display resolution for non-collection objects. Honors the repo +# `to_omp_mime` convention first, then the IRuby protocol +# (`to_iruby_mimebundle` -> [data, metadata], `to_iruby` -> [mime, data]) so plot +# and image objects (gruff, rubyplot, gnuplotrb, chunky_png, daru, ...) render +# inline — the Ruby analog of IPython's _repr_*_ methods. Returns nil when the +# value advertises no rich representation. +def __omp_rich_mime_bundle(value) + if value.respond_to?(:to_omp_mime) + mime = (value.to_omp_mime rescue nil) + return __omp_finalize_bundle(__omp_normalize_bundle(mime), value) if mime.is_a?(Hash) && !mime.empty? + end + if value.respond_to?(:to_iruby_mimebundle) + data = + begin + value.to_iruby_mimebundle + rescue ArgumentError + (value.to_iruby_mimebundle(include: []) rescue nil) + rescue StandardError + nil + end + data = data.first if data.is_a?(Array) + return __omp_finalize_bundle(__omp_normalize_bundle(data), value) if data.is_a?(Hash) && !data.empty? + end + if value.respond_to?(:to_iruby) + pair = (value.to_iruby rescue nil) + if pair.is_a?(Array) && pair.size == 2 && !pair[0].nil? + return __omp_finalize_bundle(__omp_normalize_bundle({ pair[0].to_s => pair[1] }), value) + end + end + # Last resort: probe well-known image emitters. Named methods (to_png/to_jpeg) + # are trusted; the generic to_blob is accepted only when its bytes sniff as an + # image. Covers gems that render via IRuby's registry rather than to_iruby + # (Gruff#to_blob, ChunkyPNG#to_blob, RMagick, Vips, ...). + if value.respond_to?(:to_png) + png = (value.to_png rescue nil) + return __omp_finalize_bundle({ "image/png" => __omp_image_payload(png) }, value) if png + end + jpeg_method = %i[to_jpeg to_jpg].find { |m| value.respond_to?(m) } + if jpeg_method + jpg = (value.public_send(jpeg_method) rescue nil) + return __omp_finalize_bundle({ "image/jpeg" => __omp_image_payload(jpg) }, value) if jpg + end + if value.respond_to?(:to_blob) + blob = (value.to_blob rescue nil) + if blob.is_a?(String) && (mime = __omp_sniff_image_mime(blob)) + return __omp_finalize_bundle({ mime => __omp_image_payload(blob) }, value) + end + end + + nil +end + # Build a Jupyter-style MIME bundle for a value. Strings render as plain text, -# Hash/Array render as JSON (plus a text/plain repr) so the model sees structure, -# and anything else falls back to its inspect string. Objects may opt into a -# richer bundle by defining `to_omp_mime` returning a Hash of mime => value. +# Hash/Array render as JSON (plus a text/plain repr) so the model sees structure. +# Other objects may expose a rich representation via `to_omp_mime` or the IRuby +# protocol (`to_iruby`/`to_iruby_mimebundle`); otherwise they fall back to inspect. def __omp_mime_bundle(value) case value when String @@ -151,12 +263,7 @@ def __omp_mime_bundle(value) when nil { "text/plain" => "nil" } else - if value.respond_to?(:to_omp_mime) - mime = (value.to_omp_mime rescue nil) - mime.is_a?(Hash) ? mime : { "text/plain" => __omp_scrub(value.inspect) } - else - { "text/plain" => __omp_scrub(value.inspect) } - end + __omp_rich_mime_bundle(value) || { "text/plain" => __omp_scrub(value.inspect) } end end From 9e6eb98db5ab88fcbcc1fad58bafc4560d5d3bd6 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:27:45 +0200 Subject: [PATCH 17/43] refactor(coding-agent/eval): removed unused sh cell magic - Removed the `_magic_cell_sh` function as `sh` cell magic was no longer required. --- packages/coding-agent/src/eval/py/runner.py | 6 ------ 1 file changed, 6 deletions(-) diff --git a/packages/coding-agent/src/eval/py/runner.py b/packages/coding-agent/src/eval/py/runner.py index 9db1513e5..dbb40e820 100644 --- a/packages/coding-agent/src/eval/py/runner.py +++ b/packages/coding-agent/src/eval/py/runner.py @@ -557,12 +557,6 @@ def _magic_run(args: str) -> None: def _magic_cell_bash(args: str, body: str) -> int: return _run_shell_body(body, shell_arg="/bin/bash") - -@cell_magic("sh") -def _magic_cell_sh(args: str, body: str) -> int: - return _run_shell_body(body, shell_arg="/bin/sh") - - @cell_magic("capture") def _magic_cell_capture(args: str, body: str) -> str: """Capture stdout/stderr of body; bind to ``args`` (a name) if provided.""" From cdc83c1de7417e3c16049594d416cd4716946943 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:30:26 +0200 Subject: [PATCH 18/43] feat(coding-agent/utils): added minimum dimension scaling to image resize - Added `minDimension` option to ensure images meet minimum size requirements for vision backends. - Implemented logic to scale up undersized input images while respecting maximum constraints. - Clamped minimum dimension floor to avoid resolution conflicts with defined maximum bounds. --- .../coding-agent/src/utils/image-resize.ts | 34 ++++++++++++++ .../test/utils/image-resize.test.ts | 44 +++++++++++++++++++ 2 files changed, 78 insertions(+) diff --git a/packages/coding-agent/src/utils/image-resize.ts b/packages/coding-agent/src/utils/image-resize.ts index f56088f29..79dd52812 100644 --- a/packages/coding-agent/src/utils/image-resize.ts +++ b/packages/coding-agent/src/utils/image-resize.ts @@ -3,6 +3,8 @@ import type { ImageContent } from "@oh-my-pi/pi-ai"; export interface ImageResizeOptions { maxWidth?: number; maxHeight?: number; + /** Smallest allowed edge length (px). Inputs below this are scaled up. */ + minDimension?: number; maxBytes?: number; jpegQuality?: number; excludeWebP?: boolean; @@ -23,6 +25,13 @@ export interface ResizedImage { // binding constraint once images are downsized to 1568px (Anthropic's internal threshold). const DEFAULT_MAX_BYTES = 500 * 1024; +// Smallest edge length (px) vision backends reliably accept. They tile images into +// fixed patches (Anthropic uses 28px) and reject degenerate sub-patch images — e.g. +// the 1x1 PNG an empty chart render emits — with a hard 400 ("Could not process +// image") that can poison the whole request. 200px is the smallest size Anthropic +// documents as valid (200x200 = 64 visual tokens); undersized images are scaled up. +const DEFAULT_MIN_DIMENSION = 200; + const DEFAULT_OPTIONS: Required> = { // Anthropic's "internal recommended size" — Claude internally caps images at // 1568px on the longest edge before vision processing. @@ -30,6 +39,7 @@ const DEFAULT_OPTIONS: Required> = { maxHeight: 1568, maxBytes: DEFAULT_MAX_BYTES, jpegQuality: 80, + minDimension: DEFAULT_MIN_DIMENSION, }; /** @@ -87,7 +97,12 @@ export async function resizeImage(img: ImageContent, options?: ImageResizeOption // still get JPEG-compressed. const originalSize = inputBuffer.length; const comfortableSize = opts.maxBytes / 4; + // Clamp the floor to the caps so an unusually small max can't demand an + // impossible "≥ min and ≤ max" target. + const minDimension = Math.min(opts.minDimension, opts.maxWidth, opts.maxHeight); if ( + originalWidth >= minDimension && + originalHeight >= minDimension && originalWidth <= opts.maxWidth && originalHeight <= opts.maxHeight && originalSize <= comfortableSize && @@ -120,6 +135,25 @@ export async function resizeImage(img: ImageContent, options?: ImageResizeOption targetHeight = opts.maxHeight; } + // Lift undersized inputs up to the minimum. A uniform scale covers the + // common case (icons, the 1x1 chart) without distortion; an aspect ratio + // too extreme to satisfy both floor and cap falls back to stretching the + // lagging edge up to the floor via the default fit:"fill" resize. + if (targetWidth < minDimension || targetHeight < minDimension) { + const shortEdge = Math.min(targetWidth, targetHeight); + const upscale = Math.min( + minDimension / shortEdge, + opts.maxWidth / targetWidth, + opts.maxHeight / targetHeight, + ); + if (upscale > 1) { + targetWidth = Math.round(targetWidth * upscale); + targetHeight = Math.round(targetHeight * upscale); + } + targetWidth = Math.min(opts.maxWidth, Math.max(minDimension, targetWidth)); + targetHeight = Math.min(opts.maxHeight, Math.max(minDimension, targetHeight)); + } + // First-attempt encoder: try PNG and JPEG (+ WebP if not excluded) — return smallest. // PNG wins for line art / few-color UI; JPEG wins for photographic content; // WebP usually beats JPEG by 25–35% but is disabled when OMP_NO_WEBP is set diff --git a/packages/coding-agent/test/utils/image-resize.test.ts b/packages/coding-agent/test/utils/image-resize.test.ts index 391f82344..7a092cf6b 100644 --- a/packages/coding-agent/test/utils/image-resize.test.ts +++ b/packages/coding-agent/test/utils/image-resize.test.ts @@ -128,6 +128,50 @@ describe("resizeImage defaults", () => { }); }); +describe("resizeImage minimum dimension", () => { + it("upscales a degenerate 1x1 image up to the 200px floor", async () => { + // A 1x1 PNG (e.g. an empty chart render) would sail through the fast path + // untouched and trip a provider 400 "Could not process image". + const result = await resizeImage({ type: "image", data: RED_1X1_PNG_BASE64, mimeType: "image/png" }); + + expect(result.wasResized).toBe(true); + // Square source stays square at the floor. + expect(result.width).toBe(200); + expect(result.height).toBe(200); + // The encoded bytes actually carry those dimensions. + const meta = await new Bun.Image(Buffer.from(result.data, "base64")).metadata(); + expect(meta.width).toBeGreaterThanOrEqual(200); + expect(meta.height).toBeGreaterThanOrEqual(200); + }); + + it("honors a custom minDimension override", async () => { + const result = await resizeImage( + { type: "image", data: RED_1X1_PNG_BASE64, mimeType: "image/png" }, + { minDimension: 64 }, + ); + + expect(result.width).toBe(64); + expect(result.height).toBe(64); + }); + + it("stretches a degenerate aspect ratio so both edges clear the floor and stay within the cap", async () => { + // 1x1600 strip: the cap pulls the long edge to 1568 while the short edge + // stays at 1px, so a uniform scale can't satisfy both bounds — the floor + // must be reached by fill-stretching the short edge. + const strip = await makeRedPng(1, 1600); + const result = await resizeImage({ type: "image", data: strip, mimeType: "image/png" }); + + expect(result.wasResized).toBe(true); + expect(result.width).toBeGreaterThanOrEqual(200); + expect(result.height).toBeGreaterThanOrEqual(200); + expect(result.width).toBeLessThanOrEqual(1568); + expect(result.height).toBeLessThanOrEqual(1568); + const meta = await new Bun.Image(Buffer.from(result.data, "base64")).metadata(); + expect(meta.width).toBeGreaterThanOrEqual(200); + expect(meta.height).toBeGreaterThanOrEqual(200); + }); +}); + describe("resizeImage env wiring", () => { const prior = Bun.env.OMP_NO_WEBP; From 9569e355b79235ba0de2df4b559aaa47ab5e1931 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:35:23 +0200 Subject: [PATCH 19/43] refactor(coding-agent/edit): introduced a dynamic row budget for the diff - Introduced a dynamic row budget for the diff preview to prevent overflow during streaming. - Integrated `previewWindowRows()` into the cache key to ensure proper re-rendering upon viewport resizing. - Simplified tail window logic to consistently apply the preview budget. --- packages/coding-agent/src/edit/renderer.ts | 25 +++++++++++++++------- 1 file changed, 17 insertions(+), 8 deletions(-) diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index 5aeb6e63d..cda934ec0 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -22,6 +22,7 @@ import { invalidateRenderedStringCache, type LspBatchRequest, PREVIEW_LIMITS, + previewWindowRows, type RenderedStringCache, replaceTabs, shortenPath, @@ -347,15 +348,23 @@ function formatStreamingDiff( cache?: RenderedStringCache, ): string { if (!diff) return ""; - let text = cachedRenderedString(cache, uiTheme, expanded, rawPath, diff, () => { - // Collapsed uses a "Cursor" tail window: pin the last - // EDIT_STREAMING_PREVIEW_LINES rows to the bottom so freshly streamed changes - // stay on screen. The whole-file diff is recomputed on every streamed chunk - // and its Myers alignment is not monotonic in payload length, so a hunk-aware - // window stutters as rows move between hunks. Expanded deliberately lifts that - // cap for the approval-time full view. + // Clamp the collapsed tail to the viewport so a tall or fast-growing diff + // cannot outgrow the live window. Otherwise its mutating tail scrolls above + // the native-scrollback commit boundary and the engine re-commits a fresh + // snapshot every streamed frame, stacking duplicate "… more lines above" + // previews in history. Budgeted in logical rows (only the tail is colored — + // recoloring the whole growing diff per chunk would be costly) with the + // viewport reserve absorbing wrapping; `previewWindowRows()` is in the cache + // salt so a resize re-slices. + const budget = expanded ? Number.POSITIVE_INFINITY : Math.min(EDIT_STREAMING_PREVIEW_LINES, previewWindowRows()); + let text = cachedRenderedString(cache, uiTheme, expanded, `${rawPath}:${budget}`, diff, () => { + // Collapsed uses a "Cursor" tail window: pin the last rows to the bottom so + // freshly streamed changes stay on screen. The whole-file diff is recomputed + // on every streamed chunk and its Myers alignment is not monotonic in payload + // length, so a hunk-aware window stutters as rows move between hunks. Expanded + // deliberately lifts the cap for the approval-time full view. const allLines = diff.replace(/\n+$/u, "").split("\n"); - const hiddenLines = expanded ? 0 : Math.max(0, allLines.length - EDIT_STREAMING_PREVIEW_LINES); + const hiddenLines = Math.max(0, allLines.length - budget); const visible = hiddenLines > 0 ? allLines.slice(hiddenLines) : allLines; let rendered = "\n\n"; if (hiddenLines > 0) { From 2ff005a35456e5051157ec12e55c653dcae6aa0a Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:36:42 +0200 Subject: [PATCH 20/43] fix(coding-agent/modes): resolved escape key handling under kitty keyboard protocol - Update input handler to recognize CSI-u escape sequences using `matchesKey`. - Ensure legacy bare escape sequences remain supported in environments without the protocol. - Add test coverage for both CSI-u and legacy escape inputs. --- .../src/modes/components/plugin-settings.ts | 8 +++- .../components/handle-input-or-escape.test.ts | 44 +++++++++++++++++++ 2 files changed, 51 insertions(+), 1 deletion(-) create mode 100644 packages/coding-agent/test/modes/components/handle-input-or-escape.test.ts diff --git a/packages/coding-agent/src/modes/components/plugin-settings.ts b/packages/coding-agent/src/modes/components/plugin-settings.ts index 110ed135f..89f60bba9 100644 --- a/packages/coding-agent/src/modes/components/plugin-settings.ts +++ b/packages/coding-agent/src/modes/components/plugin-settings.ts @@ -11,6 +11,7 @@ import { Container, Input, + matchesKey, type SelectItem, SelectList, type SettingItem, @@ -36,13 +37,18 @@ import { DynamicBorder } from "./dynamic-border"; /** * Forwards a keystroke to `input`, but cancels via `onCancel` when the user presses Escape. + * + * Escape is decoded via `matchesKey` rather than a raw `\x1b` compare: inside the + * fullscreen settings overlay the kitty keyboard protocol is active (ghostty/kitty), + * where the Escape key arrives as the CSI-u sequence `\x1b[27u`, not a bare `\x1b`. + * The literal fallbacks preserve legacy single/double-escape on terminals without it. */ export function handleInputOrEscape( data: string, input: { handleInput(data: string): void }, onCancel: () => void, ): void { - if (data === "\x1b" || data === "\x1b\x1b") { + if (data === "\x1b" || data === "\x1b\x1b" || matchesKey(data, "escape")) { onCancel(); return; } diff --git a/packages/coding-agent/test/modes/components/handle-input-or-escape.test.ts b/packages/coding-agent/test/modes/components/handle-input-or-escape.test.ts new file mode 100644 index 000000000..35e6f2a8b --- /dev/null +++ b/packages/coding-agent/test/modes/components/handle-input-or-escape.test.ts @@ -0,0 +1,44 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { handleInputOrEscape } from "@oh-my-pi/pi-coding-agent/modes/components/plugin-settings"; +import { setKittyProtocolActive } from "@oh-my-pi/pi-tui"; + +afterEach(() => { + setKittyProtocolActive(false); +}); + +describe("handleInputOrEscape", () => { + it("cancels on a kitty CSI-u escape (the fullscreen settings overlay encoding)", () => { + // Ghostty/kitty report Escape as `\x1b[27u` once the keyboard protocol is + // active (which it is inside the fullscreen settings overlay). A raw `\x1b` + // compare misses it, so Esc looked dead in the text-input submenu. + setKittyProtocolActive(true); + let cancelled = false; + const forwarded: string[] = []; + handleInputOrEscape("\x1b[27u", { handleInput: data => forwarded.push(data) }, () => { + cancelled = true; + }); + expect(cancelled).toBe(true); + expect(forwarded).toEqual([]); + }); + + it("cancels on a legacy bare escape", () => { + let cancelled = false; + const forwarded: string[] = []; + handleInputOrEscape("\x1b", { handleInput: data => forwarded.push(data) }, () => { + cancelled = true; + }); + expect(cancelled).toBe(true); + expect(forwarded).toEqual([]); + }); + + it("forwards a printable keystroke to the input instead of cancelling", () => { + setKittyProtocolActive(true); + let cancelled = false; + const forwarded: string[] = []; + handleInputOrEscape("g", { handleInput: data => forwarded.push(data) }, () => { + cancelled = true; + }); + expect(cancelled).toBe(false); + expect(forwarded).toEqual(["g"]); + }); +}); From 873e62c816156c53585191a2499d8dc45c7a8c93 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:37:03 +0200 Subject: [PATCH 21/43] feat(coding-agent/edit): implemented visual row budgeting for streaming diff renders - Updated streaming diff renderer to use visual line wrapping instead of line count for preview budget. - Introduced width-aware slicing to ensure displayed content stays within the provided UI bounds. - Refactored cache salt parameters to incorporate width for consistent rendering across window resizes. --- packages/coding-agent/CHANGELOG.md | 1 + packages/coding-agent/src/edit/renderer.ts | 43 ++++++++++++------- .../coding-agent/src/prompts/tools/eval.md | 1 + 3 files changed, 30 insertions(+), 15 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index ad881301f..e3392d2ad 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -17,6 +17,7 @@ ### Fixed +- Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path. - Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default. - Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners - Fixed streaming `bash`/`eval` tool output duplicating its `… (N earlier lines, showing 10 of M) (ctrl+o to expand)` preview into native scrollback. The collapsed output is a sliding tail window fixed at 10 lines, so when the box outgrew the live viewport (a tall command/output under a still-live predecessor such as a parallel tool) its mutating tail scrolled above the commit window and the renderer re-committed a fresh snapshot every frame, stacking dozens of stale preview banners and chunks. The output preview is now clamped to the viewport tail (`Math.min(10, previewWindowRows())`) and measured in visual rows at the box's inner content width (via the new `outputBlockContentWidth` helper), so on short terminals the volatile tail shrinks to stay on-screen and is never committed. Fixes the duplication introduced when scroll-off commits were made loss-free. diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index cda934ec0..061872e87 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -341,6 +341,7 @@ function renderPlainTextPreview(text: string, uiTheme: Theme, _filePath?: string function formatStreamingDiff( diff: string, rawPath: string, + width: number, uiTheme: Theme, expanded: boolean, label = "streaming", @@ -352,19 +353,28 @@ function formatStreamingDiff( // cannot outgrow the live window. Otherwise its mutating tail scrolls above // the native-scrollback commit boundary and the engine re-commits a fresh // snapshot every streamed frame, stacking duplicate "… more lines above" - // previews in history. Budgeted in logical rows (only the tail is colored — - // recoloring the whole growing diff per chunk would be costly) with the - // viewport reserve absorbing wrapping; `previewWindowRows()` is in the cache - // salt so a resize re-slices. + // previews in history. The budget is VISUAL rows (a long wrapped line counts + // for more than one) at the framed block's inner width (border only — + // contentPaddingLeft is 0); only the visible suffix is syntax-colored, so the + // cheap raw-line wrap walk keeps the per-chunk cost bounded. innerWidth/budget + // are in the cache salt so a resize re-slices. + const innerWidth = Math.max(1, width - 2); const budget = expanded ? Number.POSITIVE_INFINITY : Math.min(EDIT_STREAMING_PREVIEW_LINES, previewWindowRows()); - let text = cachedRenderedString(cache, uiTheme, expanded, `${rawPath}:${budget}`, diff, () => { - // Collapsed uses a "Cursor" tail window: pin the last rows to the bottom so - // freshly streamed changes stay on screen. The whole-file diff is recomputed - // on every streamed chunk and its Myers alignment is not monotonic in payload - // length, so a hunk-aware window stutters as rows move between hunks. Expanded - // deliberately lifts the cap for the approval-time full view. + let text = cachedRenderedString(cache, uiTheme, expanded, `${rawPath}:${innerWidth}:${budget}`, diff, () => { + // "Cursor" tail window: pin the last rows to the bottom so freshly streamed + // changes stay on screen. The whole-file diff is recomputed every chunk and + // its Myers alignment is not monotonic in payload length, so a hunk-aware + // window stutters as rows move between hunks. Expanded lifts the cap. const allLines = diff.replace(/\n+$/u, "").split("\n"); - const hiddenLines = Math.max(0, allLines.length - budget); + let visualUsed = 0; + let cut = allLines.length; + for (let i = allLines.length - 1; i >= 0; i--) { + const lineRows = Math.max(1, wrapTextWithAnsi(replaceTabs(allLines[i]!), innerWidth).length); + if (visualUsed + lineRows > budget && visualUsed > 0) break; + visualUsed += lineRows; + cut = i; + } + const hiddenLines = cut; const visible = hiddenLines > 0 ? allLines.slice(hiddenLines) : allLines; let rendered = "\n\n"; if (hiddenLines > 0) { @@ -393,6 +403,7 @@ function formatStreamingDiff( function formatMultiFileStreamingDiff( previews: PerFileDiffPreview[], + width: number, uiTheme: Theme, expanded: boolean, spinnerFrame?: number, @@ -414,7 +425,7 @@ function formatMultiFileStreamingDiff( const isLast = index === previews.length - 1; const cache = previewCacheAt(caches, index); parts.push( - `${header}${formatStreamingDiff(preview.diff, preview.path, uiTheme, expanded, "preview", isLast ? spinnerFrame : undefined, cache)}`, + `${header}${formatStreamingDiff(preview.diff, preview.path, width, uiTheme, expanded, "preview", isLast ? spinnerFrame : undefined, cache)}`, ); } } @@ -424,6 +435,7 @@ function formatMultiFileStreamingDiff( function getCallPreview( args: EditRenderArgs, rawPath: string, + width: number, uiTheme: Theme, renderContext: EditRenderContext | undefined, expanded: boolean, @@ -432,14 +444,14 @@ function getCallPreview( ): string { const multi = renderContext?.perFileDiffPreview; if (multi && multi.length > 1 && multi.some(p => p.diff || p.error)) { - return formatMultiFileStreamingDiff(multi, uiTheme, expanded, spinnerFrame, caches); + return formatMultiFileStreamingDiff(multi, width, uiTheme, expanded, spinnerFrame, caches); } const cache = previewCacheAt(caches, 0); if (args.previewDiff) { - return formatStreamingDiff(args.previewDiff, rawPath, uiTheme, expanded, "preview", spinnerFrame, cache); + return formatStreamingDiff(args.previewDiff, rawPath, width, uiTheme, expanded, "preview", spinnerFrame, cache); } if (args.diff && args.op) { - return formatStreamingDiff(args.diff, rawPath, uiTheme, expanded, "streaming", spinnerFrame, cache); + return formatStreamingDiff(args.diff, rawPath, width, uiTheme, expanded, "streaming", spinnerFrame, cache); } if (args.diff) { return renderPlainTextPreview(args.diff, uiTheme, rawPath); @@ -637,6 +649,7 @@ export const editToolRenderer = { let body = getCallPreview( editArgs, rawPath, + width, uiTheme, renderContext, options.expanded, diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index c379579ef..14a1c7923 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -13,6 +13,7 @@ Cell fields: Work incrementally — one logical step per cell (imports, define, test, use), many small cells per call; workflow notes in the assistant message or `title`, never in cell code. {{#if py}}Live event loop: use top-level `await` directly; `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}} +{{#if js}}JS runs under **Bun**: Bun globals/APIs are available (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`); top-level `await`/`return` work directly.{{/if}} {{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `tree(".", max_depth: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}} {{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `tree(max_depth=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}} Errors name the failing cell ("Cell 3 failed") — resubmit the fixed cell + any remaining. From 899c0ef08b1bdf2a4eb569f4fdf4bb213f4c741d Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:54:54 +0200 Subject: [PATCH 22/43] feat: simplified todo tool to single operation interface - Refactored `todo` tool to accept a single operation object instead of an `ops` array. - Implemented parameter normalization to maintain backward compatibility with legacy array-based tool calls. - Updated tool instructions, documentation, and UI rendering components to reflect the new interface. - Added compatibility tests to verify rendering and execution for both legacy and current operation formats. --- docs/tools/todo.md | 38 ++-- packages/coding-agent/CHANGELOG.md | 4 +- .../src/cli/gallery-fixtures/interaction.ts | 15 +- .../coding-agent/src/prompts/tools/todo.md | 2 +- packages/coding-agent/src/tools/todo.ts | 124 ++++++----- .../test/agent-session-eager-todo.test.ts | 8 +- packages/coding-agent/test/tools/todo.test.ts | 193 +++++++----------- packages/collab-web/CHANGELOG.md | 5 +- .../collab-web/src/tool-render/tools/todo.tsx | 14 +- 9 files changed, 182 insertions(+), 221 deletions(-) diff --git a/docs/tools/todo.md b/docs/tools/todo.md index 3abb699f1..c59c6efa8 100644 --- a/docs/tools/todo.md +++ b/docs/tools/todo.md @@ -1,6 +1,6 @@ # todo -> Applies ordered mutations to the session todo list and returns a text summary plus the full phase/task state. +> Applies one mutation to the session todo list and returns a text summary plus the full phase/task state. ## Source - Entry: `packages/coding-agent/src/tools/todo.ts` @@ -14,11 +14,7 @@ ## Inputs -| Field | Type | Required | Description | -| --- | --- | --- | --- | -| `ops` | `TodoOpEntry[]` | Yes | Ordered operations to apply. `minItems: 1`. - -### `TodoOpEntry` +The params object **is** a single op — the discriminator and its fields live at the top level (no `ops` array wrapper). | Op | Required fields | Optional fields | Effect | | --- | --- | --- | --- | @@ -28,9 +24,9 @@ | `drop` | `task` or `phase` or neither | None | Marks the target task, phase, or all tasks `abandoned`. | | `rm` | `task` or `phase` or neither | None | Removes the target task, clears the phase's task list, or clears all task lists. | | `append` | `phase`, `items` | None | Appends new `pending` tasks to a phase; creates the phase if missing. | -| `view` | None | None | Echoes the current list. A call whose ops are all `view` is read-only: no normalization, no state write. | +| `view` | None | None | Echoes the current list. A `view` call is read-only: no normalization, no state write. | -### Fields used inside ops +### Fields | Field | Type | Required | Description | | --- | --- | --- | --- | @@ -46,11 +42,11 @@ The tool returns a single-shot `AgentToolResult`: - `content`: one text part containing the summary from `formatSummary(...)`. - Empty final state with no errors: `Todo list cleared.` (`Todo list is empty.` for a pure-`view` call). - Non-empty final state: remaining-item list, current phase progress, then a per-phase tree. - - If any op produced validation/runtime errors, the summary starts with `Errors: ...` and the result is marked `isError: true`; the whole batch is discarded — the returned and persisted state stay at the pre-call list. + - If the op produced validation/runtime errors, the summary starts with `Errors: ...` and the result is marked `isError: true`; the mutation is discarded — the returned and persisted state stay at the pre-call list. - `details`: - `phases: TodoPhase[]` - `storage: "session" | "memory"` - - `completedTasks?: TodoCompletionTransition[]` when a task changed from non-completed to `completed` during the batch + - `completedTasks?: TodoCompletionTransition[]` when a task changed from non-completed to `completed` during the call `TodoPhase` / `TodoItem` state model: @@ -61,19 +57,19 @@ The TUI renderer (`todoToolRenderer`) merges call and result into one transcript ## Flow 1. `TodoTool.execute(...)` clones the current cached phases from `session.getTodoPhases?.() ?? []` (`packages/coding-agent/src/tools/todo.ts`). -2. `applyParams(...)` walks `params.ops` in order and applies each entry with `applyEntry(...)`. +2. `applyParams(...)` applies the single op (`params`) with `applyEntry(...)`. 3. Each op mutates the working phase array: - `initPhases(...)` rebuilds the list from scratch. - `start` resolves a task by exact `content`, demotes every other `in_progress` task to `pending`, then marks the target `in_progress`. - `done` / `drop` use `getTaskTargets(...)` to target one task, one phase, or every task. - `rm` removes one task, clears one phase's `tasks`, or clears all phases' task arrays. - `appendItems(...)` resolves or creates the target phase and pushes new `pending` tasks unless the same task content already exists anywhere. -4. Missing task/phase references are recorded in an `errors` array by `resolveTaskOrError(...)` / `resolvePhaseOrError(...)`; execution continues through the rest of the batch, but any error discards the batch's mutations at the end. -5. After the full batch, `normalizeInProgressTask(...)` enforces the single-active-task invariant: +4. Missing task/phase references are recorded in an `errors` array by `resolveTaskOrError(...)` / `resolvePhaseOrError(...)`; any error discards the op's mutations at the end. +5. After the op, `normalizeInProgressTask(...)` enforces the single-active-task invariant: - if multiple tasks are `in_progress`, only the first stays active and the rest become `pending`; - if none are `in_progress`, the first `pending` task in phase/task order is auto-promoted to `in_progress`. -6. `execute(...)` stores the updated phases with `session.setTodoPhases?.(...)` only when the batch produced no errors and was not pure-`view`; a failed batch is discarded wholesale (persisting a half-applied batch would make the natural retry hit "already exists"). `storage` is `"session"` when `session.getSessionFile()` exists, else `"memory"`. -7. `getCompletionTransitions(...)` compares the previous and updated phases (skipped for failed or pure-`view` calls); newly completed tasks are returned in `details.completedTasks`. +6. `execute(...)` stores the updated phases with `session.setTodoPhases?.(...)` only when the op produced no errors and was not a `view`; a failed op is discarded (persisting a half-applied mutation would make the natural retry hit "already exists"). `storage` is `"session"` when `session.getSessionFile()` exists, else `"memory"`. +7. `getCompletionTransitions(...)` compares the previous and updated phases (skipped for failed or `view` calls); newly completed tasks are returned in `details.completedTasks`. 8. The agent runtime also watches `todo` tool results in `packages/coding-agent/src/session/agent-session.ts`; successful results refresh cached todos, failed results inject a hidden next-turn reminder telling the model that todo progress is not visible until it retries. 9. The event controller updates the visible todo UI from `result.details.phases` on success, or shows a warning on error (`packages/coding-agent/src/modes/controllers/event-controller.ts`). @@ -87,7 +83,7 @@ The TUI renderer (`todoToolRenderer`) merges call and result into one transcript | `completed` | Can be set back to `in_progress` if targeted | Stays `completed` | Becomes `abandoned` if targeted | Removed | No status change | | `abandoned` | Can be set back to `in_progress` if targeted | Becomes `completed` if targeted | Stays `abandoned` | Removed | No status change | -Normalization then re-applies the single-active-task rule after the full op batch. +Normalization then re-applies the single-active-task rule after the op runs. ### Op targeting rules - `done`, `drop`, `rm`: @@ -118,7 +114,7 @@ The same file also exposes non-tool helpers used by `/todo`: - Session-level auto-clear of `completed`/`abandoned` tasks was removed (the timer mutated canonical phases between tool calls); the TUI todo widget still clears closed entries after `tasks.todoClearDelay` (display-only, `packages/coding-agent/src/modes/interactive-mode.ts`). ## Limits & Caps -- `ops` array: `minItems: 1` (`todoSchema`). +- `init.list`: applies to a single op (`todoSchema`). The params object carries exactly one op. - `init.list[*].items`: `minItems: 1`. - `append.items`: `minItems: 1`. - Renderer collapsed preview: `PREVIEW_LIMITS.COLLAPSED_ITEMS = 8` (`packages/coding-agent/src/tools/render-utils.ts`). @@ -126,7 +122,7 @@ The same file also exposes non-tool helpers used by `/todo`: - Tool execution mode: `concurrency = "exclusive"`, `strict = true`, `loadMode = "discoverable"`. ## Errors -- Ordinary bad op payloads are accumulated as human-readable strings in `errors`; the result is marked `isError: true` and the whole batch is discarded — the returned and persisted state stay at the pre-call list. +- Ordinary bad op payloads are accumulated as human-readable strings in `errors`; the result is marked `isError: true` and the mutation is discarded — the returned and persisted state stay at the pre-call list. - Error strings come from the helpers in `packages/coding-agent/src/tools/todo.ts`, including: - `Missing list for init operation` - `Missing task content` @@ -137,18 +133,18 @@ The same file also exposes non-tool helpers used by `/todo`: - `Missing phase name for append operation` - `Missing items for append operation` - `Task "..." already exists` -- Ops are processed in order and an early error does not stop later ops from being attempted, but any error in the batch discards every mutation the batch made. +- A `todo` call carries a single op; any error in it discards every mutation the op made. - Runtime-level tool failure is handled outside the tool body: `agent-session` injects a hidden reminder and the event controller warns the user that visible progress may be stale. - Idempotency is op-specific: - `init` is a full replacement; replaying the same payload yields the same state. - `start`, `done`, and `drop` are effectively idempotent on an existing target state, but `start` also demotes any other active task. - `rm` is not idempotent for targeted removals: the second call errors because the task or phase is gone. - - `append` is not idempotent: duplicate task content is rejected with `Task "..." already exists`; the whole `append` op validates up front, so a batch with any duplicate appends nothing. + - `append` is not idempotent: duplicate task content is rejected with `Task "..." already exists`; the `append` op validates up front, so an op with any duplicate appends nothing. ## Notes - Task lookup is exact string equality inside the tool. The model-facing prompt says task content and phase names are identifiers and should stay unique; `append` enforces task uniqueness globally, and `init` rejects duplicate phase names and duplicate task contents in its payload. - `findTaskByContent(...)` returns the first matching task across phases. Duplicate task contents make later targeted ops ambiguous. -- `normalizeInProgressTask(...)` runs after the whole batch, not after each op. A single call can intentionally build an intermediate invalid state and rely on final normalization. +- `normalizeInProgressTask(...)` runs once after the op, not mid-op. A single op (e.g. `init`) can build an intermediate invalid state and rely on final normalization. - `storage: "session"` means the session has a session-file backing; it does not mean this tool wrote a durable custom entry. - Reload persistence differs by path: - plain `todo` calls survive in transcript tool-result details; diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e3392d2ad..06061a941 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,7 +1,6 @@ # Changelog ## [Unreleased] - ### Breaking Changes - Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. @@ -12,6 +11,7 @@ ### Changed +- Simplified `todo` tool interface to accept a single operation directly instead of an array of ops - Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised). - Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`. @@ -12367,4 +12367,4 @@ Initial public release. ## [0.7.6] - 2025-11-13 -Previous releases did not maintain a changelog. +Previous releases did not maintain a changelog. \ No newline at end of file diff --git a/packages/coding-agent/src/cli/gallery-fixtures/interaction.ts b/packages/coding-agent/src/cli/gallery-fixtures/interaction.ts index 34da85a14..7ca065139 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/interaction.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/interaction.ts @@ -5,17 +5,14 @@ export const interactionFixtures: Record = { todo: { label: "Todo", streamingArgs: { - ops: [{ op: "init", list: [{ phase: "Foundation", items: ["Scaffold crate"] }] }], + op: "init", + list: [{ phase: "Foundation", items: ["Scaffold crate"] }], }, args: { - ops: [ - { - op: "init", - list: [ - { phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] }, - { phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] }, - ], - }, + op: "init", + list: [ + { phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] }, + { phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] }, ], }, result: { diff --git a/packages/coding-agent/src/prompts/tools/todo.md b/packages/coding-agent/src/prompts/tools/todo.md index 95a52dda2..11fe60f66 100644 --- a/packages/coding-agent/src/prompts/tools/todo.md +++ b/packages/coding-agent/src/prompts/tools/todo.md @@ -1,6 +1,6 @@ **Tasks referenced by verbatim content string, NEVER an auto-generated ID — no "task-1"/"task-N" exists. Pass the content text in the `task` field.** -Manages a phased task list. Pass `ops`: flat array of operations. Next pending task auto-promotes to `in_progress` on each completion. `pending` is a status, not an `op` — leave not-yet-started tasks implicit in `init`/`append`. +Next pending task auto-promotes to `in_progress` on each completion. ## Operations diff --git a/packages/coding-agent/src/tools/todo.ts b/packages/coding-agent/src/tools/todo.ts index 04a343feb..2ae213c3e 100644 --- a/packages/coding-agent/src/tools/todo.ts +++ b/packages/coding-agent/src/tools/todo.ts @@ -52,21 +52,18 @@ const InitListEntry = type({ items: type("string").describe("task content").array().atLeastLength(1).describe("tasks for this phase"), }); -const TodoOpEntry = type({ +const todoSchema = type({ op: TodoOp, "list?": InitListEntry.array().describe("phased task list (init)"), "task?": type("string").describe("task content"), "phase?": type("string").describe("phase name"), "items?": type("string").describe("task content").array().atLeastLength(1).describe("tasks to append"), -}); - -const todoSchema = type({ - ops: TodoOpEntry.array().atLeastLength(1).describe("ordered todo operations"), -}).describe("apply ordered todo operations"); +}).describe("apply a single todo operation"); type TodoParams = TodoSchema; type TodoSchema = typeof todoSchema.infer; -type TodoOpEntryValue = TodoParams["ops"][number]; +/** A single todo op entry (the params object itself). */ +type TodoOpEntryValue = TodoParams; // ============================================================================= // State helpers @@ -402,10 +399,7 @@ function applyEntry(phases: TodoPhase[], entry: TodoOpEntryValue, errors: string function applyParams(phases: TodoPhase[], params: TodoParams): { phases: TodoPhase[]; errors: string[] } { const errors: string[] = []; - let next = phases; - for (const entry of params.ops) { - next = applyEntry(next, entry, errors); - } + const next = applyEntry(phases, params, errors); normalizeInProgressTask(next); return { phases: next, errors }; } @@ -413,9 +407,15 @@ function applyParams(phases: TodoPhase[], params: TodoParams): { phases: TodoPha /** Apply an array of `todo`-style ops to existing phases. Used by /todo slash command. */ export function applyOpsToPhases( currentPhases: TodoPhase[], - ops: TodoParams["ops"], + ops: TodoParams[], ): { phases: TodoPhase[]; errors: string[] } { - return applyParams(clonePhases(currentPhases), { ops }); + const errors: string[] = []; + let next = clonePhases(currentPhases); + for (const op of ops) { + next = applyEntry(next, op, errors); + } + normalizeInProgressTask(next); + return { phases: next, errors }; } // ============================================================================= @@ -572,64 +572,44 @@ export class TodoTool implements AgentTool { { caption: "Initial setup (multi-phase)", call: { - ops: [ - { - op: "init", - list: [ - { phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] }, - { phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] }, - { phase: "Verification", items: ["Run cargo test"] }, - ], - }, + op: "init", + list: [ + { phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] }, + { phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] }, + { phase: "Verification", items: ["Run cargo test"] }, ], }, }, { caption: "View current state (read-only)", - call: { - ops: [{ op: "view" }], - }, + call: { op: "view" }, }, { caption: "Initial setup (single phase)", call: { - ops: [ - { - op: "init", - list: [{ phase: "Implementation", items: ["Apply fix", "Run tests"] }], - }, - ], + op: "init", + list: [{ phase: "Implementation", items: ["Apply fix", "Run tests"] }], }, }, { caption: "Complete one task", - call: { - ops: [{ op: "done", task: "Wire workspace" }], - }, + call: { op: "done", task: "Wire workspace" }, }, { caption: "Complete a whole phase", - call: { - ops: [{ op: "done", phase: "Auth" }], - }, + call: { op: "done", phase: "Auth" }, }, { caption: "Remove all tasks", - call: { - ops: [{ op: "rm" }], - }, + call: { op: "rm" }, }, { caption: "Drop one task", - call: { - ops: [{ op: "drop", task: "Run cargo test" }], - }, + call: { op: "drop", task: "Run cargo test" }, }, { caption: "Append tasks to a phase", - call: { - ops: [{ op: "append", phase: "Auth", items: ["Handle retries", "Run tests"] }], - }, + call: { op: "append", phase: "Auth", items: ["Handle retries", "Run tests"] }, }, ]; readonly loadMode = "discoverable"; @@ -646,7 +626,7 @@ export class TodoTool implements AgentTool { ): Promise> { const previousPhases = clonePhases(this.session.getTodoPhases?.() ?? []); // Pure-view calls are reads: no normalization, no state write. - const readOnly = params.ops.every(entry => entry.op === "view"); + const readOnly = params.op === "view"; const { phases: updated, errors } = readOnly ? { phases: previousPhases, errors: [] as string[] } : applyParams(clonePhases(previousPhases), params); @@ -673,15 +653,32 @@ export class TodoTool implements AgentTool { // TUI Renderer // ============================================================================= -type TodoRenderArgs = { - ops?: Array<{ - op?: string; - task?: string; - phase?: string; - items?: string[]; - }>; +type TodoRenderOp = { + op?: string; + task?: string; + phase?: string; + items?: string[]; }; +/** New single-op shape `{op,...}`; legacy `{ops:[...]}` still seen in old transcripts. */ +type TodoRenderArgs = TodoRenderOp & { + ops?: TodoRenderOp[]; +}; + +/** + * Normalize streaming/legacy render args to a flat op list. Accepts the new + * top-level `{op,...}` shape (returned as a one-element list), the legacy + * `{ops:[...]}` batch from old transcripts/collab-web, and partially-parsed + * streaming deltas (non-array `ops`, non-object entries) without crashing. + */ +function normalizeTodoArg(args: TodoRenderArgs | undefined): TodoRenderOp[] { + if (!args || typeof args !== "object") return []; + if (Array.isArray(args.ops)) { + return args.ops.filter((entry): entry is TodoRenderOp => !!entry && typeof entry === "object"); + } + return typeof args.op === "string" ? [args] : []; +} + // ============================================================================= // Phase numbering (display-only) // ============================================================================= @@ -794,7 +791,7 @@ function computeTouchedPhases( for (const transition of completedTasks) touched.add(transition.phase); // Phases explicitly named by the ops that ran. `init` replaces the whole // list, so the entire plan is fresh and every phase counts as touched. - const ops = Array.isArray(args?.ops) ? args.ops : []; + const ops = normalizeTodoArg(args); for (const op of ops) { if (!op || typeof op !== "object") continue; if (op.op === "init") { @@ -823,18 +820,17 @@ function formatPhaseSummary(phase: TodoPhase, oneBasedIndex: number, uiTheme: Th export const todoToolRenderer = { renderCall(args: TodoRenderArgs, options: RenderResultOptions, uiTheme: Theme): Component { - // `args` here is the raw partially-parsed JSON from the streaming - // tool-call delta and may not satisfy `TodoRenderArgs` at runtime: - // `parseStreamingJson` can hand back `{ ops: "[" }` mid-delta, or - // entries that are `null` / strings before fields stream. Guard - // against non-array `ops` and non-object entries so a malformed - // delta never breaks the TUI render loop (#2005). - const opsList = Array.isArray(args?.ops) ? args.ops : []; + // `args` is the raw partially-parsed JSON from the streaming tool-call + // delta and may not satisfy `TodoRenderArgs` at runtime: + // `parseStreamingJson` can hand back `{ op: 1 }` mid-delta, or a legacy + // `{ ops: "[" }` shape before fields stream. `normalizeTodoArg` guards + // both the new single-op and legacy batch shapes so a malformed delta + // never breaks the TUI render loop (#2005). + const opsList = normalizeTodoArg(args); const ops = opsList.length === 0 ? ["update"] - : opsList.map(entry => { - const e = entry && typeof entry === "object" ? entry : ({} as NonNullable); + : opsList.map(e => { const parts = [e.op ?? "update"]; if (e.task) parts.push(e.task); if (e.phase) parts.push(e.phase); diff --git a/packages/coding-agent/test/agent-session-eager-todo.test.ts b/packages/coding-agent/test/agent-session-eager-todo.test.ts index d5feea1d2..ed8b952fc 100644 --- a/packages/coding-agent/test/agent-session-eager-todo.test.ts +++ b/packages/coding-agent/test/agent-session-eager-todo.test.ts @@ -220,12 +220,8 @@ describe("AgentSession eager todo enforcement", () => { it("initializes todos once, then continues within the same user turn", async () => { scriptedResponses = [ createToolCallAssistantMessage("todo", { - ops: [ - { - op: "init", - list: [{ phase: "List worktrees", items: ["List all git worktrees in the current repository"] }], - }, - ], + op: "init", + list: [{ phase: "List worktrees", items: ["List all git worktrees in the current repository"] }], }), createAssistantMessage("real user turn handled"), ]; diff --git a/packages/coding-agent/test/tools/todo.test.ts b/packages/coding-agent/test/tools/todo.test.ts index 8266667e8..2a6232426 100644 --- a/packages/coding-agent/test/tools/todo.test.ts +++ b/packages/coding-agent/test/tools/todo.test.ts @@ -59,12 +59,8 @@ describe("TodoTool auto-start behavior", () => { it("auto-starts the first task after init", async () => { const tool = new TodoTool(createSession()); const result = await tool.execute("call-1", { - ops: [ - { - op: "init", - list: [{ phase: "Execution", items: ["status", "diagnostics"] }], - }, - ], + op: "init", + list: [{ phase: "Execution", items: ["status", "diagnostics"] }], }); const tasks = result.details?.phases[0]?.tasks ?? []; @@ -79,15 +75,11 @@ describe("TodoTool auto-start behavior", () => { it("auto-promotes the next pending task when current task is completed", async () => { const tool = new TodoTool(createSession()); await tool.execute("call-1", { - ops: [ - { - op: "init", - list: [{ phase: "Execution", items: ["status", "diagnostics"] }], - }, - ], + op: "init", + list: [{ phase: "Execution", items: ["status", "diagnostics"] }], }); - const result = await tool.execute("call-2", { ops: [{ op: "done", task: "status" }] }); + const result = await tool.execute("call-2", { op: "done", task: "status" }); const tasks = result.details?.phases[0]?.tasks ?? []; expect(tasks.map(task => task.status)).toEqual(["completed", "in_progress"]); @@ -96,7 +88,7 @@ describe("TodoTool auto-start behavior", () => { if (summary?.type !== "text") throw new Error("Expected text summary from todo"); expect(summary.text).toContain("Remaining items (1):"); expect(summary.text).toContain("diagnostics [in_progress] (Execution)"); - const completedResult = await tool.execute("call-3", { ops: [{ op: "done", task: "diagnostics" }] }); + const completedResult = await tool.execute("call-3", { op: "done", task: "diagnostics" }); const completedSummary = completedResult.content.find(part => part.type === "text"); if (completedSummary?.type !== "text") { throw new Error("Expected text summary from todo"); @@ -107,10 +99,8 @@ describe("TodoTool auto-start behavior", () => { it("renders completed tasks as checked before revealing strikethrough", async () => { const tool = new TodoTool(createSession()); - await tool.execute("call-1", { - ops: [{ op: "init", list: [{ phase: "Execution", items: ["finish"] }] }], - }); - const result = await tool.execute("call-2", { ops: [{ op: "done", task: "finish" }] }); + await tool.execute("call-1", { op: "init", list: [{ phase: "Execution", items: ["finish"] }] }); + const result = await tool.execute("call-2", { op: "done", task: "finish" }); const options = { expanded: true, isPartial: false, spinnerFrame: 0 }; const component = todoToolRenderer.renderResult(result, options, theme); @@ -124,19 +114,15 @@ it("renders completed tasks as checked before revealing strikethrough", async () expect(revealFrame).toContain("\x1b[9m"); }); -describe("TodoTool ops operations", () => { +describe("TodoTool operations", () => { it("jumps to a specific task out of order", async () => { const tool = new TodoTool(createSession()); await tool.execute("call-1", { - ops: [ - { - op: "init", - list: [{ phase: "Phase A", items: ["first", "second", "third"] }], - }, - ], + op: "init", + list: [{ phase: "Phase A", items: ["first", "second", "third"] }], }); - const result = await tool.execute("call-2", { ops: [{ op: "start", task: "third" }] }); + const result = await tool.execute("call-2", { op: "start", task: "third" }); const tasks = result.details?.phases[0]?.tasks ?? []; expect(tasks.map(task => task.status)).toEqual(["pending", "pending", "in_progress"]); @@ -145,18 +131,14 @@ describe("TodoTool ops operations", () => { it("demotes the current in_progress task when starting another", async () => { const tool = new TodoTool(createSession()); await tool.execute("call-1", { - ops: [ - { - op: "init", - list: [ - { phase: "A", items: ["a1", "a2"] }, - { phase: "B", items: ["b1"] }, - ], - }, + op: "init", + list: [ + { phase: "A", items: ["a1", "a2"] }, + { phase: "B", items: ["b1"] }, ], }); - const result = await tool.execute("call-2", { ops: [{ op: "start", task: "b1" }] }); + const result = await tool.execute("call-2", { op: "start", task: "b1" }); const allTasks = result.details?.phases.flatMap(phase => phase.tasks) ?? []; expect(allTasks.map(task => task.status)).toEqual(["pending", "pending", "in_progress"]); @@ -164,18 +146,12 @@ describe("TodoTool ops operations", () => { it("appends items to an existing phase", async () => { const tool = new TodoTool(createSession()); - await tool.execute("call-1", { - ops: [{ op: "init", list: [{ phase: "Work", items: ["First"] }] }], - }); + await tool.execute("call-1", { op: "init", list: [{ phase: "Work", items: ["First"] }] }); const result = await tool.execute("call-2", { - ops: [ - { - op: "append", - phase: "Work", - items: ["Second"], - }, - ], + op: "append", + phase: "Work", + items: ["Second"], }); const tasks = result.details?.phases[0]?.tasks ?? []; @@ -187,18 +163,12 @@ describe("TodoTool ops operations", () => { it("creates a phase when append targets a missing phase", async () => { const tool = new TodoTool(createSession()); - await tool.execute("call-1", { - ops: [{ op: "init", list: [{ phase: "Work", items: ["First"] }] }], - }); + await tool.execute("call-1", { op: "init", list: [{ phase: "Work", items: ["First"] }] }); const result = await tool.execute("call-2", { - ops: [ - { - op: "append", - phase: "Cleanup", - items: ["Remove dead code"], - }, - ], + op: "append", + phase: "Cleanup", + items: ["Remove dead code"], }); expect(result.details?.phases.map(phase => phase.name)).toEqual(["Work", "Cleanup"]); @@ -208,18 +178,14 @@ describe("TodoTool ops operations", () => { it("marks all tasks in a phase done", async () => { const tool = new TodoTool(createSession()); await tool.execute("call-1", { - ops: [ - { - op: "init", - list: [ - { phase: "Work", items: ["First", "Second"] }, - { phase: "Later", items: ["Third"] }, - ], - }, + op: "init", + list: [ + { phase: "Work", items: ["First", "Second"] }, + { phase: "Later", items: ["Third"] }, ], }); - const result = await tool.execute("call-2", { ops: [{ op: "done", phase: "Work" }] }); + const result = await tool.execute("call-2", { op: "done", phase: "Work" }); const allTasks = result.details?.phases.flatMap(phase => phase.tasks) ?? []; expect(allTasks.map(task => task.status)).toEqual(["completed", "completed", "in_progress"]); }); @@ -227,15 +193,11 @@ describe("TodoTool ops operations", () => { it("removes all tasks when rm omits task and phase", async () => { const tool = new TodoTool(createSession()); await tool.execute("call-1", { - ops: [ - { - op: "init", - list: [{ phase: "Work", items: ["First", "Second"] }], - }, - ], + op: "init", + list: [{ phase: "Work", items: ["First", "Second"] }], }); - const result = await tool.execute("call-2", { ops: [{ op: "rm" }] }); + const result = await tool.execute("call-2", { op: "rm" }); expect(result.details?.phases[0]?.tasks).toEqual([]); const summary = result.content.find(part => part.type === "text"); if (summary?.type !== "text") throw new Error("Expected text summary"); @@ -245,15 +207,11 @@ describe("TodoTool ops operations", () => { it("drops all tasks in a phase", async () => { const tool = new TodoTool(createSession()); await tool.execute("call-1", { - ops: [ - { - op: "init", - list: [{ phase: "Work", items: ["First", "Second"] }], - }, - ], + op: "init", + list: [{ phase: "Work", items: ["First", "Second"] }], }); - const result = await tool.execute("call-2", { ops: [{ op: "drop", phase: "Work" }] }); + const result = await tool.execute("call-2", { op: "drop", phase: "Work" }); const tasks = result.details?.phases[0]?.tasks ?? []; expect(tasks.map(task => task.status)).toEqual(["abandoned", "abandoned"]); }); @@ -270,7 +228,7 @@ describe("TodoTool ops operations", () => { ]); const tool = new TodoTool(session); - const result = await tool.execute("call-1", { ops: [{ op: "view" }] }); + const result = await tool.execute("call-1", { op: "view" }); const tasks = result.details?.phases[0]?.tasks ?? []; expect(tasks.map(task => task.status)).toEqual(["pending", "pending"]); @@ -284,7 +242,7 @@ describe("TodoTool ops operations", () => { it("view on an empty list reports empty, not cleared", async () => { const tool = new TodoTool(createSession()); - const result = await tool.execute("call-1", { ops: [{ op: "view" }] }); + const result = await tool.execute("call-1", { op: "view" }); const summary = result.content.find(part => part.type === "text"); if (summary?.type !== "text") throw new Error("Expected text summary"); expect(summary.text).toContain("Todo list is empty."); @@ -295,9 +253,7 @@ describe("TodoTool ops operations", () => { describe("TodoTool lenient init shapes", () => { it("accepts a flattened init with bare items and no phase", async () => { const tool = new TodoTool(createSession()); - const result = await tool.execute("call-1", { - ops: [{ op: "init", items: ["First", "Second"] }], - }); + const result = await tool.execute("call-1", { op: "init", items: ["First", "Second"] }); expect(result.isError).toBeUndefined(); expect(result.details?.phases.map(phase => phase.name)).toEqual(["Tasks"]); @@ -310,9 +266,7 @@ describe("TodoTool lenient init shapes", () => { it("honors a bare phase on a flattened init", async () => { const tool = new TodoTool(createSession()); - const result = await tool.execute("call-1", { - ops: [{ op: "init", phase: "Cleanup", items: ["Remove dead code"] }], - }); + const result = await tool.execute("call-1", { op: "init", phase: "Cleanup", items: ["Remove dead code"] }); expect(result.isError).toBeUndefined(); expect(result.details?.phases.map(phase => phase.name)).toEqual(["Cleanup"]); @@ -321,7 +275,7 @@ describe("TodoTool lenient init shapes", () => { it("still errors when init has neither list nor items", async () => { const tool = new TodoTool(createSession()); - const result = await tool.execute("call-1", { ops: [{ op: "init" }] }); + const result = await tool.execute("call-1", { op: "init" }); expect(result.isError).toBe(true); const summary = result.content.find(part => part.type === "text"); @@ -443,20 +397,16 @@ describe("todoToolRenderer.renderResult phase collapsing", () => { async function buildThreePhaseAfterDone() { const tool = new TodoTool(createSession()); await tool.execute("init", { - ops: [ - { - op: "init", - list: [ - { phase: "Alpha", items: ["a1", "a2"] }, - { phase: "Beta", items: ["b1", "b2"] }, - { phase: "Gamma", items: ["c1", "c2"] }, - ], - }, + op: "init", + list: [ + { phase: "Alpha", items: ["a1", "a2"] }, + { phase: "Beta", items: ["b1", "b2"] }, + { phase: "Gamma", items: ["c1", "c2"] }, ], }); // `done a1` keeps the active task inside Alpha (auto-promotes a2), leaving // Beta and Gamma untouched by this update. - return tool.execute("done", { ops: [{ op: "done", task: "a1" }] }); + return tool.execute("done", { op: "done", task: "a1" }); } function innerLines(component: Component): string[] { const lines = Bun.stripANSI(component.render(100).join("\n")).split("\n"); @@ -465,7 +415,8 @@ describe("todoToolRenderer.renderResult phase collapsing", () => { it("collapses untouched phases to a one-line summary while expanding the active phase", async () => { const result = await buildThreePhaseAfterDone(); const component = todoToolRenderer.renderResult(result, { expanded: false, isPartial: false }, theme, { - ops: [{ op: "done", task: "a1" }], + op: "done", + task: "a1", }); const rendered = Bun.stripANSI(component.render(100).join("\n")); // Active phase renders its full task list. @@ -493,7 +444,8 @@ describe("todoToolRenderer.renderResult phase collapsing", () => { it("shows every phase fully when manually expanded", async () => { const result = await buildThreePhaseAfterDone(); const component = todoToolRenderer.renderResult(result, { expanded: true, isPartial: false }, theme, { - ops: [{ op: "done", task: "a1" }], + op: "done", + task: "a1", }); const rendered = Bun.stripANSI(component.render(100).join("\n")); expect(rendered).toContain("b1"); @@ -504,7 +456,8 @@ describe("todoToolRenderer.renderResult phase collapsing", () => { it("drops blank separator lines between phases", async () => { const result = await buildThreePhaseAfterDone(); const component = todoToolRenderer.renderResult(result, { expanded: true, isPartial: false }, theme, { - ops: [{ op: "done", task: "a1" }], + op: "done", + task: "a1", }); // No empty body line survives between phases. expect(innerLines(component).every(line => line.length > 0)).toBe(true); @@ -520,27 +473,39 @@ describe("todoToolRenderer.renderCall malformed-args regression (#2005)", () => // retry cascade. const renderOptions = { expanded: false, isPartial: true } as const; - it("does not throw when ops is a streaming-truncated string", () => { - // Mid-stream `partialJson === '{"ops":"[{'` parses into `{ops: "[{"}`. - const args = { ops: '[{"op":"init"' } as unknown as Parameters[0]; + it("does not throw when op is a streaming-truncated number", () => { + // Mid-stream the new flat shape can surface `{ op: 1 }` before the + // discriminator string lands. + const args = { op: 1 } as unknown as Parameters[0]; expect(() => todoToolRenderer.renderCall(args, renderOptions, theme)).not.toThrow(); }); - it("does not throw when ops entries are null", () => { - // `partialParse` of `'{"ops":[null'` can hand back `{ops: [null]}` in - // intermediate states before the entry object opens. - const args = { ops: [null] } as unknown as Parameters[0]; - expect(() => todoToolRenderer.renderCall(args, renderOptions, theme)).not.toThrow(); - }); - - it("does not throw when an entry's items field is a non-array", () => { + it("does not throw when a flat op's items field is a non-array", () => { const args = { - ops: [{ op: "append", phase: "Work", items: "Second" as unknown as string[] }], + op: "append", + phase: "Work", + items: "Second" as unknown as string[], } as unknown as Parameters[0]; expect(() => todoToolRenderer.renderCall(args, renderOptions, theme)).not.toThrow(); }); - it("still renders ops summary metadata for well-formed args", () => { + it("does not throw on the legacy streaming-truncated `ops` string", () => { + // Old transcripts/collab-web still carry `{ ops: "[{" }` mid-stream; + // `normalizeTodoArg` must keep tolerating the legacy batch shape. + const args = { ops: '[{"op":"init"' } as unknown as Parameters[0]; + expect(() => todoToolRenderer.renderCall(args, renderOptions, theme)).not.toThrow(); + }); + + it("renders op summary metadata for a well-formed flat call", () => { + const args = { op: "init", items: ["a", "b", "c"] }; + const component = todoToolRenderer.renderCall(args, renderOptions, theme); + // `Text(text, 0, 0)` from `@oh-my-pi/pi-tui` exposes the content via .render(). + const rendered = Bun.stripANSI(component.render(120).join("\n")); + expect(rendered).toContain("init"); + expect(rendered).toContain("3 items"); + }); + + it("still renders legacy multi-op `ops` arrays from old transcripts", () => { const args = { ops: [ { op: "init", items: ["a", "b", "c"] }, @@ -549,12 +514,10 @@ describe("todoToolRenderer.renderCall malformed-args regression (#2005)", () => ], }; const component = todoToolRenderer.renderCall(args, renderOptions, theme); - // `Text(text, 0, 0)` from `@oh-my-pi/pi-tui` exposes the content via .render(). const rendered = Bun.stripANSI(component.render(120).join("\n")); expect(rendered).toContain("init"); expect(rendered).toContain("3 items"); expect(rendered).toContain("done"); - expect(rendered).toContain("a"); expect(rendered).toContain("append"); expect(rendered).toContain("Cleanup"); expect(rendered).toContain("1 item"); diff --git a/packages/collab-web/CHANGELOG.md b/packages/collab-web/CHANGELOG.md index 503b478a7..7a2d99efd 100644 --- a/packages/collab-web/CHANGELOG.md +++ b/packages/collab-web/CHANGELOG.md @@ -1,6 +1,9 @@ # Changelog ## [Unreleased] +### Fixed + +- Improved compatibility with legacy todo task transcripts ## [16.1.8] - 2026-06-20 @@ -118,4 +121,4 @@ ### Security -- Hardened transcript Markdown rendering by escaping embedded HTML and allowing only safe link schemes +- Hardened transcript Markdown rendering by escaping embedded HTML and allowing only safe link schemes \ No newline at end of file diff --git a/packages/collab-web/src/tool-render/tools/todo.tsx b/packages/collab-web/src/tool-render/tools/todo.tsx index a3614ef00..591e60df9 100644 --- a/packages/collab-web/src/tool-render/tools/todo.tsx +++ b/packages/collab-web/src/tool-render/tools/todo.tsx @@ -43,8 +43,18 @@ function roman(n: number): string { return out; } +/** + * Normalize call args to a flat op list. The current `todo` contract sends a + * single top-level op `{op,...}`; legacy transcripts still carry the batched + * `{ops:[...]}` shape. Non-record entries (streaming deltas) are dropped. + */ +function toOps(args: ToolRenderProps["args"]): unknown[] { + if (Array.isArray(args.ops)) return args.ops; + return typeof args.op === "string" ? [args] : []; +} + function Summary({ args }: ToolRenderProps): ReactNode { - const ops = Array.isArray(args.ops) ? args.ops : []; + const ops = toOps(args); const counts: Record = {}; const order: string[] = []; let firstTask: string | null = null; @@ -128,7 +138,7 @@ function Board({ phases }: { phases: unknown[] }): ReactNode { } function Body({ args, result }: ToolRenderProps): ReactNode { - const ops = Array.isArray(args.ops) ? args.ops : []; + const ops = toOps(args); const rec = detailsRecord(result); const phases = rec && Array.isArray(rec.phases) && !result?.isError ? rec.phases : null; return ( From 060f4004e7918221ebbeb8833034d67d97243445 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 00:55:07 +0200 Subject: [PATCH 23/43] feat(coding-agent): refactored eval tool to single-step execution - Transitioned the eval tool from batch multi-cell execution to a single-step input structure with flat parameters. - Updated core agent logic, UI components, and documentation to support state persistence across incremental eval calls. - Restricted bash tool capabilities by requiring explicit use of `read` or `find` instead of `ls` or `find`. - Added support for Ruby and Julia language runtimes to the eval tool and associated web renderers. --- packages/coding-agent/CHANGELOG.md | 5 + .../src/cli/gallery-fixtures/shell.ts | 38 ++-- .../src/modes/acp/acp-event-mapper.ts | 9 +- .../src/modes/utils/copy-targets.ts | 9 +- .../src/prompts/system/workflow-notice.md | 4 +- .../coding-agent/src/prompts/tools/bash.md | 1 + .../coding-agent/src/prompts/tools/eval.md | 19 +- .../coding-agent/src/tools/eval-render.ts | 7 +- packages/coding-agent/src/tools/eval.ts | 190 ++++++++---------- packages/coding-agent/src/tui/code-cell.ts | 2 +- .../coding-agent/src/utils/image-resize.ts | 6 +- .../test/acp-event-mapper.test.ts | 6 +- .../test/agent-session-python-cleanup.test.ts | 12 +- .../test/modes/utils/copy-targets.test.ts | 19 +- .../utils/render-initial-messages.test.ts | 6 +- .../test/streaming-preview-height.test.ts | 4 +- .../test/tools/eval-code-preview.test.ts | 2 +- .../test/tools/eval-description.test.ts | 33 +-- .../test/tools/eval-display-text.test.ts | 25 ++- .../test/tools/eval-fallback.test.ts | 25 +-- .../test/tools/eval-timeout.test.ts | 4 +- packages/collab-web/CHANGELOG.md | 9 + .../collab-web/src/tool-render/tools/eval.tsx | 35 +++- 23 files changed, 248 insertions(+), 222 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 06061a941..2dcffed17 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,9 +1,11 @@ # Changelog ## [Unreleased] + ### Breaking Changes - Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. +- Changed the `eval` tool to take a single cell per call (`{ language, code, title?, timeout?, reset? }`) instead of a `cells` array. State still persists per language across separate eval calls, tool calls, and `task` subagents, so each call is one logical step that reuses everything earlier calls defined — the array only encouraged re-importing/re-declaring the same setup in every batch. The schema, field descriptions, examples, system `eval.md`/`workflowz` helper docs, and the `[i/n]` cell-counter (now hidden for single cells) were updated to match; the renderer, ACP start-text, copy-targets, and collab-web tool view still parse legacy multi-cell transcripts. ### Added @@ -11,6 +13,9 @@ ### Changed +- Simplified `eval` tool to accept a single logical step (code block) instead of an array of cells +- Updated `eval` tool documentation to emphasize incremental, single-step execution +- Restricted `bash` tool from using `ls` or `find`, requiring the use of `read` or `find` tools - Simplified `todo` tool interface to accept a single operation directly instead of an array of ops - Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised). - Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`. diff --git a/packages/coding-agent/src/cli/gallery-fixtures/shell.ts b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts index 81bb0739d..2dc810bef 100644 --- a/packages/coding-agent/src/cli/gallery-fixtures/shell.ts +++ b/packages/coding-agent/src/cli/gallery-fixtures/shell.ts @@ -59,31 +59,23 @@ export const shellFixtures: Record = { eval: { label: "Eval", streamingArgs: { - cells: [ - { - language: "py", - code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js', - title: "load config", - }, - ], + language: "py", + code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js', + title: "load config", }, args: { - cells: [ - { - language: "py", - title: "load config", - code: [ - "import json", - "from pathlib import Path", - "", - 'data = json.loads(Path("package.json").read_text())', - 'deps = data.get("dependencies", {})', - 'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")', - 'print(f"{len(deps)} dependencies")', - "display(sorted(deps)[:3])", - ].join("\n"), - }, - ], + language: "py", + title: "load config", + code: [ + "import json", + "from pathlib import Path", + "", + 'data = json.loads(Path("package.json").read_text())', + 'deps = data.get("dependencies", {})', + 'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")', + 'print(f"{len(deps)} dependencies")', + "display(sorted(deps)[:3])", + ].join("\n"), }, result: { content: [ diff --git a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts index 4a595c699..c87f18abf 100644 --- a/packages/coding-agent/src/modes/acp/acp-event-mapper.ts +++ b/packages/coding-agent/src/modes/acp/acp-event-mapper.ts @@ -468,8 +468,13 @@ function buildEvalStartText(args: unknown): string | undefined { if (typeof args !== "object" || args === null || Array.isArray(args)) { return undefined; } - const cells = (args as EvalCellContainer).cells; - if (!Array.isArray(cells) || cells.length === 0) { + const container = args as EvalCellContainer & EvalCellLike; + const cells = Array.isArray(container.cells) + ? container.cells + : typeof container.code === "string" + ? [container] + : []; + if (cells.length === 0) { return undefined; } const lines: string[] = []; diff --git a/packages/coding-agent/src/modes/utils/copy-targets.ts b/packages/coding-agent/src/modes/utils/copy-targets.ts index 3eb0bdce0..4e830fc27 100644 --- a/packages/coding-agent/src/modes/utils/copy-targets.ts +++ b/packages/coding-agent/src/modes/utils/copy-targets.ts @@ -127,8 +127,13 @@ export function extractQuoteBlocks(text: string): QuoteBlock[] { function extractEvalCode(args: unknown): { code: string; language: string } | undefined { if (!args || typeof args !== "object") return undefined; - const cells = (args as { cells?: unknown }).cells; - if (!Array.isArray(cells)) return undefined; + const argsObj = args as { cells?: unknown; code?: unknown }; + const cells = Array.isArray(argsObj.cells) + ? argsObj.cells + : typeof argsObj.code === "string" + ? [argsObj] + : undefined; + if (!cells) return undefined; const codeBlocks: string[] = []; let language = "python"; diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index f2c79d7b6..dd392ba51 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -11,7 +11,7 @@ Worth it when the task benefits from decomposition + parallel coverage, or from -State persists across cells, so scout in one cell and fan out in the next. Every cell has: +State persists across eval calls, so scout in one call and fan out in the next. Every eval call has: - `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts. - `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one. @@ -20,7 +20,7 @@ State persists across cells, so scout in one cell and fan out in the next. Every - `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it. - `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget. -Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across cells and turns for multi-phase work, reading each result before you decide the next phase. +Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across calls and turns for multi-phase work, reading each result before you decide the next phase. diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 2b64f9948..3ca00cfd1 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -30,6 +30,7 @@ Anything below → `eval` cell, not bash: - Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps. - NEVER shell out to search content or files: `grep/rg` → `search`. +- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `find` tool (globbing). This is non-negotiable, even for a single quick listing. - Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`. diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 14a1c7923..60c428959 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -1,22 +1,23 @@ -Run code in a persistent kernel using a list of cells. +Run one step of code in a persistent kernel. -Cells run in array order. State persists per language across cells, tool calls, and `task` subagents — stage helpers/datasets/clients once, subagents reuse directly, no re-import/serialize. +**One eval call = one cell = one logical step.** State persists per language across separate eval calls, tool calls, and `task` subagents — define helpers/datasets/clients in one call, then later calls reuse them directly. -Cell fields: +Work incrementally: imports in one call, define in the next, test, then use — each its own eval call. Re-run setup ONLY after `reset`, a kernel crash, or a `NameError`/`ReferenceError` proving the state is gone. Parallelize work *within* a cell with the `parallel(thunks)` helper, not by batching steps. + +Fields: - `language` — {{#if py}}`"py"` IPython kernel{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`"js"` persistent JavaScript VM{{/if}}{{#if rb}}{{#ifAny py js}}, {{/ifAny}}`"rb"` persistent Ruby kernel{{/if}}{{#if jl}}{{#ifAny py js rb}}, {{/ifAny}}`"jl"` persistent Julia kernel{{/if}}. - `code` — cell body, verbatim. Newlines/quotes JSON-encoded; no fences, no headers. - `title` (optional) — short transcript label (e.g. `"imports"`). -- `timeout` (optional) — per-cell seconds. Raise only for heavy compute or long non-agent tool calls. -- `reset` (optional) — wipe this cell's language kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}} +- `timeout` (optional) — seconds. Raise only for heavy compute or long non-agent tool calls. +- `reset` (optional) — wipe this language's kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}} -Work incrementally — one logical step per cell (imports, define, test, use), many small cells per call; workflow notes in the assistant message or `title`, never in cell code. {{#if py}}Live event loop: use top-level `await` directly; `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}} {{#if js}}JS runs under **Bun**: Bun globals/APIs are available (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`); top-level `await`/`return` work directly.{{/if}} {{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `tree(".", max_depth: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}} {{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `tree(max_depth=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}} -Errors name the failing cell ("Cell 3 failed") — resubmit the fixed cell + any remaining. +On error, fix and re-run only the failing step — prior calls' state survives. @@ -71,3 +72,7 @@ Pipe handles through stage helpers to build a dependency graph — acyclic waves - **Acyclic only.** A node never waits on its own descendant. {{/if}} + + +Prior top-level names (`data`, `sessions`, helpers, imports) survive into the next eval call — reuse them; NEVER re-import, re-require, or re-declare a helper. Re-read a file only if it may have changed since the last read. Re-run setup only after `reset`, a crash, or a `NameError`/`ReferenceError`. + diff --git a/packages/coding-agent/src/tools/eval-render.ts b/packages/coding-agent/src/tools/eval-render.ts index ae0341fcc..58f71ba94 100644 --- a/packages/coding-agent/src/tools/eval-render.ts +++ b/packages/coding-agent/src/tools/eval-render.ts @@ -56,6 +56,9 @@ interface EvalRenderCellArg { } interface EvalRenderArgs { + language?: string; + code?: string; + title?: string; cells?: EvalRenderCellArg[]; __partialJson?: string; } @@ -81,8 +84,8 @@ function normalizeRenderLanguage(value: string | undefined): EvalLanguage { } function getRenderCells(args: EvalRenderArgs | undefined): EvalRenderCell[] { - const raw = args?.cells; - if (!Array.isArray(raw)) return []; + if (!args) return []; + const raw = Array.isArray(args.cells) ? args.cells : typeof args.code === "string" ? [args] : []; const out: EvalRenderCell[] = []; for (const cell of raw) { if (!cell || typeof cell !== "object") continue; diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index 05bb79360..45a98e873 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -38,8 +38,6 @@ const EVAL_LANGUAGE_NAME: Record = { rb: "Ruby", jl: "Julia", }; -const EVAL_CELLS_DESCRIPTION = - "cells executed in order. State persists within each language across cells and tool calls."; /** Join names as an English "or" list: ["A"]→"A", ["A","B"]→"A or B", 3+→"A, B, or C". */ function joinWithOr(items: readonly string[]): string { @@ -56,12 +54,12 @@ function describeCodeField(langs: readonly EvalLanguageToken[]): string { const replLangs = langs.filter(lang => lang === "rb" || lang === "jl"); // No persistent REPL backends → keep the original py/js phrasing verbatim so the // default (rb/jl off) wire schema stays byte-identical to the pre-feature one. - if (replLangs.length === 0) return "cell body, verbatim. Use top-level await freely."; + if (replLangs.length === 0) return "code to run in this eval call, verbatim. Use top-level await freely."; const awaitLangs = langs.filter(lang => lang === "py" || lang === "js"); const clauses: string[] = []; if (awaitLangs.length > 0) clauses.push(`Top-level \`await\` is available in ${awaitLangs.join("/")}`); clauses.push(`${replLangs.join("/")} auto-display the last expression like a REPL`); - return `cell body, verbatim. ${clauses.join("; ")}.`; + return `code to run in this eval call, verbatim. ${clauses.join("; ")}.`; } /** One-line discovery summary listing the runtimes available this session. */ @@ -86,29 +84,24 @@ function enabledEvalLanguages(backends: EvalBackendsAllowance): EvalLanguageToke const evalCellCommonFields = { "title?": type("string").describe('short label shown in transcript (e.g. "imports", "load config")'), - "timeout?": type("number").describe("per-cell timeout in seconds"), - "reset?": type("boolean").describe( - "wipe this cell's language kernel before running. Other languages are untouched.", - ), + "timeout?": type("number").describe("timeout for this eval call in seconds"), + "reset?": type("boolean").describe("wipe this language's kernel before running. Other languages are untouched."), }; /** - * Per-cell input. Each cell runs in order; state persists within a language - * across cells and across tool calls. This static schema carries the full - * language union for typing; {@link buildEvalSchema} narrows the wire copy per - * session so disabled backends are never advertised to the model. + * Per-call input: a single cell. State persists within a language across + * separate eval calls and across tool calls, so each call is one logical step + * and later calls reuse what earlier ones defined. This static schema carries + * the full language union for typing; {@link buildEvalSchema} narrows the wire + * copy per session so disabled backends are never advertised to the model. */ -const evalCellSchema = type({ - language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)), - code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)), - ...evalCellCommonFields, -}); -export type EvalCellInput = typeof evalCellSchema.infer; - export const evalSchema = type({ - cells: evalCellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION), + language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)), + ...evalCellCommonFields, + code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)), }); export type EvalToolParams = typeof evalSchema.infer; +export type EvalCellInput = EvalToolParams; /** * Build a session-scoped copy of the eval schema whose `language` enum and field @@ -118,14 +111,11 @@ export type EvalToolParams = typeof evalSchema.infer; * {@link evalSchema} (full union) remains the type-level source of truth. */ function buildEvalSchema(langs: readonly EvalLanguageToken[]): typeof evalSchema { - const cellSchema = type({ + const schema = type({ language: type.enumerated(...langs).describe(describeLanguageField(langs)), code: type("string").describe(describeCodeField(langs)), ...evalCellCommonFields, }); - const schema = type({ - cells: cellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION), - }); return schema as unknown as typeof evalSchema; } @@ -290,17 +280,10 @@ export class EvalTool implements AgentTool { readonly approval = "exec" as const; readonly formatApprovalDetails = (args: unknown): string[] => { const params = args as Partial; - const cells = Array.isArray(params.cells) ? params.cells : []; - const firstCell = cells[0] as Partial | undefined; - if (!firstCell) return []; const language = - typeof firstCell.language === "string" ? formatEvalInputLanguage(firstCell.language) : "javascript (default)"; - const code = typeof firstCell.code === "string" ? firstCell.code : ""; - const lines = [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`]; - if (cells.length > 1) { - lines.push(`+${cells.length - 1} more cell${cells.length === 2 ? "" : "s"}`); - } - return lines; + typeof params.language === "string" ? formatEvalInputLanguage(params.language) : "javascript (default)"; + const code = typeof params.code === "string" ? params.code : ""; + return [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`]; }; get summary(): string { return summarizeEvalLanguages(this.#enabledLanguages()); @@ -320,25 +303,53 @@ export class EvalTool implements AgentTool { spawns: spawnsAllowed, }); } - readonly examples: readonly ToolExample[] = [ + /** All reuse-chain examples; the `examples` getter filters by enabled languages. */ + private static readonly ALL_EXAMPLES: readonly ToolExample[] = [ { + caption: "First call — set up once", call: { - cells: [ - { - language: "py", - title: "imports", - timeout: 10, - code: "import json\nfrom pathlib import Path", - }, - { - language: "py", - title: "load config", - code: "data = json.loads(read('package.json'))\ndisplay(data)", - }, - ], + language: "py", + title: "imports", + code: "import json\nfrom pathlib import Path", + }, + }, + { + caption: "Second call — reuse, do NOT re-import", + call: { + language: "py", + title: "load config", + code: "data = json.loads(read('package.json'))\ndisplay(data)", + }, + }, + { + caption: "Third call — reuse the loaded config", + call: { + language: "py", + title: "scan deps", + code: "display(sorted(data['dependencies']))", + }, + }, + { + caption: "Ruby first call — set up once", + call: { + language: "rb", + title: "setup", + code: "require 'json'\npkg_path = 'package.json'", + }, + }, + { + caption: "Ruby second call — reuse, do NOT re-require", + call: { + language: "rb", + title: "load config", + code: "pkg = JSON.parse(read(pkg_path))\ndisplay(pkg.keys.sort)", }, }, ]; + get examples(): readonly ToolExample[] { + const langs = new Set(this.#enabledLanguages()); + return EvalTool.ALL_EXAMPLES.filter(ex => "call" in ex && langs.has(ex.call.language as EvalLanguageToken)); + } get parameters(): typeof evalSchema { const langs = this.#enabledLanguages(); if (langs.length === 0 || langs.length === EVAL_LANGUAGE_ORDER.length) return evalSchema; @@ -352,13 +363,9 @@ export class EvalTool implements AgentTool { readonly concurrency = "exclusive"; readonly strict = true; readonly intent = (args: Partial): string | undefined => { - const cells = Array.isArray(args.cells) ? args.cells : []; - const first = cells.find(c => c && typeof c === "object"); - if (!first) return "evaluating"; - const title = typeof first.title === "string" ? first.title : undefined; - const language = typeof first.language === "string" ? formatEvalInputLanguage(first.language) : "javascript"; - const label = title || `running ${language}`; - return cells.length > 1 ? `${label} (+${cells.length - 1})` : label; + const title = typeof args.title === "string" ? args.title : undefined; + const language = typeof args.language === "string" ? formatEvalInputLanguage(args.language) : "javascript"; + return title || `running ${language}`; }; readonly #proxyExecutor?: EvalProxyExecutor; @@ -398,27 +405,25 @@ export class EvalTool implements AgentTool { const session = this.session; const excludeWebP = webpExclusionForModel(session.getActiveModel?.()); - const cells: ResolvedEvalCell[] = []; - for (let i = 0; i < params.cells.length; i++) { - const cell = params.cells[i]; - const language: EvalLanguage = - cell.language === "py" - ? "python" - : cell.language === "rb" - ? "ruby" - : cell.language === "jl" - ? "julia" - : "js"; - const resolved = await resolveBackend(session, language); - cells.push({ - index: i, - title: cell.title, - code: cell.code, - timeoutMs: (cell.timeout ?? 30) * 1000, - reset: cell.reset ?? false, + const cellLanguage: EvalLanguage = + params.language === "py" + ? "python" + : params.language === "rb" + ? "ruby" + : params.language === "jl" + ? "julia" + : "js"; + const resolved = await resolveBackend(session, cellLanguage); + const cells: ResolvedEvalCell[] = [ + { + index: 0, + title: params.title, + code: params.code, + timeoutMs: (params.timeout ?? 30) * 1000, + reset: params.reset ?? false, resolved, - }); - } + }, + ]; const languages = uniqueEvalLanguages(cells); const notice = detailsNotice(cells); const sessionAbortController = new AbortController(); @@ -623,24 +628,9 @@ export class EvalTool implements AgentTool { cellResult.statusEvents = cellStatusEvents.length > 0 ? cellStatusEvents : undefined; cellResult.hasMarkdown = cellHasMarkdown || undefined; - let combinedCellOutput = ""; - if (cells.length > 1) { - const cellHeader = `[${i + 1}/${cells.length}]`; - const cellTitle = cell.title ? ` ${cell.title}` : ""; - if (cellOutput) { - combinedCellOutput = `${cellHeader}${cellTitle}\n${cellOutput}`; - } else { - combinedCellOutput = `${cellHeader}${cellTitle} (ok)`; - } - cellOutputs.push(combinedCellOutput); - } else if (cellOutput) { - combinedCellOutput = cellOutput; - cellOutputs.push(combinedCellOutput); - } - - if (combinedCellOutput) { - const prefix = cellOutputs.length > 1 ? "\n\n" : ""; - appendTail(`${prefix}${combinedCellOutput}`); + if (cellOutput) { + cellOutputs.push(cellOutput); + appendTail(cellOutput); } if (result.cancelled) { @@ -648,10 +638,7 @@ export class EvalTool implements AgentTool { pushUpdate(); const errorMsg = result.output || "Command aborted"; const combinedOutput = cellOutputs.join("\n\n"); - const outputText = - cells.length > 1 - ? `${combinedOutput}\n\nCell ${i + 1} aborted: ${errorMsg}` - : combinedOutput || errorMsg; + const outputText = combinedOutput || errorMsg; const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput); const details: EvalToolDetails = { @@ -674,12 +661,9 @@ export class EvalTool implements AgentTool { cellResult.status = "error"; pushUpdate(); const combinedOutput = cellOutputs.join("\n\n"); - const outputText = - cells.length > 1 - ? `${combinedOutput}\n\nCell ${i + 1} failed (exit code ${result.exitCode}). Earlier cells succeeded—their state persists. Fix only cell ${i + 1}.` - : combinedOutput - ? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}` - : `Command exited with code ${result.exitCode}`; + const outputText = combinedOutput + ? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}` + : `Command exited with code ${result.exitCode}`; const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput); const details: EvalToolDetails = { diff --git a/packages/coding-agent/src/tui/code-cell.ts b/packages/coding-agent/src/tui/code-cell.ts index 143e78f3d..c8c2810d6 100644 --- a/packages/coding-agent/src/tui/code-cell.ts +++ b/packages/coding-agent/src/tui/code-cell.ts @@ -80,7 +80,7 @@ function formatHeader(options: CodeCellOptions, theme: Theme): { title: string; parts.push(icon); } } - if (index !== undefined && total !== undefined) { + if (index !== undefined && total !== undefined && total > 1) { parts.push(theme.fg("accent", `[${index + 1}/${total}]`)); } if (title) { diff --git a/packages/coding-agent/src/utils/image-resize.ts b/packages/coding-agent/src/utils/image-resize.ts index 79dd52812..31be477bc 100644 --- a/packages/coding-agent/src/utils/image-resize.ts +++ b/packages/coding-agent/src/utils/image-resize.ts @@ -141,11 +141,7 @@ export async function resizeImage(img: ImageContent, options?: ImageResizeOption // lagging edge up to the floor via the default fit:"fill" resize. if (targetWidth < minDimension || targetHeight < minDimension) { const shortEdge = Math.min(targetWidth, targetHeight); - const upscale = Math.min( - minDimension / shortEdge, - opts.maxWidth / targetWidth, - opts.maxHeight / targetHeight, - ); + const upscale = Math.min(minDimension / shortEdge, opts.maxWidth / targetWidth, opts.maxHeight / targetHeight); if (upscale > 1) { targetWidth = Math.round(targetWidth * upscale); targetHeight = Math.round(targetHeight * upscale); diff --git a/packages/coding-agent/test/acp-event-mapper.test.ts b/packages/coding-agent/test/acp-event-mapper.test.ts index f9885f3f5..897f96b24 100644 --- a/packages/coding-agent/test/acp-event-mapper.test.ts +++ b/packages/coding-agent/test/acp-event-mapper.test.ts @@ -247,7 +247,7 @@ describe("ACP event mapper", () => { type: "tool_execution_start", toolCallId: "tc-eval-start", toolName: "eval", - args: { cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] }, + args: { language: "js", title: "sum", code: "return 1 + 1;" }, intent: "sum", } as AgentSessionEvent, "session-1", @@ -267,7 +267,7 @@ describe("ACP event mapper", () => { expect(update.title).toBe("[js] sum\nreturn 1 + 1;"); expect(update.kind).toBe("execute"); expect(update.status).toBe("pending"); - expect(update.rawInput).toEqual({ cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] }); + expect(update.rawInput).toEqual({ language: "js", title: "sum", code: "return 1 + 1;" }); expect(update.content).toContainEqual({ type: "content", content: { type: "text", text: "[js] sum\nreturn 1 + 1;" }, @@ -305,7 +305,7 @@ describe("ACP event mapper", () => { type: "tool_execution_start", toolCallId: "tc-eval-long-source", toolName: "eval", - args: { cells: [{ language: "js", code: source }] }, + args: { language: "js", code: source }, } as AgentSessionEvent, "session-1", ); diff --git a/packages/coding-agent/test/agent-session-python-cleanup.test.ts b/packages/coding-agent/test/agent-session-python-cleanup.test.ts index 65044c822..fe19cc47a 100644 --- a/packages/coding-agent/test/agent-session-python-cleanup.test.ts +++ b/packages/coding-agent/test/agent-session-python-cleanup.test.ts @@ -426,7 +426,7 @@ describe("AgentSession python cleanup", () => { expect(EvalTool).toBeDefined(); let toolExecutionSettled = false; const toolExecution = EvalTool! - .execute("call-id", { cells: [{ language: "py", code: "print('tool')" }] }, undefined, undefined, undefined) + .execute("call-id", { language: "py", code: "print('tool')" }, undefined, undefined, undefined) .finally(() => { toolExecutionSettled = true; }); @@ -652,13 +652,7 @@ describe("AgentSession python cleanup", () => { expect(EvalTool).toBeDefined(); const disposeSession = session.dispose(); await expect( - EvalTool!.execute( - "call-id", - { cells: [{ language: "py", code: "print('late')" }] }, - undefined, - undefined, - undefined, - ), + EvalTool!.execute("call-id", { language: "py", code: "print('late')" }, undefined, undefined, undefined), ).rejects.toThrow("Python execution is unavailable while session disposal is in progress"); await disposeSession; expect(executeSpy).not.toHaveBeenCalled(); @@ -693,7 +687,7 @@ describe("AgentSession python cleanup", () => { expect(EvalTool).toBeDefined(); const execution = EvalTool!.execute( "call-id", - { cells: [{ language: "py", code: "print('late after artifact')" }] }, + { language: "py", code: "print('late after artifact')" }, undefined, undefined, undefined, diff --git a/packages/coding-agent/test/modes/utils/copy-targets.test.ts b/packages/coding-agent/test/modes/utils/copy-targets.test.ts index d041ce4df..09be4091c 100644 --- a/packages/coding-agent/test/modes/utils/copy-targets.test.ts +++ b/packages/coding-agent/test/modes/utils/copy-targets.test.ts @@ -76,18 +76,25 @@ describe("extractLastCommand", () => { expect(extractLastCommand(messages)).toEqual({ kind: "bash", code: "echo b", language: "bash" }); }); - it("joins eval cell code and reports the cell language", () => { + it("extracts eval code from flat args and reports the language", () => { + const py = [ + assistantCalls([{ name: "eval", arguments: { language: "py", code: "print(1)" } }]), + ] as unknown as AgentMessage[]; + expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)", language: "python" }); + + const js = [ + assistantCalls([{ name: "eval", arguments: { language: "js", code: "log(1)" } }]), + ] as unknown as AgentMessage[]; + expect(extractLastCommand(js)?.language).toBe("javascript"); + }); + + it("still joins legacy multi-cell eval args from older transcripts", () => { const py = [ assistantCalls([ { name: "eval", arguments: { cells: [{ language: "py", code: "print(1)" }, { code: "print(2)" }] } }, ]), ] as unknown as AgentMessage[]; expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)\n\nprint(2)", language: "python" }); - - const js = [ - assistantCalls([{ name: "eval", arguments: { cells: [{ language: "js", code: "log(1)" }] } }]), - ] as unknown as AgentMessage[]; - expect(extractLastCommand(js)?.language).toBe("javascript"); }); }); diff --git a/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts b/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts index 63c97b1f2..010da59ff 100644 --- a/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts +++ b/packages/coding-agent/test/modes/utils/render-initial-messages.test.ts @@ -242,7 +242,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => { await Settings.init({ inMemory: true, overrides: { "terminal.showImages": true } }); setTerminalImageProtocol(ImageProtocol.Sixel); const transcript = transcriptWith([ - assistantToolCall("eval-image", "eval", { cells: [{ language: "py", code: "display(image)" }] }), + assistantToolCall("eval-image", "eval", { language: "py", code: "display(image)" }), { role: "toolResult", toolCallId: "eval-image", @@ -279,9 +279,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => { isError: false, timestamp: 2, }); - session.appendMessage( - assistantToolCall("eval-reopened", "eval", { cells: [{ language: "py", code: "display(image)" }] }), - ); + session.appendMessage(assistantToolCall("eval-reopened", "eval", { language: "py", code: "display(image)" })); session.appendMessage({ role: "toolResult", toolCallId: "eval-reopened", diff --git a/packages/coding-agent/test/streaming-preview-height.test.ts b/packages/coding-agent/test/streaming-preview-height.test.ts index 2c73301dd..0edb7191c 100644 --- a/packages/coding-agent/test/streaming-preview-height.test.ts +++ b/packages/coding-agent/test/streaming-preview-height.test.ts @@ -435,7 +435,9 @@ describe("streaming tool call preview height (bounded across renderers)", () => const hidden = total - window; const longLines = Array.from({ length: total }, (_, i) => `line-${i}`); const { lines, text } = renderPending("eval", { - cells: [{ language: "js", title: "big", code: longLines.map(line => `const ${line} = 1;`).join("\n") }], + language: "js", + title: "big", + code: longLines.map(line => `const ${line} = 1;`).join("\n"), }); expect(lines.length, "eval code preview should stay bounded").toBeLessThan(window + 10); diff --git a/packages/coding-agent/test/tools/eval-code-preview.test.ts b/packages/coding-agent/test/tools/eval-code-preview.test.ts index bfed55f27..99a3046a8 100644 --- a/packages/coding-agent/test/tools/eval-code-preview.test.ts +++ b/packages/coding-agent/test/tools/eval-code-preview.test.ts @@ -61,7 +61,7 @@ describe("eval renderer: viewport tail window for cell code", () => { it("bounds the pending preview to the same live tail window", () => { const component = evalToolRenderer.renderCall( - { cells: [{ language: "py", code }] }, + { language: "py", code }, { expanded: false, isPartial: true }, theme, ); diff --git a/packages/coding-agent/test/tools/eval-description.test.ts b/packages/coding-agent/test/tools/eval-description.test.ts index 6df2e99f4..c6c83d36f 100644 --- a/packages/coding-agent/test/tools/eval-description.test.ts +++ b/packages/coding-agent/test/tools/eval-description.test.ts @@ -17,7 +17,7 @@ function makeSession(opts: { spawns?: string | null; backends?: Record { } }); - it("hides rb/jl from the wire schema, summary, and description by default", () => { - const fields = wireCellFields(new EvalTool(makeSession({}))); + it("hides rb/jl from the wire schema, summary, description, and examples by default", () => { + const tool = new EvalTool(makeSession({})); + const fields = wireCellFields(tool); // Default config: rb/jl off → the wire schema is byte-identical to the pre-feature py/js one. expect(fields.languages).toEqual(["js", "py"]); expect(fields.languageDescription).toBe('runtime: "py" for the IPython kernel, "js" for the persistent JS VM'); - expect(fields.codeDescription).toBe("cell body, verbatim. Use top-level await freely."); - const tool = new EvalTool(makeSession({})); + expect(fields.codeDescription).toBe("code to run in this eval call, verbatim. Use top-level await freely."); expect(tool.summary).toBe("Execute Python or JavaScript code in an in-process eval backend"); expect(tool.description).not.toMatch(/ruby|julia/i); + // Examples must not advertise a disabled backend. + const exampleLangs = tool.examples.map(ex => ("call" in ex ? ex.call.language : null)); + expect(exampleLangs).toEqual(["py", "py", "py"]); + expect(tool.examples.some(ex => "call" in ex && ex.call.language === "rb")).toBe(false); }); it("advertises rb/jl across enum, descriptions, summary, and prelude once enabled", () => { @@ -104,12 +102,15 @@ describe("eval tool dynamic schema", () => { expect(fields.languageDescription).toBe( 'runtime: "py" for the IPython kernel, "js" for the persistent JS VM, "rb" for the persistent Ruby kernel, "jl" for the persistent Julia kernel', ); - expect(fields.codeDescription).toBe( - "cell body, verbatim. Top-level `await` is available in py/js; rb/jl auto-display the last expression like a REPL.", + expect(fields.codeDescription).toContain( + "code to run in this eval call, verbatim. Top-level `await` is available in py/js; rb/jl auto-display the last expression like a REPL.", ); expect(tool.summary).toBe("Execute Python, JavaScript, Ruby, or Julia code in a persistent eval backend"); expect(tool.description).toMatch(/ruby/i); expect(tool.description).toMatch(/julia/i); + // Ruby examples appear once rb is enabled. + const rbExampleLangs = tool.examples.filter(ex => "call" in ex && ex.call.language === "rb"); + expect(rbExampleLangs.length).toBe(2); }); it("advertises only the enabled subset of optional backends", () => { diff --git a/packages/coding-agent/test/tools/eval-display-text.test.ts b/packages/coding-agent/test/tools/eval-display-text.test.ts index 238c81087..963fbeb33 100644 --- a/packages/coding-agent/test/tools/eval-display-text.test.ts +++ b/packages/coding-agent/test/tools/eval-display-text.test.ts @@ -55,7 +55,8 @@ describe("EvalTool display() text surfacing", () => { const tool = new EvalTool(makeSession()); const result = await tool.execute("call-display-json", { - cells: [{ language: "js", code: "```js\ndisplay({ stdout: 'hi', exit_code: 0 });\n```\n" }], + language: "js", + code: "```js\ndisplay({ stdout: 'hi', exit_code: 0 });\n```\n", }); const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); @@ -76,7 +77,8 @@ describe("EvalTool display() text surfacing", () => { const tool = new EvalTool(makeSession()); const result = await tool.execute("call-mixed", { - cells: [{ language: "js", code: "```js\nprint('before'); display([1,2,3]);\n```\n" }], + language: "js", + code: "```js\nprint('before'); display([1,2,3]);\n```\n", }); const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); @@ -96,9 +98,8 @@ describe("EvalTool display() text surfacing", () => { const tool = new EvalTool(makeSession()); const result = await tool.execute("call-image", { - cells: [ - { language: "js", code: "```js\ndisplay({ type: 'image', data: '...', mimeType: 'image/png' });\n```\n" }, - ], + language: "js", + code: "```js\ndisplay({ type: 'image', data: '...', mimeType: 'image/png' });\n```\n", }); const imageBlocks = result.content.filter(c => c.type === "image"); @@ -125,12 +126,8 @@ describe("EvalTool display() text surfacing", () => { const tool = new EvalTool(makeSession()); const result = await tool.execute("call-large-image", { - cells: [ - { - language: "js", - code: "```js\ndisplay({ type: 'image', data: largePng, mimeType: 'image/png' });\n```\n", - }, - ], + language: "js", + code: "```js\ndisplay({ type: 'image', data: largePng, mimeType: 'image/png' });\n```\n", }); const image = result.content.find(c => c.type === "image"); @@ -154,7 +151,8 @@ describe("EvalTool display() text surfacing", () => { const tool = new EvalTool(makeSession()); const result = await tool.execute("call-empty", { - cells: [{ language: "js", code: "```js\nconst x = 1;\n```\n" }], + language: "js", + code: "```js\nconst x = 1;\n```\n", }); const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); @@ -172,7 +170,8 @@ describe("EvalTool display() text surfacing", () => { const tool = new EvalTool(makeSession()); const result = await tool.execute("call-huge", { - cells: [{ language: "js", code: "```js\ndisplay({ payload: 'x'.repeat(20000) });\n```\n" }], + language: "js", + code: "```js\ndisplay({ payload: 'x'.repeat(20000) });\n```\n", }); const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); diff --git a/packages/coding-agent/test/tools/eval-fallback.test.ts b/packages/coding-agent/test/tools/eval-fallback.test.ts index ff64bc9fd..4f710c2b2 100644 --- a/packages/coding-agent/test/tools/eval-fallback.test.ts +++ b/packages/coding-agent/test/tools/eval-fallback.test.ts @@ -67,7 +67,8 @@ describe("EvalTool language dispatch", () => { const tool = new EvalTool(makeSession()); await tool.execute("call-js", { - cells: [{ language: "js", code: "const x = 1;" }], + language: "js", + code: "const x = 1;", }); expect(jsExecuteSpy).toHaveBeenCalledTimes(1); @@ -82,26 +83,23 @@ describe("EvalTool language dispatch", () => { const tool = new EvalTool(makeSession()); await tool.execute("call-py", { - cells: [{ language: "py", code: "print('hi')" }], + language: "py", + code: "print('hi')", }); expect(pythonExecuteSpy).toHaveBeenCalledTimes(1); expect(jsExecuteSpy).not.toHaveBeenCalled(); }); - it("interleaves backends across cells in a single call", async () => { + it("dispatches each call to the backend named by its language", async () => { vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); vi.spyOn(evalIndex.pythonBackend, "isAvailable").mockResolvedValue(true); const pythonExecuteSpy = vi.spyOn(evalIndex.pythonBackend, "execute").mockResolvedValue(mockResult); const jsExecuteSpy = vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue(mockResult); const tool = new EvalTool(makeSession()); - await tool.execute("call-mixed", { - cells: [ - { language: "py", code: "x = 1" }, - { language: "js", code: "const y = 2;" }, - ], - }); + await tool.execute("call-py", { language: "py", code: "x = 1" }); + await tool.execute("call-js", { language: "js", code: "const y = 2;" }); expect(pythonExecuteSpy).toHaveBeenCalledTimes(1); expect(jsExecuteSpy).toHaveBeenCalledTimes(1); @@ -113,7 +111,8 @@ describe("EvalTool language dispatch", () => { const tool = new EvalTool(makeSession(settings)); await expect( tool.execute("call-py-disabled", { - cells: [{ language: "py", code: "print('hi')" }], + language: "py", + code: "print('hi')", }), ).rejects.toThrow(/eval\.py = false/); }); @@ -124,7 +123,8 @@ describe("EvalTool language dispatch", () => { const tool = new EvalTool(makeSession(settings)); await expect( tool.execute("call-js-disabled", { - cells: [{ language: "js", code: "const x = 1;" }], + language: "js", + code: "const x = 1;", }), ).rejects.toThrow(/eval\.js = false/); }); @@ -151,7 +151,8 @@ describe("EvalTool language dispatch", () => { await expect( tool.execute("call-js-env-disabled", { - cells: [{ language: "js", code: "const x = 1;" }], + language: "js", + code: "const x = 1;", }), ).rejects.toThrow(/PI_JS=0/); }); diff --git a/packages/coding-agent/test/tools/eval-timeout.test.ts b/packages/coding-agent/test/tools/eval-timeout.test.ts index 3f215bc82..6df40380d 100644 --- a/packages/coding-agent/test/tools/eval-timeout.test.ts +++ b/packages/coding-agent/test/tools/eval-timeout.test.ts @@ -31,7 +31,9 @@ describe("EvalTool timeout semantics", () => { // 1s budget; the cell idles for 5s and emits no status, so nothing extends // the budget — it must be cut off at the wall-clock limit. const result = await tool.execute("call-compute-timeout", { - cells: [{ language: "js", code: "await Bun.sleep(2000); return 'never';", timeout: 1 }], + language: "js", + code: "await Bun.sleep(2000); return 'never';", + timeout: 1, }); const text = result.content diff --git a/packages/collab-web/CHANGELOG.md b/packages/collab-web/CHANGELOG.md index 7a2d99efd..a8f40829b 100644 --- a/packages/collab-web/CHANGELOG.md +++ b/packages/collab-web/CHANGELOG.md @@ -1,6 +1,15 @@ # Changelog ## [Unreleased] + +### Added + +- Added support for Ruby and Julia code cells in the eval tool + +### Changed + +- Updated the eval tool view to render the new single-cell eval args (flat `language`/`code`/`title`/`timeout`/`reset`) and to highlight Ruby (`rb`) and Julia (`jl`) cells with their own syntax instead of collapsing them to Python, while still parsing legacy multi-cell `cells` arrays and framed `input` strings from older transcripts. + ### Fixed - Improved compatibility with legacy todo task transcripts diff --git a/packages/collab-web/src/tool-render/tools/eval.tsx b/packages/collab-web/src/tool-render/tools/eval.tsx index e72fae75e..ed42390ea 100644 --- a/packages/collab-web/src/tool-render/tools/eval.tsx +++ b/packages/collab-web/src/tool-render/tools/eval.tsx @@ -1,9 +1,10 @@ /** * `eval` (aliases: js, python, notebook) — code cells executed in the - * persistent kernel. Args arrive either as the modern `cells` array or as a - * legacy framed `input` string (`*** Cell`, `*** Begin LANG`, `===== info =====`); - * both render as discrete highlighted cells. When the result carries typed - * per-cell details, each cell's output is interleaved beneath its code. + * persistent kernel (py/js/rb/jl). Args arrive either as the modern single-cell + * flat shape (`language`/`code`/`title`/`timeout`/`reset`), a legacy `cells` + * array, or a legacy framed `input` string (`*** Cell`, `*** Begin LANG`, + * `===== info =====`); all render as discrete highlighted cells. When the result + * carries typed per-cell details, each cell's output is interleaved beneath its code. */ import type { ReactNode } from "react"; import { Badges, CodeBlock, InvalidArg, Note, Output, ResultImages, ResultText } from "../parts"; @@ -18,13 +19,22 @@ interface EvalCell { code: string; } -const HLJS_LANG: Record = { py: "python", js: "javascript", ts: "typescript" }; +const HLJS_LANG: Record = { + py: "python", + js: "javascript", + ts: "typescript", + rb: "ruby", + jl: "julia", +}; +/** Map an eval language token to its canonical short id, or null when unknown. */ function evalLangAlias(token: string | undefined): string | null { const t = (token ?? "").toUpperCase(); if (t === "PY" || t === "PYTHON" || t === "IPY" || t === "IPYTHON") return "py"; if (t === "JS" || t === "JAVASCRIPT") return "js"; if (t === "TS" || t === "TYPESCRIPT") return "ts"; + if (t === "RB" || t === "RUBY") return "rb"; + if (t === "JL" || t === "JULIA") return "jl"; return null; } @@ -215,7 +225,7 @@ function parseEvalCellsLegacy(input: string): EvalCell[] { const info = m[1] ?? ""; let lang = inheritedLang; let title = ""; - const langMatch = info.match(/^(py|js|ts)(?::"([^"]*)")?/); + const langMatch = info.match(/^(py|js|ts|rb|jl)(?::"([^"]*)")?/); if (langMatch) { lang = langMatch[1]; if (langMatch[2]) title = langMatch[2]; @@ -257,7 +267,7 @@ function cellsFromArgs(args: Record, name: string): EvalCell[] if (timeout !== null) attrs.push(`t=${timeout}s`); if (item.reset === true) attrs.push("rst"); out.push({ - lang: str(item.language) === "js" ? "js" : "py", + lang: evalLangAlias(str(item.language) ?? undefined) ?? "py", title: str(item.title) ?? "", attrs, code: str(item.code) ?? "", @@ -268,7 +278,14 @@ function cellsFromArgs(args: Record, name: string): EvalCell[] const input = str(args.input); if (input !== null) return parseEvalCells(input).filter(c => c.code !== "" || c.title !== ""); const code = str(args.code); - if (code !== null) return [{ lang: name === "js" ? "js" : "py", title: "", attrs: [], code }]; + if (code !== null) { + const attrs: string[] = []; + const timeout = num(args.timeout); + if (timeout !== null) attrs.push(`t=${timeout}s`); + if (args.reset === true) attrs.push("rst"); + const lang = evalLangAlias(str(args.language) ?? undefined) ?? (name === "js" ? "js" : "py"); + return [{ lang, title: str(args.title) ?? "", attrs, code }]; + } return []; } @@ -296,7 +313,7 @@ function detailCellsOf(details: Record | null): DetailCell[] { index: num(item.index) ?? i, title: str(item.title) ?? "", code: str(item.code) ?? "", - lang: language === "js" ? "js" : language !== null ? "py" : null, + lang: language !== null ? (evalLangAlias(language) ?? "py") : null, output: str(item.output) ?? "", status: str(item.status) ?? "", durationMs: num(item.durationMs), From ec16b09c76d78e0073f538f1e652e3181ea3a737 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 01:05:40 +0200 Subject: [PATCH 24/43] chore: bump models --- packages/catalog/src/models.json | 689 ++++++++++++++++++++++++++++++- 1 file changed, 673 insertions(+), 16 deletions(-) diff --git a/packages/catalog/src/models.json b/packages/catalog/src/models.json index b3da3a23f..9132b0e74 100644 --- a/packages/catalog/src/models.json +++ b/packages/catalog/src/models.json @@ -20926,7 +20926,7 @@ "huggingface": { "deepseek-ai/DeepSeek-R1": { "id": "deepseek-ai/DeepSeek-R1", - "name": "DeepSeek R1", + "name": "DeepSeek-R1", "api": "openai-completions", "provider": "huggingface", "baseUrl": "https://router.huggingface.co/v1", @@ -20935,13 +20935,13 @@ "text" ], "cost": { - "input": 3, - "output": 7, - "cacheRead": 3, - "cacheWrite": 3 + "input": 0.7, + "output": 2.5, + "cacheRead": 0, + "cacheWrite": 0 }, - "contextWindow": 131072, - "maxTokens": 8192, + "contextWindow": 64000, + "maxTokens": 32768, "thinking": { "mode": "effort", "efforts": [ @@ -21051,6 +21051,42 @@ } } }, + "deepseek-ai/DeepSeek-V4-Flash": { + "id": "deepseek-ai/DeepSeek-V4-Flash", + "name": "DeepSeek V4 Flash", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.14, + "output": 0.28, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1048576, + "maxTokens": 384000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ], + "effortMap": { + "minimal": "high", + "low": "high", + "medium": "high", + "high": "high", + "xhigh": "max" + } + } + }, "deepseek-ai/DeepSeek-V4-Pro": { "id": "deepseek-ai/DeepSeek-V4-Pro", "name": "DeepSeek V4 Pro", @@ -21087,6 +21123,85 @@ } } }, + "google/gemma-4-26B-A4B-it": { + "id": "google/gemma-4-26B-A4B-it", + "name": "Gemma 4 26B A4B IT", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.13, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "google/gemma-4-31B-it": { + "id": "google/gemma-4-31B-it", + "name": "Gemma 4 31B IT", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.14, + "output": 0.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "meta-llama/Llama-3.3-70B-Instruct": { + "id": "meta-llama/Llama-3.3-70B-Instruct", + "name": "Llama-3.3-70B-Instruct", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.59, + "output": 0.79, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 4096 + }, "meta-llama/Llama-3.3-70B-Instruct-Turbo": { "id": "meta-llama/Llama-3.3-70B-Instruct-Turbo", "name": "Llama 3.3 70B", @@ -21106,6 +21221,34 @@ "contextWindow": 131072, "maxTokens": 8192 }, + "MiniMaxAI/MiniMax-M2": { + "id": "MiniMaxAI/MiniMax-M2", + "name": "MiniMax-M2", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "low", + "medium", + "high" + ], + "requiresEffort": true + } + }, "MiniMaxAI/MiniMax-M2.1": { "id": "MiniMaxAI/MiniMax-M2.1", "name": "MiniMax-M2.1", @@ -21190,6 +21333,36 @@ "requiresEffort": true } }, + "MiniMaxAI/MiniMax-M3": { + "id": "MiniMaxAI/MiniMax-M3", + "name": "MiniMax-M3", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 1.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 524288, + "maxTokens": 128000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "moonshotai/Kimi-K2-Instruct": { "id": "moonshotai/Kimi-K2-Instruct", "name": "Kimi-K2-Instruct", @@ -21318,6 +21491,36 @@ ] } }, + "moonshotai/Kimi-K2.7-Code": { + "id": "moonshotai/Kimi-K2.7-Code", + "name": "Kimi K2.7 Code", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.95, + "output": 4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", @@ -21345,6 +21548,34 @@ ] } }, + "Qwen/Qwen3-235B-A22B": { + "id": "Qwen/Qwen3-235B-A22B", + "name": "Qwen3 235B-A22B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.2, + "output": 0.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 40960, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "Qwen/Qwen3-235B-A22B-Thinking-2507": { "id": "Qwen/Qwen3-235B-A22B-Thinking-2507", "name": "Qwen3-235B-A22B-Thinking-2507", @@ -21374,6 +21605,53 @@ "requiresEffort": true } }, + "Qwen/Qwen3-32B": { + "id": "Qwen/Qwen3-32B", + "name": "Qwen3 32B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.29, + "output": 0.59, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "Qwen/Qwen3-Coder-30B-A3B-Instruct": { + "id": "Qwen/Qwen3-Coder-30B-A3B-Instruct", + "name": "Qwen3-Coder 30B-A3B Instruct", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.07, + "output": 0.26, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536 + }, "Qwen/Qwen3-Coder-480B-A35B-Instruct": { "id": "Qwen/Qwen3-Coder-480B-A35B-Instruct", "name": "Qwen3-Coder-480B-A35B-Instruct", @@ -21450,6 +21728,93 @@ "contextWindow": 262144, "maxTokens": 131072 }, + "Qwen/Qwen3.5-122B-A10B": { + "id": "Qwen/Qwen3.5-122B-A10B", + "name": "Qwen3.5 122B-A10B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.4, + "output": 3.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "Qwen/Qwen3.5-27B": { + "id": "Qwen/Qwen3.5-27B", + "name": "Qwen3.5 27B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.3, + "output": 2.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "Qwen/Qwen3.5-35B-A3B": { + "id": "Qwen/Qwen3.5-35B-A3B", + "name": "Qwen3.5 35B-A3B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, "Qwen/Qwen3.5-397B-A17B": { "id": "Qwen/Qwen3.5-397B-A17B", "name": "Qwen3.5-397B-A17B", @@ -21479,6 +21844,93 @@ ] } }, + "Qwen/Qwen3.5-9B": { + "id": "Qwen/Qwen3.5-9B", + "name": "Qwen3.5 9B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.17, + "output": 0.25, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "Qwen/Qwen3.6-35B-A3B": { + "id": "Qwen/Qwen3.6-35B-A3B", + "name": "Qwen3.6 35B-A3B", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.15, + "output": 0.95, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } + }, + "stepfun-ai/Step-3.5-Flash": { + "id": "stepfun-ai/Step-3.5-Flash", + "name": "Step 3.5 Flash", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.1, + "output": 0.3, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 256000, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "XiaomiMiMo/MiMo-V2-Flash": { "id": "XiaomiMiMo/MiMo-V2-Flash", "name": "MiMo-V2-Flash", @@ -21506,6 +21958,123 @@ ] } }, + "zai-org/GLM-4.5": { + "id": "zai-org/GLM-4.5", + "name": "GLM-4.5", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.6, + "output": 2.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 98304, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/GLM-4.5-Air": { + "id": "zai-org/GLM-4.5-Air", + "name": "GLM-4.5-Air", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.13, + "output": 0.85, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 98304, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/GLM-4.5V": { + "id": "zai-org/GLM-4.5V", + "name": "GLM-4.5V", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.6, + "output": 1.8, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 65536, + "maxTokens": 16384, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, + "zai-org/GLM-4.6": { + "id": "zai-org/GLM-4.6", + "name": "GLM-4.6", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.55, + "output": 2.2, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 204800, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "zai-org/GLM-4.7": { "id": "zai-org/GLM-4.7", "name": "GLM-4.7", @@ -21621,6 +22190,35 @@ "xhigh" ] } + }, + "zai-org/GLM-5.2": { + "id": "zai-org/GLM-5.2", + "name": "GLM-5.2", + "api": "openai-completions", + "provider": "huggingface", + "baseUrl": "https://router.huggingface.co/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 1.4, + "output": 4.4, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 131072, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } } }, "kilo": { @@ -46112,11 +46710,11 @@ }, "Qwen/Qwen3-235B-A22B": { "id": "Qwen/Qwen3-235B-A22B", - "name": "Qwen/Qwen3-235B-A22B", + "name": "Qwen3 235B-A22B", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text" ], @@ -46127,7 +46725,16 @@ "cacheWrite": 0 }, "contextWindow": 131072, - "maxTokens": 8192 + "maxTokens": 8192, + "thinking": { + "mode": "effort", + "efforts": [ + "minimal", + "low", + "medium", + "high" + ] + } }, "qwen/Qwen3-235B-A22B-Instruct-2507": { "id": "qwen/Qwen3-235B-A22B-Instruct-2507", @@ -46647,13 +47254,14 @@ }, "Qwen/Qwen3.6-35B-A3B": { "id": "Qwen/Qwen3.6-35B-A3B", - "name": "Qwen/Qwen3.6-35B-A3B", + "name": "Qwen3.6 35B-A3B", "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -47716,6 +48324,25 @@ "contextWindow": null, "maxTokens": null }, + "sakana/fugu-ultra": { + "id": "sakana/fugu-ultra", + "name": "sakana/fugu-ultra", + "api": "openai-completions", + "provider": "nanogpt", + "baseUrl": "https://nano-gpt.com/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": null + }, "Salesforce/Llama-xLAM-2-70b-fc-r": { "id": "Salesforce/Llama-xLAM-2-70b-fc-r", "name": "Salesforce/Llama-xLAM-2-70b-fc-r", @@ -69124,9 +69751,9 @@ "text" ], "cost": { - "input": 1, - "output": 4, - "cacheRead": 0.18, + "input": 0.98, + "output": 3.08, + "cacheRead": 0.182, "cacheWrite": 0 }, "contextWindow": 1048576, @@ -69694,7 +70321,7 @@ "together": { "deepseek-ai/DeepSeek-R1": { "id": "deepseek-ai/DeepSeek-R1", - "name": "DeepSeek R1", + "name": "DeepSeek-R1", "api": "openai-completions", "provider": "together", "baseUrl": "https://api.together.xyz/v1", @@ -77765,6 +78392,36 @@ ] } }, + "sakana/fugu-ultra": { + "id": "sakana/fugu-ultra", + "name": "Fugu Ultra", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 1000000, + "thinking": { + "mode": "budget", + "efforts": [ + "minimal", + "low", + "medium", + "high", + "xhigh" + ] + } + }, "stepfun/step-3.5-flash": { "id": "stepfun/step-3.5-flash", "name": "Step 3.5 Flash", From ecdff42513f0c639001c1b36acfa77a82c82d4a2 Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 23:38:11 +0000 Subject: [PATCH 25/43] fix(session-selector): keep /resume header pinned after delete MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The delete-confirmation dialog mounted as a sibling below the picker's bottom border briefly grew the picker past the terminal height. The TUI's append-only renderer committed the picker's top rows (header + first sessions) into native scrollback to fit the dialog within the viewport. When the dialog closed and the picker re-rendered shorter, `windowTop` stayed pinned at `#committedRows`, leaving the picker stranded below the committed prefix — the user saw the header scrolled off the top. SessionList now exposes `setExternalReserveRows`; SessionSelectorComponent reserves the dialog's worst-case height while the dialog is mounted so the picker's total rendered output stays within the terminal viewport and the TUI never commits its rows. Fixes #3283 --- packages/coding-agent/CHANGELOG.md | 1 + .../src/modes/components/session-selector.ts | 85 +++++++++---- .../session-selector-scroll-stability.test.ts | 117 ++++++++++++++++++ 3 files changed, 182 insertions(+), 21 deletions(-) create mode 100644 packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index cde28978f..deb544cc5 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,6 +13,7 @@ - Prevented `/handoff` from executing while a response is streaming to avoid session corruption - Fixed `/handoff` cold-missing the provider prompt cache. Handoff generation now builds its request through the same pipeline a live turn uses (`convertMessagesToLlm` + `Agent.buildSideRequestContext` + `prepareSimpleStreamOptions`, via the new `generateHandoffFromContext`), so it reuses the live system prompt, normalized tools, transformed/obfuscated message history, and — critically — a stable `promptCacheKey` with a unique side `sessionId`. Previously the oneshot sent no cache-routing key and skipped the `transformContext`/`transformProviderContext` and tool/message normalization the loop applies, so its prefix never matched what the turn populated and every handoff re-read the whole context uncached. Mirrors the cache-preserving path already used by `/btw` and `/omfg`. - Fixed `/handoff` (and the RPC `handoff` command) resetting the agent while a response was still streaming, which let the live turn keep emitting into the torn-down session. Manual handoff now refuses while a prompt is in flight (matching `/fork` and `/move`); the auto-handoff path is unaffected. +- Fixed the `/resume` session picker scrolling down after a session is deleted. The delete-confirmation dialog mounted below the picker's bottom border briefly grew the picker past the terminal height; the TUI committed the picker's header rows into native scrollback to fit, and when the dialog closed the picker re-rendered shorter with `windowTop` pinned at the new commit boundary — leaving the header stranded above the viewport. The picker now reserves a row budget for the dialog while it is mounted so the picker's total rendered output stays within the terminal viewport and never commits ([#3283](https://github.com/can1357/oh-my-pi/issues/3283)) ## [16.1.15] - 2026-06-22 diff --git a/packages/coding-agent/src/modes/components/session-selector.ts b/packages/coding-agent/src/modes/components/session-selector.ts index c6f403894..055dd3116 100644 --- a/packages/coding-agent/src/modes/components/session-selector.ts +++ b/packages/coding-agent/src/modes/components/session-selector.ts @@ -172,6 +172,15 @@ class SessionList implements Component { onDeleteRequest?: (session: SessionInfo) => void; + // Extra row count the picker must keep free for a sibling component + // (currently the delete-confirmation dialog) mounted below the bottom + // border. Set by `SessionSelectorComponent` while the dialog is on screen + // so the SessionList shrinks its visible window and the picker's total + // rendered output stays within the terminal height. Without this the + // picker overflows by ~dialog-height rows, the TUI commits those top rows + // to native scrollback to fit, and when the dialog closes the picker + // header is stranded above the viewport — issue #3283. + #externalReserveRows = 0; #allSessions: SessionInfo[]; #showCwd: boolean; readonly #historyMatcher?: SessionHistoryMatcher; @@ -198,23 +207,41 @@ class SessionList implements Component { }; } + /** + * Reserve `rows` of vertical budget for a sibling component rendered + * outside the SessionList (e.g. the delete-confirmation dialog). The + * visible-entry window shrinks accordingly so the picker frame stays + * within the terminal viewport while the sibling is mounted. + */ + setExternalReserveRows(rows: number): void { + const next = Math.max(0, Math.trunc(rows)); + if (next === this.#externalReserveRows) return; + this.#externalReserveRows = next; + } + /** * Number of sessions to show at once, sized so the whole picker fits the * current viewport instead of pushing its header/search off the top. * - * Budget = rows − chrome − reserve, divided by the worst-case per-session - * height. Chrome (12) is the surrounding spacers/borders/header (7) plus the - * list's search line, blank, scroll indicator, blank, and hint (5). A titled - * session is the tallest item at 4 lines (title + preview + metadata + - * blank); budgeting for that guarantees no overflow even when every visible - * entry has a title. The reserve covers below-editor hook widgets / cursor. + * Budget = rows − chrome − reserve − externalReserve, divided by the + * worst-case per-session height. Chrome (12) is the surrounding + * spacers/borders/header (7) plus the list's search line, blank, scroll + * indicator, blank, and hint (5). A titled session is the tallest item at + * 4 lines (title + preview + metadata + blank); budgeting for that + * guarantees no overflow even when every visible entry has a title. The + * reserve covers below-editor hook widgets / cursor. The external reserve + * is non-zero only while a sibling overlay (the delete-confirmation + * dialog) is mounted below the bottom border, and is allowed to drive + * the count down to zero so the dialog never pushes the picker past the + * terminal height. */ #visibleCount(): number { const CHROME = 12; const PER_SESSION = 4; const RESERVE = 1; - const budget = this.#getTerminalRows() - CHROME - RESERVE; - return Math.max(2, Math.floor(budget / PER_SESSION)); + const budget = this.#getTerminalRows() - CHROME - RESERVE - this.#externalReserveRows; + const minimum = this.#externalReserveRows > 0 ? 0 : 2; + return Math.max(minimum, Math.floor(budget / PER_SESSION)); } /** Replace the visible dataset, e.g. when toggling folder/all-projects scope. */ @@ -580,8 +607,28 @@ export class SessionSelectorComponent extends Container { this.#messageContainer.addChild(new Spacer(1)); } + // Rows the delete-confirmation dialog adds below the picker's bottom + // border. Used to shrink the SessionList while the dialog is mounted so + // the picker's total rendered output never exceeds the terminal height. + // Sized generously (title + session-name + spacer + 2 options + spacer + + // hint + leading/trailing blanks) so the SessionList always concedes + // enough room for the dialog as it grows, even when the displayed session + // name wraps. + static readonly #DELETE_DIALOG_RESERVE_ROWS = 12; + #showDeleteConfirmation(session: SessionInfo): void { const displayName = session.title || session.firstMessage.slice(0, 40) || session.id; + const closeDialog = () => { + this.removeChild(this.#confirmationDialog!); + this.#confirmationDialog = null; + // Release the dialog's row reservation BEFORE requesting the + // rerender so the SessionList can grow back to its full window + // in the same frame the dialog disappears. Otherwise the picker + // re-renders short, leaving a band of blank rows beneath it for + // one frame (visible as a flicker / "still scrolled" feel). + this.#sessionList.setExternalReserveRows(0); + this.#onRequestRender?.(); + }; this.#confirmationDialog = new HookSelectorComponent( `Delete session?\n${displayName}`, ["Yes", "No"], @@ -597,21 +644,17 @@ export class SessionSelectorComponent extends Container { this.#showError(err instanceof Error ? err.message : String(err)); } } - // Close confirmation dialog - this.removeChild(this.#confirmationDialog!); - this.#confirmationDialog = null; - // Request rerender - this.#onRequestRender?.(); - }, - () => { - // Cancel - close confirmation dialog - this.removeChild(this.#confirmationDialog!); - this.#confirmationDialog = null; - // Request rerender - this.#onRequestRender?.(); + closeDialog(); }, + closeDialog, ); - // Show confirmation dialog + // Shrink the SessionList by the dialog's worst-case height BEFORE + // mounting the dialog so the very first frame containing the dialog + // already fits the terminal viewport. Without this the dialog's first + // render still overflows and the TUI commits the picker's top rows + // to native scrollback before the SessionList has a chance to react + // — issue #3283. + this.#sessionList.setExternalReserveRows(SessionSelectorComponent.#DELETE_DIALOG_RESERVE_ROWS); this.addChild(this.#confirmationDialog); } diff --git a/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts b/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts new file mode 100644 index 000000000..0e2c37f60 --- /dev/null +++ b/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts @@ -0,0 +1,117 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-listing"; +import { TUI } from "@oh-my-pi/pi-tui"; +import { StressRenderScheduler } from "../../../../tui/test/render-stress-scheduler"; +import { VirtualTerminal } from "../../../../tui/test/virtual-terminal"; + +beforeAll(() => { + initTheme(); +}); + +function makeSessions(count: number): SessionInfo[] { + return Array.from({ length: count }, (_, i) => ({ + path: `/work/SESSION_${i}.jsonl`, + id: `id-${i}`, + cwd: "/work", + title: `SESSION_${i}`, + created: new Date("2024-01-01T00:00:00Z"), + modified: new Date("2024-01-02T00:00:00Z"), + messageCount: 1, + size: 1024, + firstMessage: `body content ${i}`, + allMessagesText: `body content ${i}`, + })); +} + +describe("issue #3283: /resume picker scrolls down after deleting a session", () => { + it("keeps the picker header pinned at the same viewport row before and after a delete", async () => { + const term = new VirtualTerminal(80, 24, 4096); + const scheduler = new StressRenderScheduler(); + const tui = new TUI(term, undefined, { renderScheduler: scheduler }); + const selector = new SessionSelectorComponent( + makeSessions(20), + () => {}, + () => {}, + () => {}, + { + getTerminalRows: () => term.rows, + onDelete: async () => true, + }, + ); + selector.setOnRequestRender(() => tui.requestRender()); + tui.addChild(selector); + tui.setFocus(selector); + + try { + tui.start(); + await scheduler.drain(term); + + const headerRowBefore = term.getViewport().findIndex(row => Bun.stripANSI(row).includes("Resume Session")); + expect(headerRowBefore).toBeGreaterThanOrEqual(0); + + // Press Delete (CSI 3 ~) to open the confirmation dialog, then + // Enter to accept "Yes". + selector.handleInput("\x1b[3~"); + tui.requestRender(); + await scheduler.drain(term); + selector.handleInput("\n"); + // onDelete is async; let its microtasks flush before draining renders. + for (let i = 0; i < 8; i++) await Promise.resolve(); + await scheduler.drain(term); + const viewport = term.getViewport().map(row => Bun.stripANSI(row).trimEnd()); + const headerRowAfter = viewport.findIndex(row => row.includes("Resume Session")); + + // Regression: dialog growing the frame and then shrinking must + // not push the picker header further down into the viewport + // (committed scrollback rows from the dialog frame). + expect(headerRowAfter).toBeGreaterThanOrEqual(0); + expect(headerRowAfter).toBe(headerRowBefore); + } finally { + tui.stop(); + await term.flush(); + } + }); + it("keeps the picker header pinned even when the delete dialog is canceled", async () => { + const term = new VirtualTerminal(80, 24, 4096); + const scheduler = new StressRenderScheduler(); + const tui = new TUI(term, undefined, { renderScheduler: scheduler }); + const selector = new SessionSelectorComponent( + makeSessions(20), + () => {}, + () => {}, + () => {}, + { + getTerminalRows: () => term.rows, + onDelete: async () => true, + }, + ); + selector.setOnRequestRender(() => tui.requestRender()); + tui.addChild(selector); + tui.setFocus(selector); + + try { + tui.start(); + await scheduler.drain(term); + const headerRowBefore = term.getViewport().findIndex(row => Bun.stripANSI(row).includes("Resume Session")); + expect(headerRowBefore).toBeGreaterThanOrEqual(0); + + // Open dialog, then Esc to cancel without deleting. + selector.handleInput("\x1b[3~"); + tui.requestRender(); + await scheduler.drain(term); + selector.handleInput("\x1b"); + await scheduler.drain(term); + + const viewport = term.getViewport().map(row => Bun.stripANSI(row).trimEnd()); + const headerRowAfter = viewport.findIndex(row => row.includes("Resume Session")); + expect(headerRowAfter).toBe(headerRowBefore); + // Dialog gone, no scroll-down artefact. + expect(viewport.some(row => row.includes("Delete session?"))).toBe(false); + } finally { + tui.stop(); + await term.flush(); + } + }); +}); From 1dd78b207eb4dd3d7c542cddce36a7f815960a78 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 01:39:24 +0200 Subject: [PATCH 26/43] feat(coding-agent): removed unused eval helper functions - Removed deprecated eval prelude helpers `append`, `tree`, `diff`, `sort`, `uniq`, and `counter` from all supported runtimes. - Cleaned up runtime implementations, protocol definitions, and UI rendering logic associated with the removed helpers. - Updated project documentation, prompts, and test suites to reflect the reduced helper API surface. - Recorded functional changes in the package changelog. --- docs/tools/eval.md | 10 +- packages/coding-agent/CHANGELOG.md | 8 +- .../__tests__/helpers-local-roots.test.ts | 7 +- .../src/eval/__tests__/julia-prelude.test.ts | 29 --- packages/coding-agent/src/eval/jl/prelude.jl | 215 ------------------ .../src/eval/js/shared/helpers.ts | 115 +--------- .../src/eval/js/shared/prelude.txt | 23 -- .../src/eval/js/shared/runtime.ts | 6 - .../coding-agent/src/eval/js/shared/types.ts | 2 +- .../src/eval/js/worker-protocol.ts | 2 +- packages/coding-agent/src/eval/py/executor.ts | 2 +- packages/coding-agent/src/eval/py/prelude.py | 97 -------- packages/coding-agent/src/eval/rb/prelude.rb | 187 +-------------- .../src/internal-urls/local-protocol.ts | 2 +- .../coding-agent/src/prompts/tools/eval.md | 10 +- .../src/tools/browser/tab-worker.ts | 2 +- .../coding-agent/src/tools/eval-render.ts | 12 - .../test/core/ruby-runner.integration.test.ts | 5 +- 18 files changed, 23 insertions(+), 711 deletions(-) diff --git a/docs/tools/eval.md b/docs/tools/eval.md index edbb2f730..be4dda38a 100644 --- a/docs/tools/eval.md +++ b/docs/tools/eval.md @@ -129,16 +129,14 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod - Module cache is busted for **local** imports between cells so edits to source files are picked up without restarting the runtime. `__omp_import__` deletes `require.cache[absPath]` before re-importing whenever the original specifier is a filesystem path: relative (`./x`, `../x`, `.`, `..`), POSIX-absolute (`/...`), home-prefixed (`~/...`), or Windows drive-letter (`C:\...` / `C:/...`). Bare specifiers (`react`, `lodash/x`) and URL/scheme specifiers (`node:fs`, `file://...`, `https://...`) are left in cache so package identity stays stable across cells. The cache-bust only fires when the resolved target is an absolute path — unresolved bare-package fallbacks (`resolveImportSpecifier()` returning the original specifier) skip it. - The prelude installs globals: - `display`, `print`, and a `console` bridge - - `read`, `write`, `append`, `sort`, `uniq`, `counter`, `diff`, `tree`, `env`, `output` + - `read`, `write`, `env`, `output` - `tool.(args)` proxy for arbitrary session tool calls - `completion(prompt, opts?)` for oneshot, stateless model calls (see _Oneshot completion helper_ below) - `agent(prompt, opts?)` for a single subagent call, plus `parallel()` / `pipeline()` bounded-pool helpers (see _Subagent helper_ below) - `log(message)`, `phase(title)`, and `budget` (live token-budget view via async `budget.total()` / `budget.spent()` / `budget.remaining()` / `budget.hard()`) -- JS helpers that touch the host/runtime boundary are async and `await`able; pure text helpers (`sort`, `uniq`, `counter`) return synchronously but may still be safely awaited. +- JS host/runtime helpers (`read`, `write`, `output`) are async and `await`able; `env` returns synchronously. - JS helper options may be passed either positionally in the Python order or as a trailing options object. `null` and `undefined` skip positional slots: - `await read(path, offset?, limit?)` or `await read(path, { offset?, limit? })` - - `await tree(path = ".", maxDepth?, showHidden?)` or `await tree(path, { maxDepth?, showHidden? })` - - `sort(text, reverse?, unique?)`, `uniq(text, count?)`, `counter(items, limit?, reverse?)` - `await agent(prompt, agent?, model?, label?, schema?)` or `await agent(prompt, { agent?, model?, label?, schema?, handle? })` - `await parallel([() => agent("a"), () => agent("b")])` - `await pipeline(items, stage1, stage2)` @@ -215,7 +213,7 @@ A single tool call can mix Python and JS cells. Persistence is per language runt ## Side Effects - Filesystem - - JS/Python prelude helpers can read, write, append, diff, and traverse filesystem paths under the session cwd or absolute paths. + - JS/Python prelude helpers can read and write filesystem paths under the session cwd or absolute paths. - JS helper `read()` auto-delegates any non-`local://` scheme URI (`agent://`, `artifact://`, `https://`, ...) to `tool.read(...)` (honoring an `offset`/`limit` line selector), resolves `local://` under its mapped root, reads plain/absolute filesystem paths directly, and rejects directory paths. - Output may spill to an artifact file via `OutputSink`. - Network @@ -280,7 +278,7 @@ A single tool call can mix Python and JS cells. Persistence is per language runt - Backend selection is strictly explicit per cell: `language` must be `"py"` or `"js"`. The previous `*** Cell` header parser, the `eval.lark` constrained grammar, and the sniffer-based fallback have all been removed. - `EvalTool.customFormat` no longer exists. Tool calls flow through the standard JSON schema; there is no Lark-constrained sampling path. - `tool.()` exists in both JS and Python. Python calls route through a per-run loopback bridge keyed by the current cell id. -- `read()` delegates non-`local://` scheme URIs to `tool.read`, resolves `local://` under its injected root, and resolves plain paths against the session cwd or an absolute filesystem path; `resolveRegularFile()` rejects directory paths. `write()`/`append()` accept `local://` and plain paths but reject any other `scheme://` via `resolveHelperPath()` (`Protocol paths are not supported by write()`). +- `read()` delegates non-`local://` scheme URIs to `tool.read`, resolves `local://` under its injected root, and resolves plain paths against the session cwd or an absolute filesystem path; `resolveRegularFile()` rejects directory paths. `write()` accepts `local://` and plain paths but rejects any other `scheme://` via `resolveHelperPath()` (`Protocol paths are not supported by write()`). - Python helper `output(...)` depends on `PI_ARTIFACTS_DIR` or `PI_SESSION_FILE`; it fails outside a session-backed run. - `display()` can produce text and structured outputs from the same value; the renderer prefers markdown over `text/plain` when both exist. - JS static imports are rewritten only at top level. Nested imports stay invalid and surface normal JS syntax/runtime errors. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 2dcffed17..af44d0ce7 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,7 +1,6 @@ # Changelog ## [Unreleased] - ### Breaking Changes - Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. @@ -20,6 +19,13 @@ - Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised). - Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`. +### Removed + +- Removed `append`, `tree`, and `diff` eval helper functions from Python, JavaScript, and Ruby +- Removed `sort`, `uniq`, and `counter` text processing eval helpers from Python, JavaScript, and Ruby +- Removed the `append(path, content)`, `tree(path, max_depth?, show_hidden?)`, and `diff(a, b)` eval prelude helpers from every workflow runtime (Python, JavaScript, Ruby, Julia), along with their status renderers, icon entries, and tool/`docs` references. Use `write`/`read` for file mutation and `tool.(...)` for richer filesystem operations. +- Removed the `sort(text, reverse?, unique?)`, `uniq(text, count?)`, and `counter(items, limit?, reverse?)` eval text helpers from the Python, JavaScript, and Ruby prelude surfaces (Julia never defined them), along with the JS `HelperBundle`/`HelperOptions` members and `docs` references. Sort/dedupe/count inline in cell code instead. + ### Fixed - Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path. diff --git a/packages/coding-agent/src/eval/__tests__/helpers-local-roots.test.ts b/packages/coding-agent/src/eval/__tests__/helpers-local-roots.test.ts index 8bde5f649..a225ad506 100644 --- a/packages/coding-agent/src/eval/__tests__/helpers-local-roots.test.ts +++ b/packages/coding-agent/src/eval/__tests__/helpers-local-roots.test.ts @@ -4,7 +4,7 @@ import { TempDir } from "@oh-my-pi/pi-utils/temp"; import { createHelpers, type HelperContext } from "../js/shared/helpers"; /** - * The eval helpers (`read`/`write`/`append`) must substitute injected on-disk + * The eval helpers (`read`/`write`) must substitute injected on-disk * roots for internal-URL schemes. Without it, `write("local://x.md")` hits a * stdlib `path.resolve` that collapses `local://` to `local:/`, creating a junk * `local:` directory under the cwd instead of landing where `read local://x.md` @@ -20,7 +20,7 @@ function makeCtx(cwd: string, roots: Record): HelperContext { } describe("eval js helpers internal-url resolution", () => { - it("writes, reads, and appends local:// under the injected root", async () => { + it("writes and reads local:// under the injected root", async () => { using tmp = TempDir.createSync("@eval-helpers-local-"); const root = path.join(tmp.path(), "local"); const helpers = createHelpers(makeCtx(tmp.path(), { local: root })); @@ -30,9 +30,6 @@ describe("eval js helpers internal-url resolution", () => { expect(await Bun.file(written).text()).toBe("hello"); expect(await helpers.read("local://notes/merge-map.md")).toBe("hello"); - await helpers.append("local://notes/merge-map.md", " world"); - expect(await helpers.read("local://notes/merge-map.md")).toBe("hello world"); - // Regression: no literal `local:` directory created under the cwd. expect(await Bun.file(path.join(tmp.path(), "local:")).exists()).toBe(false); expect(await Bun.file(path.join(tmp.path(), "local:", "notes", "merge-map.md")).exists()).toBe(false); diff --git a/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts b/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts index b59a92db5..3775d8b2c 100644 --- a/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts +++ b/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts @@ -11,35 +11,6 @@ describe.skipIf(!HAS_JULIA)("eval Julia prelude helpers", () => { await disposeJuliaKernelSessionsByOwner(OWNER_ID); }); - it("supports tree keyword options and unified diff", async () => { - using tempDir = TempDir.createSync("@omp-eval-julia-helpers-"); - await Bun.write(path.join(tempDir.path(), "a.txt"), "same\nold\n"); - await Bun.write(path.join(tempDir.path(), "b.txt"), "same\nnew\n"); - await Bun.write(path.join(tempDir.path(), "dir", "child.txt"), "child"); - - const result = await executeJulia( - ` -d = diff("a.txt", "b.txt") -println("DIFF_DELETE=", occursin("-old", d)) -println("DIFF_ADD=", occursin("+new", d)) -t = tree(".", max_depth=2) -println("TREE_CHILD=", occursin("child.txt", t)) -nothing -`, - { - cwd: tempDir.path(), - sessionId: `julia-prelude-diff:${crypto.randomUUID()}`, - kernelOwnerId: OWNER_ID, - reset: true, - }, - ); - - expect(result.exitCode).toBe(0); - expect(result.output).toContain("DIFF_DELETE=true"); - expect(result.output).toContain("DIFF_ADD=true"); - expect(result.output).toContain("TREE_CHILD=true"); - }, 30_000); - it("supports output ranges, JSON queries, metadata, and ANSI stripping", async () => { using tempDir = TempDir.createSync("@omp-eval-julia-output-"); const artifactsDir = path.join(tempDir.path(), "session-artifacts"); diff --git a/packages/coding-agent/src/eval/jl/prelude.jl b/packages/coding-agent/src/eval/jl/prelude.jl index f199cf4f0..8136d2dbd 100644 --- a/packages/coding-agent/src/eval/jl/prelude.jl +++ b/packages/coding-agent/src/eval/jl/prelude.jl @@ -146,221 +146,6 @@ function Base.write(path::AbstractString, content::Any) return resolved end -function append(path, content) - resolved = __omp_resolve_path(string(path)) - mkpath(dirname(resolved)) - open(resolved, "a") do f - Base.write(f, string(content)) - end - - Main.emit_frame(Dict( - "type" => "display", - "id" => Main.current_rid, - "bundle" => Dict( - "application/x-omp-status" => Dict( - "op" => "append", - "path" => resolved, - "chars" => length(string(content)) - ) - ) - )) - return resolved -end - -function tree(path=".", positional_max_depth=3, positional_show_hidden=false; max_depth=positional_max_depth, show_hidden=positional_show_hidden) - base = string(path) - resolved = __omp_resolve_path(base) - lines = String[] - - function walk(dir, prefix, depth) - if depth > max_depth - return - end - entries = try - readdir(dir) - catch - String[] - end - if !show_hidden - entries = filter(e -> !startswith(e, '.'), entries) - end - sort!(entries, by = e -> (ispath(joinpath(dir, e)) && isdir(joinpath(dir, e)) ? 0 : 1, lowercase(e))) - - for (i, name) in enumerate(entries) - full = joinpath(dir, name) - is_last = i == length(entries) - is_dir = isdir(full) - push!(lines, "$(prefix)$(is_last ? "└── " : "├── ")$(name)$(is_dir ? "/" : "")") - if is_dir - walk(full, prefix * (is_last ? " " : "│ "), depth + 1) - end - end - end - - walk(resolved, "", 1) - out = join(lines, '\n') - - Main.emit_frame(Dict( - "type" => "display", - "id" => Main.current_rid, - "bundle" => Dict( - "application/x-omp-status" => Dict( - "op" => "tree", - "path" => resolved, - "lines" => length(lines) - ) - ) - )) - return out -end - -function __omp_lines_keepends(content::String) - parts = split(content, '\n'; keepempty=true) - if length(parts) == 1 && isempty(parts[1]) - return String[] - end - lines = String[] - for i in eachindex(parts) - if i < length(parts) - push!(lines, string(parts[i], "\n")) - elseif !isempty(parts[i]) - push!(lines, string(parts[i])) - end - end - return lines -end - -function __omp_diff_ops(a::Vector{String}, b::Vector{String}) - n = length(a) - m = length(b) - ops = Vector{Tuple{Symbol, Int, Int}}() - if n * m > 4_000_000 - for i in 1:n - push!(ops, (:delete, i, 1)) - end - for j in 1:m - push!(ops, (:insert, n + 1, j)) - end - return ops - end - - dp = [zeros(Int, m + 1) for _ in 1:(n + 1)] - for i in n:-1:1 - for j in m:-1:1 - dp[i][j] = a[i] == b[j] ? dp[i + 1][j + 1] + 1 : max(dp[i + 1][j], dp[i][j + 1]) - end - end - - i = 1 - j = 1 - while i <= n && j <= m - if a[i] == b[j] - push!(ops, (:equal, i, j)) - i += 1 - j += 1 - elseif dp[i + 1][j] >= dp[i][j + 1] - push!(ops, (:delete, i, j)) - i += 1 - else - push!(ops, (:insert, i, j)) - j += 1 - end - end - while i <= n - push!(ops, (:delete, i, j)) - i += 1 - end - while j <= m - push!(ops, (:insert, i, j)) - j += 1 - end - return ops -end - -function __omp_unified_diff(a::Vector{String}, b::Vector{String}, from_file::String, to_file::String, context::Int=3) - ops = __omp_diff_ops(a, b) - if !any(op -> op[1] != :equal, ops) - return "" - end - - entries = [Dict{Symbol, Any}(:tag => tag, :ai => ai, :bi => bi, :text => tag == :insert ? b[bi] : a[ai]) for (tag, ai, bi) in ops] - changed = [i for i in eachindex(entries) if entries[i][:tag] != :equal] - groups = Vector{Tuple{Int, Int}}() - start = nothing - prev = nothing - for idx in changed - if start === nothing - start = idx - prev = idx - elseif idx - prev <= (2 * context) + 1 - prev = idx - else - push!(groups, (start, prev)) - start = idx - prev = idx - end - end - if start !== nothing - push!(groups, (start, prev)) - end - - out = IOBuffer() - write(out, "--- $from_file\n") - write(out, "+++ $to_file\n") - for (group_start, group_end) in groups - lo = max(group_start - context, 1) - hi = min(group_end + context, length(entries)) - slice = entries[lo:hi] - a_start = nothing - a_count = 0 - b_start = nothing - b_count = 0 - for entry in slice - if entry[:tag] != :insert - if a_start === nothing - a_start = entry[:ai] - end - a_count += 1 - end - if entry[:tag] != :delete - if b_start === nothing - b_start = entry[:bi] - end - b_count += 1 - end - end - write(out, "@@ -$(a_start === nothing ? 1 : a_start),$a_count +$(b_start === nothing ? 1 : b_start),$b_count @@\n") - for entry in slice - prefix = entry[:tag] == :equal ? " " : (entry[:tag] == :delete ? "-" : "+") - text = string(entry[:text]) - if !endswith(text, "\n") - text *= "\n" - end - write(out, prefix * text) - end - end - return String(take!(out)) -end - -function Base.diff(a::AbstractString, b::AbstractString) - path_a = __omp_resolve_path(string(a)) - path_b = __omp_resolve_path(string(b)) - lines_a = __omp_lines_keepends(open(path_a, "r") do io - Base.read(io, String) - end) - lines_b = __omp_lines_keepends(open(path_b, "r") do io - Base.read(io, String) - end) - out = __omp_unified_diff(lines_a, lines_b, path_a, path_b) - __omp_emit_status("diff", Dict{String, Any}( - "file_a" => path_a, - "file_b" => path_b, - "identical" => isempty(out), - "preview" => first(out, min(500, length(out))) - )) - return out -end - function __omp_apply_query(data, query) if query === nothing || isempty(string(query)) return data diff --git a/packages/coding-agent/src/eval/js/shared/helpers.ts b/packages/coding-agent/src/eval/js/shared/helpers.ts index 03242aadd..0cbbb0fbd 100644 --- a/packages/coding-agent/src/eval/js/shared/helpers.ts +++ b/packages/coding-agent/src/eval/js/shared/helpers.ts @@ -1,19 +1,11 @@ -import * as fs from "node:fs"; import * as path from "node:path"; -import * as Diff from "diff"; import { ToolError } from "../../../tools/tool-errors"; import type { JsStatusEvent } from "./types"; export interface HelperOptions { - path?: string; - hidden?: boolean; - maxDepth?: number; limit?: number; offset?: number; - reverse?: boolean; - unique?: boolean; - count?: boolean; } /** @@ -35,18 +27,12 @@ export interface HelperContext { /** * The set of functions exposed to user code via `globalThis.__omp_helpers__`. The JS - * prelude reads from this bag and attaches short aliases (`read`, `write`, `tree`, ...) + * prelude reads from this bag and attaches short aliases (`read`, `write`, `env`, ...) * onto the global scope. */ export interface HelperBundle { read(rawPath: string, options?: HelperOptions): Promise; writeFile(rawPath: string, data: unknown): Promise; - append(rawPath: string, content: string): Promise; - sortText(text: string, options?: HelperOptions): string; - uniqText(text: string, options?: HelperOptions): string | Array<[number, string]>; - counter(items: string | string[], options?: HelperOptions): Array<[number, string]>; - diff(rawA: string, rawB: string): Promise; - tree(searchPath?: string, options?: HelperOptions): Promise; env(key?: string, value?: string): string | Record | undefined; } @@ -81,105 +67,6 @@ export function createHelpers(ctx: HelperContext): HelperBundle { ctx.emitStatus({ op: "write", path: filePath, bytes: getDataSize(data) }); return filePath; }, - append: async (rawPath, content) => { - const target = resolveHelperPath(ctx, rawPath, "write"); - // O(1) append; read-all+rewrite both raced concurrent writers and went - // quadratic when called in a loop. Bun.write creates parent dirs, so - // keep that behavior for the append path too. - await fs.promises.mkdir(path.dirname(target), { recursive: true }); - await fs.promises.appendFile(target, content, "utf-8"); - ctx.emitStatus({ - op: "append", - path: target, - chars: content.length, - bytes: utf8Encoder.encode(content).byteLength, - }); - return target; - }, - sortText: (text, options = {}) => { - const lines = String(text).split(/\r?\n/); - const deduped = options.unique ? Array.from(new Set(lines)) : lines; - const sorted = deduped.sort((a, b) => a.localeCompare(b)); - if (options.reverse) sorted.reverse(); - const result = sorted.join("\n"); - ctx.emitStatus({ - op: "sort", - lines: sorted.length, - reverse: options.reverse === true, - unique: options.unique === true, - }); - return result; - }, - uniqText: (text, options = {}) => { - const lines = String(text) - .split(/\r?\n/) - .filter(line => line.length > 0); - const groups: Array<[number, string]> = []; - for (const line of lines) { - const last = groups.at(-1); - if (last && last[1] === line) { - last[0] += 1; - continue; - } - groups.push([1, line]); - } - ctx.emitStatus({ op: "uniq", groups: groups.length, count_mode: options.count === true }); - if (options.count) return groups; - return groups.map(([, line]) => line).join("\n"); - }, - counter: (items, options = {}) => { - const values = Array.isArray(items) ? items : String(items).split(/\r?\n/).filter(Boolean); - const counts = new Map(); - for (const item of values) counts.set(item, (counts.get(item) ?? 0) + 1); - const entries = Array.from(counts.entries()) - .map(([item, count]) => [count, item] as [number, string]) - .sort((a, b) => (options.reverse === false ? a[0] - b[0] : b[0] - a[0]) || a[1].localeCompare(b[1])); - const limited = entries.slice(0, options.limit ?? entries.length); - ctx.emitStatus({ op: "counter", unique: counts.size, total: values.length, top: limited.slice(0, 10) }); - return limited; - }, - diff: async (rawA, rawB) => { - const fileA = resolvePath(ctx, rawA); - const fileB = resolvePath(ctx, rawB); - const [a, b] = await Promise.all([Bun.file(fileA).text(), Bun.file(fileB).text()]); - const result = Diff.createTwoFilesPatch(fileA, fileB, a, b, "", "", { context: 3 }); - ctx.emitStatus({ - op: "diff", - file_a: fileA, - file_b: fileB, - identical: a === b, - preview: result.slice(0, 500), - }); - return result; - }, - tree: async (searchPath = ".", options = {}) => { - const root = resolvePath(ctx, searchPath); - const maxDepth = options.maxDepth ?? 3; - const showHidden = options.hidden ?? false; - const lines: string[] = [`${root}/`]; - let entryCount = 0; - const walk = async (dir: string, prefix: string, depth: number): Promise => { - if (depth > maxDepth) return; - const entries = (await fs.promises.readdir(dir, { withFileTypes: true })) - .filter(entry => showHidden || !entry.name.startsWith(".")) - .sort((a, b) => a.name.localeCompare(b.name)); - for (let index = 0; index < entries.length; index++) { - const entry = entries[index]; - const isLast = index === entries.length - 1; - const connector = isLast ? "└── " : "├── "; - const suffix = entry.isDirectory() ? "/" : ""; - lines.push(`${prefix}${connector}${entry.name}${suffix}`); - entryCount += 1; - if (entry.isDirectory()) { - await walk(path.join(dir, entry.name), `${prefix}${isLast ? " " : "│ "}`, depth + 1); - } - } - }; - await walk(root, "", 1); - const result = lines.join("\n"); - ctx.emitStatus({ op: "tree", path: root, entries: entryCount, preview: result.slice(0, 1000) }); - return result; - }, env: (key, value) => { if (!key) { const merged = Object.fromEntries(Object.entries(getMergedEnv(ctx)).sort(([a], [b]) => a.localeCompare(b))); diff --git a/packages/coding-agent/src/eval/js/shared/prelude.txt b/packages/coding-agent/src/eval/js/shared/prelude.txt index abf230123..60fd12123 100644 --- a/packages/coding-agent/src/eval/js/shared/prelude.txt +++ b/packages/coding-agent/src/eval/js/shared/prelude.txt @@ -63,23 +63,6 @@ if (!globalThis.__omp_js_prelude_loaded__) { return callHelper("read", path, options); }; const write = async (path, data) => callHelper("writeFile", path, data); - const append = (path, content) => callHelper("append", path, content); - const sort = (text, opts, ...rest) => - callHelper("sortText", text, optionsArg("sort", opts, rest, ["reverse", "unique"], "{ reverse, unique }")); - const uniq = (text, opts, ...rest) => callHelper("uniqText", text, optionsArg("uniq", opts, rest, ["count"], "{ count }")); - const counter = (items, opts, ...rest) => - callHelper("counter", items, optionsArg("counter", opts, rest, ["limit", "reverse"], "{ limit, reverse }")); - const diff = (a, b) => callHelper("diff", a, b); - const tree = (path = ".", opts, ...rest) => { - if (isPlainObject(path) && opts === undefined && rest.length === 0) { - return callHelper("tree", ".", path); - } - return callHelper( - "tree", - isNil(path) ? "." : path, - optionsArg("tree", opts, rest, ["maxDepth", "showHidden"], "{ maxDepth, showHidden }"), - ); - }; const env = (key, value) => callHelper("env", key, value); const tool = new Proxy( @@ -306,11 +289,5 @@ if (!globalThis.__omp_js_prelude_loaded__) { globalThis.__pool = __pool; globalThis.read = read; globalThis.write = write; - globalThis.append = append; - globalThis.sort = sort; - globalThis.uniq = uniq; - globalThis.counter = counter; - globalThis.diff = diff; - globalThis.tree = tree; globalThis.env = env; } diff --git a/packages/coding-agent/src/eval/js/shared/runtime.ts b/packages/coding-agent/src/eval/js/shared/runtime.ts index 47b590bcd..1d88465be 100644 --- a/packages/coding-agent/src/eval/js/shared/runtime.ts +++ b/packages/coding-agent/src/eval/js/shared/runtime.ts @@ -72,12 +72,6 @@ const PRELUDE_GLOBAL_KEYS = [ "__pool", "read", "write", - "append", - "sort", - "uniq", - "counter", - "diff", - "tree", "env", ]; diff --git a/packages/coding-agent/src/eval/js/shared/types.ts b/packages/coding-agent/src/eval/js/shared/types.ts index 985154a8b..2f540dbb9 100644 --- a/packages/coding-agent/src/eval/js/shared/types.ts +++ b/packages/coding-agent/src/eval/js/shared/types.ts @@ -1,5 +1,5 @@ /** - * Structured status payload emitted by helpers (`read`, `write`, `tree`, etc.) and the + * Structured status payload emitted by helpers (`read`, `write`, `env`, etc.) and the * tool-call bridge. Surfaces to the model as part of `displays` so it has machine-readable * context about what side effects happened. */ diff --git a/packages/coding-agent/src/eval/js/worker-protocol.ts b/packages/coding-agent/src/eval/js/worker-protocol.ts index 118aae9ff..65586f312 100644 --- a/packages/coding-agent/src/eval/js/worker-protocol.ts +++ b/packages/coding-agent/src/eval/js/worker-protocol.ts @@ -7,7 +7,7 @@ export interface SessionSnapshot { sessionId: string; /** * On-disk roots the helpers substitute for internal-URL schemes - * (e.g. `{ local: "/…/artifacts/local" }`). Lets `read`/`write`/`append` + * (e.g. `{ local: "/…/artifacts/local" }`). Lets `read`/`write` * accept `local://…` paths instead of writing a literal `local:/` directory. */ localRoots?: Record; diff --git a/packages/coding-agent/src/eval/py/executor.ts b/packages/coding-agent/src/eval/py/executor.ts index 540feceaa..fa44c074b 100644 --- a/packages/coding-agent/src/eval/py/executor.ts +++ b/packages/coding-agent/src/eval/py/executor.ts @@ -71,7 +71,7 @@ export interface PythonExecutorOptions { artifactPath?: string; artifactId?: string; /** - * On-disk roots the prelude helpers (`read`/`write`/`append`) substitute for + * On-disk roots the prelude helpers (`read`/`write`) substitute for * internal-URL schemes (e.g. `{ local: "/…/artifacts/local" }`). Exported to * the kernel as `PI_EVAL_LOCAL_ROOTS` (JSON) so `write("local://x")` lands * where `read local://x` resolves instead of a literal `local:/` directory. diff --git a/packages/coding-agent/src/eval/py/prelude.py b/packages/coding-agent/src/eval/py/prelude.py index 6cb3bfb6d..663b2a07a 100644 --- a/packages/coding-agent/src/eval/py/prelude.py +++ b/packages/coding-agent/src/eval/py/prelude.py @@ -115,103 +115,6 @@ if "__omp_prelude_loaded__" not in globals(): _emit_status("write", path=str(p), chars=len(content)) return p - def append(path: str | Path, content: str) -> Path: - """Append to file.""" - p = _resolve_omp_path(path) - p.parent.mkdir(parents=True, exist_ok=True) - with p.open("a", encoding="utf-8") as f: - f.write(content) - _emit_status("append", path=str(p), chars=len(content)) - return p - - def sort(text: str, *, reverse: bool = False, unique: bool = False) -> str: - """Sort lines of text.""" - lines = text.splitlines() - if unique: - lines = list(dict.fromkeys(lines)) - lines = sorted(lines, reverse=reverse) - out = "\n".join(lines) - _emit_status("sort", lines=len(lines), unique=unique, reverse=reverse) - return out - - def uniq(text: str, *, count: bool = False) -> str | list[tuple[int, str]]: - """Remove duplicate adjacent lines (like uniq).""" - lines = text.splitlines() - if not lines: - _emit_status("uniq", groups=0) - return [] if count else "" - groups: list[tuple[int, str]] = [] - current = lines[0] - current_count = 1 - for line in lines[1:]: - if line == current: - current_count += 1 - continue - groups.append((current_count, current)) - current = line - current_count = 1 - groups.append((current_count, current)) - _emit_status("uniq", groups=len(groups), count_mode=count) - if count: - return groups - return "\n".join(line for _, line in groups) - - def counter( - items: str | list, - *, - limit: int | None = None, - reverse: bool = True, - ) -> list[tuple[int, str]]: - """Count occurrences and sort by frequency. Like sort | uniq -c | sort -rn. - - items: text (splits into lines) or list of strings - reverse: True for descending (most common first), False for ascending - Returns: [(count, item), ...] sorted by count - """ - from collections import Counter - if isinstance(items, str): - items = items.splitlines() - counts = Counter(items) - sorted_items = sorted(counts.items(), key=lambda x: (x[1], x[0]), reverse=reverse) - if limit is not None: - sorted_items = sorted_items[:limit] - result = [(count, item) for item, count in sorted_items] - _emit_status("counter", unique=len(counts), total=sum(counts.values()), top=result[:10]) - return result - def tree(path: str | Path = ".", *, max_depth: int = 3, show_hidden: bool = False) -> str: - """Return directory tree.""" - base = Path(path) - lines = [] - def walk(p: Path, prefix: str, depth: int): - if depth > max_depth: - return - items = sorted(p.iterdir(), key=lambda x: (not x.is_dir(), x.name.lower())) - items = [i for i in items if show_hidden or not i.name.startswith(".")] - for i, item in enumerate(items): - is_last = i == len(items) - 1 - connector = "└── " if is_last else "├── " - suffix = "/" if item.is_dir() else "" - lines.append(f"{prefix}{connector}{item.name}{suffix}") - if item.is_dir(): - ext = " " if is_last else "│ " - walk(item, prefix + ext, depth + 1) - lines.append(str(base) + "/") - walk(base, "", 1) - out = "\n".join(lines) - _emit_status("tree", path=str(base), entries=len(lines) - 1, preview=out[:1000]) - return out - - def diff(a: str | Path, b: str | Path) -> str: - """Compare two files, return unified diff.""" - import difflib - path_a, path_b = Path(a), Path(b) - lines_a = path_a.read_text(encoding="utf-8").splitlines(keepends=True) - lines_b = path_b.read_text(encoding="utf-8").splitlines(keepends=True) - result = difflib.unified_diff(lines_a, lines_b, fromfile=str(path_a), tofile=str(path_b)) - out = "".join(result) - _emit_status("diff", file_a=str(path_a), file_b=str(path_b), identical=not out, preview=out[:500]) - return out - def output( *ids: str, format: str = "raw", diff --git a/packages/coding-agent/src/eval/rb/prelude.rb b/packages/coding-agent/src/eval/rb/prelude.rb index 4615c185b..ee6df2de5 100644 --- a/packages/coding-agent/src/eval/rb/prelude.rb +++ b/packages/coding-agent/src/eval/rb/prelude.rb @@ -2,7 +2,7 @@ # OMP Ruby prelude helpers (loaded once into the runner's TOPLEVEL_BINDING). # # Mirrors eval/py/prelude.py: defines the cross-runtime helper surface -# (display/read/write/append/tree/diff/env/output, the `tool` bridge proxy, +# (display/read/write/env/output, the `tool` bridge proxy, # completion/agent/parallel/pipeline/log/phase/budget). Host-side helpers reach # the coding-agent over the same loopback HTTP tool bridge the Python prelude # uses (PI_TOOL_BRIDGE_URL/TOKEN/SESSION). Path helpers honor PI_EVAL_LOCAL_ROOTS @@ -112,191 +112,6 @@ unless defined?($__omp_prelude_loaded) && $__omp_prelude_loaded resolved.to_s end - def append(path, content) - resolved = __omp_resolve_path(path) - require "fileutils" - FileUtils.mkdir_p(File.dirname(resolved.to_s)) - File.open(resolved.to_s, "a") { |f| f.write(content.to_s) } - __omp_emit_status("append", "path" => resolved.to_s, "chars" => content.to_s.length) - resolved.to_s - end - - def tree(path = ".", max_depth: 3, show_hidden: false) - base = path.to_s - lines = [] - walk = lambda do |dir, prefix, depth| - return if depth > max_depth - entries = (Dir.children(dir) rescue []) - entries = entries.reject { |e| e.start_with?(".") } unless show_hidden - entries = entries.sort_by { |e| [File.directory?(File.join(dir, e)) ? 0 : 1, e.downcase] } - entries.each_with_index do |name, i| - full = File.join(dir, name) - is_last = i == entries.length - 1 - is_dir = File.directory?(full) - lines << "#{prefix}#{is_last ? "└── " : "├── "}#{name}#{is_dir ? "/" : ""}" - walk.call(full, prefix + (is_last ? " " : "│ "), depth + 1) if is_dir - end - end - lines << "#{base}/" - walk.call(base, "", 1) - out = lines.join("\n") - __omp_emit_status("tree", "path" => base, "entries" => lines.length - 1, "preview" => __omp_scrub(out[0, 1000].to_s)) - out - end - - def diff(a, b) - pa = a.to_s - pb = b.to_s - lines_a = File.read(pa, encoding: Encoding::UTF_8).lines - lines_b = File.read(pb, encoding: Encoding::UTF_8).lines - out = __omp_unified_diff(lines_a, lines_b, pa, pb) - __omp_emit_status("diff", "file_a" => pa, "file_b" => pb, "identical" => out.empty?, "preview" => __omp_scrub(out[0, 500].to_s)) - out - end - - # LCS-based op list ([:equal/:delete/:insert, aIndex, bIndex]) per consumed line. - def __omp_diff_ops(a, b) - n = a.length - m = b.length - if n * m > 4_000_000 - # Too large for the DP table — fall back to a coarse all-delete/all-insert. - ops = [] - n.times { |i| ops << [:delete, i, 0] } - m.times { |j| ops << [:insert, n, j] } - return ops - end - dp = Array.new(n + 1) { Array.new(m + 1, 0) } - (n - 1).downto(0) do |i| - row = dp[i] - nrow = dp[i + 1] - (m - 1).downto(0) do |j| - row[j] = a[i] == b[j] ? nrow[j + 1] + 1 : (nrow[j] >= row[j + 1] ? nrow[j] : row[j + 1]) - end - end - ops = [] - i = 0 - j = 0 - while i < n && j < m - if a[i] == b[j] - ops << [:equal, i, j]; i += 1; j += 1 - elsif dp[i + 1][j] >= dp[i][j + 1] - ops << [:delete, i, j]; i += 1 - else - ops << [:insert, i, j]; j += 1 - end - end - ops << [:delete, i, j].tap { i += 1 } while i < n - ops << [:insert, i, j].tap { j += 1 } while j < m - ops - end - - def __omp_unified_diff(a, b, from_file, to_file, context = 3) - ops = __omp_diff_ops(a, b) - return "" unless ops.any? { |tag, _, _| tag != :equal } - - entries = ops.map do |tag, ai, bi| - { tag: tag, ai: ai, bi: bi, text: (tag == :insert ? b[bi] : a[ai]) } - end - changed = entries.each_index.select { |k| entries[k][:tag] != :equal } - - groups = [] - start = nil - prev = nil - changed.each do |k| - if start.nil? - start = k - prev = k - elsif k - prev <= (2 * context) + 1 - prev = k - else - groups << [start, prev] - start = k - prev = k - end - end - groups << [start, prev] unless start.nil? - - out = +"" - out << "--- #{from_file}\n" - out << "+++ #{to_file}\n" - groups.each do |gs, ge| - lo = [gs - context, 0].max - hi = [ge + context, entries.length - 1].min - slice = entries[lo..hi] - a_start = nil - a_count = 0 - b_start = nil - b_count = 0 - slice.each do |e| - if e[:tag] != :insert - a_start ||= e[:ai] - a_count += 1 - end - if e[:tag] != :delete - b_start ||= e[:bi] - b_count += 1 - end - end - out << "@@ -#{(a_start || 0) + 1},#{a_count} +#{(b_start || 0) + 1},#{b_count} @@\n" - slice.each do |e| - prefix = e[:tag] == :equal ? " " : (e[:tag] == :delete ? "-" : "+") - text = e[:text].to_s - text = "#{text}\n" unless text.end_with?("\n") - out << "#{prefix}#{text}" - end - end - out - end - - # ------------------------------------------------------------------------- - # Text helpers (sort / uniq / counter) - # ------------------------------------------------------------------------- - - def sort(text, reverse: false, unique: false) - lines = text.to_s.lines.map(&:chomp) - lines = lines.uniq if unique - lines = lines.sort - lines = lines.reverse if reverse - out = lines.join("\n") - __omp_emit_status("sort", "lines" => lines.length, "unique" => unique, "reverse" => reverse) - out - end - - def uniq(text, count: false) - lines = text.to_s.lines.map(&:chomp) - if lines.empty? - __omp_emit_status("uniq", "groups" => 0) - return count ? [] : "" - end - groups = [] - current = lines[0] - run = 1 - lines[1..].each do |line| - if line == current - run += 1 - else - groups << [run, current] - current = line - run = 1 - end - end - groups << [run, current] - __omp_emit_status("uniq", "groups" => groups.length, "count_mode" => count) - count ? groups : groups.map { |_, l| l }.join("\n") - end - - def counter(items, limit: nil, reverse: true) - arr = items.is_a?(String) ? items.lines.map(&:chomp) : items.to_a - counts = Hash.new(0) - arr.each { |i| counts[i] += 1 } - sorted = counts.sort_by { |item, c| [c, item] } - sorted = sorted.reverse if reverse - sorted = sorted.first(limit) if limit - result = sorted.map { |item, c| [c, item] } - __omp_emit_status("counter", "unique" => counts.size, "total" => arr.length, "top" => result.first(10)) - result - end - # ------------------------------------------------------------------------- # Task/agent output reader # ------------------------------------------------------------------------- diff --git a/packages/coding-agent/src/internal-urls/local-protocol.ts b/packages/coding-agent/src/internal-urls/local-protocol.ts index 310ed6aaa..0fe00cf0f 100644 --- a/packages/coding-agent/src/internal-urls/local-protocol.ts +++ b/packages/coding-agent/src/internal-urls/local-protocol.ts @@ -157,7 +157,7 @@ export function resolveLocalUrlToPath( } /** - * On-disk roots the eval helpers (`read`/`write`/`append`) substitute for + * On-disk roots the eval helpers (`read`/`write`) substitute for * internal-URL schemes so e.g. `write("local://x.md")` lands where a later * `read local://x.md` resolves — instead of a literal `local:/` directory under * the cwd (a stdlib `pathlib.Path`/`path.resolve` collapses `local://` to diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 60c428959..fbe4204f7 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -15,8 +15,8 @@ Fields: {{#if py}}Live event loop: use top-level `await` directly; `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}} {{#if js}}JS runs under **Bun**: Bun globals/APIs are available (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`); top-level `await`/`return` work directly.{{/if}} -{{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `tree(".", max_depth: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}} -{{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `tree(max_depth=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}} +{{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `output("id", limit: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}} +{{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `output("id", limit=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}} On error, fix and re-run only the failing step — prior calls' state survives. @@ -31,12 +31,6 @@ read(path, offset?=1, limit?=None) → str File as text; offset/limit 1-indexed lines. Accepts `local://…`. write(path, content) → str Write file (creates parents) → resolved path. `local://…` persists across turns/subagents. -append(path, content) → str - Append → resolved path. Accepts `local://…`. -tree(path?=".", max_depth?=3, show_hidden?=False) → str - Directory tree. -diff(a, b) → str - Unified diff of two files. env(key?=None, value?=None) → str | None | dict No args → full env dict; one → value of `key`; two → set `key=value`, return value. output(*ids, format?="raw", query?=None, offset?=None, limit?=None) → str | dict | list[dict] diff --git a/packages/coding-agent/src/tools/browser/tab-worker.ts b/packages/coding-agent/src/tools/browser/tab-worker.ts index 3cfbe1b63..3d11b87b4 100644 --- a/packages/coding-agent/src/tools/browser/tab-worker.ts +++ b/packages/coding-agent/src/tools/browser/tab-worker.ts @@ -800,7 +800,7 @@ export class WorkerCore { displays.push({ type: "text", text: safeJsonStringify(output.data) }); return; } - // status — surface as compact JSON so helper side effects (read/write/tree) appear in + // status — surface as compact JSON so helper side effects (read/write/env) appear in // the cell result alongside explicit display() output. displays.push({ type: "text", text: safeJsonStringify(output.event) }); } diff --git a/packages/coding-agent/src/tools/eval-render.ts b/packages/coding-agent/src/tools/eval-render.ts index 58f71ba94..68c8e205a 100644 --- a/packages/coding-agent/src/tools/eval-render.ts +++ b/packages/coding-agent/src/tools/eval-render.ts @@ -241,14 +241,12 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string { const opIcons: Record = { read: "icon.file", write: "icon.file", - append: "icon.file", cat: "icon.file", touch: "icon.file", ls: "icon.folder", cd: "icon.folder", pwd: "icon.folder", mkdir: "icon.folder", - tree: "icon.folder", git_status: "icon.git", git_diff: "icon.git", git_log: "icon.git", @@ -280,7 +278,6 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string { if (data.path) parts.push(`from ${shortenPath(String(data.path))}`); break; case "write": - case "append": parts.push(`${data.chars ?? data.bytes ?? 0} chars`); if (data.path) parts.push(`to ${shortenPath(String(data.path))}`); break; @@ -319,13 +316,6 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string { parts.push(`${data.lines} line${(data.lines as number) !== 1 ? "s" : ""}`); if (data.staged) parts.push("(staged)"); break; - case "diff": - if (data.identical) { - parts.push("files identical"); - } else { - parts.push("files differ"); - } - break; case "batch": parts.push(`${data.files} file${(data.files as number) !== 1 ? "s" : ""} processed`); break; @@ -415,8 +405,6 @@ function formatStatusEventExpanded(event: EvalStatusEvent, theme: Theme): string case "cat": case "head": case "tail": - case "tree": - case "diff": case "git_diff": case "sh": if (data.preview) addPreview(String(data.preview)); diff --git a/packages/coding-agent/test/core/ruby-runner.integration.test.ts b/packages/coding-agent/test/core/ruby-runner.integration.test.ts index c7ccd1fa4..c7ad56a76 100644 --- a/packages/coding-agent/test/core/ruby-runner.integration.test.ts +++ b/packages/coding-agent/test/core/ruby-runner.integration.test.ts @@ -144,13 +144,10 @@ describe.skipIf(!SHOULD_RUN)("ruby runner subprocess", () => { } }); - it("exposes prelude file + text helpers", async () => { + it("exposes prelude file helpers", async () => { using tempDir = TempDir.createSync("@ruby-runner-prelude-"); const kernel = await RubyKernel.start({ cwd: tempDir.path() }); try { - const sorted = await executeRubyWithKernel(kernel, 'sort("b\\na\\nb", unique: true)', {}); - expect(sorted.output).toContain("a\nb"); - const written = await executeRubyWithKernel(kernel, 'write("note.txt", "hello"); read("note.txt")', {}); expect(written.output).toContain("hello"); } finally { From 6ff37e346a53623f9bd9727d3e39aa9d3038369e Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 01:46:36 +0200 Subject: [PATCH 27/43] feat(coding-agent): extended --thinking CLI flag options - Added `off` and `auto` as valid inputs for the `--thinking` CLI flag. - Centralized thinking level definitions in `CLI_THINKING_LEVELS` to keep flag options, shell completions, and validation in sync. - Configured CLI parsing to reject `inherit` as an explicit input to prevent unintended configuration suppression. --- packages/coding-agent/src/cli/args.ts | 9 ++++----- packages/coding-agent/src/cli/flag-tables.ts | 6 +++--- packages/coding-agent/src/commands/launch.ts | 6 +++--- packages/coding-agent/src/thinking.ts | 20 +++++++++++++++++++ .../test/auto-thinking-classifier.test.ts | 9 +++++++++ .../test/cli-hide-thinking-flag.test.ts | 20 +++++++++++++++++++ .../coding-agent/test/cli/completions.test.ts | 2 +- 7 files changed, 60 insertions(+), 12 deletions(-) diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 488826561..364bb94ac 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -1,10 +1,9 @@ /** * CLI argument parsing and help display */ -import { type Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-catalog/effort"; import { APP_NAME, CONFIG_DIR_NAME, logger } from "@oh-my-pi/pi-utils"; import chalk from "chalk"; -import { parseEffort } from "../thinking"; +import { CLI_THINKING_LEVELS, type ConfiguredThinkingLevel, parseCliThinkingLevel } from "../thinking"; import { BUILTIN_TOOL_NAMES } from "../tools/builtin-names"; import { OPTIONAL_FLAGS, @@ -32,7 +31,7 @@ export interface Args { apiKey?: string; systemPrompt?: string; appendSystemPrompt?: string; - thinking?: Effort; + thinking?: ConfiguredThinkingLevel; hideThinking?: boolean; advisor?: boolean; continue?: boolean; @@ -89,9 +88,9 @@ export interface Args { */ const PARSE_DEPS: ParseDeps = { logger, - parseEffort, + parseThinking: parseCliThinkingLevel, builtinToolNames: BUILTIN_TOOL_NAMES, - thinkingEfforts: THINKING_EFFORTS, + thinkingEfforts: CLI_THINKING_LEVELS, }; export function parseArgs(inputArgs: string[], extensionFlags?: Map): Args { diff --git a/packages/coding-agent/src/cli/flag-tables.ts b/packages/coding-agent/src/cli/flag-tables.ts index 63dcb13ec..fd315145e 100644 --- a/packages/coding-agent/src/cli/flag-tables.ts +++ b/packages/coding-agent/src/cli/flag-tables.ts @@ -30,7 +30,7 @@ * real implementations at the dispatch site. */ -import type { Effort } from "@oh-my-pi/pi-ai"; +import type { ConfiguredThinkingLevel } from "../thinking"; import type { Args } from "./args"; /** @@ -44,7 +44,7 @@ import type { Args } from "./args"; */ export interface ParseDeps { logger: { warn: (message: string, meta?: Record) => void }; - parseEffort: (value: string | null | undefined) => Effort | undefined; + parseThinking: (value: string | null | undefined) => ConfiguredThinkingLevel | undefined; builtinToolNames: readonly string[]; thinkingEfforts: readonly string[]; } @@ -165,7 +165,7 @@ export const STRING_SETTERS: Record = { result.tools = valid; }, "--thinking": (result, value, deps) => { - const thinking = deps.parseEffort(value); + const thinking = deps.parseThinking(value); if (thinking !== undefined) { result.thinking = thinking; } else { diff --git a/packages/coding-agent/src/commands/launch.ts b/packages/coding-agent/src/commands/launch.ts index d0a623bf1..5c559d6ea 100644 --- a/packages/coding-agent/src/commands/launch.ts +++ b/packages/coding-agent/src/commands/launch.ts @@ -2,12 +2,12 @@ * Root command for the coding agent CLI. */ -import { THINKING_EFFORTS } from "@oh-my-pi/pi-catalog/effort"; import { APP_NAME } from "@oh-my-pi/pi-utils"; import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli"; import { parseArgs } from "../cli/args"; import { runRootCommand } from "../main"; import { prepareAcpTerminalAuthArgs } from "../modes/acp/terminal-auth"; +import { CLI_THINKING_LEVELS } from "../thinking"; export default class Index extends Command { static description = "AI coding assistant"; @@ -100,8 +100,8 @@ export default class Index extends Command { description: "Comma-separated list of tools to enable (default: all)", }), thinking: Flags.string({ - description: `Set thinking level: ${THINKING_EFFORTS.join(", ")}`, - options: [...THINKING_EFFORTS], + description: `Set thinking level: ${CLI_THINKING_LEVELS.join(", ")}`, + options: [...CLI_THINKING_LEVELS], }), "hide-thinking": Flags.boolean({ description: "Hide thinking blocks in TUI output (display only, does not disable model thinking)", diff --git a/packages/coding-agent/src/thinking.ts b/packages/coding-agent/src/thinking.ts index 261e61be0..2e17f7fa5 100644 --- a/packages/coding-agent/src/thinking.ts +++ b/packages/coding-agent/src/thinking.ts @@ -153,6 +153,26 @@ export function getConfiguredThinkingLevelMetadata(level: ConfiguredThinkingLeve return level === AUTO_THINKING ? AUTO_THINKING_METADATA : getThinkingLevelMetadata(level); } +/** + * Thinking selectors accepted by the `--thinking` CLI flag, in display order: + * `off`, every concrete effort (`minimal`..`xhigh`), then `auto`. Single source + * for the flag's `options` list, shell completions, and the "invalid level" + * warning so all three stay in sync. + */ +export const CLI_THINKING_LEVELS: readonly string[] = [ThinkingLevel.Off, ...THINKING_EFFORTS, AUTO_THINKING]; + +/** + * Parses a `--thinking` CLI value. Accepts every {@link parseConfiguredThinkingLevel} + * selector (`off`, `auto`, `minimal`..`xhigh`, plus the `max` alias) but rejects + * `inherit`: an explicit `inherit` on the command line would suppress the + * settings/scoped-model fallback during startup resolution only to resolve back + * to the provider default, which is never what the user means. + */ +export function parseCliThinkingLevel(value: string | null | undefined): ConfiguredThinkingLevel | undefined { + const level = parseConfiguredThinkingLevel(value); + return level === ThinkingLevel.Inherit ? undefined : level; +} + /** * Resolves an auto-classified effort against the active model's supported * range. Unlike {@link clampThinkingLevelForModel}, `auto` never resolves below diff --git a/packages/coding-agent/test/auto-thinking-classifier.test.ts b/packages/coding-agent/test/auto-thinking-classifier.test.ts index 1e8141eba..a51e70d22 100644 --- a/packages/coding-agent/test/auto-thinking-classifier.test.ts +++ b/packages/coding-agent/test/auto-thinking-classifier.test.ts @@ -14,6 +14,7 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { AUTO_THINKING, clampAutoThinkingEffort, + parseCliThinkingLevel, parseConfiguredThinkingLevel, parseEffort, parseThinkingLevel, @@ -65,6 +66,14 @@ describe("auto thinking classifier helpers", () => { expect(parseThinkingLevel(ThinkingLevel.Off)).toBe(ThinkingLevel.Off); }); + it("parses CLI --thinking selectors while rejecting inherit", () => { + expect(parseCliThinkingLevel(ThinkingLevel.Off)).toBe(ThinkingLevel.Off); + expect(parseCliThinkingLevel(AUTO_THINKING)).toBe(AUTO_THINKING); + expect(parseCliThinkingLevel("max")).toBe(ThinkingLevel.XHigh); + expect(parseCliThinkingLevel(ThinkingLevel.Inherit)).toBeUndefined(); + expect(parseCliThinkingLevel("bogus")).toBeUndefined(); + }); + it("maps online 4-way classifier labels to effort levels", () => { expect(parseDifficultyLevel("x-high")).toBe(Effort.XHigh); expect(parseDifficultyLevel("The answer is HIGH.")).toBe(Effort.High); diff --git a/packages/coding-agent/test/cli-hide-thinking-flag.test.ts b/packages/coding-agent/test/cli-hide-thinking-flag.test.ts index a58634a1c..40c4d548b 100644 --- a/packages/coding-agent/test/cli-hide-thinking-flag.test.ts +++ b/packages/coding-agent/test/cli-hide-thinking-flag.test.ts @@ -1,6 +1,8 @@ import { describe, expect, it } from "bun:test"; +import { ThinkingLevel } from "@oh-my-pi/pi-agent-core"; import { Effort } from "@oh-my-pi/pi-ai"; import { parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args"; +import { AUTO_THINKING } from "@oh-my-pi/pi-coding-agent/thinking"; describe("parseArgs — --hide-thinking flag", () => { it("parses --hide-thinking as a boolean flag", () => { @@ -43,3 +45,21 @@ describe("parseArgs — --hide-thinking flag", () => { expect(result.messages).toEqual([]); }); }); + +describe("parseArgs — --thinking flag", () => { + it("accepts off so reasoning can be disabled from the CLI", () => { + expect(parseArgs(["--thinking", "off"]).thinking).toBe(ThinkingLevel.Off); + expect(parseArgs(["--thinking=off"]).thinking).toBe(ThinkingLevel.Off); + }); + + it("accepts auto, concrete efforts, and the max alias", () => { + expect(parseArgs(["--thinking", "auto"]).thinking).toBe(AUTO_THINKING); + expect(parseArgs(["--thinking", "medium"]).thinking).toBe(Effort.Medium); + expect(parseArgs(["--thinking", "max"]).thinking).toBe(ThinkingLevel.XHigh); + }); + + it("ignores invalid levels and the internal inherit selector", () => { + expect(parseArgs(["--thinking", "bogus"]).thinking).toBeUndefined(); + expect(parseArgs(["--thinking", "inherit"]).thinking).toBeUndefined(); + }); +}); diff --git a/packages/coding-agent/test/cli/completions.test.ts b/packages/coding-agent/test/cli/completions.test.ts index 2dcf3a17c..4da7f19ef 100644 --- a/packages/coding-agent/test/cli/completions.test.ts +++ b/packages/coding-agent/test/cli/completions.test.ts @@ -211,7 +211,7 @@ describe("omp completions (integration / drift)", () => { } expect(stdout).toContain("{-r,--resume}"); // Real enum option sets flow through unchanged. - expect(stdout).toContain(":value:(minimal low medium high xhigh)"); + expect(stdout).toContain(":value:(off minimal low medium high xhigh auto)"); expect(stdout).toContain(":value:(always-ask write yolo)"); // Real subcommands present; dynamic callbacks wired. expect(stdout).toContain("_omp_cmd_commit"); From c4e23fed15fe25d893f587e82f2a6c388e2728ce Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 01:46:45 +0200 Subject: [PATCH 28/43] fix(coding-agent/tools): enabled live stdout streaming for running cells - Update the eval tool to stream stdout chunks directly into the active cell's output buffer while the process is still running. - Prevent long-running cells from appearing empty in the UI by surfacing incremental output before the backend resolves. - Add regression tests to ensure streamed output is captured mid-execution and reconciled with final results. --- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/src/tools/eval.ts | 13 +++ .../test/tools/eval-streaming-output.test.ts | 87 +++++++++++++++++++ 3 files changed, 102 insertions(+) create mode 100644 packages/coding-agent/test/tools/eval-streaming-output.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index af44d0ce7..b2f80d4b1 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -28,6 +28,8 @@ ### Fixed +- Fixed the `eval` tool card not streaming a still-running cell's stdout: a long-running cell (e.g. a `time.sleep()` monitor loop) showed nothing until it returned or was interrupted, then dumped everything at once. The renderer draws cell output from `details.cells[i].output`, which was only populated after `backend.execute()` resolved — live stdout streamed into the transient result `content` tail (and `renderContext.output`), which the per-cell render branch ignores. Streamed chunks now append to the active cell's `output` (a dedicated per-cell tail buffer, capped like the aggregate) as they arrive, so the card shows progress live; on completion the authoritative full output overwrites the live tail. `log()`/`phase()`/`display()` and status ops were unaffected because they already stream via the status channel. + - Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path. - Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default. - Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index 45a98e873..b4b33c2cc 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -458,6 +458,13 @@ export class EvalTool implements AgentTool { status: "pending", })); const cellOutputs: string[] = []; + // The cell currently inside backend.execute(). Streamed stdout is + // appended to its rendered `output` live so a long-running cell (e.g. a + // sleep loop) shows progress instead of nothing until it returns. A + // dedicated per-cell tail buffer keeps attribution correct and avoids + // double-counting against the aggregate `tailBuffer`; on completion the + // authoritative `cellResult.output` (below) overwrites this live tail. + let activeLiveCell: { result: EvalCellResult; buf: TailBuffer } | undefined; const appendTail = (text: string) => { tailBuffer.append(text); @@ -507,6 +514,10 @@ export class EvalTool implements AgentTool { maxColumns: resolveOutputMaxColumns(session.settings), onChunk: chunk => { appendTail(chunk); + if (activeLiveCell) { + activeLiveCell.buf.append(chunk); + activeLiveCell.result.output = activeLiveCell.buf.text(); + } pushUpdate(); }, }); @@ -534,6 +545,7 @@ export class EvalTool implements AgentTool { cellResult.statusEvents = undefined; cellResult.exitCode = undefined; cellResult.durationMs = undefined; + activeLiveCell = { result: cellResult, buf: new TailBuffer(DEFAULT_MAX_BYTES * 2) }; pushUpdate(); const startTime = Date.now(); @@ -567,6 +579,7 @@ export class EvalTool implements AgentTool { }); } finally { idle.dispose(); + activeLiveCell = undefined; } const durationMs = Date.now() - startTime; diff --git a/packages/coding-agent/test/tools/eval-streaming-output.test.ts b/packages/coding-agent/test/tools/eval-streaming-output.test.ts new file mode 100644 index 000000000..b7d7af46e --- /dev/null +++ b/packages/coding-agent/test/tools/eval-streaming-output.test.ts @@ -0,0 +1,87 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import * as evalIndex from "@oh-my-pi/pi-coding-agent/eval"; +import type { EvalToolDetails } from "@oh-my-pi/pi-coding-agent/eval/types"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { EvalTool } from "@oh-my-pi/pi-coding-agent/tools/eval"; + +function makeSession(): ToolSession { + return { + cwd: "/tmp/eval-test", + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => null, + settings: Settings.isolated(), + }; +} + +function baseResult(overrides: Record = {}) { + return { + output: "", + exitCode: 0, + cancelled: false, + truncated: false, + artifactId: undefined, + totalLines: 0, + totalBytes: 0, + outputLines: 0, + outputBytes: 0, + displayOutputs: [] as unknown[], + ...overrides, + }; +} + +/** + * Defends the contract that stdout streamed by a still-running cell lands in the + * running cell's rendered `output` *before* `backend.execute()` returns — so a + * long-running cell (e.g. a `time.sleep()` monitor loop) shows progress live in + * the eval card instead of dumping everything at once on completion/interrupt. + * + * The eval card renderer draws cell output from `details.cells[i].output`; if + * that field is only filled after the backend resolves, the card stays blank for + * the whole run. This pins the live bridge that keeps it populated mid-flight. + */ +describe("EvalTool live stdout streaming", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("populates the running cell's output with streamed chunks before the cell returns", async () => { + const updates: EvalToolDetails[] = []; + vi.spyOn(evalIndex.jsBackend, "execute").mockImplementation((async ( + _code: string, + options: { onChunk?: (chunk: string) => void }, + ) => { + // Emit a chunk while the cell is still running, mirroring a print() + // before a long sleep. The host's onChunk path runs synchronously. + options.onChunk?.("tick 1\n"); + return baseResult({ output: "tick 1\ntick 2\n" }); + }) as never); + + const tool = new EvalTool(makeSession()); + const result = await tool.execute( + "call-stream", + { language: "js", code: "for (let i = 0; i < 2; i++) print('tick ' + i)" }, + undefined, + update => { + if (update.details) updates.push(update.details as EvalToolDetails); + }, + ); + + // A snapshot taken while the cell was still running carried the streamed + // chunk — proving the output bridge fires mid-execution, not just at the end. + const liveRunning = updates.find( + d => d.cells?.[0]?.status === "running" && (d.cells?.[0]?.output ?? "").includes("tick 1"), + ); + expect(liveRunning).toBeDefined(); + // The live snapshot shows only what streamed so far, not the post-return total. + expect(liveRunning?.cells?.[0]?.output).not.toContain("tick 2"); + + // Completion still overwrites with the authoritative full output. + const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); + expect(text).toContain("tick 1"); + expect(text).toContain("tick 2"); + expect(result.details?.cells?.[0]?.status).toBe("complete"); + expect(result.details?.cells?.[0]?.output).toContain("tick 2"); + }); +}); From 964dc480c94b473d8df4ac307fc6ab816a0943db Mon Sep 17 00:00:00 2001 From: roboomp Date: Mon, 22 Jun 2026 23:56:25 +0000 Subject: [PATCH 29/43] fix(session-selector): derive delete-dialog reserve from rendered height The first round of the issue #3283 fix reserved a fixed 12 SessionList rows for the delete confirmation dialog. On a narrow terminal or against a long session name, HookSelectorComponent's Markdown title and help text wrap past 12 rows; the picker would still overflow even after the SessionList shrank to zero entries, and the TUI committed the picker header into native scrollback again. SessionSelectorComponent now overrides render() to measure the dialog's actual rendered height at the live width before super.render() walks the children, and pushes that as the SessionList's external-row reserve. The dialog's own Container memoization makes the extra pre-render essentially free. Addresses PR #3285 review feedback. --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/modes/components/session-selector.ts | 49 +++++++------- .../session-selector-scroll-stability.test.ts | 64 +++++++++++++++++++ 3 files changed, 92 insertions(+), 23 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index deb544cc5..e9800f2af 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,7 +13,7 @@ - Prevented `/handoff` from executing while a response is streaming to avoid session corruption - Fixed `/handoff` cold-missing the provider prompt cache. Handoff generation now builds its request through the same pipeline a live turn uses (`convertMessagesToLlm` + `Agent.buildSideRequestContext` + `prepareSimpleStreamOptions`, via the new `generateHandoffFromContext`), so it reuses the live system prompt, normalized tools, transformed/obfuscated message history, and — critically — a stable `promptCacheKey` with a unique side `sessionId`. Previously the oneshot sent no cache-routing key and skipped the `transformContext`/`transformProviderContext` and tool/message normalization the loop applies, so its prefix never matched what the turn populated and every handoff re-read the whole context uncached. Mirrors the cache-preserving path already used by `/btw` and `/omfg`. - Fixed `/handoff` (and the RPC `handoff` command) resetting the agent while a response was still streaming, which let the live turn keep emitting into the torn-down session. Manual handoff now refuses while a prompt is in flight (matching `/fork` and `/move`); the auto-handoff path is unaffected. -- Fixed the `/resume` session picker scrolling down after a session is deleted. The delete-confirmation dialog mounted below the picker's bottom border briefly grew the picker past the terminal height; the TUI committed the picker's header rows into native scrollback to fit, and when the dialog closed the picker re-rendered shorter with `windowTop` pinned at the new commit boundary — leaving the header stranded above the viewport. The picker now reserves a row budget for the dialog while it is mounted so the picker's total rendered output stays within the terminal viewport and never commits ([#3283](https://github.com/can1357/oh-my-pi/issues/3283)) +- Fixed the `/resume` session picker scrolling down after a session is deleted. The delete-confirmation dialog mounted below the picker's bottom border briefly grew the picker past the terminal height; the TUI committed the picker's header rows into native scrollback to fit, and when the dialog closed the picker re-rendered shorter with `windowTop` pinned at the new commit boundary — leaving the header stranded above the viewport. The picker now overrides `render()` to measure the dialog's actual rendered height at the live width (a `Markdown` title plus a long session name can wrap past any fixed guess) and reserves that many rows in the `SessionList` budget while the dialog is mounted, so the picker's total rendered output stays within the terminal viewport and never commits ([#3283](https://github.com/can1357/oh-my-pi/issues/3283)) ## [16.1.15] - 2026-06-22 diff --git a/packages/coding-agent/src/modes/components/session-selector.ts b/packages/coding-agent/src/modes/components/session-selector.ts index 055dd3116..d901d8e43 100644 --- a/packages/coding-agent/src/modes/components/session-selector.ts +++ b/packages/coding-agent/src/modes/components/session-selector.ts @@ -551,6 +551,24 @@ export class SessionSelectorComponent extends Container { this.addChild(new DynamicBorder()); } + /** + * Re-derive the SessionList's external-row reserve from the actual + * rendered height of the confirmation dialog at the live width before + * the regular composition walks the children. The dialog's title is + * Markdown and its option list, hint, and any countdown can all wrap + * on narrow terminals or against long session names; a fixed reserve + * would underestimate that and let the picker top still scroll into + * native scrollback (issue #3283 review feedback). The dialog's own + * `Container` memoization makes the extra pre-render essentially free + * — `super.render(width)` reuses the same cached array reference, so + * the engine still sees a stable child render. + */ + override render(width: number): readonly string[] { + const reserveRows = this.#confirmationDialog?.render(Math.max(1, width)).length ?? 0; + this.#sessionList.setExternalReserveRows(reserveRows); + return super.render(width); + } + #headerLabel(): string { const scopeLabel = this.#scope === "all" ? "all projects" : "current folder"; return `${theme.bold("Resume Session")} ${theme.fg("muted", `(${scopeLabel})`)}`; @@ -607,26 +625,14 @@ export class SessionSelectorComponent extends Container { this.#messageContainer.addChild(new Spacer(1)); } - // Rows the delete-confirmation dialog adds below the picker's bottom - // border. Used to shrink the SessionList while the dialog is mounted so - // the picker's total rendered output never exceeds the terminal height. - // Sized generously (title + session-name + spacer + 2 options + spacer + - // hint + leading/trailing blanks) so the SessionList always concedes - // enough room for the dialog as it grows, even when the displayed session - // name wraps. - static readonly #DELETE_DIALOG_RESERVE_ROWS = 12; - #showDeleteConfirmation(session: SessionInfo): void { const displayName = session.title || session.firstMessage.slice(0, 40) || session.id; const closeDialog = () => { this.removeChild(this.#confirmationDialog!); this.#confirmationDialog = null; - // Release the dialog's row reservation BEFORE requesting the - // rerender so the SessionList can grow back to its full window - // in the same frame the dialog disappears. Otherwise the picker - // re-renders short, leaving a band of blank rows beneath it for - // one frame (visible as a flicker / "still scrolled" feel). - this.#sessionList.setExternalReserveRows(0); + // The next render() override pass will see no dialog and reset + // the reserve to 0 before walking children, so the SessionList + // grows back inside the very same frame the dialog disappears. this.#onRequestRender?.(); }; this.#confirmationDialog = new HookSelectorComponent( @@ -648,13 +654,12 @@ export class SessionSelectorComponent extends Container { }, closeDialog, ); - // Shrink the SessionList by the dialog's worst-case height BEFORE - // mounting the dialog so the very first frame containing the dialog - // already fits the terminal viewport. Without this the dialog's first - // render still overflows and the TUI commits the picker's top rows - // to native scrollback before the SessionList has a chance to react - // — issue #3283. - this.#sessionList.setExternalReserveRows(SessionSelectorComponent.#DELETE_DIALOG_RESERVE_ROWS); + // The reserve is recomputed every frame inside render() from the + // dialog's actual rendered height at the live width — see the + // `render` override on this class — so no pre-mount reserve hint + // is needed here. That measurement runs BEFORE the children walk, + // so the first frame containing the dialog already fits the + // viewport even when the dialog's title or session-name wraps. this.addChild(this.#confirmationDialog); } diff --git a/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts b/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts index 0e2c37f60..02ac16f19 100644 --- a/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts +++ b/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts @@ -114,4 +114,68 @@ describe("issue #3283: /resume picker scrolls down after deleting a session", () await term.flush(); } }); + + it("derives the SessionList reserve from the dialog's actual rendered height", () => { + // Direct contract: the picker's render() override must size the + // SessionList's external reserve to the dialog's *actual* rendered + // height at the live width — not a fixed constant — so a narrow + // terminal or a long session title that wraps the dialog past the + // constant never leaves the picker overflowing the viewport + // (PR #3285 review feedback). + const longName = "a-very-very-very-very-long-session-title-that-must-wrap-on-a-narrow-terminal"; + const sessions: SessionInfo[] = [ + { + path: `/work/${longName}.jsonl`, + id: "id-long", + cwd: "/work", + title: longName, + created: new Date("2024-01-01T00:00:00Z"), + modified: new Date("2024-01-02T00:00:00Z"), + messageCount: 1, + size: 1024, + firstMessage: longName, + allMessagesText: longName, + }, + ...makeSessions(10), + ]; + + const NARROW_WIDTH = 30; + const TERMINAL_ROWS = 50; + const selector = new SessionSelectorComponent( + sessions, + () => {}, + () => {}, + () => {}, + { getTerminalRows: () => TERMINAL_ROWS, onDelete: async () => true }, + ); + + // Baseline: render the picker before the dialog opens. + const beforeOpen = selector.render(NARROW_WIDTH).length; + + // Open the delete confirmation. The picker's render override + // measures the dialog and pushes the reserve into the SessionList + // before super.render() walks the children, so the very first + // render after the dialog mounts already reflects the dynamic + // reserve. + selector.handleInput("\x1b[3~"); + const afterOpen = selector.render(NARROW_WIDTH); + const dialog = selector.children.at(-1); + expect(dialog).toBeDefined(); + const dialogHeight = dialog!.render(NARROW_WIDTH).length; + + // On a narrow width with a long title the dialog wraps past the + // previous hard-coded 12-row reserve. + expect(dialogHeight).toBeGreaterThan(12); + + // Contract: when the dialog wraps past the previous constant + // reserve (12), the dynamic reserve correctly shrinks the + // SessionList by the dialog's *actual* height, so the picker + // frame growth is ≤ 0 (sessions freed ≥ dialog rows added). A + // constant reserve only frees 12 rows regardless, so the picker + // frame grows by `dialogHeight - 12` rows on every dialog open. + // Allow a one-session rounding slack (4 rows) for the floor + // inside `#visibleCount`. + const growth = afterOpen.length - beforeOpen; + expect(growth).toBeLessThanOrEqual(0); + }); }); From e266782604504106feed3f11b4bf2f1743f91a65 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 23 Jun 2026 00:05:53 +0000 Subject: [PATCH 30/43] fix(session-selector): swap delete dialog into SessionList slot Earlier rounds shrank the SessionList by the dialog's row count to keep the picker inside the viewport, but the SessionList could only claw back whole session rows and bottomed out at zero entries. On a narrow terminal with a long session title the dialog still wrapped past what the SessionList could free, the picker overflowed the viewport, and the TUI committed the header into native scrollback. The picker now hosts the SessionList inside a single contentSlot Container. Opening the delete confirmation swaps the dialog INTO that slot (replacing the SessionList); closing it swaps the SessionList back. The dialog therefore competes only with the SessionList's rendered budget, not with the SessionList AND the picker chrome, so the picker frame stays bounded by terminalRows even when the dialog wraps to many rows. SessionList's external-reserve plumbing is no longer needed and is removed. Addresses PR #3285 second-round review feedback. --- packages/coding-agent/CHANGELOG.md | 2 +- .../src/modes/components/session-selector.ts | 99 +++++++------------ .../session-selector-scroll-stability.test.ts | 53 ++++------ 3 files changed, 54 insertions(+), 100 deletions(-) diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index e9800f2af..cea0818bc 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -13,7 +13,7 @@ - Prevented `/handoff` from executing while a response is streaming to avoid session corruption - Fixed `/handoff` cold-missing the provider prompt cache. Handoff generation now builds its request through the same pipeline a live turn uses (`convertMessagesToLlm` + `Agent.buildSideRequestContext` + `prepareSimpleStreamOptions`, via the new `generateHandoffFromContext`), so it reuses the live system prompt, normalized tools, transformed/obfuscated message history, and — critically — a stable `promptCacheKey` with a unique side `sessionId`. Previously the oneshot sent no cache-routing key and skipped the `transformContext`/`transformProviderContext` and tool/message normalization the loop applies, so its prefix never matched what the turn populated and every handoff re-read the whole context uncached. Mirrors the cache-preserving path already used by `/btw` and `/omfg`. - Fixed `/handoff` (and the RPC `handoff` command) resetting the agent while a response was still streaming, which let the live turn keep emitting into the torn-down session. Manual handoff now refuses while a prompt is in flight (matching `/fork` and `/move`); the auto-handoff path is unaffected. -- Fixed the `/resume` session picker scrolling down after a session is deleted. The delete-confirmation dialog mounted below the picker's bottom border briefly grew the picker past the terminal height; the TUI committed the picker's header rows into native scrollback to fit, and when the dialog closed the picker re-rendered shorter with `windowTop` pinned at the new commit boundary — leaving the header stranded above the viewport. The picker now overrides `render()` to measure the dialog's actual rendered height at the live width (a `Markdown` title plus a long session name can wrap past any fixed guess) and reserves that many rows in the `SessionList` budget while the dialog is mounted, so the picker's total rendered output stays within the terminal viewport and never commits ([#3283](https://github.com/can1357/oh-my-pi/issues/3283)) +- Fixed the `/resume` session picker scrolling down after a session is deleted. The delete-confirmation dialog was mounted as a sibling below the picker's bottom border, briefly growing the picker past the terminal height; the TUI committed the picker's header rows into native scrollback to fit, and when the dialog closed `windowTop` stayed pinned at the new commit boundary, leaving the header stranded above the viewport. The picker now hosts the `SessionList` inside a single content slot and swaps the dialog INTO that slot (replacing the `SessionList`) while it is open, so the dialog only competes with the `SessionList`'s rendered budget — not with the `SessionList` AND the picker chrome — and the picker frame stays inside the terminal viewport even on narrow terminals where the dialog's `Markdown` title plus a long session name wrap past any fixed reserve guess ([#3283](https://github.com/can1357/oh-my-pi/issues/3283)) ## [16.1.15] - 2026-06-22 diff --git a/packages/coding-agent/src/modes/components/session-selector.ts b/packages/coding-agent/src/modes/components/session-selector.ts index d901d8e43..f54ea7785 100644 --- a/packages/coding-agent/src/modes/components/session-selector.ts +++ b/packages/coding-agent/src/modes/components/session-selector.ts @@ -172,15 +172,6 @@ class SessionList implements Component { onDeleteRequest?: (session: SessionInfo) => void; - // Extra row count the picker must keep free for a sibling component - // (currently the delete-confirmation dialog) mounted below the bottom - // border. Set by `SessionSelectorComponent` while the dialog is on screen - // so the SessionList shrinks its visible window and the picker's total - // rendered output stays within the terminal height. Without this the - // picker overflows by ~dialog-height rows, the TUI commits those top rows - // to native scrollback to fit, and when the dialog closes the picker - // header is stranded above the viewport — issue #3283. - #externalReserveRows = 0; #allSessions: SessionInfo[]; #showCwd: boolean; readonly #historyMatcher?: SessionHistoryMatcher; @@ -207,41 +198,24 @@ class SessionList implements Component { }; } - /** - * Reserve `rows` of vertical budget for a sibling component rendered - * outside the SessionList (e.g. the delete-confirmation dialog). The - * visible-entry window shrinks accordingly so the picker frame stays - * within the terminal viewport while the sibling is mounted. - */ - setExternalReserveRows(rows: number): void { - const next = Math.max(0, Math.trunc(rows)); - if (next === this.#externalReserveRows) return; - this.#externalReserveRows = next; - } - /** * Number of sessions to show at once, sized so the whole picker fits the * current viewport instead of pushing its header/search off the top. * - * Budget = rows − chrome − reserve − externalReserve, divided by the - * worst-case per-session height. Chrome (12) is the surrounding - * spacers/borders/header (7) plus the list's search line, blank, scroll - * indicator, blank, and hint (5). A titled session is the tallest item at - * 4 lines (title + preview + metadata + blank); budgeting for that - * guarantees no overflow even when every visible entry has a title. The - * reserve covers below-editor hook widgets / cursor. The external reserve - * is non-zero only while a sibling overlay (the delete-confirmation - * dialog) is mounted below the bottom border, and is allowed to drive - * the count down to zero so the dialog never pushes the picker past the - * terminal height. + * Budget = rows − chrome − reserve, divided by the worst-case per-session + * height. Chrome (12) is the surrounding spacers/borders/header (7) plus + * the list's search line, blank, scroll indicator, blank, and hint (5). + * A titled session is the tallest item at 4 lines (title + preview + + * metadata + blank); budgeting for that guarantees no overflow even when + * every visible entry has a title. The reserve covers below-editor hook + * widgets / cursor. */ #visibleCount(): number { const CHROME = 12; const PER_SESSION = 4; const RESERVE = 1; - const budget = this.#getTerminalRows() - CHROME - RESERVE - this.#externalReserveRows; - const minimum = this.#externalReserveRows > 0 ? 0 : 2; - return Math.max(minimum, Math.floor(budget / PER_SESSION)); + const budget = this.#getTerminalRows() - CHROME - RESERVE; + return Math.max(2, Math.floor(budget / PER_SESSION)); } /** Replace the visible dataset, e.g. when toggling folder/all-projects scope. */ @@ -497,6 +471,15 @@ export interface SessionSelectorOptions { export class SessionSelectorComponent extends Container { #sessionList: SessionList; #confirmationDialog: HookSelectorComponent | null = null; + // Hosts whichever of `#sessionList` / `#confirmationDialog` is live this + // frame. The dialog REPLACES the SessionList (not augments it) so the + // picker layout becomes `chrome + max(sessionList, dialog) + chrome` — + // the dialog never adds rows on top of the SessionList's own chrome. + // Without this, on a narrow terminal the dialog could still push the + // picker past the viewport once the SessionList shrank to zero entries + // (issue #3283 review, PR #3285): the dialog had no rows left to claw + // back and was rendered alongside the SessionList chrome. + #contentSlot: Container; #messageContainer: Container; #headerText: Text; #onDelete?: (session: SessionInfo) => Promise; @@ -544,31 +527,15 @@ export class SessionSelectorComponent extends Container { void this.#toggleScope(); }; } - this.addChild(this.#sessionList); + this.#contentSlot = new Container(); + this.#contentSlot.addChild(this.#sessionList); + this.addChild(this.#contentSlot); // Add bottom border this.addChild(new Spacer(1)); this.addChild(new DynamicBorder()); } - /** - * Re-derive the SessionList's external-row reserve from the actual - * rendered height of the confirmation dialog at the live width before - * the regular composition walks the children. The dialog's title is - * Markdown and its option list, hint, and any countdown can all wrap - * on narrow terminals or against long session names; a fixed reserve - * would underestimate that and let the picker top still scroll into - * native scrollback (issue #3283 review feedback). The dialog's own - * `Container` memoization makes the extra pre-render essentially free - * — `super.render(width)` reuses the same cached array reference, so - * the engine still sees a stable child render. - */ - override render(width: number): readonly string[] { - const reserveRows = this.#confirmationDialog?.render(Math.max(1, width)).length ?? 0; - this.#sessionList.setExternalReserveRows(reserveRows); - return super.render(width); - } - #headerLabel(): string { const scopeLabel = this.#scope === "all" ? "all projects" : "current folder"; return `${theme.bold("Resume Session")} ${theme.fg("muted", `(${scopeLabel})`)}`; @@ -628,11 +595,12 @@ export class SessionSelectorComponent extends Container { #showDeleteConfirmation(session: SessionInfo): void { const displayName = session.title || session.firstMessage.slice(0, 40) || session.id; const closeDialog = () => { - this.removeChild(this.#confirmationDialog!); this.#confirmationDialog = null; - // The next render() override pass will see no dialog and reset - // the reserve to 0 before walking children, so the SessionList - // grows back inside the very same frame the dialog disappears. + // Restore the SessionList in the content slot so the picker is + // back to its normal layout on the very next render — same + // frame the dialog disappears. + this.#contentSlot.clear(); + this.#contentSlot.addChild(this.#sessionList); this.#onRequestRender?.(); }; this.#confirmationDialog = new HookSelectorComponent( @@ -654,13 +622,14 @@ export class SessionSelectorComponent extends Container { }, closeDialog, ); - // The reserve is recomputed every frame inside render() from the - // dialog's actual rendered height at the live width — see the - // `render` override on this class — so no pre-mount reserve hint - // is needed here. That measurement runs BEFORE the children walk, - // so the first frame containing the dialog already fits the - // viewport even when the dialog's title or session-name wraps. - this.addChild(this.#confirmationDialog); + // Swap the SessionList out of the content slot and mount the + // dialog in its place. The picker's vertical budget for the + // content slot is `terminalRows - pickerChrome`, so as long as + // the dialog fits that bound the picker frame stays inside the + // terminal viewport and the TUI never commits anything. + this.#contentSlot.clear(); + this.#contentSlot.addChild(this.#confirmationDialog); + this.#onRequestRender?.(); } handleInput(keyData: string): void { diff --git a/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts b/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts index 02ac16f19..42b67b83e 100644 --- a/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts +++ b/packages/coding-agent/test/modes/components/session-selector-scroll-stability.test.ts @@ -115,13 +115,17 @@ describe("issue #3283: /resume picker scrolls down after deleting a session", () } }); - it("derives the SessionList reserve from the dialog's actual rendered height", () => { - // Direct contract: the picker's render() override must size the - // SessionList's external reserve to the dialog's *actual* rendered - // height at the live width — not a fixed constant — so a narrow - // terminal or a long session title that wraps the dialog past the - // constant never leaves the picker overflowing the viewport - // (PR #3285 review feedback). + it("keeps the picker frame bounded on a narrow terminal where the dialog title wraps", () => { + // Direct structural contract: even on a small terminal with a + // session name that wraps the dialog past any plausible fixed + // reserve, the picker's total rendered output stays within the + // terminal height. The dialog REPLACES the SessionList inside + // the picker frame, so its rows compete only with the + // SessionList's rendered budget — not with both the SessionList + // AND picker chrome (PR #3285 second review round). Without the + // structural fix the dialog could push the picker past the + // viewport once the SessionList reserve could not shrink any + // further, and the TUI committed the header into scrollback. const longName = "a-very-very-very-very-long-session-title-that-must-wrap-on-a-narrow-terminal"; const sessions: SessionInfo[] = [ { @@ -140,7 +144,7 @@ describe("issue #3283: /resume picker scrolls down after deleting a session", () ]; const NARROW_WIDTH = 30; - const TERMINAL_ROWS = 50; + const TERMINAL_ROWS = 24; const selector = new SessionSelectorComponent( sessions, () => {}, @@ -149,33 +153,14 @@ describe("issue #3283: /resume picker scrolls down after deleting a session", () { getTerminalRows: () => TERMINAL_ROWS, onDelete: async () => true }, ); - // Baseline: render the picker before the dialog opens. - const beforeOpen = selector.render(NARROW_WIDTH).length; + // Baseline: picker fits the viewport before the dialog opens. + expect(selector.render(NARROW_WIDTH).length).toBeLessThanOrEqual(TERMINAL_ROWS); - // Open the delete confirmation. The picker's render override - // measures the dialog and pushes the reserve into the SessionList - // before super.render() walks the children, so the very first - // render after the dialog mounts already reflects the dynamic - // reserve. + // Open the delete confirmation; the dialog takes the SessionList + // slot inside the picker frame. The picker's rendered total must + // STILL fit the viewport even though the dialog wraps past the + // previous 12-row reserve guess. selector.handleInput("\x1b[3~"); - const afterOpen = selector.render(NARROW_WIDTH); - const dialog = selector.children.at(-1); - expect(dialog).toBeDefined(); - const dialogHeight = dialog!.render(NARROW_WIDTH).length; - - // On a narrow width with a long title the dialog wraps past the - // previous hard-coded 12-row reserve. - expect(dialogHeight).toBeGreaterThan(12); - - // Contract: when the dialog wraps past the previous constant - // reserve (12), the dynamic reserve correctly shrinks the - // SessionList by the dialog's *actual* height, so the picker - // frame growth is ≤ 0 (sessions freed ≥ dialog rows added). A - // constant reserve only frees 12 rows regardless, so the picker - // frame grows by `dialogHeight - 12` rows on every dialog open. - // Allow a one-session rounding slack (4 rows) for the floor - // inside `#visibleCount`. - const growth = afterOpen.length - beforeOpen; - expect(growth).toBeLessThanOrEqual(0); + expect(selector.render(NARROW_WIDTH).length).toBeLessThanOrEqual(TERMINAL_ROWS); }); }); From cffb804d3a0dfdf5e13cd1c41b19600deac5fef5 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 02:25:30 +0200 Subject: [PATCH 31/43] feat(coding-agent): added mouse support and fullscreen rendering to session picker - Enabled fullscreen overlay rendering for the terminal session picker. - Implemented full mouse support including wheel-based scrolling and click-to-select functionality. - Anchored the session picker footer to the bottom of the viewport to correct UI flickering. - Added comprehensive unit tests for mouse interaction and layout constancy during resizing. --- packages/coding-agent/CHANGELOG.md | 5 +- .../coding-agent/src/cli/session-picker.ts | 20 ++- .../src/modes/components/session-selector.ts | 128 ++++++++++++++-- .../components/session-selector-mouse.test.ts | 140 ++++++++++++++++++ 4 files changed, 275 insertions(+), 18 deletions(-) create mode 100644 packages/coding-agent/test/modes/components/session-selector-mouse.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b2f80d4b1..2d28e2b5b 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Breaking Changes - Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. @@ -12,12 +13,15 @@ ### Changed +- Made the session picker fullscreen with mouse support for clicking rows and scrolling +- Pinned the session picker footer to the bottom of the screen to prevent layout flickering - Simplified `eval` tool to accept a single logical step (code block) instead of an array of cells - Updated `eval` tool documentation to emphasize incremental, single-step execution - Restricted `bash` tool from using `ls` or `find`, requiring the use of `read` or `find` tools - Simplified `todo` tool interface to accept a single operation directly instead of an array of ops - Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised). - Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`. +- Made the `--resume` session picker fullscreen on the terminal's alternate screen, so the list scrolls with the mouse wheel and a row resumes its session on left click. Rows are hit-tested against the live scroll window, and the keybinding hint + bottom border are now pinned to the screen bottom instead of drifting up and down as the visible window changes height. ### Removed @@ -29,7 +33,6 @@ ### Fixed - Fixed the `eval` tool card not streaming a still-running cell's stdout: a long-running cell (e.g. a `time.sleep()` monitor loop) showed nothing until it returned or was interrupted, then dumped everything at once. The renderer draws cell output from `details.cells[i].output`, which was only populated after `backend.execute()` resolved — live stdout streamed into the transient result `content` tail (and `renderContext.output`), which the per-cell render branch ignores. Streamed chunks now append to the active cell's `output` (a dedicated per-cell tail buffer, capped like the aggregate) as they arrive, so the card shows progress live; on completion the authoritative full output overwrites the live tail. `log()`/`phase()`/`display()` and status ops were unaffected because they already stream via the status channel. - - Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path. - Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default. - Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners diff --git a/packages/coding-agent/src/cli/session-picker.ts b/packages/coding-agent/src/cli/session-picker.ts index 0a7e5facd..0debd5274 100644 --- a/packages/coding-agent/src/cli/session-picker.ts +++ b/packages/coding-agent/src/cli/session-picker.ts @@ -8,8 +8,10 @@ import { FileSessionStorage } from "../session/session-storage"; /** * Show the TUI session selector and return the selected session, or null if - * cancelled. Tab toggles between current-folder and all-projects scope; the - * all-projects list is loaded lazily via `SessionManager.listAll`. + * cancelled. Rendered as a fullscreen overlay on the terminal's alternate + * screen, so the list scrolls and rows are clickable with the mouse. Tab + * toggles between current-folder and all-projects scope; the all-projects list + * is loaded lazily via `SessionManager.listAll`. */ export async function selectSession( sessions: SessionInfo[], @@ -65,6 +67,7 @@ export async function selectSession( loadAllSessions: () => SessionManager.listAll(storage), allSessions: options?.allSessions, getTerminalRows: () => ui.terminal.rows, + fillHeight: true, }, ); return selector; @@ -72,7 +75,18 @@ export async function selectSession( const selector = showSelector(); selector.setOnRequestRender(() => ui.requestRender()); - ui.addChild(selector); + // Present as a fullscreen overlay so the picker borrows the terminal's + // alternate screen buffer (vim/less idiom): the list scrolls and rows are + // clickable via the mouse tracking the overlay enables for its lifetime. + // Anchored top-left at full size so a mouse row maps directly to a rendered + // line (the overlay paints from screen row 0). + ui.showOverlay(selector, { + anchor: "top-left", + width: "100%", + maxHeight: "100%", + margin: 0, + fullscreen: true, + }); ui.setFocus(selector); ui.start(); return promise; diff --git a/packages/coding-agent/src/modes/components/session-selector.ts b/packages/coding-agent/src/modes/components/session-selector.ts index c6f403894..8d9537faa 100644 --- a/packages/coding-agent/src/modes/components/session-selector.ts +++ b/packages/coding-agent/src/modes/components/session-selector.ts @@ -5,6 +5,7 @@ import { Input, matchesKey, padding, + parseSgrMouse, replaceTabs, ScrollView, Spacer, @@ -161,6 +162,12 @@ export function mergeSessionRanking( class SessionList implements Component { #filteredSessions: SessionInfo[] = []; #selectedIndex: number = 0; + // Maps a 0-based line within this list's own render to a filtered-session + // index, or undefined for chrome rows (search line, blanks, scrollbar gap). + // Rebuilt every render so the picker's mouse hit-testing tracks the live + // scroll window. Only consulted while the picker holds the alternate screen + // (where the overlay enables mouse tracking and paints from screen row 0). + #hitRows: (number | undefined)[] = []; readonly #searchInput: Input; onSelect?: (session: SessionInfo) => void; onCancel?: () => void; @@ -257,12 +264,32 @@ class SessionList implements Component { } } + /** Resolve a list-local rendered-line index to a filtered-session index. */ + hitTestSession(line: number): number | undefined { + return this.#hitRows[line]; + } + + /** Wheel notch: move the selection one step (clamped, no wrap). */ + handleWheel(delta: -1 | 1): void { + if (this.#filteredSessions.length === 0) return; + this.#selectedIndex = Math.max(0, Math.min(this.#filteredSessions.length - 1, this.#selectedIndex + delta)); + } + + /** Mouse click: select the session under the pointer and resume it. */ + selectAndConfirm(index: number): void { + const session = this.#filteredSessions[index]; + if (!session) return; + this.#selectedIndex = index; + this.onSelect?.(session); + } + invalidate(): void { // No cached state to invalidate currently } render(width: number): readonly string[] { const lines: string[] = []; + this.#hitRows = []; // Render search input lines.push(...this.#searchInput.render(width)); @@ -311,9 +338,11 @@ class SessionList implements Component { // Each session block is built into sessionLines, then wrapped by ScrollView // so the right-edge scrollbar is proportional at the physical-line level. const sessionLines: string[] = []; + const sessionRowIndex: number[] = []; const overflow = this.#filteredSessions.length > maxVisible; const rowWidth = Math.max(0, width - (overflow ? 1 : 0)); for (let i = startIndex; i < endIndex; i++) { + const blockStart = sessionLines.length; const session = this.#filteredSessions[i]; const isSelected = i === this.#selectedIndex; @@ -363,6 +392,7 @@ class SessionList implements Component { sessionLines.push(metadataLine); sessionLines.push(""); // Blank line between sessions + for (let k = blockStart; k < sessionLines.length; k++) sessionRowIndex[k] = i; } // Wrap the rendered window in a ScrollView for a proportional right-edge bar. @@ -375,16 +405,10 @@ class SessionList implements Component { theme: { track: t => theme.fg("muted", t), thumb: t => theme.fg("accent", t) }, }); sv.setScrollOffset(Math.round(startIndex * linesPerItem)); - lines.push(...sv.render(width)); - - // Add keybinding hint - lines.push(""); - lines.push( - theme.fg( - "muted", - ` [Del delete · Enter select · Tab ${this.#showCwd ? "current folder" : "all projects"} · Esc cancel]`, - ), - ); + const sessionRegionStart = lines.length; + const svLines = sv.render(width); + for (let k = 0; k < svLines.length; k++) this.#hitRows[sessionRegionStart + k] = sessionRowIndex[k]; + lines.push(...svLines); return lines; } @@ -462,6 +486,13 @@ export interface SessionSelectorOptions { * Omitted only in tests; defaults to a conservative 24 rows. */ getTerminalRows?: () => number; + /** + * Fill the whole viewport and pin the footer (hint + bottom border) to the + * last rows, so the footer stops drifting as the list window changes height. + * Set by the standalone `--resume` picker (fullscreen alternate screen); the + * in-editor selector leaves it off and renders compactly. + */ + fillHeight?: boolean; } /** @@ -479,6 +510,18 @@ export class SessionSelectorComponent extends Container { #globalSessions: SessionInfo[] | null = null; #scope: "folder" | "all" = "folder"; #toggling = false; + // 0-based line where the session list begins within this component's own + // render, captured each frame. The fullscreen picker overlay paints from + // screen row 0, so a mouse row maps to `row - #listLineOffset` inside the + // list. Only meaningful while the picker holds the alternate screen. + #listLineOffset = 0; + // 0-based line where the pinned footer begins; clicks at or below it never + // hit-test the list, so a footer click on a cramped (trimmed) frame can't + // resume a session scrolled off-screen. + #footerStart = 0; + readonly #getTerminalRows: () => number; + readonly #fillHeight: boolean; + readonly #bottomBorder = new DynamicBorder(); constructor( sessions: SessionInfo[], @@ -494,6 +537,8 @@ export class SessionSelectorComponent extends Container { this.#loadAllSessions = options.loadAllSessions; this.#folderSessions = sessions; this.#globalSessions = options.allSessions ?? null; + this.#getTerminalRows = options.getTerminalRows ?? (() => 24); + this.#fillHeight = options.fillHeight ?? false; // Add header this.addChild(new Spacer(1)); this.#headerText = new Text(this.#headerLabel(), 1, 0); @@ -518,10 +563,6 @@ export class SessionSelectorComponent extends Container { }; } this.addChild(this.#sessionList); - - // Add bottom border - this.addChild(new Spacer(1)); - this.addChild(new DynamicBorder()); } #headerLabel(): string { @@ -615,7 +656,47 @@ export class SessionSelectorComponent extends Container { this.addChild(this.#confirmationDialog); } + /** + * Concatenate the children's renders (like {@link Container}) while recording + * the line where the session list begins, so the fullscreen picker can hit- + * test mouse rows against the live list window. SessionList rebuilds its lines + * every frame, so Container's reference-memoization never applied here. + * + * In fill-height mode the body is padded (or, on a cramped terminal, trimmed) + * to leave exactly enough room for the footer at the screen bottom, so the + * footer is always visible and never drifts as the list window resizes. The + * in-editor selector just appends the footer directly. + */ + render(width: number): readonly string[] { + const lines: string[] = []; + for (const child of this.children) { + const childLines = child.render(width); + if (child === this.#sessionList) this.#listLineOffset = lines.length; + for (const line of childLines) lines.push(line); + } + const footer = this.#footerLines(width); + if (this.#fillHeight) { + const target = Math.max(0, this.#getTerminalRows() - footer.length); + if (lines.length > target) lines.length = target; + else for (let i = lines.length; i < target; i++) lines.push(""); + } + this.#footerStart = lines.length; + for (const line of footer) lines.push(line); + return lines; + } + + /** Blank · keybinding hint · bottom border. Rendered by {@link render}. */ + #footerLines(width: number): string[] { + const scopeHint = this.#scope === "all" ? "current folder" : "all projects"; + const hint = theme.fg("muted", ` [Del delete · Enter select · Tab ${scopeHint} · Esc cancel]`); + return ["", hint, "", ...this.#bottomBorder.render(width)]; + } + handleInput(keyData: string): void { + if (keyData.startsWith("\x1b[<")) { + this.#handleMouse(keyData); + return; + } if (this.#confirmationDialog) { this.#confirmationDialog.handleInput(keyData); } else { @@ -623,6 +704,25 @@ export class SessionSelectorComponent extends Container { } } + /** + * SGR mouse reports, delivered only while the picker holds the alternate + * screen (the fullscreen overlay enables tracking and paints from screen row + * 0). Wheel scrolls the list; a left click resumes the session under the + * pointer. Mouse is inert while the delete-confirmation dialog is open. + */ + #handleMouse(data: string): void { + if (this.#confirmationDialog) return; + const event = parseSgrMouse(data); + if (!event) return; + if (event.wheel !== null) { + this.#sessionList.handleWheel(event.wheel); + return; + } + if (!event.leftClick || event.row >= this.#footerStart) return; + const index = this.#sessionList.hitTestSession(event.row - this.#listLineOffset); + if (index !== undefined) this.#sessionList.selectAndConfirm(index); + } + getSessionList(): SessionList { return this.#sessionList; } diff --git a/packages/coding-agent/test/modes/components/session-selector-mouse.test.ts b/packages/coding-agent/test/modes/components/session-selector-mouse.test.ts new file mode 100644 index 000000000..87fc5fb90 --- /dev/null +++ b/packages/coding-agent/test/modes/components/session-selector-mouse.test.ts @@ -0,0 +1,140 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-listing"; + +beforeAll(async () => { + await initTheme(); +}); + +function makeSession(id: string, title: string | undefined): SessionInfo { + return { + path: `/work/${id}.jsonl`, + id, + cwd: "/work", + title, + created: new Date("2024-01-01T00:00:00Z"), + modified: new Date("2024-01-02T00:00:00Z"), + messageCount: 1, + size: 1024, + firstMessage: `body for ${id}`, + allMessagesText: `body for ${id}`, + }; +} + +/** SGR left-button press at a 1-based screen row (column is irrelevant for row hit-testing). */ +function leftClick(row1Based: number, col1Based = 4): string { + return `\x1b[<0;${col1Based};${row1Based}M`; +} + +/** SGR wheel notch: button 64 = up, 65 = down. */ +function wheel(direction: "up" | "down"): string { + return `\x1b[<${direction === "down" ? 65 : 64};1;1M`; +} + +function makeSelector( + sessions: SessionInfo[], + onSelect: (s: SessionInfo) => void, + rows = 40, +): SessionSelectorComponent { + return new SessionSelectorComponent( + sessions, + onSelect, + () => {}, + () => {}, + { + getTerminalRows: () => rows, + fillHeight: true, + }, + ); +} + +describe("SessionSelectorComponent mouse", () => { + it("resumes the session under a left click", () => { + const sessions = [ + makeSession("aaaa", "Alpha session"), + makeSession("bbbb", "Beta session"), + makeSession("cccc", "Gamma session"), + ]; + let picked: SessionInfo | undefined; + const selector = makeSelector(sessions, s => { + picked = s; + }); + + // Render first so the hit-test map and list offset reflect this frame. + const lines = selector.render(80); + const betaRow = lines.findIndex(line => line.includes("Beta session")); + expect(betaRow).toBeGreaterThanOrEqual(0); + + // Mouse rows are 1-based; the fullscreen overlay paints from screen row 0. + selector.handleInput(leftClick(betaRow + 1)); + expect(picked?.id).toBe("bbbb"); + }); + + it("scrolls the selection with the wheel, then resumes it on Enter", () => { + const sessions = [ + makeSession("aaaa", "Alpha session"), + makeSession("bbbb", "Beta session"), + makeSession("cccc", "Gamma session"), + ]; + let picked: SessionInfo | undefined; + const selector = makeSelector(sessions, s => { + picked = s; + }); + + selector.render(80); + // Selection starts at the first row; two notches down lands on Gamma. + selector.handleInput(wheel("down")); + selector.handleInput(wheel("down")); + selector.handleInput("\n"); + expect(picked?.id).toBe("cccc"); + }); + + it("ignores a click on the pinned footer (never resumes a hidden session)", () => { + const sessions = Array.from({ length: 20 }, (_, i) => makeSession(`s${i}`, `Title ${i}`)); + let picked: SessionInfo | undefined; + const selector = makeSelector( + sessions, + s => { + picked = s; + }, + 40, + ); + + const lines = selector.render(80); + const footerRow = lines.findIndex(line => line.includes("Esc cancel")); + expect(footerRow).toBeGreaterThanOrEqual(0); + + // Click directly on the footer hint row: must not resume anything. + selector.handleInput(leftClick(footerRow + 1)); + expect(picked).toBeUndefined(); + }); +}); + +describe("SessionSelectorComponent fill-height footer", () => { + // First half titled (4 rows each), second half untitled (3 rows each), so the + // scrolled window changes height — the regression that made the footer drift. + function mixedSessions(count: number): SessionInfo[] { + return Array.from({ length: count }, (_, i) => makeSession(`s${i}`, i < count / 2 ? `Titled ${i}` : undefined)); + } + + it("fills the viewport and pins the footer to the bottom regardless of scroll", () => { + const rows = 40; + const selector = makeSelector(mixedSessions(20), () => {}, rows); + + const top = selector.render(80); + const topHint = top.findIndex(line => line.includes("Esc cancel")); + expect(top.length).toBe(rows); + expect(topHint).toBe(rows - 3); + expect(top[rows - 1]!.trim().length).toBeGreaterThan(0); // bottom border on the last row + + // Scroll to the bottom of the list (now an untitled window of a different + // height); the footer must not move. + for (let i = 0; i < 25; i++) selector.handleInput(wheel("down")); + const bottom = selector.render(80); + const bottomHint = bottom.findIndex(line => line.includes("Esc cancel")); + expect(bottom.length).toBe(rows); + expect(bottomHint).toBe(topHint); + expect(bottom[rows - 1]!.trim().length).toBeGreaterThan(0); + }); +}); From c686c1c80349d0af6539b63207d1c91a882ee36f Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 23 Jun 2026 01:27:52 +0000 Subject: [PATCH 32/43] fix(ai): kept anthropic thinking context Sent context_management.keep=all for all enabled Anthropic thinking requests so API-key and Anthropic-compatible providers preserve replayed reasoning blocks across turns. Added regression coverage for budget and adaptive thinking payloads. Fixes #3288 --- packages/ai/CHANGELOG.md | 4 ++++ packages/ai/src/providers/anthropic.ts | 16 +++++++++---- packages/ai/test/anthropic-alignment.test.ts | 8 ++++--- ...anthropic-unsigned-thinking-replay.test.ts | 23 +++++++++++++++++++ 4 files changed, 43 insertions(+), 8 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 5a45d955e..6aad3230a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Fixed + +- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) + ## [16.1.15] - 2026-06-22 ### Fixed diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index bfa6f6773..f681f7f54 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2944,11 +2944,17 @@ function buildParams( } } - // Pre-compute context_management (depends on thinking). - const contextManagement = - isOAuthToken && thinking?.type === "adaptive" - ? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] } - : undefined; + // Pre-compute context_management. Send keep: "all" for every enabled or + // adaptive thinking request (OAuth + API-key) — not just OAuth. Without + // this directive Anthropic-compatible backends (Z.AI, Kimi, DeepSeek, …) + // strip the replayed thinking blocks `replayUnsignedThinking` puts back + // on the wire, so the model loses the prior reasoning chain across turns + // and the KV cache misses every turn (#3288). Narrowing this guard back + // to `isOAuthToken` regresses every API-key thinking provider. + const shouldKeepThinkingContext = thinking?.type === "adaptive" || thinking?.type === "enabled"; + const contextManagement = shouldKeepThinkingContext + ? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] } + : undefined; // Pre-compute output_config. const outputConfigEntries: AnthropicOutputConfig = {}; diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index 47a966f68..47a2714fc 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -1870,7 +1870,7 @@ describe("Anthropic request fingerprint alignment", () => { expect(maxPayload.output_config).toEqual({ effort: "max" }); }); - it("keeps summarized adaptive thinking by default for API-key Opus 4.7+ requests", async () => { + it("keeps summarized adaptive thinking and context management for API-key Opus 4.7+ requests", async () => { const payload = (await captureAnthropicPayload( buildModel({ ...ANTHROPIC_MODEL_SPEC, @@ -1892,12 +1892,14 @@ describe("Anthropic request fingerprint alignment", () => { }, )) as { thinking?: { type?: string; display?: string }; - context_management?: unknown; + context_management?: { edits?: Array<{ type?: string; keep?: string | number }> }; output_config?: { effort?: string }; }; expect(payload.thinking).toEqual({ type: "adaptive", display: "summarized" }); - expect(payload.context_management).toBeUndefined(); + expect(payload.context_management).toEqual({ + edits: [{ type: "clear_thinking_20251015", keep: "all" }], + }); expect(payload.output_config).toEqual({ effort: "xhigh" }); }); diff --git a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts index 1aa908bd8..6cd7369b2 100644 --- a/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts +++ b/packages/ai/test/anthropic-unsigned-thinking-replay.test.ts @@ -101,6 +101,29 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => { expect(blocks[1]).toEqual({ type: "text", text: "Sure." }); }); + it("sends context_management for API-key Anthropic-compatible thinking requests", async () => { + const { promise, resolve } = Promise.withResolvers(); + streamAnthropic( + makeModel(), + { systemPrompt: [], messages: [makeUser("continue")] }, + { + apiKey: "sk-ant-api-test", + signal: AbortSignal.abort(), + thinkingEnabled: true, + onPayload: payload => resolve(payload), + }, + ); + + const payload = (await promise) as { + thinking?: { type?: string }; + context_management?: { edits?: Array<{ type?: string; keep?: string }> }; + }; + expect(payload.thinking?.type).toBe("enabled"); + expect(payload.context_management).toEqual({ + edits: [{ type: "clear_thinking_20251015", keep: "all" }], + }); + }); + it("sanitizes lone surrogates in tool arguments regardless of origin API", () => { const loneSurrogate = "broken \ud83d end"; const makeToolCallAssistant = (api: AssistantMessage["api"]): AssistantMessage => ({ From 6b3c7ad3766cbe092a5786b3ebd1f080a32069c5 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 23 Jun 2026 01:35:42 +0000 Subject: [PATCH 33/43] fix(ai): advertised context-management beta for api-key thinking Anthropic rejects context_management.clear_thinking_20251015 without the context-management-2025-06-27 beta header. OAuth requests carried it via claudeCodeAgentBetaDefaults; API-key requests pushed the new context_management field without the beta, so the server rejected them. Push the beta into extraBetas alongside the field for every API-key thinking request, and exclude the GitHub Copilot proxy (which strips Anthropic betas and demotes thinking blocks upstream) from emitting the field. Fixes #3288 --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/providers/anthropic.ts | 28 ++++++++++++--- packages/ai/test/anthropic-alignment.test.ts | 35 +++++++++++++++++++ .../ai/test/github-copilot-reasoning.test.ts | 6 ++++ 4 files changed, 66 insertions(+), 5 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 6aad3230a..f167fe2d3 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) +- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. GitHub Copilot's Anthropic proxy is excluded because it strips Anthropic betas and demotes thinking blocks to text upstream. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) ## [16.1.15] - 2026-06-22 diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index f681f7f54..7543d0594 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -120,10 +120,11 @@ export function buildBetaHeader(baseBetas: readonly string[], extraBetas: readon } const midConversationSystemBeta = "mid-conversation-system-2026-04-07"; +const contextManagementBeta = "context-management-2025-06-27"; const claudeCodeUtilityBetaDefaults = [ "oauth-2025-04-20", "interleaved-thinking-2025-05-14", - "context-management-2025-06-27", + contextManagementBeta, "prompt-caching-scope-2026-01-05", "structured-outputs-2025-12-15", ] as const; @@ -131,7 +132,7 @@ const claudeCodeAgentBetaDefaults = [ "claude-code-20250219", "oauth-2025-04-20", "interleaved-thinking-2025-05-14", - "context-management-2025-06-27", + contextManagementBeta, "prompt-caching-scope-2026-01-05", midConversationSystemBeta, "advanced-tool-use-2025-11-20", @@ -1680,6 +1681,20 @@ const streamAnthropicOnce = ( // carry it in the Claude Code list). extraBetas.push(midConversationSystemBeta); } + // `context_management.clear_thinking_20251015` requires this beta. OAuth + // requests carry it in `claudeCodeAgentBetaDefaults`; API-key requests + // need it added explicitly so the field is honored instead of rejected + // (#3288). Skip Copilot — its proxy strips Anthropic betas and the + // upstream compat flag demotes thinking blocks to text, so there is + // nothing for `keep: "all"` to preserve. + if ( + model.reasoning && + options?.thinkingEnabled && + model.provider !== "github-copilot" && + !extraBetas.includes(contextManagementBeta) + ) { + extraBetas.push(contextManagementBeta); + } const created = createClient(model, { model, @@ -2950,8 +2965,13 @@ function buildParams( // strip the replayed thinking blocks `replayUnsignedThinking` puts back // on the wire, so the model loses the prior reasoning chain across turns // and the KV cache misses every turn (#3288). Narrowing this guard back - // to `isOAuthToken` regresses every API-key thinking provider. - const shouldKeepThinkingContext = thinking?.type === "adaptive" || thinking?.type === "enabled"; + // to `isOAuthToken` regresses every API-key thinking provider. Skip + // Copilot — its proxy strips Anthropic betas (so the required + // `context-management-2025-06-27` header never lands) and the compat + // flag demotes thinking blocks to text, so `keep: "all"` is a no-op + // that risks the proxy rejecting an unrecognized field. + const shouldKeepThinkingContext = + model.provider !== "github-copilot" && (thinking?.type === "adaptive" || thinking?.type === "enabled"); const contextManagement = shouldKeepThinkingContext ? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] } : undefined; diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index 47a2714fc..f4efa80f7 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -406,6 +406,41 @@ describe("Anthropic request fingerprint alignment", () => { expect(capturedBeta).toContain("mid-conversation-system-2026-04-07"); }); + it("adds the context-management beta to API-key thinking requests", async () => { + let capturedBeta: string | undefined; + const fetchMock = (async (_input: string | URL | Request, init?: RequestInit) => { + capturedBeta = (init?.headers as Record | undefined)?.["anthropic-beta"]; + return new Response( + JSON.stringify({ type: "error", error: { type: "invalid_request_error", message: "captured" } }), + { status: 400, headers: { "Content-Type": "application/json" } }, + ); + }) as typeof fetch; + + // `context_management.clear_thinking_20251015` is rejected without + // the `context-management-2025-06-27` beta. OAuth requests carry it + // via `claudeCodeAgentBetaDefaults`; API-key requests must add it + // explicitly whenever thinking is enabled so the field is honored + // instead of dropped on the floor (#3288). + await streamAnthropic( + ANTHROPIC_MODEL, + { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }] }, + { apiKey: "sk-ant-api-test", thinkingEnabled: true, fetch: fetchMock }, + ).result(); + + expect(capturedBeta).toContain("context-management-2025-06-27"); + + capturedBeta = undefined; + await streamAnthropic( + ANTHROPIC_MODEL, + { systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }] }, + { apiKey: "sk-ant-api-test", thinkingEnabled: false, fetch: fetchMock }, + ).result(); + + // No context_management field is sent when thinking is disabled, so the + // beta MUST NOT be advertised either. + expect(capturedBeta ?? "").not.toContain("context-management-2025-06-27"); + }); + it("billing-header fingerprint uses first user message, not leading developer message", async () => { const userText = "Hello from user with enough chars padding here"; diff --git a/packages/ai/test/github-copilot-reasoning.test.ts b/packages/ai/test/github-copilot-reasoning.test.ts index 6360c40dd..e96a2c710 100644 --- a/packages/ai/test/github-copilot-reasoning.test.ts +++ b/packages/ai/test/github-copilot-reasoning.test.ts @@ -56,9 +56,15 @@ describe("GitHub Copilot reasoning request construction", () => { const payload = (await captureAnthropicPayload(model)) as { thinking?: { type?: string }; output_config?: { effort?: string }; + context_management?: unknown; }; expect(payload.thinking).toEqual({ type: "adaptive" }); expect(payload.output_config).toEqual({ effort: "high" }); + // The Copilot Anthropic proxy strips Anthropic betas and demotes + // thinking blocks to text upstream — the `context_management` field + // would have no replayed thinking to keep and risks proxy rejection + // of an unrecognized field. The field MUST NOT be sent (#3288). + expect(payload.context_management).toBeUndefined(); }); }); From 5291b2f5a07f81e64a8312f00c7c7e2f04a1e02a Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 23 Jun 2026 01:42:53 +0000 Subject: [PATCH 34/43] fix(ai): skipped context management for injected clients Injected Anthropic clients bypass buildAnthropicClientOptions, so this package cannot add the context-management beta header their SDK instance would need before accepting context_management.clear_thinking_20251015. Omit context_management for options.client requests while preserving thinking itself, and add regression coverage for injected-client payload shaping. Fixes #3288 --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/providers/anthropic.ts | 13 ++++---- .../ai/test/anthropic-stream-envelope.test.ts | 30 +++++++++++++++++++ 3 files changed, 39 insertions(+), 6 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index f167fe2d3..bd038ba1a 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. GitHub Copilot's Anthropic proxy is excluded because it strips Anthropic betas and demotes thinking blocks to text upstream. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) +- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. Injected SDK clients and GitHub Copilot's Anthropic proxy are excluded because this code path cannot add the beta to caller-owned clients, while Copilot strips Anthropic betas and demotes thinking blocks to text upstream. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) ## [16.1.15] - 2026-06-22 diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 7543d0594..8300316a2 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -2966,12 +2966,15 @@ function buildParams( // on the wire, so the model loses the prior reasoning chain across turns // and the KV cache misses every turn (#3288). Narrowing this guard back // to `isOAuthToken` regresses every API-key thinking provider. Skip - // Copilot — its proxy strips Anthropic betas (so the required - // `context-management-2025-06-27` header never lands) and the compat - // flag demotes thinking blocks to text, so `keep: "all"` is a no-op - // that risks the proxy rejecting an unrecognized field. + // injected clients because this code cannot add the required + // `context-management-2025-06-27` beta to caller-owned SDK clients. Skip + // Copilot because its proxy strips Anthropic betas and demotes thinking + // blocks to text upstream, so `keep: "all"` is a no-op that risks proxy + // rejection of an unrecognized field. const shouldKeepThinkingContext = - model.provider !== "github-copilot" && (thinking?.type === "adaptive" || thinking?.type === "enabled"); + !options?.client && + model.provider !== "github-copilot" && + (thinking?.type === "adaptive" || thinking?.type === "enabled"); const contextManagement = shouldKeepThinkingContext ? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] } : undefined; diff --git a/packages/ai/test/anthropic-stream-envelope.test.ts b/packages/ai/test/anthropic-stream-envelope.test.ts index f73748009..27cee8e5c 100644 --- a/packages/ai/test/anthropic-stream-envelope.test.ts +++ b/packages/ai/test/anthropic-stream-envelope.test.ts @@ -576,6 +576,36 @@ describe("anthropic stream envelope handling", () => { expect(capturedParams?.tools?.map(tool => tool.name)).toEqual(["web_search"]); expect(capturedOptions?.headers).toEqual({ "X-Umans-Websearch-Provider": "exa" }); }); + + it("does not send context_management through injected clients", async () => { + type CapturedPayload = { + thinking?: { type?: string }; + context_management?: unknown; + }; + let capturedParams: CapturedPayload | undefined; + const client: AnthropicMessagesClientLike = { + messages: { + create(params) { + capturedParams = params as CapturedPayload; + return createMockRequest(createTextSuccessEvents("done")); + }, + }, + }; + + const stream = streamAnthropic(model, context, { + client, + thinkingEnabled: true, + }); + const events: AssistantMessageEvent[] = []; + for await (const event of stream) { + events.push(event); + } + const result = await stream.result(); + + expect(result.content).toEqual([{ type: "text", text: "done" }]); + expect(capturedParams?.thinking?.type).toBe("enabled"); + expect(capturedParams?.context_management).toBeUndefined(); + }); it("unwraps thinking blocks that Anthropic streams with literal thinking tags", async () => { const wrappedThinking = "\n\nCheck logs before accepting container health.\n"; From c04fa895694203a961c323e364365c2615a03633 Mon Sep 17 00:00:00 2001 From: roboomp Date: Tue, 23 Jun 2026 01:54:33 +0000 Subject: [PATCH 35/43] fix(ai): skipped context management for vertex rawpredict Vertex Claude rawPredict expects Anthropic beta flags in the JSON body as anthropic_beta. The Anthropic stream path can only add context-management-2025-06-27 as an HTTP header there, so sending context_management would make reasoning requests fail. Omit context_management and its beta header for google-vertex Anthropic models while preserving thinking itself, and add regression coverage to the rawPredict routing test. Fixes #3288 --- packages/ai/CHANGELOG.md | 2 +- packages/ai/src/providers/anthropic.ts | 13 +++++++++---- packages/ai/test/stream.test.ts | 12 +++++++++++- 3 files changed, 21 insertions(+), 6 deletions(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index bd038ba1a..329d36472 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -4,7 +4,7 @@ ### Fixed -- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. Injected SDK clients and GitHub Copilot's Anthropic proxy are excluded because this code path cannot add the beta to caller-owned clients, while Copilot strips Anthropic betas and demotes thinking blocks to text upstream. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) +- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. Injected SDK clients, GitHub Copilot's Anthropic proxy, and Vertex rawPredict are excluded because this code path cannot add the beta to caller-owned clients, Copilot strips Anthropic betas and demotes thinking blocks to text upstream, and Vertex expects betas in the JSON body rather than the Anthropic HTTP beta header. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) ## [16.1.15] - 2026-06-22 diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 8300316a2..ab5974f3a 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -1684,13 +1684,15 @@ const streamAnthropicOnce = ( // `context_management.clear_thinking_20251015` requires this beta. OAuth // requests carry it in `claudeCodeAgentBetaDefaults`; API-key requests // need it added explicitly so the field is honored instead of rejected - // (#3288). Skip Copilot — its proxy strips Anthropic betas and the - // upstream compat flag demotes thinking blocks to text, so there is - // nothing for `keep: "all"` to preserve. + // (#3288). Skip transports where this package cannot deliver the beta + // in the form their adapter accepts: Copilot strips Anthropic betas, + // and Vertex rawPredict needs betas in the body (`anthropic_beta`), + // not as an `anthropic-beta` HTTP header. if ( model.reasoning && options?.thinkingEnabled && model.provider !== "github-copilot" && + model.provider !== "google-vertex" && !extraBetas.includes(contextManagementBeta) ) { extraBetas.push(contextManagementBeta); @@ -2970,10 +2972,13 @@ function buildParams( // `context-management-2025-06-27` beta to caller-owned SDK clients. Skip // Copilot because its proxy strips Anthropic betas and demotes thinking // blocks to text upstream, so `keep: "all"` is a no-op that risks proxy - // rejection of an unrecognized field. + // rejection of an unrecognized field. Skip Vertex rawPredict because that + // adapter requires betas in the JSON body (`anthropic_beta`) instead of the + // Anthropic HTTP beta header this code can add. const shouldKeepThinkingContext = !options?.client && model.provider !== "github-copilot" && + model.provider !== "google-vertex" && (thinking?.type === "adaptive" || thinking?.type === "enabled"); const contextManagement = shouldKeepThinkingContext ? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] } diff --git a/packages/ai/test/stream.test.ts b/packages/ai/test/stream.test.ts index bdf1a7218..c3891d1a5 100644 --- a/packages/ai/test/stream.test.ts +++ b/packages/ai/test/stream.test.ts @@ -580,7 +580,12 @@ describe("Generate E2E Tests", () => { contextWindow: 200_000, maxTokens: 64_000, }); - const captured = Promise.withResolvers<{ url: string; authorization: string | null; body: unknown }>(); + const captured = Promise.withResolvers<{ + url: string; + authorization: string | null; + betaHeader: string | null; + body: unknown; + }>(); try { __resetVertexTokenCache(); @@ -598,6 +603,7 @@ describe("Generate E2E Tests", () => { { messages: [{ role: "user", content: "Hello", timestamp: Date.now() }] }, { apiKey: "", + thinkingEnabled: true, fetch: async (input, init) => { const url = input instanceof Request ? input.url : input.toString(); if ( @@ -611,6 +617,7 @@ describe("Generate E2E Tests", () => { captured.resolve({ url, authorization: headers.get("authorization"), + betaHeader: headers.get("anthropic-beta"), body: JSON.parse(bodyText), }); return new Response(JSON.stringify({ error: { message: "stop after capture" } }), { status: 400 }); @@ -632,6 +639,9 @@ describe("Generate E2E Tests", () => { stream: true, }); expect((request.body as Record).model).toBeUndefined(); + expect((request.body as Record).thinking?.type).toBe("enabled"); + expect((request.body as Record).context_management).toBeUndefined(); + expect(request.betaHeader ?? "").not.toContain("context-management-2025-06-27"); } finally { __resetVertexTokenCache(); homedirSpy.mockRestore(); From 26443eefacce86a635fd3fc2dc7e7512e090ab05 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 05:15:08 +0200 Subject: [PATCH 36/43] fix(coding-agent): terminated process on cancelled startup resume picker - Force process exit when the startup session picker is cancelled instead of returning. - Prevent hanging the event loop caused by long-lived startup handles such as theme listeners and timers. - Add regression test case to verify clean process termination upon picker cancellation. --- packages/coding-agent/src/main.ts | 19 ++++- .../test/main-resume-cancel-exit.test.ts | 84 +++++++++++++++++++ 2 files changed, 99 insertions(+), 4 deletions(-) create mode 100644 packages/coding-agent/test/main-resume-cancel-exit.test.ts diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index e8a09f817..a9d9b893b 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -945,6 +945,7 @@ async function buildSessionOptions( interface RunRootCommandDependencies { createAgentSession?: typeof createAgentSession; discoverAuthStorage?: typeof discoverAuthStorage; + selectSession?: typeof selectSession; runAcpMode?: RunAcpMode; settings?: Settings; forceSetupWizard?: boolean; @@ -1131,7 +1132,8 @@ export async function runRootCommand( // (see issue #1668). if (typeof parsedArgs.resume === "string" && !sessionManager) { writeStartupNotice(parsedArgs, `${chalk.dim("Resume cancelled: session is in another project.")}\n`); - return; + stopStartupWatchdog(); + process.exit(0); } // Handle --resume (no value): show session picker @@ -1147,17 +1149,26 @@ export async function runRootCommand( preloadedAllSessions = await logger.time("SessionManager.listAll", SessionManager.listAll); if (preloadedAllSessions.length === 0) { writeStartupNotice(parsedArgs, `${chalk.dim("No sessions found")}\n`); - return; + stopStartupWatchdog(); + process.exit(0); } } pauseStartupWatchdog(); - const selected = await logger.time("selectSession", selectSession, folderSessions, { + const selected = await logger.time("selectSession", deps.selectSession ?? selectSession, folderSessions, { allSessions: preloadedAllSessions, }); resumeStartupWatchdog(); if (!selected) { writeStartupNotice(parsedArgs, `${chalk.dim("No session selected")}\n`); - return; + // Quit instead of returning: startup already armed long-lived handles + // (theme watcher + SIGWINCH/macOS appearance listeners via initTheme, + // settings save timer, model registry) that keep the event loop alive, + // so a bare return hangs the process after the picker leaves the alt + // screen. No session was built here, so there is nothing to flush. The + // in-session `/resume` picker (selector-controller.ts) takes a different + // onCancel that just closes the overlay — only this startup path exits. + stopStartupWatchdog(); + process.exit(0); } // Resuming a session from another project: switch the process into that // project's directory and refresh cwd-derived caches before the session is diff --git a/packages/coding-agent/test/main-resume-cancel-exit.test.ts b/packages/coding-agent/test/main-resume-cancel-exit.test.ts new file mode 100644 index 000000000..c1884f81f --- /dev/null +++ b/packages/coding-agent/test/main-resume-cancel-exit.test.ts @@ -0,0 +1,84 @@ +/** + * Regression: cancelling the startup `--resume` session picker (e.g. pressing + * Esc) must terminate the process cleanly. Startup arms long-lived handles + * (theme/appearance listeners via initTheme, settings save timer, model + * registry), so the previous bare `return` left the event loop with live + * handles and the process hung after the picker left the alternate screen. + * + * The fix exits via `process.exit(0)` — matching the `--version`/`--export` + * early-exit convention in the same function. Only this startup call site + * exits; the in-session `/resume` picker (selector-controller.ts) keeps its own + * onCancel that just closes the overlay. + */ +import { describe, expect, it, vi } from "bun:test"; +import * as path from "node:path"; +import { parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { runRootCommand } from "@oh-my-pi/pi-coding-agent/main"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { TempDir } from "@oh-my-pi/pi-utils"; + +class ProcessExitSignal extends Error { + constructor(readonly code: number) { + super(`process.exit(${code})`); + this.name = "ProcessExitSignal"; + } +} + +describe("runRootCommand — startup --resume picker cancellation", () => { + it("exits cleanly (process.exit 0) when the picker is cancelled instead of returning and hanging", async () => { + using tempDir = TempDir.createSync("@omp-resume-cancel-"); + const sessionDir = tempDir.path(); + // One valid session so folderSessions is non-empty and the picker (not the + // "No sessions found" probe) is the path under test. + await Bun.write( + path.join(sessionDir, "existing.jsonl"), + `${JSON.stringify({ type: "session", id: "existing-session", cwd: sessionDir, timestamp: new Date().toISOString() })}\n`, + ); + + const authStorage = await AuthStorage.create(path.join(sessionDir, "auth.db")); + const settings = Settings.isolated({ "marketplace.autoUpdate": "off" }); + + // --print keeps initTheme non-interactive so no global appearance/SIGWINCH + // listeners leak into the rest of the suite; the picker branch is gated on + // `resume === true`, not on interactivity, so it still runs. + const parsed = parseArgs(["--resume", "--print"]); + parsed.noExtensions = true; + parsed.noSkills = true; + parsed.noRules = true; + parsed.noTools = true; + parsed.noLsp = true; + parsed.sessionDir = sessionDir; + + const exitCodes: number[] = []; + vi.spyOn(process, "exit").mockImplementation(((code?: number) => { + exitCodes.push(code ?? 0); + throw new ProcessExitSignal(code ?? 0); + }) as typeof process.exit); + vi.spyOn(process.stdout, "write").mockImplementation(() => true); + + let pickerCalled = false; + let thrown: unknown; + try { + await runRootCommand(parsed, ["--resume", "--print"], { + discoverAuthStorage: async () => authStorage, + settings, + selectSession: async () => { + pickerCalled = true; + return null; // user cancelled (Esc) + }, + }); + } catch (err) { + thrown = err; + } finally { + vi.restoreAllMocks(); + authStorage.close(); + } + + expect(pickerCalled).toBe(true); + expect(thrown).toBeInstanceOf(ProcessExitSignal); + // Exactly one clean exit — proves the cancel branch terminates instead of + // falling through to session creation or returning into a hang. + expect(exitCodes).toEqual([0]); + }, 15_000); +}); From c311c30cad7cb728333859685c4946e03805f09d Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 05:16:14 +0200 Subject: [PATCH 37/43] fix(coding-agent): resolved local image protocol and process handling - Standardized `local://` image processing to prevent file corruption during decoding. - Refactored local path resolution logic to enforce safety constraints and path containment. - Implemented an image fast-path in `ReadTool` to correctly render local images before text decoding. - Added comprehensive test coverage for image rendering, text compatibility, and path security. - Resolved an event loop hang associated with `omp --resume` operations. --- packages/coding-agent/CHANGELOG.md | 3 +- .../src/internal-urls/local-protocol.ts | 153 +++++++++----- packages/coding-agent/src/tools/read.ts | 196 ++++++++++++------ .../test/tools/read-local-image.test.ts | 106 ++++++++++ 4 files changed, 344 insertions(+), 114 deletions(-) create mode 100644 packages/coding-agent/test/tools/read-local-image.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 29a47fa41..a50ea4dcd 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,7 +1,6 @@ # Changelog ## [Unreleased] - ### Breaking Changes - Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. @@ -32,6 +31,8 @@ ### Fixed +- Fixed `local://` URLs decoding images as corrupted text (mojibake) instead of showing the image +- Fixed `omp --resume` hanging instead of exiting when the startup session picker is cancelled (Esc) or there are no sessions to resume. Startup arms long-lived handles (theme/appearance listeners, settings save timer, model registry), so the cancel/empty paths' bare `return` left the event loop alive and the process stuck after the picker cleared the alternate screen. These paths now exit cleanly via `process.exit(0)`, matching the `--version`/`--export` early-exit convention. The in-session `/resume` picker is unaffected — it keeps its own cancel handler that just closes the overlay. - Fixed the `/resume` session picker scrolling down after a session is deleted. The delete-confirmation dialog was mounted as a sibling below the picker's bottom border, briefly growing the picker past the terminal height; the TUI committed the picker's header rows into native scrollback to fit, and when the dialog closed `windowTop` stayed pinned at the new commit boundary, leaving the header stranded above the viewport. The picker now hosts the `SessionList` in a single content slot and swaps the dialog INTO that slot (replacing the `SessionList`) while it is open, so the dialog only competes with the `SessionList`'s rendered budget — not the `SessionList` AND the picker chrome — and the picker frame stays inside the viewport. ([#3283](https://github.com/can1357/oh-my-pi/issues/3283)) - Fixed the `eval` tool card not streaming a still-running cell's stdout: a long-running cell (e.g. a `time.sleep()` monitor loop) showed nothing until it returned or was interrupted, then dumped everything at once. The renderer draws cell output from `details.cells[i].output`, which was only populated after `backend.execute()` resolved — live stdout streamed into the transient result `content` tail (and `renderContext.output`), which the per-cell render branch ignores. Streamed chunks now append to the active cell's `output` (a dedicated per-cell tail buffer, capped like the aggregate) as they arrive, so the card shows progress live; on completion the authoritative full output overwrites the live tail. `log()`/`phase()`/`display()` and status ops were unaffected because they already stream via the status channel. - Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path. diff --git a/packages/coding-agent/src/internal-urls/local-protocol.ts b/packages/coding-agent/src/internal-urls/local-protocol.ts index 0fe00cf0f..055891240 100644 --- a/packages/coding-agent/src/internal-urls/local-protocol.ts +++ b/packages/coding-agent/src/internal-urls/local-protocol.ts @@ -169,6 +169,96 @@ export function buildEvalUrlRoots(options: LocalProtocolOptions): Record { + const localRoot = path.resolve(resolveLocalRoot(opts)); + await fs.mkdir(localRoot, { recursive: true }); + + let resolvedRoot: string; + try { + resolvedRoot = await fs.realpath(localRoot); + } catch (error) { + if (isEnoent(error)) { + throw new Error("Unable to initialize local:// root"); + } + throw error; + } + + const relativePath = extractRelativePath(url); + const targetPath = relativePath ? path.resolve(resolvedRoot, relativePath) : resolvedRoot; + ensureWithinRoot(targetPath, resolvedRoot); + + if (targetPath === resolvedRoot) { + return { kind: "listing", root: resolvedRoot }; + } + + const parentDir = path.dirname(targetPath); + try { + const realParent = await fs.realpath(parentDir); + ensureWithinRoot(realParent, resolvedRoot); + } catch (error) { + if (!isEnoent(error)) throw error; + } + + let realTargetPath: string; + try { + realTargetPath = await fs.realpath(targetPath); + } catch (error) { + if (isEnoent(error)) { + throw new Error(`Local file not found: ${url.href}`); + } + throw error; + } + + ensureWithinRoot(realTargetPath, resolvedRoot); + + const stat = await fs.stat(realTargetPath); + if (stat.isDirectory()) { + return { kind: "directory", path: realTargetPath }; + } + if (!stat.isFile()) { + throw new Error(`local:// URL must resolve to a file or directory: ${url.href}`); + } + return { kind: "file", path: realTargetPath, size: stat.size }; +} + +/** + * Resolve a local:// URL to a regular on-disk file, applying the same + * realpath + containment guarantees as {@link LocalProtocolHandler.resolve} + * but WITHOUT reading or UTF-8-decoding its contents. Returns null when there + * is no active session or when the URL targets the root listing or a directory; + * throws the handler's not-found and "escapes local root" errors for missing + * files and symlink escapes. + * + * Options are resolved via {@link LocalProtocolHandler.resolveOptions} so the + * caller-options → override → registry order matches router resolution exactly. + * The read tool uses this to detect and emit image files from their real path + * before the text-only resource contract would decode the binary into mojibake. + */ +export async function resolveLocalUrlToFile( + input: string | InternalUrl, + context?: ResolveContext, +): Promise<{ path: string; size: number } | null> { + const opts = LocalProtocolHandler.resolveOptions(context); + if (!opts) return null; + const url = typeof input === "string" ? parseLocalUrl(input) : input; + const resolved = await resolveLocalTarget(url, opts); + return resolved.kind === "file" ? { path: resolved.path, size: resolved.size } : null; +} + /** * Protocol handler for local:// URLs. * @@ -238,65 +328,22 @@ export class LocalProtocolHandler implements ProtocolHandler { throw new Error("No session - local:// unavailable"); } - const localRoot = path.resolve(resolveLocalRoot(opts)); - await fs.mkdir(localRoot, { recursive: true }); - - let resolvedRoot: string; - try { - resolvedRoot = await fs.realpath(localRoot); - } catch (error) { - if (isEnoent(error)) { - throw new Error("Unable to initialize local:// root"); - } - throw error; + const resolved = await resolveLocalTarget(url, opts); + if (resolved.kind === "listing") { + return buildListing(url, resolved.root); + } + if (resolved.kind === "directory") { + return buildDirectoryResource(url.href, resolved.path, [LOCAL_WRITE_NOTE]); } - const relativePath = extractRelativePath(url); - const targetPath = relativePath ? path.resolve(resolvedRoot, relativePath) : resolvedRoot; - ensureWithinRoot(targetPath, resolvedRoot); - - if (targetPath === resolvedRoot) { - return buildListing(url, resolvedRoot); - } - - const parentDir = path.dirname(targetPath); - try { - const realParent = await fs.realpath(parentDir); - ensureWithinRoot(realParent, resolvedRoot); - } catch (error) { - if (!isEnoent(error)) throw error; - } - - let realTargetPath: string; - try { - realTargetPath = await fs.realpath(targetPath); - } catch (error) { - if (isEnoent(error)) { - throw new Error(`Local file not found: ${url.href}`); - } - throw error; - } - - ensureWithinRoot(realTargetPath, resolvedRoot); - - const stat = await fs.stat(realTargetPath); - if (stat.isDirectory()) { - return buildDirectoryResource(url.href, realTargetPath, [ - "Use write path local:// to persist large intermediate artifacts across turns.", - ]); - } - if (!stat.isFile()) { - throw new Error(`local:// URL must resolve to a file or directory: ${url.href}`); - } - - const content = await Bun.file(realTargetPath).text(); + const content = await Bun.file(resolved.path).text(); return { url: url.href, content, - contentType: getContentType(realTargetPath), + contentType: getContentType(resolved.path), size: Buffer.byteLength(content, "utf-8"), - sourcePath: realTargetPath, - notes: ["Use write path local:// to persist large intermediate artifacts across turns."], + sourcePath: resolved.path, + notes: [LOCAL_WRITE_NOTE], }; } diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index 30791e8f1..3bc5a32da 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -8,7 +8,7 @@ import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; import { glob, type SummaryResult, summarizeCode } from "@oh-my-pi/pi-natives"; import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; -import { getRemoteDir, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils"; +import { getRemoteDir, type ImageMetadata, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils"; import { type } from "arktype"; import { LRUCache } from "lru-cache/raw"; import { @@ -22,7 +22,7 @@ import { import { normalizeToLF } from "../edit/normalize"; import { isNotebookPath, readEditableNotebookText } from "../edit/notebook"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; -import { InternalUrlRouter } from "../internal-urls"; +import { InternalUrlRouter, resolveLocalUrlToFile } from "../internal-urls"; import { parseInternalUrl } from "../internal-urls/parse"; import type { InternalUrl } from "../internal-urls/types"; import { getLanguageFromPath, type Theme } from "../modes/theme/theme"; @@ -1112,6 +1112,79 @@ export class ReadTool implements AgentTool { .done(); } + /** + * Build content blocks for an on-disk image file: an `inspect_image` + * metadata note when inspection is enabled, otherwise the decoded image + * block. Shared by the plain-file read path and the `local://` image fast + * path so both honor `inspect_image.enabled`, the size cap, and auto-resize + * identically. Too-large / unsupported images surface as {@link ToolError}. + */ + async #loadImageContent(options: { + readPath: string; + absolutePath: string; + mimeType: string; + imageMetadata: ImageMetadata | null; + fileSize: number; + }): Promise<{ content: Array; details: ReadToolDetails; sourcePath: string }> { + const { readPath, absolutePath, mimeType, imageMetadata, fileSize } = options; + if (this.#inspectImageEnabled) { + const outputMime = imageMetadata?.mimeType ?? mimeType; + const metadataLines = [ + "Image metadata:", + `- MIME: ${outputMime}`, + `- Bytes: ${fileSize} (${formatBytes(fileSize)})`, + imageMetadata?.width !== undefined && imageMetadata.height !== undefined + ? `- Dimensions: ${imageMetadata.width}x${imageMetadata.height}` + : "- Dimensions: unknown", + imageMetadata?.channels !== undefined ? `- Channels: ${imageMetadata.channels}` : "- Channels: unknown", + imageMetadata?.hasAlpha === true + ? "- Alpha: yes" + : imageMetadata?.hasAlpha === false + ? "- Alpha: no" + : "- Alpha: unknown", + "", + `If you want to analyze the image, call inspect_image with path="${formatPathRelativeToCwd( + absolutePath, + this.session.cwd, + )}" and a question describing what to inspect and the desired output format.`, + ]; + return { content: [{ type: "text", text: metadataLines.join("\n") }], details: {}, sourcePath: absolutePath }; + } + + if (fileSize > MAX_IMAGE_SIZE) { + const sizeStr = formatBytes(fileSize); + const maxStr = formatBytes(MAX_IMAGE_SIZE); + throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`); + } + try { + const imageInput = await loadImageInput({ + path: readPath, + cwd: this.session.cwd, + autoResize: this.#autoResizeImages, + maxBytes: MAX_IMAGE_SIZE, + resolvedPath: absolutePath, + detectedMimeType: mimeType, + excludeWebP: webpExclusionForModel(this.session.getActiveModel?.()), + }); + if (!imageInput) { + throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`); + } + return { + content: [ + { type: "text", text: imageInput.textNote }, + { type: "image", data: imageInput.data, mimeType: imageInput.mimeType }, + ], + details: {}, + sourcePath: imageInput.resolvedPath, + }; + } catch (error) { + if (error instanceof ImageInputTooLargeError) { + throw new ToolError(error.message); + } + throw error; + } + } + #buildInMemoryTextResult( text: string, offset: number | undefined, @@ -2132,64 +2205,13 @@ export class ReadTool implements AgentTool { | undefined; if (mimeType) { - if (this.#inspectImageEnabled) { - const metadata = imageMetadata; - const outputMime = metadata?.mimeType ?? mimeType; - const outputBytes = fileSize; - const metadataLines = [ - "Image metadata:", - `- MIME: ${outputMime}`, - `- Bytes: ${outputBytes} (${formatBytes(outputBytes)})`, - metadata?.width !== undefined && metadata.height !== undefined - ? `- Dimensions: ${metadata.width}x${metadata.height}` - : "- Dimensions: unknown", - metadata?.channels !== undefined ? `- Channels: ${metadata.channels}` : "- Channels: unknown", - metadata?.hasAlpha === true - ? "- Alpha: yes" - : metadata?.hasAlpha === false - ? "- Alpha: no" - : "- Alpha: unknown", - "", - `If you want to analyze the image, call inspect_image with path="${formatPathRelativeToCwd( - absolutePath, - this.session.cwd, - )}" and a question describing what to inspect and the desired output format.`, - ]; - content = [{ type: "text", text: metadataLines.join("\n") }]; - details = {}; - sourcePath = absolutePath; - } else { - if (fileSize > MAX_IMAGE_SIZE) { - const sizeStr = formatBytes(fileSize); - const maxStr = formatBytes(MAX_IMAGE_SIZE); - throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`); - } - try { - const imageInput = await loadImageInput({ - path: readPath, - cwd: this.session.cwd, - autoResize: this.#autoResizeImages, - maxBytes: MAX_IMAGE_SIZE, - resolvedPath: absolutePath, - detectedMimeType: mimeType, - excludeWebP: webpExclusionForModel(this.session.getActiveModel?.()), - }); - if (!imageInput) { - throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`); - } - content = [ - { type: "text", text: imageInput.textNote }, - { type: "image", data: imageInput.data, mimeType: imageInput.mimeType }, - ]; - details = {}; - sourcePath = imageInput.resolvedPath; - } catch (error) { - if (error instanceof ImageInputTooLargeError) { - throw new ToolError(error.message); - } - throw error; - } - } + ({ content, details, sourcePath } = await this.#loadImageContent({ + readPath, + absolutePath, + mimeType, + imageMetadata, + fileSize, + })); } else if (isNotebookPath(absolutePath) && !isRawSelector(parsed)) { const notebookText = await readEditableNotebookText(absolutePath, localReadPath); if (isMultiRange(parsed) && parsed.kind === "lines") { @@ -2727,6 +2749,17 @@ export class ReadTool implements AgentTool { hasExtraction = hasPathExtraction || hasQueryExtraction; } + // local:// files are real on-disk paths. Detect image files and emit a + // decoded image block before the text-only resource contract UTF-8 + // decodes the binary into mojibake. The fast path returns null for + // non-images, directories, listings, or any resolution failure, so the + // text path below reproduces the router's not-found / symlink-escape + // behavior unchanged. + if (scheme === "local") { + const imageResult = await this.#tryReadLocalImage(urlMeta, signal); + if (imageResult) return imageResult; + } + // Reject line selectors when query extraction is used if (hasExtraction && parsedSel.kind !== "none" && parsedSel.kind !== "raw") { throw new ToolError("Cannot combine query extraction with line selectors"); @@ -2770,6 +2803,49 @@ export class ReadTool implements AgentTool { }); } + /** + * Fast path for `local://` image files. Resolves the URL to its real + * on-disk path with the same realpath + containment checks as + * {@link LocalProtocolHandler.resolve} (via {@link resolveLocalUrlToFile}), + * and — only when the target is a genuine image — emits a decoded image + * block. Returns null for non-images, directories, listings, or any + * resolution failure (not-found, symlink escape) so the caller falls back to + * normal text resolution, which reproduces the router's errors. Errors from + * a confirmed image (too large / unsupported) propagate rather than + * degrading into a corrupted text read. + */ + async #tryReadLocalImage(url: InternalUrl, signal?: AbortSignal): Promise | null> { + let file: { path: string; size: number } | null; + try { + file = await resolveLocalUrlToFile(url, { + cwd: this.session.cwd, + settings: this.session.settings, + signal, + localProtocolOptions: this.session.localProtocolOptions, + }); + } catch { + // Not found / containment escape / no session — let the text path + // surface the router's canonical error. + return null; + } + if (!file) return null; + + const imageMetadata = await readImageMetadata(file.path); + const mimeType = imageMetadata?.mimeType; + if (!mimeType) return null; + + const { content, details, sourcePath } = await this.#loadImageContent({ + readPath: url.href, + absolutePath: file.path, + mimeType, + imageMetadata, + fileSize: file.size, + }); + const resultBuilder = toolResult(details).content(content).sourceInternal(url.href); + if (sourcePath) resultBuilder.sourcePath(sourcePath); + return resultBuilder.done(); + } + /** Read directory contents as a formatted listing */ async #readDirectory( absolutePath: string, diff --git a/packages/coding-agent/test/tools/read-local-image.test.ts b/packages/coding-agent/test/tools/read-local-image.test.ts new file mode 100644 index 000000000..7bf0d5d55 --- /dev/null +++ b/packages/coding-agent/test/tools/read-local-image.test.ts @@ -0,0 +1,106 @@ +/** + * `local://` is routed through the internal-URL handler, whose resource + * contract is text-only (`content: string`). Before the image fast path, a + * `local://photo.png` read UTF-8-decoded the PNG bytes into mojibake. These + * lock the fix: genuine image files under the session local root decode into an + * inline image block, text files still read as text, and a file symlinked + * outside the local root is rejected by the same realpath guard the router uses + * (the fast path must not become a containment bypass). + */ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { InternalUrlRouter, LocalProtocolHandler } from "@oh-my-pi/pi-coding-agent/internal-urls"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; + +// 1x1 transparent PNG — small enough to pass through image loading untouched. +const TINY_PNG = Buffer.from( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg==", + "base64", +); + +function makeSession(testDir: string): ToolSession { + const sessionFile = path.join(testDir, "session.jsonl"); + const artifactsDir = sessionFile.slice(0, -6); + return { + cwd: testDir, + hasUI: false, + getSessionFile: () => sessionFile, + getArtifactsDir: () => artifactsDir, + getSessionSpawns: () => null, + settings: Settings.isolated({ "images.autoResize": false }), + } as unknown as ToolSession; +} + +function joinText(content: Array<{ type: string; text?: string }>): string { + return content + .filter(c => c.type === "text") + .map(c => c.text ?? "") + .join("\n"); +} + +describe("read local:// images", () => { + let testDir: string; + let localRoot: string; + + beforeEach(async () => { + LocalProtocolHandler.resetOverrideForTests(); + InternalUrlRouter.resetForTests(); + testDir = await fs.mkdtemp(path.join(os.tmpdir(), "read-local-image-")); + const artifactsDir = path.join(testDir, "artifacts"); + localRoot = path.join(artifactsDir, "local"); + await fs.mkdir(localRoot, { recursive: true }); + LocalProtocolHandler.setOverride({ + getArtifactsDir: () => artifactsDir, + getSessionId: () => "session-local-image", + }); + }); + + afterEach(async () => { + LocalProtocolHandler.resetOverrideForTests(); + InternalUrlRouter.resetForTests(); + await fs.rm(testDir, { recursive: true, force: true }); + }); + + it("decodes a local:// PNG into an inline image block", async () => { + await Bun.write(path.join(localRoot, "clifford.png"), TINY_PNG); + const tool = new ReadTool(makeSession(testDir)); + + const result = await tool.execute("call", { path: "local://clifford.png" }); + + const image = result.content.find(c => c.type === "image"); + expect(image).toBeDefined(); + expect(image && "mimeType" in image ? image.mimeType : undefined).toBe("image/png"); + // The pre-fix bug surfaced the PNG signature byte (0x89) UTF-8-decoded to + // the replacement char; the fixed path must never emit it as text. + expect(joinText(result.content)).not.toContain("\uFFFDPNG"); + }); + + it("still reads a local:// text file as text (fast path falls through)", async () => { + await Bun.write(path.join(localRoot, "notes.txt"), "hello world"); + const tool = new ReadTool(makeSession(testDir)); + + const result = await tool.execute("call", { path: "local://notes.txt" }); + + expect(result.content.some(c => c.type === "image")).toBe(false); + expect(joinText(result.content)).toContain("hello world"); + }); + + it("does not read an image symlinked outside the local root", async () => { + if (process.platform === "win32") return; + const outsideDir = path.join(testDir, "outside"); + await fs.mkdir(outsideDir, { recursive: true }); + await Bun.write(path.join(outsideDir, "secret.png"), TINY_PNG); + await fs.symlink(outsideDir, path.join(localRoot, "linked")); + const tool = new ReadTool(makeSession(testDir)); + + // The realpath/containment guard the router applies must still reject the + // escape; the image fast path must not silently read it. + await expect(tool.execute("call", { path: "local://linked/secret.png" })).rejects.toThrow( + "local:// URL escapes local root", + ); + }); +}); From c6fcc155a29b3c7a9ccf02ccd5688a57e18e2e41 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 05:34:17 +0200 Subject: [PATCH 38/43] fix: nohup --- crates/pi-shell/src/shell.rs | 84 ++++++++++---- packages/coding-agent/scripts/build-binary.ts | 20 +++- packages/terminal-bench/agent/omp_local.py | 103 ++++++++++++------ packages/terminal-bench/src/runner.ts | 17 ++- 4 files changed, 165 insertions(+), 59 deletions(-) diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index c3943920c..ad3ffcf70 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -496,6 +496,39 @@ fn normalize_path_segment(segment: &str) -> String { normalized.to_string_lossy().to_ascii_lowercase() } +/// Check if a command is resolvable in the given PATH string. +/// Returns true if the command exists and is executable in one of the PATH +/// directories. +fn command_is_resolvable(command: &str, path: &str) -> bool { + for dir in std::env::split_paths(path) { + let full_path = dir.join(command); + #[cfg(unix)] + { + use std::os::unix::fs::PermissionsExt; + if full_path.exists() { + if let Ok(metadata) = full_path.metadata() { + let permissions = metadata.permissions(); + if permissions.mode() & 0o111 != 0 { + return true; + } + } + } + } + #[cfg(windows)] + { + if full_path.exists() { + // On Windows, .exe/.bat/.cmd extensions are automatically tried + for ext in ["", ".exe", ".bat", ".cmd"] { + let with_ext = dir.join(format!("{}{}", command, ext)); + if with_ext.exists() { + return true; + } + } + } + } + } + false +} #[cfg(not(windows))] fn merge_path_values(_existing: &str, incoming: &str) -> String { incoming.to_string() @@ -519,10 +552,6 @@ async fn create_session(config: &ShellConfig) -> Result { } shell.register_builtin("sleep", builtins::builtin::()); shell.register_builtin("timeout", builtins::builtin::()); - shell.register_builtin( - "nohup", - builtins::builtin::().transparent_background_wrapper(), - ); let mut merged_path: Option = None; for (key, value) in std::env::vars() { @@ -552,8 +581,8 @@ async fn create_session(config: &ShellConfig) -> Result { merged_path = Some(value.to_string_lossy().into_owned()); } - if let Some(path_value) = merged_path { - let mut var = ShellVariable::new(ShellValue::String(path_value)); + if let Some(path_value) = &merged_path { + let mut var = ShellVariable::new(ShellValue::String(path_value.clone())); var.export(); shell .env_mut() @@ -576,6 +605,27 @@ async fn create_session(config: &ShellConfig) -> Result { } } apply_env_fallback(&mut shell)?; + // The nohup builtin detaches its operand into a new session (see + // NohupCommand) so a backgrounded server survives this embedded shell's + // kill-on-drop teardown. It therefore shadows any system `nohup` (which does + // NOT escape the process-group kill) — unless explicitly opted out via + // PI_DISABLE_NOHUP_BUILTIN (session env or process env), in which case bare + // `nohup` resolves to the real coreutils binary. + let nohup_builtin_disabled = { + let raw = config + .session_env + .as_ref() + .and_then(|env| env.get("PI_DISABLE_NOHUP_BUILTIN").cloned()) + .or_else(|| std::env::var("PI_DISABLE_NOHUP_BUILTIN").ok()); + matches!(raw.as_deref(), Some(v) if !v.is_empty() && v != "0" && !v.eq_ignore_ascii_case("false")) + }; + let should_register_nohup = !nohup_builtin_disabled; + if should_register_nohup { + shell.register_builtin( + "nohup", + builtins::builtin::().transparent_background_wrapper(), + ); + } #[cfg(windows)] configure_windows_path(&mut shell)?; @@ -1916,18 +1966,13 @@ impl builtins::Command for NohupCommand { return Ok(ExecutionResult::new(125)); } - // Deliberately *not* nohup: we neither ignore SIGHUP nor detach the - // child into a new session. The command runs as an ordinary brush - // descendant so it is reaped together with the host instead of - // lingering as an orphan once the host process goes away. Agents - // reach for `nohup` assuming the shell is one-shot; in this - // persistent embedded shell that assumption is wrong and the only - // effect of real `nohup` would be to leak background processes. - // - // coreutils `nohup` additionally redirects stdin from /dev/null and - // stdout/stderr to `nohup.out`, but *only* when those streams are - // terminals. The embedded host always hands commands a pipe with a - // /dev/null stdin, so none of that redirection ever applies here. + // Detach the operand into a new session / process group (like `setsid`) + // so a backgrounded server survives this embedded shell's kill-on-drop + // teardown, which SIGKILLs the shell's own process group when the host + // process exits. Agents reach for `nohup &` expecting exactly + // this persistence; a real coreutils `nohup` would NOT help, since it + // stays in the shell's process group and dies with it. The new session + // is applied below via ProcessGroupPolicy::NewProcessGroup. let mut command_line = String::new(); for (idx, arg) in command.iter().enumerate() { if idx > 0 { @@ -1936,7 +1981,8 @@ impl builtins::Command for NohupCommand { command_line.push_str("e_arg(arg)); } - let params = context.params.clone(); + let mut params = context.params.clone(); + params.process_group_policy = ProcessGroupPolicy::NewProcessGroup; let source_info = SourceInfo::from("pi-natives:nohup"); context .shell diff --git a/packages/coding-agent/scripts/build-binary.ts b/packages/coding-agent/scripts/build-binary.ts index 7a5779b28..a4cfabeb1 100644 --- a/packages/coding-agent/scripts/build-binary.ts +++ b/packages/coding-agent/scripts/build-binary.ts @@ -5,7 +5,15 @@ import * as path from "node:path"; const packageDir = path.join(import.meta.dir, ".."); const repoRoot = path.join(packageDir, "..", ".."); -const outputPath = path.join(packageDir, "dist", "omp"); +// Optional cross-compile target, e.g. CROSS_TARGET=linux-arm64 → bun build +// --target=bun-linux-arm64, embeds the matching native, outputs dist/omp-. +const crossTarget = Bun.env.CROSS_TARGET || null; +const [crossPlatform, crossArch] = crossTarget ? crossTarget.split("-") : [null, null]; +// x64 uses the baseline bun runtime so it runs under Rosetta / pre-AVX2 CPUs +// (the modern bun-linux-x64 target SIGILLs under Apple-Silicon Rosetta). +const bunTarget = crossTarget ? (crossTarget === "linux-x64" ? "bun-linux-x64-baseline" : `bun-${crossTarget}`) : null; +const outName = crossTarget ? `omp-${crossTarget}` : "omp"; +const outputPath = path.join(packageDir, "dist", outName); // Transformers.js is an optional, native-heavy dependency that is never bundled // into the binary; the tiny-model worker `bun install`s it into a runtime cache @@ -17,7 +25,7 @@ const transformersVersion = ( ).version; function shouldAdhocSignDarwinBinary(): boolean { - return process.platform === "darwin"; + return process.platform === "darwin" && !crossTarget; } async function runCommand( @@ -43,7 +51,10 @@ async function main(): Promise { try { await runCommand(["bun", "--cwd=../stats", "scripts/generate-client-bundle.ts", "--generate"]); await runCommand(["bun", "scripts/generate-docs-index.ts", "--generate"]); - await runCommand(["bun", "--cwd=../natives", "run", "embed:native"]); + await runCommand( + ["bun", "--cwd=../natives", "run", "embed:native"], + crossTarget ? { ...Bun.env, TARGET_PLATFORM: crossPlatform as string, TARGET_ARCH: crossArch as string } : Bun.env, + ); await runCommand(["bun", "scripts/embed-mupdf-wasm.ts", "--generate"]); try { const buildEnv = shouldAdhocSignDarwinBinary() ? { ...Bun.env, BUN_NO_CODESIGN_MACHO_BINARY: "1" } : Bun.env; @@ -52,6 +63,7 @@ async function main(): Promise { "bun", "build", "--compile", + ...(bunTarget ? ["--target", bunTarget] : []), "--no-compile-autoload-bunfig", "--no-compile-autoload-dotenv", "--no-compile-autoload-tsconfig", @@ -85,7 +97,7 @@ async function main(): Promise { "./packages/coding-agent/src/extensibility/legacy-pi-ai-shim.ts", "./packages/coding-agent/src/extensibility/legacy-pi-coding-agent-shim.ts", "--outfile", - "packages/coding-agent/dist/omp", + `packages/coding-agent/dist/${outName}`, ], buildEnv, repoRoot, diff --git a/packages/terminal-bench/agent/omp_local.py b/packages/terminal-bench/agent/omp_local.py index 65f640c32..b647221f1 100644 --- a/packages/terminal-bench/agent/omp_local.py +++ b/packages/terminal-bench/agent/omp_local.py @@ -195,6 +195,9 @@ class OmpLocal(BaseInstalledAgent): self._home = "/root" self._bun = "/root/.bun/bin/bun" self._cli = "/root/.omp-bench/app/dist/cli.js" + self._binary_arm64 = _env("OMP_TB_BINARY_ARM64") + self._binary_x64 = _env("OMP_TB_BINARY_X64") + self._binary = bool(self._binary_arm64 or self._binary_x64) @staticmethod @override @@ -207,6 +210,8 @@ class OmpLocal(BaseInstalledAgent): @override def get_version_command(self) -> str | None: + if self._binary: + return f"{shlex.quote(self._cli)} --version" return self._wrap(f"{shlex.quote(self._bun)} {shlex.quote(self._cli)} --version") @override @@ -229,41 +234,44 @@ class OmpLocal(BaseInstalledAgent): @override async def install(self, environment: BaseEnvironment) -> None: - # 1) System deps (root). curl+unzip for the Bun installer; ca-certs for TLS. - await self.exec_as_root( - environment, - command=( - "set -e; " - "if command -v apt-get >/dev/null 2>&1; then " - " apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y curl unzip ca-certificates tar; " - "elif command -v apk >/dev/null 2>&1; then " - " echo 'ERROR: Alpine/musl base image; @oh-my-pi/pi-natives ships no musl prebuilt' >&2; exit 3; " - "elif command -v dnf >/dev/null 2>&1; then dnf install -y curl unzip tar; " - "elif command -v yum >/dev/null 2>&1; then yum install -y curl unzip tar; " - "fi" - ), - ) - - # Resolve the agent user's HOME (root vs non-root tasks differ). + # Resolve the agent user's HOME first (root vs non-root tasks differ). home = (await self.exec_as_agent(environment, command='printf %s "$HOME"')).stdout self._home = (home or "/root").strip() or "/root" - # 2) Bun (agent user). - await self.exec_as_agent( - environment, - command=( - "set -e; " - f"export BUN_INSTALL={shlex.quote(self._home + '/.bun')}; " - f'curl -fsSL https://bun.sh/install | bash -s "bun-v{self._bun_version}"; ' - f'{shlex.quote(self._home + "/.bun/bin/bun")} --version' - ), - ) - self._bun = f"{self._home}/.bun/bin/bun" - - if self._install_mode == "published": - self._cli = await self._install_published(environment) + if self._binary: + # Self-contained binary mode: upload + chmod only. No apt/curl/bun/npm, so + # trial setup needs zero outbound network (no_network tasks set up cleanly). + await self._install_binary(environment) else: - self._cli = await self._install_local(environment) + # 1) System deps (root). curl+unzip for the Bun installer; ca-certs for TLS. + await self.exec_as_root( + environment, + command=( + "set -e; " + "if command -v apt-get >/dev/null 2>&1; then " + " apt-get update && DEBIAN_FRONTEND=noninteractive apt-get install -y curl unzip ca-certificates tar; " + "elif command -v apk >/dev/null 2>&1; then " + " echo 'ERROR: Alpine/musl base image; @oh-my-pi/pi-natives ships no musl prebuilt' >&2; exit 3; " + "elif command -v dnf >/dev/null 2>&1; then dnf install -y curl unzip tar; " + "elif command -v yum >/dev/null 2>&1; then yum install -y curl unzip tar; " + "fi" + ), + ) + # 2) Bun (agent user). + await self.exec_as_agent( + environment, + command=( + "set -e; " + f"export BUN_INSTALL={shlex.quote(self._home + '/.bun')}; " + f'curl -fsSL https://bun.sh/install | bash -s "bun-v{self._bun_version}"; ' + f'{shlex.quote(self._home + "/.bun/bin/bun")} --version' + ), + ) + self._bun = f"{self._home}/.bun/bin/bun" + if self._install_mode == "published": + self._cli = await self._install_published(environment) + else: + self._cli = await self._install_local(environment) # 3) Auth + model config under $HOME/.omp/agent. if self._gateway_on: @@ -299,6 +307,29 @@ class OmpLocal(BaseInstalledAgent): ) return f"{app}/dist/cli.js" + async def _install_binary(self, environment: BaseEnvironment) -> str: + """Probe container arch, upload only the matching self-contained omp binary.""" + arch = (await self.exec_as_agent(environment, command="uname -m")).stdout.strip() + if arch in ("aarch64", "arm64"): + hostbin = self._binary_arm64 + elif arch in ("x86_64", "amd64"): + hostbin = self._binary_x64 + else: + raise RuntimeError(f"binary mode: unsupported container arch {arch!r}") + if not hostbin: + raise RuntimeError(f"binary mode: no omp binary provided for container arch {arch}") + app_dir = f"{self._home}/.omp-bench" + dst = f"{app_dir}/omp" + staging = "/tmp/omp-bin" + await self.exec_as_agent(environment, command=f"mkdir -p {shlex.quote(app_dir)}") + await environment.upload_file(hostbin, staging) + await self.exec_as_agent( + environment, + command=f"cp {shlex.quote(staging)} {shlex.quote(dst)} && chmod +x {shlex.quote(dst)}", + ) + self._cli = dst + return dst + async def _install_published(self, environment: BaseEnvironment) -> str: app = f"{self._home}/.omp-bench/app" spec = f"@oh-my-pi/pi-coding-agent@{self._pkg_version}" @@ -415,9 +446,11 @@ class OmpLocal(BaseInstalledAgent): raise ValueError("model must be 'provider/model' (e.g. anthropic/claude-sonnet-4-6)") provider, model = self.model_name.split("/", 1) - parts = [ - shlex.quote(self._bun), - shlex.quote(self._cli), + if self._binary: + parts = [shlex.quote(self._cli)] + else: + parts = [shlex.quote(self._bun), shlex.quote(self._cli)] + parts += [ "--print", "--mode json", f"--provider {shlex.quote(provider)}", @@ -451,7 +484,7 @@ class OmpLocal(BaseInstalledAgent): if not self._gateway_on: run_env.update(self._collect_provider_keys(provider)) run_env.update(self._forward_env) - await self.exec_as_agent(environment, command=self._wrap(run), env=run_env or None) + await self.exec_as_agent(environment, command=run if self._binary else self._wrap(run), env=run_env or None) @override def populate_context_post_run(self, context: AgentContext) -> None: diff --git a/packages/terminal-bench/src/runner.ts b/packages/terminal-bench/src/runner.ts index 068db1d5a..33b24c52e 100755 --- a/packages/terminal-bench/src/runner.ts +++ b/packages/terminal-bench/src/runner.ts @@ -42,6 +42,8 @@ export interface Config { install: "local" | "published"; version: string | null; tarball: string | null; + binaryArm64: string | null; + binaryX64: string | null; build: boolean; jobsDir: string; jobName: string | null; @@ -77,6 +79,8 @@ function defaultConfig(): Config { install: "local", version: null, tarball: null, + binaryArm64: null, + binaryX64: null, build: true, jobsDir: path.join(REPO_ROOT, "runs", "tb2"), jobName: null, @@ -196,6 +200,15 @@ export function parseArgs(argv: string[]): Config { cfg.tarball = path.resolve(take(arg)); cfg.build = false; break; + case "--binary": { + const p = path.resolve(take(arg)); + const base = path.basename(p); + if (/arm64|aarch64/.test(base)) cfg.binaryArm64 = p; + else if (/x64|x86[_-]?64|amd64/.test(base)) cfg.binaryX64 = p; + else throw new Error(`--binary: cannot infer arch from ${base} (expect arm64/x64 in filename)`); + cfg.build = false; + break; + } case "--no-build": cfg.build = false; break; @@ -918,6 +931,8 @@ export function buildHarborEnv( env.OMP_TB_INSTALL = cfg.install; env.OMP_TB_VERSION = cfg.version ?? version; if (tarball) env.OMP_TB_TARBALL = tarball; + if (cfg.binaryArm64) env.OMP_TB_BINARY_ARM64 = cfg.binaryArm64; + if (cfg.binaryX64) env.OMP_TB_BINARY_X64 = cfg.binaryX64; if (cfg.thinking) env.OMP_TB_THINKING = cfg.thinking; if (cfg.advisorModel) { env.OMP_TB_ADVISOR_MODEL = cfg.advisorModel; @@ -1068,7 +1083,7 @@ async function main(): Promise { // tarball (local install only) let tarball: string | null = cfg.tarball; - if (cfg.agent === "omp" && cfg.install === "local") { + if (cfg.agent === "omp" && cfg.install === "local" && !cfg.binaryArm64 && !cfg.binaryX64) { if (tarball) { process.stdout.write(dim(`using tarball ${tarball}\n`)); } else if (cfg.build) { From 00dcd545977e91f288b0ebfc69ae622dc7ecc205 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 08:06:54 +0200 Subject: [PATCH 39/43] feat: implemented reparenting for backgrounded wrappers - Introduce `detach_reparent` parameter to command execution to support process reparenting. - Add `detach_session_reparent` to Unix command extensions using a double-fork technique to orphan processes from the shell descendant tree. - Update background pipeline logic to automatically apply reparenting when unwrapping transparent wrappers like `nohup`. - Remove unused `command_is_resolvable` helper. --- crates/brush-core-vendored/src/commands.rs | 8 ++- crates/brush-core-vendored/src/interp.rs | 24 +++++++-- .../src/sys/stubs/commands.rs | 7 +++ .../src/sys/unix/commands.rs | 41 +++++++++++++++ .../src/sys/windows/commands.rs | 8 +++ crates/pi-shell/src/shell.rs | 50 ++++--------------- packages/coding-agent/scripts/build-binary.ts | 4 +- 7 files changed, 96 insertions(+), 46 deletions(-) diff --git a/crates/brush-core-vendored/src/commands.rs b/crates/brush-core-vendored/src/commands.rs index eef2c3244..1c5e1ce70 100644 --- a/crates/brush-core-vendored/src/commands.rs +++ b/crates/brush-core-vendored/src/commands.rs @@ -632,7 +632,13 @@ pub(crate) fn execute_external_command( match session_action { ChildSessionAction::DetachSession => { // setsid() creates the fresh session + process group; no process_group(). - cmd.detach_session(); + // A reparenting operand (`nohup cmd &`) additionally double-forks so it + // leaves the host's descendant tree and survives the teardown walk. + if context.params.detach_reparent { + cmd.detach_session_reparent(); + } else { + cmd.detach_session(); + } } ChildSessionAction::TakeForeground if command_leads_session => { // Don't set process_group(0) - setsid() in pre_exec will handle it. diff --git a/crates/brush-core-vendored/src/interp.rs b/crates/brush-core-vendored/src/interp.rs index e3aed2309..cd1d98f66 100644 --- a/crates/brush-core-vendored/src/interp.rs +++ b/crates/brush-core-vendored/src/interp.rs @@ -73,6 +73,11 @@ pub struct ExecutionParameters { open_files: openfiles::OpenFiles, /// Policy for how to manage spawned external processes. pub process_group_policy: ProcessGroupPolicy, + /// Whether external commands spawned in this context should reparent out of + /// the shell's descendant tree (double-fork on Unix) so they survive the + /// host's descendant-walk teardown. Set for the operand of a transparent + /// background wrapper such as `nohup cmd &`. + pub detach_reparent: bool, /// Optional cancellation token shared with callers. cancel_token: Option, /// Optional command-output marker hook. @@ -382,7 +387,12 @@ async fn spawn_async_ao_list_as_job<'a, SE: extensions::ShellExtensions>( let direct_pipeline = background_process_pipeline_for_async_job(ao_list, shell, &async_params).await?; - let job = if let Some(pipeline) = direct_pipeline { + let job = if let Some((pipeline, detach_reparent)) = direct_pipeline { + // A transparent background wrapper (e.g. `nohup cmd &`) was unwrapped to its + // operand. Reparent that operand out of the shell's descendant tree so it + // survives the host's descendant-walk teardown — the persistence agents + // reach for `nohup` expecting. + async_params.detach_reparent = detach_reparent; match try_spawn_pipeline_as_job(&pipeline, ao_list.to_string(), shell, &async_params).await? { Some(job) => job, None => spawn_async_ao_list_in_task(ao_list, shell, &async_params), @@ -404,16 +414,22 @@ async fn background_process_pipeline_for_async_job, params: &ExecutionParameters, -) -> Result, error::Error> { +) -> Result, error::Error> { if !ao_list.additional.is_empty() { return Ok(None); } let mut pipeline = ao_list.first.clone(); + let mut detach_reparent = false; for _ in 0..8 { match classify_background_process_pipeline(&pipeline, shell, params).await? { - BackgroundProcessPipeline::Direct => return Ok(Some(pipeline)), - BackgroundProcessPipeline::Wrapper(unwrapped) => pipeline = unwrapped, + BackgroundProcessPipeline::Direct => return Ok(Some((pipeline, detach_reparent))), + BackgroundProcessPipeline::Wrapper(unwrapped) => { + // Unwrapping a transparent background wrapper (`nohup`) means the + // operand should reparent away from the shell when finally spawned. + detach_reparent = true; + pipeline = unwrapped; + }, BackgroundProcessPipeline::Internal => return Ok(None), } } diff --git a/crates/brush-core-vendored/src/sys/stubs/commands.rs b/crates/brush-core-vendored/src/sys/stubs/commands.rs index 50689ee78..527c82659 100644 --- a/crates/brush-core-vendored/src/sys/stubs/commands.rs +++ b/crates/brush-core-vendored/src/sys/stubs/commands.rs @@ -98,10 +98,17 @@ impl CommandFgControlExt for std::process::Command { pub trait CommandSessionExt { /// Arranges for the command to run in a new session with no controlling terminal. fn detach_session(&mut self); + /// Like [`CommandSessionExt::detach_session`]. No-op on platforms without + /// `setsid`/`fork` reparenting. + fn detach_session_reparent(&mut self); } impl CommandSessionExt for std::process::Command { fn detach_session(&mut self) { // NOTE: This is a no-op on platforms without setsid support. } + + fn detach_session_reparent(&mut self) { + // NOTE: This is a no-op on platforms without setsid/fork support. + } } diff --git a/crates/brush-core-vendored/src/sys/unix/commands.rs b/crates/brush-core-vendored/src/sys/unix/commands.rs index 91f4cce4a..46759a3d6 100644 --- a/crates/brush-core-vendored/src/sys/unix/commands.rs +++ b/crates/brush-core-vendored/src/sys/unix/commands.rs @@ -73,6 +73,10 @@ impl CommandFgControlExt for std::process::Command { pub trait CommandSessionExt { /// Arranges for the command to run in a new POSIX session with no controlling terminal. fn detach_session(&mut self); + /// Like [`CommandSessionExt::detach_session`], but additionally double-forks + /// so the spawned process reparents to init (PID 1) and leaves the caller's + /// descendant tree. + fn detach_session_reparent(&mut self); } impl CommandSessionExt for std::process::Command { @@ -84,6 +88,15 @@ impl CommandSessionExt for std::process::Command { self.pre_exec(pre_exec_detach_session); } } + + fn detach_session_reparent(&mut self) { + // SAFETY: + // This arranges for a provided function to run in the forked child before + // exec. Only async-signal-safe calls (`setsid`, `fork`, `_exit`) are used. + unsafe { + self.pre_exec(pre_exec_detach_session_reparent); + } + } } fn pre_exec_take_foreground() -> Result<(), std::io::Error> { @@ -119,3 +132,31 @@ fn pre_exec_detach_session() -> Result<(), std::io::Error> { Err(errno) => Err(std::io::Error::from_raw_os_error(errno as i32)), } } + +fn pre_exec_detach_session_reparent() -> Result<(), std::io::Error> { + // New session first: drop any controlling terminal. Ignore EPERM, which means + // the child is already a session leader from an outer policy. + match nix::unistd::setsid() { + Ok(_) | Err(nix::errno::Errno::EPERM) => {}, + Err(errno) => return Err(std::io::Error::from_raw_os_error(errno as i32)), + } + + // Double-fork: the intermediate child — the pid the parent's spawn machinery + // tracks — exits immediately, so the grandchild that goes on to `exec` the + // operand reparents to init (PID 1) and is no longer a descendant of the + // shell. This is what lets `nohup cmd &` survive the host's descendant-walk + // teardown without relying on an external `setsid(1)` binary. + // + // SAFETY: the post-`fork` child here is single-threaded, and only + // async-signal-safe primitives (`fork`, `_exit`) run before `exec`. + let pid = unsafe { libc::fork() }; + if pid < 0 { + return Err(std::io::Error::last_os_error()); + } + if pid > 0 { + // Intermediate parent: exit now to orphan the grandchild. `_exit` avoids + // running atexit handlers or flushing inherited buffers in the fork. + unsafe { libc::_exit(0) }; + } + Ok(()) +} diff --git a/crates/brush-core-vendored/src/sys/windows/commands.rs b/crates/brush-core-vendored/src/sys/windows/commands.rs index c22d0a3a3..54ec635c9 100644 --- a/crates/brush-core-vendored/src/sys/windows/commands.rs +++ b/crates/brush-core-vendored/src/sys/windows/commands.rs @@ -114,10 +114,18 @@ pub trait CommandSessionExt { /// terminal. On Windows this is a no-op; process-group and console behavior /// are handled uniformly by `sys::process::spawn`. fn detach_session(&mut self); + /// Like [`CommandSessionExt::detach_session`]. On Windows there is no session + /// or `fork`-based reparenting, so this is a no-op: the operand stays a child + /// of the shell. + fn detach_session_reparent(&mut self); } impl CommandSessionExt for std::process::Command { fn detach_session(&mut self) { // NOTE: Windows has no setsid; intentionally a no-op. } + + fn detach_session_reparent(&mut self) { + // NOTE: no reparenting primitive on Windows; intentionally a no-op. + } } diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index ad3ffcf70..2e0be99b6 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -496,39 +496,6 @@ fn normalize_path_segment(segment: &str) -> String { normalized.to_string_lossy().to_ascii_lowercase() } -/// Check if a command is resolvable in the given PATH string. -/// Returns true if the command exists and is executable in one of the PATH -/// directories. -fn command_is_resolvable(command: &str, path: &str) -> bool { - for dir in std::env::split_paths(path) { - let full_path = dir.join(command); - #[cfg(unix)] - { - use std::os::unix::fs::PermissionsExt; - if full_path.exists() { - if let Ok(metadata) = full_path.metadata() { - let permissions = metadata.permissions(); - if permissions.mode() & 0o111 != 0 { - return true; - } - } - } - } - #[cfg(windows)] - { - if full_path.exists() { - // On Windows, .exe/.bat/.cmd extensions are automatically tried - for ext in ["", ".exe", ".bat", ".cmd"] { - let with_ext = dir.join(format!("{}{}", command, ext)); - if with_ext.exists() { - return true; - } - } - } - } - } - false -} #[cfg(not(windows))] fn merge_path_values(_existing: &str, incoming: &str) -> String { incoming.to_string() @@ -1966,13 +1933,16 @@ impl builtins::Command for NohupCommand { return Ok(ExecutionResult::new(125)); } - // Detach the operand into a new session / process group (like `setsid`) - // so a backgrounded server survives this embedded shell's kill-on-drop - // teardown, which SIGKILLs the shell's own process group when the host - // process exits. Agents reach for `nohup &` expecting exactly - // this persistence; a real coreutils `nohup` would NOT help, since it - // stays in the shell's process group and dies with it. The new session - // is applied below via ProcessGroupPolicy::NewProcessGroup. + // `nohup ` (foreground) runs the operand directly and surfaces its + // exit status — the contract pinned by + // `nohup_builtin_propagates_command_exit_code`. Persistence across the + // host's teardown is a *background* concern that never reaches this + // builtin: the agent writes `nohup &`, and brush's + // `transparent_background_wrapper` unwraps that to spawn the operand + // directly with `detach_reparent`, double-forking it out of the shell's + // descendant tree (see `execute_external_command` / `detach_session_reparent`). + // Like coreutils, we run the operand here; we only differ by not masking + // SIGHUP (see `nohup_builtin_does_not_mask_sighup`). let mut command_line = String::new(); for (idx, arg) in command.iter().enumerate() { if idx > 0 { diff --git a/packages/coding-agent/scripts/build-binary.ts b/packages/coding-agent/scripts/build-binary.ts index a4cfabeb1..b90498214 100644 --- a/packages/coding-agent/scripts/build-binary.ts +++ b/packages/coding-agent/scripts/build-binary.ts @@ -53,7 +53,9 @@ async function main(): Promise { await runCommand(["bun", "scripts/generate-docs-index.ts", "--generate"]); await runCommand( ["bun", "--cwd=../natives", "run", "embed:native"], - crossTarget ? { ...Bun.env, TARGET_PLATFORM: crossPlatform as string, TARGET_ARCH: crossArch as string } : Bun.env, + crossTarget + ? { ...Bun.env, TARGET_PLATFORM: crossPlatform as string, TARGET_ARCH: crossArch as string } + : Bun.env, ); await runCommand(["bun", "scripts/embed-mupdf-wasm.ts", "--generate"]); try { From 2787b6dff70d0ce5cd7bbd9133d88b05cb9229ce Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 08:08:31 +0200 Subject: [PATCH 40/43] chore: update changelogs --- packages/coding-agent/CHANGELOG.md | 17 +- ...selector-controller-session-delete.test.ts | 284 ------------------ packages/coding-agent/test/tools.test.ts | 2 +- .../test/tools/inspect-image.test.ts | 2 +- .../test/utils/image-resize.test.ts | 10 +- packages/collab-web/CHANGELOG.md | 2 +- 6 files changed, 17 insertions(+), 300 deletions(-) delete mode 100644 packages/coding-agent/test/modes/controllers/selector-controller-session-delete.test.ts diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a50ea4dcd..5f093ed04 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Breaking Changes - Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. @@ -22,13 +23,6 @@ - Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`. - Made the `--resume` session picker fullscreen on the terminal's alternate screen, so the list scrolls with the mouse wheel and a row resumes its session on left click. Rows are hit-tested against the live scroll window, and the keybinding hint + bottom border are now pinned to the screen bottom instead of drifting up and down as the visible window changes height. -### Removed - -- Removed `append`, `tree`, and `diff` eval helper functions from Python, JavaScript, and Ruby -- Removed `sort`, `uniq`, and `counter` text processing eval helpers from Python, JavaScript, and Ruby -- Removed the `append(path, content)`, `tree(path, max_depth?, show_hidden?)`, and `diff(a, b)` eval prelude helpers from every workflow runtime (Python, JavaScript, Ruby, Julia), along with their status renderers, icon entries, and tool/`docs` references. Use `write`/`read` for file mutation and `tool.(...)` for richer filesystem operations. -- Removed the `sort(text, reverse?, unique?)`, `uniq(text, count?)`, and `counter(items, limit?, reverse?)` eval text helpers from the Python, JavaScript, and Ruby prelude surfaces (Julia never defined them), along with the JS `HelperBundle`/`HelperOptions` members and `docs` references. Sort/dedupe/count inline in cell code instead. - ### Fixed - Fixed `local://` URLs decoding images as corrupted text (mojibake) instead of showing the image @@ -47,6 +41,13 @@ - Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Active-goal compaction continuations now also resolve completed retry gates before returning, preventing `isRetrying` from staying stuck after a retry succeeds over the threshold. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) - Fixed active `/goal` runs that never reached `agent_end` because the model kept emitting tool calls inside one agent run. Threshold maintenance now runs between tool-call turns, compacts the live loop context in place, and suppresses queued continuations that would race the still-running goal loop. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174)) +### Removed + +- Removed `append`, `tree`, and `diff` eval helper functions from Python, JavaScript, and Ruby +- Removed `sort`, `uniq`, and `counter` text processing eval helpers from Python, JavaScript, and Ruby +- Removed the `append(path, content)`, `tree(path, max_depth?, show_hidden?)`, and `diff(a, b)` eval prelude helpers from every workflow runtime (Python, JavaScript, Ruby, Julia), along with their status renderers, icon entries, and tool/`docs` references. Use `write`/`read` for file mutation and `tool.(...)` for richer filesystem operations. +- Removed the `sort(text, reverse?, unique?)`, `uniq(text, count?)`, and `counter(items, limit?, reverse?)` eval text helpers from the Python, JavaScript, and Ruby prelude surfaces (Julia never defined them), along with the JS `HelperBundle`/`HelperOptions` members and `docs` references. Sort/dedupe/count inline in cell code instead. + ## [16.1.15] - 2026-06-22 ### Added @@ -12385,4 +12386,4 @@ Initial public release. ## [0.7.6] - 2025-11-13 -Previous releases did not maintain a changelog. \ No newline at end of file +Previous releases did not maintain a changelog. diff --git a/packages/coding-agent/test/modes/controllers/selector-controller-session-delete.test.ts b/packages/coding-agent/test/modes/controllers/selector-controller-session-delete.test.ts deleted file mode 100644 index 87d8f9870..000000000 --- a/packages/coding-agent/test/modes/controllers/selector-controller-session-delete.test.ts +++ /dev/null @@ -1,284 +0,0 @@ -import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test"; -import { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector"; -import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller"; -import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; -import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-listing"; -import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; -import { FileSessionStorage } from "@oh-my-pi/pi-coding-agent/session/session-storage"; - -type TestContext = InteractiveModeContext & { - editorContainer: { - children: unknown[]; - clear: () => void; - addChild: (child: unknown) => void; - }; -}; - -function makeSessionInfo(path: string): SessionInfo { - return { - path, - id: path, - cwd: "/tmp/project", - title: "Active session", - created: new Date("2025-01-01T00:00:00Z"), - modified: new Date("2025-01-01T00:00:00Z"), - messageCount: 1, - size: 0, - firstMessage: "hello", - allMessagesText: "hello", - }; -} - -function createContext(currentSessionFile: string): { - ctx: TestContext; - calls: string[]; - setCurrentSessionFile: (path: string) => void; - showHookConfirm: (title: string, message: string) => Promise; - newSession: () => Promise; -} { - const calls: string[] = []; - let sessionFile = currentSessionFile; - const editorContainer = { - children: [] as unknown[], - clear() { - this.children = []; - calls.push("editorContainer.clear"); - }, - addChild(child: unknown) { - this.children.push(child); - calls.push("editorContainer.addChild"); - }, - }; - const showHookConfirm = vi.fn(async () => true); - const newSession = vi.fn(async () => { - calls.push("session.newSession"); - sessionFile = "/tmp/project/sessions/detached.jsonl"; - return true; - }); - const session = { - newSession, - switchSession: vi.fn(async () => true), - }; - const ctx = { - editorContainer, - editor: {}, - ui: { - setFocus: vi.fn(), - requestRender: vi.fn(() => { - calls.push("ui.requestRender"); - }), - terminal: { columns: 120 }, - }, - session, - get viewSession() { - return session; - }, - sessionManager: { - getCwd: () => "/tmp/project", - getSessionDir: () => "/tmp/project/sessions", - getSessionFile: () => sessionFile, - }, - statusContainer: { - clear: vi.fn(() => { - calls.push("statusContainer.clear"); - }), - }, - pendingMessagesContainer: { - clear: vi.fn(() => { - calls.push("pendingMessagesContainer.clear"); - }), - }, - compactionQueuedMessages: [] as unknown[], - streamingComponent: { active: true }, - streamingMessage: { active: true }, - pendingTools: { - clear: vi.fn(() => { - calls.push("pendingTools.clear"); - }), - }, - loadingAnimation: { - stop: vi.fn(() => { - calls.push("loadingAnimation.stop"); - }), - }, - statusLine: { - invalidate: vi.fn(() => { - calls.push("statusLine.invalidate"); - }), - setSessionStartTime: vi.fn(() => { - calls.push("statusLine.setSessionStartTime"); - }), - }, - updateEditorTopBorder: vi.fn(() => { - calls.push("updateEditorTopBorder"); - }), - updateEditorBorderColor: vi.fn(() => { - calls.push("updateEditorBorderColor"); - }), - renderInitialMessages: vi.fn(() => { - calls.push("renderInitialMessages"); - }), - reloadTodos: vi.fn(async () => { - calls.push("reloadTodos"); - }), - showStatus: vi.fn((message: string) => { - calls.push(`showStatus:${message}`); - }), - showError: vi.fn(), - showHookConfirm, - shutdown: vi.fn(async () => undefined), - clearTransientSessionUi() { - ctx.loadingAnimation?.stop(); - ctx.statusContainer.clear(); - ctx.pendingMessagesContainer.clear(); - ctx.pendingTools.clear(); - }, - } as unknown as TestContext; - - return { - ctx, - calls, - setCurrentSessionFile(path: string) { - sessionFile = path; - }, - showHookConfirm, - newSession, - }; -} - -function renderText(selector: SessionSelectorComponent): string { - return selector.render(120).join("\n"); -} - -beforeAll(() => { - initTheme(); -}); - -describe("SelectorController session deletion", () => { - beforeEach(() => { - vi.spyOn(SessionManager, "list").mockResolvedValue([]); - vi.spyOn(SessionManager, "listAll").mockResolvedValue([]); - }); - - afterEach(() => { - vi.restoreAllMocks(); - }); - - it("detaches the active session before selector deletion removes it", async () => { - const activeSession = makeSessionInfo("/tmp/project/sessions/active.jsonl"); - const { ctx, calls } = createContext(activeSession.path); - vi.spyOn(SessionManager, "list").mockResolvedValue([activeSession]); - const deleteSessionWithArtifacts = vi - .spyOn(FileSessionStorage.prototype, "deleteSessionWithArtifacts") - .mockImplementation(async sessionPath => { - calls.push(`delete:${sessionPath}`); - }); - const controller = new SelectorController(ctx); - - await controller.showSessionSelector(); - const selector = ctx.editorContainer.children[0]; - if (!(selector instanceof SessionSelectorComponent)) { - throw new Error("Expected session selector component"); - } - - const sessionList = selector.getSessionList() as unknown as { - onDeleteRequest?: (session: SessionInfo) => void; - }; - sessionList.onDeleteRequest?.(activeSession); - selector.handleInput("\n"); - await Bun.sleep(0); - - expect(deleteSessionWithArtifacts).toHaveBeenCalledWith(activeSession.path); - expect(calls).toEqual([ - "editorContainer.clear", - "editorContainer.addChild", - "ui.requestRender", - "session.newSession", - "loadingAnimation.stop", - "statusContainer.clear", - "pendingMessagesContainer.clear", - "pendingTools.clear", - "statusLine.invalidate", - "statusLine.setSessionStartTime", - "updateEditorTopBorder", - "updateEditorBorderColor", - "renderInitialMessages", - "reloadTodos", - "ui.requestRender", - `delete:${activeSession.path}`, - "ui.requestRender", - ]); - expect(ctx.sessionManager.getSessionFile()).toBe("/tmp/project/sessions/detached.jsonl"); - }); - - it("shows inline selector errors when session deletion fails after detach", async () => { - const activeSession = makeSessionInfo("/tmp/project/sessions/active.jsonl"); - const { ctx, newSession } = createContext(activeSession.path); - vi.spyOn(SessionManager, "list").mockResolvedValue([activeSession]); - const deleteSessionWithArtifacts = vi - .spyOn(FileSessionStorage.prototype, "deleteSessionWithArtifacts") - .mockRejectedValue(new Error("disk failed")); - const controller = new SelectorController(ctx); - - await controller.showSessionSelector(); - const selector = ctx.editorContainer.children[0]; - if (!(selector instanceof SessionSelectorComponent)) { - throw new Error("Expected session selector component"); - } - - const sessionList = selector.getSessionList() as unknown as { - onDeleteRequest?: (session: SessionInfo) => void; - }; - sessionList.onDeleteRequest?.(activeSession); - selector.handleInput("\n"); - await Bun.sleep(0); - - expect(newSession).toHaveBeenCalledTimes(1); - expect(deleteSessionWithArtifacts).toHaveBeenCalledWith(activeSession.path); - expect(ctx.showError).not.toHaveBeenCalled(); - expect(ctx.sessionManager.getSessionFile()).toBe("/tmp/project/sessions/detached.jsonl"); - expect(renderText(selector)).toContain("Error: Failed to delete session: disk failed"); - }); - - it("creates a fresh session before deleting via slash command and then shows the selector", async () => { - const activeSessionPath = "/tmp/project/sessions/active.jsonl"; - const { ctx, calls, showHookConfirm, newSession } = createContext(activeSessionPath); - const deleteSessionWithArtifacts = vi - .spyOn(FileSessionStorage.prototype, "deleteSessionWithArtifacts") - .mockImplementation(async sessionPath => { - calls.push(`delete:${sessionPath}`); - }); - const exists = vi.spyOn(FileSessionStorage.prototype, "exists").mockResolvedValue(true); - const controller = new SelectorController(ctx); - - await controller.handleSessionDeleteCommand(); - - expect(exists).toHaveBeenCalledWith(activeSessionPath); - expect(showHookConfirm).toHaveBeenCalledWith( - "Delete Session", - "This will permanently delete the current session.\nYou will be returned to the session selector.", - ); - expect(newSession).toHaveBeenCalledTimes(1); - expect(deleteSessionWithArtifacts).toHaveBeenCalledWith(activeSessionPath); - expect(calls).toEqual([ - "session.newSession", - "loadingAnimation.stop", - "statusContainer.clear", - "pendingMessagesContainer.clear", - "pendingTools.clear", - "statusLine.invalidate", - "statusLine.setSessionStartTime", - "updateEditorTopBorder", - "updateEditorBorderColor", - "renderInitialMessages", - "reloadTodos", - "ui.requestRender", - `delete:${activeSessionPath}`, - "showStatus:Session deleted", - "editorContainer.clear", - "editorContainer.addChild", - "ui.requestRender", - ]); - }); -}); diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index 0720f31c9..cae087d36 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -803,7 +803,7 @@ describe("Coding Agent Tools", () => { fs.writeFileSync(testFile, pngBuffer); const legacyReadTool = wrapToolWithMetaNotice( - new ReadTool(createTestToolSession(testDir, Settings.isolated({ "inspect_image.enabled": false }))), + new ReadTool(createTestToolSession(testDir, Settings.isolated({ "inspect_image.enabled": false, "images.autoResize": false }))), ); const result = await legacyReadTool.execute("test-call-img-1", { path: testFile }); diff --git a/packages/coding-agent/test/tools/inspect-image.test.ts b/packages/coding-agent/test/tools/inspect-image.test.ts index f3149ab54..3393cd57f 100644 --- a/packages/coding-agent/test/tools/inspect-image.test.ts +++ b/packages/coding-agent/test/tools/inspect-image.test.ts @@ -161,7 +161,7 @@ describe("InspectImageTool", () => { const stub = createCompleteSimpleSuccessStub("Attached image inspected"); const missingCwd = path.join(testDir, "missing-cwd"); const tool = new InspectImageTool( - createSession(missingCwd, visionModel, "test-key", Settings.isolated(), { + createSession(missingCwd, visionModel, "test-key", Settings.isolated({ "images.autoResize": false }), { imageAttachments: [{ label: "Image #1", uri: "attachment://1", image }], }), stub.fn, diff --git a/packages/coding-agent/test/utils/image-resize.test.ts b/packages/coding-agent/test/utils/image-resize.test.ts index 7a092cf6b..7e145f277 100644 --- a/packages/coding-agent/test/utils/image-resize.test.ts +++ b/packages/coding-agent/test/utils/image-resize.test.ts @@ -25,8 +25,8 @@ async function makeRedWebP(width: number, height: number): Promise { // its input, so a single decodable source per shape serves every test. Real image // encode/decode is the only cost here, so each source is the smallest solid-red // image that still crosses the threshold under test: -// - oversizedPng: a thin strip whose long edge exceeds the 1568 default cap, so -// re-encodes touch ~1568×98 px instead of 1568×1568. A uniform red keeps format +// - oversizedPng: a strip whose long edge exceeds the 1568 default cap, so +// re-encodes touch ~1568×392 px instead of 1568×1568. A uniform red keeps format // selection deterministic (WebP is always smallest; PNG always beats JPEG), so // the strip exercises the same format/budget logic as a large square. // - smallPng / smallWebp: 200×200, comfortably inside every default cap (fast path). @@ -36,7 +36,7 @@ let smallWebp: string; beforeAll(async () => { [oversizedPng, smallPng, smallWebp] = await Promise.all([ - makeRedPng(1600, 100), + makeRedPng(1600, 400), makeRedPng(200, 200), makeRedWebP(200, 200), ]); @@ -50,8 +50,8 @@ describe("resizeImage defaults", () => { expect(result.wasResized).toBe(true); expect(result.width).toBeLessThanOrEqual(1568); expect(result.height).toBeLessThanOrEqual(1568); - // Aspect ratio of the 1600x100 source preserved (with rounding tolerance). - expect(Math.abs(result.width / result.height - 1600 / 100)).toBeLessThan(0.01); + // Aspect ratio of the 1600x400 source preserved (with rounding tolerance). + expect(Math.abs(result.width / result.height - 1600 / 400)).toBeLessThan(0.01); }); it("preserves inputs already within budget and dimensions (fast path)", async () => { diff --git a/packages/collab-web/CHANGELOG.md b/packages/collab-web/CHANGELOG.md index a8f40829b..981da4464 100644 --- a/packages/collab-web/CHANGELOG.md +++ b/packages/collab-web/CHANGELOG.md @@ -130,4 +130,4 @@ ### Security -- Hardened transcript Markdown rendering by escaping embedded HTML and allowing only safe link schemes \ No newline at end of file +- Hardened transcript Markdown rendering by escaping embedded HTML and allowing only safe link schemes From a6bdef85a28075971cc84c95444b469c31aaf816 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 08:18:45 +0200 Subject: [PATCH 41/43] feat(ai): added support for reasoning items in replay sanitization - Added `sanitizeOpenAIResponsesReasoningItemForReplay` to process reasoning-type items by stripping unique identifiers and filtering properties. - Updated the main sanitization utility to route reasoning items through the new logic. --- packages/ai/CHANGELOG.md | 1 + packages/ai/src/utils.ts | 14 ++++++++++++++ .../ai/test/openai-responses-openrouter.test.ts | 1 + packages/coding-agent/test/tools.test.ts | 7 ++++++- 4 files changed, 22 insertions(+), 1 deletion(-) diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 329d36472..dfb4ed404 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. Injected SDK clients, GitHub Copilot's Anthropic proxy, and Vertex rawPredict are excluded because this code path cannot add the beta to caller-owned clients, Copilot strips Anthropic betas and demotes thinking blocks to text upstream, and Vertex expects betas in the JSON body rather than the Anthropic HTTP beta header. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) +- Fixed OpenRouter Responses native history replay leaking Gemini reasoning item `format` metadata back into follow-up requests, which caused HTTP 400 rejections while preserving encrypted reasoning replay. ## [16.1.15] - 2026-06-22 diff --git a/packages/ai/src/utils.ts b/packages/ai/src/utils.ts index 1c2be665c..4eb074951 100644 --- a/packages/ai/src/utils.ts +++ b/packages/ai/src/utils.ts @@ -78,6 +78,7 @@ function sanitizeOpenAIResponsesHistoryItemForReplay( ): OpenAIResponsesReplayItem | undefined { if (item.type === "item_reference") return undefined; if (item.type === "image_generation_call") return sanitizeOpenAIResponsesImageGenerationCallForReplay(item); + if (item.type === "reasoning") return sanitizeOpenAIResponsesReasoningItemForReplay(item); // providerPayload stores raw output items; replay strips item ids and keeps only normalized call_id. const { id: _id, ...sanitizedItem } = item; @@ -88,6 +89,19 @@ function sanitizeOpenAIResponsesHistoryItemForReplay( return sanitizedItem as unknown as OpenAIResponsesReplayItem; } +function sanitizeOpenAIResponsesReasoningItemForReplay(item: Record): OpenAIResponsesReplayItem { + const sanitizedItem: Record = { type: "reasoning" }; + if (Array.isArray(item.summary)) sanitizedItem.summary = item.summary; + if (Array.isArray(item.content)) sanitizedItem.content = item.content; + if (typeof item.encrypted_content === "string" || item.encrypted_content === null) { + sanitizedItem.encrypted_content = item.encrypted_content; + } + if (item.status === "in_progress" || item.status === "completed" || item.status === "incomplete") { + sanitizedItem.status = item.status; + } + return sanitizedItem as unknown as OpenAIResponsesReplayItem; +} + function sanitizeOpenAIResponsesImageGenerationCallForReplay( item: Record, ): ResponseInputItem.ImageGenerationCall | undefined { diff --git a/packages/ai/test/openai-responses-openrouter.test.ts b/packages/ai/test/openai-responses-openrouter.test.ts index fc5f55cff..20515f726 100644 --- a/packages/ai/test/openai-responses-openrouter.test.ts +++ b/packages/ai/test/openai-responses-openrouter.test.ts @@ -351,6 +351,7 @@ describe("OpenRouter Responses request shape", () => { id: "rs_1", encrypted_content: "encrypted-reasoning", summary: [], + format: "google-gemini-v1", }; const replayItem = { type: nativeItem.type, diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index cae087d36..7a4c2ec20 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -803,7 +803,12 @@ describe("Coding Agent Tools", () => { fs.writeFileSync(testFile, pngBuffer); const legacyReadTool = wrapToolWithMetaNotice( - new ReadTool(createTestToolSession(testDir, Settings.isolated({ "inspect_image.enabled": false, "images.autoResize": false }))), + new ReadTool( + createTestToolSession( + testDir, + Settings.isolated({ "inspect_image.enabled": false, "images.autoResize": false }), + ), + ), ); const result = await legacyReadTool.execute("test-call-img-1", { path: testFile }); From 2c3e0d79a426a1ec44461655c340f3a92140a025 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 08:22:21 +0200 Subject: [PATCH 42/43] chore: bump version to 16.1.16 --- Cargo.lock | 12 +++--- Cargo.toml | 2 +- bun.lock | 54 +++++++++++++-------------- crates/pi-natives/src/lib.rs | 2 +- package.json | 24 ++++++------ packages/agent/CHANGELOG.md | 2 + packages/agent/package.json | 2 +- packages/ai/CHANGELOG.md | 2 + packages/ai/package.json | 2 +- packages/catalog/package.json | 2 +- packages/coding-agent/CHANGELOG.md | 2 + packages/coding-agent/package.json | 2 +- packages/collab-web/CHANGELOG.md | 2 + packages/hashline/package.json | 2 +- packages/mnemopi/package.json | 2 +- packages/natives/native/index.d.ts | 2 +- packages/natives/native/index.js | 2 +- packages/natives/package.json | 2 +- packages/snapcompact/package.json | 2 +- packages/stats/package.json | 2 +- packages/swarm-extension/package.json | 2 +- packages/tui/package.json | 2 +- packages/utils/package.json | 2 +- packages/wire/package.json | 2 +- 24 files changed, 70 insertions(+), 62 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 9e58de330..417c9eefa 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1769,9 +1769,9 @@ checksum = "88904434abc2901f197fe8cc55f0445e7ded921dba5911dad2e2b39b48e663c4" [[package]] name = "memmap2" -version = "0.9.10" +version = "0.9.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714098028fe011992e1c3962653c96b2d578c4b4bce9036e15ff220319b1e0e3" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" dependencies = [ "libc", ] @@ -2320,7 +2320,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "16.1.15" +version = "16.1.16" dependencies = [ "anyhow", "ast-grep-core", @@ -2390,7 +2390,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "16.1.15" +version = "16.1.16" dependencies = [ "async-trait", "libc", @@ -2402,7 +2402,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "16.1.15" +version = "16.1.16" dependencies = [ "anyhow", "arboard", @@ -2450,7 +2450,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "16.1.15" +version = "16.1.16" dependencies = [ "anyhow", "brush-builtins", diff --git a/Cargo.toml b/Cargo.toml index 4b11b31d9..6210ff090 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "16.1.15" +version = "16.1.16" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index d08f3adb0..8d2db424d 100644 --- a/bun.lock +++ b/bun.lock @@ -21,7 +21,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "16.1.15", + "version": "16.1.16", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -39,7 +39,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "16.1.15", + "version": "16.1.16", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-catalog": "catalog:", @@ -55,7 +55,7 @@ }, "packages/catalog": { "name": "@oh-my-pi/pi-catalog", - "version": "16.1.15", + "version": "16.1.16", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -69,7 +69,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "16.1.15", + "version": "16.1.16", "bin": { "omp": "src/cli.ts", }, @@ -137,7 +137,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "16.1.15", + "version": "16.1.16", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -148,7 +148,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "16.1.15", + "version": "16.1.16", "bin": { "mnemopi": "src/cli.ts", }, @@ -174,7 +174,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "16.1.15", + "version": "16.1.16", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -182,7 +182,7 @@ }, "packages/snapcompact": { "name": "@oh-my-pi/snapcompact", - "version": "16.1.15", + "version": "16.1.16", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -195,7 +195,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "16.1.15", + "version": "16.1.16", "bin": { "omp-stats": "./src/index.ts", }, @@ -221,7 +221,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "16.1.15", + "version": "16.1.16", "bin": { "omp-swarm": "src/cli.ts", }, @@ -247,7 +247,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "16.1.15", + "version": "16.1.16", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -288,7 +288,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "16.1.15", + "version": "16.1.16", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "handlebars": "catalog:", @@ -301,7 +301,7 @@ }, "packages/wire": { "name": "@oh-my-pi/pi-wire", - "version": "16.1.15", + "version": "16.1.16", "devDependencies": { "@types/bun": "catalog:", }, @@ -338,18 +338,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.1.15", - "@oh-my-pi/omp-stats": "16.1.15", - "@oh-my-pi/pi-agent-core": "16.1.15", - "@oh-my-pi/pi-ai": "16.1.15", - "@oh-my-pi/pi-catalog": "16.1.15", - "@oh-my-pi/pi-coding-agent": "16.1.15", - "@oh-my-pi/pi-mnemopi": "16.1.15", - "@oh-my-pi/pi-natives": "16.1.15", - "@oh-my-pi/pi-tui": "16.1.15", - "@oh-my-pi/pi-utils": "16.1.15", - "@oh-my-pi/pi-wire": "16.1.15", - "@oh-my-pi/snapcompact": "16.1.15", + "@oh-my-pi/hashline": "16.1.16", + "@oh-my-pi/omp-stats": "16.1.16", + "@oh-my-pi/pi-agent-core": "16.1.16", + "@oh-my-pi/pi-ai": "16.1.16", + "@oh-my-pi/pi-catalog": "16.1.16", + "@oh-my-pi/pi-coding-agent": "16.1.16", + "@oh-my-pi/pi-mnemopi": "16.1.16", + "@oh-my-pi/pi-natives": "16.1.16", + "@oh-my-pi/pi-tui": "16.1.16", + "@oh-my-pi/pi-utils": "16.1.16", + "@oh-my-pi/pi-wire": "16.1.16", + "@oh-my-pi/snapcompact": "16.1.16", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -969,7 +969,7 @@ "chalk": ["chalk@5.6.2", "", {}, "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA=="], - "chardet": ["chardet@2.1.1", "", {}, "sha512-PsezH1rqdV9VvyNhxxOW32/d75r01NY7TQCmOqomRo15ZSOKbpTFVsfjghxo6JloQUCGnH4k1LGu0R4yCLlWQQ=="], + "chardet": ["chardet@2.2.0", "", {}, "sha512-rddelWYNPRrXq6PtNEN2S3f6t9ILzvqaN5pVgi4kqt9jHQaXIial9PznB5iSPVlQSLNaaH22ItWz3EJtQ10+OA=="], "chart.js": ["chart.js@4.5.1", "", { "dependencies": { "@kurkle/color": "^0.3.0" } }, "sha512-GIjfiT9dbmHRiYi6Nl2yFCq7kkwdkp1W/lp2J99rX0yo9tgJGn3lKQATztIjb5tVtevcBtIdICNWqlq5+E8/Pw=="], @@ -1323,7 +1323,7 @@ "scheduler": ["scheduler@0.27.0", "", {}, "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q=="], - "semver": ["semver@7.8.4", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-rUCObTnP32Q08R2uuIrt7r9PlEonuTmtuXYcW6s5kjdlj3xbnwe+21yXptAUYcMAABLkYYTtnmzb3w3EDZfueA=="], + "semver": ["semver@7.8.5", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA=="], "semver-compare": ["semver-compare@1.0.0", "", {}, "sha512-YM3/ITh2MJ5MtzaM429anh+x2jiLVjqILF4m4oyQB18W7Ggea7BfqdH/wGMK7dDiMghv/6WG7znWMwUDzJiXow=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index bc2b7a5be..3ebeea9f9 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -172,7 +172,7 @@ fn create_windows_napi_tokio_runtime() -> Option { /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV16_1_15")] +#[napi(js_name = "__piNativesV16_1_16")] pub const fn pi_natives_version_sentinel() {} /// Native module entry point: install crash diagnostics before any tool can diff --git a/package.json b/package.json index a7823b0f4..772ce431e 100644 --- a/package.json +++ b/package.json @@ -25,18 +25,18 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "16.1.15", - "@oh-my-pi/omp-stats": "16.1.15", - "@oh-my-pi/pi-agent-core": "16.1.15", - "@oh-my-pi/pi-ai": "16.1.15", - "@oh-my-pi/pi-catalog": "16.1.15", - "@oh-my-pi/pi-coding-agent": "16.1.15", - "@oh-my-pi/pi-mnemopi": "16.1.15", - "@oh-my-pi/pi-natives": "16.1.15", - "@oh-my-pi/pi-tui": "16.1.15", - "@oh-my-pi/pi-utils": "16.1.15", - "@oh-my-pi/pi-wire": "16.1.15", - "@oh-my-pi/snapcompact": "16.1.15", + "@oh-my-pi/hashline": "16.1.16", + "@oh-my-pi/omp-stats": "16.1.16", + "@oh-my-pi/pi-agent-core": "16.1.16", + "@oh-my-pi/pi-ai": "16.1.16", + "@oh-my-pi/pi-catalog": "16.1.16", + "@oh-my-pi/pi-coding-agent": "16.1.16", + "@oh-my-pi/pi-mnemopi": "16.1.16", + "@oh-my-pi/pi-natives": "16.1.16", + "@oh-my-pi/pi-tui": "16.1.16", + "@oh-my-pi/pi-utils": "16.1.16", + "@oh-my-pi/pi-wire": "16.1.16", + "@oh-my-pi/snapcompact": "16.1.16", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 3c2e82d06..04ad9fb1d 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.1.16] - 2026-06-23 + ### Added - Added `generateHandoffFromContext(context, model, options)` to `@oh-my-pi/pi-agent-core/compaction`: runs the handoff oneshot against a fully-built provider `Context` (system prompt, normalized tools, transformed history, trailing handoff prompt) with `streamOptions` mirroring the live turn's cache routing, so a host that owns the transform pipeline can make the handoff request share the prompt cache the main turn populated. `generateHandoff(messages, …)` is unchanged and now delegates to it. diff --git a/packages/agent/package.json b/packages/agent/package.json index 54f511099..784534b23 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "16.1.15", + "version": "16.1.16", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index dfb4ed404..34c31852e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.1.16] - 2026-06-23 + ### Fixed - Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. Injected SDK clients, GitHub Copilot's Anthropic proxy, and Vertex rawPredict are excluded because this code path cannot add the beta to caller-owned clients, Copilot strips Anthropic betas and demotes thinking blocks to text upstream, and Vertex expects betas in the JSON body rather than the Anthropic HTTP beta header. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288)) diff --git a/packages/ai/package.json b/packages/ai/package.json index 6af7452c8..a218534ce 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "16.1.15", + "version": "16.1.16", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/catalog/package.json b/packages/catalog/package.json index 40462bd30..9b52cab77 100644 --- a/packages/catalog/package.json +++ b/packages/catalog/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-catalog", - "version": "16.1.15", + "version": "16.1.16", "description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 5f093ed04..1bff46648 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.1.16] - 2026-06-23 + ### Breaking Changes - Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index d0ab7e8d2..905474e2d 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "16.1.15", + "version": "16.1.16", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/collab-web/CHANGELOG.md b/packages/collab-web/CHANGELOG.md index 981da4464..02e9b21e2 100644 --- a/packages/collab-web/CHANGELOG.md +++ b/packages/collab-web/CHANGELOG.md @@ -2,6 +2,8 @@ ## [Unreleased] +## [16.1.16] - 2026-06-23 + ### Added - Added support for Ruby and Julia code cells in the eval tool diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 69e8ee64e..f245750b7 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "16.1.15", + "version": "16.1.16", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 42f5552bb..f9504a6f5 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "16.1.15", + "version": "16.1.16", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 4a6041195..87eda6161 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -162,7 +162,7 @@ export declare function __ompInstallTokioRuntime(): void * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV16_1_15(): void +export declare function __piNativesV16_1_16(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index e62fde553..c33c65ffc 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -24,7 +24,7 @@ export const Shell = nativeBindings.Shell; // functions export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime; -export const __piNativesV16_1_15 = nativeBindings.__piNativesV16_1_15; +export const __piNativesV16_1_16 = nativeBindings.__piNativesV16_1_16; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 12f7e847e..be1233873 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "16.1.15", + "version": "16.1.16", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/snapcompact/package.json b/packages/snapcompact/package.json index fc5029378..5457a81ce 100644 --- a/packages/snapcompact/package.json +++ b/packages/snapcompact/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/snapcompact", - "version": "16.1.15", + "version": "16.1.16", "description": "Bitmap-frame context compression for vision-capable LLMs", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/stats/package.json b/packages/stats/package.json index e4233ed07..7a3f93815 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "16.1.15", + "version": "16.1.16", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index 073f5ffb1..1d68a3793 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "16.1.15", + "version": "16.1.16", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/package.json b/packages/tui/package.json index 641bb284b..c2dc0a43d 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "16.1.15", + "version": "16.1.16", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/package.json b/packages/utils/package.json index 3bd2b4f5e..a13c66759 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "16.1.15", + "version": "16.1.16", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/wire/package.json b/packages/wire/package.json index 103b11ce4..1ac0c94f8 100644 --- a/packages/wire/package.json +++ b/packages/wire/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-wire", - "version": "16.1.15", + "version": "16.1.16", "description": "Shared wire protocol types for Oh My Pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", From af2e53e0701a6392d6f8853d0d0fd6c44c2af760 Mon Sep 17 00:00:00 2001 From: can1357 Date: Tue, 23 Jun 2026 09:45:11 +0200 Subject: [PATCH 43/43] fix(test): align :async: background-retention test with nohup reparenting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `nohup cmd &` is now a transparent background wrapper that double-forks the operand so it reparents to init (commit 00dcd54597). The shell only tracks the short-lived intermediate fork, so `$!` is no longer the surviving process — the prior test read `$!`, then `process.kill(pid, 0)` checked an already-reaped pid and failed on Linux (the failing CI job). Split into two contracts: - plain `&` retention: stays a child of the shell, counted by `liveBackgroundJobCount`, kept alive by the retain map; `$!` is the real child pid we assert on. - nohup reparenting: the operand writes its own pid before `exec`ing the long sleep, and that (post-exec-stable) pid is asserted to survive across turns — independent of `$!`. --- .../coding-agent/test/bash-executor.test.ts | 55 +++++++++++++++++-- 1 file changed, 50 insertions(+), 5 deletions(-) diff --git a/packages/coding-agent/test/bash-executor.test.ts b/packages/coding-agent/test/bash-executor.test.ts index 8f8cebec0..db46ac62a 100644 --- a/packages/coding-agent/test/bash-executor.test.ts +++ b/packages/coding-agent/test/bash-executor.test.ts @@ -974,7 +974,7 @@ describe("executeBash :async: background retention", () => { }); it.skipIf(process.platform === "win32")( - "keeps a per-job :async: shell's background process alive across turns", + "keeps a per-job :async: shell's plain-`&` background process alive across turns", async () => { const pidFile = path.join(tmp, "pid"); const sleepBin = fs.existsSync("/bin/sleep") ? "/bin/sleep" : "sleep"; @@ -982,10 +982,11 @@ describe("executeBash :async: background retention", () => { try { // A per-job `:async:` key: its shell is removed from the reuse map at // teardown, which would SIGKILL the backgrounded child (kill-on-drop). - // The retain logic keeps the shell alive while a background process is - // still running. `$!` is the external child's pid (nohup is a - // transparent background wrapper). - const res = await executeBash(`nohup ${sleepBin} 30 >/dev/null 2>&1 & echo $! > ${shellQuote(pidFile)}`, { + // A plain `&` job stays a child of the shell, so `liveBackgroundJobCount` + // sees it and the retain logic keeps the shell alive while the child + // runs. `$!` is the external child's own pid (no transparent wrapper to + // unwrap), so it is the process we assert on. + const res = await executeBash(`${sleepBin} 30 >/dev/null 2>&1 & echo $! > ${shellQuote(pidFile)}`, { sessionKey: "retain-probe:async:job1", cwd: tmp, }); @@ -1012,4 +1013,48 @@ describe("executeBash :async: background retention", () => { } }, ); + + it.skipIf(process.platform === "win32")( + "keeps a nohup-detached background process alive across turns (reparenting)", + async () => { + const pidFile = path.join(tmp, "nohup-pid"); + const sleepBin = fs.existsSync("/bin/sleep") ? "/bin/sleep" : "sleep"; + let pid: number | undefined; + try { + // `nohup cmd &` is a transparent background wrapper: brush unwraps it and + // double-forks the operand so it reparents to init and survives teardown + // independently of the retain map. The shell only ever tracked the + // short-lived intermediate fork, so `$!` is NOT the surviving process — + // the operand writes its own pid before `exec`ing the long sleep, and + // that pid (unchanged across exec) is the one we assert stays alive. + const operand = `echo $$ > ${pidFile}; exec ${sleepBin} 30`; + const res = await executeBash(`nohup sh -c ${shellQuote(operand)} >/dev/null 2>&1 &`, { + sessionKey: "reparent-probe:async:job1", + cwd: tmp, + }); + expect(res.cancelled).toBe(false); + + await pollUntil(() => fs.existsSync(pidFile), Date.now() + 4000); + pid = Number.parseInt(fs.readFileSync(pidFile, "utf8").trim(), 10); + expect(Number.isInteger(pid)).toBe(true); + + // A later turn on a different per-job shell must not have killed it. + await executeBash("true", { sessionKey: "reparent-probe:async:job2", cwd: tmp }); + + let alive = true; + try { + process.kill(pid, 0); + } catch { + alive = false; + } + expect(alive).toBe(true); + } finally { + if (pid !== undefined) { + try { + process.kill(pid, "SIGKILL"); + } catch {} + } + } + }, + ); });