diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index af44d0ce7..b2f80d4b1 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -28,6 +28,8 @@ ### Fixed +- Fixed the `eval` tool card not streaming a still-running cell's stdout: a long-running cell (e.g. a `time.sleep()` monitor loop) showed nothing until it returned or was interrupted, then dumped everything at once. The renderer draws cell output from `details.cells[i].output`, which was only populated after `backend.execute()` resolved — live stdout streamed into the transient result `content` tail (and `renderContext.output`), which the per-cell render branch ignores. Streamed chunks now append to the active cell's `output` (a dedicated per-cell tail buffer, capped like the aggregate) as they arrive, so the card shows progress live; on completion the authoritative full output overwrites the live tail. `log()`/`phase()`/`display()` and status ops were unaffected because they already stream via the status channel. + - Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path. - Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default. - Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index 45a98e873..b4b33c2cc 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -458,6 +458,13 @@ export class EvalTool implements AgentTool { status: "pending", })); const cellOutputs: string[] = []; + // The cell currently inside backend.execute(). Streamed stdout is + // appended to its rendered `output` live so a long-running cell (e.g. a + // sleep loop) shows progress instead of nothing until it returns. A + // dedicated per-cell tail buffer keeps attribution correct and avoids + // double-counting against the aggregate `tailBuffer`; on completion the + // authoritative `cellResult.output` (below) overwrites this live tail. + let activeLiveCell: { result: EvalCellResult; buf: TailBuffer } | undefined; const appendTail = (text: string) => { tailBuffer.append(text); @@ -507,6 +514,10 @@ export class EvalTool implements AgentTool { maxColumns: resolveOutputMaxColumns(session.settings), onChunk: chunk => { appendTail(chunk); + if (activeLiveCell) { + activeLiveCell.buf.append(chunk); + activeLiveCell.result.output = activeLiveCell.buf.text(); + } pushUpdate(); }, }); @@ -534,6 +545,7 @@ export class EvalTool implements AgentTool { cellResult.statusEvents = undefined; cellResult.exitCode = undefined; cellResult.durationMs = undefined; + activeLiveCell = { result: cellResult, buf: new TailBuffer(DEFAULT_MAX_BYTES * 2) }; pushUpdate(); const startTime = Date.now(); @@ -567,6 +579,7 @@ export class EvalTool implements AgentTool { }); } finally { idle.dispose(); + activeLiveCell = undefined; } const durationMs = Date.now() - startTime; diff --git a/packages/coding-agent/test/tools/eval-streaming-output.test.ts b/packages/coding-agent/test/tools/eval-streaming-output.test.ts new file mode 100644 index 000000000..b7d7af46e --- /dev/null +++ b/packages/coding-agent/test/tools/eval-streaming-output.test.ts @@ -0,0 +1,87 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import * as evalIndex from "@oh-my-pi/pi-coding-agent/eval"; +import type { EvalToolDetails } from "@oh-my-pi/pi-coding-agent/eval/types"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { EvalTool } from "@oh-my-pi/pi-coding-agent/tools/eval"; + +function makeSession(): ToolSession { + return { + cwd: "/tmp/eval-test", + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => null, + settings: Settings.isolated(), + }; +} + +function baseResult(overrides: Record = {}) { + return { + output: "", + exitCode: 0, + cancelled: false, + truncated: false, + artifactId: undefined, + totalLines: 0, + totalBytes: 0, + outputLines: 0, + outputBytes: 0, + displayOutputs: [] as unknown[], + ...overrides, + }; +} + +/** + * Defends the contract that stdout streamed by a still-running cell lands in the + * running cell's rendered `output` *before* `backend.execute()` returns — so a + * long-running cell (e.g. a `time.sleep()` monitor loop) shows progress live in + * the eval card instead of dumping everything at once on completion/interrupt. + * + * The eval card renderer draws cell output from `details.cells[i].output`; if + * that field is only filled after the backend resolves, the card stays blank for + * the whole run. This pins the live bridge that keeps it populated mid-flight. + */ +describe("EvalTool live stdout streaming", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("populates the running cell's output with streamed chunks before the cell returns", async () => { + const updates: EvalToolDetails[] = []; + vi.spyOn(evalIndex.jsBackend, "execute").mockImplementation((async ( + _code: string, + options: { onChunk?: (chunk: string) => void }, + ) => { + // Emit a chunk while the cell is still running, mirroring a print() + // before a long sleep. The host's onChunk path runs synchronously. + options.onChunk?.("tick 1\n"); + return baseResult({ output: "tick 1\ntick 2\n" }); + }) as never); + + const tool = new EvalTool(makeSession()); + const result = await tool.execute( + "call-stream", + { language: "js", code: "for (let i = 0; i < 2; i++) print('tick ' + i)" }, + undefined, + update => { + if (update.details) updates.push(update.details as EvalToolDetails); + }, + ); + + // A snapshot taken while the cell was still running carried the streamed + // chunk — proving the output bridge fires mid-execution, not just at the end. + const liveRunning = updates.find( + d => d.cells?.[0]?.status === "running" && (d.cells?.[0]?.output ?? "").includes("tick 1"), + ); + expect(liveRunning).toBeDefined(); + // The live snapshot shows only what streamed so far, not the post-return total. + expect(liveRunning?.cells?.[0]?.output).not.toContain("tick 2"); + + // Completion still overwrites with the authoritative full output. + const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); + expect(text).toContain("tick 1"); + expect(text).toContain("tick 2"); + expect(result.details?.cells?.[0]?.status).toBe("complete"); + expect(result.details?.cells?.[0]?.output).toContain("tick 2"); + }); +});