fix(coding-agent/tools): enabled live stdout streaming for running cells

- Update the eval tool to stream stdout chunks directly into the active cell's output buffer while the process is still running.
- Prevent long-running cells from appearing empty in the UI by surfacing incremental output before the backend resolves.
- Add regression tests to ensure streamed output is captured mid-execution and reconciled with final results.
This commit is contained in:
can1357
2026-06-23 01:46:45 +02:00
parent 6ff37e346a
commit c4e23fed15
3 changed files with 102 additions and 0 deletions
+2
View File
@@ -28,6 +28,8 @@
### Fixed
- Fixed the `eval` tool card not streaming a still-running cell's stdout: a long-running cell (e.g. a `time.sleep()` monitor loop) showed nothing until it returned or was interrupted, then dumped everything at once. The renderer draws cell output from `details.cells[i].output`, which was only populated after `backend.execute()` resolved — live stdout streamed into the transient result `content` tail (and `renderContext.output`), which the per-cell render branch ignores. Streamed chunks now append to the active cell's `output` (a dedicated per-cell tail buffer, capped like the aggregate) as they arrive, so the card shows progress live; on completion the authoritative full output overwrites the live tail. `log()`/`phase()`/`display()` and status ops were unaffected because they already stream via the status channel.
- Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path.
- Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default.
- Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners
+13
View File
@@ -458,6 +458,13 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
status: "pending",
}));
const cellOutputs: string[] = [];
// The cell currently inside backend.execute(). Streamed stdout is
// appended to its rendered `output` live so a long-running cell (e.g. a
// sleep loop) shows progress instead of nothing until it returns. A
// dedicated per-cell tail buffer keeps attribution correct and avoids
// double-counting against the aggregate `tailBuffer`; on completion the
// authoritative `cellResult.output` (below) overwrites this live tail.
let activeLiveCell: { result: EvalCellResult; buf: TailBuffer } | undefined;
const appendTail = (text: string) => {
tailBuffer.append(text);
@@ -507,6 +514,10 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
maxColumns: resolveOutputMaxColumns(session.settings),
onChunk: chunk => {
appendTail(chunk);
if (activeLiveCell) {
activeLiveCell.buf.append(chunk);
activeLiveCell.result.output = activeLiveCell.buf.text();
}
pushUpdate();
},
});
@@ -534,6 +545,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
cellResult.statusEvents = undefined;
cellResult.exitCode = undefined;
cellResult.durationMs = undefined;
activeLiveCell = { result: cellResult, buf: new TailBuffer(DEFAULT_MAX_BYTES * 2) };
pushUpdate();
const startTime = Date.now();
@@ -567,6 +579,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
});
} finally {
idle.dispose();
activeLiveCell = undefined;
}
const durationMs = Date.now() - startTime;
@@ -0,0 +1,87 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import * as evalIndex from "@oh-my-pi/pi-coding-agent/eval";
import type { EvalToolDetails } from "@oh-my-pi/pi-coding-agent/eval/types";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { EvalTool } from "@oh-my-pi/pi-coding-agent/tools/eval";
function makeSession(): ToolSession {
return {
cwd: "/tmp/eval-test",
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => null,
settings: Settings.isolated(),
};
}
function baseResult(overrides: Record<string, unknown> = {}) {
return {
output: "",
exitCode: 0,
cancelled: false,
truncated: false,
artifactId: undefined,
totalLines: 0,
totalBytes: 0,
outputLines: 0,
outputBytes: 0,
displayOutputs: [] as unknown[],
...overrides,
};
}
/**
* Defends the contract that stdout streamed by a still-running cell lands in the
* running cell's rendered `output` *before* `backend.execute()` returns — so a
* long-running cell (e.g. a `time.sleep()` monitor loop) shows progress live in
* the eval card instead of dumping everything at once on completion/interrupt.
*
* The eval card renderer draws cell output from `details.cells[i].output`; if
* that field is only filled after the backend resolves, the card stays blank for
* the whole run. This pins the live bridge that keeps it populated mid-flight.
*/
describe("EvalTool live stdout streaming", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("populates the running cell's output with streamed chunks before the cell returns", async () => {
const updates: EvalToolDetails[] = [];
vi.spyOn(evalIndex.jsBackend, "execute").mockImplementation((async (
_code: string,
options: { onChunk?: (chunk: string) => void },
) => {
// Emit a chunk while the cell is still running, mirroring a print()
// before a long sleep. The host's onChunk path runs synchronously.
options.onChunk?.("tick 1\n");
return baseResult({ output: "tick 1\ntick 2\n" });
}) as never);
const tool = new EvalTool(makeSession());
const result = await tool.execute(
"call-stream",
{ language: "js", code: "for (let i = 0; i < 2; i++) print('tick ' + i)" },
undefined,
update => {
if (update.details) updates.push(update.details as EvalToolDetails);
},
);
// A snapshot taken while the cell was still running carried the streamed
// chunk — proving the output bridge fires mid-execution, not just at the end.
const liveRunning = updates.find(
d => d.cells?.[0]?.status === "running" && (d.cells?.[0]?.output ?? "").includes("tick 1"),
);
expect(liveRunning).toBeDefined();
// The live snapshot shows only what streamed so far, not the post-return total.
expect(liveRunning?.cells?.[0]?.output).not.toContain("tick 2");
// Completion still overwrites with the authoritative full output.
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
expect(text).toContain("tick 1");
expect(text).toContain("tick 2");
expect(result.details?.cells?.[0]?.status).toBe("complete");
expect(result.details?.cells?.[0]?.output).toContain("tick 2");
});
});