fix(coding-agent/tools): enabled live stdout streaming for running cells
- Update the eval tool to stream stdout chunks directly into the active cell's output buffer while the process is still running. - Prevent long-running cells from appearing empty in the UI by surfacing incremental output before the backend resolves. - Add regression tests to ensure streamed output is captured mid-execution and reconciled with final results.
This commit is contained in:
@@ -28,6 +28,8 @@
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed the `eval` tool card not streaming a still-running cell's stdout: a long-running cell (e.g. a `time.sleep()` monitor loop) showed nothing until it returned or was interrupted, then dumped everything at once. The renderer draws cell output from `details.cells[i].output`, which was only populated after `backend.execute()` resolved — live stdout streamed into the transient result `content` tail (and `renderContext.output`), which the per-cell render branch ignores. Streamed chunks now append to the active cell's `output` (a dedicated per-cell tail buffer, capped like the aggregate) as they arrive, so the card shows progress live; on completion the authoritative full output overwrites the live tail. `log()`/`phase()`/`display()` and status ops were unaffected because they already stream via the status channel.
|
||||
|
||||
- Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path.
|
||||
- Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default.
|
||||
- Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners
|
||||
|
||||
@@ -458,6 +458,13 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
status: "pending",
|
||||
}));
|
||||
const cellOutputs: string[] = [];
|
||||
// The cell currently inside backend.execute(). Streamed stdout is
|
||||
// appended to its rendered `output` live so a long-running cell (e.g. a
|
||||
// sleep loop) shows progress instead of nothing until it returns. A
|
||||
// dedicated per-cell tail buffer keeps attribution correct and avoids
|
||||
// double-counting against the aggregate `tailBuffer`; on completion the
|
||||
// authoritative `cellResult.output` (below) overwrites this live tail.
|
||||
let activeLiveCell: { result: EvalCellResult; buf: TailBuffer } | undefined;
|
||||
|
||||
const appendTail = (text: string) => {
|
||||
tailBuffer.append(text);
|
||||
@@ -507,6 +514,10 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
maxColumns: resolveOutputMaxColumns(session.settings),
|
||||
onChunk: chunk => {
|
||||
appendTail(chunk);
|
||||
if (activeLiveCell) {
|
||||
activeLiveCell.buf.append(chunk);
|
||||
activeLiveCell.result.output = activeLiveCell.buf.text();
|
||||
}
|
||||
pushUpdate();
|
||||
},
|
||||
});
|
||||
@@ -534,6 +545,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
cellResult.statusEvents = undefined;
|
||||
cellResult.exitCode = undefined;
|
||||
cellResult.durationMs = undefined;
|
||||
activeLiveCell = { result: cellResult, buf: new TailBuffer(DEFAULT_MAX_BYTES * 2) };
|
||||
pushUpdate();
|
||||
|
||||
const startTime = Date.now();
|
||||
@@ -567,6 +579,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
|
||||
});
|
||||
} finally {
|
||||
idle.dispose();
|
||||
activeLiveCell = undefined;
|
||||
}
|
||||
const durationMs = Date.now() - startTime;
|
||||
|
||||
|
||||
@@ -0,0 +1,87 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import * as evalIndex from "@oh-my-pi/pi-coding-agent/eval";
|
||||
import type { EvalToolDetails } from "@oh-my-pi/pi-coding-agent/eval/types";
|
||||
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { EvalTool } from "@oh-my-pi/pi-coding-agent/tools/eval";
|
||||
|
||||
function makeSession(): ToolSession {
|
||||
return {
|
||||
cwd: "/tmp/eval-test",
|
||||
hasUI: false,
|
||||
getSessionFile: () => null,
|
||||
getSessionSpawns: () => null,
|
||||
settings: Settings.isolated(),
|
||||
};
|
||||
}
|
||||
|
||||
function baseResult(overrides: Record<string, unknown> = {}) {
|
||||
return {
|
||||
output: "",
|
||||
exitCode: 0,
|
||||
cancelled: false,
|
||||
truncated: false,
|
||||
artifactId: undefined,
|
||||
totalLines: 0,
|
||||
totalBytes: 0,
|
||||
outputLines: 0,
|
||||
outputBytes: 0,
|
||||
displayOutputs: [] as unknown[],
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Defends the contract that stdout streamed by a still-running cell lands in the
|
||||
* running cell's rendered `output` *before* `backend.execute()` returns — so a
|
||||
* long-running cell (e.g. a `time.sleep()` monitor loop) shows progress live in
|
||||
* the eval card instead of dumping everything at once on completion/interrupt.
|
||||
*
|
||||
* The eval card renderer draws cell output from `details.cells[i].output`; if
|
||||
* that field is only filled after the backend resolves, the card stays blank for
|
||||
* the whole run. This pins the live bridge that keeps it populated mid-flight.
|
||||
*/
|
||||
describe("EvalTool live stdout streaming", () => {
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("populates the running cell's output with streamed chunks before the cell returns", async () => {
|
||||
const updates: EvalToolDetails[] = [];
|
||||
vi.spyOn(evalIndex.jsBackend, "execute").mockImplementation((async (
|
||||
_code: string,
|
||||
options: { onChunk?: (chunk: string) => void },
|
||||
) => {
|
||||
// Emit a chunk while the cell is still running, mirroring a print()
|
||||
// before a long sleep. The host's onChunk path runs synchronously.
|
||||
options.onChunk?.("tick 1\n");
|
||||
return baseResult({ output: "tick 1\ntick 2\n" });
|
||||
}) as never);
|
||||
|
||||
const tool = new EvalTool(makeSession());
|
||||
const result = await tool.execute(
|
||||
"call-stream",
|
||||
{ language: "js", code: "for (let i = 0; i < 2; i++) print('tick ' + i)" },
|
||||
undefined,
|
||||
update => {
|
||||
if (update.details) updates.push(update.details as EvalToolDetails);
|
||||
},
|
||||
);
|
||||
|
||||
// A snapshot taken while the cell was still running carried the streamed
|
||||
// chunk — proving the output bridge fires mid-execution, not just at the end.
|
||||
const liveRunning = updates.find(
|
||||
d => d.cells?.[0]?.status === "running" && (d.cells?.[0]?.output ?? "").includes("tick 1"),
|
||||
);
|
||||
expect(liveRunning).toBeDefined();
|
||||
// The live snapshot shows only what streamed so far, not the post-return total.
|
||||
expect(liveRunning?.cells?.[0]?.output).not.toContain("tick 2");
|
||||
|
||||
// Completion still overwrites with the authoritative full output.
|
||||
const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n");
|
||||
expect(text).toContain("tick 1");
|
||||
expect(text).toContain("tick 2");
|
||||
expect(result.details?.cells?.[0]?.status).toBe("complete");
|
||||
expect(result.details?.cells?.[0]?.output).toContain("tick 2");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user