From 1d414c1383eaebb2b1ef7397cb6f69aa00471023 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 18 Jan 2026 19:27:45 +0100 Subject: [PATCH] feat: added IPython-backed Python tool with streaming output and Jupyter integration - Added IPython-backed Python tool with streaming output and image/JSON rendering. - Implemented Jupyter kernel gateway integration with WebSocket communication. - Added Python prelude with 30+ shell-like utility functions for file operations. - Migrated environment variables from PI_ to OMP_ prefix with automatic migration. - Added streaming output system with automatic spill-to-disk for large outputs. - Reorganized settings interface into behavior, tools, display, voice, status, lsp, and exa tabs. --- docs/ipy-integration-plan.md | 63 ++ docs/ipy-integration-points.md | 188 ++++ docs/ipy-kernel-protocol.md | 92 ++ docs/ipy-prelude.md | 305 ++++++ docs/ipy-tui-ux.md | 246 +++++ docs/python-dynamic-docs.md | 266 +++++ packages/ai/CHANGELOG.md | 3 + packages/ai/src/cli.ts | 2 +- packages/ai/src/index.ts | 2 + .../src/providers/openai-codex-responses.ts | 2 +- packages/ai/src/utils/migrate-env.ts | 8 + packages/ai/test/context-overflow.test.ts | 4 +- packages/ai/test/openai-codex-stream.test.ts | 12 +- packages/ai/test/stream.test.ts | 2 +- packages/coding-agent/CHANGELOG.md | 7 +- packages/coding-agent/docs/python-repl.md | 77 ++ packages/coding-agent/src/bun-imports.d.ts | 6 + .../coding-agent/src/core/bash-executor.ts | 87 +- .../src/core/python-executor-display.test.ts | 42 + .../core/python-executor-lifecycle.test.ts | 99 ++ .../src/core/python-executor-mapping.test.ts | 41 + .../src/core/python-executor-per-call.test.ts | 49 + .../src/core/python-executor-session.test.ts | 103 ++ .../core/python-executor-streaming.test.ts | 77 ++ .../src/core/python-executor-timeout.test.ts | 35 + .../core/python-executor.lifecycle.test.ts | 139 +++ .../src/core/python-executor.result.test.ts | 49 + .../src/core/python-executor.test.ts | 178 ++++ .../coding-agent/src/core/python-executor.ts | 294 ++++++ .../src/core/python-kernel-display.test.ts | 54 + .../src/core/python-kernel-env.test.ts | 138 +++ .../src/core/python-kernel-session.test.ts | 87 ++ .../src/core/python-kernel-ws.test.ts | 104 ++ .../src/core/python-kernel.lifecycle.test.ts | 256 +++++ .../src/core/python-kernel.test.ts | 535 ++++++++++ .../coding-agent/src/core/python-kernel.ts | 952 ++++++++++++++++++ .../coding-agent/src/core/python-prelude.py | 442 ++++++++ .../src/core/python-prelude.test.ts | 77 ++ .../coding-agent/src/core/python-prelude.ts | 3 + packages/coding-agent/src/core/sdk.ts | 14 + .../coding-agent/src/core/session-manager.ts | 59 +- .../src/core/settings-manager-python.test.ts | 23 + .../coding-agent/src/core/settings-manager.ts | 152 ++- .../src/core/streaming-output.test.ts | 26 + .../coding-agent/src/core/streaming-output.ts | 86 ++ .../src/core/system-prompt.python.test.ts | 17 + .../coding-agent/src/core/system-prompt.ts | 4 +- packages/coding-agent/src/core/timings.ts | 2 +- .../coding-agent/src/core/tools/index.test.ts | 73 +- packages/coding-agent/src/core/tools/index.ts | 70 +- .../src/core/tools/python-execution.test.ts | 68 ++ .../src/core/tools/python-fallback.test.ts | 72 ++ .../src/core/tools/python-renderer.test.ts | 36 + .../src/core/tools/python-tool-mode.test.ts | 43 + .../src/core/tools/python.test.ts | 120 +++ .../coding-agent/src/core/tools/python.ts | 380 +++++++ .../coding-agent/src/core/tools/renderers.ts | 2 + .../src/core/tools/task/executor.ts | 78 +- .../src/core/tools/task/worker-protocol.ts | 24 +- .../src/core/tools/task/worker.ts | 131 ++- packages/coding-agent/src/index.ts | 1 + .../interactive/components/settings-defs.ts | 643 +++++++----- .../components/settings-selector.ts | 9 +- .../interactive/components/tool-execution.ts | 7 + .../src/prompts/agents/explore.md | 4 +- .../coding-agent/src/prompts/agents/plan.md | 4 +- .../src/prompts/agents/reviewer.md | 4 +- .../coding-agent/src/prompts/agents/task.md | 2 +- .../src/prompts/system/system-prompt.md | 33 +- .../coding-agent/src/prompts/tools/python.md | 66 ++ .../test/python-tool-settings.test.ts | 87 ++ packages/tui/src/tui.ts | 2 +- 72 files changed, 7087 insertions(+), 381 deletions(-) create mode 100644 docs/ipy-integration-plan.md create mode 100644 docs/ipy-integration-points.md create mode 100644 docs/ipy-kernel-protocol.md create mode 100644 docs/ipy-prelude.md create mode 100644 docs/ipy-tui-ux.md create mode 100644 docs/python-dynamic-docs.md create mode 100644 packages/ai/src/utils/migrate-env.ts create mode 100644 packages/coding-agent/docs/python-repl.md create mode 100644 packages/coding-agent/src/core/python-executor-display.test.ts create mode 100644 packages/coding-agent/src/core/python-executor-lifecycle.test.ts create mode 100644 packages/coding-agent/src/core/python-executor-mapping.test.ts create mode 100644 packages/coding-agent/src/core/python-executor-per-call.test.ts create mode 100644 packages/coding-agent/src/core/python-executor-session.test.ts create mode 100644 packages/coding-agent/src/core/python-executor-streaming.test.ts create mode 100644 packages/coding-agent/src/core/python-executor-timeout.test.ts create mode 100644 packages/coding-agent/src/core/python-executor.lifecycle.test.ts create mode 100644 packages/coding-agent/src/core/python-executor.result.test.ts create mode 100644 packages/coding-agent/src/core/python-executor.test.ts create mode 100644 packages/coding-agent/src/core/python-executor.ts create mode 100644 packages/coding-agent/src/core/python-kernel-display.test.ts create mode 100644 packages/coding-agent/src/core/python-kernel-env.test.ts create mode 100644 packages/coding-agent/src/core/python-kernel-session.test.ts create mode 100644 packages/coding-agent/src/core/python-kernel-ws.test.ts create mode 100644 packages/coding-agent/src/core/python-kernel.lifecycle.test.ts create mode 100644 packages/coding-agent/src/core/python-kernel.test.ts create mode 100644 packages/coding-agent/src/core/python-kernel.ts create mode 100644 packages/coding-agent/src/core/python-prelude.py create mode 100644 packages/coding-agent/src/core/python-prelude.test.ts create mode 100644 packages/coding-agent/src/core/python-prelude.ts create mode 100644 packages/coding-agent/src/core/settings-manager-python.test.ts create mode 100644 packages/coding-agent/src/core/streaming-output.test.ts create mode 100644 packages/coding-agent/src/core/streaming-output.ts create mode 100644 packages/coding-agent/src/core/system-prompt.python.test.ts create mode 100644 packages/coding-agent/src/core/tools/python-execution.test.ts create mode 100644 packages/coding-agent/src/core/tools/python-fallback.test.ts create mode 100644 packages/coding-agent/src/core/tools/python-renderer.test.ts create mode 100644 packages/coding-agent/src/core/tools/python-tool-mode.test.ts create mode 100644 packages/coding-agent/src/core/tools/python.test.ts create mode 100644 packages/coding-agent/src/core/tools/python.ts create mode 100644 packages/coding-agent/src/prompts/tools/python.md create mode 100644 packages/coding-agent/test/python-tool-settings.test.ts diff --git a/docs/ipy-integration-plan.md b/docs/ipy-integration-plan.md new file mode 100644 index 000000000..b5f4d5625 --- /dev/null +++ b/docs/ipy-integration-plan.md @@ -0,0 +1,63 @@ +# IPython Kernel REPL Integration Plan + +## Assumptions +- No sandboxing or persistence beyond the current agent session. +- Local Python + ipykernel available (install step needed if missing). +- Integration follows existing `bash` tool behavior for streaming, truncation, and TUI rendering. + +## Reference Docs +- Kernel protocol + message flow: `docs/ipy-kernel-protocol.md` +- TUI streaming + truncation mapping: `docs/ipy-tui-ux.md` +- Integration points and adapter interfaces: `docs/ipy-integration-points.md` +- Python prelude helpers: `docs/ipy-prelude.md` + +## Open Decisions to Lock +- Kernel reuse policy: per-session shared kernel vs per-call kernel. Recommendation: per-session shared kernel with per-call queueing. +- Parallel tool calls: queue per kernel, spawn new kernel for concurrent calls if requested. +- Kernel restart policy: restart on crash or after N executions / memory threshold (start with crash-only). +- Startup policy: lazy init by default; optional warm start at session begin. +- Interrupt mode: prefer `interrupt_request` when supported; fall back to SIGINT. + +## Phase 1 — Kernel IPC + Executor + Tool Wiring +Goal: end-to-end `python` tool that executes code via ipykernel, streams output, and returns bash-style results. + +### TODO +- [ ] Add a kernel lifecycle module (spawn, connect, shutdown) per `docs/ipy-kernel-protocol.md`. +- [ ] Implement connection file generation with free-port allocation and HMAC signing. +- [ ] Implement ZMQ client wiring for shell/iopub/control/stdin/hb channels. +- [ ] Implement message encoding/decoding (`` framing + JSON parts + HMAC signature). +- [ ] Implement `KernelConnection.execute()` that streams IOPub messages and resolves on `status: idle`. +- [ ] Implement `python-executor.ts` mirroring `BashExecutorOptions` and `BashResult` semantics (see `docs/ipy-integration-points.md`). +- [ ] Reuse or extract `createOutputSink` logic so Python executor gets the same truncation + spill behavior as bash. +- [ ] Add `python` tool module mirroring `bash.ts` streaming/onUpdate behavior (see `docs/ipy-tui-ux.md`). +- [ ] Add tool registration and prompt template for `python` tool. +- [ ] Add settings flag to select tool exposure: bash-only, ipy-only, or both (default ipy-only). +- [ ] Auto-fallback: if kernel launch fails, expose bash tool only for that session. +- [ ] Update all prompts that reference the bash tool via template rendering so tool availability is dynamic. +- [ ] Add a Python tool prompt describing built-in helpers and shell bridge (see `docs/ipy-prelude.md`), plus matplotlib guidance (`plt.show()` headless; use `plt.savefig()` or return figure). +- [ ] Environment propagation: pass user env vars (with explicit allowlist/denylist), set PYTHONPATH, and detect venv when launching kernel. +- [ ] Dependency detection: detect `python` + `ipykernel` availability; return actionable error if missing. +- [ ] Manual validation: stdout, stderr, errors, timeouts, large output truncation, image/png, and application/json rendering. + +## Phase 2 — TUI/UX Polish + Reliability + Tests +Goal: solid UX parity with bash tool and reliable CI coverage. + +### TODO +- [ ] Add a dedicated renderer or reuse `bashToolRenderer` with Python-specific label/metadata. +- [ ] Ensure expand/collapse behavior uses `details.fullOutput` and `truncateTail` consistently. +- [ ] Implement queueing for concurrent calls to the same kernel; add a config option for per-call kernels. +- [ ] Error recovery: detect kernel crash/ZMQ drop and auto-restart once per session. +- [ ] stdin handling: on `stdin_request`, return a tool error after timeout with guidance (no silent hangs). +- [ ] Output types: support `text/plain`, render `image/png` (kitty images), and render `application/json` as a collapsible property tree (reuse task tool tree renderer patterns). If only `text/html` is present, convert to markdown and emit as text output. +- [ ] Kernel selection: document a future `kernel` parameter and strategy for multiple Python versions. +- [ ] Startup optimization: measure kernel spawn latency; if >2s, consider warm start on session init or prelude injection after kernel_info. +- [ ] Heartbeat monitoring: periodic HB ping (e.g., every 5s) to detect silent kernel death; restart on failure. +- [ ] Tests: mock kernel messages for unit tests; add a dev logging flag for IPC trace output. +- [ ] Add developer docs for kernel requirements, shell bridge behavior, and troubleshooting. + +## Exit Criteria +- `python` tool can execute multi-line code and stream output in TUI. +- Output truncation and full-output expansion match bash tool behavior. +- Timeouts/cancellations produce consistent error messaging. +- Kernel crash detection triggers automatic restart. +- Tests cover core execution and error paths. diff --git a/docs/ipy-integration-points.md b/docs/ipy-integration-points.md new file mode 100644 index 000000000..97cc64618 --- /dev/null +++ b/docs/ipy-integration-points.md @@ -0,0 +1,188 @@ +# Python REPL via IPython kernel: plan + +## Assumptions +- No security/sandboxing requirements. +- No persistence across sessions; kernel lifetime is scoped to the agent session (or explicit reset/close). +- REPL is per agent session (not shared across agents unless explicitly wired). +- Python available locally; IPython/Jupyter kernel deps can be added as runtime deps later. +- Streaming output to TUI should mirror bash tool behavior (chunked, truncation, expandable output). + +## Goals +- Add a Python REPL tool backed by IPython kernels. +- Integrate with existing bash tool surface (schema/renderer/streaming/truncation). +- Reuse bash-executor patterns for streaming, cancellation, and output capture. +- Define adapter interfaces to support local and future remote kernel execution. + +## Key integration points +### `packages/coding-agent/src/core/tools/bash.ts` +- Pattern to follow: tool schema, execute handler, streaming updates, truncation, error surfacing, renderer. +- Target additions: + - A new tool module `python.ts` (or `pyrepl.ts`) mirroring bash tool structure. + - Similar `ToolDetails` with truncation metadata, fullOutputPath, fullOutput. + - Reuse `truncateTail` + TUI renderer patterns. +- Adapter use: inject a `PythonOperations` (analogous to `BashOperations`) to allow local vs remote kernel implementations. + +### `packages/coding-agent/src/core/bash-executor.ts` +- Patterns to reuse: + - `BashExecutorOptions` fields: `cwd`, `timeout`, `onChunk`, `signal`. + - `BashResult` shape for output/truncation/cancelled. + - `createOutputSink` and `pumpStream` for streaming + truncation. +- Target additions: + - A parallel module `python-executor.ts` or generic `stream-executor.ts` that wraps kernel I/O to match `BashResult` semantics. + - Option to share `createOutputSink` logic (split into shared module) if preferred. + +## Kernel lifecycle +### Lifecycle states +1. **Provisioning**: start kernel process + open IPC channels. +2. **Ready**: kernel info request succeeds; execute requests accepted. +3. **Executing**: process `execute_request` and stream outputs. +4. **Interrupted** (optional): on timeout or cancellation; send interrupt. +5. **Shutdown**: request kernel shutdown; clean IPC; kill if unresponsive. + +### Lifecycle management +- Kernel is created at tool invocation start (or on first execution if implementing a multi-call session). +- Kernel must be shut down on: + - successful completion + - error (startup failure, exec failure) + - abort signal + - timeout +- Use a `KernelController` interface to own lifecycle and cleanup. + +## IPC strategy +### IPython/Jupyter protocol +- Use ZeroMQ sockets: `shell`, `iopub`, `control`, `stdin`, `hb`. +- Send `kernel_info_request` to verify readiness. +- Execute with `execute_request` on `shell` channel. +- Stream results from `iopub`: + - `stream` (stdout/stderr) + - `execute_result` (repr, display data) + - `error` (traceback) + - `status` (idle/busy) +- Treat `status: idle` as execution completion signal. + +### Session IDs +- Generate `session_id` and `msg_id` per execute call. +- Filter iopub messages by `parent_header.msg_id` to avoid cross-talk. + +## Bun integration +### Process spawn +- Use `Bun.spawn()` for local kernel process: `python -m ipykernel_launcher -f `. +- Allocate free ports, then use `Bun.file()` and `Bun.write()` to create the connection file (JSON) before spawn. +- Use `AbortSignal` to cancel execution and trigger interrupt/termination. + +### ZMQ handling +- Use a JS ZMQ library compatible with Bun (investigate `zeromq` Bun support). +- If Bun is incompatible, consider a tiny Node sidecar for ZMQ until Bun support is verified (documented in plan, not implemented now). + +## Tool availability settings +- Add a setting to control tool exposure: `bash-only`, `ipy-only`, or `both`. +- Default: `ipy-only` with automatic fallback to `bash-only` if kernel startup fails. +- Tool registration should consult settings and runtime availability per session. + +## Shell bridge in Python +- Provide a `bash()` helper in the Python prelude that shells out via `bash -lc` and uses the snapshot path when available. +- Reuse the existing TypeScript `shell-snapshot.ts` to generate the snapshot file, then pass its path to the kernel via an env var. +- Python helper should prefer the snapshot env var if present; otherwise run plain `bash -lc`. + +## Execute flow +1. Create kernel controller (spawn process + load connection file). +2. Open ZMQ sockets and send `kernel_info_request`. +3. Send `execute_request` with code. +4. Collect output: + - `stream` → append to output + - `execute_result`/`display_data` → serialize as text/plain or JSON text + - `error` → include traceback, mark as failure +5. Complete when `status: idle` for matching `parent_header.msg_id`. +6. Apply truncation rules and return tool result. +7. Shutdown kernel. + +## Streaming to TUI +- Same mechanics as bash tool: + - incremental `onUpdate` with truncated tail + - final output includes truncation metadata and `fullOutputPath` +- Use `truncateTail` for preview + `fullOutput` in details for expansion. +- For display data, prefer text/plain; fallback to JSON string. + +## Adapter design +### Interfaces +```ts +export interface PythonKernelOperations { + startKernel(options: { cwd: string; env?: Record }): Promise; + sendExecute(handle: KernelHandle, code: string, options: ExecuteOptions): AsyncIterable; + interrupt(handle: KernelHandle): Promise; + shutdown(handle: KernelHandle): Promise; +} + +export interface KernelHandle { + id: string; + connectionInfo: KernelConnectionInfo; +} + +export interface KernelConnectionInfo { + transport: "tcp" | "ipc"; + ip: string; + shellPort: number; + iopubPort: number; + controlPort: number; + stdinPort: number; + hbPort: number; + key: string; + signatureScheme: string; // e.g. "hmac-sha256" +} + +export interface ExecuteOptions { + signal?: AbortSignal; + timeoutMs?: number; + onChunk?: (text: string) => void; // sanitized +} +``` + +### Executor adapter +- Create `executePython` that mirrors `executeBash`: + - input: code + `PythonExecutorOptions` (`cwd`, `timeout`, `signal`, `onChunk`) + - output: `PythonResult` mirroring `BashResult` +- Use a shared `createOutputSink` (move to `streaming-executor.ts`). + +### Mapping to existing executor shape +- `PythonExecutorOptions` mirrors `BashExecutorOptions` for compatibility. +- `PythonResult` mirrors `BashResult` for renderer reuse. +- Tool details/truncation identical to `bash.ts`. + +## Tool surface (schema) +- `command` → `code` (string) +- `timeout` (seconds) +- `workdir` (optional) +- `kernel` (optional): for future extension (e.g., python version); unused now. + +## Concrete implementation steps (design plan) +1. **Design shared streaming utilities** + - Factor `createOutputSink` and `pumpStream` into `stream-executor.ts`. + - Keep bash-executor using it to avoid duplication. +2. **Define kernel operations interface** + - Create `PythonKernelOperations` + `KernelHandle` types. + - Provide default local implementation using Bun + ZMQ. +3. **Implement python executor** + - `executePython(code, options)` to stream outputs and return `PythonResult`. + - Handle timeouts → interrupt kernel + annotate output. +4. **Implement python tool** + - Mirror bash tool behavior: interception not needed. + - Use `executePython` with onUpdate streaming + truncation. + - Renderer can reuse `bashToolRenderer` logic or clone with label changes. +5. **Integrate into tool registry** + - Add new tool to tools index with appropriate schema. + - Ensure tool description prompt is written (new prompt file if required). +6. **Wire into TUI** + - Ensure `tool-execution` supports expansion via details.fullOutput. +7. **Add unit tests** + - Mirror `prompt-templates` tests; add REPL output/truncation tests. + +## Notes for `docs/ipy-*.md` +Include the following sections: +- Motivation + non-goals +- Kernel lifecycle diagram +- IPC message flow (execute_request → iopub stream/error/result → status idle) +- Executor interface and data structures +- Tool schema and output format +- Timeout/cancel behavior +- TUI streaming and truncation handling +- Future extensions (persistent kernels, multiple executions, rich display handling) diff --git a/docs/ipy-kernel-protocol.md b/docs/ipy-kernel-protocol.md new file mode 100644 index 000000000..d6e480d0f --- /dev/null +++ b/docs/ipy-kernel-protocol.md @@ -0,0 +1,92 @@ +# IPython Kernel REPL Tool (coding-agent) + +## Implementation Status: COMPLETE + +The Python REPL tool has been implemented using **Jupyter Kernel Gateway** for IPC instead of direct ZeroMQ connections. This avoids Bun's incompatibility with the zeromq NAPI module (libuv `uv_async_init` not supported). + +## Architecture + +``` +TypeScript (coding-agent) + │ + ├── REST API (kernel lifecycle) + │ POST /api/kernels → create kernel + │ DELETE /api/kernels/{id} → shutdown kernel + │ POST /api/kernels/{id}/interrupt → interrupt execution + │ + └── WebSocket (message passing) + ws://host:port/api/kernels/{id}/channels + └── Multiplexed Jupyter protocol (shell, iopub, stdin, control) + │ + ▼ + Jupyter Kernel Gateway (Python process) + │ + └── ZeroMQ (internal, handled by pyzmq) + │ + ▼ + ipykernel (Python kernel) +``` + +## Key Files + +- `packages/coding-agent/src/core/python-kernel.ts` - Kernel Gateway client +- `packages/coding-agent/src/core/python-executor.ts` - Execution wrapper +- `packages/coding-agent/src/core/tools/python.ts` - Tool definition + +## Dependencies + +**Python (user must install):** +```bash +pip install jupyter_kernel_gateway ipykernel +``` + +**TypeScript:** No native dependencies. Uses standard WebSocket API. + +## WebSocket Wire Protocol + +The Jupyter Kernel Gateway uses a binary WebSocket protocol: + +``` +┌─────────────┬──────────┬──────────┬─────┬─────────┬──────────┬─────┐ +│ offset_count│ offset_0 │ offset_1 │ ... │ msg │ buffer_0 │ ... │ +│ (4 bytes) │ (4 bytes)│ (4 bytes)│ │ (JSON) │ (binary) │ │ +└─────────────┴──────────┴──────────┴─────┴─────────┴──────────┴─────┘ +``` + +Message JSON structure: +```json +{ + "channel": "shell|iopub|stdin|control", + "header": { "msg_id", "session", "username", "date", "msg_type", "version" }, + "parent_header": {}, + "metadata": {}, + "content": {} +} +``` + +## Kernel Lifecycle + +1. **Start Gateway**: Spawn `python -m jupyter_kernel_gateway --port=` +2. **Wait for Ready**: Poll `GET /api/kernelspecs` until 200 +3. **Create Kernel**: `POST /api/kernels` with `{ "name": "python3" }` +4. **Connect WebSocket**: `ws://host:port/api/kernels/{id}/channels` +5. **Execute Code**: Send `execute_request` on shell channel +6. **Receive Output**: Handle `stream`, `execute_result`, `display_data`, `error` on iopub +7. **Shutdown**: `DELETE /api/kernels/{id}`, then kill gateway process + +## Settings + +| Setting | Values | Default | Description | +|---------|--------|---------|-------------| +| `python.toolMode` | `ipy-only`, `bash-only`, `both` | `ipy-only` | How Python code is executed | +| `python.kernelMode` | `session`, `per-call` | `session` | Whether to keep kernel alive | + +## Previous Approach (Deprecated) + +The original plan used direct ZeroMQ connections from TypeScript to ipykernel. This was abandoned because: + +1. The `zeromq` npm package uses NAPI with libuv internals +2. Bun doesn't support `uv_async_init` (see https://github.com/oven-sh/bun/issues/18546) +3. No pure-JS ZeroMQ implementation exists + +Jupyter Kernel Gateway solves this by handling ZeroMQ internally (via Python's pyzmq) and exposing a standard HTTP/WebSocket interface. diff --git a/docs/ipy-prelude.md b/docs/ipy-prelude.md new file mode 100644 index 000000000..75fdea488 --- /dev/null +++ b/docs/ipy-prelude.md @@ -0,0 +1,305 @@ +# IPython REPL Prelude + +## Purpose +Define the Python-side helpers injected into the IPython kernel so the agent can perform common shell/file tasks without needing the bash tool. The goal is to cover the frequent bash use cases (file ops, grep/find, git status, basic exec), while keeping outputs visible in the tool stream. + +## Design Principles +- **Visibility**: helpers must print concise, useful output by default. +- **Predictability**: return structured values *and* print summaries. +- **Safety**: no sandboxing; focus on correctness and clarity. +- **Parity**: cover the common bash operations seen in sessions. +- **Minimal state**: no persistence beyond kernel lifetime. + +## Prelude Injection +The prelude is executed once per kernel at startup. It should: +- import common modules +- define helper functions +- set a small display utility for consistent output + +### Core Imports +```python +from pathlib import Path +import os, sys, re, json, shutil, subprocess, glob, textwrap +from datetime import datetime +``` + +## Helpers to Expose + +### 1) Working directory +```python +def pwd() -> Path: + """Print and return current working directory.""" + p = Path.cwd() + print(str(p)) + return p + + +def cd(path: str | Path) -> Path: + """Change directory and print the new cwd.""" + p = Path(path).expanduser().resolve() + os.chdir(p) + print(str(p)) + return p +``` + +### 2) Environment +```python +def env(key: str | None = None, value: str | None = None): + """Get/set environment variables. + + - env() -> prints all env vars (sorted) and returns dict + - env("PATH") -> prints and returns value + - env("FOO", "bar") -> sets and prints assignment + """ + if key is None: + items = dict(sorted(os.environ.items())) + for k, v in items.items(): + print(f"{k}={v}") + print(f"[env] {len(items)} variables") + return items + if value is not None: + os.environ[key] = value + print(f"{key}={value}") + return value + val = os.environ.get(key) + print(f"{key}={val}") + return val +``` + +### 3) File read/write +```python +def read(path: str | Path, *, limit: int | None = None) -> str: + """Read file contents. Prints a short preview + length.""" + p = Path(path) + data = p.read_text(encoding="utf-8") + if limit is not None: + preview = data[:limit] + print(preview) + print(f"[read {len(data)} chars from {p}]") + else: + print(data) + print(f"[read {len(data)} chars from {p}]") + return data + + +def write(path: str | Path, content: str) -> Path: + """Write file contents (create parents). Prints bytes written.""" + p = Path(path) + p.parent.mkdir(parents=True, exist_ok=True) + p.write_text(content, encoding="utf-8") + print(f"[wrote {len(content)} chars to {p}]") + return p + + +def append(path: str | Path, content: str) -> Path: + """Append to file. Prints bytes appended.""" + p = Path(path) + p.parent.mkdir(parents=True, exist_ok=True) + with p.open("a", encoding="utf-8") as f: + f.write(content) + print(f"[appended {len(content)} chars to {p}]") + return p +``` + +### 4) File ops (mkdir/rm/mv/cp) +```python +def mkdir(path: str | Path) -> Path: + p = Path(path) + p.mkdir(parents=True, exist_ok=True) + print(f"[mkdir] {p}") + return p + + +def rm(path: str | Path, *, recursive: bool = False) -> None: + p = Path(path) + if p.is_dir() and recursive: + shutil.rmtree(p) + print(f"[rm -r] {p}") + elif p.exists(): + p.unlink() + print(f"[rm] {p}") + else: + print(f"[rm] {p} (missing)") + + +def mv(src: str | Path, dst: str | Path) -> Path: + src_p = Path(src) + dst_p = Path(dst) + dst_p.parent.mkdir(parents=True, exist_ok=True) + shutil.move(str(src_p), str(dst_p)) + print(f"[mv] {src_p} -> {dst_p}") + return dst_p + + +def cp(src: str | Path, dst: str | Path) -> Path: + src_p = Path(src) + dst_p = Path(dst) + dst_p.parent.mkdir(parents=True, exist_ok=True) + if src_p.is_dir(): + shutil.copytree(src_p, dst_p, dirs_exist_ok=True) + else: + shutil.copy2(src_p, dst_p) + print(f"[cp] {src_p} -> {dst_p}") + return dst_p +``` + +### 5) Listing / find / glob +```python +def ls(path: str | Path = ".") -> list[Path]: + p = Path(path) + items = sorted(p.iterdir()) + for item in items: + suffix = "/" if item.is_dir() else "" + print(f"{item.name}{suffix}") + print(f"[ls] {len(items)} entries in {p}") + return items + + +def find(pattern: str, path: str | Path = ".", *, files_only: bool = True) -> list[Path]: + """Recursive glob find. Defaults to files only.""" + p = Path(path) + matches = [] + for m in p.rglob(pattern): + if files_only and m.is_dir(): + continue + matches.append(m) + matches = sorted(matches) + for m in matches: + print(str(m)) + print(f"[find] {len(matches)} matches for '{pattern}' in {p}") + return matches +``` + +### 6) Grep +```python +def grep(pattern: str, path: str | Path, *, ignore_case: bool = False, context: int = 0) -> list[tuple[int, str]]: + """Grep a single file.""" + flags = re.IGNORECASE if ignore_case else 0 + rx = re.compile(pattern, flags) + p = Path(path) + lines = p.read_text(encoding="utf-8").splitlines() + hits: list[tuple[int, str]] = [] + for i, line in enumerate(lines, 1): + if rx.search(line): + hits.append((i, line)) + print(f"{i}: {line}") + if context: + start = max(0, i - 1 - context) + end = min(len(lines), i - 1 + context + 1) + for j in range(start, end): + if j + 1 == i: + continue + print(f"{j+1}- {lines[j]}") + print(f"[grep] {len(hits)} matches in {p}") + return hits + + +def rgrep(pattern: str, path: str | Path = ".", *, glob_pattern: str = "*", ignore_case: bool = False) -> list[tuple[Path, int, str]]: + """Recursive grep across files matching glob_pattern.""" + flags = re.IGNORECASE if ignore_case else 0 + rx = re.compile(pattern, flags) + base = Path(path) + hits: list[tuple[Path, int, str]] = [] + for file_path in base.rglob(glob_pattern): + if file_path.is_dir(): + continue + try: + lines = file_path.read_text(encoding="utf-8").splitlines() + except Exception: + continue + for i, line in enumerate(lines, 1): + if rx.search(line): + hits.append((file_path, i, line)) + print(f"{file_path}:{i}: {line}") + print(f"[rgrep] {len(hits)} matches in {base}") + return hits +``` + +### 7) Text utilities (head/tail/sed-like replace) +```python +def head(text: str, n: int = 10) -> str: + lines = text.splitlines()[:n] + out = "\n".join(lines) + print(out) + print(f"[head] {len(lines)} lines") + return out + + +def tail(text: str, n: int = 10) -> str: + lines = text.splitlines()[-n:] + out = "\n".join(lines) + print(out) + print(f"[tail] {len(lines)} lines") + return out + + +def replace(path: str | Path, pattern: str, repl: str, *, regex: bool = False) -> int: + p = Path(path) + data = p.read_text(encoding="utf-8") + if regex: + new, count = re.subn(pattern, repl, data) + else: + new = data.replace(pattern, repl) + count = data.count(pattern) + p.write_text(new, encoding="utf-8") + print(f"[replace] {count} replacements in {p}") + return count +``` + +### 7) Simple command runner +```python +def run(cmd: str, *, cwd: str | Path | None = None, timeout: int | None = None) -> subprocess.CompletedProcess[str]: + """Run a shell command and print stdout/stderr.""" + result = subprocess.run( + cmd, + cwd=str(cwd) if cwd else None, + shell=True, + capture_output=True, + text=True, + timeout=timeout, + ) + if result.stdout: + print(result.stdout, end="" if result.stdout.endswith("\n") else "\n") + if result.stderr: + print(result.stderr, end="" if result.stderr.endswith("\n") else "\n") + print(f"[run] exit={result.returncode}") + return result +``` + +## Bash bridge (snapshot-aware) +Expose a `bash()` helper that uses the shell snapshot when available (generated on the TypeScript side by `shell-snapshot.ts`). The kernel should receive the snapshot path via env var, e.g. `OMP_SHELL_SNAPSHOT`. + +```python +def bash(cmd: str, *, cwd: str | Path | None = None, timeout: int | None = None) -> subprocess.CompletedProcess[str]: + """Run a shell command via bash when available; fallback on Windows or missing bash.""" + snapshot = os.environ.get("OMP_SHELL_SNAPSHOT") + prefix = f"source '{snapshot}' 2>/dev/null && " if snapshot else "" + final = f"{prefix}{cmd}" + + # Prefer bash when present (Unix), otherwise fall back to sh/cmd. + bash_path = shutil.which("bash") + if bash_path: + return run(f"{bash_path} -lc {json.dumps(final)}", cwd=cwd, timeout=timeout) + + # Windows fallback + if sys.platform.startswith("win"): + return run(f"cmd /c {json.dumps(cmd)}", cwd=cwd, timeout=timeout) + + # Last-resort POSIX shell + sh_path = shutil.which("sh") + if sh_path: + return run(f"{sh_path} -lc {json.dumps(cmd)}", cwd=cwd, timeout=timeout) + + raise RuntimeError("No suitable shell found for bash() bridge") +``` + +## Behavior Notes +- All helpers print useful summaries so outputs are visible in tool streaming. +- Functions return values for programmatic use while still printing. +- `run()` is intentionally verbose and not a replacement for the bash tool; it’s a bridge for cases where Python lacks a direct helper. +- **Grep context**: if two matches are close, context lines can overlap. Low priority improvement: merge overlapping ranges before printing. + +## Follow-ups (Optional) +- Add a `git` helper wrapper for common git operations (status/diff/log). +- Add a `cat()` alias for `read()` and `touch()` helper. +- Add `walk()` helper to show directories with depth limit. diff --git a/docs/ipy-tui-ux.md b/docs/ipy-tui-ux.md new file mode 100644 index 000000000..007f08a4f --- /dev/null +++ b/docs/ipy-tui-ux.md @@ -0,0 +1,246 @@ +# IPython kernel REPL tool plan (research/design) + +## Assumptions +- REPL state lives only for the current agent session (no disk persistence, no restore on restart). +- No sandboxing/security constraints (trusted code execution only). +- New NPM dependency is acceptable if required for ZeroMQ (confirm with maintainers). +- Tool surface will be parallel to `bash` (streaming output, truncation, expand/collapse behavior). + +## Goals +- Add a Python REPL tool backed by IPython kernels. +- Integrate streaming/truncation behavior with the existing `bash` tool UX and renderer expectations. +- Provide a clean executor adapter that fits the `bash-executor`/`bash.ts` streaming model. +- Support concurrent kernels (parallel REPL sessions) without persistence. + +## Non-goals +- Security sandboxing, containerization, or permission gating. +- Kernel persistence across agent restarts. +- Rich notebook UX (no cell metadata editing, no execution history storage). + +## Architecture overview +**New components** +1. **KernelManager** (session-scoped registry) + - Owns kernel lifecycles, keyed by `kernelId` or `sessionId`. + - Spawns kernels, tracks connection info, routes execute requests. +2. **KernelProcess** + - Spawns `python -m ipykernel_launcher -f `. + - Produces `KernelConnection` (ZMQ sockets) and manages shutdown. +3. **KernelConnection** + - ZMQ sockets: `shell`, `iopub`, `stdin`, `control`, `hb`. + - Implements Jupyter messaging protocol (JSON frames + HMAC signature). + - Provides `execute(code)` returning outputs + status. +4. **KernelExecutorAdapter** + - Implements a streaming executor interface mirroring `BashOperations.exec`. + - Converts IOPub messages to text chunks for `onChunk` and captures full output. +5. **PythonTool** (new tool surface) + - Tool name: `python` (or `ipython`), parameters: `code`, optional `kernelId`, `timeout`. + - Uses `KernelExecutorAdapter` with `executeBashWithOperations`-like flow or parallel executor. + +**Integration points** +- Tool execution and streaming should match `packages/coding-agent/src/core/tools/bash.ts` behavior. +- Truncation uses `truncateTail` and `DEFAULT_MAX_BYTES` from `tools/truncate`. +- TUI uses existing renderer logic (collapsed/expanded, visual truncation). + +## Kernel lifecycle +1. **Create** + - Generate `connection.json` (temp dir). + - Spawn kernel process: + - `python -m ipykernel_launcher -f /tmp/.json` + - Wait for heartbeat (HB) or a `kernel_info_request` response to confirm readiness. +2. **Use** + - For each `execute_request`, set `parent_header.msg_id` to correlate IOPub messages. + - Stream IOPub outputs to the tool renderer. +3. **Shutdown** + - Send `shutdown_request` over control channel. + - Kill process tree if shutdown fails or on timeout. +4. **Disposal** + - Remove from registry when agent session ends or user explicitly closes. + +**Lifecycle constraints** +- No persistence; kernel is per session. +- A kernel ID can be auto-generated per tool call unless user supplies `kernelId`. +- Kernel cleanup on `AbortSignal`/timeout. + +## IPC details (Jupyter protocol essentials) +**Connection file structure** (generated per kernel): +```json +{ + "ip": "127.0.0.1", + "transport": "tcp", + "signature_scheme": "hmac-sha256", + "key": "", + "shell_port": 57541, + "iopub_port": 57542, + "stdin_port": 57543, + "control_port": 57544, + "hb_port": 57545 +} +``` + +**Message frames** (ZMQ multipart): +- `identities...` (routing frames) +- `DELIM` (``) +- `signature` +- `header` (JSON) +- `parent_header` (JSON) +- `metadata` (JSON) +- `content` (JSON) +- `buffers...` (binary) + +**Signature** +- HMAC SHA-256 of `header|parent_header|metadata|content` using `key`. + +**Channels** +- `shell`: send `execute_request`, receive `execute_reply`. +- `iopub`: receive streamed outputs, status (`busy`/`idle`). +- `stdin`: input requests (can reject/auto-respond as non-interactive). +- `control`: `shutdown_request`, `interrupt_request`. +- `hb`: heartbeat for liveness. + +## Execute flow (mapping to current executor interface) +1. **Tool call**: `python` tool receives `{ code, kernelId?, timeout? }`. +2. **Kernel selection**: KernelManager returns existing kernel by id or creates a new one. +3. **Executor adapter** + - Calls `KernelConnection.execute(code, { onMessage, signal, timeout })`. + - Emits `onChunk` events by transforming IOPub `stream`, `execute_result`, `display_data`, `error`. +4. **Result capture** + - Aggregate output in rolling buffer (same behavior as `createOutputSink`). + - On completion, return final output + `exitCode` equivalent (0/1). + +**ExitCode mapping** +- `exitCode = 0` if `execute_reply.status === "ok"`. +- `exitCode = 1` if `status === "error"`. +- `cancelled = true` if `AbortSignal` triggered or timeout. + +## Streaming output mapping to renderer expectations +**Source messages → text chunks** +- `iopub: stream` + - `content.name` in `stdout|stderr` → direct text. +- `iopub: execute_result` + - Convert `data["text/plain"]` to text chunk. + - If `image/png` present, surface as image output (kitty image rendering supported). + - If `application/json` present, surface as a collapsible JSON tree (reuse task renderer tree format). +- `iopub: display_data` + - Prefer `text/plain`. + - Render `image/png` and `application/json` the same way as `execute_result` when present. + - If `text/html` is present without `text/plain`, convert HTML to markdown and emit that as text output (same pattern as tools that return `_rendered` content, e.g. `git-tool` fetch rendering). +- `iopub: error` + - Join `traceback[]` into lines; stream immediately. + +**Mapping to existing tool streaming** +- Use the same `currentOutput += chunk` strategy as `bash.ts`. +- For each chunk, call `onUpdate` with `{ content: [{ type: "text", text: truncateTail(currentOutput).content }] }`. +- Populate `details.truncation` and `details.fullOutput` when truncated (same as bash). + +**Collapsed/expanded behavior** +- Render context should match `bashToolRenderer`: + - `renderContext.output` → truncated text (tail). + - `details.fullOutput` → full output buffer when expanded. + - Use same preview line count as `BASH_DEFAULT_PREVIEW_LINES`. +- Expectation: collapsed view shows tail, with expand note; expanded shows full output when available. + +## Bun integration notes +- Use `Bun.spawn` to launch `python` kernel process. +- Use `node:fs` for directory creation; use `Bun.write` for connection file contents. +- Use WebCrypto `subtle.importKey` + `subtle.sign` for HMAC signing (avoid `node:crypto`). +- ZMQ library compatibility with Bun must be validated (native deps). + +## Tool surface design +**Tool name**: `python` (or `ipy`) + +**Prompt updates** +- Update all prompts that mention the bash tool to be rendered via prompt templates so tool availability is dynamic per settings (bash-only vs ipy-only vs both). +- Add a Python tool prompt similar to `prompts/tools/bash.md` explaining: + - Execution semantics (kernel-backed, persistent within session) + - Streaming output and truncation + - Recommended uses vs. when to use other tools + - Built-in helpers (e.g., `bash()` bridge) + - Matplotlib note: `plt.show()` is headless; prefer `plt.savefig()` or returning the figure object for display + +**Schema** +```ts +Type.Object({ + code: Type.String({ description: "Python code to execute" }), + kernelId: Type.Optional(Type.String({ description: "Kernel session id" })), + timeout: Type.Optional(Type.Number({ description: "Timeout in seconds" })) +}) +``` + +**Tool result** +- `{ content: [{ type: "text", text: outputText }], details?: { truncation, fullOutputPath?, fullOutput? } }` +- Use the same detail structure as `BashToolDetails` for consistency. + +## Adapter design (data structures) +**KernelSession** +```ts +interface KernelSession { + id: string; + connection: KernelConnection; + process: Subprocess; + createdAt: number; + lastUsedAt: number; +} +``` + +**KernelExecuteResult** +```ts +interface KernelExecuteResult { + output: string; + exitCode: number | undefined; + cancelled: boolean; + truncated: boolean; + fullOutputPath?: string; +} +``` + +**KernelExecutorOptions** +```ts +interface KernelExecutorOptions { + timeout?: number; // ms + onChunk?: (chunk: string) => void; + signal?: AbortSignal; +} +``` + +**KernelOperations (bash-style adapter)** +```ts +interface KernelOperations { + exec: ( + code: string, + kernelId: string, + options: { onData: (data: Buffer) => void; signal?: AbortSignal; timeout?: number } + ) => Promise<{ exitCode: number | null }>; +} +``` + +## Execution state machine +1. `execute_request` sent; status = busy +2. `iopub` stream/output/error messages emitted → `onChunk` +3. `execute_reply` (shell channel) sets `exitCode` +4. `iopub` status = idle → complete + +## Concrete implementation steps +1. **KernelManager** + - Session-scoped registry with `getOrCreateKernel(kernelId?)` and `shutdown(kernelId)`. +2. **KernelProcess** + - Create connection file (temp), spawn kernel, connect ZMQ sockets. + - Implement handshake: `kernel_info_request` until response or timeout. +3. **KernelConnection** + - Implement message signing, send/recv, routing. + - Correlate `parent_header.msg_id` to filter IOPub outputs per execute. +4. **KernelExecutorAdapter** + - Provide `exec` method compatible with `executeBashWithOperations` or a parallel executor. + - Convert IOPub messages → `onData(Buffer.from(text))`. +5. **Python tool** + - Wire to executor; update `toolRenderers` or reuse bash renderer with context. +6. **Streaming integration** + - Use `truncateTail` during updates and final output. + - Provide `details.fullOutput` when truncated. +7. **Shutdown/cleanup** + - On tool abort or session end, dispose kernel and temp files. + +## Open questions (need confirmation) +- Preferred ZMQ dependency (native `zeromq` vs pure JS fallback). +- Whether to reuse `bashToolRenderer` or create a dedicated python renderer. +- How to handle rich outputs (images/HTML) in future iterations. +- Whether to support input requests (stdin channel) or always error. diff --git a/docs/python-dynamic-docs.md b/docs/python-dynamic-docs.md new file mode 100644 index 000000000..32b4e9078 --- /dev/null +++ b/docs/python-dynamic-docs.md @@ -0,0 +1,266 @@ +# Dynamic Python Tool Documentation + +## Problem + +`packages/coding-agent/src/prompts/tools/python.md` contains hardcoded documentation for prelude helpers: + +```markdown +### File I/O +- `read(path, limit=None)` — read file, optional char limit +- `write(path, content)` — write file (creates parents) +... +``` + +This duplicates information already present in `python-prelude.py` (function signatures and docstrings). When the prelude changes, the markdown must be manually updated — they drift. + +## Goal + +Extract helper documentation from the running Python environment via runtime introspection, then populate `python.md` using Handlebars templating (same system as `system-prompt.md`). + +## Current Flow + +``` +startup + └─> checkPythonKernelAvailability(cwd) // just checks if kernel_gateway is installed + └─> returns { ok: true/false, reason } + +createTools(session) + └─> if python available && mode != "bash-only" + └─> createPythonTool(session) + └─> description: renderPromptTemplate(pythonDescription) // no context, static +``` + +## Proposed Flow + +``` +startup + └─> warmPythonKernel(cwd) + ├─> start kernel gateway + kernel + ├─> run prelude + ├─> run introspection snippet + │ └─> returns JSON: [{ name, signature, docstring, category }, ...] + ├─> cache extracted docs in module-level state + └─> returns { ok, reason, kernel } + +createTools(session) + └─> if python available && mode != "bash-only" + └─> createPythonTool(session) + └─> description: renderPromptTemplate(pythonDescription, { helpers: cachedDocs }) + +session reset + └─> kernel.restart() // reuse same gateway, just restart kernel +``` + +## Key Changes + +### 1. `python-kernel.ts` — Add introspection method + +```typescript +interface PreludeHelper { + name: string; + signature: string; + docstring: string; +} + +class PythonKernel { + // existing... + + async introspectPrelude(): Promise { + const result = await this.execute(INTROSPECTION_SNIPPET); + return JSON.parse(result.output); + } +} +``` + +### 2. `python-executor.ts` — Expose warm kernel + cached docs + +```typescript +let cachedPreludeDocs: PreludeHelper[] | null = null; + +export async function warmPythonEnvironment(cwd: string): Promise<{ + ok: boolean; + reason?: string; + docs: PreludeHelper[]; +}> { + // 1. Check availability (existing) + // 2. Start kernel (new - currently deferred to first execute) + // 3. Run prelude (already happens on kernel start) + // 4. Introspect and cache + // 5. Return docs +} + +export function getPreludeDocs(): PreludeHelper[] { + return cachedPreludeDocs ?? []; +} +``` + +### 3. `tools/index.ts` — Warm kernel during tool creation + +```typescript +export async function createTools(session: ToolSession, toolNames?: string[]): Promise { + // ...existing python availability check... + + if (shouldCheckPython) { + const warmup = await warmPythonEnvironment(session.cwd); + pythonAvailable = warmup.ok; + // docs now cached for use by createPythonTool + } + + // ...rest unchanged... +} +``` + +### 4. `tools/python.ts` — Pass docs to template + +```typescript +import { getPreludeDocs } from "../python-executor"; + +export function createPythonTool(session: ToolSession): AgentTool { + const helpers = getPreludeDocs(); + const categories = groupByCategory(helpers); // group into File I/O, Navigation, etc. + + return { + name: "python", + description: renderPromptTemplate(pythonDescription, { categories }), + // ... + }; +} +``` + +### 5. `prompts/tools/python.md` — Use Handlebars + +```markdown +## Prelude helpers + +All helpers auto-print results and return values for chaining. + +{{#each categories}} +### {{name}} +{{#each functions}} +- `{{name}}{{signature}}` — {{docstring}} +{{/each}} + +{{/each}} +``` + +## Introspection Snippet + +```python +import inspect, json + +CATEGORIES = { + 'read': 'File I/O', 'write': 'File I/O', 'append': 'File I/O', ... + 'cp': 'File operations', 'mv': 'File operations', ... +} + +helpers = [] +for name, cat in CATEGORIES.items(): + obj = globals().get(name) + if not callable(obj): + continue + sig = str(inspect.signature(obj)) + doc = (inspect.getdoc(obj) or '').split('\n')[0] + helpers.append({'name': name, 'signature': sig, 'docstring': doc, 'category': cat}) + +print(json.dumps(helpers)) +``` + +## Category Mapping + +Categories are defined in the introspection snippet (Python side), not TypeScript. This keeps the source of truth in one place. Order: + +1. File I/O: `read`, `write`, `append`, `touch`, `cat` +2. File operations: `cp`, `mv`, `rm`, `mkdir` +3. Navigation: `pwd`, `cd`, `ls`, `tree`, `stat` +4. Search: `find`, `glob_files`, `grep`, `rgrep` +5. Text processing: `head`, `tail`, `sort_lines`, `uniq`, `cols`, `wc` +6. Find and replace: `replace`, `sed`, `rsed` +7. Batch operations: `batch`, `diff` +8. Shell bridge: `run`, `bash`, `env` + +## Session Reset Behavior + +Currently: `reset: true` parameter triggers full kernel restart via `restartKernelSession()`. + +Proposed: Same behavior, but the pre-warmed kernel is the one being restarted. No change needed — the existing session management already handles this. + +## Environment Variable Override + +Add `OMP_PY` environment variable to override the settings preference: + +| Value | Mode | Description | +|-------|------|-------------| +| `0` or `bash` | bash-only | Disable Python tool, use bash only | +| `1` or `py` | ipy-only | Disable bash tool, use Python only | +| `mix` or `both` | both | Enable both bash and Python tools | + +Priority: `OMP_PY` env var > settings preference > default (`ipy-only`) + +### Implementation + +In `tools/index.ts`: + +```typescript +function getPythonModeFromEnv(): "ipy-only" | "bash-only" | "both" | null { + const value = process.env.OMP_PY?.toLowerCase(); + if (!value) return null; + + switch (value) { + case "0": + case "bash": + return "bash-only"; + case "1": + case "py": + return "ipy-only"; + case "mix": + case "both": + return "both"; + default: + return null; + } +} + +export async function createTools(session: ToolSession, toolNames?: string[]): Promise { + // ... + const pythonMode = getPythonModeFromEnv() ?? session.settings?.getPythonToolMode?.() ?? "ipy-only"; + // ... +} +``` + +### Use Cases + +- `OMP_PY=0 omp` — Force bash mode for compatibility testing +- `OMP_PY=1 omp` — Force Python mode even if settings say otherwise +- `OMP_PY=mix omp` — Enable both for users who want choice + +## Fallback + +If kernel warmup fails (no Python, no kernel_gateway), `getPreludeDocs()` returns empty array. The template should handle this gracefully: + +```markdown +{{#if categories.length}} +## Prelude helpers +... +{{else}} +## Prelude helpers + +(Documentation unavailable — Python kernel failed to start) +{{/if}} +``` + +## Files to Modify + +1. `packages/coding-agent/src/core/python-kernel.ts` — Add `introspectPrelude()` method +2. `packages/coding-agent/src/core/python-executor.ts` — Add `warmPythonEnvironment()`, `getPreludeDocs()` +3. `packages/coding-agent/src/core/tools/index.ts` — Call warmup during `createTools()`, add `OMP_PY` env override +4. `packages/coding-agent/src/core/tools/python.ts` — Pass docs context to template +5. `packages/coding-agent/src/prompts/tools/python.md` — Convert to Handlebars template +6. `packages/coding-agent/src/core/python-prelude.py` — Add category markers or rely on introspection snippet + +## Testing + +1. Unit test: `introspectPrelude()` returns expected structure +2. Unit test: `warmPythonEnvironment()` populates cache +3. Unit test: Template renders correctly with mock docs +4. Integration test: Full flow from startup to tool description containing extracted docs +5. Fallback test: Empty docs when Python unavailable diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index faaaf943c..5f6157998 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] + ### Added - Added automatic retry logic for OpenAI Codex responses with configurable delay and max retries @@ -9,6 +10,8 @@ ### Changed +- Updated environment variable prefix from PI_ to OMP_ for better consistency +- Added automatic migration for legacy PI_ environment variables to OMP_ equivalents - Adjusted Bedrock Claude thinking budgets to reserve output tokens when maxTokens is too low ### Fixed diff --git a/packages/ai/src/cli.ts b/packages/ai/src/cli.ts index e6281626c..c1c226b87 100755 --- a/packages/ai/src/cli.ts +++ b/packages/ai/src/cli.ts @@ -1,5 +1,5 @@ #!/usr/bin/env bun - +import "./utils/migrate-env"; import { existsSync, readFileSync, writeFileSync } from "node:fs"; import { createInterface } from "readline"; import { loginAnthropic } from "./utils/oauth/anthropic"; diff --git a/packages/ai/src/index.ts b/packages/ai/src/index.ts index b1b71bd06..078772d5e 100644 --- a/packages/ai/src/index.ts +++ b/packages/ai/src/index.ts @@ -1,3 +1,5 @@ +import "./utils/migrate-env"; + export * from "./models"; export * from "./providers/anthropic"; export * from "./providers/cursor"; diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index 149c8dd5d..3095dbdc3 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -49,7 +49,7 @@ export interface OpenAICodexResponsesOptions extends StreamOptions { codexMode?: boolean; } -const CODEX_DEBUG = process.env.PI_CODEX_DEBUG === "1" || process.env.PI_CODEX_DEBUG === "true"; +const CODEX_DEBUG = process.env.OMP_CODEX_DEBUG === "1" || process.env.OMP_CODEX_DEBUG === "true"; const CODEX_MAX_RETRIES = 2; const CODEX_RETRYABLE_STATUS = new Set([408, 429, 500, 502, 503, 504]); const CODEX_RETRY_DELAY_MS = 500; diff --git a/packages/ai/src/utils/migrate-env.ts b/packages/ai/src/utils/migrate-env.ts new file mode 100644 index 000000000..367e79494 --- /dev/null +++ b/packages/ai/src/utils/migrate-env.ts @@ -0,0 +1,8 @@ +for (const [key, value] of Object.entries(process.env)) { + if (key.startsWith("PI_") && value !== undefined) { + const ompKey = `OMP_${key.slice(3)}`; // PI_FOO -> OMP_FOO + if (process.env[ompKey] === undefined) { + process.env[ompKey] = value; + } + } +} diff --git a/packages/ai/test/context-overflow.test.ts b/packages/ai/test/context-overflow.test.ts index 4c41d228f..2c17866b0 100644 --- a/packages/ai/test/context-overflow.test.ts +++ b/packages/ai/test/context-overflow.test.ts @@ -447,7 +447,7 @@ describe("Context overflow error handling", () => { // Check if ollama is installed and local LLM tests are enabled let ollamaInstalled = false; - if (!process.env.PI_NO_LOCAL_LLM) { + if (!process.env.OMP_NO_LOCAL_LLM) { try { execSync("which ollama", { stdio: "ignore" }); ollamaInstalled = true; @@ -541,7 +541,7 @@ describe("Context overflow error handling", () => { // ============================================================================= let lmStudioRunning = false; - if (!process.env.PI_NO_LOCAL_LLM) { + if (!process.env.OMP_NO_LOCAL_LLM) { try { execSync("curl -s --max-time 1 http://localhost:1234/v1/models > /dev/null", { stdio: "ignore" }); lmStudioRunning = true; diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 52e6e18d8..701ed4c72 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -6,14 +6,14 @@ import { streamOpenAICodexResponses } from "../src/providers/openai-codex-respon import type { Context, Model } from "../src/types"; const originalFetch = global.fetch; -const originalAgentDir = process.env.PI_CODING_AGENT_DIR; +const originalAgentDir = process.env.OMP_CODING_AGENT_DIR; afterEach(() => { global.fetch = originalFetch; if (originalAgentDir === undefined) { - delete process.env.PI_CODING_AGENT_DIR; + delete process.env.OMP_CODING_AGENT_DIR; } else { - process.env.PI_CODING_AGENT_DIR = originalAgentDir; + process.env.OMP_CODING_AGENT_DIR = originalAgentDir; } vi.restoreAllMocks(); }); @@ -21,7 +21,7 @@ afterEach(() => { describe("openai-codex streaming", () => { it("streams SSE responses into AssistantMessageEventStream", async () => { const tempDir = mkdtempSync(join(tmpdir(), "pi-codex-stream-")); - process.env.PI_CODING_AGENT_DIR = tempDir; + process.env.OMP_CODING_AGENT_DIR = tempDir; const payload = Buffer.from( JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: "acc_test" } }), @@ -132,7 +132,7 @@ describe("openai-codex streaming", () => { it("sets conversation_id/session_id headers and prompt_cache_key when sessionId is provided", async () => { const tempDir = mkdtempSync(join(tmpdir(), "pi-codex-stream-")); - process.env.PI_CODING_AGENT_DIR = tempDir; + process.env.OMP_CODING_AGENT_DIR = tempDir; const payload = Buffer.from( JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: "acc_test" } }), @@ -233,7 +233,7 @@ describe("openai-codex streaming", () => { it("does not set conversation_id/session_id headers when sessionId is not provided", async () => { const tempDir = mkdtempSync(join(tmpdir(), "pi-codex-stream-")); - process.env.PI_CODING_AGENT_DIR = tempDir; + process.env.OMP_CODING_AGENT_DIR = tempDir; const payload = Buffer.from( JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: "acc_test" } }), diff --git a/packages/ai/test/stream.test.ts b/packages/ai/test/stream.test.ts index ea30a3bca..cbc913a0e 100644 --- a/packages/ai/test/stream.test.ts +++ b/packages/ai/test/stream.test.ts @@ -911,7 +911,7 @@ describe("Generate E2E Tests", () => { // Check if ollama is installed and local LLM tests are enabled let ollamaInstalled = false; - if (!process.env.PI_NO_LOCAL_LLM) { + if (!process.env.OMP_NO_LOCAL_LLM) { try { execSync("which ollama", { stdio: "ignore" }); ollamaInstalled = true; diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 7e7f71399..7cde3e154 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,9 +1,12 @@ # Changelog ## [Unreleased] - ### Added +- Added IPython-backed Python tool with streaming output, image/JSON rendering, and Jupyter kernel gateway integration +- Added Python prelude with 30+ shell-like utility functions for file operations +- Added Python tool exposure settings with session-scoped kernel reuse and fallback behavior +- Added streaming output system with automatic spill-to-disk for large outputs - Added extension input interception with source metadata and command argument completion - Added extension command context `compact()` helper plus context usage accessors - Added ExtensionAPI `setLabel()` for extension and entry labels @@ -21,6 +24,8 @@ ### Changed +- Reorganized settings interface into behavior, tools, display, voice, status, lsp, and exa tabs +- Migrated environment variables from PI_ to OMP_ prefix with automatic migration - Updated model selector to use TabBar component for provider navigation - Changed role badges to inverted style with colored backgrounds - Added support for /models command alias in addition to /model diff --git a/packages/coding-agent/docs/python-repl.md b/packages/coding-agent/docs/python-repl.md new file mode 100644 index 000000000..5cd2a3b2c --- /dev/null +++ b/packages/coding-agent/docs/python-repl.md @@ -0,0 +1,77 @@ +# Python REPL (Jupyter Kernel Gateway) + +## Requirements + +- Python 3 available on PATH (or via an active virtualenv) +- `jupyter-kernel-gateway` (`kernel_gateway` module) and `ipykernel` installed in the selected Python environment + +Install: +```bash +python -m pip install jupyter_kernel_gateway ipykernel +``` + +## How It Works + +The Python tool starts a Jupyter Kernel Gateway process locally, which manages an IPython kernel. All code execution goes through the gateway's REST and WebSocket APIs. + +Startup flow: +1. Spawn `python -m kernel_gateway` on a random available port +2. Wait for gateway to become ready (`GET /api/kernelspecs`) +3. Create a kernel (`POST /api/kernels`) +4. Connect WebSocket for execution messages +5. Run prelude code (helper functions) + +## External Gateway Support + +Instead of spawning a local gateway, you can connect to an already-running Jupyter Kernel Gateway: + +```bash +# Connect to external gateway +export OMP_PYTHON_GATEWAY_URL="http://127.0.0.1:8888" + +# Optional: auth token if gateway requires it (KG_AUTH_TOKEN) +export OMP_PYTHON_GATEWAY_TOKEN="your-token-here" +``` + +When `OMP_PYTHON_GATEWAY_URL` is set: +- No local gateway process is spawned +- Kernels are created on the external gateway +- The gateway process is not killed on shutdown +- Availability check uses `/api/kernelspecs` endpoint instead of local module check + +This is useful for: +- Remote kernel execution +- Shared kernel environments +- Pre-configured gateway setups + +## Environment Propagation + +- The kernel inherits a filtered environment (explicit allowlist + denylist) +- `PYTHONPATH` includes the working directory and any existing `PYTHONPATH` value +- Virtual environments are detected via `VIRTUAL_ENV`, `.venv/`, or `venv/` and preferred when present + +## Kernel Modes + +Settings under `python` control exposure and reuse: +- `toolMode`: `ipy-only` (default), `bash-only`, `both` +- `kernelMode`: `session` (default, queued), `per-call` + +## Shell Bridge + +The Python prelude exposes `bash()` which: +- Sources the shell snapshot when `OMP_SHELL_SNAPSHOT` is set +- Runs via `bash -lc` when available, with OS fallbacks + +## Output Handling + +- Streams `stdout`/`stderr` as text +- `image/png` display data renders inline in TUI +- `application/json` display data renders as a collapsible tree +- `text/html` display data is converted to basic markdown + +## Troubleshooting + +- **Kernel unavailable**: Ensure `python` + `jupyter-kernel-gateway` + `ipykernel` are installed; the session will fall back to bash-only. +- **External gateway unreachable**: Check the URL is correct and the gateway is running. If auth is required, set `OMP_PYTHON_GATEWAY_TOKEN`. +- **IPC tracing**: Set `OMP_PYTHON_IPC_TRACE=1` to log kernel message flow. +- **Stdin requests**: Interactive input is not supported; refactor code to avoid `input()` or provide data programmatically. diff --git a/packages/coding-agent/src/bun-imports.d.ts b/packages/coding-agent/src/bun-imports.d.ts index 169a6791d..e7dc00dff 100644 --- a/packages/coding-agent/src/bun-imports.d.ts +++ b/packages/coding-agent/src/bun-imports.d.ts @@ -14,3 +14,9 @@ declare module "*.txt" { const content: string; export default content; } + +// Python files imported as text +declare module "*.py" { + const content: string; + export default content; +} diff --git a/packages/coding-agent/src/core/bash-executor.ts b/packages/coding-agent/src/core/bash-executor.ts index 82fbbee8c..ce60105c8 100644 --- a/packages/coding-agent/src/core/bash-executor.ts +++ b/packages/coding-agent/src/core/bash-executor.ts @@ -6,16 +6,12 @@ * - Direct calls from modes that need bash execution */ -import { createWriteStream, type WriteStream } from "node:fs"; -import { tmpdir } from "node:os"; -import { join } from "node:path"; import type { Subprocess } from "bun"; -import { nanoid } from "nanoid"; -import stripAnsi from "strip-ansi"; -import { getShellConfig, killProcessTree, sanitizeBinaryOutput } from "../utils/shell"; +import { getShellConfig, killProcessTree } from "../utils/shell"; import { getOrCreateSnapshot, getSnapshotSourceCommand } from "../utils/shell-snapshot"; +import { createOutputSink, pumpStream } from "./streaming-output"; import type { BashOperations } from "./tools/bash"; -import { DEFAULT_MAX_BYTES, truncateTail } from "./tools/truncate"; +import { DEFAULT_MAX_BYTES } from "./tools/truncate"; import { ScopeSignal } from "./utils"; // ============================================================================ @@ -50,83 +46,6 @@ export interface BashResult { // Implementation // ============================================================================ -function createSanitizer(): TransformStream { - const decoder = new TextDecoder(); - return new TransformStream({ - transform(chunk, controller) { - const text = sanitizeBinaryOutput(stripAnsi(decoder.decode(chunk, { stream: true }))).replace(/\r/g, ""); - controller.enqueue(text); - }, - }); -} - -async function pumpStream(readable: ReadableStream, writer: WritableStreamDefaultWriter) { - const reader = readable.pipeThrough(createSanitizer()).getReader(); - try { - while (true) { - const { done, value } = await reader.read(); - if (done) break; - await writer.write(value); - } - } finally { - reader.releaseLock(); - } -} - -function createOutputSink( - spillThreshold: number, - maxBuffer: number, - onChunk?: (text: string) => void, -): WritableStream & { - dump: (annotation?: string) => { output: string; truncated: boolean; fullOutputPath?: string }; -} { - const chunks: string[] = []; - let chunkBytes = 0; - let totalBytes = 0; - let fullOutputPath: string | undefined; - let fullOutputStream: WriteStream | undefined; - - const sink = new WritableStream({ - write(text) { - totalBytes += text.length; - - // Spill to temp file if needed - if (totalBytes > spillThreshold && !fullOutputPath) { - fullOutputPath = join(tmpdir(), `omp-${nanoid()}.buffer`); - const ts = createWriteStream(fullOutputPath); - chunks.forEach((c) => { - ts.write(c); - }); - fullOutputStream = ts; - } - fullOutputStream?.write(text); - - // Rolling buffer - chunks.push(text); - chunkBytes += text.length; - while (chunkBytes > maxBuffer && chunks.length > 1) { - chunkBytes -= chunks.shift()!.length; - } - - onChunk?.(text); - }, - close() { - fullOutputStream?.end(); - }, - }); - - return Object.assign(sink, { - dump(annotation?: string) { - if (annotation) { - chunks.push(`\n\n${annotation}`); - } - const full = chunks.join(""); - const { content, truncated } = truncateTail(full); - return { output: truncated ? content : full, truncated, fullOutputPath: fullOutputPath }; - }, - }); -} - /** * Execute a bash command with optional streaming and cancellation support. * diff --git a/packages/coding-agent/src/core/python-executor-display.test.ts b/packages/coding-agent/src/core/python-executor-display.test.ts new file mode 100644 index 000000000..e6ef1b985 --- /dev/null +++ b/packages/coding-agent/src/core/python-executor-display.test.ts @@ -0,0 +1,42 @@ +import { describe, expect, it } from "bun:test"; +import { executePythonWithKernel, type PythonKernelExecutor } from "./python-executor"; +import type { KernelDisplayOutput, KernelExecuteOptions, KernelExecuteResult } from "./python-kernel"; + +class FakeKernel implements PythonKernelExecutor { + private result: KernelExecuteResult; + private onExecute: (options?: KernelExecuteOptions) => Promise | void; + + constructor(result: KernelExecuteResult, onExecute: (options?: KernelExecuteOptions) => Promise | void) { + this.result = result; + this.onExecute = onExecute; + } + + async execute(_code: string, options?: KernelExecuteOptions): Promise { + await this.onExecute(options); + return this.result; + } +} + +describe("executePythonWithKernel display outputs", () => { + it("aggregates display outputs in order", async () => { + const outputs: KernelDisplayOutput[] = [ + { type: "json", data: { foo: "bar" } }, + { type: "image", data: "abc", mimeType: "image/png" }, + ]; + + const kernel = new FakeKernel( + { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }, + async (options) => { + if (!options?.onDisplay) return; + for (const output of outputs) { + await options.onDisplay(output); + } + }, + ); + + const result = await executePythonWithKernel(kernel, "print('hi')"); + + expect(result.exitCode).toBe(0); + expect(result.displayOutputs).toEqual(outputs); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor-lifecycle.test.ts b/packages/coding-agent/src/core/python-executor-lifecycle.test.ts new file mode 100644 index 000000000..4bf732013 --- /dev/null +++ b/packages/coding-agent/src/core/python-executor-lifecycle.test.ts @@ -0,0 +1,99 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { disposeAllKernelSessions, executePython } from "./python-executor"; +import type { KernelExecuteResult } from "./python-kernel"; +import * as pythonKernel from "./python-kernel"; + +class FakeKernel { + execute = vi.fn(async () => this.result); + shutdown = vi.fn(async () => {}); + ping = vi.fn(async () => true); + alive = true; + + constructor(private readonly result: KernelExecuteResult) {} + + isAlive(): boolean { + return this.alive; + } +} + +const OK_RESULT: KernelExecuteResult = { + status: "ok", + cancelled: false, + timedOut: false, + stdinRequested: false, +}; + +afterEach(async () => { + vi.restoreAllMocks(); + await disposeAllKernelSessions(); +}); + +describe("executePython lifecycle", () => { + it("starts and shuts down per-call kernels", async () => { + const kernel = new FakeKernel(OK_RESULT); + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const startSpy = vi + .spyOn(pythonKernel.PythonKernel, "start") + .mockResolvedValue(kernel as unknown as pythonKernel.PythonKernel); + + await executePython("print('hi')", { kernelMode: "per-call", cwd: process.cwd() }); + + expect(startSpy).toHaveBeenCalledTimes(1); + expect(kernel.execute).toHaveBeenCalledTimes(1); + expect(kernel.shutdown).toHaveBeenCalledTimes(1); + }); + + it("reuses session kernels until reset", async () => { + const kernel = new FakeKernel(OK_RESULT); + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const startSpy = vi + .spyOn(pythonKernel.PythonKernel, "start") + .mockResolvedValue(kernel as unknown as pythonKernel.PythonKernel); + + await executePython("1 + 1", { kernelMode: "session", sessionId: "test-session", cwd: process.cwd() }); + await executePython("2 + 2", { kernelMode: "session", sessionId: "test-session", cwd: process.cwd() }); + + expect(startSpy).toHaveBeenCalledTimes(1); + expect(kernel.execute).toHaveBeenCalledTimes(2); + }); + + it("resets session kernels when requested", async () => { + const kernel = new FakeKernel(OK_RESULT); + const kernelNext = new FakeKernel(OK_RESULT); + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const startSpy = vi + .spyOn(pythonKernel.PythonKernel, "start") + .mockResolvedValueOnce(kernel as unknown as pythonKernel.PythonKernel) + .mockResolvedValueOnce(kernelNext as unknown as pythonKernel.PythonKernel); + + await executePython("1 + 1", { kernelMode: "session", sessionId: "reset-session", cwd: process.cwd() }); + await executePython("2 + 2", { + kernelMode: "session", + sessionId: "reset-session", + reset: true, + cwd: process.cwd(), + }); + + expect(startSpy).toHaveBeenCalledTimes(2); + expect(kernel.shutdown).toHaveBeenCalledTimes(1); + expect(kernelNext.execute).toHaveBeenCalledTimes(1); + }); + + it("restarts session kernels when they are dead", async () => { + const kernel = new FakeKernel(OK_RESULT); + const kernelNext = new FakeKernel(OK_RESULT); + kernel.alive = false; + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const startSpy = vi + .spyOn(pythonKernel.PythonKernel, "start") + .mockResolvedValueOnce(kernel as unknown as pythonKernel.PythonKernel) + .mockResolvedValueOnce(kernelNext as unknown as pythonKernel.PythonKernel); + + await executePython("1 + 1", { kernelMode: "session", sessionId: "dead-session", cwd: process.cwd() }); + + expect(startSpy).toHaveBeenCalledTimes(2); + expect(kernel.shutdown).toHaveBeenCalledTimes(1); + expect(kernel.execute).toHaveBeenCalledTimes(0); + expect(kernelNext.execute).toHaveBeenCalledTimes(1); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor-mapping.test.ts b/packages/coding-agent/src/core/python-executor-mapping.test.ts new file mode 100644 index 000000000..3551aba09 --- /dev/null +++ b/packages/coding-agent/src/core/python-executor-mapping.test.ts @@ -0,0 +1,41 @@ +import { describe, expect, it } from "bun:test"; +import { executePythonWithKernel, type PythonKernelExecutor } from "./python-executor"; +import type { KernelExecuteOptions, KernelExecuteResult } from "./python-kernel"; + +class FakeKernel implements PythonKernelExecutor { + constructor( + private result: KernelExecuteResult, + private onExecute: (options?: KernelExecuteOptions) => void = () => {}, + ) {} + + async execute(_code: string, options?: KernelExecuteOptions): Promise { + this.onExecute(options); + return this.result; + } +} + +describe("executePythonWithKernel mapping", () => { + it("annotates timeout cancellations", async () => { + const kernel = new FakeKernel({ status: "ok", cancelled: true, timedOut: true, stdinRequested: false }); + const result = await executePythonWithKernel(kernel, "sleep(10)", { timeout: 5000 }); + + expect(result.cancelled).toBe(true); + expect(result.exitCode).toBeUndefined(); + expect(result.output).toContain("Command timed out after 5 seconds"); + }); + + it("maps error status to non-zero exit code", async () => { + const kernel = new FakeKernel( + { status: "error", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.("traceback\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "1 / 0"); + + expect(result.exitCode).toBe(1); + expect(result.output).toContain("traceback"); + expect(result.stdinRequested).toBe(false); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor-per-call.test.ts b/packages/coding-agent/src/core/python-executor-per-call.test.ts new file mode 100644 index 000000000..0ba404692 --- /dev/null +++ b/packages/coding-agent/src/core/python-executor-per-call.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from "bun:test"; +import { executePython } from "./python-executor"; +import type { KernelExecuteOptions, KernelExecuteResult } from "./python-kernel"; +import { PythonKernel } from "./python-kernel"; + +interface KernelStub { + execute: (code: string, options?: KernelExecuteOptions) => Promise; + shutdown: () => Promise; +} + +describe("executePython (per-call)", () => { + it("shuts down kernel on timed-out cancellation", async () => { + process.env.OMP_PYTHON_SKIP_CHECK = "1"; + + let shutdownCalls = 0; + const kernel: KernelStub = { + execute: async () => ({ + status: "ok", + cancelled: true, + timedOut: true, + stdinRequested: false, + }), + shutdown: async () => { + shutdownCalls += 1; + }, + }; + + const kernelClass = PythonKernel as unknown as { + start: (options: { cwd: string }) => Promise; + }; + const originalStart = kernelClass.start; + kernelClass.start = async () => kernel; + + try { + const result = await executePython("sleep(10)", { + kernelMode: "per-call", + timeout: 2000, + cwd: "/tmp", + }); + + expect(result.cancelled).toBe(true); + expect(result.exitCode).toBeUndefined(); + expect(result.output).toContain("Command timed out after 2 seconds"); + expect(shutdownCalls).toBe(1); + } finally { + kernelClass.start = originalStart; + } + }); +}); diff --git a/packages/coding-agent/src/core/python-executor-session.test.ts b/packages/coding-agent/src/core/python-executor-session.test.ts new file mode 100644 index 000000000..2c8c895df --- /dev/null +++ b/packages/coding-agent/src/core/python-executor-session.test.ts @@ -0,0 +1,103 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { disposeAllKernelSessions, executePython } from "./python-executor"; +import * as pythonKernel from "./python-kernel"; + +class FakeKernel { + executeCalls = 0; + shutdownCalls = 0; + alive = true; + constructor(private readonly shouldThrow: boolean = false) {} + + isAlive(): boolean { + return this.alive; + } + + async execute(): Promise<{ status: "ok"; cancelled: false; timedOut: false; stdinRequested: false }> { + this.executeCalls += 1; + if (this.shouldThrow) { + this.alive = false; + throw new Error("kernel crashed"); + } + return { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }; + } + + async ping(): Promise { + return this.alive; + } + + async shutdown(): Promise { + this.shutdownCalls += 1; + this.alive = false; + } +} + +describe("executePython session lifecycle", () => { + afterEach(async () => { + vi.restoreAllMocks(); + await disposeAllKernelSessions(); + }); + + it("restarts session when kernel is not alive", async () => { + const kernel1 = new FakeKernel(); + kernel1.alive = false; + const kernel2 = new FakeKernel(); + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const startSpy = vi + .spyOn(pythonKernel.PythonKernel, "start") + .mockResolvedValueOnce(kernel1 as unknown as pythonKernel.PythonKernel) + .mockResolvedValueOnce(kernel2 as unknown as pythonKernel.PythonKernel); + + await executePython("print('hi')", { cwd: "/tmp", sessionId: "session-1", kernelMode: "session" }); + + expect(startSpy).toHaveBeenCalledTimes(2); + expect(kernel1.executeCalls).toBe(0); + expect(kernel1.shutdownCalls).toBe(1); + expect(kernel2.executeCalls).toBe(1); + }); + + it("restarts after an execution failure when kernel is dead", async () => { + const kernel1 = new FakeKernel(true); + const kernel2 = new FakeKernel(); + const starts = [kernel1, kernel2]; + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const startSpy = vi.spyOn(pythonKernel.PythonKernel, "start").mockImplementation(async () => { + const next = starts.shift(); + if (!next) { + throw new Error("No kernel available"); + } + return next as unknown as pythonKernel.PythonKernel; + }); + + await executePython("raise", { cwd: "/tmp", sessionId: "session-2", kernelMode: "session" }); + + expect(startSpy).toHaveBeenCalledTimes(2); + expect(kernel1.executeCalls).toBe(1); + expect(kernel2.executeCalls).toBe(1); + }); + + it("resets existing session when requested", async () => { + const kernel1 = new FakeKernel(); + const kernel2 = new FakeKernel(); + const starts = [kernel1, kernel2]; + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const startSpy = vi.spyOn(pythonKernel.PythonKernel, "start").mockImplementation(async () => { + const next = starts.shift(); + if (!next) { + throw new Error("No kernel available"); + } + return next as unknown as pythonKernel.PythonKernel; + }); + + await executePython("print('one')", { cwd: "/tmp", sessionId: "session-3", kernelMode: "session" }); + await executePython("print('two')", { + cwd: "/tmp", + sessionId: "session-3", + kernelMode: "session", + reset: true, + }); + + expect(startSpy).toHaveBeenCalledTimes(2); + expect(kernel1.shutdownCalls).toBe(1); + expect(kernel2.executeCalls).toBe(1); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor-streaming.test.ts b/packages/coding-agent/src/core/python-executor-streaming.test.ts new file mode 100644 index 000000000..83d24c248 --- /dev/null +++ b/packages/coding-agent/src/core/python-executor-streaming.test.ts @@ -0,0 +1,77 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { rmSync } from "node:fs"; +import { executePythonWithKernel, type PythonKernelExecutor } from "./python-executor"; +import type { KernelExecuteOptions, KernelExecuteResult } from "./python-kernel"; +import { DEFAULT_MAX_BYTES } from "./tools/truncate"; + +class FakeKernel implements PythonKernelExecutor { + private result: KernelExecuteResult; + private onExecute: (options?: KernelExecuteOptions) => void; + + constructor(result: KernelExecuteResult, onExecute: (options?: KernelExecuteOptions) => void) { + this.result = result; + this.onExecute = onExecute; + } + + async execute(_code: string, options?: KernelExecuteOptions): Promise { + this.onExecute(options); + return this.result; + } +} + +const cleanupPaths: string[] = []; + +afterEach(() => { + while (cleanupPaths.length > 0) { + const path = cleanupPaths.pop(); + if (path) { + try { + rmSync(path, { force: true }); + } catch {} + } + } +}); + +describe("executePythonWithKernel streaming", () => { + it("truncates large output and writes full output file", async () => { + const largeOutput = "a".repeat(DEFAULT_MAX_BYTES + 128); + const kernel = new FakeKernel( + { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.(largeOutput); + }, + ); + + const result = await executePythonWithKernel(kernel, "print('hi')"); + + expect(result.truncated).toBe(true); + expect(result.fullOutputPath).toBeDefined(); + expect(result.output.length).toBeLessThan(largeOutput.length); + if (result.fullOutputPath) { + cleanupPaths.push(result.fullOutputPath); + } + }); + + it("annotates timed out runs", async () => { + const kernel = new FakeKernel({ status: "ok", cancelled: true, timedOut: true, stdinRequested: false }, () => {}); + + const result = await executePythonWithKernel(kernel, "sleep", { timeout: 2000 }); + + expect(result.cancelled).toBe(true); + expect(result.exitCode).toBeUndefined(); + expect(result.output).toContain("Command timed out after 2 seconds"); + }); + + it("sanitizes ANSI and carriage returns", async () => { + const kernel = new FakeKernel( + { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.("\u001b[31mhello\r\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "print('hello')"); + + expect(result.output).toBe("hello\n"); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor-timeout.test.ts b/packages/coding-agent/src/core/python-executor-timeout.test.ts new file mode 100644 index 000000000..7ef13a35f --- /dev/null +++ b/packages/coding-agent/src/core/python-executor-timeout.test.ts @@ -0,0 +1,35 @@ +import { describe, expect, it } from "bun:test"; +import { executePythonWithKernel, type PythonKernelExecutor } from "./python-executor"; +import type { KernelExecuteOptions, KernelExecuteResult } from "./python-kernel"; + +class FakeKernel implements PythonKernelExecutor { + private result: KernelExecuteResult; + private onExecute: (options?: KernelExecuteOptions) => void; + + constructor(result: KernelExecuteResult, onExecute: (options?: KernelExecuteOptions) => void) { + this.result = result; + this.onExecute = onExecute; + } + + async execute(_code: string, options?: KernelExecuteOptions): Promise { + this.onExecute(options); + return this.result; + } +} + +describe("executePythonWithKernel cancellation", () => { + it("annotates timeouts when cancelled", async () => { + const kernel = new FakeKernel( + { status: "ok", cancelled: true, timedOut: true, stdinRequested: false }, + (options) => { + options?.onChunk?.("tick\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "sleep(10)", { timeout: 5000 }); + + expect(result.cancelled).toBe(true); + expect(result.exitCode).toBeUndefined(); + expect(result.output).toContain("Command timed out after 5 seconds"); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor.lifecycle.test.ts b/packages/coding-agent/src/core/python-executor.lifecycle.test.ts new file mode 100644 index 000000000..1051a83f9 --- /dev/null +++ b/packages/coding-agent/src/core/python-executor.lifecycle.test.ts @@ -0,0 +1,139 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { disposeAllKernelSessions, executePython } from "./python-executor"; +import { type KernelExecuteOptions, type KernelExecuteResult, PythonKernel } from "./python-kernel"; + +process.env.OMP_PYTHON_SKIP_CHECK = "1"; + +class FakeKernel { + private result: KernelExecuteResult; + private onExecute?: (options?: KernelExecuteOptions) => void; + private alive: boolean; + readonly executeCalls: string[] = []; + shutdownCalls = 0; + + constructor( + result: KernelExecuteResult, + options: { alive?: boolean; onExecute?: (options?: KernelExecuteOptions) => void } = {}, + ) { + this.result = result; + this.onExecute = options.onExecute; + this.alive = options.alive ?? true; + } + + isAlive(): boolean { + return this.alive; + } + + async execute(code: string, options?: KernelExecuteOptions): Promise { + this.executeCalls.push(code); + this.onExecute?.(options); + return this.result; + } + + async shutdown(): Promise { + this.shutdownCalls += 1; + this.alive = false; + } + + async ping(): Promise { + return this.alive; + } +} + +const okResult: KernelExecuteResult = { + status: "ok", + cancelled: false, + timedOut: false, + stdinRequested: false, +}; + +describe("executePython session lifecycle", () => { + const originalStart = PythonKernel.start; + + afterEach(async () => { + PythonKernel.start = originalStart; + await disposeAllKernelSessions(); + }); + + it("reuses a session kernel across calls", async () => { + let startCount = 0; + const kernel = new FakeKernel(okResult, { onExecute: (options) => options?.onChunk?.("ok\n") }); + PythonKernel.start = async () => { + startCount += 1; + return kernel as unknown as PythonKernel; + }; + + const first = await executePython("print('one')", { sessionId: "session-1" }); + const second = await executePython("print('two')", { sessionId: "session-1" }); + + expect(startCount).toBe(1); + expect(kernel.executeCalls).toEqual(["print('one')", "print('two')"]); + expect(first.output).toContain("ok"); + expect(second.output).toContain("ok"); + }); + + it("restarts the session kernel when not alive", async () => { + const deadKernel = new FakeKernel(okResult, { alive: false }); + const liveKernel = new FakeKernel(okResult, { onExecute: (options) => options?.onChunk?.("live\n") }); + const kernels = [deadKernel, liveKernel]; + let startCount = 0; + + PythonKernel.start = async () => { + startCount += 1; + return kernels.shift() as unknown as PythonKernel; + }; + + const result = await executePython("print('restart')", { sessionId: "session-restart" }); + + expect(startCount).toBe(2); + expect(deadKernel.shutdownCalls).toBe(1); + expect(deadKernel.executeCalls).toEqual([]); + expect(liveKernel.executeCalls).toEqual(["print('restart')"]); + expect(result.output).toContain("live"); + }); + + it("resets the session kernel when requested", async () => { + const firstKernel = new FakeKernel(okResult); + const secondKernel = new FakeKernel(okResult); + const kernels = [firstKernel, secondKernel]; + let startCount = 0; + + PythonKernel.start = async () => { + startCount += 1; + return kernels.shift() as unknown as PythonKernel; + }; + + await executePython("print('one')", { sessionId: "session-reset" }); + await executePython("print('two')", { sessionId: "session-reset", reset: true }); + + expect(startCount).toBe(2); + expect(firstKernel.shutdownCalls).toBe(1); + expect(secondKernel.executeCalls).toEqual(["print('two')"]); + }); + + it("uses per-call kernels when configured", async () => { + const kernelA = new FakeKernel(okResult); + const kernelB = new FakeKernel(okResult); + const kernels = [kernelA, kernelB]; + let startCount = 0; + let shutdownCount = 0; + + PythonKernel.start = async () => { + startCount += 1; + return kernels.shift() as unknown as PythonKernel; + }; + + kernelA.shutdown = async () => { + shutdownCount += 1; + }; + kernelB.shutdown = async () => { + shutdownCount += 1; + }; + + await executePython("print('one')", { kernelMode: "per-call" }); + await executePython("print('two')", { kernelMode: "per-call" }); + + expect(startCount).toBe(2); + expect(shutdownCount).toBe(2); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor.result.test.ts b/packages/coding-agent/src/core/python-executor.result.test.ts new file mode 100644 index 000000000..977b247fd --- /dev/null +++ b/packages/coding-agent/src/core/python-executor.result.test.ts @@ -0,0 +1,49 @@ +import { describe, expect, it } from "bun:test"; +import { executePythonWithKernel, type PythonKernelExecutor } from "./python-executor"; +import type { KernelExecuteOptions, KernelExecuteResult } from "./python-kernel"; + +class FakeKernel implements PythonKernelExecutor { + private result: KernelExecuteResult; + private onExecute?: (options?: KernelExecuteOptions) => void; + + constructor(result: KernelExecuteResult, onExecute?: (options?: KernelExecuteOptions) => void) { + this.result = result; + this.onExecute = onExecute; + } + + async execute(_code: string, options?: KernelExecuteOptions): Promise { + this.onExecute?.(options); + return this.result; + } +} + +describe("executePythonWithKernel result mapping", () => { + it("adds timeout annotation when cancelled", async () => { + const kernel = new FakeKernel({ + status: "ok", + cancelled: true, + timedOut: true, + stdinRequested: false, + }); + + const result = await executePythonWithKernel(kernel, "sleep()", { timeout: 5000 }); + + expect(result.exitCode).toBeUndefined(); + expect(result.cancelled).toBe(true); + expect(result.output).toContain("Command timed out after 5 seconds"); + }); + + it("maps kernel error status to exit code 1", async () => { + const kernel = new FakeKernel( + { status: "error", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.("Traceback...\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "raise ValueError('boom')"); + + expect(result.exitCode).toBe(1); + expect(result.output).toContain("Traceback"); + }); +}); diff --git a/packages/coding-agent/src/core/python-executor.test.ts b/packages/coding-agent/src/core/python-executor.test.ts new file mode 100644 index 000000000..13425be9e --- /dev/null +++ b/packages/coding-agent/src/core/python-executor.test.ts @@ -0,0 +1,178 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { rmSync } from "node:fs"; +import { + disposeAllKernelSessions, + executePythonWithKernel, + getPreludeDocs, + type PythonKernelExecutor, + warmPythonEnvironment, +} from "./python-executor"; +import { type KernelExecuteOptions, type KernelExecuteResult, type PreludeHelper, PythonKernel } from "./python-kernel"; +import { DEFAULT_MAX_BYTES } from "./tools/truncate"; + +class FakeKernel implements PythonKernelExecutor { + private result: KernelExecuteResult; + private onExecute: (options?: KernelExecuteOptions) => void; + + constructor(result: KernelExecuteResult, onExecute: (options?: KernelExecuteOptions) => void) { + this.result = result; + this.onExecute = onExecute; + } + + async execute(_code: string, options?: KernelExecuteOptions): Promise { + this.onExecute(options); + return this.result; + } +} + +describe("executePythonWithKernel", () => { + it("captures text and display outputs", async () => { + const kernel = new FakeKernel( + { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.("hello\n"); + options?.onDisplay?.({ type: "json", data: { foo: "bar" } }); + }, + ); + + const result = await executePythonWithKernel(kernel, "print('hello')"); + + expect(result.exitCode).toBe(0); + expect(result.output).toContain("hello"); + expect(result.displayOutputs).toHaveLength(1); + }); + + it("marks stdin request as error", async () => { + const kernel = new FakeKernel( + { status: "ok", cancelled: false, timedOut: false, stdinRequested: true }, + () => {}, + ); + + const result = await executePythonWithKernel(kernel, "input('prompt')"); + + expect(result.exitCode).toBe(1); + expect(result.stdinRequested).toBe(true); + expect(result.output).toContain("Kernel requested stdin; interactive input is not supported."); + }); + + it("maps error status to exit code 1", async () => { + const kernel = new FakeKernel( + { status: "error", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.("Traceback\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "raise ValueError('nope')"); + + expect(result.exitCode).toBe(1); + expect(result.cancelled).toBe(false); + expect(result.output).toContain("Traceback"); + }); + + it("sanitizes streamed chunks", async () => { + const kernel = new FakeKernel( + { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.("\u001b[31mred\r\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "print('red')"); + + expect(result.output).toBe("red\n"); + }); + + it("returns cancelled result with timeout annotation", async () => { + const kernel = new FakeKernel( + { status: "ok", cancelled: true, timedOut: true, stdinRequested: false }, + (options) => { + options?.onChunk?.("partial output\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "while True: pass", { timeout: 4100 }); + + expect(result.exitCode).toBeUndefined(); + expect(result.cancelled).toBe(true); + expect(result.output).toContain("Command timed out after 4 seconds"); + }); + + it("returns cancelled result without timeout annotation", async () => { + const kernel = new FakeKernel( + { status: "ok", cancelled: true, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.("cancelled output\n"); + }, + ); + + const result = await executePythonWithKernel(kernel, "while True: pass"); + + expect(result.exitCode).toBeUndefined(); + expect(result.cancelled).toBe(true); + expect(result.output).toContain("cancelled output"); + expect(result.output).not.toContain("Command timed out"); + }); + + it("truncates large output and stores full output file", async () => { + const largeOutput = `${"x".repeat(DEFAULT_MAX_BYTES + 1024)}TAIL`; + const kernel = new FakeKernel( + { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }, + (options) => { + options?.onChunk?.(largeOutput); + }, + ); + + const result = await executePythonWithKernel(kernel, "print('big')"); + + expect(result.truncated).toBe(true); + expect(result.fullOutputPath).toBeDefined(); + expect(result.output).toContain("TAIL"); + + const fullText = await Bun.file(result.fullOutputPath as string).text(); + expect(fullText).toBe(largeOutput); + + rmSync(result.fullOutputPath as string, { force: true }); + }); +}); + +afterEach(async () => { + await disposeAllKernelSessions(); + vi.restoreAllMocks(); +}); + +describe("warmPythonEnvironment", () => { + it("caches prelude docs on warmup", async () => { + const previousSkip = process.env.OMP_PYTHON_SKIP_CHECK; + process.env.OMP_PYTHON_SKIP_CHECK = "1"; + const docs: PreludeHelper[] = [ + { + name: "read", + signature: "(path)", + docstring: "Read file contents.", + category: "File I/O", + }, + ]; + const kernel = { + introspectPrelude: vi.fn().mockResolvedValue(docs), + ping: vi.fn().mockResolvedValue(true), + isAlive: () => true, + shutdown: vi.fn().mockResolvedValue(undefined), + }; + const startSpy = vi.spyOn(PythonKernel, "start").mockResolvedValue(kernel as unknown as PythonKernel); + + const result = await warmPythonEnvironment("/tmp/test", "session-1"); + + expect(result.ok).toBe(true); + expect(result.docs).toEqual(docs); + expect(getPreludeDocs()).toEqual(docs); + expect(kernel.introspectPrelude).toHaveBeenCalledTimes(1); + + startSpy.mockRestore(); + if (previousSkip === undefined) { + delete process.env.OMP_PYTHON_SKIP_CHECK; + } else { + process.env.OMP_PYTHON_SKIP_CHECK = previousSkip; + } + }); +}); diff --git a/packages/coding-agent/src/core/python-executor.ts b/packages/coding-agent/src/core/python-executor.ts new file mode 100644 index 000000000..fcc919799 --- /dev/null +++ b/packages/coding-agent/src/core/python-executor.ts @@ -0,0 +1,294 @@ +import stripAnsi from "strip-ansi"; +import { sanitizeBinaryOutput } from "../utils/shell"; +import { logger } from "./logger"; +import { + checkPythonKernelAvailability, + type KernelDisplayOutput, + type KernelExecuteOptions, + type KernelExecuteResult, + type PreludeHelper, + PythonKernel, +} from "./python-kernel"; +import { createOutputSink } from "./streaming-output"; +import { DEFAULT_MAX_BYTES } from "./tools/truncate"; + +export type PythonKernelMode = "session" | "per-call"; + +export interface PythonExecutorOptions { + /** Working directory for command execution */ + cwd?: string; + /** Timeout in milliseconds */ + timeout?: number; + /** Callback for streaming output chunks (already sanitized) */ + onChunk?: (chunk: string) => void; + /** AbortSignal for cancellation */ + signal?: AbortSignal; + /** Session identifier for kernel reuse */ + sessionId?: string; + /** Kernel mode (session reuse vs per-call) */ + kernelMode?: PythonKernelMode; + /** Restart the kernel before executing */ + reset?: boolean; +} + +export interface PythonKernelExecutor { + execute: (code: string, options?: KernelExecuteOptions) => Promise; +} + +export interface PythonResult { + /** Combined stdout + stderr output (sanitized, possibly truncated) */ + output: string; + /** Execution exit code (0 ok, 1 error, undefined if cancelled) */ + exitCode: number | undefined; + /** Whether the execution was cancelled via signal */ + cancelled: boolean; + /** Whether the output was truncated */ + truncated: boolean; + /** Path to temp file containing full output (if output exceeded truncation threshold) */ + fullOutputPath?: string; + /** Rich display outputs captured from display_data/execute_result */ + displayOutputs: KernelDisplayOutput[]; + /** Whether stdin was requested */ + stdinRequested: boolean; +} + +interface KernelSession { + id: string; + kernel: PythonKernel; + queue: Promise; + restartCount: number; + dead: boolean; + lastUsedAt: number; + heartbeatTimer?: NodeJS.Timeout; +} + +const kernelSessions = new Map(); +let cachedPreludeDocs: PreludeHelper[] | null = null; + +export async function disposeAllKernelSessions(): Promise { + const sessions = Array.from(kernelSessions.values()); + await Promise.allSettled(sessions.map((session) => disposeKernelSession(session))); +} + +function sanitizeChunk(text: string): string { + return sanitizeBinaryOutput(stripAnsi(text)).replace(/\r/g, ""); +} + +async function ensureKernelAvailable(cwd: string): Promise { + const availability = await checkPythonKernelAvailability(cwd); + if (!availability.ok) { + throw new Error(availability.reason ?? "Python kernel unavailable"); + } +} + +export async function warmPythonEnvironment( + cwd: string, + sessionId?: string, +): Promise<{ ok: boolean; reason?: string; docs: PreludeHelper[] }> { + try { + await ensureKernelAvailable(cwd); + } catch (err: unknown) { + const reason = err instanceof Error ? err.message : String(err); + cachedPreludeDocs = []; + return { ok: false, reason, docs: [] }; + } + if (cachedPreludeDocs && cachedPreludeDocs.length > 0) { + return { ok: true, docs: cachedPreludeDocs }; + } + const resolvedSessionId = sessionId ?? `session:${cwd}`; + try { + const docs = await withKernelSession(resolvedSessionId, cwd, async (kernel) => kernel.introspectPrelude()); + cachedPreludeDocs = docs; + return { ok: true, docs }; + } catch (err: unknown) { + const reason = err instanceof Error ? err.message : String(err); + cachedPreludeDocs = []; + return { ok: false, reason, docs: [] }; + } +} + +export function getPreludeDocs(): PreludeHelper[] { + return cachedPreludeDocs ?? []; +} + +async function createKernelSession(sessionId: string, cwd: string): Promise { + const kernel = await PythonKernel.start({ cwd }); + const session: KernelSession = { + id: sessionId, + kernel, + queue: Promise.resolve(), + restartCount: 0, + dead: false, + lastUsedAt: Date.now(), + }; + + session.heartbeatTimer = setInterval(async () => { + if (session.dead) return; + const ok = await session.kernel.ping().catch(() => false); + if (!ok) { + session.dead = true; + } + }, 5000); + + return session; +} + +async function restartKernelSession(session: KernelSession, cwd: string): Promise { + session.restartCount += 1; + if (session.restartCount > 1) { + throw new Error("Python kernel restarted too many times in this session"); + } + try { + await session.kernel.shutdown(); + } catch (err) { + logger.warn("Failed to shutdown crashed kernel", { error: err instanceof Error ? err.message : String(err) }); + } + const kernel = await PythonKernel.start({ cwd }); + session.kernel = kernel; + session.dead = false; + session.lastUsedAt = Date.now(); +} + +async function disposeKernelSession(session: KernelSession): Promise { + if (session.heartbeatTimer) { + clearInterval(session.heartbeatTimer); + } + try { + await session.kernel.shutdown(); + } catch (err) { + logger.warn("Failed to shutdown kernel", { error: err instanceof Error ? err.message : String(err) }); + } + kernelSessions.delete(session.id); +} + +async function withKernelSession( + sessionId: string, + cwd: string, + handler: (kernel: PythonKernel) => Promise, +): Promise { + let session = kernelSessions.get(sessionId); + if (!session) { + session = await createKernelSession(sessionId, cwd); + kernelSessions.set(sessionId, session); + } + + const run = async (): Promise => { + session!.lastUsedAt = Date.now(); + if (session!.dead || !session!.kernel.isAlive()) { + await restartKernelSession(session!, cwd); + } + try { + const result = await handler(session!.kernel); + session!.restartCount = 0; + return result; + } catch (err) { + if (!session!.dead && session!.kernel.isAlive()) { + throw err; + } + await restartKernelSession(session!, cwd); + const result = await handler(session!.kernel); + session!.restartCount = 0; + return result; + } + }; + + const task = session.queue.then(run, run); + session.queue = task.then( + () => undefined, + () => undefined, + ); + return task; +} + +async function executeWithKernel( + kernel: PythonKernelExecutor, + code: string, + options: PythonExecutorOptions | undefined, +): Promise { + const sink = createOutputSink(DEFAULT_MAX_BYTES, DEFAULT_MAX_BYTES * 2, options?.onChunk); + const writer = sink.getWriter(); + const displayOutputs: KernelDisplayOutput[] = []; + + try { + const result = await kernel.execute(code, { + signal: options?.signal, + timeoutMs: options?.timeout, + onChunk: async (text) => { + await writer.write(sanitizeChunk(text)); + }, + onDisplay: async (output) => { + displayOutputs.push(output); + }, + }); + + if (result.cancelled) { + const secs = options?.timeout ? Math.round(options.timeout / 1000) : undefined; + const annotation = + result.timedOut && secs !== undefined ? `Command timed out after ${secs} seconds` : undefined; + return { + exitCode: undefined, + cancelled: true, + displayOutputs, + stdinRequested: result.stdinRequested, + ...sink.dump(annotation), + }; + } + + if (result.stdinRequested) { + return { + exitCode: 1, + cancelled: false, + displayOutputs, + stdinRequested: true, + ...sink.dump("Kernel requested stdin; interactive input is not supported."), + }; + } + + const exitCode = result.status === "ok" ? 0 : 1; + return { + exitCode, + cancelled: false, + displayOutputs, + stdinRequested: false, + ...sink.dump(), + }; + } catch (err) { + const error = err instanceof Error ? err : new Error(String(err)); + logger.error("Python execution failed", { error: error.message }); + throw error; + } finally { + await writer.close(); + } +} + +export async function executePythonWithKernel( + kernel: PythonKernelExecutor, + code: string, + options?: PythonExecutorOptions, +): Promise { + return await executeWithKernel(kernel, code, options); +} + +export async function executePython(code: string, options?: PythonExecutorOptions): Promise { + const cwd = options?.cwd ?? process.cwd(); + await ensureKernelAvailable(cwd); + + const kernelMode = options?.kernelMode ?? "session"; + if (kernelMode === "per-call") { + const kernel = await PythonKernel.start({ cwd }); + try { + return await executeWithKernel(kernel, code, options); + } finally { + await kernel.shutdown(); + } + } + + const sessionId = options?.sessionId ?? `session:${cwd}`; + if (options?.reset) { + const existing = kernelSessions.get(sessionId); + if (existing) { + await disposeKernelSession(existing); + } + } + return await withKernelSession(sessionId, cwd, async (kernel) => executeWithKernel(kernel, code, options)); +} diff --git a/packages/coding-agent/src/core/python-kernel-display.test.ts b/packages/coding-agent/src/core/python-kernel-display.test.ts new file mode 100644 index 000000000..37611044d --- /dev/null +++ b/packages/coding-agent/src/core/python-kernel-display.test.ts @@ -0,0 +1,54 @@ +import { describe, expect, it } from "bun:test"; +import { type KernelDisplayOutput, PythonKernel } from "./python-kernel"; + +const renderDisplay = ( + PythonKernel as unknown as { + prototype: { + renderDisplay: (content: Record) => { + text: string; + outputs: KernelDisplayOutput[]; + }; + }; + } +).prototype.renderDisplay; + +describe("PythonKernel display rendering", () => { + it("normalizes text/plain output and returns no display outputs", () => { + const { text, outputs } = renderDisplay.call({} as PythonKernel, { + data: { "text/plain": "hello" }, + }); + + expect(text).toBe("hello\n"); + expect(outputs).toHaveLength(0); + }); + + it("collects image and json display outputs without text", () => { + const { text, outputs } = renderDisplay.call({} as PythonKernel, { + data: { "image/png": "abc", "application/json": { foo: "bar" } }, + }); + + expect(text).toBe(""); + expect(outputs).toEqual([ + { type: "image", data: "abc", mimeType: "image/png" }, + { type: "json", data: { foo: "bar" } }, + ]); + }); + + it("converts text/html to markdown", () => { + const { text, outputs } = renderDisplay.call({} as PythonKernel, { + data: { "text/html": "

Hello

" }, + }); + + expect(outputs).toHaveLength(0); + expect(text).toBe("**Hello**\n"); + }); + + it("combines text/plain with json output", () => { + const { text, outputs } = renderDisplay.call({} as PythonKernel, { + data: { "text/plain": "value", "application/json": { ok: true } }, + }); + + expect(text).toBe("value\n"); + expect(outputs).toEqual([{ type: "json", data: { ok: true } }]); + }); +}); diff --git a/packages/coding-agent/src/core/python-kernel-env.test.ts b/packages/coding-agent/src/core/python-kernel-env.test.ts new file mode 100644 index 000000000..ec642e47b --- /dev/null +++ b/packages/coding-agent/src/core/python-kernel-env.test.ts @@ -0,0 +1,138 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import * as shell from "../utils/shell"; +import * as shellSnapshot from "../utils/shell-snapshot"; +import { PythonKernel } from "./python-kernel"; +import { PYTHON_PRELUDE } from "./python-prelude"; + +class FakeWebSocket { + static OPEN = 1; + static CLOSED = 3; + readyState = FakeWebSocket.OPEN; + binaryType = "arraybuffer"; + url: string; + onopen?: () => void; + onerror?: (event: unknown) => void; + onclose?: () => void; + onmessage?: (event: { data: ArrayBuffer }) => void; + + constructor(url: string) { + this.url = url; + queueMicrotask(() => { + this.onopen?.(); + }); + } + + send(_data: ArrayBuffer) {} + + close() { + this.readyState = FakeWebSocket.CLOSED; + this.onclose?.(); + } +} + +describe("PythonKernel.start (local gateway)", () => { + const originalEnv = { ...process.env }; + const originalFetch = globalThis.fetch; + const originalWebSocket = globalThis.WebSocket; + + beforeEach(() => { + process.env.BUN_ENV = "test"; + delete process.env.OMP_PYTHON_GATEWAY_URL; + delete process.env.OMP_PYTHON_GATEWAY_TOKEN; + globalThis.WebSocket = FakeWebSocket as unknown as typeof WebSocket; + }); + + afterEach(() => { + for (const key of Object.keys(process.env)) { + if (!(key in originalEnv)) { + delete process.env[key]; + } + } + for (const [key, value] of Object.entries(originalEnv)) { + process.env[key] = value; + } + globalThis.fetch = originalFetch; + globalThis.WebSocket = originalWebSocket; + vi.restoreAllMocks(); + }); + + it("filters environment variables before spawning gateway", async () => { + const fetchSpy = vi.fn(async (input: string | URL, init?: RequestInit) => { + const url = typeof input === "string" ? input : input.toString(); + if (url.endsWith("/api/kernelspecs")) { + return new Response(JSON.stringify({}), { status: 200 }); + } + if (url.endsWith("/api/kernels") && init?.method === "POST") { + return new Response(JSON.stringify({ id: "kernel-1" }), { status: 201 }); + } + return new Response("", { status: 200 }); + }); + globalThis.fetch = fetchSpy as unknown as typeof fetch; + + const shellSpy = vi.spyOn(shell, "getShellConfig").mockResolvedValue({ + shell: "/bin/bash", + args: ["-lc"], + env: { + PATH: "/bin", + HOME: "/home/test", + OPENAI_API_KEY: "secret", + UNSAFE_TOKEN: "nope", + OMP_CUSTOM: "1", + LC_ALL: "en_US.UTF-8", + }, + prefix: undefined, + }); + const snapshotSpy = vi.spyOn(shellSnapshot, "getOrCreateSnapshot").mockResolvedValue(null); + const whichSpy = vi.spyOn(Bun, "which").mockReturnValue("/usr/bin/python"); + + let spawnEnv: Record | undefined; + let spawnArgs: string[] | undefined; + const spawnSpy = vi.spyOn(Bun, "spawn").mockImplementation(((...args: unknown[]) => { + const [cmd, options] = args as [string[] | { cmd: string[] }, { env?: Record }?]; + spawnArgs = Array.isArray(cmd) ? cmd : cmd.cmd; + spawnEnv = options?.env; + return { pid: 1234, exited: Promise.resolve(0) } as unknown as Bun.Subprocess; + }) as unknown as typeof Bun.spawn); + + const executeSpy = vi + .spyOn(PythonKernel.prototype, "execute") + .mockResolvedValue({ status: "ok", cancelled: false, timedOut: false, stdinRequested: false }); + + const kernel = await PythonKernel.start({ cwd: "/tmp/project", env: { CUSTOM_VAR: "ok" } }); + + const createCall = fetchSpy.mock.calls.find(([input, init]) => { + const url = typeof input === "string" ? input : input.toString(); + return url.endsWith("/api/kernels") && init?.method === "POST"; + }); + expect(createCall).toBeDefined(); + if (createCall) { + expect(JSON.parse(String(createCall[1]?.body ?? "{}"))).toEqual({ name: "python3" }); + } + + expect(spawnArgs).toContain("kernel_gateway"); + expect(spawnEnv?.PATH).toBe("/bin"); + expect(spawnEnv?.HOME).toBe("/home/test"); + expect(spawnEnv?.OMP_CUSTOM).toBe("1"); + expect(spawnEnv?.LC_ALL).toBe("en_US.UTF-8"); + expect(spawnEnv?.CUSTOM_VAR).toBe("ok"); + expect(spawnEnv?.OPENAI_API_KEY).toBeUndefined(); + expect(spawnEnv?.UNSAFE_TOKEN).toBeUndefined(); + expect(spawnEnv?.PYTHONPATH).toBe("/tmp/project"); + + expect(executeSpy).toHaveBeenCalledWith( + PYTHON_PRELUDE, + expect.objectContaining({ + silent: true, + storeHistory: false, + }), + ); + + await kernel.shutdown(); + + shellSpy.mockRestore(); + snapshotSpy.mockRestore(); + whichSpy.mockRestore(); + spawnSpy.mockRestore(); + executeSpy.mockRestore(); + }); +}); diff --git a/packages/coding-agent/src/core/python-kernel-session.test.ts b/packages/coding-agent/src/core/python-kernel-session.test.ts new file mode 100644 index 000000000..09f44c61b --- /dev/null +++ b/packages/coding-agent/src/core/python-kernel-session.test.ts @@ -0,0 +1,87 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { disposeAllKernelSessions, executePython } from "./python-executor"; +import type { KernelExecuteOptions, KernelExecuteResult } from "./python-kernel"; +import { PythonKernel } from "./python-kernel"; + +class FakeKernel { + executeCalls = 0; + shutdownCalls = 0; + alive = true; + readonly id: string; + + constructor(id: string) { + this.id = id; + } + + async execute(_code: string, options?: KernelExecuteOptions): Promise { + this.executeCalls += 1; + options?.onChunk?.("ok\n"); + return { status: "ok", cancelled: false, timedOut: false, stdinRequested: false }; + } + + async shutdown(): Promise { + this.shutdownCalls += 1; + this.alive = false; + } + + isAlive(): boolean { + return this.alive; + } + + async ping(): Promise { + return this.alive; + } +} + +describe("executePython kernel reuse", () => { + const originalStart = PythonKernel.start; + let startCalls = 0; + let kernels: FakeKernel[] = []; + + beforeEach(() => { + process.env.OMP_PYTHON_SKIP_CHECK = "1"; + startCalls = 0; + kernels = []; + PythonKernel.start = (async () => { + startCalls += 1; + const kernel = new FakeKernel(`kernel-${startCalls}`); + kernels.push(kernel); + return kernel as unknown as PythonKernel; + }) as typeof PythonKernel.start; + }); + + afterEach(async () => { + PythonKernel.start = originalStart; + await disposeAllKernelSessions(); + }); + + it("reuses kernels for session mode", async () => { + await executePython("print('one')", { cwd: "/tmp", sessionId: "session-a", kernelMode: "session" }); + await executePython("print('two')", { cwd: "/tmp", sessionId: "session-a", kernelMode: "session" }); + + expect(startCalls).toBe(1); + expect(kernels[0]?.executeCalls).toBe(2); + }); + + it("creates and disposes per-call kernels", async () => { + await executePython("print('one')", { cwd: "/tmp", kernelMode: "per-call" }); + await executePython("print('two')", { cwd: "/tmp", kernelMode: "per-call" }); + + expect(startCalls).toBe(2); + expect(kernels[0]?.shutdownCalls).toBe(1); + expect(kernels[1]?.shutdownCalls).toBe(1); + }); + + it("resets the session kernel when requested", async () => { + await executePython("print('one')", { cwd: "/tmp", sessionId: "session-b", kernelMode: "session" }); + await executePython("print('two')", { + cwd: "/tmp", + sessionId: "session-b", + kernelMode: "session", + reset: true, + }); + + expect(startCalls).toBe(2); + expect(kernels[0]?.shutdownCalls).toBe(1); + }); +}); diff --git a/packages/coding-agent/src/core/python-kernel-ws.test.ts b/packages/coding-agent/src/core/python-kernel-ws.test.ts new file mode 100644 index 000000000..ebcc60e04 --- /dev/null +++ b/packages/coding-agent/src/core/python-kernel-ws.test.ts @@ -0,0 +1,104 @@ +import { describe, expect, it } from "bun:test"; +import { deserializeWebSocketMessage, type JupyterMessage, serializeWebSocketMessage } from "./python-kernel"; + +const encoder = new TextEncoder(); + +function buildFrame(message: Omit, buffers: Uint8Array[] = []): ArrayBuffer { + const msgBytes = encoder.encode(JSON.stringify(message)); + const offsetCount = 1 + buffers.length; + const headerSize = 4 + offsetCount * 4; + + let totalSize = headerSize + msgBytes.length; + for (const buffer of buffers) { + totalSize += buffer.length; + } + + const frame = new ArrayBuffer(totalSize); + const view = new DataView(frame); + const bytes = new Uint8Array(frame); + + view.setUint32(0, offsetCount, true); + view.setUint32(4, headerSize, true); + bytes.set(msgBytes, headerSize); + + let offset = headerSize + msgBytes.length; + for (let i = 0; i < buffers.length; i++) { + view.setUint32(4 + (i + 1) * 4, offset, true); + bytes.set(buffers[i], offset); + offset += buffers[i].length; + } + + return frame; +} + +describe("deserializeWebSocketMessage", () => { + it("parses offset tables and buffers", () => { + const message = { + channel: "iopub", + header: { + msg_id: "msg-1", + session: "session-1", + username: "omp", + date: "2024-01-01T00:00:00Z", + msg_type: "stream", + version: "5.5", + }, + parent_header: {}, + metadata: {}, + content: { text: "hello" }, + }; + const buffer = new Uint8Array([1, 2, 3]); + const frame = buildFrame(message, [buffer]); + + const parsed = deserializeWebSocketMessage(frame); + + expect(parsed).not.toBeNull(); + expect(parsed?.header.msg_id).toBe("msg-1"); + expect(parsed?.content).toEqual({ text: "hello" }); + expect(parsed?.buffers?.[0]).toEqual(buffer); + }); + + it("returns null for invalid frames", () => { + const headerSize = 8; + const bytes = encoder.encode("not-json"); + const frame = new ArrayBuffer(headerSize + bytes.length); + const view = new DataView(frame); + const data = new Uint8Array(frame); + view.setUint32(0, 1, true); + view.setUint32(4, headerSize, true); + data.set(bytes, headerSize); + + expect(deserializeWebSocketMessage(frame)).toBeNull(); + const emptyFrame = new ArrayBuffer(4); + new DataView(emptyFrame).setUint32(0, 0, true); + expect(deserializeWebSocketMessage(emptyFrame)).toBeNull(); + }); +}); + +describe("serializeWebSocketMessage", () => { + it("round trips message payloads", () => { + const message: JupyterMessage = { + channel: "shell", + header: { + msg_id: "msg-2", + session: "session-2", + username: "omp", + date: "2024-02-01T00:00:00Z", + msg_type: "execute_request", + version: "5.5", + }, + parent_header: { parent: "root" }, + metadata: { tag: "meta" }, + content: { code: "print('hi')" }, + buffers: [new Uint8Array([9, 8, 7])], + }; + + const frame = serializeWebSocketMessage(message); + const parsed = deserializeWebSocketMessage(frame); + + expect(parsed).not.toBeNull(); + expect(parsed?.header.msg_type).toBe("execute_request"); + expect(parsed?.content).toEqual({ code: "print('hi')" }); + expect(parsed?.buffers?.[0]).toEqual(new Uint8Array([9, 8, 7])); + }); +}); diff --git a/packages/coding-agent/src/core/python-kernel.lifecycle.test.ts b/packages/coding-agent/src/core/python-kernel.lifecycle.test.ts new file mode 100644 index 000000000..b88bd20cc --- /dev/null +++ b/packages/coding-agent/src/core/python-kernel.lifecycle.test.ts @@ -0,0 +1,256 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import type { Subprocess } from "bun"; +import { PythonKernel } from "./python-kernel"; + +type SpawnOptions = Parameters[1]; + +type FetchCall = { url: string; init?: RequestInit }; + +type FetchResponse = { + ok: boolean; + status: number; + json: () => Promise; + text: () => Promise; +}; + +type MockEnvironment = { + fetchCalls: FetchCall[]; + spawnCalls: { cmd: string[]; options: SpawnOptions }[]; +}; + +type MessageEventPayload = { data: ArrayBuffer }; + +type WebSocketHandler = (event: unknown) => void; + +type WebSocketMessageHandler = (event: MessageEventPayload) => void; + +class FakeWebSocket { + static OPEN = 1; + static CLOSED = 3; + static instances: FakeWebSocket[] = []; + + readyState = FakeWebSocket.OPEN; + binaryType = "arraybuffer"; + url: string; + sent: ArrayBuffer[] = []; + + onopen: WebSocketHandler | null = null; + onerror: WebSocketHandler | null = null; + onclose: WebSocketHandler | null = null; + onmessage: WebSocketMessageHandler | null = null; + + constructor(url: string) { + this.url = url; + FakeWebSocket.instances.push(this); + queueMicrotask(() => { + this.onopen?.(undefined); + }); + } + + send(data: ArrayBuffer): void { + this.sent.push(data); + } + + close(): void { + this.readyState = FakeWebSocket.CLOSED; + this.onclose?.(undefined); + } +} + +const createResponse = (options: { ok: boolean; status?: number; json?: unknown; text?: string }): FetchResponse => { + return { + ok: options.ok, + status: options.status ?? (options.ok ? 200 : 500), + json: async () => options.json ?? {}, + text: async () => options.text ?? "", + }; +}; + +const createTempDir = () => mkdtempSync(join(tmpdir(), "omp-python-kernel-")); + +const createFakeProcess = (): Subprocess => { + const exited = new Promise(() => undefined); + return { pid: 999999, exited } as Subprocess; +}; + +describe("PythonKernel gateway lifecycle", () => { + const originalFetch = globalThis.fetch; + const originalWebSocket = globalThis.WebSocket; + const originalSpawn = Bun.spawn; + const originalSleep = Bun.sleep; + const originalWhich = Bun.which; + const originalExecute = PythonKernel.prototype.execute; + const originalGatewayUrl = process.env.OMP_PYTHON_GATEWAY_URL; + const originalGatewayToken = process.env.OMP_PYTHON_GATEWAY_TOKEN; + const originalBunEnv = process.env.BUN_ENV; + + let tempDir: string; + let env: MockEnvironment; + + beforeEach(() => { + tempDir = createTempDir(); + env = { fetchCalls: [], spawnCalls: [] }; + + process.env.BUN_ENV = "test"; + delete process.env.OMP_PYTHON_GATEWAY_URL; + delete process.env.OMP_PYTHON_GATEWAY_TOKEN; + + FakeWebSocket.instances = []; + globalThis.WebSocket = FakeWebSocket as unknown as typeof WebSocket; + + Object.defineProperty(Bun, "spawn", { + value: ((cmd: string[] | string, options?: SpawnOptions) => { + const normalized = Array.isArray(cmd) ? cmd : [cmd]; + env.spawnCalls.push({ cmd: normalized, options: options ?? {} }); + return createFakeProcess(); + }) as typeof Bun.spawn, + configurable: true, + }); + + Object.defineProperty(Bun, "sleep", { + value: (async () => undefined) as typeof Bun.sleep, + configurable: true, + }); + + Object.defineProperty(Bun, "which", { + value: (() => "/usr/bin/python") as typeof Bun.which, + configurable: true, + }); + + Object.defineProperty(PythonKernel.prototype, "execute", { + value: (async () => ({ + status: "ok", + cancelled: false, + timedOut: false, + stdinRequested: false, + })) as typeof PythonKernel.prototype.execute, + configurable: true, + }); + }); + + afterEach(() => { + if (tempDir) { + rmSync(tempDir, { recursive: true, force: true }); + } + + if (originalBunEnv === undefined) { + delete process.env.BUN_ENV; + } else { + process.env.BUN_ENV = originalBunEnv; + } + if (originalGatewayUrl === undefined) { + delete process.env.OMP_PYTHON_GATEWAY_URL; + } else { + process.env.OMP_PYTHON_GATEWAY_URL = originalGatewayUrl; + } + if (originalGatewayToken === undefined) { + delete process.env.OMP_PYTHON_GATEWAY_TOKEN; + } else { + process.env.OMP_PYTHON_GATEWAY_TOKEN = originalGatewayToken; + } + + globalThis.fetch = originalFetch; + globalThis.WebSocket = originalWebSocket; + + Object.defineProperty(Bun, "spawn", { value: originalSpawn, configurable: true }); + Object.defineProperty(Bun, "sleep", { value: originalSleep, configurable: true }); + Object.defineProperty(Bun, "which", { value: originalWhich, configurable: true }); + Object.defineProperty(PythonKernel.prototype, "execute", { value: originalExecute, configurable: true }); + }); + + it("starts local gateway, polls readiness, interrupts, and shuts down", async () => { + let kernelspecAttempts = 0; + globalThis.fetch = (async (input: string | URL, init?: RequestInit) => { + const url = String(input); + env.fetchCalls.push({ url, init }); + + if (url.endsWith("/api/kernelspecs")) { + kernelspecAttempts += 1; + const ok = kernelspecAttempts >= 2; + return createResponse({ ok }) as unknown as Response; + } + + if (url.endsWith("/api/kernels") && init?.method === "POST") { + return createResponse({ ok: true, json: { id: "kernel-123" } }) as unknown as Response; + } + + return createResponse({ ok: true }) as unknown as Response; + }) as typeof fetch; + + const kernel = await PythonKernel.start({ cwd: tempDir }); + + expect(env.spawnCalls).toHaveLength(1); + expect(env.spawnCalls[0].cmd).toEqual( + expect.arrayContaining([ + "-m", + "kernel_gateway", + "--KernelGatewayApp.allow_origin=*", + "--JupyterApp.answer_yes=true", + ]), + ); + expect(env.fetchCalls.filter((call) => call.url.endsWith("/api/kernelspecs"))).toHaveLength(3); + expect(env.fetchCalls.some((call) => call.url.endsWith("/api/kernels") && call.init?.method === "POST")).toBe( + true, + ); + + await kernel.interrupt(); + expect(env.fetchCalls.some((call) => call.url.includes("/interrupt") && call.init?.method === "POST")).toBe(true); + expect(FakeWebSocket.instances[0]?.sent.length).toBe(1); + + await kernel.shutdown(); + expect(env.fetchCalls.some((call) => call.init?.method === "DELETE")).toBe(true); + expect(kernel.isAlive()).toBe(false); + }); + + it("throws when gateway readiness never succeeds", async () => { + const originalNow = Date.now; + let now = 0; + Date.now = () => { + now += 1000; + return now; + }; + + try { + globalThis.fetch = (async (input: string | URL, init?: RequestInit) => { + const url = String(input); + env.fetchCalls.push({ url, init }); + if (url.endsWith("/api/kernelspecs")) { + return createResponse({ ok: false, status: 503 }) as unknown as Response; + } + return createResponse({ ok: true }) as unknown as Response; + }) as typeof fetch; + + await expect(PythonKernel.start({ cwd: tempDir })).rejects.toThrow("Kernel gateway failed to start"); + expect(env.spawnCalls).toHaveLength(1); + } finally { + Date.now = originalNow; + } + }); + + it("does not throw when shutdown API fails", async () => { + let kernelspecAttempts = 0; + globalThis.fetch = (async (input: string | URL, init?: RequestInit) => { + const url = String(input); + env.fetchCalls.push({ url, init }); + if (url.endsWith("/api/kernelspecs")) { + kernelspecAttempts += 1; + const ok = kernelspecAttempts >= 1; + return createResponse({ ok }) as unknown as Response; + } + if (url.endsWith("/api/kernels") && init?.method === "POST") { + return createResponse({ ok: true, json: { id: "kernel-456" } }) as unknown as Response; + } + if (init?.method === "DELETE") { + throw new Error("delete failed"); + } + return createResponse({ ok: true }) as unknown as Response; + }) as typeof fetch; + + const kernel = await PythonKernel.start({ cwd: tempDir }); + + await expect(kernel.shutdown()).resolves.toBeUndefined(); + }); +}); diff --git a/packages/coding-agent/src/core/python-kernel.test.ts b/packages/coding-agent/src/core/python-kernel.test.ts new file mode 100644 index 000000000..317864a07 --- /dev/null +++ b/packages/coding-agent/src/core/python-kernel.test.ts @@ -0,0 +1,535 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { type KernelDisplayOutput, PythonKernel } from "./python-kernel"; +import { PYTHON_PRELUDE } from "./python-prelude"; + +type JupyterMessage = { + channel: string; + header: { + msg_id: string; + session: string; + username: string; + date: string; + msg_type: string; + version: string; + }; + parent_header: Record; + metadata: Record; + content: Record; + buffers?: Uint8Array[]; +}; + +const textEncoder = new TextEncoder(); +const textDecoder = new TextDecoder(); + +function encodeMessage(msg: JupyterMessage): ArrayBuffer { + const msgText = JSON.stringify({ + channel: msg.channel, + header: msg.header, + parent_header: msg.parent_header, + metadata: msg.metadata, + content: msg.content, + }); + const msgBytes = textEncoder.encode(msgText); + const buffers = msg.buffers ?? []; + const offsetCount = 1 + buffers.length; + const headerSize = 4 + offsetCount * 4; + let totalSize = headerSize + msgBytes.length; + for (const buffer of buffers) { + totalSize += buffer.length; + } + const result = new ArrayBuffer(totalSize); + const view = new DataView(result); + const bytes = new Uint8Array(result); + view.setUint32(0, offsetCount, true); + let offset = headerSize; + view.setUint32(4, offset, true); + bytes.set(msgBytes, offset); + offset += msgBytes.length; + buffers.forEach((buffer, index) => { + view.setUint32(4 + (index + 1) * 4, offset, true); + bytes.set(buffer, offset); + offset += buffer.length; + }); + return result; +} + +function decodeMessage(data: ArrayBuffer): JupyterMessage { + const view = new DataView(data); + const offsetCount = view.getUint32(0, true); + const offsets: number[] = []; + for (let i = 0; i < offsetCount; i++) { + offsets.push(view.getUint32(4 + i * 4, true)); + } + const msgStart = offsets[0]; + const msgEnd = offsets.length > 1 ? offsets[1] : data.byteLength; + const msgBytes = new Uint8Array(data, msgStart, msgEnd - msgStart); + const msgText = textDecoder.decode(msgBytes); + return JSON.parse(msgText) as JupyterMessage; +} + +class FakeWebSocket { + static OPEN = 1; + static CLOSED = 3; + static lastInstance: FakeWebSocket | null = null; + readyState = FakeWebSocket.OPEN; + binaryType = "arraybuffer"; + onopen?: () => void; + onmessage?: (event: { data: ArrayBuffer }) => void; + onerror?: (event: unknown) => void; + onclose?: () => void; + readonly url: string; + readonly sent: ArrayBuffer[] = []; + private handleSend: ((data: ArrayBuffer) => void) | null = null; + + constructor(url: string) { + this.url = url; + FakeWebSocket.lastInstance = this; + queueMicrotask(() => this.onopen?.()); + } + + setSendHandler(handler: (data: ArrayBuffer) => void) { + this.handleSend = handler; + } + + send(data: ArrayBuffer) { + this.sent.push(data); + this.handleSend?.(data); + } + + close() { + this.readyState = FakeWebSocket.CLOSED; + this.onclose?.(); + } +} + +describe("PythonKernel (external gateway)", () => { + const originalEnv = { ...process.env }; + const originalFetch = globalThis.fetch; + const originalWebSocket = globalThis.WebSocket; + + beforeEach(() => { + process.env.OMP_PYTHON_GATEWAY_URL = "http://gateway.test"; + globalThis.WebSocket = FakeWebSocket as unknown as typeof WebSocket; + }); + + afterEach(() => { + for (const key of Object.keys(process.env)) { + if (!(key in originalEnv)) { + delete process.env[key]; + } + } + for (const [key, value] of Object.entries(originalEnv)) { + process.env[key] = value; + } + globalThis.fetch = originalFetch; + globalThis.WebSocket = originalWebSocket; + FakeWebSocket.lastInstance = null; + vi.restoreAllMocks(); + }); + + it("executes code via websocket stream and display data", async () => { + const fetchMock = vi.fn(async (url: string, init?: RequestInit) => { + if (url.endsWith("/api/kernels") && init?.method === "POST") { + return new Response(JSON.stringify({ id: "kernel-1" }), { status: 201 }); + } + if (url.includes("/api/kernels/") && init?.method === "DELETE") { + return new Response("", { status: 204 }); + } + return new Response("", { status: 200 }); + }); + globalThis.fetch = fetchMock as unknown as typeof fetch; + + const responseQueue: Array<(msgId: string, ws: FakeWebSocket) => void> = []; + responseQueue.push((msgId, ws) => { + const reply: JupyterMessage = { + channel: "shell", + header: { + msg_id: "reply-1", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "execute_reply", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { status: "ok", execution_count: 1 }, + }; + const status: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "status-1", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "status", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { execution_state: "idle" }, + }; + ws.onmessage?.({ data: encodeMessage(reply) }); + ws.onmessage?.({ data: encodeMessage(status) }); + }); + responseQueue.push((msgId, ws) => { + const stream: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "stream-1", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "stream", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { text: "hello\n" }, + }; + const display: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "display-1", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "execute_result", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { + data: { + "text/plain": "result", + "application/json": { answer: 42 }, + }, + }, + }; + const reply: JupyterMessage = { + channel: "shell", + header: { + msg_id: "reply-2", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "execute_reply", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { status: "ok", execution_count: 2 }, + }; + const status: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "status-2", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "status", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { execution_state: "idle" }, + }; + ws.onmessage?.({ data: encodeMessage(stream) }); + ws.onmessage?.({ data: encodeMessage(display) }); + ws.onmessage?.({ data: encodeMessage(reply) }); + ws.onmessage?.({ data: encodeMessage(status) }); + }); + + const kernelPromise = PythonKernel.start({ cwd: "/" }); + const ws = FakeWebSocket.lastInstance; + if (!ws) throw new Error("WebSocket not initialized"); + ws.setSendHandler((data) => { + const msg = decodeMessage(data); + const handler = responseQueue.shift(); + if (!handler) { + throw new Error(`Unexpected message: ${msg.header.msg_type}`); + } + handler(msg.header.msg_id, ws); + }); + + const kernel = await kernelPromise; + const chunks: string[] = []; + const displays: KernelDisplayOutput[] = []; + + const result = await kernel.execute("print('hello')", { + onChunk: (text) => { + chunks.push(text); + }, + onDisplay: (output) => { + displays.push(output); + }, + }); + + expect(result.status).toBe("ok"); + expect(chunks.join("")).toContain("hello"); + expect(chunks.join("")).toContain("result"); + expect(displays).toEqual([{ type: "json", data: { answer: 42 } }]); + + await kernel.shutdown(); + expect(fetchMock).toHaveBeenCalledWith("http://gateway.test/api/kernels/kernel-1", { + method: "DELETE", + headers: {}, + }); + }); + + it("marks kernel dead after repeated ping failures", async () => { + const fetchMock = vi.fn(async (url: string, init?: RequestInit) => { + if (url.endsWith("/api/kernels") && init?.method === "POST") { + return new Response(JSON.stringify({ id: "kernel-2" }), { status: 201 }); + } + if (url.includes("/api/kernels/kernel-2") && !init?.method) { + throw new Error("ping failed"); + } + return new Response("", { status: 200 }); + }); + globalThis.fetch = fetchMock as unknown as typeof fetch; + + const responseQueue: Array<(msgId: string, ws: FakeWebSocket) => void> = [ + (msgId, ws) => { + const reply: JupyterMessage = { + channel: "shell", + header: { + msg_id: "reply-prelude", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "execute_reply", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { status: "ok", execution_count: 1 }, + }; + const status: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "status-prelude", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "status", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { execution_state: "idle" }, + }; + ws.onmessage?.({ data: encodeMessage(reply) }); + ws.onmessage?.({ data: encodeMessage(status) }); + }, + ]; + + const kernelPromise = PythonKernel.start({ cwd: "/" }); + const ws = FakeWebSocket.lastInstance; + if (!ws) throw new Error("WebSocket not initialized"); + ws.setSendHandler((data) => { + const msg = decodeMessage(data); + const handler = responseQueue.shift(); + if (!handler) { + throw new Error(`Unexpected message: ${msg.header.msg_type}`); + } + handler(msg.header.msg_id, ws); + }); + + const kernel = await kernelPromise; + const firstPing = await kernel.ping(1); + const secondPing = await kernel.ping(1); + + expect(firstPing).toBe(false); + expect(secondPing).toBe(false); + expect(kernel.isAlive()).toBe(false); + + await kernel.shutdown(); + }); + + it("initializes the IPython prelude", async () => { + const fetchMock = vi.fn(async (url: string, init?: RequestInit) => { + if (url.endsWith("/api/kernels") && init?.method === "POST") { + return new Response(JSON.stringify({ id: "kernel-3" }), { status: 201 }); + } + return new Response("", { status: 200 }); + }); + globalThis.fetch = fetchMock as unknown as typeof fetch; + + const responseQueue: Array<(msgId: string, ws: FakeWebSocket) => void> = [ + (msgId, ws) => { + const reply: JupyterMessage = { + channel: "shell", + header: { + msg_id: "reply-prelude", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "execute_reply", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { status: "ok", execution_count: 1 }, + }; + const status: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "status-prelude", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "status", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { execution_state: "idle" }, + }; + ws.onmessage?.({ data: encodeMessage(reply) }); + ws.onmessage?.({ data: encodeMessage(status) }); + }, + ]; + + const kernelPromise = PythonKernel.start({ cwd: "/" }); + const ws = FakeWebSocket.lastInstance; + if (!ws) throw new Error("WebSocket not initialized"); + ws.setSendHandler((data) => { + const msg = decodeMessage(data); + const handler = responseQueue.shift(); + if (!handler) { + throw new Error(`Unexpected message: ${msg.header.msg_type}`); + } + expect(msg.content.code).toBe(PYTHON_PRELUDE); + handler(msg.header.msg_id, ws); + }); + + const kernel = await kernelPromise; + expect(kernel.isAlive()).toBe(true); + await kernel.shutdown(); + }); + + it("introspects prelude helpers", async () => { + const fetchMock = vi.fn(async (url: string, init?: RequestInit) => { + if (url.endsWith("/api/kernels") && init?.method === "POST") { + return new Response(JSON.stringify({ id: "kernel-4" }), { status: 201 }); + } + return new Response("", { status: 200 }); + }); + globalThis.fetch = fetchMock as unknown as typeof fetch; + + const docs = [ + { + name: "read", + signature: "(path, limit=None)", + docstring: "Read file contents.", + category: "File I/O", + }, + ]; + const payload = JSON.stringify(docs); + + const responseQueue: Array<(msgId: string, ws: FakeWebSocket) => void> = [ + (msgId, ws) => { + const reply: JupyterMessage = { + channel: "shell", + header: { + msg_id: "reply-prelude", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "execute_reply", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { status: "ok", execution_count: 1 }, + }; + const status: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "status-prelude", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "status", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { execution_state: "idle" }, + }; + ws.onmessage?.({ data: encodeMessage(reply) }); + ws.onmessage?.({ data: encodeMessage(status) }); + }, + (msgId, ws) => { + const stream: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "stream-docs", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "stream", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { text: `${payload}\n` }, + }; + const reply: JupyterMessage = { + channel: "shell", + header: { + msg_id: "reply-docs", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "execute_reply", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { status: "ok", execution_count: 2 }, + }; + const status: JupyterMessage = { + channel: "iopub", + header: { + msg_id: "status-docs", + session: "session", + username: "omp", + date: new Date().toISOString(), + msg_type: "status", + version: "5.5", + }, + parent_header: { msg_id: msgId }, + metadata: {}, + content: { execution_state: "idle" }, + }; + ws.onmessage?.({ data: encodeMessage(stream) }); + ws.onmessage?.({ data: encodeMessage(reply) }); + ws.onmessage?.({ data: encodeMessage(status) }); + }, + ]; + + const kernelPromise = PythonKernel.start({ cwd: "/" }); + const ws = FakeWebSocket.lastInstance; + if (!ws) throw new Error("WebSocket not initialized"); + ws.setSendHandler((data) => { + const msg = decodeMessage(data); + const handler = responseQueue.shift(); + if (!handler) { + throw new Error(`Unexpected message: ${msg.header.msg_type}`); + } + if (msg.content.code !== PYTHON_PRELUDE) { + expect(String(msg.content.code)).toContain("__omp_prelude_docs__"); + } + handler(msg.header.msg_id, ws); + }); + + const kernel = await kernelPromise; + const result = await kernel.introspectPrelude(); + expect(result).toEqual(docs); + await kernel.shutdown(); + }); +}); + +// TODO: add coverage for gateway process exit handling once PythonKernel exposes a test hook. diff --git a/packages/coding-agent/src/core/python-kernel.ts b/packages/coding-agent/src/core/python-kernel.ts new file mode 100644 index 000000000..8981c0597 --- /dev/null +++ b/packages/coding-agent/src/core/python-kernel.ts @@ -0,0 +1,952 @@ +import { createServer } from "node:net"; +import { delimiter, join } from "node:path"; +import type { Subprocess } from "bun"; +import { nanoid } from "nanoid"; +import { getShellConfig, killProcessTree } from "../utils/shell"; +import { getOrCreateSnapshot } from "../utils/shell-snapshot"; +import { logger } from "./logger"; +import { PYTHON_PRELUDE } from "./python-prelude"; +import { htmlToBasicMarkdown } from "./tools/web-scrapers/types"; +import { ScopeSignal } from "./utils"; + +const TEXT_ENCODER = new TextEncoder(); +const TEXT_DECODER = new TextDecoder(); +const HEARTBEAT_INTERVAL_MS = 5000; +const HEARTBEAT_TIMEOUT_MS = 2000; +const HEARTBEAT_FAILURE_LIMIT = 1; +const GATEWAY_STARTUP_TIMEOUT_MS = 30000; +const GATEWAY_STARTUP_ATTEMPTS = 3; +const TRACE_IPC = process.env.OMP_PYTHON_IPC_TRACE === "1"; +const PRELUDE_INTROSPECTION_SNIPPET = "import json\nprint(json.dumps(__omp_prelude_docs__()))"; + +interface ExternalGatewayConfig { + url: string; + token?: string; +} + +function getExternalGatewayConfig(): ExternalGatewayConfig | null { + const url = process.env.OMP_PYTHON_GATEWAY_URL; + if (!url) return null; + return { + url: url.replace(/\/$/, ""), + token: process.env.OMP_PYTHON_GATEWAY_TOKEN, + }; +} + +const DEFAULT_ENV_ALLOWLIST = new Set([ + "PATH", + "HOME", + "USER", + "LOGNAME", + "SHELL", + "LANG", + "LC_ALL", + "LC_CTYPE", + "LC_MESSAGES", + "TERM", + "TERM_PROGRAM", + "TERM_PROGRAM_VERSION", + "TMPDIR", + "TEMP", + "TMP", + "XDG_CACHE_HOME", + "XDG_CONFIG_HOME", + "XDG_DATA_HOME", + "XDG_RUNTIME_DIR", + "SSH_AUTH_SOCK", + "SSH_AGENT_PID", + "CONDA_PREFIX", + "CONDA_DEFAULT_ENV", + "VIRTUAL_ENV", + "PYTHONPATH", +]); + +const DEFAULT_ENV_ALLOW_PREFIXES = ["LC_", "XDG_", "OMP_"]; + +const DEFAULT_ENV_DENYLIST = new Set([ + "OPENAI_API_KEY", + "ANTHROPIC_API_KEY", + "GOOGLE_API_KEY", + "GEMINI_API_KEY", + "OPENROUTER_API_KEY", + "PERPLEXITY_API_KEY", + "EXA_API_KEY", + "AZURE_OPENAI_API_KEY", + "MISTRAL_API_KEY", +]); + +export interface JupyterHeader { + msg_id: string; + session: string; + username: string; + date: string; + msg_type: string; + version: string; +} + +export interface JupyterMessage { + channel: string; + header: JupyterHeader; + parent_header: Record; + metadata: Record; + content: Record; + buffers?: Uint8Array[]; +} + +export type KernelDisplayOutput = { type: "json"; data: unknown } | { type: "image"; data: string; mimeType: string }; + +export interface KernelExecuteOptions { + signal?: AbortSignal; + onChunk?: (text: string) => Promise | void; + onDisplay?: (output: KernelDisplayOutput) => Promise | void; + timeoutMs?: number; + silent?: boolean; + storeHistory?: boolean; + allowStdin?: boolean; +} + +export interface KernelExecuteResult { + status: "ok" | "error"; + executionCount?: number; + error?: { name: string; value: string; traceback: string[] }; + cancelled: boolean; + timedOut: boolean; + stdinRequested: boolean; +} + +export interface PreludeHelper { + name: string; + signature: string; + docstring: string; + category: string; +} + +interface KernelStartOptions { + cwd: string; + env?: Record; +} + +export interface PythonKernelAvailability { + ok: boolean; + pythonPath?: string; + reason?: string; +} + +function filterEnv(env: Record): Record { + const filtered: Record = {}; + for (const [key, value] of Object.entries(env)) { + if (value === undefined) continue; + if (DEFAULT_ENV_DENYLIST.has(key)) continue; + if (DEFAULT_ENV_ALLOWLIST.has(key)) { + filtered[key] = value; + continue; + } + if (DEFAULT_ENV_ALLOW_PREFIXES.some((prefix) => key.startsWith(prefix))) { + filtered[key] = value; + } + } + return filtered; +} + +async function resolveVenvPath(cwd: string): Promise { + if (process.env.VIRTUAL_ENV) return process.env.VIRTUAL_ENV; + const candidates = [join(cwd, ".venv"), join(cwd, "venv")]; + for (const candidate of candidates) { + if (await Bun.file(candidate).exists()) { + return candidate; + } + } + return null; +} + +async function resolvePythonRuntime(cwd: string, baseEnv: Record) { + const env = { ...baseEnv }; + const venvPath = env.VIRTUAL_ENV ?? (await resolveVenvPath(cwd)); + if (venvPath) { + env.VIRTUAL_ENV = venvPath; + const binDir = process.platform === "win32" ? join(venvPath, "Scripts") : join(venvPath, "bin"); + const pythonCandidate = join(binDir, process.platform === "win32" ? "python.exe" : "python"); + if (await Bun.file(pythonCandidate).exists()) { + env.PATH = env.PATH ? `${binDir}${delimiter}${env.PATH}` : binDir; + return { pythonPath: pythonCandidate, env }; + } + } + + const pythonPath = Bun.which("python") ?? Bun.which("python3"); + if (!pythonPath) { + throw new Error("Python executable not found on PATH"); + } + return { pythonPath, env }; +} + +export async function checkPythonKernelAvailability(cwd: string): Promise { + if (process.env.BUN_ENV === "test" || process.env.NODE_ENV === "test" || process.env.OMP_PYTHON_SKIP_CHECK === "1") { + return { ok: true }; + } + + const externalConfig = getExternalGatewayConfig(); + if (externalConfig) { + return checkExternalGatewayAvailability(externalConfig); + } + + try { + const { env } = await getShellConfig(); + const baseEnv = filterEnv(env); + const runtime = await resolvePythonRuntime(cwd, baseEnv); + const result = Bun.spawnSync( + [ + runtime.pythonPath, + "-c", + "import importlib.util,sys;sys.exit(0 if importlib.util.find_spec('kernel_gateway') and importlib.util.find_spec('ipykernel') else 1)", + ], + { cwd, env: runtime.env, stdin: "ignore", stdout: "pipe", stderr: "pipe" }, + ); + if (result.exitCode === 0) { + return { ok: true, pythonPath: runtime.pythonPath }; + } + return { + ok: false, + pythonPath: runtime.pythonPath, + reason: + "kernel_gateway (jupyter-kernel-gateway) or ipykernel not installed. Run: python -m pip install jupyter_kernel_gateway ipykernel", + }; + } catch (err: unknown) { + return { ok: false, reason: err instanceof Error ? err.message : String(err) }; + } +} + +async function checkExternalGatewayAvailability(config: ExternalGatewayConfig): Promise { + try { + const headers: Record = {}; + if (config.token) { + headers.Authorization = `token ${config.token}`; + } + + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), 5000); + + const response = await fetch(`${config.url}/api/kernelspecs`, { + headers, + signal: controller.signal, + }); + clearTimeout(timeout); + + if (response.ok) { + return { ok: true }; + } + + if (response.status === 401 || response.status === 403) { + return { + ok: false, + reason: `External gateway at ${config.url} requires authentication. Set OMP_PYTHON_GATEWAY_TOKEN.`, + }; + } + + return { + ok: false, + reason: `External gateway at ${config.url} returned status ${response.status}`, + }; + } catch (err: unknown) { + const message = err instanceof Error ? err.message : String(err); + if (message.includes("abort") || message.includes("timeout")) { + return { + ok: false, + reason: `External gateway at ${config.url} is not reachable (timeout)`, + }; + } + return { + ok: false, + reason: `External gateway at ${config.url} is not reachable: ${message}`, + }; + } +} + +async function allocatePort(): Promise { + return await new Promise((resolve, reject) => { + const server = createServer(); + server.unref(); + server.on("error", reject); + server.listen(0, "127.0.0.1", () => { + const address = server.address(); + if (address && typeof address === "object") { + const port = address.port; + server.close((err: Error | null | undefined) => { + if (err) { + reject(err); + } else { + resolve(port); + } + }); + } else { + server.close(); + reject(new Error("Failed to allocate port")); + } + }); + }); +} + +function normalizeDisplayText(text: string): string { + return text.endsWith("\n") ? text : `${text}\n`; +} + +export function deserializeWebSocketMessage(data: ArrayBuffer): JupyterMessage | null { + const view = new DataView(data); + const offsetCount = view.getUint32(0, true); + + if (offsetCount < 1) return null; + + const offsets: number[] = []; + for (let i = 0; i < offsetCount; i++) { + offsets.push(view.getUint32(4 + i * 4, true)); + } + + const msgStart = offsets[0]; + const msgEnd = offsets.length > 1 ? offsets[1] : data.byteLength; + const msgBytes = new Uint8Array(data, msgStart, msgEnd - msgStart); + const msgText = TEXT_DECODER.decode(msgBytes); + + try { + const msg = JSON.parse(msgText) as { + channel: string; + header: JupyterHeader; + parent_header: Record; + metadata: Record; + content: Record; + }; + + const buffers: Uint8Array[] = []; + for (let i = 1; i < offsets.length; i++) { + const start = offsets[i]; + const end = i + 1 < offsets.length ? offsets[i + 1] : data.byteLength; + buffers.push(new Uint8Array(data, start, end - start)); + } + + return { ...msg, buffers }; + } catch { + return null; + } +} + +export function serializeWebSocketMessage(msg: JupyterMessage): ArrayBuffer { + const msgText = JSON.stringify({ + channel: msg.channel, + header: msg.header, + parent_header: msg.parent_header, + metadata: msg.metadata, + content: msg.content, + }); + const msgBytes = TEXT_ENCODER.encode(msgText); + + const buffers = msg.buffers ?? []; + const offsetCount = 1 + buffers.length; + const headerSize = 4 + offsetCount * 4; + + let totalSize = headerSize + msgBytes.length; + for (const buf of buffers) { + totalSize += buf.length; + } + + const result = new ArrayBuffer(totalSize); + const view = new DataView(result); + const bytes = new Uint8Array(result); + + view.setUint32(0, offsetCount, true); + + let offset = headerSize; + view.setUint32(4, offset, true); + bytes.set(msgBytes, offset); + offset += msgBytes.length; + + for (let i = 0; i < buffers.length; i++) { + view.setUint32(4 + (i + 1) * 4, offset, true); + bytes.set(buffers[i], offset); + offset += buffers[i].length; + } + + return result; +} + +export class PythonKernel { + readonly id: string; + readonly kernelId: string; + readonly gatewayProcess: Subprocess | null; + readonly gatewayUrl: string; + readonly sessionId: string; + readonly username: string; + readonly #authToken?: string; + + #ws: WebSocket | null = null; + #disposed = false; + #alive = true; + #heartbeatTimer?: NodeJS.Timeout; + #heartbeatFailures = 0; + #messageHandlers = new Map void>(); + #channelHandlers = new Map void>>(); + + private constructor( + id: string, + kernelId: string, + gatewayProcess: Subprocess | null, + gatewayUrl: string, + sessionId: string, + username: string, + authToken?: string, + ) { + this.id = id; + this.kernelId = kernelId; + this.gatewayProcess = gatewayProcess; + this.gatewayUrl = gatewayUrl; + this.sessionId = sessionId; + this.username = username; + this.#authToken = authToken; + + if (this.gatewayProcess) { + this.gatewayProcess.exited.then(() => { + this.#alive = false; + }); + } + } + + #authHeaders(): Record { + if (!this.#authToken) return {}; + return { Authorization: `token ${this.#authToken}` }; + } + + static async start(options: KernelStartOptions): Promise { + const availability = await checkPythonKernelAvailability(options.cwd); + if (!availability.ok) { + throw new Error(availability.reason ?? "Python kernel unavailable"); + } + + const externalConfig = getExternalGatewayConfig(); + if (externalConfig) { + return PythonKernel.startWithExternalGateway(externalConfig); + } + + return PythonKernel.startWithLocalGateway(options); + } + + private static async startWithExternalGateway(config: ExternalGatewayConfig): Promise { + const headers: Record = { "Content-Type": "application/json" }; + if (config.token) { + headers.Authorization = `token ${config.token}`; + } + + const createResponse = await fetch(`${config.url}/api/kernels`, { + method: "POST", + headers, + body: JSON.stringify({ name: "python3" }), + }); + + if (!createResponse.ok) { + throw new Error(`Failed to create kernel on external gateway: ${await createResponse.text()}`); + } + + const kernelInfo = (await createResponse.json()) as { id: string }; + const kernelId = kernelInfo.id; + + const kernel = new PythonKernel(nanoid(), kernelId, null, config.url, nanoid(), "omp", config.token); + + try { + await kernel.connectWebSocket(); + kernel.startHeartbeat(); + const preludeResult = await kernel.execute(PYTHON_PRELUDE, { silent: true, storeHistory: false }); + if (preludeResult.cancelled || preludeResult.status === "error") { + throw new Error("Failed to initialize Python kernel prelude"); + } + return kernel; + } catch (err: unknown) { + await kernel.shutdown(); + throw err; + } + } + + private static async startWithLocalGateway(options: KernelStartOptions): Promise { + const { shell, env } = await getShellConfig(); + const filteredEnv = filterEnv(env); + const runtime = await resolvePythonRuntime(options.cwd, filteredEnv); + const snapshotPath = await getOrCreateSnapshot(shell, env).catch((err: unknown) => { + logger.warn("Failed to resolve shell snapshot for Python kernel", { + error: err instanceof Error ? err.message : String(err), + }); + return null; + }); + + const kernelEnv: Record = { + ...runtime.env, + ...options.env, + PYTHONUNBUFFERED: "1", + OMP_SHELL_SNAPSHOT: snapshotPath ?? undefined, + }; + + const pythonPathParts = [options.cwd, kernelEnv.PYTHONPATH].filter(Boolean).join(delimiter); + if (pythonPathParts) { + kernelEnv.PYTHONPATH = pythonPathParts; + } + + let gatewayProcess: Subprocess | null = null; + let gatewayUrl: string | null = null; + let lastError: string | null = null; + + for (let attempt = 0; attempt < GATEWAY_STARTUP_ATTEMPTS; attempt += 1) { + const gatewayPort = await allocatePort(); + const candidateUrl = `http://127.0.0.1:${gatewayPort}`; + const candidateProcess = Bun.spawn( + [ + runtime.pythonPath, + "-m", + "kernel_gateway", + "--KernelGatewayApp.ip=127.0.0.1", + `--KernelGatewayApp.port=${gatewayPort}`, + "--KernelGatewayApp.port_retries=0", + "--KernelGatewayApp.allow_origin=*", + "--JupyterApp.answer_yes=true", + ], + { + cwd: options.cwd, + stdin: "ignore", + stdout: "pipe", + stderr: "pipe", + env: kernelEnv, + }, + ); + + let exited = false; + candidateProcess.exited + .then(() => { + exited = true; + }) + .catch(() => { + exited = true; + }); + + const startTime = Date.now(); + while (Date.now() - startTime < GATEWAY_STARTUP_TIMEOUT_MS) { + if (exited) break; + try { + const response = await fetch(`${candidateUrl}/api/kernelspecs`); + if (response.ok) { + gatewayProcess = candidateProcess; + gatewayUrl = candidateUrl; + break; + } + } catch { + // Gateway not ready yet + } + await Bun.sleep(100); + } + + if (gatewayProcess && gatewayUrl) break; + + killProcessTree(candidateProcess.pid); + lastError = exited ? "Kernel gateway process exited during startup" : "Kernel gateway failed to start"; + } + + if (!gatewayProcess || !gatewayUrl) { + throw new Error(lastError ?? "Kernel gateway failed to start"); + } + + const createResponse = await fetch(`${gatewayUrl}/api/kernels`, { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ name: "python3" }), + }); + + if (!createResponse.ok) { + killProcessTree(gatewayProcess.pid); + throw new Error(`Failed to create kernel: ${await createResponse.text()}`); + } + + const kernelInfo = (await createResponse.json()) as { id: string }; + const kernelId = kernelInfo.id; + + const kernel = new PythonKernel(nanoid(), kernelId, gatewayProcess, gatewayUrl, nanoid(), "omp"); + + try { + await kernel.connectWebSocket(); + kernel.startHeartbeat(); + const preludeResult = await kernel.execute(PYTHON_PRELUDE, { silent: true, storeHistory: false }); + if (preludeResult.cancelled || preludeResult.status === "error") { + throw new Error("Failed to initialize Python kernel prelude"); + } + return kernel; + } catch (err: unknown) { + await kernel.shutdown(); + throw err; + } + } + + private async connectWebSocket(): Promise { + const wsBase = this.gatewayUrl.replace(/^http/, "ws"); + let wsUrl = `${wsBase}/api/kernels/${this.kernelId}/channels`; + if (this.#authToken) { + wsUrl += `?token=${encodeURIComponent(this.#authToken)}`; + } + + return new Promise((resolve, reject) => { + const ws = new WebSocket(wsUrl); + ws.binaryType = "arraybuffer"; + + const timeout = setTimeout(() => { + ws.close(); + reject(new Error("WebSocket connection timeout")); + }, 10000); + + ws.onopen = () => { + clearTimeout(timeout); + this.#ws = ws; + resolve(); + }; + + ws.onerror = (event) => { + clearTimeout(timeout); + reject(new Error(`WebSocket error: ${event}`)); + }; + + ws.onclose = () => { + this.#alive = false; + }; + + ws.onmessage = (event) => { + let msg: JupyterMessage | null = null; + if (event.data instanceof ArrayBuffer) { + msg = deserializeWebSocketMessage(event.data); + } else if (typeof event.data === "string") { + try { + msg = JSON.parse(event.data) as JupyterMessage; + } catch { + return; + } + } + if (!msg) return; + + if (TRACE_IPC) { + logger.debug("Kernel IPC recv", { channel: msg.channel, msgType: msg.header.msg_type }); + } + + const parentId = (msg.parent_header as { msg_id?: string }).msg_id; + if (parentId) { + const handler = this.#messageHandlers.get(parentId); + if (handler) handler(msg); + } + + const channelHandlers = this.#channelHandlers.get(msg.channel); + if (channelHandlers) { + for (const handler of channelHandlers) { + handler(msg); + } + } + }; + }); + } + + isAlive(): boolean { + return this.#alive && !this.#disposed && this.#ws?.readyState === WebSocket.OPEN; + } + + async execute(code: string, options?: KernelExecuteOptions): Promise { + if (!this.isAlive()) { + throw new Error("Python kernel is not running"); + } + + const msgId = nanoid(); + const msg: JupyterMessage = { + channel: "shell", + header: { + msg_id: msgId, + session: this.sessionId, + username: this.username, + date: new Date().toISOString(), + msg_type: "execute_request", + version: "5.5", + }, + parent_header: {}, + metadata: {}, + content: { + code, + silent: options?.silent ?? false, + store_history: options?.storeHistory ?? !(options?.silent ?? false), + user_expressions: {}, + allow_stdin: options?.allowStdin ?? false, + stop_on_error: true, + }, + }; + + let status: "ok" | "error" = "ok"; + let executionCount: number | undefined; + let error: { name: string; value: string; traceback: string[] } | undefined; + let replyReceived = false; + let idleReceived = false; + let stdinRequested = false; + let cancelled = false; + let timedOut = false; + + using executionSignal = new ScopeSignal({ signal: options?.signal, timeout: options?.timeoutMs }); + + return new Promise((resolve) => { + const cleanup = () => { + this.#messageHandlers.delete(msgId); + resolve({ status, executionCount, error, cancelled, timedOut, stdinRequested }); + }; + + const checkDone = () => { + if (replyReceived && idleReceived) { + cleanup(); + } + }; + + executionSignal.catch(async () => { + cancelled = true; + timedOut = executionSignal.timedOut(); + await this.interrupt(); + cleanup(); + }); + + this.#messageHandlers.set(msgId, async (response) => { + switch (response.header.msg_type) { + case "execute_reply": { + replyReceived = true; + const replyStatus = response.content.status; + status = replyStatus === "error" ? "error" : "ok"; + if (typeof response.content.execution_count === "number") { + executionCount = response.content.execution_count; + } + checkDone(); + break; + } + case "stream": { + const text = String(response.content.text ?? ""); + if (text && options?.onChunk) { + await options.onChunk(text); + } + break; + } + case "execute_result": + case "display_data": { + const { text, outputs } = this.renderDisplay(response.content); + if (text && options?.onChunk) { + await options.onChunk(text); + } + if (outputs.length > 0 && options?.onDisplay) { + for (const output of outputs) { + await options.onDisplay(output); + } + } + break; + } + case "error": { + const traceback = Array.isArray(response.content.traceback) + ? response.content.traceback.map((line: unknown) => String(line)) + : []; + error = { + name: String(response.content.ename ?? "Error"), + value: String(response.content.evalue ?? ""), + traceback, + }; + const text = traceback.length > 0 ? `${traceback.join("\n")}\n` : `${error.name}: ${error.value}\n`; + if (options?.onChunk) { + await options.onChunk(text); + } + break; + } + case "status": { + const state = response.content.execution_state; + if (state === "idle") { + idleReceived = true; + checkDone(); + } + break; + } + case "input_request": { + stdinRequested = true; + if (options?.onChunk) { + await options.onChunk( + "[stdin] Kernel requested input. Interactive stdin is not supported; provide input programmatically.\n", + ); + } + this.sendMessage({ + channel: "stdin", + header: { + msg_id: nanoid(), + session: this.sessionId, + username: this.username, + date: new Date().toISOString(), + msg_type: "input_reply", + version: "5.5", + }, + parent_header: response.header as unknown as Record, + metadata: {}, + content: { value: "" }, + }); + break; + } + } + }); + + this.sendMessage(msg); + }); + } + + async introspectPrelude(): Promise { + let output = ""; + const result = await this.execute(PRELUDE_INTROSPECTION_SNIPPET, { + silent: false, + storeHistory: false, + onChunk: (text) => { + output += text; + }, + }); + if (result.cancelled || result.status === "error") { + throw new Error("Failed to introspect Python prelude"); + } + const trimmed = output.trim(); + if (!trimmed) return []; + try { + return JSON.parse(trimmed) as PreludeHelper[]; + } catch (err: unknown) { + const message = err instanceof Error ? err.message : String(err); + throw new Error(`Failed to parse Python prelude docs: ${message}`); + } + } + + async interrupt(): Promise { + try { + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), 2000); + await fetch(`${this.gatewayUrl}/api/kernels/${this.kernelId}/interrupt`, { + method: "POST", + headers: this.#authHeaders(), + signal: controller.signal, + }); + clearTimeout(timeout); + } catch (err: unknown) { + logger.warn("Failed to interrupt kernel via API", { error: err instanceof Error ? err.message : String(err) }); + } + + try { + const msg: JupyterMessage = { + channel: "control", + header: { + msg_id: nanoid(), + session: this.sessionId, + username: this.username, + date: new Date().toISOString(), + msg_type: "interrupt_request", + version: "5.5", + }, + parent_header: {}, + metadata: {}, + content: {}, + }; + this.sendMessage(msg); + } catch (err: unknown) { + logger.warn("Failed to send interrupt request", { error: err instanceof Error ? err.message : String(err) }); + } + } + + async shutdown(): Promise { + if (this.#disposed) return; + this.#disposed = true; + this.#alive = false; + + if (this.#heartbeatTimer) { + clearInterval(this.#heartbeatTimer); + this.#heartbeatTimer = undefined; + } + + try { + await fetch(`${this.gatewayUrl}/api/kernels/${this.kernelId}`, { + method: "DELETE", + headers: this.#authHeaders(), + }); + } catch (err: unknown) { + logger.warn("Failed to delete kernel via API", { error: err instanceof Error ? err.message : String(err) }); + } + + if (this.#ws) { + this.#ws.close(); + this.#ws = null; + } + + if (this.gatewayProcess) { + try { + killProcessTree(this.gatewayProcess.pid); + } catch (err: unknown) { + logger.warn("Failed to terminate gateway process", { + error: err instanceof Error ? err.message : String(err), + }); + } + } + } + + async ping(timeoutMs: number = HEARTBEAT_TIMEOUT_MS): Promise { + if (!this.isAlive()) return false; + try { + const controller = new AbortController(); + const timeout = setTimeout(() => controller.abort(), timeoutMs); + const response = await fetch(`${this.gatewayUrl}/api/kernels/${this.kernelId}`, { + signal: controller.signal, + headers: this.#authHeaders(), + }); + clearTimeout(timeout); + if (response.ok) { + this.#heartbeatFailures = 0; + return true; + } + throw new Error(`Kernel status check failed: ${response.status}`); + } catch (err: unknown) { + this.#heartbeatFailures += 1; + if (this.#heartbeatFailures > HEARTBEAT_FAILURE_LIMIT) { + this.#alive = false; + logger.warn("Kernel heartbeat failed", { error: err instanceof Error ? err.message : String(err) }); + } + return false; + } + } + + private startHeartbeat(): void { + if (this.#heartbeatTimer) return; + this.#heartbeatTimer = setInterval(() => { + void this.ping(); + }, HEARTBEAT_INTERVAL_MS); + } + + private renderDisplay(content: Record): { text: string; outputs: KernelDisplayOutput[] } { + const data = content.data as Record | undefined; + if (!data) return { text: "", outputs: [] }; + + const outputs: KernelDisplayOutput[] = []; + if (typeof data["image/png"] === "string") { + outputs.push({ type: "image", data: data["image/png"] as string, mimeType: "image/png" }); + } + if (data["application/json"] !== undefined) { + outputs.push({ type: "json", data: data["application/json"] }); + } + + if (typeof data["text/plain"] === "string") { + return { text: normalizeDisplayText(String(data["text/plain"])), outputs }; + } + if (data["text/html"] !== undefined) { + const markdown = htmlToBasicMarkdown(String(data["text/html"])) || ""; + return { text: markdown ? normalizeDisplayText(markdown) : "", outputs }; + } + return { text: "", outputs }; + } + + private sendMessage(msg: JupyterMessage): void { + if (!this.#ws || this.#ws.readyState !== WebSocket.OPEN) { + throw new Error("WebSocket not connected"); + } + + if (TRACE_IPC) { + logger.debug("Kernel IPC send", { + channel: msg.channel, + msgType: msg.header.msg_type, + msgId: msg.header.msg_id, + }); + } + + const data = serializeWebSocketMessage(msg); + this.#ws.send(data); + } +} diff --git a/packages/coding-agent/src/core/python-prelude.py b/packages/coding-agent/src/core/python-prelude.py new file mode 100644 index 000000000..4c9969143 --- /dev/null +++ b/packages/coding-agent/src/core/python-prelude.py @@ -0,0 +1,442 @@ +# OMP IPython prelude helpers +if "__omp_prelude_loaded__" not in globals(): + __omp_prelude_loaded__ = True + from pathlib import Path + import os, sys, re, json, shutil, subprocess, glob, textwrap, inspect + from datetime import datetime + + def pwd() -> Path: + """Print and return current working directory.""" + p = Path.cwd() + print(str(p)) + return p + + def cd(path: str | Path) -> Path: + """Change directory and print the new cwd.""" + p = Path(path).expanduser().resolve() + os.chdir(p) + print(str(p)) + return p + + def env(key: str | None = None, value: str | None = None): + """Get/set environment variables.""" + if key is None: + items = dict(sorted(os.environ.items())) + for k, v in items.items(): + print(f"{k}={v}") + print(f"[env] {len(items)} variables") + return items + if value is not None: + os.environ[key] = value + print(f"{key}={value}") + return value + val = os.environ.get(key) + print(f"{key}={val}") + return val + + def read(path: str | Path, *, limit: int | None = None) -> str: + """Read file contents. Prints a short preview + length.""" + p = Path(path) + data = p.read_text(encoding="utf-8") + if limit is not None: + preview = data[:limit] + print(preview) + print(f"[read {len(data)} chars from {p}]") + else: + print(data) + print(f"[read {len(data)} chars from {p}]") + return data + + def write(path: str | Path, content: str) -> Path: + """Write file contents (create parents). Prints bytes written.""" + p = Path(path) + p.parent.mkdir(parents=True, exist_ok=True) + p.write_text(content, encoding="utf-8") + print(f"[wrote {len(content)} chars to {p}]") + return p + + def append(path: str | Path, content: str) -> Path: + """Append to file. Prints bytes appended.""" + p = Path(path) + p.parent.mkdir(parents=True, exist_ok=True) + with p.open("a", encoding="utf-8") as f: + f.write(content) + print(f"[appended {len(content)} chars to {p}]") + return p + + def mkdir(path: str | Path) -> Path: + """Create directory (parents=True).""" + p = Path(path) + p.mkdir(parents=True, exist_ok=True) + print(f"[mkdir] {p}") + return p + + def rm(path: str | Path, *, recursive: bool = False) -> None: + """Delete file or directory (recursive optional).""" + p = Path(path) + if p.is_dir(): + if recursive: + shutil.rmtree(p) + print(f"[rm -r] {p}") + return + print(f"[rm] {p} (directory, use recursive=True)") + return + if p.exists(): + p.unlink() + print(f"[rm] {p}") + else: + print(f"[rm] {p} (missing)") + + def mv(src: str | Path, dst: str | Path) -> Path: + """Move or rename a file/directory.""" + src_p = Path(src) + dst_p = Path(dst) + dst_p.parent.mkdir(parents=True, exist_ok=True) + shutil.move(str(src_p), str(dst_p)) + print(f"[mv] {src_p} -> {dst_p}") + return dst_p + + def cp(src: str | Path, dst: str | Path) -> Path: + """Copy a file or directory.""" + src_p = Path(src) + dst_p = Path(dst) + dst_p.parent.mkdir(parents=True, exist_ok=True) + if src_p.is_dir(): + shutil.copytree(src_p, dst_p, dirs_exist_ok=True) + else: + shutil.copy2(src_p, dst_p) + print(f"[cp] {src_p} -> {dst_p}") + return dst_p + + def ls(path: str | Path = ".") -> list[Path]: + """List directory contents.""" + p = Path(path) + items = sorted(p.iterdir()) + for item in items: + suffix = "/" if item.is_dir() else "" + print(f"{item.name}{suffix}") + print(f"[ls] {len(items)} entries in {p}") + return items + + def find(pattern: str, path: str | Path = ".", *, files_only: bool = True) -> list[Path]: + """Recursive glob find. Defaults to files only.""" + p = Path(path) + matches = [] + for m in p.rglob(pattern): + if files_only and m.is_dir(): + continue + matches.append(m) + matches = sorted(matches) + for m in matches: + print(str(m)) + print(f"[find] {len(matches)} matches for '{pattern}' in {p}") + return matches + + def grep(pattern: str, path: str | Path, *, ignore_case: bool = False, context: int = 0) -> list[tuple[int, str]]: + """Grep a single file.""" + flags = re.IGNORECASE if ignore_case else 0 + rx = re.compile(pattern, flags) + p = Path(path) + lines = p.read_text(encoding="utf-8").splitlines() + hits: list[tuple[int, str]] = [] + for i, line in enumerate(lines, 1): + if rx.search(line): + hits.append((i, line)) + print(f"{i}: {line}") + if context: + start = max(0, i - 1 - context) + end = min(len(lines), i - 1 + context + 1) + for j in range(start, end): + if j + 1 == i: + continue + print(f"{j+1}- {lines[j]}") + print(f"[grep] {len(hits)} matches in {p}") + return hits + + def rgrep(pattern: str, path: str | Path = ".", *, glob_pattern: str = "*", ignore_case: bool = False) -> list[tuple[Path, int, str]]: + """Recursive grep across files matching glob_pattern.""" + flags = re.IGNORECASE if ignore_case else 0 + rx = re.compile(pattern, flags) + base = Path(path) + hits: list[tuple[Path, int, str]] = [] + for file_path in base.rglob(glob_pattern): + if file_path.is_dir(): + continue + try: + lines = file_path.read_text(encoding="utf-8").splitlines() + except Exception: + continue + for i, line in enumerate(lines, 1): + if rx.search(line): + hits.append((file_path, i, line)) + print(f"{file_path}:{i}: {line}") + print(f"[rgrep] {len(hits)} matches in {base}") + return hits + + def head(text: str, n: int = 10) -> str: + """Return the first n lines of text.""" + lines = text.splitlines()[:n] + out = "\n".join(lines) + print(out) + print(f"[head] {len(lines)} lines") + return out + + def tail(text: str, n: int = 10) -> str: + """Return the last n lines of text.""" + lines = text.splitlines()[-n:] + out = "\n".join(lines) + print(out) + print(f"[tail] {len(lines)} lines") + return out + + def replace(path: str | Path, pattern: str, repl: str, *, regex: bool = False) -> int: + """Replace text in a file (regex optional).""" + p = Path(path) + data = p.read_text(encoding="utf-8") + if regex: + new, count = re.subn(pattern, repl, data) + else: + new = data.replace(pattern, repl) + count = data.count(pattern) + p.write_text(new, encoding="utf-8") + print(f"[replace] {count} replacements in {p}") + return count + + def run(cmd: str, *, cwd: str | Path | None = None, timeout: int | None = None) -> subprocess.CompletedProcess[str]: + """Run a shell command and print stdout/stderr.""" + result = subprocess.run( + cmd, + cwd=str(cwd) if cwd else None, + shell=True, + capture_output=True, + text=True, + timeout=timeout, + ) + if result.stdout: + print(result.stdout, end="" if result.stdout.endswith("\n") else "\n") + if result.stderr: + print(result.stderr, end="" if result.stderr.endswith("\n") else "\n") + print(f"[run] exit={result.returncode}") + return result + + def bash(cmd: str, *, cwd: str | Path | None = None, timeout: int | None = None) -> subprocess.CompletedProcess[str]: + """Run a shell command via bash when available; fallback when missing.""" + snapshot = os.environ.get("OMP_SHELL_SNAPSHOT") + prefix = f"source '{snapshot}' 2>/dev/null && " if snapshot else "" + final = f"{prefix}{cmd}" + + bash_path = shutil.which("bash") + if bash_path: + return run(f"{bash_path} -lc {json.dumps(final)}", cwd=cwd, timeout=timeout) + + if sys.platform.startswith("win"): + return run(f"cmd /c {json.dumps(cmd)}", cwd=cwd, timeout=timeout) + + sh_path = shutil.which("sh") + if sh_path: + return run(f"{sh_path} -lc {json.dumps(cmd)}", cwd=cwd, timeout=timeout) + + raise RuntimeError("No suitable shell found for bash() bridge") + + # --- Extended shell-like utilities --- + + def cat(*paths: str | Path, separator: str = "\n") -> str: + """Concatenate multiple files and print. Like shell cat.""" + parts = [] + for p in paths: + parts.append(Path(p).read_text(encoding="utf-8")) + out = separator.join(parts) + print(out) + print(f"[cat] {len(paths)} files, {len(out)} chars") + return out + + def touch(path: str | Path) -> Path: + """Create empty file or update mtime.""" + p = Path(path) + p.parent.mkdir(parents=True, exist_ok=True) + p.touch() + print(f"[touch] {p}") + return p + + def wc(text: str) -> dict: + """Word/line/char count.""" + lines = text.splitlines() + words = text.split() + result = {"lines": len(lines), "words": len(words), "chars": len(text)} + print(f"{result['lines']} lines, {result['words']} words, {result['chars']} chars") + return result + + def sort_lines(text: str, *, reverse: bool = False, unique: bool = False) -> str: + """Sort lines of text.""" + lines = text.splitlines() + if unique: + lines = list(dict.fromkeys(lines)) + lines = sorted(lines, reverse=reverse) + out = "\n".join(lines) + print(out) + return out + + def uniq(text: str, *, count: bool = False) -> str | list[tuple[int, str]]: + """Remove duplicate adjacent lines (like uniq).""" + lines = text.splitlines() + if not lines: + return [] if count else "" + groups: list[tuple[int, str]] = [] + current = lines[0] + current_count = 1 + for line in lines[1:]: + if line == current: + current_count += 1 + continue + groups.append((current_count, current)) + current = line + current_count = 1 + groups.append((current_count, current)) + if count: + for c, l in groups: + print(f"{c:>4} {l}") + return groups + out = "\n".join(line for _, line in groups) + print(out) + return out + + def cols(text: str, *indices: int, sep: str | None = None) -> str: + """Extract columns from text (0-indexed). Like cut.""" + result_lines = [] + for line in text.splitlines(): + parts = line.split(sep) if sep else line.split() + selected = [parts[i] for i in indices if i < len(parts)] + result_lines.append(" ".join(selected)) + out = "\n".join(result_lines) + print(out) + return out + + def tree(path: str | Path = ".", *, max_depth: int = 3, show_hidden: bool = False) -> str: + """Print directory tree.""" + base = Path(path) + lines = [] + def walk(p: Path, prefix: str, depth: int): + if depth > max_depth: + return + items = sorted(p.iterdir(), key=lambda x: (not x.is_dir(), x.name.lower())) + items = [i for i in items if show_hidden or not i.name.startswith(".")] + for i, item in enumerate(items): + is_last = i == len(items) - 1 + connector = "└── " if is_last else "├── " + suffix = "/" if item.is_dir() else "" + lines.append(f"{prefix}{connector}{item.name}{suffix}") + if item.is_dir(): + ext = " " if is_last else "│ " + walk(item, prefix + ext, depth + 1) + lines.append(str(base) + "/") + walk(base, "", 1) + out = "\n".join(lines) + print(out) + return out + + def stat(path: str | Path) -> dict: + """Get file/directory info.""" + p = Path(path) + s = p.stat() + info = { + "path": str(p), + "size": s.st_size, + "is_file": p.is_file(), + "is_dir": p.is_dir(), + "mtime": datetime.fromtimestamp(s.st_mtime).isoformat(), + "mode": oct(s.st_mode), + } + for k, v in info.items(): + print(f"{k}: {v}") + return info + + def diff(a: str | Path, b: str | Path) -> str: + """Compare two files, print unified diff.""" + import difflib + path_a, path_b = Path(a), Path(b) + lines_a = path_a.read_text(encoding="utf-8").splitlines(keepends=True) + lines_b = path_b.read_text(encoding="utf-8").splitlines(keepends=True) + result = difflib.unified_diff(lines_a, lines_b, fromfile=str(path_a), tofile=str(path_b)) + out = "".join(result) + if out: + print(out) + else: + print("[diff] files are identical") + return out + + def glob_files(pattern: str, path: str | Path = ".") -> list[Path]: + """Non-recursive glob (use find() for recursive).""" + p = Path(path) + matches = sorted(p.glob(pattern)) + for m in matches: + print(str(m)) + print(f"[glob] {len(matches)} matches") + return matches + + def batch(paths: list[str | Path], fn) -> list: + """Apply function to multiple files. Returns list of results.""" + results = [] + for p in paths: + result = fn(Path(p)) + results.append(result) + print(f"[batch] processed {len(paths)} files") + return results + + def sed(path: str | Path, pattern: str, repl: str, *, flags: int = 0) -> int: + """Regex replace in file (like sed -i). Returns count.""" + p = Path(path) + data = p.read_text(encoding="utf-8") + new, count = re.subn(pattern, repl, data, flags=flags) + p.write_text(new, encoding="utf-8") + print(f"[sed] {count} replacements in {p}") + return count + + def rsed(pattern: str, repl: str, path: str | Path = ".", *, glob_pattern: str = "*", flags: int = 0) -> int: + """Recursive sed across files matching glob_pattern.""" + base = Path(path) + total = 0 + for file_path in base.rglob(glob_pattern): + if file_path.is_dir(): + continue + try: + data = file_path.read_text(encoding="utf-8") + new, count = re.subn(pattern, repl, data, flags=flags) + if count > 0: + file_path.write_text(new, encoding="utf-8") + print(f"{file_path}: {count} replacements") + total += count + except Exception: + continue + print(f"[rsed] {total} total replacements") + return total + + def __omp_prelude_docs__() -> list[dict[str, str]]: + """Return prelude helper docs for templating.""" + categories = [ + ("File I/O", ["read", "write", "append", "touch", "cat"]), + ("File operations", ["cp", "mv", "rm", "mkdir"]), + ("Navigation", ["pwd", "cd", "ls", "tree", "stat"]), + ("Search", ["find", "glob_files", "grep", "rgrep"]), + ("Text processing", ["head", "tail", "sort_lines", "uniq", "cols", "wc"]), + ("Find and replace", ["replace", "sed", "rsed"]), + ("Batch operations", ["batch", "diff"]), + ("Shell bridge", ["run", "bash", "env"]), + ] + helpers: list[dict[str, str]] = [] + for category, names in categories: + for name in names: + obj = globals().get(name) + if not callable(obj): + continue + signature = str(inspect.signature(obj)) + doc = inspect.getdoc(obj) or "" + docline = doc.splitlines()[0] if doc else "" + helpers.append( + { + "name": name, + "signature": signature, + "docstring": docline, + "category": category, + } + ) + return helpers diff --git a/packages/coding-agent/src/core/python-prelude.test.ts b/packages/coding-agent/src/core/python-prelude.test.ts new file mode 100644 index 000000000..ffa7002db --- /dev/null +++ b/packages/coding-agent/src/core/python-prelude.test.ts @@ -0,0 +1,77 @@ +import { describe, expect, it } from "bun:test"; +import { createPythonTool } from "./tools/python"; + +const pythonPath = Bun.which("python") ?? Bun.which("python3"); +const hasKernelDeps = (() => { + if (!pythonPath) return false; + const result = Bun.spawnSync( + [ + pythonPath, + "-c", + "import importlib.util,sys;sys.exit(0 if importlib.util.find_spec('kernel_gateway') and importlib.util.find_spec('ipykernel') else 1)", + ], + { stdin: "ignore", stdout: "pipe", stderr: "pipe" }, + ); + return result.exitCode === 0; +})(); + +const shouldRun = Boolean(pythonPath) && hasKernelDeps; + +describe.skipIf(!shouldRun)("PYTHON_PRELUDE integration", () => { + it("exposes prelude helpers via python tool", async () => { + const helpers = [ + "pwd", + "cd", + "env", + "read", + "write", + "append", + "mkdir", + "rm", + "mv", + "cp", + "ls", + "find", + "grep", + "rgrep", + "head", + "tail", + "replace", + "run", + "bash", + ]; + + const session = { + cwd: process.cwd(), + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => null, + settings: { + getImageAutoResize: () => true, + getLspFormatOnWrite: () => false, + getLspDiagnosticsOnWrite: () => false, + getLspDiagnosticsOnEdit: () => false, + getEditFuzzyMatch: () => true, + getGitToolEnabled: () => true, + getBashInterceptorEnabled: () => true, + getBashInterceptorSimpleLsEnabled: () => true, + getBashInterceptorRules: () => [], + getPythonToolMode: () => "ipy-only" as const, + getPythonKernelMode: () => "per-call" as const, + }, + }; + + const tool = createPythonTool(session); + const code = ` +helpers = ${JSON.stringify(helpers)} +missing = [name for name in helpers if name not in globals() or not callable(globals()[name])] +print("HELPERS_OK=" + ("1" if not missing else "0")) +if missing: + print("MISSING=" + ",".join(missing)) +`; + + const result = await tool.execute("tool-call-1", { code }); + const output = result.content.find((item) => item.type === "text")?.text ?? ""; + expect(output).toContain("HELPERS_OK=1"); + }); +}); diff --git a/packages/coding-agent/src/core/python-prelude.ts b/packages/coding-agent/src/core/python-prelude.ts new file mode 100644 index 000000000..8621ca60d --- /dev/null +++ b/packages/coding-agent/src/core/python-prelude.ts @@ -0,0 +1,3 @@ +import pythonPrelude from "./python-prelude.py" with { type: "text" }; + +export const PYTHON_PRELUDE = pythonPrelude; diff --git a/packages/coding-agent/src/core/sdk.ts b/packages/coding-agent/src/core/sdk.ts index 6ef825e7a..3fdba5e3f 100644 --- a/packages/coding-agent/src/core/sdk.ts +++ b/packages/coding-agent/src/core/sdk.ts @@ -66,6 +66,7 @@ import { convertToLlm } from "./messages"; import { ModelRegistry } from "./model-registry"; import { formatModelString, parseModelString } from "./model-resolver"; import { loadPromptTemplates as loadPromptTemplatesInternal, type PromptTemplate } from "./prompt-templates"; +import { disposeAllKernelSessions } from "./python-executor"; import { SessionManager } from "./session-manager"; import { type Settings, SettingsManager, type SkillsSettings } from "./settings-manager"; import { loadSkills as loadSkillsInternal, type Skill, type SkillWarning } from "./skills"; @@ -88,6 +89,7 @@ import { createGitTool, createGrepTool, createLsTool, + createPythonTool, createReadTool, createSshTool, createTools, @@ -217,6 +219,7 @@ export { // Individual tool factories (for custom usage) createReadTool, createBashTool, + createPythonTool, createSshTool, createEditTool, createWriteTool, @@ -441,6 +444,16 @@ function registerSshCleanup(): void { registerAsyncCleanup(() => cleanupSshResources()); } +let pythonCleanupRegistered = false; + +function registerPythonCleanup(): void { + if (pythonCleanupRegistered) return; + pythonCleanupRegistered = true; + registerAsyncCleanup(async () => { + await disposeAllKernelSessions(); + }); +} + function customToolToDefinition(tool: CustomTool): ToolDefinition { const definition: ToolDefinition & { [TOOL_DEFINITION_MARKER]: true } = { name: tool.name, @@ -541,6 +554,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} const eventBus = options.eventBus ?? createEventBus(); registerSshCleanup(); + registerPythonCleanup(); // Use provided or create AuthStorage and ModelRegistry const authStorage = options.authStorage ?? (await discoverAuthStorage(agentDir)); diff --git a/packages/coding-agent/src/core/session-manager.ts b/packages/coding-agent/src/core/session-manager.ts index e3a7145e9..8cf085aa7 100644 --- a/packages/coding-agent/src/core/session-manager.ts +++ b/packages/coding-agent/src/core/session-manager.ts @@ -468,6 +468,14 @@ export function loadEntriesFromFile(filePath: string, storage: SessionStorage = * Lightweight metadata for a session file, used in session picker UI. * Uses lazy getters to defer string formatting until actually displayed. */ +function sanitizeSessionName(value: string | undefined): string | undefined { + if (!value) return undefined; + const firstLine = value.split(/\r?\n/)[0] ?? ""; + const stripped = firstLine.replace(/[\x00-\x1F\x7F]/g, ""); + const trimmed = stripped.trim(); + return trimmed.length > 0 ? trimmed : undefined; +} + class RecentSessionInfo { readonly path: string; readonly mtime: number; @@ -476,13 +484,16 @@ class RecentSessionInfo { #name: string | undefined; #timeAgo: string | undefined; - constructor(path: string, mtime: number, header: Record) { + constructor(path: string, mtime: number, header: Record, firstPrompt?: string) { this.path = path; this.mtime = mtime; - // Extract title from session header, falling back to id if title is missing + // Extract title from session header, falling back to first user prompt, then id const trystr = (v: unknown) => (typeof v === "string" ? v : undefined); - this.#fullName = trystr(header.title) ?? trystr(header.id); + this.#fullName = + sanitizeSessionName(trystr(header.title)) ?? + sanitizeSessionName(firstPrompt) ?? + sanitizeSessionName(trystr(header.id)); } /** Full session name from header, or filename without extension as fallback */ @@ -508,26 +519,58 @@ class RecentSessionInfo { } } +/** + * Extracts the text content from a user message entry. + * Returns undefined if the entry is not a user message or has no text. + */ +function extractFirstUserPrompt(lines: string[]): string | undefined { + for (let i = 1; i < lines.length; i++) { + const line = lines[i]; + if (!line?.trim()) continue; + try { + const entry = JSON.parse(line) as Record; + if (entry.type !== "message") continue; + const message = entry.message as Record | undefined; + if (message?.role !== "user") continue; + const content = message.content; + if (typeof content === "string") return content; + if (Array.isArray(content)) { + for (const block of content) { + if (typeof block === "object" && block !== null && "text" in block) { + const text = (block as { text: unknown }).text; + if (typeof text === "string") return text; + } + } + } + } catch { + // Invalid JSON, skip to next line + } + } + return undefined; +} + /** * Reads all session files from the directory and returns them sorted by mtime (newest first). - * Uses low-level file I/O to efficiently read only the first 512 bytes of each file - * to extract the JSON header without loading entire session logs into memory. + * Uses low-level file I/O to efficiently read only the first 4KB of each file + * to extract the JSON header and first user message without loading entire session logs into memory. */ function getSortedSessions(sessionDir: string, storage: SessionStorage): RecentSessionInfo[] { try { - const buf = Buffer.alloc(512); + const buf = Buffer.alloc(4096); const files: string[] = storage.listFilesSync(sessionDir, "*.jsonl"); return files .map((path: string) => { try { const length = storage.readTextPrefixSync(path, buf); const content = buf.toString("utf-8", 0, length); - const firstLine = content.split("\n")[0]; + const lines = content.split("\n"); + const firstLine = lines[0]; if (!firstLine || !firstLine.trim()) return null; const header = JSON.parse(firstLine) as Record; if (header.type !== "session" || typeof header.id !== "string") return null; const mtime = storage.statSync(path).mtimeMs; - return new RecentSessionInfo(path, mtime, header); + const firstPrompt = header.title ? undefined : extractFirstUserPrompt(lines); + return new RecentSessionInfo(path, mtime, header, firstPrompt); } catch { return null; } diff --git a/packages/coding-agent/src/core/settings-manager-python.test.ts b/packages/coding-agent/src/core/settings-manager-python.test.ts new file mode 100644 index 000000000..42a9fc022 --- /dev/null +++ b/packages/coding-agent/src/core/settings-manager-python.test.ts @@ -0,0 +1,23 @@ +import { describe, expect, it } from "bun:test"; +import { SettingsManager } from "./settings-manager"; + +describe("SettingsManager python settings", () => { + it("defaults to ipy-only and session", () => { + const settings = SettingsManager.inMemory(); + + expect(settings.getPythonToolMode()).toBe("ipy-only"); + expect(settings.getPythonKernelMode()).toBe("session"); + }); + + it("persists python tool and kernel modes", async () => { + const settings = SettingsManager.inMemory(); + + await settings.setPythonToolMode("bash-only"); + await settings.setPythonKernelMode("per-call"); + + expect(settings.getPythonToolMode()).toBe("bash-only"); + expect(settings.getPythonKernelMode()).toBe("per-call"); + expect(settings.serialize().python?.toolMode).toBe("bash-only"); + expect(settings.serialize().python?.kernelMode).toBe("per-call"); + }); +}); diff --git a/packages/coding-agent/src/core/settings-manager.ts b/packages/coding-agent/src/core/settings-manager.ts index 35db70410..94ec26e7a 100644 --- a/packages/coding-agent/src/core/settings-manager.ts +++ b/packages/coding-agent/src/core/settings-manager.ts @@ -112,6 +112,14 @@ export interface LspSettings { diagnosticsOnEdit?: boolean; // default: false (return LSP diagnostics after edit tool edits code files) } +export type PythonToolMode = "ipy-only" | "bash-only" | "both"; +export type PythonKernelMode = "session" | "per-call"; + +export interface PythonSettings { + toolMode?: PythonToolMode; + kernelMode?: PythonKernelMode; +} + export interface EditSettings { fuzzyMatch?: boolean; // default: true (accept high-confidence fuzzy matches for whitespace/indentation) } @@ -215,6 +223,7 @@ export interface Settings { git?: GitSettings; mcp?: MCPSettings; lsp?: LspSettings; + python?: PythonSettings; edit?: EditSettings; ttsr?: TtsrSettings; todoCompletion?: TodoCompletionSettings; @@ -307,6 +316,7 @@ const DEFAULT_SETTINGS: Settings = { git: { enabled: false }, mcp: { enableProjectConfig: true }, lsp: { formatOnWrite: false, diagnosticsOnWrite: true, diagnosticsOnEdit: false }, + python: { toolMode: "ipy-only", kernelMode: "session" }, edit: { fuzzyMatch: true }, ttsr: { enabled: true, contextMode: "discard", repeatMode: "once", repeatGap: 10 }, voice: { @@ -383,6 +393,22 @@ function normalizeSettings(settings: Settings): Settings { ...merged, symbolPreset, bashInterceptor: normalizeBashInterceptorSettings(merged.bashInterceptor), + python: normalizePythonSettings(merged.python), + }; +} + +function normalizePythonSettings(settings: PythonSettings | undefined): PythonSettings { + const toolMode = settings?.toolMode; + const kernelMode = settings?.kernelMode; + return { + toolMode: + toolMode === "ipy-only" || toolMode === "bash-only" || toolMode === "both" + ? toolMode + : (DEFAULT_SETTINGS.python?.toolMode ?? "ipy-only"), + kernelMode: + kernelMode === "session" || kernelMode === "per-call" + ? kernelMode + : (DEFAULT_SETTINGS.python?.kernelMode ?? "session"), }; } @@ -787,6 +813,30 @@ export class SettingsManager { }; } + getRetryMaxRetries(): number { + return this.settings.retry?.maxRetries ?? 3; + } + + async setRetryMaxRetries(maxRetries: number): Promise { + if (!this.globalSettings.retry) { + this.globalSettings.retry = {}; + } + this.globalSettings.retry.maxRetries = maxRetries; + await this.save(); + } + + getRetryBaseDelayMs(): number { + return this.settings.retry?.baseDelayMs ?? 2000; + } + + async setRetryBaseDelayMs(baseDelayMs: number): Promise { + if (!this.globalSettings.retry) { + this.globalSettings.retry = {}; + } + this.globalSettings.retry.baseDelayMs = baseDelayMs; + await this.save(); + } + getTodoCompletionSettings(): { enabled: boolean; maxReminders: number } { return { enabled: this.settings.todoCompletion?.enabled ?? false, @@ -901,6 +951,14 @@ export class SettingsManager { return this.settings.skills?.enableSkillCommands ?? true; } + async setEnableSkillCommands(enabled: boolean): Promise { + if (!this.globalSettings.skills) { + this.globalSettings.skills = {}; + } + this.globalSettings.skills.enableSkillCommands = enabled; + await this.save(); + } + getCommandsSettings(): Required { return { enableClaudeUser: this.settings.commands?.enableClaudeUser ?? true, @@ -908,6 +966,30 @@ export class SettingsManager { }; } + getCommandsEnableClaudeUser(): boolean { + return this.settings.commands?.enableClaudeUser ?? true; + } + + async setCommandsEnableClaudeUser(enabled: boolean): Promise { + if (!this.globalSettings.commands) { + this.globalSettings.commands = {}; + } + this.globalSettings.commands.enableClaudeUser = enabled; + await this.save(); + } + + getCommandsEnableClaudeProject(): boolean { + return this.settings.commands?.enableClaudeProject ?? true; + } + + async setCommandsEnableClaudeProject(enabled: boolean): Promise { + if (!this.globalSettings.commands) { + this.globalSettings.commands = {}; + } + this.globalSettings.commands.enableClaudeProject = enabled; + await this.save(); + } + getShowImages(): boolean { return this.settings.terminal?.showImages ?? true; } @@ -1064,10 +1146,42 @@ export class SettingsManager { await this.save(); } + async setBashInterceptorSimpleLsEnabled(enabled: boolean): Promise { + if (!this.globalSettings.bashInterceptor) { + this.globalSettings.bashInterceptor = {}; + } + this.globalSettings.bashInterceptor.simpleLs = enabled; + await this.save(); + } + getGitToolEnabled(): boolean { return this.settings.git?.enabled ?? false; } + getPythonToolMode(): PythonToolMode { + return this.settings.python?.toolMode ?? "ipy-only"; + } + + async setPythonToolMode(mode: PythonToolMode): Promise { + if (!this.globalSettings.python) { + this.globalSettings.python = {}; + } + this.globalSettings.python.toolMode = mode; + await this.save(); + } + + getPythonKernelMode(): PythonKernelMode { + return this.settings.python?.kernelMode ?? "session"; + } + + async setPythonKernelMode(mode: PythonKernelMode): Promise { + if (!this.globalSettings.python) { + this.globalSettings.python = {}; + } + this.globalSettings.python.kernelMode = mode; + await this.save(); + } + async setGitToolEnabled(enabled: boolean): Promise { if (!this.globalSettings.git) { this.globalSettings.git = {}; @@ -1262,6 +1376,42 @@ export class SettingsManager { await this.save(); } + getVoiceTtsModel(): string { + return this.settings.voice?.ttsModel ?? "gpt-4o-mini-tts"; + } + + async setVoiceTtsModel(model: string): Promise { + if (!this.globalSettings.voice) { + this.globalSettings.voice = {}; + } + this.globalSettings.voice.ttsModel = model; + await this.save(); + } + + getVoiceTtsVoice(): string { + return this.settings.voice?.ttsVoice ?? "alloy"; + } + + async setVoiceTtsVoice(voice: string): Promise { + if (!this.globalSettings.voice) { + this.globalSettings.voice = {}; + } + this.globalSettings.voice.ttsVoice = voice; + await this.save(); + } + + getVoiceTtsFormat(): "wav" | "mp3" | "opus" | "aac" | "flac" { + return this.settings.voice?.ttsFormat ?? "wav"; + } + + async setVoiceTtsFormat(format: "wav" | "mp3" | "opus" | "aac" | "flac"): Promise { + if (!this.globalSettings.voice) { + this.globalSettings.voice = {}; + } + this.globalSettings.voice.ttsFormat = format; + await this.save(); + } + // ═══════════════════════════════════════════════════════════════════════════ // Status Line Settings // ═══════════════════════════════════════════════════════════════════════════ @@ -1398,7 +1548,7 @@ export class SettingsManager { return this.settings.showHardwareCursor; } // Check env var override - const envVar = process.env.PI_HARDWARE_CURSOR?.toLowerCase(); + const envVar = process.env.OMP_HARDWARE_CURSOR?.toLowerCase(); if (envVar === "0" || envVar === "false") return false; if (envVar === "1" || envVar === "true") return true; // Default to true on Linux/macOS for IME support diff --git a/packages/coding-agent/src/core/streaming-output.test.ts b/packages/coding-agent/src/core/streaming-output.test.ts new file mode 100644 index 000000000..caa3ea3d7 --- /dev/null +++ b/packages/coding-agent/src/core/streaming-output.test.ts @@ -0,0 +1,26 @@ +import { describe, expect, it } from "bun:test"; +import { createOutputSink } from "./streaming-output"; + +function makeLargeOutput(size: number): string { + return "x".repeat(size); +} + +describe("createOutputSink", () => { + it("spills to disk and truncates large output", async () => { + const largeOutput = makeLargeOutput(60_000); + const sink = createOutputSink(10, 70_000); + const writer = sink.getWriter(); + + await writer.write(largeOutput); + await writer.close(); + + const result = sink.dump(); + + expect(result.truncated).toBe(true); + expect(result.fullOutputPath).toBeDefined(); + expect(result.output.length).toBeLessThan(largeOutput.length); + + const fullOutput = await Bun.file(result.fullOutputPath!).text(); + expect(fullOutput).toBe(largeOutput); + }); +}); diff --git a/packages/coding-agent/src/core/streaming-output.ts b/packages/coding-agent/src/core/streaming-output.ts new file mode 100644 index 000000000..c0923c90f --- /dev/null +++ b/packages/coding-agent/src/core/streaming-output.ts @@ -0,0 +1,86 @@ +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { nanoid } from "nanoid"; +import stripAnsi from "strip-ansi"; +import { sanitizeBinaryOutput } from "../utils/shell"; +import { truncateTail } from "./tools/truncate"; + +interface OutputFileSink { + write(data: string): number | Promise; + end(): void; +} + +export function createSanitizer(): TransformStream { + const decoder = new TextDecoder(); + return new TransformStream({ + transform(chunk, controller) { + const text = sanitizeBinaryOutput(stripAnsi(decoder.decode(chunk, { stream: true }))).replace(/\r/g, ""); + controller.enqueue(text); + }, + }); +} + +export async function pumpStream(readable: ReadableStream, writer: WritableStreamDefaultWriter) { + const reader = readable.pipeThrough(createSanitizer()).getReader(); + try { + while (true) { + const { done, value } = await reader.read(); + if (done) break; + await writer.write(value); + } + } finally { + reader.releaseLock(); + } +} + +export function createOutputSink( + spillThreshold: number, + maxBuffer: number, + onChunk?: (text: string) => void, +): WritableStream & { + dump: (annotation?: string) => { output: string; truncated: boolean; fullOutputPath?: string }; +} { + const chunks: string[] = []; + let chunkBytes = 0; + let totalBytes = 0; + let fullOutputPath: string | undefined; + let fullOutputStream: OutputFileSink | undefined; + + const sink = new WritableStream({ + write(text) { + totalBytes += text.length; + + if (totalBytes > spillThreshold && !fullOutputPath) { + fullOutputPath = join(tmpdir(), `omp-${nanoid()}.buffer`); + const stream = Bun.file(fullOutputPath).writer(); + chunks.forEach((chunk) => { + stream.write(chunk); + }); + fullOutputStream = stream; + } + fullOutputStream?.write(text); + + chunks.push(text); + chunkBytes += text.length; + while (chunkBytes > maxBuffer && chunks.length > 1) { + chunkBytes -= chunks.shift()!.length; + } + + onChunk?.(text); + }, + close() { + fullOutputStream?.end(); + }, + }); + + return Object.assign(sink, { + dump(annotation?: string) { + if (annotation) { + chunks.push(`\n\n${annotation}`); + } + const full = chunks.join(""); + const { content, truncated } = truncateTail(full); + return { output: truncated ? content : full, truncated, fullOutputPath }; + }, + }); +} diff --git a/packages/coding-agent/src/core/system-prompt.python.test.ts b/packages/coding-agent/src/core/system-prompt.python.test.ts new file mode 100644 index 000000000..00bd26e69 --- /dev/null +++ b/packages/coding-agent/src/core/system-prompt.python.test.ts @@ -0,0 +1,17 @@ +import { describe, expect, it } from "bun:test"; +import { buildSystemPrompt } from "./system-prompt"; + +describe("buildSystemPrompt", () => { + it("includes python tool details when enabled", async () => { + const prompt = await buildSystemPrompt({ + cwd: "/tmp", + toolNames: ["python"], + contextFiles: [], + skills: [], + rules: [], + }); + + expect(prompt).toContain("python: Execute Python code via a session-backed IPython kernel"); + expect(prompt).toContain("What python IS for"); + }); +}); diff --git a/packages/coding-agent/src/core/system-prompt.ts b/packages/coding-agent/src/core/system-prompt.ts index 032ff7548..cd504f2b7 100644 --- a/packages/coding-agent/src/core/system-prompt.ts +++ b/packages/coding-agent/src/core/system-prompt.ts @@ -76,6 +76,7 @@ const toolDescriptions: Record = { ask: "Ask user for input or clarification", read: "Read file contents", bash: "Execute bash commands (npm, docker, etc.)", + python: "Execute Python code via a session-backed IPython kernel", calc: "{ calculations: array of { expression: string, prefix: string, suffix: string } } Basic calculations.", ssh: "Execute commands on remote hosts via SSH", edit: "Make surgical edits to files (find exact text and replace)", @@ -672,7 +673,8 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): // Build tool descriptions array // Priority: toolNames (explicit list) > tools (Map) > defaults - const defaultToolNames: ToolName[] = ["read", "bash", "edit", "write"]; + // Default includes both bash and python; actual availability determined by settings in createTools + const defaultToolNames: ToolName[] = ["read", "bash", "python", "edit", "write"]; let toolNamesArray: string[]; if (toolNames !== undefined) { // Explicit toolNames list provided (could be empty) diff --git a/packages/coding-agent/src/core/timings.ts b/packages/coding-agent/src/core/timings.ts index 8adb696ae..c1df9eda1 100644 --- a/packages/coding-agent/src/core/timings.ts +++ b/packages/coding-agent/src/core/timings.ts @@ -3,7 +3,7 @@ * Enable with OMP_TIMING=1 or PI_TIMING=1 environment variable. */ -const ENABLED = process.env.OMP_TIMING === "1" || process.env.PI_TIMING === "1"; +const ENABLED = process.env.OMP_TIMING === "1"; const timings: Array<{ label: string; ms: number }> = []; let lastTime = Date.now(); diff --git a/packages/coding-agent/src/core/tools/index.test.ts b/packages/coding-agent/src/core/tools/index.test.ts index df7a18e62..7808d387b 100644 --- a/packages/coding-agent/src/core/tools/index.test.ts +++ b/packages/coding-agent/src/core/tools/index.test.ts @@ -1,6 +1,8 @@ import { describe, expect, it } from "bun:test"; import { BUILTIN_TOOLS, createTools, HIDDEN_TOOLS, type ToolSession } from "./index"; +process.env.OMP_PYTHON_SKIP_CHECK = "1"; + function createTestSession(overrides: Partial = {}): ToolSession { return { cwd: "/tmp/test", @@ -11,6 +13,21 @@ function createTestSession(overrides: Partial = {}): ToolSession { }; } +function createBaseSettings(overrides: Partial> = {}) { + return { + getImageAutoResize: () => true, + getLspFormatOnWrite: () => true, + getLspDiagnosticsOnWrite: () => true, + getLspDiagnosticsOnEdit: () => false, + getEditFuzzyMatch: () => true, + getGitToolEnabled: () => true, + getBashInterceptorEnabled: () => true, + getBashInterceptorSimpleLsEnabled: () => true, + getBashInterceptorRules: () => [], + ...overrides, + }; +} + describe("createTools", () => { it("creates all builtin tools by default", async () => { const session = createTestSession(); @@ -18,7 +35,8 @@ describe("createTools", () => { const names = tools.map((t) => t.name); // Core tools should always be present - expect(names).toContain("bash"); + expect(names).toContain("python"); + expect(names).not.toContain("bash"); expect(names).toContain("calc"); expect(names).toContain("read"); expect(names).toContain("edit"); @@ -35,6 +53,34 @@ describe("createTools", () => { expect(names).toContain("web_search"); }); + it("includes bash and python when python mode is both", async () => { + const session = createTestSession({ + settings: createBaseSettings({ + getPythonToolMode: () => "both", + getPythonKernelMode: () => "session", + }), + }); + const tools = await createTools(session); + const names = tools.map((t) => t.name); + + expect(names).toContain("bash"); + expect(names).toContain("python"); + }); + + it("includes bash only when python mode is bash-only", async () => { + const session = createTestSession({ + settings: createBaseSettings({ + getPythonToolMode: () => "bash-only", + getPythonKernelMode: () => "session", + }), + }); + const tools = await createTools(session); + const names = tools.map((t) => t.name); + + expect(names).toContain("bash"); + expect(names).not.toContain("python"); + }); + it("excludes lsp tool when session disables LSP", async () => { const session = createTestSession({ enableLsp: false }); const tools = await createTools(session, ["read", "lsp", "write"]); @@ -93,17 +139,7 @@ describe("createTools", () => { it("excludes git tool when disabled in settings", async () => { const session = createTestSession({ - settings: { - getImageAutoResize: () => true, - getLspFormatOnWrite: () => true, - getLspDiagnosticsOnWrite: () => true, - getLspDiagnosticsOnEdit: () => false, - getEditFuzzyMatch: () => true, - getGitToolEnabled: () => false, - getBashInterceptorEnabled: () => true, - getBashInterceptorSimpleLsEnabled: () => true, - getBashInterceptorRules: () => [], - }, + settings: createBaseSettings({ getGitToolEnabled: () => false }), }); const tools = await createTools(session); const names = tools.map((t) => t.name); @@ -113,17 +149,7 @@ describe("createTools", () => { it("includes git tool when enabled in settings", async () => { const session = createTestSession({ - settings: { - getImageAutoResize: () => true, - getLspFormatOnWrite: () => true, - getLspDiagnosticsOnWrite: () => true, - getLspDiagnosticsOnEdit: () => false, - getEditFuzzyMatch: () => true, - getGitToolEnabled: () => true, - getBashInterceptorEnabled: () => true, - getBashInterceptorSimpleLsEnabled: () => true, - getBashInterceptorRules: () => [], - }, + settings: createBaseSettings({ getGitToolEnabled: () => true }), }); const tools = await createTools(session); const names = tools.map((t) => t.name); @@ -153,6 +179,7 @@ describe("createTools", () => { const expectedTools = [ "ask", "bash", + "python", "calc", "ssh", "edit", diff --git a/packages/coding-agent/src/core/tools/index.ts b/packages/coding-agent/src/core/tools/index.ts index 877a47cb5..b00208013 100644 --- a/packages/coding-agent/src/core/tools/index.ts +++ b/packages/coding-agent/src/core/tools/index.ts @@ -24,6 +24,7 @@ export { } from "./lsp/index"; export { createNotebookTool, type NotebookToolDetails } from "./notebook"; export { createOutputTool, type OutputToolDetails } from "./output"; +export { createPythonTool, type PythonToolDetails } from "./python"; export { createReadTool, type ReadToolDetails } from "./read"; export { reportFindingTool, type SubmitReviewDetails } from "./review"; export { createSshTool, type SSHToolDetails } from "./ssh"; @@ -63,6 +64,8 @@ export { createWriteTool, type WriteToolDetails } from "./write"; import type { AgentTool } from "@oh-my-pi/pi-agent-core"; import type { EventBus } from "../event-bus"; +import { logger } from "../logger"; +import { warmPythonEnvironment } from "../python-executor"; import type { BashInterceptorRule } from "../settings-manager"; import { createAskTool } from "./ask"; import { createBashTool } from "./bash"; @@ -76,6 +79,7 @@ import { createLsTool } from "./ls"; import { createLspTool } from "./lsp/index"; import { createNotebookTool } from "./notebook"; import { createOutputTool } from "./output"; +import { createPythonTool } from "./python"; import { createReadTool } from "./read"; import { reportFindingTool } from "./review"; import { createSshTool } from "./ssh"; @@ -129,6 +133,8 @@ export interface ToolSession { getBashInterceptorEnabled(): boolean; getBashInterceptorSimpleLsEnabled(): boolean; getBashInterceptorRules(): BashInterceptorRule[]; + getPythonToolMode?(): "ipy-only" | "bash-only" | "both"; + getPythonKernelMode?(): "session" | "per-call"; }; } @@ -137,6 +143,7 @@ type ToolFactory = (session: ToolSession) => Tool | null | Promise; export const BUILTIN_TOOLS: Record = { ask: createAskTool, bash: createBashTool, + python: createPythonTool, calc: createCalculatorTool, ssh: createSshTool, edit: createEditTool, @@ -162,6 +169,36 @@ export const HIDDEN_TOOLS: Record = { export type ToolName = keyof typeof BUILTIN_TOOLS; +export type PythonToolMode = "ipy-only" | "bash-only" | "both"; + +/** + * Parse OMP_PY environment variable to determine Python tool mode. + * Returns null if not set or invalid. + * + * Values: + * - "0" or "bash" → bash-only + * - "1" or "py" → ipy-only + * - "mix" or "both" → both + */ +function getPythonModeFromEnv(): PythonToolMode | null { + const value = process.env.OMP_PY?.toLowerCase(); + if (!value) return null; + + switch (value) { + case "0": + case "bash": + return "bash-only"; + case "1": + case "py": + return "ipy-only"; + case "mix": + case "both": + return "both"; + default: + return null; + } +} + /** * Create tools from BUILTIN_TOOLS registry. */ @@ -169,8 +206,39 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P const includeComplete = session.requireCompleteTool === true; const enableLsp = session.enableLsp ?? true; const requestedTools = toolNames && toolNames.length > 0 ? [...new Set(toolNames)] : undefined; + const pythonMode = getPythonModeFromEnv() ?? session.settings?.getPythonToolMode?.() ?? "ipy-only"; + let pythonAvailable = true; + const shouldCheckPython = + pythonMode !== "bash-only" && + (requestedTools === undefined || requestedTools.includes("python") || pythonMode === "ipy-only"); + if (shouldCheckPython) { + const warmup = await warmPythonEnvironment(session.cwd, session.getSessionFile?.() ?? `cwd:${session.cwd}`); + pythonAvailable = warmup.ok; + if (!warmup.ok) { + logger.warn("Python kernel unavailable, falling back to bash", { + reason: warmup.reason, + }); + } + } + const effectiveMode = pythonAvailable ? pythonMode : "bash-only"; + const allowBash = effectiveMode !== "ipy-only"; + const allowPython = effectiveMode !== "bash-only"; + if ( + requestedTools && + allowBash && + !allowPython && + requestedTools.includes("python") && + !requestedTools.includes("bash") + ) { + requestedTools.push("bash"); + } const allTools: Record = { ...BUILTIN_TOOLS, ...HIDDEN_TOOLS }; - const isToolAllowed = (name: string) => (name === "lsp" ? enableLsp : true); + const isToolAllowed = (name: string) => { + if (name === "lsp") return enableLsp; + if (name === "bash") return allowBash; + if (name === "python") return allowPython; + return true; + }; if (includeComplete && requestedTools && !requestedTools.includes("complete")) { requestedTools.push("complete"); } diff --git a/packages/coding-agent/src/core/tools/python-execution.test.ts b/packages/coding-agent/src/core/tools/python-execution.test.ts new file mode 100644 index 000000000..aab0255b5 --- /dev/null +++ b/packages/coding-agent/src/core/tools/python-execution.test.ts @@ -0,0 +1,68 @@ +import { describe, expect, it, vi } from "bun:test"; +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import * as pythonExecutor from "../python-executor"; +import type { ToolSession } from "./index"; +import { createPythonTool } from "./python"; + +function createSession(cwd: string): ToolSession { + return { + cwd, + hasUI: false, + getSessionFile: () => "session-file", + getSessionSpawns: () => "*", + settings: { + getImageAutoResize: () => true, + getLspFormatOnWrite: () => true, + getLspDiagnosticsOnWrite: () => true, + getLspDiagnosticsOnEdit: () => false, + getEditFuzzyMatch: () => true, + getGitToolEnabled: () => true, + getBashInterceptorEnabled: () => true, + getBashInterceptorSimpleLsEnabled: () => true, + getBashInterceptorRules: () => [], + getPythonToolMode: () => "ipy-only", + getPythonKernelMode: () => "per-call", + }, + }; +} + +describe("python tool execution", () => { + it("passes kernel options from settings and args", async () => { + const tempDir = mkdtempSync(join(tmpdir(), "python-tool-")); + const executeSpy = vi.spyOn(pythonExecutor, "executePython").mockResolvedValue({ + output: "ok", + exitCode: 0, + cancelled: false, + truncated: false, + displayOutputs: [], + stdinRequested: false, + }); + + const tool = createPythonTool(createSession(tempDir)); + const result = await tool.execute( + "call-id", + { code: "print('hi')", timeout: 5, workdir: tempDir, reset: true }, + undefined, + undefined, + undefined, + ); + + expect(executeSpy).toHaveBeenCalledWith( + "print('hi')", + expect.objectContaining({ + cwd: tempDir, + timeout: 5000, + sessionId: "session-file", + kernelMode: "per-call", + reset: true, + }), + ); + const text = result.content.find((item) => item.type === "text")?.text; + expect(text).toBe("ok"); + + executeSpy.mockRestore(); + rmSync(tempDir, { recursive: true, force: true }); + }); +}); diff --git a/packages/coding-agent/src/core/tools/python-fallback.test.ts b/packages/coding-agent/src/core/tools/python-fallback.test.ts new file mode 100644 index 000000000..61693f7f3 --- /dev/null +++ b/packages/coding-agent/src/core/tools/python-fallback.test.ts @@ -0,0 +1,72 @@ +import { describe, expect, it, vi } from "bun:test"; +import * as pythonKernelModule from "../python-kernel"; +import type { ToolSession } from "./index"; +import { createTools } from "./index"; + +function createTestSession(overrides: Partial = {}): ToolSession { + return { + cwd: "/tmp/test", + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + ...overrides, + }; +} + +function createBaseSettings(overrides: Partial> = {}) { + return { + getImageAutoResize: () => true, + getLspFormatOnWrite: () => true, + getLspDiagnosticsOnWrite: () => true, + getLspDiagnosticsOnEdit: () => false, + getEditFuzzyMatch: () => true, + getGitToolEnabled: () => true, + getBashInterceptorEnabled: () => true, + getBashInterceptorSimpleLsEnabled: () => true, + getBashInterceptorRules: () => [], + ...overrides, + }; +} + +describe("createTools python fallback", () => { + it("falls back to bash-only when kernel unavailable", async () => { + const availabilitySpy = vi + .spyOn(pythonKernelModule, "checkPythonKernelAvailability") + .mockResolvedValue({ ok: false, reason: "unavailable" }); + + const session = createTestSession({ + settings: createBaseSettings({ + getPythonToolMode: () => "ipy-only", + getPythonKernelMode: () => "session", + }), + }); + + const tools = await createTools(session, ["python"]); + const names = tools.map((tool) => tool.name).sort(); + + expect(names).toEqual(["bash"]); + + availabilitySpy.mockRestore(); + }); + + it("keeps bash when python mode is both but unavailable", async () => { + const availabilitySpy = vi + .spyOn(pythonKernelModule, "checkPythonKernelAvailability") + .mockResolvedValue({ ok: false, reason: "unavailable" }); + + const session = createTestSession({ + settings: createBaseSettings({ + getPythonToolMode: () => "both", + getPythonKernelMode: () => "session", + }), + }); + + const tools = await createTools(session); + const names = tools.map((tool) => tool.name); + + expect(names).toContain("bash"); + expect(names).not.toContain("python"); + + availabilitySpy.mockRestore(); + }); +}); diff --git a/packages/coding-agent/src/core/tools/python-renderer.test.ts b/packages/coding-agent/src/core/tools/python-renderer.test.ts new file mode 100644 index 000000000..3aaf0a293 --- /dev/null +++ b/packages/coding-agent/src/core/tools/python-renderer.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it } from "bun:test"; +import stripAnsi from "strip-ansi"; +import { getThemeByName } from "../../modes/interactive/theme/theme"; +import { pythonToolRenderer } from "./python"; +import { truncateTail } from "./truncate"; + +describe("pythonToolRenderer", () => { + it("renders truncated output when collapsed and full output when expanded", () => { + const theme = getThemeByName("dark"); + expect(theme).toBeDefined(); + const uiTheme = theme!; + + const fullOutput = ["line 1", "line 2", "line 3", "line 4"].join("\n"); + const truncation = truncateTail(fullOutput, { maxLines: 2, maxBytes: 128 }); + + const result = { + content: [{ type: "text", text: truncation.content }], + details: { + truncation, + fullOutput, + }, + }; + + const collapsed = pythonToolRenderer.renderResult(result, { expanded: false, isPartial: false }, uiTheme); + const collapsedLines = stripAnsi(collapsed.render(80).join("\n")); + expect(collapsedLines).toContain("line 4"); + expect(collapsedLines).not.toContain("line 1"); + expect(collapsedLines).toContain("Truncated:"); + + const expanded = pythonToolRenderer.renderResult(result, { expanded: true, isPartial: false }, uiTheme); + const expandedLines = stripAnsi(expanded.render(80).join("\n")); + expect(expandedLines).toContain("line 1"); + expect(expandedLines).toContain("line 4"); + expect(expandedLines).not.toContain("Truncated:"); + }); +}); diff --git a/packages/coding-agent/src/core/tools/python-tool-mode.test.ts b/packages/coding-agent/src/core/tools/python-tool-mode.test.ts new file mode 100644 index 000000000..2fc70e8b3 --- /dev/null +++ b/packages/coding-agent/src/core/tools/python-tool-mode.test.ts @@ -0,0 +1,43 @@ +import { describe, expect, it } from "bun:test"; +import { createTools, type ToolSession } from "./index"; + +function createSession(overrides: Partial = {}): ToolSession { + return { + cwd: "/tmp/test", + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: { + getImageAutoResize: () => true, + getLspFormatOnWrite: () => true, + getLspDiagnosticsOnWrite: () => true, + getLspDiagnosticsOnEdit: () => false, + getEditFuzzyMatch: () => true, + getGitToolEnabled: () => true, + getBashInterceptorEnabled: () => true, + getBashInterceptorSimpleLsEnabled: () => true, + getBashInterceptorRules: () => [], + getPythonToolMode: () => "bash-only", + getPythonKernelMode: () => "session", + }, + ...overrides, + }; +} + +describe("createTools python fallback", () => { + it("falls back to bash when python is requested but disabled", async () => { + const previous = process.env.OMP_PYTHON_SKIP_CHECK; + process.env.OMP_PYTHON_SKIP_CHECK = "1"; + const session = createSession(); + const tools = await createTools(session, ["python"]); + const names = tools.map((tool) => tool.name); + + expect(names).toEqual(["bash"]); + + if (previous === undefined) { + delete process.env.OMP_PYTHON_SKIP_CHECK; + } else { + process.env.OMP_PYTHON_SKIP_CHECK = previous; + } + }); +}); diff --git a/packages/coding-agent/src/core/tools/python.test.ts b/packages/coding-agent/src/core/tools/python.test.ts new file mode 100644 index 000000000..11f9e3713 --- /dev/null +++ b/packages/coding-agent/src/core/tools/python.test.ts @@ -0,0 +1,120 @@ +import { afterAll, beforeAll, describe, expect, it, vi } from "bun:test"; +import * as pythonExecutor from "../python-executor"; +import { createTools, type ToolSession } from "./index"; +import { createPythonTool } from "./python"; + +let previousSkipCheck: string | undefined; + +beforeAll(() => { + previousSkipCheck = process.env.OMP_PYTHON_SKIP_CHECK; + process.env.OMP_PYTHON_SKIP_CHECK = "1"; +}); + +afterAll(() => { + if (previousSkipCheck === undefined) { + delete process.env.OMP_PYTHON_SKIP_CHECK; + return; + } + process.env.OMP_PYTHON_SKIP_CHECK = previousSkipCheck; +}); + +function createSession(overrides: Partial = {}): ToolSession { + return { + cwd: "/tmp/test", + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + ...overrides, + }; +} + +function createSettings(toolMode: "ipy-only" | "bash-only" | "both") { + return { + getImageAutoResize: () => true, + getLspFormatOnWrite: () => true, + getLspDiagnosticsOnWrite: () => true, + getLspDiagnosticsOnEdit: () => false, + getEditFuzzyMatch: () => true, + getGitToolEnabled: () => true, + getBashInterceptorEnabled: () => true, + getBashInterceptorSimpleLsEnabled: () => true, + getBashInterceptorRules: () => [], + getPythonToolMode: () => toolMode, + getPythonKernelMode: () => "session" as const, + }; +} + +describe("python tool schema", () => { + it("exposes expected parameters", () => { + const tool = createPythonTool(createSession()); + const schema = tool.parameters as { + type: string; + properties: Record; + required?: string[]; + }; + + expect(schema.type).toBe("object"); + expect(schema.properties.code.type).toBe("string"); + expect(schema.properties.timeout.type).toBe("number"); + expect(schema.properties.workdir.type).toBe("string"); + expect(schema.properties.reset.type).toBe("boolean"); + expect(schema.required).toEqual(["code"]); + }); +}); + +describe("python tool docs template", () => { + it("renders dynamic helper docs", () => { + const docs = [ + { + name: "read", + signature: "(path)", + docstring: "Read file contents.", + category: "File I/O", + }, + ]; + const spy = vi.spyOn(pythonExecutor, "getPreludeDocs").mockReturnValue(docs); + + const tool = createPythonTool(createSession()); + + expect(tool.description).toContain("### File I/O"); + expect(tool.description).toContain("- `read(path)` — Read file contents."); + + spy.mockRestore(); + }); + + it("renders fallback when docs are unavailable", () => { + const spy = vi.spyOn(pythonExecutor, "getPreludeDocs").mockReturnValue([]); + + const tool = createPythonTool(createSession()); + + expect(tool.description).toContain("Documentation unavailable — Python kernel failed to start"); + + spy.mockRestore(); + }); +}); + +describe("python tool exposure", () => { + it("includes python only in ipy-only mode", async () => { + const session = createSession({ settings: createSettings("ipy-only") }); + const tools = await createTools(session); + const names = tools.map((tool) => tool.name); + expect(names).toContain("python"); + expect(names).not.toContain("bash"); + }); + + it("includes bash only in bash-only mode", async () => { + const session = createSession({ settings: createSettings("bash-only") }); + const tools = await createTools(session); + const names = tools.map((tool) => tool.name); + expect(names).toContain("bash"); + expect(names).not.toContain("python"); + }); + + it("includes bash and python in both mode", async () => { + const session = createSession({ settings: createSettings("both") }); + const tools = await createTools(session); + const names = tools.map((tool) => tool.name); + expect(names).toContain("bash"); + expect(names).toContain("python"); + }); +}); diff --git a/packages/coding-agent/src/core/tools/python.ts b/packages/coding-agent/src/core/tools/python.ts new file mode 100644 index 000000000..6a1581faa --- /dev/null +++ b/packages/coding-agent/src/core/tools/python.ts @@ -0,0 +1,380 @@ +import { relative, resolve, sep } from "node:path"; +import type { AgentTool, AgentToolContext } from "@oh-my-pi/pi-agent-core"; +import type { ImageContent } from "@oh-my-pi/pi-ai"; +import type { Component } from "@oh-my-pi/pi-tui"; +import { Text, truncateToWidth } from "@oh-my-pi/pi-tui"; +import { Type } from "@sinclair/typebox"; +import { truncateToVisualLines } from "../../modes/interactive/components/visual-truncate"; +import type { Theme } from "../../modes/interactive/theme/theme"; +import pythonDescription from "../../prompts/tools/python.md" with { type: "text" }; +import type { RenderResultOptions } from "../custom-tools/types"; +import { renderPromptTemplate } from "../prompt-templates"; +import { executePython, getPreludeDocs, type PythonExecutorOptions } from "../python-executor"; +import type { PreludeHelper } from "../python-kernel"; +import type { ToolSession } from "./index"; +import { resolveToCwd } from "./path-utils"; +import { createToolUIKit, getTreeBranch, getTreeContinuePrefix } from "./render-utils"; +import { DEFAULT_MAX_BYTES, formatSize, type TruncationResult, truncateTail } from "./truncate"; + +export const PYTHON_DEFAULT_PREVIEW_LINES = 10; + +type PreludeCategory = { + name: string; + functions: PreludeHelper[]; +}; + +function groupPreludeHelpers(helpers: PreludeHelper[]): PreludeCategory[] { + const categories: PreludeCategory[] = []; + const byName = new Map(); + for (const helper of helpers) { + let bucket = byName.get(helper.category); + if (!bucket) { + bucket = []; + byName.set(helper.category, bucket); + categories.push({ name: helper.category, functions: bucket }); + } + bucket.push(helper); + } + return categories; +} + +export const pythonSchema = Type.Object({ + code: Type.String({ description: "Python code to execute" }), + timeout: Type.Optional(Type.Number({ description: "Timeout in seconds (optional, no default timeout)" })), + workdir: Type.Optional( + Type.String({ description: "Working directory for the command (default: current directory)" }), + ), + reset: Type.Optional(Type.Boolean({ description: "Restart the kernel before executing this code" })), +}); + +export interface PythonToolDetails { + truncation?: TruncationResult; + fullOutputPath?: string; + fullOutput?: string; + jsonOutputs?: unknown[]; + images?: ImageContent[]; +} + +function formatJsonScalar(value: unknown): string { + if (value === null) return "null"; + if (value === undefined) return "undefined"; + if (typeof value === "string") return JSON.stringify(value); + if (typeof value === "number" || typeof value === "boolean" || typeof value === "bigint") return String(value); + if (typeof value === "function") return "[function]"; + return "[object]"; +} + +function renderJsonTree(value: unknown, theme: Theme, expanded: boolean, maxDepth = expanded ? 6 : 2): string[] { + const maxItems = expanded ? 20 : 5; + + const renderNode = (node: unknown, prefix: string, depth: number, isLast: boolean, label?: string): string[] => { + const branch = getTreeBranch(isLast, theme); + const displayLabel = label ? `${label}: ` : ""; + + if (depth >= maxDepth || node === null || typeof node !== "object") { + return [`${prefix}${branch} ${displayLabel}${formatJsonScalar(node)}`]; + } + + const isArray = Array.isArray(node); + const entries = isArray + ? node.map((val, index) => [String(index), val] as const) + : Object.entries(node as object); + const header = `${prefix}${branch} ${displayLabel}${isArray ? `Array(${entries.length})` : `Object(${entries.length})`}`; + const lines = [header]; + + const childPrefix = prefix + getTreeContinuePrefix(isLast, theme); + const visible = entries.slice(0, maxItems); + for (let i = 0; i < visible.length; i++) { + const [key, val] = visible[i]; + const childLast = i === visible.length - 1 && (expanded || entries.length <= maxItems); + lines.push(...renderNode(val, childPrefix, depth + 1, childLast, isArray ? `[${key}]` : key)); + } + if (!expanded && entries.length > maxItems) { + const moreBranch = theme.tree.last; + lines.push(`${childPrefix}${moreBranch} ${entries.length - maxItems} more item(s)`); + } + return lines; + }; + + return renderNode(value, "", 0, true); +} + +export function createPythonTool(session: ToolSession): AgentTool { + const helpers = getPreludeDocs(); + const categories = groupPreludeHelpers(helpers); + return { + name: "python", + label: "Python", + description: renderPromptTemplate(pythonDescription, { categories }), + parameters: pythonSchema, + execute: async ( + _toolCallId: string, + { code, timeout, workdir, reset }: { code: string; timeout?: number; workdir?: string; reset?: boolean }, + signal?: AbortSignal, + onUpdate?, + _ctx?: AgentToolContext, + ) => { + const controller = new AbortController(); + const onAbort = () => controller.abort(); + signal?.addEventListener("abort", onAbort, { once: true }); + + try { + if (signal?.aborted) { + throw new Error("Aborted"); + } + + const commandCwd = workdir ? resolveToCwd(workdir, session.cwd) : session.cwd; + let cwdStat: Awaited>; + try { + cwdStat = await Bun.file(commandCwd).stat(); + } catch { + throw new Error(`Working directory does not exist: ${commandCwd}`); + } + if (!cwdStat.isDirectory()) { + throw new Error(`Working directory is not a directory: ${commandCwd}`); + } + + const maxTailBytes = DEFAULT_MAX_BYTES * 2; + const tailChunks: Array<{ text: string; bytes: number }> = []; + let tailBytes = 0; + const jsonOutputs: unknown[] = []; + const images: ImageContent[] = []; + + const executorOptions: PythonExecutorOptions = { + cwd: commandCwd, + timeout: timeout ? timeout * 1000 : undefined, + signal: controller.signal, + sessionId: session.getSessionFile?.() ?? `cwd:${session.cwd}`, + kernelMode: session.settings?.getPythonKernelMode?.() ?? "session", + reset, + onChunk: (chunk) => { + const chunkBytes = Buffer.byteLength(chunk, "utf-8"); + tailChunks.push({ text: chunk, bytes: chunkBytes }); + tailBytes += chunkBytes; + while (tailBytes > maxTailBytes && tailChunks.length > 1) { + const removed = tailChunks.shift(); + if (removed) { + tailBytes -= removed.bytes; + } + } + if (onUpdate) { + const tailText = tailChunks.map((entry) => entry.text).join(""); + const truncation = truncateTail(tailText); + onUpdate({ + content: [{ type: "text", text: truncation.content || "" }], + details: truncation.truncated ? { truncation } : undefined, + }); + } + }, + }; + + const result = await executePython(code, executorOptions); + + for (const output of result.displayOutputs) { + if (output.type === "json") { + jsonOutputs.push(output.data); + } + if (output.type === "image") { + images.push({ type: "image", data: output.data, mimeType: output.mimeType }); + } + } + + if (result.cancelled) { + throw new Error(result.output || "Command aborted"); + } + + const truncation = truncateTail(result.output); + let outputText = + truncation.content || (jsonOutputs.length > 0 || images.length > 0 ? "(no text output)" : "(no output)"); + let details: PythonToolDetails | undefined; + + if (truncation.truncated) { + const fullOutputSuffix = result.fullOutputPath ? ` Full output: ${result.fullOutputPath}` : ""; + details = { + truncation, + fullOutputPath: result.fullOutputPath, + jsonOutputs: jsonOutputs, + images, + }; + + const startLine = truncation.totalLines - truncation.outputLines + 1; + const endLine = truncation.totalLines; + + if (truncation.lastLinePartial) { + const lastLineSize = formatSize(Buffer.byteLength(result.output.split("\n").pop() || "", "utf-8")); + outputText += `\n\n[Showing last ${formatSize(truncation.outputBytes)} of line ${endLine} (line is ${lastLineSize})${fullOutputSuffix}]`; + } else if (truncation.truncatedBy === "lines") { + outputText += `\n\n[Showing lines ${startLine}-${endLine} of ${truncation.totalLines}${fullOutputSuffix}]`; + } else { + outputText += `\n\n[Showing lines ${startLine}-${endLine} of ${truncation.totalLines} (${formatSize(DEFAULT_MAX_BYTES)} limit)${fullOutputSuffix}]`; + } + } + + if (!details && (jsonOutputs.length > 0 || images.length > 0)) { + details = { jsonOutputs: jsonOutputs, images }; + } + + if (result.exitCode !== 0 && result.exitCode !== undefined) { + outputText += `\n\nCommand exited with code ${result.exitCode}`; + throw new Error(outputText); + } + + return { content: [{ type: "text", text: outputText }], details }; + } finally { + signal?.removeEventListener("abort", onAbort); + } + }, + }; +} + +interface PythonRenderArgs { + code?: string; + timeout?: number; + workdir?: string; +} + +interface PythonRenderContext { + output?: string; + expanded?: boolean; + previewLines?: number; + timeout?: number; +} + +export const pythonToolRenderer = { + renderCall(args: PythonRenderArgs, uiTheme: Theme): Component { + const ui = createToolUIKit(uiTheme); + const code = args.code || uiTheme.format.ellipsis; + const prompt = uiTheme.fg("accent", ">>>"); + const cwd = process.cwd(); + let displayWorkdir = args.workdir; + + if (displayWorkdir) { + const resolvedCwd = resolve(cwd); + const resolvedWorkdir = resolve(displayWorkdir); + if (resolvedWorkdir === resolvedCwd) { + displayWorkdir = undefined; + } else { + const relativePath = relative(resolvedCwd, resolvedWorkdir); + const isWithinCwd = relativePath && !relativePath.startsWith("..") && !relativePath.startsWith(`..${sep}`); + if (isWithinCwd) { + displayWorkdir = relativePath; + } + } + } + + const cmdText = displayWorkdir + ? `${prompt} ${uiTheme.fg("dim", `cd ${displayWorkdir} &&`)} ${code}` + : `${prompt} ${code}`; + const text = ui.title(cmdText); + return new Text(text, 0, 0); + }, + + renderResult( + result: { content: Array<{ type: string; text?: string }>; details?: PythonToolDetails }, + options: RenderResultOptions & { renderContext?: PythonRenderContext }, + uiTheme: Theme, + ): Component { + const ui = createToolUIKit(uiTheme); + const { renderContext } = options; + const details = result.details; + + const expanded = renderContext?.expanded ?? options.expanded; + const previewLines = renderContext?.previewLines ?? PYTHON_DEFAULT_PREVIEW_LINES; + const output = renderContext?.output ?? (result.content?.find((c) => c.type === "text")?.text ?? "").trim(); + const fullOutput = details?.fullOutput; + const displayOutput = expanded ? (fullOutput ?? output) : output; + const showingFullOutput = expanded && fullOutput !== undefined; + + const jsonOutputs = details?.jsonOutputs ?? []; + const jsonLines = jsonOutputs.flatMap((value, index) => { + const header = `JSON output ${index + 1}`; + const treeLines = renderJsonTree(value, uiTheme, expanded); + return [header, ...treeLines]; + }); + const combinedOutput = [displayOutput, ...jsonLines].filter(Boolean).join("\n"); + + const truncation = details?.truncation; + const fullOutputPath = details?.fullOutputPath; + const timeoutSeconds = renderContext?.timeout; + const timeoutLine = + typeof timeoutSeconds === "number" + ? uiTheme.fg("dim", ui.wrapBrackets(`Timeout: ${timeoutSeconds}s`)) + : undefined; + let warningLine: string | undefined; + if (fullOutputPath || (truncation?.truncated && !showingFullOutput)) { + const warnings: string[] = []; + if (fullOutputPath) { + warnings.push(`Full output: ${fullOutputPath}`); + } + if (truncation?.truncated && !showingFullOutput) { + if (truncation.truncatedBy === "lines") { + warnings.push(`Truncated: showing ${truncation.outputLines} of ${truncation.totalLines} lines`); + } else { + warnings.push( + `Truncated: ${truncation.outputLines} lines shown (${ui.formatBytes(truncation.maxBytes ?? DEFAULT_MAX_BYTES)} limit)`, + ); + } + } + if (warnings.length > 0) { + warningLine = uiTheme.fg("warning", ui.wrapBrackets(warnings.join(". "))); + } + } + + if (!combinedOutput) { + const lines = [timeoutLine, warningLine].filter(Boolean) as string[]; + return new Text(lines.join("\n"), 0, 0); + } + + if (expanded) { + const styledOutput = combinedOutput + .split("\n") + .map((line) => uiTheme.fg("toolOutput", line)) + .join("\n"); + const lines = [styledOutput, timeoutLine, warningLine].filter(Boolean) as string[]; + return new Text(lines.join("\n"), 0, 0); + } + + const styledOutput = combinedOutput + .split("\n") + .map((line) => uiTheme.fg("toolOutput", line)) + .join("\n"); + const textContent = `\n${styledOutput}`; + + let cachedWidth: number | undefined; + let cachedLines: string[] | undefined; + let cachedSkipped: number | undefined; + + return { + render: (width: number): string[] => { + if (cachedLines === undefined || cachedWidth !== width) { + const result = truncateToVisualLines(textContent, previewLines, width); + cachedLines = result.visualLines; + cachedSkipped = result.skippedCount; + cachedWidth = width; + } + const outputLines: string[] = []; + if (cachedSkipped && cachedSkipped > 0) { + outputLines.push(""); + const skippedLine = uiTheme.fg( + "dim", + `${uiTheme.format.ellipsis} (${cachedSkipped} earlier lines, showing ${cachedLines.length} of ${cachedSkipped + cachedLines.length}) (ctrl+o to expand)`, + ); + outputLines.push(truncateToWidth(skippedLine, width, uiTheme.fg("dim", uiTheme.format.ellipsis))); + } + outputLines.push(...cachedLines); + if (timeoutLine) { + outputLines.push(truncateToWidth(timeoutLine, width, uiTheme.fg("dim", uiTheme.format.ellipsis))); + } + if (warningLine) { + outputLines.push(truncateToWidth(warningLine, width, uiTheme.fg("warning", uiTheme.format.ellipsis))); + } + return outputLines; + }, + invalidate: () => { + cachedWidth = undefined; + cachedLines = undefined; + cachedSkipped = undefined; + }, + }; + }, +}; diff --git a/packages/coding-agent/src/core/tools/renderers.ts b/packages/coding-agent/src/core/tools/renderers.ts index 5557d50fd..1a2f1b686 100644 --- a/packages/coding-agent/src/core/tools/renderers.ts +++ b/packages/coding-agent/src/core/tools/renderers.ts @@ -17,6 +17,7 @@ import { lsToolRenderer } from "./ls"; import { lspToolRenderer } from "./lsp/render"; import { notebookToolRenderer } from "./notebook"; import { outputToolRenderer } from "./output"; +import { pythonToolRenderer } from "./python"; import { readToolRenderer } from "./read"; import { sshToolRenderer } from "./ssh"; import { taskToolRenderer } from "./task/render"; @@ -41,6 +42,7 @@ type ToolRenderer = { export const toolRenderers: Record = { ask: askToolRenderer as ToolRenderer, bash: bashToolRenderer as ToolRenderer, + python: pythonToolRenderer as ToolRenderer, calc: calculatorToolRenderer as ToolRenderer, edit: editToolRenderer as ToolRenderer, find: findToolRenderer as ToolRenderer, diff --git a/packages/coding-agent/src/core/tools/task/executor.ts b/packages/coding-agent/src/core/tools/task/executor.ts index 7327228c3..9f3255db8 100644 --- a/packages/coding-agent/src/core/tools/task/executor.ts +++ b/packages/coding-agent/src/core/tools/task/executor.ts @@ -10,6 +10,9 @@ import type { EventBus } from "../../event-bus"; import { callTool } from "../../mcp/client"; import type { MCPManager } from "../../mcp/manager"; import type { ModelRegistry } from "../../model-registry"; +import { checkPythonKernelAvailability } from "../../python-kernel"; +import type { ToolSession } from ".."; +import { createPythonTool } from "../python"; import { ensureArtifactsDir, getArtifactPaths } from "./artifacts"; import { resolveModelPattern } from "./model-resolver"; import { subprocessToolRegistry } from "./subprocess-tool-registry"; @@ -26,6 +29,7 @@ import { import type { MCPToolCallRequest, MCPToolMetadata, + PythonToolCallRequest, SubagentWorkerRequest, SubagentWorkerResponse, } from "./worker-protocol"; @@ -52,7 +56,11 @@ export interface ExecutorOptions { mcpManager?: MCPManager; authStorage?: AuthStorage; modelRegistry?: ModelRegistry; - settingsManager?: { serialize: () => import("../../settings-manager").Settings }; + settingsManager?: { + serialize: () => import("../../settings-manager").Settings; + getPythonToolMode?: () => "ipy-only" | "bash-only" | "both"; + getPythonKernelMode?: () => "session" | "per-call"; + }; } /** @@ -269,14 +277,35 @@ export async function runSubprocess(options: ExecutorOptions): Promise name !== "exec"); + if (pythonToolMode === "bash-only") { + expanded.push("bash"); + } else if (pythonToolMode === "ipy-only") { + expanded.push("python"); + } else { + expanded.push("python", "bash"); + } + toolNames = Array.from(new Set(expanded)); + } + const serializedSettings = options.settingsManager?.serialize(); const availableModels = options.modelRegistry?.getAvailable().map((model) => `${model.provider}/${model.id}`); // Resolve and add model const resolvedModel = await resolveModelPattern(modelOverride || agent.model, availableModels, serializedSettings); - const sessionFile = subtaskSessionFile ?? options.sessionFile ?? null; + const parentSessionFile = options.sessionFile ?? null; + const sessionFile = subtaskSessionFile ?? parentSessionFile; const spawnsEnv = agent.spawns === undefined ? "" : agent.spawns === "*" ? "*" : agent.spawns.join(","); + const pythonToolRequested = toolNames === undefined || toolNames.includes("python"); + let pythonProxyEnabled = pythonToolRequested && pythonToolMode !== "bash-only"; + if (pythonProxyEnabled) { + const availability = await checkPythonKernelAvailability(cwd); + pythonProxyEnabled = availability.ok; + } + let worker: Worker; try { worker = new Worker(new URL("./worker.ts", import.meta.url), { type: "module" }); @@ -312,6 +341,17 @@ export async function runSubprocess(options: ExecutorOptions): Promise parentSessionFile, + getSessionSpawns: () => spawnsEnv, + settings: options.settingsManager as ToolSession["settings"], + settingsManager: options.settingsManager, + }; + const pythonTool = pythonProxyEnabled ? createPythonTool(pythonToolSession) : null; + // Accumulate usage incrementally from message_end events (no memory for streaming events) const accumulatedUsage = { input: 0, @@ -606,6 +646,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise { + if (!pythonTool) { + worker.postMessage({ + type: "python_tool_result", + callId: request.callId, + error: "Python proxy not available", + }); + return; + } + try { + const result = await pythonTool.execute( + request.callId, + request.params as { code: string; timeout?: number; workdir?: string; reset?: boolean }, + signal, + ); + worker.postMessage({ + type: "python_tool_result", + callId: request.callId, + result: { content: result.content ?? [], details: result.details }, + }); + } catch (error) { + worker.postMessage({ + type: "python_tool_result", + callId: request.callId, + error: error instanceof Error ? error.message : String(error), + }); + } + }; + const onMessage = (event: WorkerMessageEvent) => { const message = event.data; if (!message || resolved) return; @@ -664,6 +734,10 @@ export async function runSubprocess(options: ExecutorOptions): Promise; +} + +export interface PythonToolCallResponse { + type: "python_tool_result"; + callId: string; + result?: { + content: Array<{ type: string; text?: string; [key: string]: unknown }>; + details?: unknown; + isError?: boolean; + }; + error?: string; +} + export interface SubagentWorkerStartPayload { cwd: string; task: string; @@ -54,14 +71,17 @@ export interface SubagentWorkerStartPayload { serializedModels?: SerializedModelRegistry; serializedSettings?: Settings; mcpTools?: MCPToolMetadata[]; + pythonToolProxy?: boolean; } export type SubagentWorkerRequest = | { type: "start"; payload: SubagentWorkerStartPayload } | { type: "abort" } - | MCPToolCallResponse; + | MCPToolCallResponse + | PythonToolCallResponse; export type SubagentWorkerResponse = | { type: "event"; event: AgentEvent } | { type: "done"; exitCode: number; durationMs: number; error?: string; aborted?: boolean } - | MCPToolCallRequest; + | MCPToolCallRequest + | PythonToolCallRequest; diff --git a/packages/coding-agent/src/core/tools/task/worker.ts b/packages/coding-agent/src/core/tools/task/worker.ts index e2cd17dfc..d54823a78 100644 --- a/packages/coding-agent/src/core/tools/task/worker.ts +++ b/packages/coding-agent/src/core/tools/task/worker.ts @@ -25,9 +25,11 @@ import { createAgentSession, discoverAuthStorage, discoverModels } from "../../s import { SessionManager } from "../../session-manager"; import { SettingsManager } from "../../settings-manager"; import { untilAborted } from "../../utils"; +import { pythonSchema } from "../python"; import type { MCPToolCallResponse, MCPToolMetadata, + PythonToolCallResponse, SubagentWorkerRequest, SubagentWorkerResponse, SubagentWorkerStartPayload, @@ -49,14 +51,27 @@ interface PendingMCPCall { timeoutId: ReturnType; } +interface PendingPythonCall { + resolve: (result: PythonToolCallResponse["result"]) => void; + reject: (error: Error) => void; + timeoutId: ReturnType; +} + const pendingMCPCalls = new Map(); +const pendingPythonCalls = new Map(); const MCP_CALL_TIMEOUT_MS = 60_000; +const PYTHON_CALL_TIMEOUT_MS = 300_000; let mcpCallIdCounter = 0; +let pythonCallIdCounter = 0; function generateMCPCallId(): string { return `mcp_${Date.now()}_${++mcpCallIdCounter}`; } +function generatePythonCallId(): string { + return `python_${Date.now()}_${++pythonCallIdCounter}`; +} + function callMCPToolViaParent( toolName: string, params: Record, @@ -110,6 +125,57 @@ function callMCPToolViaParent( }); } +function callPythonToolViaParent( + params: Record, + signal?: AbortSignal, + timeoutMs = PYTHON_CALL_TIMEOUT_MS, +): Promise { + return new Promise((resolve, reject) => { + const callId = generatePythonCallId(); + if (signal?.aborted) { + reject(new Error("Aborted")); + return; + } + + const timeoutId = setTimeout(() => { + pendingPythonCalls.delete(callId); + reject(new Error(`Python call timed out after ${timeoutMs}ms`)); + }, timeoutMs); + + const cleanup = () => { + clearTimeout(timeoutId); + pendingPythonCalls.delete(callId); + }; + + signal?.addEventListener( + "abort", + () => { + cleanup(); + reject(new Error("Aborted")); + }, + { once: true }, + ); + + pendingPythonCalls.set(callId, { + resolve: (result) => { + cleanup(); + resolve(result ?? { content: [] }); + }, + reject: (error) => { + cleanup(); + reject(error); + }, + timeoutId, + }); + + postMessageSafe({ + type: "python_tool_call", + callId, + params, + } as SubagentWorkerResponse); + }); +} + function handleMCPToolResult(response: MCPToolCallResponse): void { const pending = pendingMCPCalls.get(response.callId); if (!pending) return; @@ -120,6 +186,16 @@ function handleMCPToolResult(response: MCPToolCallResponse): void { } } +function handlePythonToolResult(response: PythonToolCallResponse): void { + const pending = pendingPythonCalls.get(response.callId); + if (!pending) return; + if (response.error) { + pending.reject(new Error(response.error)); + } else { + pending.resolve(response.result); + } +} + function createMCPProxyTool(metadata: MCPToolMetadata): CustomTool { return { name: metadata.name, @@ -157,6 +233,48 @@ function createMCPProxyTool(metadata: MCPToolMetadata): CustomTool { }; } +function getPythonCallTimeoutMs(params: Record): number { + const timeout = params.timeout; + if (typeof timeout === "number" && Number.isFinite(timeout) && timeout > 0) { + return timeout * 1000 + 5000; + } + return PYTHON_CALL_TIMEOUT_MS; +} + +function createPythonProxyTool(): CustomTool { + return { + name: "python", + label: "Python", + description: "Execute Python code via the parent kernel.", + parameters: pythonSchema, + execute: async (_toolCallId, params, _onUpdate, _ctx, signal) => { + try { + const timeoutMs = getPythonCallTimeoutMs(params as Record); + const result = await callPythonToolViaParent(params as Record, signal, timeoutMs); + return { + content: + result?.content?.map((c) => + c.type === "text" + ? { type: "text" as const, text: c.text ?? "" } + : { type: "text" as const, text: JSON.stringify(c) }, + ) ?? [], + details: result?.details, + }; + } catch (error) { + return { + content: [ + { + type: "text" as const, + text: `Python error: ${error instanceof Error ? error.message : String(error)}`, + }, + ], + details: { isError: true }, + }; + } + }, + }; +} + interface WorkerMessageEvent { data: T; } @@ -284,8 +402,10 @@ async function runTask(runState: RunState, payload: SubagentWorkerStartPayload): checkAbort(); } - // Create MCP proxy tools if provided + // Create MCP/python proxy tools if provided const mcpProxyTools = payload.mcpTools?.map(createMCPProxyTool) ?? []; + const pythonProxyTools = payload.pythonToolProxy ? [createPythonProxyTool()] : []; + const proxyTools = [...mcpProxyTools, ...pythonProxyTools]; // Resolve model override (equivalent to CLI's parseModelPattern with --model) const { model, thinkingLevel: modelThinkingLevel } = resolveModelOverride(payload.model, modelRegistry); @@ -325,8 +445,8 @@ async function runTask(runState: RunState, payload: SubagentWorkerStartPayload): enableLsp: payload.enableLsp ?? true, // Disable local MCP discovery if using proxy tools enableMCP: !payload.mcpTools, - // Add MCP proxy tools - customTools: mcpProxyTools.length > 0 ? mcpProxyTools : undefined, + // Add proxy tools + customTools: proxyTools.length > 0 ? proxyTools : undefined, }); runState.session = session; @@ -568,6 +688,11 @@ globalThis.addEventListener("message", (event: WorkerMessageEvent = { /** * All settings definitions. * Order determines display order within each tab. + * + * Tabs: + * - behavior: Core agent behavior (compaction, modes, retries, notifications) + * - tools: Tool-specific settings (bash, git, python, edit, MCP, skills) + * - display: Visual/UI settings (theme, images, thinking) + * - voice: Voice mode and TTSR settings + * - status: Status line configuration + * - lsp: LSP integration settings + * - exa: Exa search tool settings */ export const SETTINGS_DEFS: SettingDef[] = [ - // Config tab + // ═══════════════════════════════════════════════════════════════════════════ + // Behavior tab - Core agent behavior + // ═══════════════════════════════════════════════════════════════════════════ { id: "autoCompact", - tab: "config", + tab: "behavior", type: "boolean", label: "Auto-compact", description: "Automatically compact context when it gets too large", get: (sm) => sm.getCompactionEnabled(), - set: (sm, v) => sm.setCompactionEnabled(v), // Also handled in session + set: (sm, v) => sm.setCompactionEnabled(v), }, { id: "branchSummaries", - tab: "config", + tab: "behavior", type: "boolean", label: "Branch summaries", description: "Prompt to summarize when leaving a branch", @@ -99,35 +112,77 @@ export const SETTINGS_DEFS: SettingDef[] = [ }, { id: "todoCompletion", - tab: "config", + tab: "behavior", type: "boolean", label: "Todo completion", - description: "Remind agent to complete todos before stopping (up to 3 reminders)", + description: "Remind agent to complete todos before stopping", get: (sm) => sm.getTodoCompletionEnabled(), set: (sm, v) => sm.setTodoCompletionEnabled(v), }, { - id: "showImages", - tab: "config", - type: "boolean", - label: "Show images", - description: "Render images inline in terminal", - get: (sm) => sm.getShowImages(), - set: (sm, v) => sm.setShowImages(v), - condition: () => !!getCapabilities().images, + id: "todoCompletionMaxReminders", + tab: "behavior", + type: "submenu", + label: "Todo max reminders", + description: "Maximum reminders to complete todos before giving up", + get: (sm) => String(sm.getTodoCompletionMaxReminders()), + set: (sm, v) => sm.setTodoCompletionMaxReminders(Number.parseInt(v, 10)), + getOptions: () => [ + { value: "1", label: "1 reminder" }, + { value: "2", label: "2 reminders" }, + { value: "3", label: "3 reminders" }, + { value: "5", label: "5 reminders" }, + ], }, { - id: "voiceEnabled", - tab: "config", - type: "boolean", - label: "Voice mode", - description: "Enable realtime voice input/output (Ctrl+Y toggle, auto-send on silence)", - get: (sm) => sm.getVoiceEnabled(), - set: (sm, v) => sm.setVoiceEnabled(v), + id: "steeringMode", + tab: "behavior", + type: "enum", + label: "Steering mode", + description: "How to process queued messages while agent is working", + values: ["one-at-a-time", "all"], + get: (sm) => sm.getSteeringMode(), + set: (sm, v) => sm.setSteeringMode(v as "all" | "one-at-a-time"), + }, + { + id: "followUpMode", + tab: "behavior", + type: "enum", + label: "Follow-up mode", + description: "How to drain follow-up messages after a turn completes", + values: ["one-at-a-time", "all"], + get: (sm) => sm.getFollowUpMode(), + set: (sm, v) => sm.setFollowUpMode(v as "one-at-a-time" | "all"), + }, + { + id: "interruptMode", + tab: "behavior", + type: "enum", + label: "Interrupt mode", + description: "When steering messages interrupt tool execution", + values: ["immediate", "wait"], + get: (sm) => sm.getInterruptMode(), + set: (sm, v) => sm.setInterruptMode(v as "immediate" | "wait"), + }, + { + id: "retryMaxRetries", + tab: "behavior", + type: "submenu", + label: "Retry max attempts", + description: "Maximum retry attempts on API errors", + get: (sm) => String(sm.getRetryMaxRetries()), + set: (sm, v) => sm.setRetryMaxRetries(Number.parseInt(v, 10)), + getOptions: () => [ + { value: "1", label: "1 retry" }, + { value: "2", label: "2 retries" }, + { value: "3", label: "3 retries" }, + { value: "5", label: "5 retries" }, + { value: "10", label: "10 retries" }, + ], }, { id: "completionNotification", - tab: "config", + tab: "behavior", type: "enum", label: "Completion notification", description: "Notify when the agent completes", @@ -135,75 +190,9 @@ export const SETTINGS_DEFS: SettingDef[] = [ get: (sm) => sm.getNotificationOnComplete(), set: (sm, v) => sm.setNotificationOnComplete(v as NotificationMethod), }, - { - id: "autoResizeImages", - tab: "config", - type: "boolean", - label: "Auto-resize images", - description: "Resize large images to 2000x2000 max for better model compatibility", - get: (sm) => sm.getImageAutoResize(), - set: (sm, v) => sm.setImageAutoResize(v), - }, - { - id: "blockImages", - tab: "config", - type: "boolean", - label: "Block images", - description: "Prevent images from being sent to LLM providers", - get: (sm) => sm.getBlockImages(), - set: (sm, v) => sm.setBlockImages(v), - }, - { - id: "steeringMode", - tab: "config", - type: "enum", - label: "Steering mode", - description: "How to process queued messages while agent is working", - values: ["one-at-a-time", "all"], - get: (sm) => sm.getSteeringMode(), - set: (sm, v) => sm.setSteeringMode(v as "all" | "one-at-a-time"), // Also handled in session - }, - { - id: "followUpMode", - tab: "config", - type: "enum", - label: "Follow-up mode", - description: "How to drain follow-up messages after a turn completes", - values: ["one-at-a-time", "all"], - get: (sm) => sm.getFollowUpMode(), - set: (sm, v) => sm.setFollowUpMode(v as "one-at-a-time" | "all"), // Also handled in session - }, - { - id: "interruptMode", - tab: "config", - type: "enum", - label: "Interrupt mode", - description: "When steering messages interrupt tool execution", - values: ["immediate", "wait"], - get: (sm) => sm.getInterruptMode(), - set: (sm, v) => sm.setInterruptMode(v as "immediate" | "wait"), // Also handled in session - }, - { - id: "hideThinking", - tab: "config", - type: "boolean", - label: "Hide thinking", - description: "Hide thinking blocks in assistant responses", - get: (sm) => sm.getHideThinkingBlock(), - set: (sm, v) => sm.setHideThinkingBlock(v), - }, - { - id: "collapseChangelog", - tab: "config", - type: "boolean", - label: "Collapse changelog", - description: "Show condensed changelog after updates", - get: (sm) => sm.getCollapseChangelog(), - set: (sm, v) => sm.setCollapseChangelog(v), - }, { id: "startupQuiet", - tab: "config", + tab: "behavior", type: "boolean", label: "Startup quiet", description: "Skip welcome screen and startup status messages", @@ -211,17 +200,17 @@ export const SETTINGS_DEFS: SettingDef[] = [ set: (sm, v) => sm.setStartupQuiet(v), }, { - id: "showHardwareCursor", - tab: "config", + id: "collapseChangelog", + tab: "behavior", type: "boolean", - label: "Hardware cursor", - description: "Show terminal cursor for IME support (default: on for Linux/macOS)", - get: (sm) => sm.getShowHardwareCursor(), - set: (sm, v) => sm.setShowHardwareCursor(v), + label: "Collapse changelog", + description: "Show condensed changelog after updates", + get: (sm) => sm.getCollapseChangelog(), + set: (sm, v) => sm.setCollapseChangelog(v), }, { id: "doubleEscapeAction", - tab: "config", + tab: "behavior", type: "enum", label: "Double-escape action", description: "Action when pressing Escape twice with empty editor", @@ -229,18 +218,31 @@ export const SETTINGS_DEFS: SettingDef[] = [ get: (sm) => sm.getDoubleEscapeAction(), set: (sm, v) => sm.setDoubleEscapeAction(v as "branch" | "tree"), }, + + // ═══════════════════════════════════════════════════════════════════════════ + // Tools tab - Tool-specific settings + // ═══════════════════════════════════════════════════════════════════════════ { id: "bashInterceptor", - tab: "config", + tab: "tools", type: "boolean", label: "Bash interceptor", description: "Block shell commands that have dedicated tools (grep, cat, etc.)", get: (sm) => sm.getBashInterceptorEnabled(), set: (sm, v) => sm.setBashInterceptorEnabled(v), }, + { + id: "bashInterceptorSimpleLs", + tab: "tools", + type: "boolean", + label: "Intercept simple ls", + description: "Intercept bare ls commands (when bash interceptor is enabled)", + get: (sm) => sm.getBashInterceptorSimpleLsEnabled(), + set: (sm, v) => sm.setBashInterceptorSimpleLsEnabled(v), + }, { id: "gitTool", - tab: "config", + tab: "tools", type: "boolean", label: "Git tool", description: "Enable structured Git tool", @@ -248,17 +250,28 @@ export const SETTINGS_DEFS: SettingDef[] = [ set: (sm, v) => sm.setGitToolEnabled(v), }, { - id: "mcpProjectConfig", - tab: "config", - type: "boolean", - label: "MCP project config", - description: "Load .mcp.json/mcp.json from project root", - get: (sm) => sm.getMCPProjectConfigEnabled(), - set: (sm, v) => sm.setMCPProjectConfigEnabled(v), + id: "pythonToolMode", + tab: "tools", + type: "enum", + label: "Python tool mode", + description: "How Python code is executed", + values: ["ipy-only", "bash-only", "both"], + get: (sm) => sm.getPythonToolMode(), + set: (sm, v) => sm.setPythonToolMode(v as PythonToolMode), + }, + { + id: "pythonKernelMode", + tab: "tools", + type: "enum", + label: "Python kernel mode", + description: "Whether to keep IPython kernel alive across calls", + values: ["session", "per-call"], + get: (sm) => sm.getPythonKernelMode(), + set: (sm, v) => sm.setPythonKernelMode(v as PythonKernelMode), }, { id: "editFuzzyMatch", - tab: "config", + tab: "tools", type: "boolean", label: "Edit fuzzy match", description: "Accept high-confidence fuzzy matches for whitespace/indentation differences", @@ -266,76 +279,44 @@ export const SETTINGS_DEFS: SettingDef[] = [ set: (sm, v) => sm.setEditFuzzyMatch(v), }, { - id: "ttsrEnabled", - tab: "config", + id: "mcpProjectConfig", + tab: "tools", type: "boolean", - label: "TTSR enabled", - description: "Time Traveling Stream Rules: interrupt agent when output matches rule patterns", - get: (sm) => sm.getTtsrEnabled(), - set: (sm, v) => sm.setTtsrEnabled(v), + label: "MCP project config", + description: "Load .mcp.json/mcp.json from project root", + get: (sm) => sm.getMCPProjectConfigEnabled(), + set: (sm, v) => sm.setMCPProjectConfigEnabled(v), }, { - id: "ttsrContextMode", - tab: "config", - type: "enum", - label: "TTSR context mode", - description: "What to do with partial output when TTSR triggers", - values: ["discard", "keep"], - get: (sm) => sm.getTtsrContextMode(), - set: (sm, v) => sm.setTtsrContextMode(v as "keep" | "discard"), + id: "skillCommands", + tab: "tools", + type: "boolean", + label: "Skill commands", + description: "Register skills as /skill:name commands", + get: (sm) => sm.getEnableSkillCommands(), + set: (sm, v) => sm.setEnableSkillCommands(v), }, { - id: "ttsrRepeatMode", - tab: "config", - type: "enum", - label: "TTSR repeat mode", - description: "How rules can repeat: once per session or after a message gap", - values: ["once", "after-gap"], - get: (sm) => sm.getTtsrRepeatMode(), - set: (sm, v) => sm.setTtsrRepeatMode(v as "once" | "after-gap"), + id: "claudeUserCommands", + tab: "tools", + type: "boolean", + label: "Claude user commands", + description: "Load commands from ~/.claude/commands/", + get: (sm) => sm.getCommandsEnableClaudeUser(), + set: (sm, v) => sm.setCommandsEnableClaudeUser(v), }, { - id: "thinkingLevel", - tab: "config", - type: "submenu", - label: "Thinking level", - description: "Reasoning depth for thinking-capable models", - get: (sm) => sm.getDefaultThinkingLevel() ?? "off", - set: (sm, v) => sm.setDefaultThinkingLevel(v as ThinkingLevel), // Also handled in session - getOptions: () => - (["off", "minimal", "low", "medium", "high", "xhigh"] as ThinkingLevel[]).map((level) => ({ - value: level, - label: level, - description: THINKING_DESCRIPTIONS[level], - })), - }, - { - id: "theme", - tab: "config", - type: "submenu", - label: "Theme", - description: "Color theme for the interface", - get: (sm) => sm.getTheme() ?? "dark", - set: (sm, v) => sm.setTheme(v), - getOptions: () => [], // Filled dynamically from context - }, - { - id: "symbolPreset", - tab: "config", - type: "submenu", - label: "Symbol preset", - description: "Icon/symbol style (overrides theme default)", - get: (sm) => sm.getSymbolPreset() ?? "unicode", - set: (sm, v) => sm.setSymbolPreset(v as SymbolPreset), - getOptions: () => [ - { value: "unicode", label: "Unicode", description: "Standard Unicode symbols (default)" }, - { value: "nerd", label: "Nerd Font", description: "Nerd Font icons (requires Nerd Font)" }, - { value: "ascii", label: "ASCII", description: "ASCII-only characters (maximum compatibility)" }, - ], + id: "claudeProjectCommands", + tab: "tools", + type: "boolean", + label: "Claude project commands", + description: "Load commands from .claude/commands/", + get: (sm) => sm.getCommandsEnableClaudeProject(), + set: (sm, v) => sm.setCommandsEnableClaudeProject(v), }, { id: "webSearchProvider", - tab: "config", + tab: "tools", type: "submenu", label: "Web search provider", description: "Provider for web search tool", @@ -350,7 +331,7 @@ export const SETTINGS_DEFS: SettingDef[] = [ }, { id: "imageProvider", - tab: "config", + tab: "tools", type: "submenu", label: "Image provider", description: "Provider for image generation tool", @@ -363,92 +344,203 @@ export const SETTINGS_DEFS: SettingDef[] = [ ], }, - // LSP tab + // ═══════════════════════════════════════════════════════════════════════════ + // Display tab - Visual/UI settings + // ═══════════════════════════════════════════════════════════════════════════ { - id: "lspFormatOnWrite", - tab: "lsp", - type: "boolean", - label: "Format on write", - description: "Automatically format code files using LSP after writing", - get: (sm) => sm.getLspFormatOnWrite(), - set: (sm, v) => sm.setLspFormatOnWrite(v), + id: "theme", + tab: "display", + type: "submenu", + label: "Theme", + description: "Color theme for the interface", + get: (sm) => sm.getTheme() ?? "dark", + set: (sm, v) => sm.setTheme(v), + getOptions: () => [], // Filled dynamically from context }, { - id: "lspDiagnosticsOnWrite", - tab: "lsp", - type: "boolean", - label: "Diagnostics on write", - description: "Return LSP diagnostics (errors/warnings) after writing code files", - get: (sm) => sm.getLspDiagnosticsOnWrite(), - set: (sm, v) => sm.setLspDiagnosticsOnWrite(v), + id: "symbolPreset", + tab: "display", + type: "submenu", + label: "Symbol preset", + description: "Icon/symbol style (overrides theme default)", + get: (sm) => sm.getSymbolPreset() ?? "unicode", + set: (sm, v) => sm.setSymbolPreset(v as SymbolPreset), + getOptions: () => [ + { value: "unicode", label: "Unicode", description: "Standard Unicode symbols (default)" }, + { value: "nerd", label: "Nerd Font", description: "Nerd Font icons (requires Nerd Font)" }, + { value: "ascii", label: "ASCII", description: "ASCII-only characters (maximum compatibility)" }, + ], }, { - id: "lspDiagnosticsOnEdit", - tab: "lsp", + id: "thinkingLevel", + tab: "display", + type: "submenu", + label: "Thinking level", + description: "Reasoning depth for thinking-capable models", + get: (sm) => sm.getDefaultThinkingLevel() ?? "off", + set: (sm, v) => sm.setDefaultThinkingLevel(v as ThinkingLevel), + getOptions: () => + (["off", "minimal", "low", "medium", "high", "xhigh"] as ThinkingLevel[]).map((level) => ({ + value: level, + label: level, + description: THINKING_DESCRIPTIONS[level], + })), + }, + { + id: "hideThinking", + tab: "display", type: "boolean", - label: "Diagnostics on edit", - description: "Return LSP diagnostics (errors/warnings) after editing code files", - get: (sm) => sm.getLspDiagnosticsOnEdit(), - set: (sm, v) => sm.setLspDiagnosticsOnEdit(v), + label: "Hide thinking", + description: "Hide thinking blocks in assistant responses", + get: (sm) => sm.getHideThinkingBlock(), + set: (sm, v) => sm.setHideThinkingBlock(v), + }, + { + id: "showImages", + tab: "display", + type: "boolean", + label: "Show images", + description: "Render images inline in terminal", + get: (sm) => sm.getShowImages(), + set: (sm, v) => sm.setShowImages(v), + condition: () => !!getCapabilities().images, + }, + { + id: "autoResizeImages", + tab: "display", + type: "boolean", + label: "Auto-resize images", + description: "Resize large images to 2000x2000 max for better model compatibility", + get: (sm) => sm.getImageAutoResize(), + set: (sm, v) => sm.setImageAutoResize(v), + }, + { + id: "blockImages", + tab: "display", + type: "boolean", + label: "Block images", + description: "Prevent images from being sent to LLM providers", + get: (sm) => sm.getBlockImages(), + set: (sm, v) => sm.setBlockImages(v), + }, + { + id: "showHardwareCursor", + tab: "display", + type: "boolean", + label: "Hardware cursor", + description: "Show terminal cursor for IME support (default: on for Linux/macOS)", + get: (sm) => sm.getShowHardwareCursor(), + set: (sm, v) => sm.setShowHardwareCursor(v), }, - // Exa tab + // ═══════════════════════════════════════════════════════════════════════════ + // Voice tab - Voice mode and TTSR settings + // ═══════════════════════════════════════════════════════════════════════════ { - id: "exaEnabled", - tab: "exa", + id: "voiceEnabled", + tab: "voice", type: "boolean", - label: "Exa enabled", - description: "Master toggle for all Exa search tools", - get: (sm) => sm.getExaSettings().enabled, - set: (sm, v) => sm.setExaEnabled(v), + label: "Voice mode", + description: "Enable realtime voice input/output (Ctrl+Y toggle, auto-send on silence)", + get: (sm) => sm.getVoiceEnabled(), + set: (sm, v) => sm.setVoiceEnabled(v), }, { - id: "exaSearch", - tab: "exa", - type: "boolean", - label: "Exa search", - description: "Basic search, deep search, code search, crawl", - get: (sm) => sm.getExaSettings().enableSearch, - set: (sm, v) => sm.setExaSearchEnabled(v), + id: "voiceTtsModel", + tab: "voice", + type: "submenu", + label: "TTS model", + description: "Text-to-speech model for voice output", + get: (sm) => sm.getVoiceTtsModel(), + set: (sm, v) => sm.setVoiceTtsModel(v), + getOptions: () => [ + { value: "gpt-4o-mini-tts", label: "GPT-4o Mini TTS", description: "Fast and efficient" }, + { value: "tts-1", label: "TTS-1", description: "Standard quality" }, + { value: "tts-1-hd", label: "TTS-1 HD", description: "Higher quality" }, + ], }, { - id: "exaLinkedin", - tab: "exa", - type: "boolean", - label: "Exa LinkedIn", - description: "Search LinkedIn for people and companies", - get: (sm) => sm.getExaSettings().enableLinkedin, - set: (sm, v) => sm.setExaLinkedinEnabled(v), + id: "voiceTtsVoice", + tab: "voice", + type: "submenu", + label: "TTS voice", + description: "Voice for text-to-speech output", + get: (sm) => sm.getVoiceTtsVoice(), + set: (sm, v) => sm.setVoiceTtsVoice(v), + getOptions: () => [ + { value: "alloy", label: "Alloy", description: "Neutral" }, + { value: "echo", label: "Echo", description: "Male" }, + { value: "fable", label: "Fable", description: "British" }, + { value: "onyx", label: "Onyx", description: "Deep male" }, + { value: "nova", label: "Nova", description: "Female" }, + { value: "shimmer", label: "Shimmer", description: "Female" }, + ], }, { - id: "exaCompany", - tab: "exa", - type: "boolean", - label: "Exa company", - description: "Comprehensive company research tool", - get: (sm) => sm.getExaSettings().enableCompany, - set: (sm, v) => sm.setExaCompanyEnabled(v), + id: "voiceTtsFormat", + tab: "voice", + type: "submenu", + label: "TTS format", + description: "Audio format for voice output", + get: (sm) => sm.getVoiceTtsFormat(), + set: (sm, v) => sm.setVoiceTtsFormat(v as "wav" | "mp3" | "opus" | "aac" | "flac"), + getOptions: () => [ + { value: "wav", label: "WAV", description: "Uncompressed, best quality" }, + { value: "mp3", label: "MP3", description: "Compressed, widely compatible" }, + { value: "opus", label: "Opus", description: "Efficient compression" }, + { value: "aac", label: "AAC", description: "Apple-friendly" }, + { value: "flac", label: "FLAC", description: "Lossless compression" }, + ], }, { - id: "exaResearcher", - tab: "exa", + id: "ttsrEnabled", + tab: "voice", type: "boolean", - label: "Exa researcher", - description: "AI-powered deep research tasks", - get: (sm) => sm.getExaSettings().enableResearcher, - set: (sm, v) => sm.setExaResearcherEnabled(v), + label: "TTSR enabled", + description: "Time Traveling Stream Rules: interrupt agent when output matches patterns", + get: (sm) => sm.getTtsrEnabled(), + set: (sm, v) => sm.setTtsrEnabled(v), }, { - id: "exaWebsets", - tab: "exa", - type: "boolean", - label: "Exa websets", - description: "Webset management and enrichment tools", - get: (sm) => sm.getExaSettings().enableWebsets, - set: (sm, v) => sm.setExaWebsetsEnabled(v), + id: "ttsrContextMode", + tab: "voice", + type: "enum", + label: "TTSR context mode", + description: "What to do with partial output when TTSR triggers", + values: ["discard", "keep"], + get: (sm) => sm.getTtsrContextMode(), + set: (sm, v) => sm.setTtsrContextMode(v as "keep" | "discard"), + }, + { + id: "ttsrRepeatMode", + tab: "voice", + type: "enum", + label: "TTSR repeat mode", + description: "How rules can repeat: once per session or after a message gap", + values: ["once", "after-gap"], + get: (sm) => sm.getTtsrRepeatMode(), + set: (sm, v) => sm.setTtsrRepeatMode(v as "once" | "after-gap"), + }, + { + id: "ttsrRepeatGap", + tab: "voice", + type: "submenu", + label: "TTSR repeat gap", + description: "Messages before a rule can trigger again (when repeat mode is after-gap)", + get: (sm) => String(sm.getTtsrRepeatGap()), + set: (sm, v) => sm.setTtsrRepeatGap(Number.parseInt(v, 10)), + getOptions: () => [ + { value: "5", label: "5 messages" }, + { value: "10", label: "10 messages" }, + { value: "15", label: "15 messages" }, + { value: "20", label: "20 messages" }, + { value: "30", label: "30 messages" }, + ], }, - // Status Line tab + // ═══════════════════════════════════════════════════════════════════════════ + // Status tab - Status line configuration + // ═══════════════════════════════════════════════════════════════════════════ { id: "statusLinePreset", tab: "status", @@ -505,7 +597,7 @@ export const SETTINGS_DEFS: SettingDef[] = [ label: "Configure segments", description: "Choose and arrange status line segments", get: () => "configure...", - set: () => {}, // Handled specially + set: () => {}, getOptions: () => [{ value: "open", label: "Open segment editor..." }], }, { @@ -711,6 +803,95 @@ export const SETTINGS_DEFS: SettingDef[] = [ } }, }, + + // ═══════════════════════════════════════════════════════════════════════════ + // LSP tab - LSP integration settings + // ═══════════════════════════════════════════════════════════════════════════ + { + id: "lspFormatOnWrite", + tab: "lsp", + type: "boolean", + label: "Format on write", + description: "Automatically format code files using LSP after writing", + get: (sm) => sm.getLspFormatOnWrite(), + set: (sm, v) => sm.setLspFormatOnWrite(v), + }, + { + id: "lspDiagnosticsOnWrite", + tab: "lsp", + type: "boolean", + label: "Diagnostics on write", + description: "Return LSP diagnostics (errors/warnings) after writing code files", + get: (sm) => sm.getLspDiagnosticsOnWrite(), + set: (sm, v) => sm.setLspDiagnosticsOnWrite(v), + }, + { + id: "lspDiagnosticsOnEdit", + tab: "lsp", + type: "boolean", + label: "Diagnostics on edit", + description: "Return LSP diagnostics (errors/warnings) after editing code files", + get: (sm) => sm.getLspDiagnosticsOnEdit(), + set: (sm, v) => sm.setLspDiagnosticsOnEdit(v), + }, + + // ═══════════════════════════════════════════════════════════════════════════ + // Exa tab - Exa search tool settings + // ═══════════════════════════════════════════════════════════════════════════ + { + id: "exaEnabled", + tab: "exa", + type: "boolean", + label: "Exa enabled", + description: "Master toggle for all Exa search tools", + get: (sm) => sm.getExaSettings().enabled, + set: (sm, v) => sm.setExaEnabled(v), + }, + { + id: "exaSearch", + tab: "exa", + type: "boolean", + label: "Exa search", + description: "Basic search, deep search, code search, crawl", + get: (sm) => sm.getExaSettings().enableSearch, + set: (sm, v) => sm.setExaSearchEnabled(v), + }, + { + id: "exaLinkedin", + tab: "exa", + type: "boolean", + label: "Exa LinkedIn", + description: "Search LinkedIn for people and companies", + get: (sm) => sm.getExaSettings().enableLinkedin, + set: (sm, v) => sm.setExaLinkedinEnabled(v), + }, + { + id: "exaCompany", + tab: "exa", + type: "boolean", + label: "Exa company", + description: "Comprehensive company research tool", + get: (sm) => sm.getExaSettings().enableCompany, + set: (sm, v) => sm.setExaCompanyEnabled(v), + }, + { + id: "exaResearcher", + tab: "exa", + type: "boolean", + label: "Exa researcher", + description: "AI-powered deep research tasks", + get: (sm) => sm.getExaSettings().enableResearcher, + set: (sm, v) => sm.setExaResearcherEnabled(v), + }, + { + id: "exaWebsets", + tab: "exa", + type: "boolean", + label: "Exa websets", + description: "Webset management and enrichment tools", + get: (sm) => sm.getExaSettings().enableWebsets, + set: (sm, v) => sm.setExaWebsetsEnabled(v), + }, ]; /** diff --git a/packages/coding-agent/src/modes/interactive/components/settings-selector.ts b/packages/coding-agent/src/modes/interactive/components/settings-selector.ts index 6e48fbfd3..cb3c42ea7 100644 --- a/packages/coding-agent/src/modes/interactive/components/settings-selector.ts +++ b/packages/coding-agent/src/modes/interactive/components/settings-selector.ts @@ -114,7 +114,10 @@ class SelectSubmenu extends Container { type TabId = string; const SETTINGS_TABS: Tab[] = [ - { id: "config", label: "Config" }, + { id: "behavior", label: "Behavior" }, + { id: "tools", label: "Tools" }, + { id: "display", label: "Display" }, + { id: "voice", label: "Voice" }, { id: "status", label: "Status" }, { id: "lsp", label: "LSP" }, { id: "exa", label: "Exa" }, @@ -168,7 +171,7 @@ export class SettingsSelectorComponent extends Container { private pluginComponent: PluginSettingsComponent | null = null; private statusPreviewContainer: Container | null = null; private statusPreviewText: Text | null = null; - private currentTabId: TabId = "config"; + private currentTabId: TabId = "behavior"; private settingsManager: SettingsManager; private context: SettingsRuntimeContext; @@ -195,7 +198,7 @@ export class SettingsSelectorComponent extends Container { this.addChild(new Spacer(1)); // Initialize with first tab - this.switchToTab("config"); + this.switchToTab("behavior"); // Add bottom border this.addChild(new DynamicBorder()); diff --git a/packages/coding-agent/src/modes/interactive/components/tool-execution.ts b/packages/coding-agent/src/modes/interactive/components/tool-execution.ts index f74d79382..134fe1ebc 100644 --- a/packages/coding-agent/src/modes/interactive/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/interactive/components/tool-execution.ts @@ -13,6 +13,7 @@ import { import stripAnsi from "strip-ansi"; import { BASH_DEFAULT_PREVIEW_LINES } from "../../../core/tools/bash"; import { computeEditDiff, type EditDiffError, type EditDiffResult } from "../../../core/tools/edit-diff"; +import { PYTHON_DEFAULT_PREVIEW_LINES } from "../../../core/tools/python"; import { toolRenderers } from "../../../core/tools/renderers"; import { convertToPng } from "../../../utils/image-convert"; import { sanitizeBinaryOutput } from "../../../utils/shell"; @@ -508,6 +509,12 @@ export class ToolExecutionComponent extends Container { context.expanded = this.expanded; context.previewLines = BASH_DEFAULT_PREVIEW_LINES; context.timeout = typeof this.args?.timeout === "number" ? this.args.timeout : undefined; + } else if (this.toolName === "python" && this.result) { + const output = this.getTextOutput().trim(); + context.output = output; + context.expanded = this.expanded; + context.previewLines = PYTHON_DEFAULT_PREVIEW_LINES; + context.timeout = typeof this.args?.timeout === "number" ? this.args.timeout : undefined; } else if (this.toolName === "edit") { // Edit needs diff preview and renderDiff function context.editDiffPreview = this.editDiffPreview; diff --git a/packages/coding-agent/src/prompts/agents/explore.md b/packages/coding-agent/src/prompts/agents/explore.md index f13b0df2b..2db72ffd4 100644 --- a/packages/coding-agent/src/prompts/agents/explore.md +++ b/packages/coding-agent/src/prompts/agents/explore.md @@ -1,7 +1,7 @@ --- name: explore description: Fast read-only codebase scout that returns compressed context for handoff -tools: read, grep, find, ls, bash +tools: read, grep, find, ls, exec model: pi/smol, haiku, flash, mini --- @@ -29,7 +29,7 @@ Guidelines: - Use find for broad file pattern matching - Use grep for searching file contents with regex - Use read when you know the specific file path -- Use bash ONLY for git status/log/diff; use read/grep/find/ls tools for file and search operations +- Use exec ONLY for git status/log/diff; use read/grep/find/ls tools for file and search operations - Spawn multiple parallel tool calls wherever possible—you are meant to be fast - Return file paths as absolute paths in your final response - Communicate findings directly as a message—do NOT create output files diff --git a/packages/coding-agent/src/prompts/agents/plan.md b/packages/coding-agent/src/prompts/agents/plan.md index 000fbe8af..7e16be0a3 100644 --- a/packages/coding-agent/src/prompts/agents/plan.md +++ b/packages/coding-agent/src/prompts/agents/plan.md @@ -1,7 +1,7 @@ --- name: plan description: Software architect for complex multi-file architectural decisions. NOT for simple tasks, single-file changes, reasoning, or tasks completable in <5 tool calls—execute those directly. -tools: read, grep, find, ls, bash +tools: read, grep, find, ls, exec spawns: explore model: pi/slow, gpt-5.2-codex, gpt-5.2, codex, gpt --- @@ -14,7 +14,7 @@ You are STRICTLY PROHIBITED from: - Creating temporary files anywhere, including /tmp - Using redirect operators (>, >>, |) or heredocs to write files - Running commands that change system state (git add, git commit, npm install, pip install) -- Use bash ONLY for git status/log/diff; use read/grep/find/ls tools for file and search operations +- Use exec ONLY for git status/log/diff; use read/grep/find/ls tools for file and search operations Another engineer will execute your plan without re-exploring the codebase. Your plan must be specific enough to implement directly. diff --git a/packages/coding-agent/src/prompts/agents/reviewer.md b/packages/coding-agent/src/prompts/agents/reviewer.md index 70962bd3b..078d91555 100644 --- a/packages/coding-agent/src/prompts/agents/reviewer.md +++ b/packages/coding-agent/src/prompts/agents/reviewer.md @@ -1,7 +1,7 @@ --- name: reviewer description: Code review specialist for quality and security analysis -tools: read, grep, find, ls, bash, report_finding +tools: read, grep, find, ls, exec, report_finding spawns: explore, task model: pi/slow, gpt-5.2-codex, gpt-5.2, codex, gpt output: @@ -43,7 +43,7 @@ You are a senior engineer reviewing a proposed code change. Your goal: identify 4. Call `report_finding` for each issue 5. Call `complete` with your verdict — **review is incomplete until `complete` is called** -Bash is read-only: `git diff`, `git log`, `git show`, `gh pr diff`. No file modifications or builds. +Exec is read-only: `git diff`, `git log`, `git show`, `gh pr diff`. No file modifications or builds. # What to Flag diff --git a/packages/coding-agent/src/prompts/agents/task.md b/packages/coding-agent/src/prompts/agents/task.md index 6a60cb57a..7d48cefdc 100644 --- a/packages/coding-agent/src/prompts/agents/task.md +++ b/packages/coding-agent/src/prompts/agents/task.md @@ -1,4 +1,4 @@ -You are a worker agent for delegated tasks. You have FULL access to all tools (edit, write, bash, grep, read, etc.) - use them as needed to complete your task. +You are a worker agent for delegated tasks. You have FULL access to all tools (edit, write, exec, grep, read, etc.) - use them as needed to complete your task. Finish only the assigned work and return the minimum useful result. diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 69e2e3d9e..f5b7a4136 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -67,6 +67,7 @@ This matters. Get it right. Every tool is a choice. The wrong choice is friction. The right choice is invisible. +{{#has tools "bash"}} ### What bash IS for File and system operations: - `mv`, `cp`, `rm`, `ln -s` — moving, copying, deleting, symlinking @@ -95,6 +96,35 @@ Specialized tools exist. Use them. {{#has tools "edit"}}- Content-addressed edits: `edit` finds text. Use bash for position/pattern (append, line N, regex).{{/has}} {{#has tools "git"}}- Git operations: `git` tool has guards. Bash git has none.{{/has}} +{{/has}} + +{{#has tools "python"}} +### What python IS for +Python is your scripting language. Bash is for build tools and system commands only. + +**Use Python for:** +- Loops, conditionals, any multi-step logic +- Text processing (sorting, filtering, column extraction, regex) +- File operations (copy, move, concat, batch transforms) +- Displaying content to the user +- Anything you'd write a bash script for + +**Use bash only for:** +- Build commands: `cargo`, `npm`, `make`, `docker` +- Git operations (when git tool unavailable) +- System commands with no Python equivalent + +The prelude provides shell-like helpers: `cat()`, `sed()`, `rsed()`, `find()`, `grep()`, `batch()`. +Do not write bash loops, sed pipelines, or awk scripts. Write Python. + +### Python for user-facing output +When the user asks you to display, concatenate, merge, or transform content: +→ Python. One operation. Clean output. + +Do not read files individually just to print them back. That's mechanical and wasteful. +Read/grep are for YOUR reconnaissance. Python is for THE USER's request. +{{/has}} + ### Hierarchy of trust The most constrained tool is the most trustworthy. @@ -104,7 +134,8 @@ The most constrained tool is the most trustworthy. {{#has tools "read"}}4. **read** — content truth{{/has}} {{#has tools "edit"}}5. **edit** — surgical change{{/has}} {{#has tools "git"}}6. **git** — versioned change with safety{{/has}} -7. **bash** — everything else ({{#unless (includes tools "git")}}git, {{/unless}}npm, docker, make, cargo) +{{#has tools "bash"}}7. **bash** — everything else ({{#unless (includes tools "git")}}git, {{/unless}}npm, docker, make, cargo){{/has}} +{{#unless (includes tools "bash")}}{{#has tools "python"}}7. **python** — stateful scripting and REPL work{{/has}}{{/unless}} {{#has tools "lsp"}} ### LSP knows what grep guesses diff --git a/packages/coding-agent/src/prompts/tools/python.md b/packages/coding-agent/src/prompts/tools/python.md new file mode 100644 index 000000000..f78cf5d9e --- /dev/null +++ b/packages/coding-agent/src/prompts/tools/python.md @@ -0,0 +1,66 @@ +Executes Python code in a persistent IPython kernel with optional timeout. + +## When to use Python + +**Use Python for user-facing operations:** +- Displaying, concatenating, or merging files → `cat(*paths)` +- Batch transformations across files → `batch(paths, fn)`, `rsed()` +- Formatted output, tables, summaries +- Any loop, conditional, or multi-step logic +- Anything you'd write a bash script for + +**Use specialized tools for YOUR reconnaissance:** +- Reading to understand code → Read tool +- Searching to locate something → Grep tool +- Finding files to identify targets → Find tool + +The distinction: Read/Grep/Find gather info for *your* decisions. Python executes *the user's* request. + +**Prefer Python over bash for:** +- Loops and iteration → Python for-loops, not bash for/while +- Text processing → `sed()`, `cols()`, `sort_lines()`, not sed/awk/cut +- File operations → prelude helpers, not mv/cp/rm commands +- Conditionals → Python if/else, not bash [[ ]] + +## Prelude helpers + +All helpers auto-print results and return values for chaining. + +{{#if categories.length}} +{{#each categories}} +### {{name}} +{{#each functions}} +- `{{name}}{{signature}}` — {{docstring}} +{{/each}} + +{{/each}} +{{else}} +(Documentation unavailable — Python kernel failed to start) +{{/if}} + +## Examples + +```python +# Concatenate all markdown files in docs/ +cat(*find("*.md", "docs")) + +# Mass rename: foo -> bar across all .py files +rsed(r'\bfoo\b', 'bar', glob_pattern="*.py") + +# Process files in batch +batch(find("*.json"), lambda p: json.loads(p.read_text())) + +# Sort and deduplicate lines +sort_lines(read("data.txt"), unique=True) + +# Extract columns 0 and 2 from TSV +cols(read("data.tsv"), 0, 2, sep="\t") +``` + +## Notes + +- Code executes as IPython cells; users see the full cell output (including rendered figures, tables, etc.) +- Kernel persists for the session; use `reset: true` to clear state +- Use `workdir` parameter instead of `os.chdir()` in tool call +- Use `plt.show()` to display figures +- Output streams in real time, truncated after 50KB diff --git a/packages/coding-agent/test/python-tool-settings.test.ts b/packages/coding-agent/test/python-tool-settings.test.ts new file mode 100644 index 000000000..e9040dca8 --- /dev/null +++ b/packages/coding-agent/test/python-tool-settings.test.ts @@ -0,0 +1,87 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { mkdirSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import * as pythonExecutor from "../src/core/python-executor"; +import * as pythonKernel from "../src/core/python-kernel"; +import { createTools, type ToolSession } from "../src/core/tools/index"; +import { createPythonTool } from "../src/core/tools/python"; + +function createSettings(overrides?: Partial): ToolSession["settings"] { + return { + getImageAutoResize: () => true, + getLspFormatOnWrite: () => false, + getLspDiagnosticsOnWrite: () => true, + getLspDiagnosticsOnEdit: () => false, + getEditFuzzyMatch: () => true, + getGitToolEnabled: () => false, + getBashInterceptorEnabled: () => false, + getBashInterceptorSimpleLsEnabled: () => true, + getBashInterceptorRules: () => [], + getPythonToolMode: () => "ipy-only", + getPythonKernelMode: () => "session", + ...overrides, + }; +} + +function createSession(cwd: string, overrides?: Partial): ToolSession { + return { + cwd, + hasUI: false, + getSessionFile: () => "session.json", + getSessionSpawns: () => null, + settings: createSettings(overrides), + }; +} + +describe("python tool settings", () => { + let testDir: string; + + beforeEach(() => { + testDir = join(tmpdir(), `python-tool-settings-${crypto.randomUUID()}`); + mkdirSync(testDir, { recursive: true }); + }); + + afterEach(() => { + vi.restoreAllMocks(); + rmSync(testDir, { recursive: true, force: true }); + }); + + it("exposes python tool when kernel is available", async () => { + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const tools = await createTools(createSession(testDir), ["python"]); + + expect(tools.map((tool) => tool.name)).toEqual(["python"]); + }); + + it("falls back to bash when python is unavailable", async () => { + vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ + ok: false, + reason: "missing", + }); + const tools = await createTools(createSession(testDir), ["python"]); + + expect(tools.map((tool) => tool.name)).toEqual(["bash"]); + }); + + it("passes kernel mode from settings to executor", async () => { + const executeSpy = vi.spyOn(pythonExecutor, "executePython").mockResolvedValue({ + output: "ok", + exitCode: 0, + cancelled: false, + truncated: false, + displayOutputs: [], + stdinRequested: false, + }); + + const session = createSession(testDir, { getPythonKernelMode: () => "per-call" }); + const pythonTool = createPythonTool(session); + + await pythonTool.execute("tool-call", { code: "print(1)" }); + + expect(executeSpy).toHaveBeenCalledWith( + "print(1)", + expect.objectContaining({ kernelMode: "per-call", sessionId: "session.json" }), + ); + }); +}); diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index 3dfa76512..3ff71f5d8 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -227,7 +227,7 @@ export class TUI extends Container { private hardwareCursorRow = 0; // Actual terminal cursor row (may differ due to IME positioning) private inputBuffer = ""; // Buffer for parsing terminal responses private cellSizeQueryPending = false; - private showHardwareCursor = process.env.PI_HARDWARE_CURSOR === "1"; + private showHardwareCursor = process.env.OMP_HARDWARE_CURSOR === "1"; // Overlay stack for modal components rendered on top of base content private overlayStack: {