Merge remote-tracking branch 'upstream/main' into feat/secret-friendly-names

# Conflicts:
#	packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts
This commit is contained in:
Mathews-Tom
2026-06-23 17:12:13 +05:30
130 changed files with 4220 additions and 1958 deletions
Generated
+6 -6
View File
@@ -1769,9 +1769,9 @@ checksum = "88904434abc2901f197fe8cc55f0445e7ded921dba5911dad2e2b39b48e663c4"
[[package]]
name = "memmap2"
version = "0.9.10"
version = "0.9.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "714098028fe011992e1c3962653c96b2d578c4b4bce9036e15ff220319b1e0e3"
checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0"
dependencies = [
"libc",
]
@@ -2320,7 +2320,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "16.1.15"
version = "16.1.16"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -2390,7 +2390,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "16.1.15"
version = "16.1.16"
dependencies = [
"async-trait",
"libc",
@@ -2402,7 +2402,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "16.1.15"
version = "16.1.16"
dependencies = [
"anyhow",
"arboard",
@@ -2450,7 +2450,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "16.1.15"
version = "16.1.16"
dependencies = [
"anyhow",
"brush-builtins",
+1 -1
View File
@@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"]
resolver = "3"
[workspace.package]
version = "16.1.15"
version = "16.1.16"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
+27 -27
View File
@@ -21,7 +21,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "16.1.15",
"version": "16.1.16",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -39,7 +39,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "16.1.15",
"version": "16.1.16",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -55,7 +55,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "16.1.15",
"version": "16.1.16",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -69,7 +69,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "16.1.15",
"version": "16.1.16",
"bin": {
"omp": "src/cli.ts",
},
@@ -137,7 +137,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "16.1.15",
"version": "16.1.16",
"dependencies": {
"diff": "catalog:",
"lru-cache": "catalog:",
@@ -148,7 +148,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "16.1.15",
"version": "16.1.16",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -174,7 +174,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "16.1.15",
"version": "16.1.16",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -182,7 +182,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "16.1.15",
"version": "16.1.16",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
@@ -195,7 +195,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "16.1.15",
"version": "16.1.16",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -221,7 +221,7 @@
},
"packages/swarm-extension": {
"name": "@oh-my-pi/swarm-extension",
"version": "16.1.15",
"version": "16.1.16",
"bin": {
"omp-swarm": "src/cli.ts",
},
@@ -247,7 +247,7 @@
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "16.1.15",
"version": "16.1.16",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -288,7 +288,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "16.1.15",
"version": "16.1.16",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"handlebars": "catalog:",
@@ -301,7 +301,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "16.1.15",
"version": "16.1.16",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -338,18 +338,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "16.1.15",
"@oh-my-pi/omp-stats": "16.1.15",
"@oh-my-pi/pi-agent-core": "16.1.15",
"@oh-my-pi/pi-ai": "16.1.15",
"@oh-my-pi/pi-catalog": "16.1.15",
"@oh-my-pi/pi-coding-agent": "16.1.15",
"@oh-my-pi/pi-mnemopi": "16.1.15",
"@oh-my-pi/pi-natives": "16.1.15",
"@oh-my-pi/pi-tui": "16.1.15",
"@oh-my-pi/pi-utils": "16.1.15",
"@oh-my-pi/pi-wire": "16.1.15",
"@oh-my-pi/snapcompact": "16.1.15",
"@oh-my-pi/hashline": "16.1.16",
"@oh-my-pi/omp-stats": "16.1.16",
"@oh-my-pi/pi-agent-core": "16.1.16",
"@oh-my-pi/pi-ai": "16.1.16",
"@oh-my-pi/pi-catalog": "16.1.16",
"@oh-my-pi/pi-coding-agent": "16.1.16",
"@oh-my-pi/pi-mnemopi": "16.1.16",
"@oh-my-pi/pi-natives": "16.1.16",
"@oh-my-pi/pi-tui": "16.1.16",
"@oh-my-pi/pi-utils": "16.1.16",
"@oh-my-pi/pi-wire": "16.1.16",
"@oh-my-pi/snapcompact": "16.1.16",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -969,7 +969,7 @@
"chalk": ["chalk@5.6.2", "", {}, "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA=="],
"chardet": ["chardet@2.1.1", "", {}, "sha512-PsezH1rqdV9VvyNhxxOW32/d75r01NY7TQCmOqomRo15ZSOKbpTFVsfjghxo6JloQUCGnH4k1LGu0R4yCLlWQQ=="],
"chardet": ["chardet@2.2.0", "", {}, "sha512-rddelWYNPRrXq6PtNEN2S3f6t9ILzvqaN5pVgi4kqt9jHQaXIial9PznB5iSPVlQSLNaaH22ItWz3EJtQ10+OA=="],
"chart.js": ["chart.js@4.5.1", "", { "dependencies": { "@kurkle/color": "^0.3.0" } }, "sha512-GIjfiT9dbmHRiYi6Nl2yFCq7kkwdkp1W/lp2J99rX0yo9tgJGn3lKQATztIjb5tVtevcBtIdICNWqlq5+E8/Pw=="],
@@ -1323,7 +1323,7 @@
"scheduler": ["scheduler@0.27.0", "", {}, "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q=="],
"semver": ["semver@7.8.4", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-rUCObTnP32Q08R2uuIrt7r9PlEonuTmtuXYcW6s5kjdlj3xbnwe+21yXptAUYcMAABLkYYTtnmzb3w3EDZfueA=="],
"semver": ["semver@7.8.5", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-Y7/KDsb8LjooZpwaqGyulO6DQlksgCncchHGk+sZIY4SBvUocMBEFH5Ur1fI4dV+Jvl0w6cjvucaIi40puRioA=="],
"semver-compare": ["semver-compare@1.0.0", "", {}, "sha512-YM3/ITh2MJ5MtzaM429anh+x2jiLVjqILF4m4oyQB18W7Ggea7BfqdH/wGMK7dDiMghv/6WG7znWMwUDzJiXow=="],
+7 -1
View File
@@ -632,7 +632,13 @@ pub(crate) fn execute_external_command(
match session_action {
ChildSessionAction::DetachSession => {
// setsid() creates the fresh session + process group; no process_group().
cmd.detach_session();
// A reparenting operand (`nohup cmd &`) additionally double-forks so it
// leaves the host's descendant tree and survives the teardown walk.
if context.params.detach_reparent {
cmd.detach_session_reparent();
} else {
cmd.detach_session();
}
}
ChildSessionAction::TakeForeground if command_leads_session => {
// Don't set process_group(0) - setsid() in pre_exec will handle it.
+20 -4
View File
@@ -73,6 +73,11 @@ pub struct ExecutionParameters {
open_files: openfiles::OpenFiles,
/// Policy for how to manage spawned external processes.
pub process_group_policy: ProcessGroupPolicy,
/// Whether external commands spawned in this context should reparent out of
/// the shell's descendant tree (double-fork on Unix) so they survive the
/// host's descendant-walk teardown. Set for the operand of a transparent
/// background wrapper such as `nohup cmd &`.
pub detach_reparent: bool,
/// Optional cancellation token shared with callers.
cancel_token: Option<CancellationToken>,
/// Optional command-output marker hook.
@@ -382,7 +387,12 @@ async fn spawn_async_ao_list_as_job<'a, SE: extensions::ShellExtensions>(
let direct_pipeline =
background_process_pipeline_for_async_job(ao_list, shell, &async_params).await?;
let job = if let Some(pipeline) = direct_pipeline {
let job = if let Some((pipeline, detach_reparent)) = direct_pipeline {
// A transparent background wrapper (e.g. `nohup cmd &`) was unwrapped to its
// operand. Reparent that operand out of the shell's descendant tree so it
// survives the host's descendant-walk teardown — the persistence agents
// reach for `nohup` expecting.
async_params.detach_reparent = detach_reparent;
match try_spawn_pipeline_as_job(&pipeline, ao_list.to_string(), shell, &async_params).await? {
Some(job) => job,
None => spawn_async_ao_list_in_task(ao_list, shell, &async_params),
@@ -404,16 +414,22 @@ async fn background_process_pipeline_for_async_job<SE: extensions::ShellExtensio
ao_list: &ast::AndOrList,
shell: &mut Shell<SE>,
params: &ExecutionParameters,
) -> Result<Option<ast::Pipeline>, error::Error> {
) -> Result<Option<(ast::Pipeline, bool)>, error::Error> {
if !ao_list.additional.is_empty() {
return Ok(None);
}
let mut pipeline = ao_list.first.clone();
let mut detach_reparent = false;
for _ in 0..8 {
match classify_background_process_pipeline(&pipeline, shell, params).await? {
BackgroundProcessPipeline::Direct => return Ok(Some(pipeline)),
BackgroundProcessPipeline::Wrapper(unwrapped) => pipeline = unwrapped,
BackgroundProcessPipeline::Direct => return Ok(Some((pipeline, detach_reparent))),
BackgroundProcessPipeline::Wrapper(unwrapped) => {
// Unwrapping a transparent background wrapper (`nohup`) means the
// operand should reparent away from the shell when finally spawned.
detach_reparent = true;
pipeline = unwrapped;
},
BackgroundProcessPipeline::Internal => return Ok(None),
}
}
@@ -98,10 +98,17 @@ impl CommandFgControlExt for std::process::Command {
pub trait CommandSessionExt {
/// Arranges for the command to run in a new session with no controlling terminal.
fn detach_session(&mut self);
/// Like [`CommandSessionExt::detach_session`]. No-op on platforms without
/// `setsid`/`fork` reparenting.
fn detach_session_reparent(&mut self);
}
impl CommandSessionExt for std::process::Command {
fn detach_session(&mut self) {
// NOTE: This is a no-op on platforms without setsid support.
}
fn detach_session_reparent(&mut self) {
// NOTE: This is a no-op on platforms without setsid/fork support.
}
}
@@ -73,6 +73,10 @@ impl CommandFgControlExt for std::process::Command {
pub trait CommandSessionExt {
/// Arranges for the command to run in a new POSIX session with no controlling terminal.
fn detach_session(&mut self);
/// Like [`CommandSessionExt::detach_session`], but additionally double-forks
/// so the spawned process reparents to init (PID 1) and leaves the caller's
/// descendant tree.
fn detach_session_reparent(&mut self);
}
impl CommandSessionExt for std::process::Command {
@@ -84,6 +88,15 @@ impl CommandSessionExt for std::process::Command {
self.pre_exec(pre_exec_detach_session);
}
}
fn detach_session_reparent(&mut self) {
// SAFETY:
// This arranges for a provided function to run in the forked child before
// exec. Only async-signal-safe calls (`setsid`, `fork`, `_exit`) are used.
unsafe {
self.pre_exec(pre_exec_detach_session_reparent);
}
}
}
fn pre_exec_take_foreground() -> Result<(), std::io::Error> {
@@ -119,3 +132,31 @@ fn pre_exec_detach_session() -> Result<(), std::io::Error> {
Err(errno) => Err(std::io::Error::from_raw_os_error(errno as i32)),
}
}
fn pre_exec_detach_session_reparent() -> Result<(), std::io::Error> {
// New session first: drop any controlling terminal. Ignore EPERM, which means
// the child is already a session leader from an outer policy.
match nix::unistd::setsid() {
Ok(_) | Err(nix::errno::Errno::EPERM) => {},
Err(errno) => return Err(std::io::Error::from_raw_os_error(errno as i32)),
}
// Double-fork: the intermediate child — the pid the parent's spawn machinery
// tracks — exits immediately, so the grandchild that goes on to `exec` the
// operand reparents to init (PID 1) and is no longer a descendant of the
// shell. This is what lets `nohup cmd &` survive the host's descendant-walk
// teardown without relying on an external `setsid(1)` binary.
//
// SAFETY: the post-`fork` child here is single-threaded, and only
// async-signal-safe primitives (`fork`, `_exit`) run before `exec`.
let pid = unsafe { libc::fork() };
if pid < 0 {
return Err(std::io::Error::last_os_error());
}
if pid > 0 {
// Intermediate parent: exit now to orphan the grandchild. `_exit` avoids
// running atexit handlers or flushing inherited buffers in the fork.
unsafe { libc::_exit(0) };
}
Ok(())
}
@@ -114,10 +114,18 @@ pub trait CommandSessionExt {
/// terminal. On Windows this is a no-op; process-group and console behavior
/// are handled uniformly by `sys::process::spawn`.
fn detach_session(&mut self);
/// Like [`CommandSessionExt::detach_session`]. On Windows there is no session
/// or `fork`-based reparenting, so this is a no-op: the operand stays a child
/// of the shell.
fn detach_session_reparent(&mut self);
}
impl CommandSessionExt for std::process::Command {
fn detach_session(&mut self) {
// NOTE: Windows has no setsid; intentionally a no-op.
}
fn detach_session_reparent(&mut self) {
// NOTE: no reparenting primitive on Windows; intentionally a no-op.
}
}
+1 -1
View File
@@ -172,7 +172,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV16_1_15")]
#[napi(js_name = "__piNativesV16_1_16")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
+35 -19
View File
@@ -519,10 +519,6 @@ async fn create_session(config: &ShellConfig) -> Result<ShellSessionCore> {
}
shell.register_builtin("sleep", builtins::builtin::<SleepCommand, _>());
shell.register_builtin("timeout", builtins::builtin::<TimeoutCommand, _>());
shell.register_builtin(
"nohup",
builtins::builtin::<NohupCommand, _>().transparent_background_wrapper(),
);
let mut merged_path: Option<String> = None;
for (key, value) in std::env::vars() {
@@ -552,8 +548,8 @@ async fn create_session(config: &ShellConfig) -> Result<ShellSessionCore> {
merged_path = Some(value.to_string_lossy().into_owned());
}
if let Some(path_value) = merged_path {
let mut var = ShellVariable::new(ShellValue::String(path_value));
if let Some(path_value) = &merged_path {
let mut var = ShellVariable::new(ShellValue::String(path_value.clone()));
var.export();
shell
.env_mut()
@@ -576,6 +572,27 @@ async fn create_session(config: &ShellConfig) -> Result<ShellSessionCore> {
}
}
apply_env_fallback(&mut shell)?;
// The nohup builtin detaches its operand into a new session (see
// NohupCommand) so a backgrounded server survives this embedded shell's
// kill-on-drop teardown. It therefore shadows any system `nohup` (which does
// NOT escape the process-group kill) — unless explicitly opted out via
// PI_DISABLE_NOHUP_BUILTIN (session env or process env), in which case bare
// `nohup` resolves to the real coreutils binary.
let nohup_builtin_disabled = {
let raw = config
.session_env
.as_ref()
.and_then(|env| env.get("PI_DISABLE_NOHUP_BUILTIN").cloned())
.or_else(|| std::env::var("PI_DISABLE_NOHUP_BUILTIN").ok());
matches!(raw.as_deref(), Some(v) if !v.is_empty() && v != "0" && !v.eq_ignore_ascii_case("false"))
};
let should_register_nohup = !nohup_builtin_disabled;
if should_register_nohup {
shell.register_builtin(
"nohup",
builtins::builtin::<NohupCommand, _>().transparent_background_wrapper(),
);
}
#[cfg(windows)]
configure_windows_path(&mut shell)?;
@@ -1916,18 +1933,16 @@ impl builtins::Command for NohupCommand {
return Ok(ExecutionResult::new(125));
}
// Deliberately *not* nohup: we neither ignore SIGHUP nor detach the
// child into a new session. The command runs as an ordinary brush
// descendant so it is reaped together with the host instead of
// lingering as an orphan once the host process goes away. Agents
// reach for `nohup` assuming the shell is one-shot; in this
// persistent embedded shell that assumption is wrong and the only
// effect of real `nohup` would be to leak background processes.
//
// coreutils `nohup` additionally redirects stdin from /dev/null and
// stdout/stderr to `nohup.out`, but *only* when those streams are
// terminals. The embedded host always hands commands a pipe with a
// /dev/null stdin, so none of that redirection ever applies here.
// `nohup <cmd>` (foreground) runs the operand directly and surfaces its
// exit status — the contract pinned by
// `nohup_builtin_propagates_command_exit_code`. Persistence across the
// host's teardown is a *background* concern that never reaches this
// builtin: the agent writes `nohup <server> &`, and brush's
// `transparent_background_wrapper` unwraps that to spawn the operand
// directly with `detach_reparent`, double-forking it out of the shell's
// descendant tree (see `execute_external_command` / `detach_session_reparent`).
// Like coreutils, we run the operand here; we only differ by not masking
// SIGHUP (see `nohup_builtin_does_not_mask_sighup`).
let mut command_line = String::new();
for (idx, arg) in command.iter().enumerate() {
if idx > 0 {
@@ -1936,7 +1951,8 @@ impl builtins::Command for NohupCommand {
command_line.push_str(&quote_arg(arg));
}
let params = context.params.clone();
let mut params = context.params.clone();
params.process_group_policy = ProcessGroupPolicy::NewProcessGroup;
let source_info = SourceInfo::from("pi-natives:nohup");
context
.shell
+1 -1
View File
@@ -160,7 +160,7 @@ The runner additionally receives `PYTHONUNBUFFERED=1` and `PYTHONIOENCODING=utf-
If Python preflight fails and `eval.js` is enabled, `eval` remains available for `js` cells; `py` cells fail with a Python-backend availability error.
Python prelude helpers include `agent(prompt, *, agent_type="task", model=None, label=None, schema=None, return_handle=False)`. It synchronously calls the host bridge, runs one subagent through the task executor, and returns the final text. When `schema` is supplied, the helper parses the subagent's JSON output and returns the object. When `return_handle=True`, it instead returns a DAG node dict (`{"text", "output", "handle", "id", "agent"}`) whose `handle` is the spawned agent's recoverable `agent://<id>` URI (the parsed object lands under `"data"` when `schema` is also set), so a downstream `pipeline`/`parallel` stage can reference the transcript by handle instead of re-inlining it.
Python prelude helpers include `agent(prompt, *, agent="task", model=None, label=None, schema=None, handle=False)`. It synchronously calls the host bridge, runs one subagent through the task executor, and returns the final text. When `schema` is supplied, the helper parses the subagent's JSON output and returns the object. When `handle=True`, it instead returns a DAG node dict (`{"text", "output", "handle", "id", "agent"}`) whose `handle` is the spawned agent's recoverable `agent://<id>` URI (the parsed object lands under `"data"` when `schema` is also set), so a downstream `pipeline`/`parallel` stage can reference the transcript by handle instead of re-inlining it.
## Execution flow and cancellation/timeout
+9 -11
View File
@@ -129,17 +129,15 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod
- Module cache is busted for **local** imports between cells so edits to source files are picked up without restarting the runtime. `__omp_import__` deletes `require.cache[absPath]` before re-importing whenever the original specifier is a filesystem path: relative (`./x`, `../x`, `.`, `..`), POSIX-absolute (`/...`), home-prefixed (`~/...`), or Windows drive-letter (`C:\...` / `C:/...`). Bare specifiers (`react`, `lodash/x`) and URL/scheme specifiers (`node:fs`, `file://...`, `https://...`) are left in cache so package identity stays stable across cells. The cache-bust only fires when the resolved target is an absolute path — unresolved bare-package fallbacks (`resolveImportSpecifier()` returning the original specifier) skip it.
- The prelude installs globals:
- `display`, `print`, and a `console` bridge
- `read`, `write`, `append`, `sort`, `uniq`, `counter`, `diff`, `tree`, `env`, `output`
- `read`, `write`, `env`, `output`
- `tool.<name>(args)` proxy for arbitrary session tool calls
- `completion(prompt, opts?)` for oneshot, stateless model calls (see _Oneshot completion helper_ below)
- `agent(prompt, opts?)` for a single subagent call, plus `parallel()` / `pipeline()` bounded-pool helpers (see _Subagent helper_ below)
- `log(message)`, `phase(title)`, and `budget` (live token-budget view via async `budget.total()` / `budget.spent()` / `budget.remaining()` / `budget.hard()`)
- JS helpers that touch the host/runtime boundary are async and `await`able; pure text helpers (`sort`, `uniq`, `counter`) return synchronously but may still be safely awaited.
- JS host/runtime helpers (`read`, `write`, `output`) are async and `await`able; `env` returns synchronously.
- JS helper options may be passed either positionally in the Python order or as a trailing options object. `null` and `undefined` skip positional slots:
- `await read(path, offset?, limit?)` or `await read(path, { offset?, limit? })`
- `await tree(path = ".", maxDepth?, showHidden?)` or `await tree(path, { maxDepth?, showHidden? })`
- `sort(text, reverse?, unique?)`, `uniq(text, count?)`, `counter(items, limit?, reverse?)`
- `await agent(prompt, agentType?, model?, label?, schema?)` or `await agent(prompt, { agentType?, model?, label?, schema?, returnHandle? })`
- `await agent(prompt, agent?, model?, label?, schema?)` or `await agent(prompt, { agent?, model?, label?, schema?, handle? })`
- `await parallel([() => agent("a"), () => agent("b")])`
- `await pipeline(items, stage1, stage2)`
- `display(value)` behavior:
@@ -193,13 +191,13 @@ Both runtimes expose `completion()` — a single stateless completion against a
Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state.
- Signatures:
- JS: `await agent(prompt, agentType?, model?, label?, schema?)` or `await agent(prompt, { agentType?, model?, label?, schema?, returnHandle? })`
- Python: `agent(prompt, *, agent_type="task", model=None, label=None, schema=None, return_handle=False)`
- `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work.
- JS: `await agent(prompt, agent?, model?, label?, schema?)` or `await agent(prompt, { agent?, model?, label?, schema?, handle? })`
- Python: `agent(prompt, *, agent="task", model=None, label=None, schema=None, handle=False)`
- `agent` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work.
- `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply.
- Shared background is passed via files: write a `local://` file and reference it in the prompt. `label` controls the `agent://<id>` output label prefix.
- `schema` passes a JSON Schema to the subagent structured-output path. When present, the helper parses the final JSON text and returns an object.
- `returnHandle` / `return_handle` (default off) returns a DAG node dict — `{ text, output, handle: "agent://<id>", id, agent }`, plus a parsed `data` field when `schema` is set — instead of the bare output, so a downstream stage can reference the transcript by handle.
- `handle` (default off) returns a DAG node dict — `{ text, output, handle: "agent://<id>", id, agent }`, plus a parsed `data` field when `schema` is set — instead of the bare output, so a downstream stage can reference the transcript by handle.
- Spawn restrictions use `session.getSessionSpawns()` exactly like the `task` tool. Eval-driven subagent recursion is capped at depth 3.
- JS and Python both expose `parallel(thunks)` and `pipeline(items, ...stages)`; both use a bounded async/threaded pool whose width tracks the `task.maxConcurrency` setting (the same ceiling the `task` tool uses; `0` = run every item at once), preserve item order, and propagate rejections. The width is fetched live from the host via the `__concurrency__` bridge, so the helpers no longer take a `concurrency` argument.
- Errors surface as exceptions: unknown or disabled agent, disallowed spawn, recursion cap, subagent failure, or invalid structured output all fail the eval cell.
@@ -215,7 +213,7 @@ A single tool call can mix Python and JS cells. Persistence is per language runt
## Side Effects
- Filesystem
- JS/Python prelude helpers can read, write, append, diff, and traverse filesystem paths under the session cwd or absolute paths.
- JS/Python prelude helpers can read and write filesystem paths under the session cwd or absolute paths.
- JS helper `read()` auto-delegates any non-`local://` scheme URI (`agent://`, `artifact://`, `https://`, ...) to `tool.read(...)` (honoring an `offset`/`limit` line selector), resolves `local://` under its mapped root, reads plain/absolute filesystem paths directly, and rejects directory paths.
- Output may spill to an artifact file via `OutputSink`.
- Network
@@ -280,7 +278,7 @@ A single tool call can mix Python and JS cells. Persistence is per language runt
- Backend selection is strictly explicit per cell: `language` must be `"py"` or `"js"`. The previous `*** Cell` header parser, the `eval.lark` constrained grammar, and the sniffer-based fallback have all been removed.
- `EvalTool.customFormat` no longer exists. Tool calls flow through the standard JSON schema; there is no Lark-constrained sampling path.
- `tool.<name>()` exists in both JS and Python. Python calls route through a per-run loopback bridge keyed by the current cell id.
- `read()` delegates non-`local://` scheme URIs to `tool.read`, resolves `local://` under its injected root, and resolves plain paths against the session cwd or an absolute filesystem path; `resolveRegularFile()` rejects directory paths. `write()`/`append()` accept `local://` and plain paths but reject any other `scheme://` via `resolveHelperPath()` (`Protocol paths are not supported by write()`).
- `read()` delegates non-`local://` scheme URIs to `tool.read`, resolves `local://` under its injected root, and resolves plain paths against the session cwd or an absolute filesystem path; `resolveRegularFile()` rejects directory paths. `write()` accepts `local://` and plain paths but rejects any other `scheme://` via `resolveHelperPath()` (`Protocol paths are not supported by write()`).
- Python helper `output(...)` depends on `PI_ARTIFACTS_DIR` or `PI_SESSION_FILE`; it fails outside a session-backed run.
- `display()` can produce text and structured outputs from the same value; the renderer prefers markdown over `text/plain` when both exist.
- JS static imports are rewritten only at top level. Nested imports stay invalid and surface normal JS syntax/runtime errors.
+17 -21
View File
@@ -1,6 +1,6 @@
# todo
> Applies ordered mutations to the session todo list and returns a text summary plus the full phase/task state.
> Applies one mutation to the session todo list and returns a text summary plus the full phase/task state.
## Source
- Entry: `packages/coding-agent/src/tools/todo.ts`
@@ -14,11 +14,7 @@
## Inputs
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `ops` | `TodoOpEntry[]` | Yes | Ordered operations to apply. `minItems: 1`.
### `TodoOpEntry`
The params object **is** a single op — the discriminator and its fields live at the top level (no `ops` array wrapper).
| Op | Required fields | Optional fields | Effect |
| --- | --- | --- | --- |
@@ -28,9 +24,9 @@
| `drop` | `task` or `phase` or neither | None | Marks the target task, phase, or all tasks `abandoned`. |
| `rm` | `task` or `phase` or neither | None | Removes the target task, clears the phase's task list, or clears all task lists. |
| `append` | `phase`, `items` | None | Appends new `pending` tasks to a phase; creates the phase if missing. |
| `view` | None | None | Echoes the current list. A call whose ops are all `view` is read-only: no normalization, no state write. |
| `view` | None | None | Echoes the current list. A `view` call is read-only: no normalization, no state write. |
### Fields used inside ops
### Fields
| Field | Type | Required | Description |
| --- | --- | --- | --- |
@@ -46,11 +42,11 @@ The tool returns a single-shot `AgentToolResult`:
- `content`: one text part containing the summary from `formatSummary(...)`.
- Empty final state with no errors: `Todo list cleared.` (`Todo list is empty.` for a pure-`view` call).
- Non-empty final state: remaining-item list, current phase progress, then a per-phase tree.
- If any op produced validation/runtime errors, the summary starts with `Errors: ...` and the result is marked `isError: true`; the whole batch is discarded — the returned and persisted state stay at the pre-call list.
- If the op produced validation/runtime errors, the summary starts with `Errors: ...` and the result is marked `isError: true`; the mutation is discarded — the returned and persisted state stay at the pre-call list.
- `details`:
- `phases: TodoPhase[]`
- `storage: "session" | "memory"`
- `completedTasks?: TodoCompletionTransition[]` when a task changed from non-completed to `completed` during the batch
- `completedTasks?: TodoCompletionTransition[]` when a task changed from non-completed to `completed` during the call
`TodoPhase` / `TodoItem` state model:
@@ -61,19 +57,19 @@ The TUI renderer (`todoToolRenderer`) merges call and result into one transcript
## Flow
1. `TodoTool.execute(...)` clones the current cached phases from `session.getTodoPhases?.() ?? []` (`packages/coding-agent/src/tools/todo.ts`).
2. `applyParams(...)` walks `params.ops` in order and applies each entry with `applyEntry(...)`.
2. `applyParams(...)` applies the single op (`params`) with `applyEntry(...)`.
3. Each op mutates the working phase array:
- `initPhases(...)` rebuilds the list from scratch.
- `start` resolves a task by exact `content`, demotes every other `in_progress` task to `pending`, then marks the target `in_progress`.
- `done` / `drop` use `getTaskTargets(...)` to target one task, one phase, or every task.
- `rm` removes one task, clears one phase's `tasks`, or clears all phases' task arrays.
- `appendItems(...)` resolves or creates the target phase and pushes new `pending` tasks unless the same task content already exists anywhere.
4. Missing task/phase references are recorded in an `errors` array by `resolveTaskOrError(...)` / `resolvePhaseOrError(...)`; execution continues through the rest of the batch, but any error discards the batch's mutations at the end.
5. After the full batch, `normalizeInProgressTask(...)` enforces the single-active-task invariant:
4. Missing task/phase references are recorded in an `errors` array by `resolveTaskOrError(...)` / `resolvePhaseOrError(...)`; any error discards the op's mutations at the end.
5. After the op, `normalizeInProgressTask(...)` enforces the single-active-task invariant:
- if multiple tasks are `in_progress`, only the first stays active and the rest become `pending`;
- if none are `in_progress`, the first `pending` task in phase/task order is auto-promoted to `in_progress`.
6. `execute(...)` stores the updated phases with `session.setTodoPhases?.(...)` only when the batch produced no errors and was not pure-`view`; a failed batch is discarded wholesale (persisting a half-applied batch would make the natural retry hit "already exists"). `storage` is `"session"` when `session.getSessionFile()` exists, else `"memory"`.
7. `getCompletionTransitions(...)` compares the previous and updated phases (skipped for failed or pure-`view` calls); newly completed tasks are returned in `details.completedTasks`.
6. `execute(...)` stores the updated phases with `session.setTodoPhases?.(...)` only when the op produced no errors and was not a `view`; a failed op is discarded (persisting a half-applied mutation would make the natural retry hit "already exists"). `storage` is `"session"` when `session.getSessionFile()` exists, else `"memory"`.
7. `getCompletionTransitions(...)` compares the previous and updated phases (skipped for failed or `view` calls); newly completed tasks are returned in `details.completedTasks`.
8. The agent runtime also watches `todo` tool results in `packages/coding-agent/src/session/agent-session.ts`; successful results refresh cached todos, failed results inject a hidden next-turn reminder telling the model that todo progress is not visible until it retries.
9. The event controller updates the visible todo UI from `result.details.phases` on success, or shows a warning on error (`packages/coding-agent/src/modes/controllers/event-controller.ts`).
@@ -87,7 +83,7 @@ The TUI renderer (`todoToolRenderer`) merges call and result into one transcript
| `completed` | Can be set back to `in_progress` if targeted | Stays `completed` | Becomes `abandoned` if targeted | Removed | No status change |
| `abandoned` | Can be set back to `in_progress` if targeted | Becomes `completed` if targeted | Stays `abandoned` | Removed | No status change |
Normalization then re-applies the single-active-task rule after the full op batch.
Normalization then re-applies the single-active-task rule after the op runs.
### Op targeting rules
- `done`, `drop`, `rm`:
@@ -118,7 +114,7 @@ The same file also exposes non-tool helpers used by `/todo`:
- Session-level auto-clear of `completed`/`abandoned` tasks was removed (the timer mutated canonical phases between tool calls); the TUI todo widget still clears closed entries after `tasks.todoClearDelay` (display-only, `packages/coding-agent/src/modes/interactive-mode.ts`).
## Limits & Caps
- `ops` array: `minItems: 1` (`todoSchema`).
- `init.list`: applies to a single op (`todoSchema`). The params object carries exactly one op.
- `init.list[*].items`: `minItems: 1`.
- `append.items`: `minItems: 1`.
- Renderer collapsed preview: `PREVIEW_LIMITS.COLLAPSED_ITEMS = 8` (`packages/coding-agent/src/tools/render-utils.ts`).
@@ -126,7 +122,7 @@ The same file also exposes non-tool helpers used by `/todo`:
- Tool execution mode: `concurrency = "exclusive"`, `strict = true`, `loadMode = "discoverable"`.
## Errors
- Ordinary bad op payloads are accumulated as human-readable strings in `errors`; the result is marked `isError: true` and the whole batch is discarded — the returned and persisted state stay at the pre-call list.
- Ordinary bad op payloads are accumulated as human-readable strings in `errors`; the result is marked `isError: true` and the mutation is discarded — the returned and persisted state stay at the pre-call list.
- Error strings come from the helpers in `packages/coding-agent/src/tools/todo.ts`, including:
- `Missing list for init operation`
- `Missing task content`
@@ -137,18 +133,18 @@ The same file also exposes non-tool helpers used by `/todo`:
- `Missing phase name for append operation`
- `Missing items for append operation`
- `Task "..." already exists`
- Ops are processed in order and an early error does not stop later ops from being attempted, but any error in the batch discards every mutation the batch made.
- A `todo` call carries a single op; any error in it discards every mutation the op made.
- Runtime-level tool failure is handled outside the tool body: `agent-session` injects a hidden reminder and the event controller warns the user that visible progress may be stale.
- Idempotency is op-specific:
- `init` is a full replacement; replaying the same payload yields the same state.
- `start`, `done`, and `drop` are effectively idempotent on an existing target state, but `start` also demotes any other active task.
- `rm` is not idempotent for targeted removals: the second call errors because the task or phase is gone.
- `append` is not idempotent: duplicate task content is rejected with `Task "..." already exists`; the whole `append` op validates up front, so a batch with any duplicate appends nothing.
- `append` is not idempotent: duplicate task content is rejected with `Task "..." already exists`; the `append` op validates up front, so an op with any duplicate appends nothing.
## Notes
- Task lookup is exact string equality inside the tool. The model-facing prompt says task content and phase names are identifiers and should stay unique; `append` enforces task uniqueness globally, and `init` rejects duplicate phase names and duplicate task contents in its payload.
- `findTaskByContent(...)` returns the first matching task across phases. Duplicate task contents make later targeted ops ambiguous.
- `normalizeInProgressTask(...)` runs after the whole batch, not after each op. A single call can intentionally build an intermediate invalid state and rely on final normalization.
- `normalizeInProgressTask(...)` runs once after the op, not mid-op. A single op (e.g. `init`) can build an intermediate invalid state and rely on final normalization.
- `storage: "session"` means the session has a session-file backing; it does not mean this tool wrote a durable custom entry.
- Reload persistence differs by path:
- plain `todo` calls survive in transcript tool-result details;
+12 -12
View File
@@ -25,18 +25,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "16.1.15",
"@oh-my-pi/omp-stats": "16.1.15",
"@oh-my-pi/pi-agent-core": "16.1.15",
"@oh-my-pi/pi-ai": "16.1.15",
"@oh-my-pi/pi-catalog": "16.1.15",
"@oh-my-pi/pi-coding-agent": "16.1.15",
"@oh-my-pi/pi-mnemopi": "16.1.15",
"@oh-my-pi/pi-natives": "16.1.15",
"@oh-my-pi/pi-tui": "16.1.15",
"@oh-my-pi/pi-utils": "16.1.15",
"@oh-my-pi/pi-wire": "16.1.15",
"@oh-my-pi/snapcompact": "16.1.15",
"@oh-my-pi/hashline": "16.1.16",
"@oh-my-pi/omp-stats": "16.1.16",
"@oh-my-pi/pi-agent-core": "16.1.16",
"@oh-my-pi/pi-ai": "16.1.16",
"@oh-my-pi/pi-catalog": "16.1.16",
"@oh-my-pi/pi-coding-agent": "16.1.16",
"@oh-my-pi/pi-mnemopi": "16.1.16",
"@oh-my-pi/pi-natives": "16.1.16",
"@oh-my-pi/pi-tui": "16.1.16",
"@oh-my-pi/pi-utils": "16.1.16",
"@oh-my-pi/pi-wire": "16.1.16",
"@oh-my-pi/snapcompact": "16.1.16",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
+4 -1
View File
@@ -1,6 +1,9 @@
# Changelog
## [Unreleased]
## [16.1.16] - 2026-06-23
### Added
- Added `generateHandoffFromContext(context, model, options)` to `@oh-my-pi/pi-agent-core/compaction`: runs the handoff oneshot against a fully-built provider `Context` (system prompt, normalized tools, transformed history, trailing handoff prompt) with `streamOptions` mirroring the live turn's cache routing, so a host that owns the transform pipeline can make the handoff request share the prompt cache the main turn populated. `generateHandoff(messages, …)` is unchanged and now delegates to it.
@@ -930,4 +933,4 @@ Initial release under @oh-my-pi scope. See previous releases at [badlogic/pi-mon
### Changed
- `Agent` constructor now has all options optional (empty options use defaults).
- `queueMessage()` is now synchronous (no longer returns a Promise).
- `queueMessage()` is now synchronous (no longer returns a Promise).
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "16.1.15",
"version": "16.1.16",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+7
View File
@@ -2,6 +2,13 @@
## [Unreleased]
## [16.1.16] - 2026-06-23
### Fixed
- Fixed Anthropic-compatible thinking requests sending replayed thinking blocks without `context_management.keep: "all"`, preserving multi-turn reasoning context for API-key providers. API-key requests now also advertise the required `context-management-2025-06-27` beta header so the field is honored instead of rejected. Injected SDK clients, GitHub Copilot's Anthropic proxy, and Vertex rawPredict are excluded because this code path cannot add the beta to caller-owned clients, Copilot strips Anthropic betas and demotes thinking blocks to text upstream, and Vertex expects betas in the JSON body rather than the Anthropic HTTP beta header. ([#3288](https://github.com/can1357/oh-my-pi/issues/3288))
- Fixed OpenRouter Responses native history replay leaking Gemini reasoning item `format` metadata back into follow-up requests, which caused HTTP 400 rejections while preserving encrypted reasoning replay.
## [16.1.15] - 2026-06-22
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "16.1.15",
"version": "16.1.16",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+41 -7
View File
@@ -120,10 +120,11 @@ export function buildBetaHeader(baseBetas: readonly string[], extraBetas: readon
}
const midConversationSystemBeta = "mid-conversation-system-2026-04-07";
const contextManagementBeta = "context-management-2025-06-27";
const claudeCodeUtilityBetaDefaults = [
"oauth-2025-04-20",
"interleaved-thinking-2025-05-14",
"context-management-2025-06-27",
contextManagementBeta,
"prompt-caching-scope-2026-01-05",
"structured-outputs-2025-12-15",
] as const;
@@ -131,7 +132,7 @@ const claudeCodeAgentBetaDefaults = [
"claude-code-20250219",
"oauth-2025-04-20",
"interleaved-thinking-2025-05-14",
"context-management-2025-06-27",
contextManagementBeta,
"prompt-caching-scope-2026-01-05",
midConversationSystemBeta,
"advanced-tool-use-2025-11-20",
@@ -1680,6 +1681,22 @@ const streamAnthropicOnce = (
// carry it in the Claude Code list).
extraBetas.push(midConversationSystemBeta);
}
// `context_management.clear_thinking_20251015` requires this beta. OAuth
// requests carry it in `claudeCodeAgentBetaDefaults`; API-key requests
// need it added explicitly so the field is honored instead of rejected
// (#3288). Skip transports where this package cannot deliver the beta
// in the form their adapter accepts: Copilot strips Anthropic betas,
// and Vertex rawPredict needs betas in the body (`anthropic_beta`),
// not as an `anthropic-beta` HTTP header.
if (
model.reasoning &&
options?.thinkingEnabled &&
model.provider !== "github-copilot" &&
model.provider !== "google-vertex" &&
!extraBetas.includes(contextManagementBeta)
) {
extraBetas.push(contextManagementBeta);
}
const created = createClient(model, {
model,
@@ -2944,11 +2961,28 @@ function buildParams(
}
}
// Pre-compute context_management (depends on thinking).
const contextManagement =
isOAuthToken && thinking?.type === "adaptive"
? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] }
: undefined;
// Pre-compute context_management. Send keep: "all" for every enabled or
// adaptive thinking request (OAuth + API-key) — not just OAuth. Without
// this directive Anthropic-compatible backends (Z.AI, Kimi, DeepSeek, …)
// strip the replayed thinking blocks `replayUnsignedThinking` puts back
// on the wire, so the model loses the prior reasoning chain across turns
// and the KV cache misses every turn (#3288). Narrowing this guard back
// to `isOAuthToken` regresses every API-key thinking provider. Skip
// injected clients because this code cannot add the required
// `context-management-2025-06-27` beta to caller-owned SDK clients. Skip
// Copilot because its proxy strips Anthropic betas and demotes thinking
// blocks to text upstream, so `keep: "all"` is a no-op that risks proxy
// rejection of an unrecognized field. Skip Vertex rawPredict because that
// adapter requires betas in the JSON body (`anthropic_beta`) instead of the
// Anthropic HTTP beta header this code can add.
const shouldKeepThinkingContext =
!options?.client &&
model.provider !== "github-copilot" &&
model.provider !== "google-vertex" &&
(thinking?.type === "adaptive" || thinking?.type === "enabled");
const contextManagement = shouldKeepThinkingContext
? { edits: [{ type: "clear_thinking_20251015" as const, keep: "all" as const }] }
: undefined;
// Pre-compute output_config.
const outputConfigEntries: AnthropicOutputConfig = {};
+14
View File
@@ -78,6 +78,7 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
): OpenAIResponsesReplayItem | undefined {
if (item.type === "item_reference") return undefined;
if (item.type === "image_generation_call") return sanitizeOpenAIResponsesImageGenerationCallForReplay(item);
if (item.type === "reasoning") return sanitizeOpenAIResponsesReasoningItemForReplay(item);
// providerPayload stores raw output items; replay strips item ids and keeps only normalized call_id.
const { id: _id, ...sanitizedItem } = item;
@@ -88,6 +89,19 @@ function sanitizeOpenAIResponsesHistoryItemForReplay(
return sanitizedItem as unknown as OpenAIResponsesReplayItem;
}
function sanitizeOpenAIResponsesReasoningItemForReplay(item: Record<string, unknown>): OpenAIResponsesReplayItem {
const sanitizedItem: Record<string, unknown> = { type: "reasoning" };
if (Array.isArray(item.summary)) sanitizedItem.summary = item.summary;
if (Array.isArray(item.content)) sanitizedItem.content = item.content;
if (typeof item.encrypted_content === "string" || item.encrypted_content === null) {
sanitizedItem.encrypted_content = item.encrypted_content;
}
if (item.status === "in_progress" || item.status === "completed" || item.status === "incomplete") {
sanitizedItem.status = item.status;
}
return sanitizedItem as unknown as OpenAIResponsesReplayItem;
}
function sanitizeOpenAIResponsesImageGenerationCallForReplay(
item: Record<string, unknown>,
): ResponseInputItem.ImageGenerationCall | undefined {
+40 -3
View File
@@ -406,6 +406,41 @@ describe("Anthropic request fingerprint alignment", () => {
expect(capturedBeta).toContain("mid-conversation-system-2026-04-07");
});
it("adds the context-management beta to API-key thinking requests", async () => {
let capturedBeta: string | undefined;
const fetchMock = (async (_input: string | URL | Request, init?: RequestInit) => {
capturedBeta = (init?.headers as Record<string, string> | undefined)?.["anthropic-beta"];
return new Response(
JSON.stringify({ type: "error", error: { type: "invalid_request_error", message: "captured" } }),
{ status: 400, headers: { "Content-Type": "application/json" } },
);
}) as typeof fetch;
// `context_management.clear_thinking_20251015` is rejected without
// the `context-management-2025-06-27` beta. OAuth requests carry it
// via `claudeCodeAgentBetaDefaults`; API-key requests must add it
// explicitly whenever thinking is enabled so the field is honored
// instead of dropped on the floor (#3288).
await streamAnthropic(
ANTHROPIC_MODEL,
{ systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }] },
{ apiKey: "sk-ant-api-test", thinkingEnabled: true, fetch: fetchMock },
).result();
expect(capturedBeta).toContain("context-management-2025-06-27");
capturedBeta = undefined;
await streamAnthropic(
ANTHROPIC_MODEL,
{ systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }] },
{ apiKey: "sk-ant-api-test", thinkingEnabled: false, fetch: fetchMock },
).result();
// No context_management field is sent when thinking is disabled, so the
// beta MUST NOT be advertised either.
expect(capturedBeta ?? "").not.toContain("context-management-2025-06-27");
});
it("billing-header fingerprint uses first user message, not leading developer message", async () => {
const userText = "Hello from user with enough chars padding here";
@@ -1870,7 +1905,7 @@ describe("Anthropic request fingerprint alignment", () => {
expect(maxPayload.output_config).toEqual({ effort: "max" });
});
it("keeps summarized adaptive thinking by default for API-key Opus 4.7+ requests", async () => {
it("keeps summarized adaptive thinking and context management for API-key Opus 4.7+ requests", async () => {
const payload = (await captureAnthropicPayload(
buildModel({
...ANTHROPIC_MODEL_SPEC,
@@ -1892,12 +1927,14 @@ describe("Anthropic request fingerprint alignment", () => {
},
)) as {
thinking?: { type?: string; display?: string };
context_management?: unknown;
context_management?: { edits?: Array<{ type?: string; keep?: string | number }> };
output_config?: { effort?: string };
};
expect(payload.thinking).toEqual({ type: "adaptive", display: "summarized" });
expect(payload.context_management).toBeUndefined();
expect(payload.context_management).toEqual({
edits: [{ type: "clear_thinking_20251015", keep: "all" }],
});
expect(payload.output_config).toEqual({ effort: "xhigh" });
});
@@ -576,6 +576,36 @@ describe("anthropic stream envelope handling", () => {
expect(capturedParams?.tools?.map(tool => tool.name)).toEqual(["web_search"]);
expect(capturedOptions?.headers).toEqual({ "X-Umans-Websearch-Provider": "exa" });
});
it("does not send context_management through injected clients", async () => {
type CapturedPayload = {
thinking?: { type?: string };
context_management?: unknown;
};
let capturedParams: CapturedPayload | undefined;
const client: AnthropicMessagesClientLike = {
messages: {
create(params) {
capturedParams = params as CapturedPayload;
return createMockRequest(createTextSuccessEvents("done"));
},
},
};
const stream = streamAnthropic(model, context, {
client,
thinkingEnabled: true,
});
const events: AssistantMessageEvent[] = [];
for await (const event of stream) {
events.push(event);
}
const result = await stream.result();
expect(result.content).toEqual([{ type: "text", text: "done" }]);
expect(capturedParams?.thinking?.type).toBe("enabled");
expect(capturedParams?.context_management).toBeUndefined();
});
it("unwraps thinking blocks that Anthropic streams with literal thinking tags", async () => {
const wrappedThinking =
"<thinking>\n<thinking>\nCheck logs before accepting container health.\n</thinking></thinking>";
@@ -101,6 +101,29 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
expect(blocks[1]).toEqual({ type: "text", text: "Sure." });
});
it("sends context_management for API-key Anthropic-compatible thinking requests", async () => {
const { promise, resolve } = Promise.withResolvers<unknown>();
streamAnthropic(
makeModel(),
{ systemPrompt: [], messages: [makeUser("continue")] },
{
apiKey: "sk-ant-api-test",
signal: AbortSignal.abort(),
thinkingEnabled: true,
onPayload: payload => resolve(payload),
},
);
const payload = (await promise) as {
thinking?: { type?: string };
context_management?: { edits?: Array<{ type?: string; keep?: string }> };
};
expect(payload.thinking?.type).toBe("enabled");
expect(payload.context_management).toEqual({
edits: [{ type: "clear_thinking_20251015", keep: "all" }],
});
});
it("sanitizes lone surrogates in tool arguments regardless of origin API", () => {
const loneSurrogate = "broken \ud83d end";
const makeToolCallAssistant = (api: AssistantMessage["api"]): AssistantMessage => ({
@@ -56,9 +56,15 @@ describe("GitHub Copilot reasoning request construction", () => {
const payload = (await captureAnthropicPayload(model)) as {
thinking?: { type?: string };
output_config?: { effort?: string };
context_management?: unknown;
};
expect(payload.thinking).toEqual({ type: "adaptive" });
expect(payload.output_config).toEqual({ effort: "high" });
// The Copilot Anthropic proxy strips Anthropic betas and demotes
// thinking blocks to text upstream — the `context_management` field
// would have no replayed thinking to keep and risks proxy rejection
// of an unrecognized field. The field MUST NOT be sent (#3288).
expect(payload.context_management).toBeUndefined();
});
});
@@ -351,6 +351,7 @@ describe("OpenRouter Responses request shape", () => {
id: "rs_1",
encrypted_content: "encrypted-reasoning",
summary: [],
format: "google-gemini-v1",
};
const replayItem = {
type: nativeItem.type,
+11 -1
View File
@@ -580,7 +580,12 @@ describe("Generate E2E Tests", () => {
contextWindow: 200_000,
maxTokens: 64_000,
});
const captured = Promise.withResolvers<{ url: string; authorization: string | null; body: unknown }>();
const captured = Promise.withResolvers<{
url: string;
authorization: string | null;
betaHeader: string | null;
body: unknown;
}>();
try {
__resetVertexTokenCache();
@@ -598,6 +603,7 @@ describe("Generate E2E Tests", () => {
{ messages: [{ role: "user", content: "Hello", timestamp: Date.now() }] },
{
apiKey: "<authenticated>",
thinkingEnabled: true,
fetch: async (input, init) => {
const url = input instanceof Request ? input.url : input.toString();
if (
@@ -611,6 +617,7 @@ describe("Generate E2E Tests", () => {
captured.resolve({
url,
authorization: headers.get("authorization"),
betaHeader: headers.get("anthropic-beta"),
body: JSON.parse(bodyText),
});
return new Response(JSON.stringify({ error: { message: "stop after capture" } }), { status: 400 });
@@ -632,6 +639,9 @@ describe("Generate E2E Tests", () => {
stream: true,
});
expect((request.body as Record<string, unknown>).model).toBeUndefined();
expect((request.body as Record<string, { type?: string }>).thinking?.type).toBe("enabled");
expect((request.body as Record<string, unknown>).context_management).toBeUndefined();
expect(request.betaHeader ?? "").not.toContain("context-management-2025-06-27");
} finally {
__resetVertexTokenCache();
homedirSpy.mockRestore();
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "16.1.15",
"version": "16.1.16",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+673 -16
View File
@@ -20926,7 +20926,7 @@
"huggingface": {
"deepseek-ai/DeepSeek-R1": {
"id": "deepseek-ai/DeepSeek-R1",
"name": "DeepSeek R1",
"name": "DeepSeek-R1",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
@@ -20935,13 +20935,13 @@
"text"
],
"cost": {
"input": 3,
"output": 7,
"cacheRead": 3,
"cacheWrite": 3
"input": 0.7,
"output": 2.5,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 8192,
"contextWindow": 64000,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
@@ -21051,6 +21051,42 @@
}
}
},
"deepseek-ai/DeepSeek-V4-Flash": {
"id": "deepseek-ai/DeepSeek-V4-Flash",
"name": "DeepSeek V4 Flash",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.14,
"output": 0.28,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1048576,
"maxTokens": 384000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
],
"effortMap": {
"minimal": "high",
"low": "high",
"medium": "high",
"high": "high",
"xhigh": "max"
}
}
},
"deepseek-ai/DeepSeek-V4-Pro": {
"id": "deepseek-ai/DeepSeek-V4-Pro",
"name": "DeepSeek V4 Pro",
@@ -21087,6 +21123,85 @@
}
}
},
"google/gemma-4-26B-A4B-it": {
"id": "google/gemma-4-26B-A4B-it",
"name": "Gemma 4 26B A4B IT",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.13,
"output": 0.4,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"google/gemma-4-31B-it": {
"id": "google/gemma-4-31B-it",
"name": "Gemma 4 31B IT",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.14,
"output": 0.4,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 32768,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"meta-llama/Llama-3.3-70B-Instruct": {
"id": "meta-llama/Llama-3.3-70B-Instruct",
"name": "Llama-3.3-70B-Instruct",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 0.59,
"output": 0.79,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 4096
},
"meta-llama/Llama-3.3-70B-Instruct-Turbo": {
"id": "meta-llama/Llama-3.3-70B-Instruct-Turbo",
"name": "Llama 3.3 70B",
@@ -21106,6 +21221,34 @@
"contextWindow": 131072,
"maxTokens": 8192
},
"MiniMaxAI/MiniMax-M2": {
"id": "MiniMaxAI/MiniMax-M2",
"name": "MiniMax-M2",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.3,
"output": 1.2,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 204800,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"low",
"medium",
"high"
],
"requiresEffort": true
}
},
"MiniMaxAI/MiniMax-M2.1": {
"id": "MiniMaxAI/MiniMax-M2.1",
"name": "MiniMax-M2.1",
@@ -21190,6 +21333,36 @@
"requiresEffort": true
}
},
"MiniMaxAI/MiniMax-M3": {
"id": "MiniMaxAI/MiniMax-M3",
"name": "MiniMax-M3",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.3,
"output": 1.2,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 524288,
"maxTokens": 128000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"moonshotai/Kimi-K2-Instruct": {
"id": "moonshotai/Kimi-K2-Instruct",
"name": "Kimi-K2-Instruct",
@@ -21318,6 +21491,36 @@
]
}
},
"moonshotai/Kimi-K2.7-Code": {
"id": "moonshotai/Kimi-K2.7-Code",
"name": "Kimi K2.7 Code",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.95,
"output": 4,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"openai/gpt-oss-120b": {
"id": "openai/gpt-oss-120b",
"name": "GPT OSS 120B",
@@ -21345,6 +21548,34 @@
]
}
},
"Qwen/Qwen3-235B-A22B": {
"id": "Qwen/Qwen3-235B-A22B",
"name": "Qwen3 235B-A22B",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.2,
"output": 0.8,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 40960,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"Qwen/Qwen3-235B-A22B-Thinking-2507": {
"id": "Qwen/Qwen3-235B-A22B-Thinking-2507",
"name": "Qwen3-235B-A22B-Thinking-2507",
@@ -21374,6 +21605,53 @@
"requiresEffort": true
}
},
"Qwen/Qwen3-32B": {
"id": "Qwen/Qwen3-32B",
"name": "Qwen3 32B",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.29,
"output": 0.59,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"Qwen/Qwen3-Coder-30B-A3B-Instruct": {
"id": "Qwen/Qwen3-Coder-30B-A3B-Instruct",
"name": "Qwen3-Coder 30B-A3B Instruct",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 0.07,
"output": 0.26,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536
},
"Qwen/Qwen3-Coder-480B-A35B-Instruct": {
"id": "Qwen/Qwen3-Coder-480B-A35B-Instruct",
"name": "Qwen3-Coder-480B-A35B-Instruct",
@@ -21450,6 +21728,93 @@
"contextWindow": 262144,
"maxTokens": 131072
},
"Qwen/Qwen3.5-122B-A10B": {
"id": "Qwen/Qwen3.5-122B-A10B",
"name": "Qwen3.5 122B-A10B",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.4,
"output": 3.2,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"Qwen/Qwen3.5-27B": {
"id": "Qwen/Qwen3.5-27B",
"name": "Qwen3.5 27B",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.3,
"output": 2.4,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"Qwen/Qwen3.5-35B-A3B": {
"id": "Qwen/Qwen3.5-35B-A3B",
"name": "Qwen3.5 35B-A3B",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.25,
"output": 2,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"Qwen/Qwen3.5-397B-A17B": {
"id": "Qwen/Qwen3.5-397B-A17B",
"name": "Qwen3.5-397B-A17B",
@@ -21479,6 +21844,93 @@
]
}
},
"Qwen/Qwen3.5-9B": {
"id": "Qwen/Qwen3.5-9B",
"name": "Qwen3.5 9B",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.17,
"output": 0.25,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"Qwen/Qwen3.6-35B-A3B": {
"id": "Qwen/Qwen3.6-35B-A3B",
"name": "Qwen3.6 35B-A3B",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.15,
"output": 0.95,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"stepfun-ai/Step-3.5-Flash": {
"id": "stepfun-ai/Step-3.5-Flash",
"name": "Step 3.5 Flash",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.1,
"output": 0.3,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 256000,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"XiaomiMiMo/MiMo-V2-Flash": {
"id": "XiaomiMiMo/MiMo-V2-Flash",
"name": "MiMo-V2-Flash",
@@ -21506,6 +21958,123 @@
]
}
},
"zai-org/GLM-4.5": {
"id": "zai-org/GLM-4.5",
"name": "GLM-4.5",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.6,
"output": 2.2,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 98304,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"zai-org/GLM-4.5-Air": {
"id": "zai-org/GLM-4.5-Air",
"name": "GLM-4.5-Air",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.13,
"output": 0.85,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 98304,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"zai-org/GLM-4.5V": {
"id": "zai-org/GLM-4.5V",
"name": "GLM-4.5V",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 0.6,
"output": 1.8,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 65536,
"maxTokens": 16384,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"zai-org/GLM-4.6": {
"id": "zai-org/GLM-4.6",
"name": "GLM-4.6",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 0.55,
"output": 2.2,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 204800,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"zai-org/GLM-4.7": {
"id": "zai-org/GLM-4.7",
"name": "GLM-4.7",
@@ -21621,6 +22190,35 @@
"xhigh"
]
}
},
"zai-org/GLM-5.2": {
"id": "zai-org/GLM-5.2",
"name": "GLM-5.2",
"api": "openai-completions",
"provider": "huggingface",
"baseUrl": "https://router.huggingface.co/v1",
"reasoning": true,
"input": [
"text"
],
"cost": {
"input": 1.4,
"output": 4.4,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 131072,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
}
},
"kilo": {
@@ -46112,11 +46710,11 @@
},
"Qwen/Qwen3-235B-A22B": {
"id": "Qwen/Qwen3-235B-A22B",
"name": "Qwen/Qwen3-235B-A22B",
"name": "Qwen3 235B-A22B",
"api": "openai-completions",
"provider": "nanogpt",
"baseUrl": "https://nano-gpt.com/api/v1",
"reasoning": false,
"reasoning": true,
"input": [
"text"
],
@@ -46127,7 +46725,16 @@
"cacheWrite": 0
},
"contextWindow": 131072,
"maxTokens": 8192
"maxTokens": 8192,
"thinking": {
"mode": "effort",
"efforts": [
"minimal",
"low",
"medium",
"high"
]
}
},
"qwen/Qwen3-235B-A22B-Instruct-2507": {
"id": "qwen/Qwen3-235B-A22B-Instruct-2507",
@@ -46647,13 +47254,14 @@
},
"Qwen/Qwen3.6-35B-A3B": {
"id": "Qwen/Qwen3.6-35B-A3B",
"name": "Qwen/Qwen3.6-35B-A3B",
"name": "Qwen3.6 35B-A3B",
"api": "openai-completions",
"provider": "nanogpt",
"baseUrl": "https://nano-gpt.com/api/v1",
"reasoning": true,
"input": [
"text"
"text",
"image"
],
"cost": {
"input": 0,
@@ -47716,6 +48324,25 @@
"contextWindow": null,
"maxTokens": null
},
"sakana/fugu-ultra": {
"id": "sakana/fugu-ultra",
"name": "sakana/fugu-ultra",
"api": "openai-completions",
"provider": "nanogpt",
"baseUrl": "https://nano-gpt.com/api/v1",
"reasoning": false,
"input": [
"text"
],
"cost": {
"input": 0,
"output": 0,
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": null
},
"Salesforce/Llama-xLAM-2-70b-fc-r": {
"id": "Salesforce/Llama-xLAM-2-70b-fc-r",
"name": "Salesforce/Llama-xLAM-2-70b-fc-r",
@@ -69124,9 +69751,9 @@
"text"
],
"cost": {
"input": 1,
"output": 4,
"cacheRead": 0.18,
"input": 0.98,
"output": 3.08,
"cacheRead": 0.182,
"cacheWrite": 0
},
"contextWindow": 1048576,
@@ -69694,7 +70321,7 @@
"together": {
"deepseek-ai/DeepSeek-R1": {
"id": "deepseek-ai/DeepSeek-R1",
"name": "DeepSeek R1",
"name": "DeepSeek-R1",
"api": "openai-completions",
"provider": "together",
"baseUrl": "https://api.together.xyz/v1",
@@ -77765,6 +78392,36 @@
]
}
},
"sakana/fugu-ultra": {
"id": "sakana/fugu-ultra",
"name": "Fugu Ultra",
"api": "anthropic-messages",
"provider": "vercel-ai-gateway",
"baseUrl": "https://ai-gateway.vercel.sh",
"reasoning": true,
"input": [
"text",
"image"
],
"cost": {
"input": 5,
"output": 30,
"cacheRead": 0.5,
"cacheWrite": 0
},
"contextWindow": 1000000,
"maxTokens": 1000000,
"thinking": {
"mode": "budget",
"efforts": [
"minimal",
"low",
"medium",
"high",
"xhigh"
]
}
},
"stepfun/step-3.5-flash": {
"id": "stepfun/step-3.5-flash",
"name": "Step 3.5 Flash",
+37 -1
View File
@@ -2,6 +2,13 @@
## [Unreleased]
## [16.1.16] - 2026-06-23
### Breaking Changes
- Renamed the eval `agent()` helper parameters `agent_type` → `agent` and `return_handle` → `handle` across every workflow runtime (Python, JavaScript, Ruby, Julia), so the names are identical in every language (no camelCase/snake_case split) and the agent-selection parameter matches the `task` tool's `agent`. The `__agent__` eval bridge wire protocol was renamed to match.
- Changed the `eval` tool to take a single cell per call (`{ language, code, title?, timeout?, reset? }`) instead of a `cells` array. State still persists per language across separate eval calls, tool calls, and `task` subagents, so each call is one logical step that reuses everything earlier calls defined — the array only encouraged re-importing/re-declaring the same setup in every batch. The schema, field descriptions, examples, system `eval.md`/`workflowz` helper docs, and the `[i/n]` cell-counter (now hidden for single cells) were updated to match; the renderer, ACP start-text, copy-targets, and collab-web tool view still parse legacy multi-cell transcripts.
### Added
- Added `friendlyName` support for hidden secrets so model-visible placeholders can carry sanitized semantic labels, content-derived hashes, and case hints while preserving exact deobfuscation ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)).
@@ -14,13 +21,42 @@
- Added `isolated`, `apply`, and `merge` options to eval `agent()` across every workflow runtime (Python, JavaScript, Ruby, Julia) so `workflowz`-driven fan-outs can request the same copy-on-write worktree isolation the `task` tool offers (strict opt-in via `isolated: true`, matching the `task` tool; `apply: false` keeps captured patches/branches without merging back; `merge: false` forces patch mode). Extracted the task-isolation lifecycle into `task/isolation-runner.ts` so the eval bridge and `TaskTool` share one implementation ([#3196](https://github.com/can1357/oh-my-pi/issues/3196))
### Changed
- Made the session picker fullscreen with mouse support for clicking rows and scrolling
- Pinned the session picker footer to the bottom of the screen to prevent layout flickering
- Simplified `eval` tool to accept a single logical step (code block) instead of an array of cells
- Updated `eval` tool documentation to emphasize incremental, single-step execution
- Restricted `bash` tool from using `ls` or `find`, requiring the use of `read` or `find` tools
- Simplified `todo` tool interface to accept a single operation directly instead of an array of ops
- Reinforced routing of fragile, multi-step shell logic to the `eval` tool over `bash`. The system-prompt tool policy, `bash.md`, and `eval.md` now treat loops, conditionals, heredocs, inline `-e`/`-c` scripts, multi-stage pipelines, and quote/JSON escaping as the signal to write an `eval` cell; bash's "compute a fact" carveout is narrowed to single short pipelines, and `eval.md` now actively claims that territory with runtime-templated examples (only enabled backends are advertised).
- Made `eval` an essential built-in tool (`loadMode: "essential"`, added to the default essential tool set) so it stays active under `tools.discoveryMode: "all"` instead of being hidden behind `search_tool_bm25`.
- Made the `--resume` session picker fullscreen on the terminal's alternate screen, so the list scrolls with the mouse wheel and a row resumes its session on left click. Rows are hit-tested against the live scroll window, and the keybinding hint + bottom border are now pinned to the screen bottom instead of drifting up and down as the visible window changes height.
### Fixed
- Fixed `local://` URLs decoding images as corrupted text (mojibake) instead of showing the image
- Fixed `omp --resume` hanging instead of exiting when the startup session picker is cancelled (Esc) or there are no sessions to resume. Startup arms long-lived handles (theme/appearance listeners, settings save timer, model registry), so the cancel/empty paths' bare `return` left the event loop alive and the process stuck after the picker cleared the alternate screen. These paths now exit cleanly via `process.exit(0)`, matching the `--version`/`--export` early-exit convention. The in-session `/resume` picker is unaffected — it keeps its own cancel handler that just closes the overlay.
- Fixed the `/resume` session picker scrolling down after a session is deleted. The delete-confirmation dialog was mounted as a sibling below the picker's bottom border, briefly growing the picker past the terminal height; the TUI committed the picker's header rows into native scrollback to fit, and when the dialog closed `windowTop` stayed pinned at the new commit boundary, leaving the header stranded above the viewport. The picker now hosts the `SessionList` in a single content slot and swaps the dialog INTO that slot (replacing the `SessionList`) while it is open, so the dialog only competes with the `SessionList`'s rendered budget — not the `SessionList` AND the picker chrome — and the picker frame stays inside the viewport. ([#3283](https://github.com/can1357/oh-my-pi/issues/3283))
- Fixed the `eval` tool card not streaming a still-running cell's stdout: a long-running cell (e.g. a `time.sleep()` monitor loop) showed nothing until it returned or was interrupted, then dumped everything at once. The renderer draws cell output from `details.cells[i].output`, which was only populated after `backend.execute()` resolved — live stdout streamed into the transient result `content` tail (and `renderContext.output`), which the per-cell render branch ignores. Streamed chunks now append to the active cell's `output` (a dedicated per-cell tail buffer, capped like the aggregate) as they arrive, so the card shows progress live; on completion the authoritative full output overwrites the live tail. `log()`/`phase()`/`display()` and status ops were unaffected because they already stream via the status channel.
- Fixed Escape doing nothing in the Settings text-input fields (e.g. "Python Interpreter") on terminals with the kitty keyboard protocol active (ghostty/kitty). Inside the fullscreen settings overlay the protocol reports Escape as the CSI-u sequence `\x1b[27u`, which the text-input submenu's raw `\x1b` compare missed; `handleInputOrEscape` now decodes Escape via `matchesKey`, matching every other Escape-to-cancel path.
- Fixed Julia `eval` graph/plot visualization (Plots.jl, GraphRecipes, Makie, etc.) never rendering inline. Two bugs: (1) the runner's `build_mime_bundle`/`emit_error` dispatched `show`/`showable`/`showerror` directly from the long-lived `main()` loop, whose world age is frozen before any cell ran, so rich `show(::IO, ::MIME"image/png", …)` methods registered when a plotting package is `using`-ed inside a cell were invisible — `show` fell back to the default struct repr (which itself threw on Julia 1.12, aborting the whole result). These calls now route through `Base.invokelatest`, and the `text/plain` probe is guarded so a failing repr can no longer suppress the image MIME. (2) The default GR backend popped up a native `gksqt` GUI window on each plot; the runner now defaults `GKSwstype=100` (headless, overridable) so plots render only as inline PNGs, mirroring the Python runner's `MPLBACKEND=Agg` default.
- Fixed streaming output blocks incorrectly calculating preview height, preventing flickering banners
- Fixed streaming `bash`/`eval` tool output duplicating its `… (N earlier lines, showing 10 of M) (ctrl+o to expand)` preview into native scrollback. The collapsed output is a sliding tail window fixed at 10 lines, so when the box outgrew the live viewport (a tall command/output under a still-live predecessor such as a parallel tool) its mutating tail scrolled above the commit window and the renderer re-committed a fresh snapshot every frame, stacking dozens of stale preview banners and chunks. The output preview is now clamped to the viewport tail (`Math.min(10, previewWindowRows())`) and measured in visual rows at the box's inner content width (via the new `outputBlockContentWidth` helper), so on short terminals the volatile tail shrinks to stay on-screen and is never committed. Fixes the duplication introduced when scroll-off commits were made loss-free.
- Prevented `/handoff` from executing while a response is streaming to avoid session corruption
- Fixed `/handoff` cold-missing the provider prompt cache. Handoff generation now builds its request through the same pipeline a live turn uses (`convertMessagesToLlm` + `Agent.buildSideRequestContext` + `prepareSimpleStreamOptions`, via the new `generateHandoffFromContext`), so it reuses the live system prompt, normalized tools, transformed/obfuscated message history, and — critically — a stable `promptCacheKey` with a unique side `sessionId`. Previously the oneshot sent no cache-routing key and skipped the `transformContext`/`transformProviderContext` and tool/message normalization the loop applies, so its prefix never matched what the turn populated and every handoff re-read the whole context uncached. Mirrors the cache-preserving path already used by `/btw` and `/omfg`.
- Fixed `/handoff` (and the RPC `handoff` command) resetting the agent while a response was still streaming, which let the live turn keep emitting into the torn-down session. Manual handoff now refuses while a prompt is in flight (matching `/fork` and `/move`); the auto-handoff path is unaffected.
- Fixed Exa web search requests firing back-to-back with no client-side pacing by adding a configurable `exa.searchDelayMs` delay (default 1000ms) between Exa search requests. ([#3271](https://github.com/can1357/oh-my-pi/issues/3271))
- Fixed `ask` returning `(cancelled)` or aborting the tool when Escape dismissed `Other (type your own)` custom input; it now returns to the option selector so the user can pick a listed answer instead. ([#3269](https://github.com/can1357/oh-my-pi/issues/3269))
- Fixed `/goal` threshold auto-compaction skipping real sessions through three paths: per-turn supersede/drop-useless pruning no longer deflates the threshold trigger below the last provider-billed context; active-goal text stops now attempt threshold maintenance before unexpected-stop retry continuations can return from post-turn handling; and empty `toolUse` stops keep the existing cleanup pass that strips the orphan assistant from active context + session history before any compaction continuation. Active-goal compaction continuations now also resolve completed retry gates before returning, preventing `isRetrying` from staying stuck after a retry succeeds over the threshold. Added `agent_end maintenance routing` and `Auto-compaction threshold decision` debug logs so future no-start reports identify the exact early-return branch and the billed/stored/resolved/post-maintenance token counts that fed `shouldCompact`. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174))
- Fixed active `/goal` runs that never reached `agent_end` because the model kept emitting tool calls inside one agent run. Threshold maintenance now runs between tool-call turns, compacts the live loop context in place, and suppresses queued continuations that would race the still-running goal loop. ([#3174](https://github.com/can1357/oh-my-pi/issues/3174))
### Removed
- Removed `append`, `tree`, and `diff` eval helper functions from Python, JavaScript, and Ruby
- Removed `sort`, `uniq`, and `counter` text processing eval helpers from Python, JavaScript, and Ruby
- Removed the `append(path, content)`, `tree(path, max_depth?, show_hidden?)`, and `diff(a, b)` eval prelude helpers from every workflow runtime (Python, JavaScript, Ruby, Julia), along with their status renderers, icon entries, and tool/`docs` references. Use `write`/`read` for file mutation and `tool.<name>(...)` for richer filesystem operations.
- Removed the `sort(text, reverse?, unique?)`, `uniq(text, count?)`, and `counter(items, limit?, reverse?)` eval text helpers from the Python, JavaScript, and Ruby prelude surfaces (Julia never defined them), along with the JS `HelperBundle`/`HelperOptions` members and `docs` references. Sort/dedupe/count inline in cell code instead.
## [16.1.15] - 2026-06-22
@@ -12360,4 +12396,4 @@ Initial public release.
## [0.7.6] - 2025-11-13
Previous releases did not maintain a changelog.
Previous releases did not maintain a changelog.
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-coding-agent",
"version": "16.1.15",
"version": "16.1.16",
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+18 -4
View File
@@ -5,7 +5,15 @@ import * as path from "node:path";
const packageDir = path.join(import.meta.dir, "..");
const repoRoot = path.join(packageDir, "..", "..");
const outputPath = path.join(packageDir, "dist", "omp");
// Optional cross-compile target, e.g. CROSS_TARGET=linux-arm64 → bun build
// --target=bun-linux-arm64, embeds the matching native, outputs dist/omp-<target>.
const crossTarget = Bun.env.CROSS_TARGET || null;
const [crossPlatform, crossArch] = crossTarget ? crossTarget.split("-") : [null, null];
// x64 uses the baseline bun runtime so it runs under Rosetta / pre-AVX2 CPUs
// (the modern bun-linux-x64 target SIGILLs under Apple-Silicon Rosetta).
const bunTarget = crossTarget ? (crossTarget === "linux-x64" ? "bun-linux-x64-baseline" : `bun-${crossTarget}`) : null;
const outName = crossTarget ? `omp-${crossTarget}` : "omp";
const outputPath = path.join(packageDir, "dist", outName);
// Transformers.js is an optional, native-heavy dependency that is never bundled
// into the binary; the tiny-model worker `bun install`s it into a runtime cache
@@ -17,7 +25,7 @@ const transformersVersion = (
).version;
function shouldAdhocSignDarwinBinary(): boolean {
return process.platform === "darwin";
return process.platform === "darwin" && !crossTarget;
}
async function runCommand(
@@ -43,7 +51,12 @@ async function main(): Promise<void> {
try {
await runCommand(["bun", "--cwd=../stats", "scripts/generate-client-bundle.ts", "--generate"]);
await runCommand(["bun", "scripts/generate-docs-index.ts", "--generate"]);
await runCommand(["bun", "--cwd=../natives", "run", "embed:native"]);
await runCommand(
["bun", "--cwd=../natives", "run", "embed:native"],
crossTarget
? { ...Bun.env, TARGET_PLATFORM: crossPlatform as string, TARGET_ARCH: crossArch as string }
: Bun.env,
);
await runCommand(["bun", "scripts/embed-mupdf-wasm.ts", "--generate"]);
try {
const buildEnv = shouldAdhocSignDarwinBinary() ? { ...Bun.env, BUN_NO_CODESIGN_MACHO_BINARY: "1" } : Bun.env;
@@ -52,6 +65,7 @@ async function main(): Promise<void> {
"bun",
"build",
"--compile",
...(bunTarget ? ["--target", bunTarget] : []),
"--no-compile-autoload-bunfig",
"--no-compile-autoload-dotenv",
"--no-compile-autoload-tsconfig",
@@ -85,7 +99,7 @@ async function main(): Promise<void> {
"./packages/coding-agent/src/extensibility/legacy-pi-ai-shim.ts",
"./packages/coding-agent/src/extensibility/legacy-pi-coding-agent-shim.ts",
"--outfile",
"packages/coding-agent/dist/omp",
`packages/coding-agent/dist/${outName}`,
],
buildEnv,
repoRoot,
+4 -5
View File
@@ -1,10 +1,9 @@
/**
* CLI argument parsing and help display
*/
import { type Effort, THINKING_EFFORTS } from "@oh-my-pi/pi-catalog/effort";
import { APP_NAME, CONFIG_DIR_NAME, logger } from "@oh-my-pi/pi-utils";
import chalk from "chalk";
import { parseEffort } from "../thinking";
import { CLI_THINKING_LEVELS, type ConfiguredThinkingLevel, parseCliThinkingLevel } from "../thinking";
import { BUILTIN_TOOL_NAMES } from "../tools/builtin-names";
import {
OPTIONAL_FLAGS,
@@ -32,7 +31,7 @@ export interface Args {
apiKey?: string;
systemPrompt?: string;
appendSystemPrompt?: string;
thinking?: Effort;
thinking?: ConfiguredThinkingLevel;
hideThinking?: boolean;
advisor?: boolean;
continue?: boolean;
@@ -89,9 +88,9 @@ export interface Args {
*/
const PARSE_DEPS: ParseDeps = {
logger,
parseEffort,
parseThinking: parseCliThinkingLevel,
builtinToolNames: BUILTIN_TOOL_NAMES,
thinkingEfforts: THINKING_EFFORTS,
thinkingEfforts: CLI_THINKING_LEVELS,
};
export function parseArgs(inputArgs: string[], extensionFlags?: Map<string, { type: "boolean" | "string" }>): Args {
+3 -3
View File
@@ -30,7 +30,7 @@
* real implementations at the dispatch site.
*/
import type { Effort } from "@oh-my-pi/pi-ai";
import type { ConfiguredThinkingLevel } from "../thinking";
import type { Args } from "./args";
/**
@@ -44,7 +44,7 @@ import type { Args } from "./args";
*/
export interface ParseDeps {
logger: { warn: (message: string, meta?: Record<string, unknown>) => void };
parseEffort: (value: string | null | undefined) => Effort | undefined;
parseThinking: (value: string | null | undefined) => ConfiguredThinkingLevel | undefined;
builtinToolNames: readonly string[];
thinkingEfforts: readonly string[];
}
@@ -165,7 +165,7 @@ export const STRING_SETTERS: Record<string, StringSetter> = {
result.tools = valid;
},
"--thinking": (result, value, deps) => {
const thinking = deps.parseEffort(value);
const thinking = deps.parseThinking(value);
if (thinking !== undefined) {
result.thinking = thinking;
} else {
@@ -5,17 +5,14 @@ export const interactionFixtures: Record<string, GalleryFixture> = {
todo: {
label: "Todo",
streamingArgs: {
ops: [{ op: "init", list: [{ phase: "Foundation", items: ["Scaffold crate"] }] }],
op: "init",
list: [{ phase: "Foundation", items: ["Scaffold crate"] }],
},
args: {
ops: [
{
op: "init",
list: [
{ phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] },
{ phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] },
],
},
op: "init",
list: [
{ phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] },
{ phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] },
],
},
result: {
@@ -59,31 +59,23 @@ export const shellFixtures: Record<string, GalleryFixture> = {
eval: {
label: "Eval",
streamingArgs: {
cells: [
{
language: "py",
code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js',
title: "load config",
},
],
language: "py",
code: 'import json\nfrom pathlib import Path\n\ndata = json.loads(Path("package.js',
title: "load config",
},
args: {
cells: [
{
language: "py",
title: "load config",
code: [
"import json",
"from pathlib import Path",
"",
'data = json.loads(Path("package.json").read_text())',
'deps = data.get("dependencies", {})',
'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")',
'print(f"{len(deps)} dependencies")',
"display(sorted(deps)[:3])",
].join("\n"),
},
],
language: "py",
title: "load config",
code: [
"import json",
"from pathlib import Path",
"",
'data = json.loads(Path("package.json").read_text())',
'deps = data.get("dependencies", {})',
'print(f"{data[\\"name\\"]} v{data[\\"version\\"]}")',
'print(f"{len(deps)} dependencies")',
"display(sorted(deps)[:3])",
].join("\n"),
},
result: {
content: [
@@ -8,8 +8,10 @@ import { FileSessionStorage } from "../session/session-storage";
/**
* Show the TUI session selector and return the selected session, or null if
* cancelled. Tab toggles between current-folder and all-projects scope; the
* all-projects list is loaded lazily via `SessionManager.listAll`.
* cancelled. Rendered as a fullscreen overlay on the terminal's alternate
* screen, so the list scrolls and rows are clickable with the mouse. Tab
* toggles between current-folder and all-projects scope; the all-projects list
* is loaded lazily via `SessionManager.listAll`.
*/
export async function selectSession(
sessions: SessionInfo[],
@@ -65,6 +67,7 @@ export async function selectSession(
loadAllSessions: () => SessionManager.listAll(storage),
allSessions: options?.allSessions,
getTerminalRows: () => ui.terminal.rows,
fillHeight: true,
},
);
return selector;
@@ -72,7 +75,18 @@ export async function selectSession(
const selector = showSelector();
selector.setOnRequestRender(() => ui.requestRender());
ui.addChild(selector);
// Present as a fullscreen overlay so the picker borrows the terminal's
// alternate screen buffer (vim/less idiom): the list scrolls and rows are
// clickable via the mouse tracking the overlay enables for its lifetime.
// Anchored top-left at full size so a mouse row maps directly to a rendered
// line (the overlay paints from screen row 0).
ui.showOverlay(selector, {
anchor: "top-left",
width: "100%",
maxHeight: "100%",
margin: 0,
fullscreen: true,
});
ui.setFocus(selector);
ui.start();
return promise;
+3 -3
View File
@@ -2,12 +2,12 @@
* Root command for the coding agent CLI.
*/
import { THINKING_EFFORTS } from "@oh-my-pi/pi-catalog/effort";
import { APP_NAME } from "@oh-my-pi/pi-utils";
import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
import { parseArgs } from "../cli/args";
import { runRootCommand } from "../main";
import { prepareAcpTerminalAuthArgs } from "../modes/acp/terminal-auth";
import { CLI_THINKING_LEVELS } from "../thinking";
export default class Index extends Command {
static description = "AI coding assistant";
@@ -100,8 +100,8 @@ export default class Index extends Command {
description: "Comma-separated list of tools to enable (default: all)",
}),
thinking: Flags.string({
description: `Set thinking level: ${THINKING_EFFORTS.join(", ")}`,
options: [...THINKING_EFFORTS],
description: `Set thinking level: ${CLI_THINKING_LEVELS.join(", ")}`,
options: [...CLI_THINKING_LEVELS],
}),
"hide-thinking": Flags.boolean({
description: "Hide thinking blocks in TUI output (display only, does not disable model thinking)",
@@ -3572,7 +3572,7 @@ export const SETTINGS_SCHEMA = {
group: "Discovery & MCP",
label: "Essential Tools Override",
description:
"Override the always-loaded built-in tools (default: read, bash, edit). Leave empty to use defaults.",
"Override the always-loaded built-in tools (default: read, bash, edit, write, find, eval). Leave empty to use defaults.",
},
},
@@ -4479,6 +4479,17 @@ export const SETTINGS_SCHEMA = {
},
},
"exa.searchDelayMs": {
type: "number",
default: 1_000,
ui: {
tab: "providers",
group: "Services",
label: "Exa Search Delay",
description: "Minimum delay between Exa web search requests in milliseconds; set 0 to disable pacing",
},
},
"exa.enableResearcher": {
type: "boolean",
default: false,
@@ -4785,6 +4796,7 @@ export interface TtsrSettings {
export interface ExaSettings {
enabled: boolean;
enableSearch: boolean;
searchDelayMs: number;
enableResearcher: boolean;
enableWebsets: boolean;
}
+34 -12
View File
@@ -22,6 +22,7 @@ import {
invalidateRenderedStringCache,
type LspBatchRequest,
PREVIEW_LIMITS,
previewWindowRows,
type RenderedStringCache,
replaceTabs,
shortenPath,
@@ -340,6 +341,7 @@ function renderPlainTextPreview(text: string, uiTheme: Theme, _filePath?: string
function formatStreamingDiff(
diff: string,
rawPath: string,
width: number,
uiTheme: Theme,
expanded: boolean,
label = "streaming",
@@ -347,15 +349,32 @@ function formatStreamingDiff(
cache?: RenderedStringCache,
): string {
if (!diff) return "";
let text = cachedRenderedString(cache, uiTheme, expanded, rawPath, diff, () => {
// Collapsed uses a "Cursor" tail window: pin the last
// EDIT_STREAMING_PREVIEW_LINES rows to the bottom so freshly streamed changes
// stay on screen. The whole-file diff is recomputed on every streamed chunk
// and its Myers alignment is not monotonic in payload length, so a hunk-aware
// window stutters as rows move between hunks. Expanded deliberately lifts that
// cap for the approval-time full view.
// Clamp the collapsed tail to the viewport so a tall or fast-growing diff
// cannot outgrow the live window. Otherwise its mutating tail scrolls above
// the native-scrollback commit boundary and the engine re-commits a fresh
// snapshot every streamed frame, stacking duplicate "… more lines above"
// previews in history. The budget is VISUAL rows (a long wrapped line counts
// for more than one) at the framed block's inner width (border only —
// contentPaddingLeft is 0); only the visible suffix is syntax-colored, so the
// cheap raw-line wrap walk keeps the per-chunk cost bounded. innerWidth/budget
// are in the cache salt so a resize re-slices.
const innerWidth = Math.max(1, width - 2);
const budget = expanded ? Number.POSITIVE_INFINITY : Math.min(EDIT_STREAMING_PREVIEW_LINES, previewWindowRows());
let text = cachedRenderedString(cache, uiTheme, expanded, `${rawPath}:${innerWidth}:${budget}`, diff, () => {
// "Cursor" tail window: pin the last rows to the bottom so freshly streamed
// changes stay on screen. The whole-file diff is recomputed every chunk and
// its Myers alignment is not monotonic in payload length, so a hunk-aware
// window stutters as rows move between hunks. Expanded lifts the cap.
const allLines = diff.replace(/\n+$/u, "").split("\n");
const hiddenLines = expanded ? 0 : Math.max(0, allLines.length - EDIT_STREAMING_PREVIEW_LINES);
let visualUsed = 0;
let cut = allLines.length;
for (let i = allLines.length - 1; i >= 0; i--) {
const lineRows = Math.max(1, wrapTextWithAnsi(replaceTabs(allLines[i]!), innerWidth).length);
if (visualUsed + lineRows > budget && visualUsed > 0) break;
visualUsed += lineRows;
cut = i;
}
const hiddenLines = cut;
const visible = hiddenLines > 0 ? allLines.slice(hiddenLines) : allLines;
let rendered = "\n\n";
if (hiddenLines > 0) {
@@ -384,6 +403,7 @@ function formatStreamingDiff(
function formatMultiFileStreamingDiff(
previews: PerFileDiffPreview[],
width: number,
uiTheme: Theme,
expanded: boolean,
spinnerFrame?: number,
@@ -405,7 +425,7 @@ function formatMultiFileStreamingDiff(
const isLast = index === previews.length - 1;
const cache = previewCacheAt(caches, index);
parts.push(
`${header}${formatStreamingDiff(preview.diff, preview.path, uiTheme, expanded, "preview", isLast ? spinnerFrame : undefined, cache)}`,
`${header}${formatStreamingDiff(preview.diff, preview.path, width, uiTheme, expanded, "preview", isLast ? spinnerFrame : undefined, cache)}`,
);
}
}
@@ -415,6 +435,7 @@ function formatMultiFileStreamingDiff(
function getCallPreview(
args: EditRenderArgs,
rawPath: string,
width: number,
uiTheme: Theme,
renderContext: EditRenderContext | undefined,
expanded: boolean,
@@ -423,14 +444,14 @@ function getCallPreview(
): string {
const multi = renderContext?.perFileDiffPreview;
if (multi && multi.length > 1 && multi.some(p => p.diff || p.error)) {
return formatMultiFileStreamingDiff(multi, uiTheme, expanded, spinnerFrame, caches);
return formatMultiFileStreamingDiff(multi, width, uiTheme, expanded, spinnerFrame, caches);
}
const cache = previewCacheAt(caches, 0);
if (args.previewDiff) {
return formatStreamingDiff(args.previewDiff, rawPath, uiTheme, expanded, "preview", spinnerFrame, cache);
return formatStreamingDiff(args.previewDiff, rawPath, width, uiTheme, expanded, "preview", spinnerFrame, cache);
}
if (args.diff && args.op) {
return formatStreamingDiff(args.diff, rawPath, uiTheme, expanded, "streaming", spinnerFrame, cache);
return formatStreamingDiff(args.diff, rawPath, width, uiTheme, expanded, "streaming", spinnerFrame, cache);
}
if (args.diff) {
return renderPlainTextPreview(args.diff, uiTheme, rawPath);
@@ -628,6 +649,7 @@ export const editToolRenderer = {
let body = getCallPreview(
editArgs,
rawPath,
width,
uiTheme,
renderContext,
options.expanded,
@@ -156,7 +156,7 @@ describe("runEvalAgent", () => {
vi.restoreAllMocks();
});
it("resolves the default task agent and agentType overrides", async () => {
it("resolves the default task agent and agent overrides", async () => {
mockAgents();
const runSpy = vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options =>
singleResult(options, {
@@ -166,7 +166,7 @@ describe("runEvalAgent", () => {
const session = makeSession();
const defaultResult = await runEvalAgent({ prompt: "hello" }, { session });
const overrideResult = await runEvalAgent({ prompt: "hello", agentType: "reviewer" }, { session });
const overrideResult = await runEvalAgent({ prompt: "hello", agent: "reviewer" }, { session });
expect(defaultResult.text).toBe("task");
expect(overrideResult.text).toBe("reviewer");
@@ -178,7 +178,7 @@ describe("runEvalAgent", () => {
mockAgents([taskAgent]);
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options));
await expect(runEvalAgent({ prompt: "hello", agentType: "missing" }, { session: makeSession() })).rejects.toThrow(
await expect(runEvalAgent({ prompt: "hello", agent: "missing" }, { session: makeSession() })).rejects.toThrow(
'Unknown agent "missing"',
);
});
@@ -844,12 +844,12 @@ describe("runEvalAgent isolation", () => {
expect(mergeSpy).toHaveBeenCalledTimes(1);
});
it("preserves temp artifacts for non-isolated returnHandle outputs", async () => {
it("preserves temp artifacts for non-isolated handle outputs", async () => {
mockAgents();
const rmSpy = vi.spyOn(fs, "rm").mockResolvedValue(undefined);
vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options));
await runEvalAgent({ prompt: "plain handle", returnHandle: true }, { session: makeSession() });
await runEvalAgent({ prompt: "plain handle", handle: true }, { session: makeSession() });
const removedArtifactsDir = rmSpy.mock.calls.some(
([target]) => typeof target === "string" && target.includes("omp-eval-agent-"),
@@ -1204,7 +1204,7 @@ describe("runEvalAgent isolation", () => {
expect(removedArtifactsDir).toBe(true);
});
it("preserves the temp artifacts dir after a successful apply when returnHandle is requested", async () => {
it("preserves the temp artifacts dir after a successful apply when handle is requested", async () => {
mockAgents();
mockIsolationContext();
const rmSpy = vi.spyOn(fs, "rm").mockResolvedValue(undefined);
@@ -1218,7 +1218,7 @@ describe("runEvalAgent isolation", () => {
mergedBranchForNestedPatches: false,
});
await runEvalAgent({ prompt: "scout", isolated: true, returnHandle: true }, { session: isolatedSession() });
await runEvalAgent({ prompt: "scout", isolated: true, handle: true }, { session: isolatedSession() });
const removedArtifactsDir = rmSpy.mock.calls.some(
([target]) => typeof target === "string" && target.includes("omp-eval-agent-"),
@@ -4,7 +4,7 @@ import { TempDir } from "@oh-my-pi/pi-utils/temp";
import { createHelpers, type HelperContext } from "../js/shared/helpers";
/**
* The eval helpers (`read`/`write`/`append`) must substitute injected on-disk
* The eval helpers (`read`/`write`) must substitute injected on-disk
* roots for internal-URL schemes. Without it, `write("local://x.md")` hits a
* stdlib `path.resolve` that collapses `local://` to `local:/`, creating a junk
* `local:` directory under the cwd instead of landing where `read local://x.md`
@@ -20,7 +20,7 @@ function makeCtx(cwd: string, roots: Record<string, string>): HelperContext {
}
describe("eval js helpers internal-url resolution", () => {
it("writes, reads, and appends local:// under the injected root", async () => {
it("writes and reads local:// under the injected root", async () => {
using tmp = TempDir.createSync("@eval-helpers-local-");
const root = path.join(tmp.path(), "local");
const helpers = createHelpers(makeCtx(tmp.path(), { local: root }));
@@ -30,9 +30,6 @@ describe("eval js helpers internal-url resolution", () => {
expect(await Bun.file(written).text()).toBe("hello");
expect(await helpers.read("local://notes/merge-map.md")).toBe("hello");
await helpers.append("local://notes/merge-map.md", " world");
expect(await helpers.read("local://notes/merge-map.md")).toBe("hello world");
// Regression: no literal `local:` directory created under the cwd.
expect(await Bun.file(path.join(tmp.path(), "local:")).exists()).toBe(false);
expect(await Bun.file(path.join(tmp.path(), "local:", "notes", "merge-map.md")).exists()).toBe(false);
@@ -11,35 +11,6 @@ describe.skipIf(!HAS_JULIA)("eval Julia prelude helpers", () => {
await disposeJuliaKernelSessionsByOwner(OWNER_ID);
});
it("supports tree keyword options and unified diff", async () => {
using tempDir = TempDir.createSync("@omp-eval-julia-helpers-");
await Bun.write(path.join(tempDir.path(), "a.txt"), "same\nold\n");
await Bun.write(path.join(tempDir.path(), "b.txt"), "same\nnew\n");
await Bun.write(path.join(tempDir.path(), "dir", "child.txt"), "child");
const result = await executeJulia(
`
d = diff("a.txt", "b.txt")
println("DIFF_DELETE=", occursin("-old", d))
println("DIFF_ADD=", occursin("+new", d))
t = tree(".", max_depth=2)
println("TREE_CHILD=", occursin("child.txt", t))
nothing
`,
{
cwd: tempDir.path(),
sessionId: `julia-prelude-diff:${crypto.randomUUID()}`,
kernelOwnerId: OWNER_ID,
reset: true,
},
);
expect(result.exitCode).toBe(0);
expect(result.output).toContain("DIFF_DELETE=true");
expect(result.output).toContain("DIFF_ADD=true");
expect(result.output).toContain("TREE_CHILD=true");
}, 120_000);
it("supports output ranges, JSON queries, metadata, and ANSI stripping", async () => {
using tempDir = TempDir.createSync("@omp-eval-julia-output-");
const artifactsDir = path.join(tempDir.path(), "session-artifacts");
@@ -3,7 +3,7 @@ import * as vm from "node:vm";
import { JAVASCRIPT_PRELUDE_SOURCE } from "../js/shared/prelude";
/**
* The eval `agent()` helper grows a `returnHandle` option that turns its bare
* The eval `agent()` helper grows a `handle` option that turns its bare
* text result into a DAG node dict carrying the spawned agent's recoverable
* `agent://<id>` handle, so a downstream `pipeline`/`parallel` stage can wire
* the transcript by reference instead of re-inlining it. These lock the node
@@ -23,8 +23,8 @@ function loadPrelude(callTool: (name: string, args: unknown) => Promise<unknown>
type AgentHelper = (prompt: string, opts?: Record<string, unknown>) => Promise<unknown>;
describe("eval js agent() returnHandle", () => {
it("returns a DAG node carrying the agent:// handle when returnHandle is set", async () => {
describe("eval js agent() handle", () => {
it("returns a DAG node carrying the agent:// handle when handle is set", async () => {
let seenName: string | undefined;
let seenArgs: Record<string, unknown> | undefined;
const sandbox = loadPrelude(async (name, args) => {
@@ -32,9 +32,9 @@ describe("eval js agent() returnHandle", () => {
seenArgs = args as Record<string, unknown>;
return { text: "hello world", details: { agent: "task", id: "abc123", model: "m", structured: false } };
});
const node = await (sandbox.agent as AgentHelper)("say hi", { returnHandle: true });
const node = await (sandbox.agent as AgentHelper)("say hi", { handle: true });
expect(seenName).toBe("__agent__");
expect(seenArgs?.returnHandle).toBe(true);
expect(seenArgs?.handle).toBe(true);
expect(node).toEqual({
text: "hello world",
output: "hello world",
@@ -53,7 +53,7 @@ describe("eval js agent() returnHandle", () => {
expect(out).toBe("hello world");
});
it("carries the parsed object under data when schema and returnHandle combine", async () => {
it("carries the parsed object under data when schema and handle combine", async () => {
const payload = JSON.stringify({ k: 1 });
const sandbox = loadPrelude(async () => ({
text: payload,
@@ -61,7 +61,7 @@ describe("eval js agent() returnHandle", () => {
}));
const node = (await (sandbox.agent as AgentHelper)("emit", {
schema: { type: "object" },
returnHandle: true,
handle: true,
})) as Record<string, unknown>;
expect(node.handle).toBe("agent://id-9");
expect(node.data).toEqual({ k: 1 });
@@ -70,7 +70,7 @@ describe("eval js agent() returnHandle", () => {
it("falls back to a null handle without throwing when the bridge omits details", async () => {
const sandbox = loadPrelude(async () => ({ text: "lonely" }));
const node = await (sandbox.agent as AgentHelper)("x", { returnHandle: true });
const node = await (sandbox.agent as AgentHelper)("x", { handle: true });
expect(node).toEqual({ text: "lonely", output: "lonely", handle: null, id: null, agent: null });
});
@@ -93,7 +93,7 @@ describe("eval js agent() returnHandle", () => {
schema: { type: "object" },
isolated: true,
apply: false,
returnHandle: true,
handle: true,
})) as Record<string, unknown>;
expect(node.handle).toBe("agent://iso-1");
expect(node.data).toEqual({ ok: true });
@@ -43,19 +43,19 @@ const DEFAULT_AGENT_LABEL = "EvalAgent";
const agentArgsSchema = type({
prompt: "string>0",
"agentType?": "string>0",
"agent?": "string>0",
"model?": "string>0|string>0[]",
"label?": "string",
"schema?": "unknown",
"isolated?": "boolean",
"apply?": "boolean",
"merge?": "boolean",
"returnHandle?": "boolean",
"handle?": "boolean",
});
interface EvalAgentArgs {
prompt: string;
agentType?: string;
agent?: string;
model?: string | string[];
label?: string;
schema?: unknown;
@@ -83,7 +83,7 @@ interface EvalAgentArgs {
*/
merge?: boolean;
/** True when a runtime helper will return an `agent://` handle backed by the output artifacts. */
returnHandle?: boolean;
handle?: boolean;
}
export interface EvalAgentBridgeOptions {
@@ -276,7 +276,7 @@ function buildSubagentFailureMessage(agentName: string, result: SingleResult): s
*/
export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOptions): Promise<EvalAgentResult> {
const parsed = parseAgentArgs(args);
const agentName = parsed.agentType ?? DEFAULT_AGENT_TYPE;
const agentName = parsed.agent ?? DEFAULT_AGENT_TYPE;
const structured = Object.hasOwn(parsed, "schema");
assertNotPlanMode(options.session);
@@ -521,8 +521,7 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
// consumes `details.patchPath` / `details.branchName` /
// `details.nestedPatches` out of band. Failed isolated applies throw
// earlier with a recovery hint, so they never reach this gate.
const shouldCleanupTempArtifacts =
tempArtifactsDir && !parsed.returnHandle && (!isIsolated || changesApplied === true);
const shouldCleanupTempArtifacts = tempArtifactsDir && !parsed.handle && (!isIsolated || changesApplied === true);
if (shouldCleanupTempArtifacts) {
await fs.rm(artifactsDir, { recursive: true, force: true });
}
+6 -228
View File
@@ -146,221 +146,6 @@ function Base.write(path::AbstractString, content::Any)
return resolved
end
function append(path, content)
resolved = __omp_resolve_path(string(path))
mkpath(dirname(resolved))
open(resolved, "a") do f
Base.write(f, string(content))
end
Main.emit_frame(Dict(
"type" => "display",
"id" => Main.current_rid,
"bundle" => Dict(
"application/x-omp-status" => Dict(
"op" => "append",
"path" => resolved,
"chars" => length(string(content))
)
)
))
return resolved
end
function tree(path=".", positional_max_depth=3, positional_show_hidden=false; max_depth=positional_max_depth, show_hidden=positional_show_hidden)
base = string(path)
resolved = __omp_resolve_path(base)
lines = String[]
function walk(dir, prefix, depth)
if depth > max_depth
return
end
entries = try
readdir(dir)
catch
String[]
end
if !show_hidden
entries = filter(e -> !startswith(e, '.'), entries)
end
sort!(entries, by = e -> (ispath(joinpath(dir, e)) && isdir(joinpath(dir, e)) ? 0 : 1, lowercase(e)))
for (i, name) in enumerate(entries)
full = joinpath(dir, name)
is_last = i == length(entries)
is_dir = isdir(full)
push!(lines, "$(prefix)$(is_last ? "└── " : "├── ")$(name)$(is_dir ? "/" : "")")
if is_dir
walk(full, prefix * (is_last ? " " : "│ "), depth + 1)
end
end
end
walk(resolved, "", 1)
out = join(lines, '\n')
Main.emit_frame(Dict(
"type" => "display",
"id" => Main.current_rid,
"bundle" => Dict(
"application/x-omp-status" => Dict(
"op" => "tree",
"path" => resolved,
"lines" => length(lines)
)
)
))
return out
end
function __omp_lines_keepends(content::String)
parts = split(content, '\n'; keepempty=true)
if length(parts) == 1 && isempty(parts[1])
return String[]
end
lines = String[]
for i in eachindex(parts)
if i < length(parts)
push!(lines, string(parts[i], "\n"))
elseif !isempty(parts[i])
push!(lines, string(parts[i]))
end
end
return lines
end
function __omp_diff_ops(a::Vector{String}, b::Vector{String})
n = length(a)
m = length(b)
ops = Vector{Tuple{Symbol, Int, Int}}()
if n * m > 4_000_000
for i in 1:n
push!(ops, (:delete, i, 1))
end
for j in 1:m
push!(ops, (:insert, n + 1, j))
end
return ops
end
dp = [zeros(Int, m + 1) for _ in 1:(n + 1)]
for i in n:-1:1
for j in m:-1:1
dp[i][j] = a[i] == b[j] ? dp[i + 1][j + 1] + 1 : max(dp[i + 1][j], dp[i][j + 1])
end
end
i = 1
j = 1
while i <= n && j <= m
if a[i] == b[j]
push!(ops, (:equal, i, j))
i += 1
j += 1
elseif dp[i + 1][j] >= dp[i][j + 1]
push!(ops, (:delete, i, j))
i += 1
else
push!(ops, (:insert, i, j))
j += 1
end
end
while i <= n
push!(ops, (:delete, i, j))
i += 1
end
while j <= m
push!(ops, (:insert, i, j))
j += 1
end
return ops
end
function __omp_unified_diff(a::Vector{String}, b::Vector{String}, from_file::String, to_file::String, context::Int=3)
ops = __omp_diff_ops(a, b)
if !any(op -> op[1] != :equal, ops)
return ""
end
entries = [Dict{Symbol, Any}(:tag => tag, :ai => ai, :bi => bi, :text => tag == :insert ? b[bi] : a[ai]) for (tag, ai, bi) in ops]
changed = [i for i in eachindex(entries) if entries[i][:tag] != :equal]
groups = Vector{Tuple{Int, Int}}()
start = nothing
prev = nothing
for idx in changed
if start === nothing
start = idx
prev = idx
elseif idx - prev <= (2 * context) + 1
prev = idx
else
push!(groups, (start, prev))
start = idx
prev = idx
end
end
if start !== nothing
push!(groups, (start, prev))
end
out = IOBuffer()
write(out, "--- $from_file\n")
write(out, "+++ $to_file\n")
for (group_start, group_end) in groups
lo = max(group_start - context, 1)
hi = min(group_end + context, length(entries))
slice = entries[lo:hi]
a_start = nothing
a_count = 0
b_start = nothing
b_count = 0
for entry in slice
if entry[:tag] != :insert
if a_start === nothing
a_start = entry[:ai]
end
a_count += 1
end
if entry[:tag] != :delete
if b_start === nothing
b_start = entry[:bi]
end
b_count += 1
end
end
write(out, "@@ -$(a_start === nothing ? 1 : a_start),$a_count +$(b_start === nothing ? 1 : b_start),$b_count @@\n")
for entry in slice
prefix = entry[:tag] == :equal ? " " : (entry[:tag] == :delete ? "-" : "+")
text = string(entry[:text])
if !endswith(text, "\n")
text *= "\n"
end
write(out, prefix * text)
end
end
return String(take!(out))
end
function Base.diff(a::AbstractString, b::AbstractString)
path_a = __omp_resolve_path(string(a))
path_b = __omp_resolve_path(string(b))
lines_a = __omp_lines_keepends(open(path_a, "r") do io
Base.read(io, String)
end)
lines_b = __omp_lines_keepends(open(path_b, "r") do io
Base.read(io, String)
end)
out = __omp_unified_diff(lines_a, lines_b, path_a, path_b)
__omp_emit_status("diff", Dict{String, Any}(
"file_a" => path_a,
"file_b" => path_b,
"identical" => isempty(out),
"preview" => first(out, min(500, length(out)))
))
return out
end
function __omp_apply_query(data, query)
if query === nothing || isempty(string(query))
return data
@@ -734,10 +519,10 @@ function completion(prompt::String; model="default", system=nothing, schema=noth
return schema === nothing ? text : Main.json_parse(string(text))
end
function agent(prompt::String; agent_type="task", model=nothing, label=nothing, schema=nothing, isolated=nothing, apply=nothing, merge=nothing, return_handle=false, kwargs...)
function agent(prompt::String; agent="task", model=nothing, label=nothing, schema=nothing, isolated=nothing, apply=nothing, merge=nothing, handle=false, kwargs...)
args_dict = Dict{String, Any}("prompt" => prompt)
if agent_type !== nothing
args_dict["agentType"] = agent_type
if agent !== nothing
args_dict["agent"] = agent
end
if model !== nothing
args_dict["model"] = model
@@ -759,20 +544,13 @@ function agent(prompt::String; agent_type="task", model=nothing, label=nothing,
if merge !== nothing
args_dict["merge"] = Bool(merge)
end
handle_result = return_handle
handle_result = handle
for (k, v) in kwargs
key = string(k)
if key == "agent_type" || key == "agentType"
args_dict["agentType"] = v
elseif key == "return_handle" || key == "returnHandle"
handle_result = Bool(v)
else
args_dict[key] = v
end
args_dict[string(k)] = v
end
# Tell the bridge a handle is wanted so it preserves the backing artifacts.
if handle_result
args_dict["returnHandle"] = true
args_dict["handle"] = true
end
res = __omp_call_bridge("__agent__", args_dict)
text = res isa AbstractDict ? get(res, "text", res) : res
+38 -12
View File
@@ -3,6 +3,12 @@
using Base64
# Force GR (the default Plots.jl backend) into a headless workstation so a plot
# never pops up a native gksqt GUI window — the harness renders the inline PNG
# from `show(io, MIME"image/png", plt)` itself. `get!` keeps an explicit
# user-provided value, mirroring the Python runner's MPLBACKEND=Agg default.
get!(ENV, "GKSwstype", "100")
const ORIGINAL_STDOUT = stdout
const ORIGINAL_STDERR = stderr
const ORIGINAL_STDIN = stdin
@@ -441,24 +447,38 @@ end
function build_mime_bundle(value)
bundle = Dict{String, Any}()
# text/plain
io_plain = IOBuffer()
show(io_plain, MIME"text/plain"(), value)
bundle["text/plain"] = String(take!(io_plain))
# text/plain — every mime probe below uses `Base.invokelatest` because this
# function runs from the frozen-world `main()` loop: `show`/`showable`
# methods that a package adds when it is `using`-ed *inside* a cell (e.g.
# Plots/Makie/GraphRecipes registering rich `show` for their plot types) are
# invisible to direct dispatch here and fall back to the default struct show,
# which can itself throw. Guard text/plain too so a failing repr never aborts
# the whole bundle before the image mime is reached.
try
io_plain = IOBuffer()
Base.invokelatest(show, io_plain, MIME"text/plain"(), value)
bundle["text/plain"] = String(take!(io_plain))
catch
bundle["text/plain"] = try
summary(value)
catch
string(typeof(value))
end
end
# rich mime types
for mime_str in ["text/html", "text/markdown", "image/png", "image/jpeg"]
m = MIME(Symbol(mime_str))
if showable(m, value)
if Base.invokelatest(showable, m, value)
try
io = IOBuffer()
if mime_str in ["image/png", "image/jpeg"]
b64_io = Base64EncodePipe(io)
show(b64_io, m, value)
Base.invokelatest(show, b64_io, m, value)
close(b64_io)
else
show(io, m, value)
Base.invokelatest(show, io, m, value)
end
bundle[mime_str] = String(take!(io))
catch
@@ -466,7 +486,7 @@ function build_mime_bundle(value)
end
end
end
if value isa AbstractDict || value isa AbstractVector
try
bundle["application/json"] = value
@@ -474,7 +494,7 @@ function build_mime_bundle(value)
# ignore
end
end
return bundle
end
@@ -493,7 +513,13 @@ pushdisplay(OmpDisplay())
function emit_error(rid, err, bt)
io = IOBuffer()
showerror(io, err)
# invokelatest + guard: custom error types from packages loaded inside the
# cell define `showerror` methods invisible to this frozen-world function.
try
Base.invokelatest(showerror, io, err)
catch
print(io, string(err))
end
err_str = String(take!(io))
tb = String[]
@@ -1,19 +1,11 @@
import * as fs from "node:fs";
import * as path from "node:path";
import * as Diff from "diff";
import { ToolError } from "../../../tools/tool-errors";
import type { JsStatusEvent } from "./types";
export interface HelperOptions {
path?: string;
hidden?: boolean;
maxDepth?: number;
limit?: number;
offset?: number;
reverse?: boolean;
unique?: boolean;
count?: boolean;
}
/**
@@ -35,18 +27,12 @@ export interface HelperContext {
/**
* The set of functions exposed to user code via `globalThis.__omp_helpers__`. The JS
* prelude reads from this bag and attaches short aliases (`read`, `write`, `tree`, ...)
* prelude reads from this bag and attaches short aliases (`read`, `write`, `env`, ...)
* onto the global scope.
*/
export interface HelperBundle {
read(rawPath: string, options?: HelperOptions): Promise<string>;
writeFile(rawPath: string, data: unknown): Promise<string>;
append(rawPath: string, content: string): Promise<string>;
sortText(text: string, options?: HelperOptions): string;
uniqText(text: string, options?: HelperOptions): string | Array<[number, string]>;
counter(items: string | string[], options?: HelperOptions): Array<[number, string]>;
diff(rawA: string, rawB: string): Promise<string>;
tree(searchPath?: string, options?: HelperOptions): Promise<string>;
env(key?: string, value?: string): string | Record<string, string> | undefined;
}
@@ -81,105 +67,6 @@ export function createHelpers(ctx: HelperContext): HelperBundle {
ctx.emitStatus({ op: "write", path: filePath, bytes: getDataSize(data) });
return filePath;
},
append: async (rawPath, content) => {
const target = resolveHelperPath(ctx, rawPath, "write");
// O(1) append; read-all+rewrite both raced concurrent writers and went
// quadratic when called in a loop. Bun.write creates parent dirs, so
// keep that behavior for the append path too.
await fs.promises.mkdir(path.dirname(target), { recursive: true });
await fs.promises.appendFile(target, content, "utf-8");
ctx.emitStatus({
op: "append",
path: target,
chars: content.length,
bytes: utf8Encoder.encode(content).byteLength,
});
return target;
},
sortText: (text, options = {}) => {
const lines = String(text).split(/\r?\n/);
const deduped = options.unique ? Array.from(new Set(lines)) : lines;
const sorted = deduped.sort((a, b) => a.localeCompare(b));
if (options.reverse) sorted.reverse();
const result = sorted.join("\n");
ctx.emitStatus({
op: "sort",
lines: sorted.length,
reverse: options.reverse === true,
unique: options.unique === true,
});
return result;
},
uniqText: (text, options = {}) => {
const lines = String(text)
.split(/\r?\n/)
.filter(line => line.length > 0);
const groups: Array<[number, string]> = [];
for (const line of lines) {
const last = groups.at(-1);
if (last && last[1] === line) {
last[0] += 1;
continue;
}
groups.push([1, line]);
}
ctx.emitStatus({ op: "uniq", groups: groups.length, count_mode: options.count === true });
if (options.count) return groups;
return groups.map(([, line]) => line).join("\n");
},
counter: (items, options = {}) => {
const values = Array.isArray(items) ? items : String(items).split(/\r?\n/).filter(Boolean);
const counts = new Map<string, number>();
for (const item of values) counts.set(item, (counts.get(item) ?? 0) + 1);
const entries = Array.from(counts.entries())
.map(([item, count]) => [count, item] as [number, string])
.sort((a, b) => (options.reverse === false ? a[0] - b[0] : b[0] - a[0]) || a[1].localeCompare(b[1]));
const limited = entries.slice(0, options.limit ?? entries.length);
ctx.emitStatus({ op: "counter", unique: counts.size, total: values.length, top: limited.slice(0, 10) });
return limited;
},
diff: async (rawA, rawB) => {
const fileA = resolvePath(ctx, rawA);
const fileB = resolvePath(ctx, rawB);
const [a, b] = await Promise.all([Bun.file(fileA).text(), Bun.file(fileB).text()]);
const result = Diff.createTwoFilesPatch(fileA, fileB, a, b, "", "", { context: 3 });
ctx.emitStatus({
op: "diff",
file_a: fileA,
file_b: fileB,
identical: a === b,
preview: result.slice(0, 500),
});
return result;
},
tree: async (searchPath = ".", options = {}) => {
const root = resolvePath(ctx, searchPath);
const maxDepth = options.maxDepth ?? 3;
const showHidden = options.hidden ?? false;
const lines: string[] = [`${root}/`];
let entryCount = 0;
const walk = async (dir: string, prefix: string, depth: number): Promise<void> => {
if (depth > maxDepth) return;
const entries = (await fs.promises.readdir(dir, { withFileTypes: true }))
.filter(entry => showHidden || !entry.name.startsWith("."))
.sort((a, b) => a.name.localeCompare(b.name));
for (let index = 0; index < entries.length; index++) {
const entry = entries[index];
const isLast = index === entries.length - 1;
const connector = isLast ? "└── " : "├── ";
const suffix = entry.isDirectory() ? "/" : "";
lines.push(`${prefix}${connector}${entry.name}${suffix}`);
entryCount += 1;
if (entry.isDirectory()) {
await walk(path.join(dir, entry.name), `${prefix}${isLast ? " " : "│ "}`, depth + 1);
}
}
};
await walk(root, "", 1);
const result = lines.join("\n");
ctx.emitStatus({ op: "tree", path: root, entries: entryCount, preview: result.slice(0, 1000) });
return result;
},
env: (key, value) => {
if (!key) {
const merged = Object.fromEntries(Object.entries(getMergedEnv(ctx)).sort(([a], [b]) => a.localeCompare(b)));
@@ -63,23 +63,6 @@ if (!globalThis.__omp_js_prelude_loaded__) {
return callHelper("read", path, options);
};
const write = async (path, data) => callHelper("writeFile", path, data);
const append = (path, content) => callHelper("append", path, content);
const sort = (text, opts, ...rest) =>
callHelper("sortText", text, optionsArg("sort", opts, rest, ["reverse", "unique"], "{ reverse, unique }"));
const uniq = (text, opts, ...rest) => callHelper("uniqText", text, optionsArg("uniq", opts, rest, ["count"], "{ count }"));
const counter = (items, opts, ...rest) =>
callHelper("counter", items, optionsArg("counter", opts, rest, ["limit", "reverse"], "{ limit, reverse }"));
const diff = (a, b) => callHelper("diff", a, b);
const tree = (path = ".", opts, ...rest) => {
if (isPlainObject(path) && opts === undefined && rest.length === 0) {
return callHelper("tree", ".", path);
}
return callHelper(
"tree",
isNil(path) ? "." : path,
optionsArg("tree", opts, rest, ["maxDepth", "showHidden"], "{ maxDepth, showHidden }"),
);
};
const env = (key, value) => callHelper("env", key, value);
const tool = new Proxy(
@@ -121,14 +104,14 @@ if (!globalThis.__omp_js_prelude_loaded__) {
"agent",
opts,
rest,
["agentType", "model", "label", "schema", "isolated", "apply", "merge"],
"{ agentType, model, label, schema, isolated, apply, merge, returnHandle }",
["agent", "model", "label", "schema", "isolated", "apply", "merge"],
"{ agent, model, label, schema, isolated, apply, merge, handle }",
);
const { returnHandle, ...callArgs } = o;
const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...callArgs, returnHandle: Boolean(returnHandle) });
const { handle, ...callArgs } = o;
const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...callArgs, handle: Boolean(handle) });
const text = res && typeof res === "object" ? res.text : res;
const parsed = hasOwn(callArgs, "schema") ? JSON.parse(text) : text;
if (!returnHandle) return parsed;
if (!handle) return parsed;
const details = res && typeof res === "object" ? res.details : undefined;
if (!details || typeof details !== "object" || details.id == null) {
return { text, output: text, handle: null, id: null, agent: null };
@@ -306,11 +289,5 @@ if (!globalThis.__omp_js_prelude_loaded__) {
globalThis.__pool = __pool;
globalThis.read = read;
globalThis.write = write;
globalThis.append = append;
globalThis.sort = sort;
globalThis.uniq = uniq;
globalThis.counter = counter;
globalThis.diff = diff;
globalThis.tree = tree;
globalThis.env = env;
}
@@ -72,12 +72,6 @@ const PRELUDE_GLOBAL_KEYS = [
"__pool",
"read",
"write",
"append",
"sort",
"uniq",
"counter",
"diff",
"tree",
"env",
];
@@ -1,5 +1,5 @@
/**
* Structured status payload emitted by helpers (`read`, `write`, `tree`, etc.) and the
* Structured status payload emitted by helpers (`read`, `write`, `env`, etc.) and the
* tool-call bridge. Surfaces to the model as part of `displays` so it has machine-readable
* context about what side effects happened.
*/
@@ -7,7 +7,7 @@ export interface SessionSnapshot {
sessionId: string;
/**
* On-disk roots the helpers substitute for internal-URL schemes
* (e.g. `{ local: "/…/artifacts/local" }`). Lets `read`/`write`/`append`
* (e.g. `{ local: "/…/artifacts/local" }`). Lets `read`/`write`
* accept `local://…` paths instead of writing a literal `local:/` directory.
*/
localRoots?: Record<string, string>;
@@ -17,8 +17,8 @@ describe("python prelude", () => {
expect(signature).toContain("limit");
});
it("exposes isolation artifacts on the agent() return_handle node", () => {
// agent(..., return_handle=True) is the only escape hatch for
it("exposes isolation artifacts on the agent() handle node", () => {
// agent(..., handle=True) is the only escape hatch for
// recovering apply=False patch/branch/nested artifacts (the bare
// schema return is just the parsed object), so the helper MUST
// translate the bridge's camelCase details onto the node — otherwise
@@ -71,7 +71,7 @@ export interface PythonExecutorOptions {
artifactPath?: string;
artifactId?: string;
/**
* On-disk roots the prelude helpers (`read`/`write`/`append`) substitute for
* On-disk roots the prelude helpers (`read`/`write`) substitute for
* internal-URL schemes (e.g. `{ local: "/…/artifacts/local" }`). Exported to
* the kernel as `PI_EVAL_LOCAL_ROOTS` (JSON) so `write("local://x")` lands
* where `read local://x` resolves instead of a literal `local:/` directory.
+9 -106
View File
@@ -115,103 +115,6 @@ if "__omp_prelude_loaded__" not in globals():
_emit_status("write", path=str(p), chars=len(content))
return p
def append(path: str | Path, content: str) -> Path:
"""Append to file."""
p = _resolve_omp_path(path)
p.parent.mkdir(parents=True, exist_ok=True)
with p.open("a", encoding="utf-8") as f:
f.write(content)
_emit_status("append", path=str(p), chars=len(content))
return p
def sort(text: str, *, reverse: bool = False, unique: bool = False) -> str:
"""Sort lines of text."""
lines = text.splitlines()
if unique:
lines = list(dict.fromkeys(lines))
lines = sorted(lines, reverse=reverse)
out = "\n".join(lines)
_emit_status("sort", lines=len(lines), unique=unique, reverse=reverse)
return out
def uniq(text: str, *, count: bool = False) -> str | list[tuple[int, str]]:
"""Remove duplicate adjacent lines (like uniq)."""
lines = text.splitlines()
if not lines:
_emit_status("uniq", groups=0)
return [] if count else ""
groups: list[tuple[int, str]] = []
current = lines[0]
current_count = 1
for line in lines[1:]:
if line == current:
current_count += 1
continue
groups.append((current_count, current))
current = line
current_count = 1
groups.append((current_count, current))
_emit_status("uniq", groups=len(groups), count_mode=count)
if count:
return groups
return "\n".join(line for _, line in groups)
def counter(
items: str | list,
*,
limit: int | None = None,
reverse: bool = True,
) -> list[tuple[int, str]]:
"""Count occurrences and sort by frequency. Like sort | uniq -c | sort -rn.
items: text (splits into lines) or list of strings
reverse: True for descending (most common first), False for ascending
Returns: [(count, item), ...] sorted by count
"""
from collections import Counter
if isinstance(items, str):
items = items.splitlines()
counts = Counter(items)
sorted_items = sorted(counts.items(), key=lambda x: (x[1], x[0]), reverse=reverse)
if limit is not None:
sorted_items = sorted_items[:limit]
result = [(count, item) for item, count in sorted_items]
_emit_status("counter", unique=len(counts), total=sum(counts.values()), top=result[:10])
return result
def tree(path: str | Path = ".", *, max_depth: int = 3, show_hidden: bool = False) -> str:
"""Return directory tree."""
base = Path(path)
lines = []
def walk(p: Path, prefix: str, depth: int):
if depth > max_depth:
return
items = sorted(p.iterdir(), key=lambda x: (not x.is_dir(), x.name.lower()))
items = [i for i in items if show_hidden or not i.name.startswith(".")]
for i, item in enumerate(items):
is_last = i == len(items) - 1
connector = "└── " if is_last else "├── "
suffix = "/" if item.is_dir() else ""
lines.append(f"{prefix}{connector}{item.name}{suffix}")
if item.is_dir():
ext = " " if is_last else "│ "
walk(item, prefix + ext, depth + 1)
lines.append(str(base) + "/")
walk(base, "", 1)
out = "\n".join(lines)
_emit_status("tree", path=str(base), entries=len(lines) - 1, preview=out[:1000])
return out
def diff(a: str | Path, b: str | Path) -> str:
"""Compare two files, return unified diff."""
import difflib
path_a, path_b = Path(a), Path(b)
lines_a = path_a.read_text(encoding="utf-8").splitlines(keepends=True)
lines_b = path_b.read_text(encoding="utf-8").splitlines(keepends=True)
result = difflib.unified_diff(lines_a, lines_b, fromfile=str(path_a), tofile=str(path_b))
out = "".join(result)
_emit_status("diff", file_a=str(path_a), file_b=str(path_b), identical=not out, preview=out[:500])
return out
def output(
*ids: str,
format: str = "raw",
@@ -520,10 +423,10 @@ if "__omp_prelude_loaded__" not in globals():
text = res.get("text") if isinstance(res, dict) else res
return json.loads(text) if schema is not None else text
def agent(prompt, *, agent_type="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, return_handle=False):
def agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False):
"""Run a subagent and return its final output.
`agent_type` selects the subagent definition (default "task"). Pass
`agent` selects the subagent definition (default "task"). Pass
`model` to override that agent's model, `label` for the output artifact
id, and `schema` to request structured JSON output; when `schema` is
supplied the parsed object is returned. Share background by writing a
@@ -539,13 +442,13 @@ if "__omp_prelude_loaded__" not in globals():
When isolated, `apply=False` keeps captured changes inside the
worktree and surfaces the root patch path, branch name, and nested
repository patches through the DAG node dict (combine with
`return_handle=True` to receive them — see below; the bare return type
`handle=True` to receive them — see below; the bare return type
stays bytes/string/parsed object and has nowhere to expose artifacts).
`merge=False` forces patch mode even when `task.isolation.merge` is
`"branch"`, avoiding the per-call git lock + repo mutation that branch
mode performs.
Set `return_handle=True` to receive a DAG node dict instead of bare
Set `handle=True` to receive a DAG node dict instead of bare
text: ``{"text", "output", "handle", "id", "agent"}`` where ``handle``
is the spawned agent's recoverable ``agent://<id>`` URI. A downstream
``pipeline``/``parallel`` stage embeds that ``handle`` (or ``output``)
@@ -560,8 +463,8 @@ if "__omp_prelude_loaded__" not in globals():
``handle=None`` — the helper never throws.
"""
args = {"prompt": prompt}
if agent_type is not None:
args["agentType"] = agent_type
if agent is not None:
args["agent"] = agent
if model is not None:
args["model"] = model
if label is not None:
@@ -574,12 +477,12 @@ if "__omp_prelude_loaded__" not in globals():
args["apply"] = bool(apply)
if merge is not None:
args["merge"] = bool(merge)
if return_handle:
args["returnHandle"] = True
if handle:
args["handle"] = True
res = _bridge_call("__agent__", args)
text = res.get("text") if isinstance(res, dict) else res
parsed = json.loads(text) if schema is not None else text
if not return_handle:
if not handle:
return parsed
details = res.get("details") if isinstance(res, dict) else None
if not isinstance(details, dict) or details.get("id") is None:
@@ -557,12 +557,6 @@ def _magic_run(args: str) -> None:
def _magic_cell_bash(args: str, body: str) -> int:
return _run_shell_body(body, shell_arg="/bin/bash")
@cell_magic("sh")
def _magic_cell_sh(args: str, body: str) -> int:
return _run_shell_body(body, shell_arg="/bin/sh")
@cell_magic("capture")
def _magic_cell_capture(args: str, body: str) -> str:
"""Capture stdout/stderr of body; bind to ``args`` (a name) if provided."""
+5 -190
View File
@@ -2,7 +2,7 @@
# OMP Ruby prelude helpers (loaded once into the runner's TOPLEVEL_BINDING).
#
# Mirrors eval/py/prelude.py: defines the cross-runtime helper surface
# (display/read/write/append/tree/diff/env/output, the `tool` bridge proxy,
# (display/read/write/env/output, the `tool` bridge proxy,
# completion/agent/parallel/pipeline/log/phase/budget). Host-side helpers reach
# the coding-agent over the same loopback HTTP tool bridge the Python prelude
# uses (PI_TOOL_BRIDGE_URL/TOKEN/SESSION). Path helpers honor PI_EVAL_LOCAL_ROOTS
@@ -112,191 +112,6 @@ unless defined?($__omp_prelude_loaded) && $__omp_prelude_loaded
resolved.to_s
end
def append(path, content)
resolved = __omp_resolve_path(path)
require "fileutils"
FileUtils.mkdir_p(File.dirname(resolved.to_s))
File.open(resolved.to_s, "a") { |f| f.write(content.to_s) }
__omp_emit_status("append", "path" => resolved.to_s, "chars" => content.to_s.length)
resolved.to_s
end
def tree(path = ".", max_depth: 3, show_hidden: false)
base = path.to_s
lines = []
walk = lambda do |dir, prefix, depth|
return if depth > max_depth
entries = (Dir.children(dir) rescue [])
entries = entries.reject { |e| e.start_with?(".") } unless show_hidden
entries = entries.sort_by { |e| [File.directory?(File.join(dir, e)) ? 0 : 1, e.downcase] }
entries.each_with_index do |name, i|
full = File.join(dir, name)
is_last = i == entries.length - 1
is_dir = File.directory?(full)
lines << "#{prefix}#{is_last ? "└── " : "├── "}#{name}#{is_dir ? "/" : ""}"
walk.call(full, prefix + (is_last ? " " : "│ "), depth + 1) if is_dir
end
end
lines << "#{base}/"
walk.call(base, "", 1)
out = lines.join("\n")
__omp_emit_status("tree", "path" => base, "entries" => lines.length - 1, "preview" => __omp_scrub(out[0, 1000].to_s))
out
end
def diff(a, b)
pa = a.to_s
pb = b.to_s
lines_a = File.read(pa, encoding: Encoding::UTF_8).lines
lines_b = File.read(pb, encoding: Encoding::UTF_8).lines
out = __omp_unified_diff(lines_a, lines_b, pa, pb)
__omp_emit_status("diff", "file_a" => pa, "file_b" => pb, "identical" => out.empty?, "preview" => __omp_scrub(out[0, 500].to_s))
out
end
# LCS-based op list ([:equal/:delete/:insert, aIndex, bIndex]) per consumed line.
def __omp_diff_ops(a, b)
n = a.length
m = b.length
if n * m > 4_000_000
# Too large for the DP table — fall back to a coarse all-delete/all-insert.
ops = []
n.times { |i| ops << [:delete, i, 0] }
m.times { |j| ops << [:insert, n, j] }
return ops
end
dp = Array.new(n + 1) { Array.new(m + 1, 0) }
(n - 1).downto(0) do |i|
row = dp[i]
nrow = dp[i + 1]
(m - 1).downto(0) do |j|
row[j] = a[i] == b[j] ? nrow[j + 1] + 1 : (nrow[j] >= row[j + 1] ? nrow[j] : row[j + 1])
end
end
ops = []
i = 0
j = 0
while i < n && j < m
if a[i] == b[j]
ops << [:equal, i, j]; i += 1; j += 1
elsif dp[i + 1][j] >= dp[i][j + 1]
ops << [:delete, i, j]; i += 1
else
ops << [:insert, i, j]; j += 1
end
end
ops << [:delete, i, j].tap { i += 1 } while i < n
ops << [:insert, i, j].tap { j += 1 } while j < m
ops
end
def __omp_unified_diff(a, b, from_file, to_file, context = 3)
ops = __omp_diff_ops(a, b)
return "" unless ops.any? { |tag, _, _| tag != :equal }
entries = ops.map do |tag, ai, bi|
{ tag: tag, ai: ai, bi: bi, text: (tag == :insert ? b[bi] : a[ai]) }
end
changed = entries.each_index.select { |k| entries[k][:tag] != :equal }
groups = []
start = nil
prev = nil
changed.each do |k|
if start.nil?
start = k
prev = k
elsif k - prev <= (2 * context) + 1
prev = k
else
groups << [start, prev]
start = k
prev = k
end
end
groups << [start, prev] unless start.nil?
out = +""
out << "--- #{from_file}\n"
out << "+++ #{to_file}\n"
groups.each do |gs, ge|
lo = [gs - context, 0].max
hi = [ge + context, entries.length - 1].min
slice = entries[lo..hi]
a_start = nil
a_count = 0
b_start = nil
b_count = 0
slice.each do |e|
if e[:tag] != :insert
a_start ||= e[:ai]
a_count += 1
end
if e[:tag] != :delete
b_start ||= e[:bi]
b_count += 1
end
end
out << "@@ -#{(a_start || 0) + 1},#{a_count} +#{(b_start || 0) + 1},#{b_count} @@\n"
slice.each do |e|
prefix = e[:tag] == :equal ? " " : (e[:tag] == :delete ? "-" : "+")
text = e[:text].to_s
text = "#{text}\n" unless text.end_with?("\n")
out << "#{prefix}#{text}"
end
end
out
end
# -------------------------------------------------------------------------
# Text helpers (sort / uniq / counter)
# -------------------------------------------------------------------------
def sort(text, reverse: false, unique: false)
lines = text.to_s.lines.map(&:chomp)
lines = lines.uniq if unique
lines = lines.sort
lines = lines.reverse if reverse
out = lines.join("\n")
__omp_emit_status("sort", "lines" => lines.length, "unique" => unique, "reverse" => reverse)
out
end
def uniq(text, count: false)
lines = text.to_s.lines.map(&:chomp)
if lines.empty?
__omp_emit_status("uniq", "groups" => 0)
return count ? [] : ""
end
groups = []
current = lines[0]
run = 1
lines[1..].each do |line|
if line == current
run += 1
else
groups << [run, current]
current = line
run = 1
end
end
groups << [run, current]
__omp_emit_status("uniq", "groups" => groups.length, "count_mode" => count)
count ? groups : groups.map { |_, l| l }.join("\n")
end
def counter(items, limit: nil, reverse: true)
arr = items.is_a?(String) ? items.lines.map(&:chomp) : items.to_a
counts = Hash.new(0)
arr.each { |i| counts[i] += 1 }
sorted = counts.sort_by { |item, c| [c, item] }
sorted = sorted.reverse if reverse
sorted = sorted.first(limit) if limit
result = sorted.map { |item, c| [c, item] }
__omp_emit_status("counter", "unique" => counts.size, "total" => arr.length, "top" => result.first(10))
result
end
# -------------------------------------------------------------------------
# Task/agent output reader
# -------------------------------------------------------------------------
@@ -577,9 +392,9 @@ unless defined?($__omp_prelude_loaded) && $__omp_prelude_loaded
schema.nil? ? text : JSON.parse(text)
end
def agent(prompt, agent_type: "task", model: nil, label: nil, schema: nil, isolated: nil, apply: nil, merge: nil, return_handle: false)
def agent(prompt, agent: "task", model: nil, label: nil, schema: nil, isolated: nil, apply: nil, merge: nil, handle: false)
args = { "prompt" => prompt }
args["agentType"] = agent_type unless agent_type.nil?
args["agent"] = agent unless agent.nil?
args["model"] = model unless model.nil?
args["label"] = label unless label.nil?
args["schema"] = schema unless schema.nil?
@@ -589,11 +404,11 @@ unless defined?($__omp_prelude_loaded) && $__omp_prelude_loaded
args["apply"] = !!apply unless apply.nil?
args["merge"] = !!merge unless merge.nil?
# Tell the bridge a handle is wanted so it preserves the backing artifacts.
args["returnHandle"] = true if return_handle
args["handle"] = true if handle
res = OmpBridge.call("__agent__", args)
text = res.is_a?(Hash) ? res["text"] : res
parsed = schema.nil? ? text : JSON.parse(text)
return parsed unless return_handle
return parsed unless handle
details = res.is_a?(Hash) ? res["details"] : nil
if !details.is_a?(Hash) || details["id"].nil?
return { "text" => text, "output" => text, "handle" => nil, "id" => nil, "agent" => nil }
+116 -9
View File
@@ -129,10 +129,122 @@ def __omp_emit_status(op, data = {})
__omp_emit_display({ "application/x-omp-status" => status }, "display")
end
OMP_IMAGE_MIMES = %w[image/png image/jpeg].freeze
# True when `str` already looks like base64 text (ASCII, base64 alphabet, length
# a multiple of 4). Raw image blobs (PNG/JPEG bytes) contain high bytes, so they
# fail the ASCII check and get encoded instead of passed through unchanged.
def __omp_base64?(str)
s = str.to_s
# ascii_only? is safe on any encoding (no regex over invalid bytes). Raw image
# blobs carry high bytes and fail here, so they get encoded rather than scanned.
return false unless s.ascii_only?
stripped = s.gsub(/\s+/, "")
return false if stripped.empty? || (stripped.bytesize % 4) != 0
stripped.match?(%r{\A[A-Za-z0-9+/]*={0,2}\z})
end
# Coerce an image payload to the base64 ASCII the host renders. IRuby-style
# `to_iruby` hands back raw binary blobs (Gruff#to_blob, ChunkyPNG, RMagick),
# which would also break JSON.generate; strict-encode them unless already base64.
def __omp_image_payload(content)
require "base64"
s = content.to_s
return s.gsub(/\s+/, "") if __omp_base64?(s)
Base64.strict_encode64(s.b)
end
# Detect a host-renderable image MIME from a binary blob's magic bytes. Lets us
# treat the generic `to_blob` (Gruff/RMagick/ChunkyPNG/Vips) as an image only
# when it really is one, avoiding false positives on unrelated `to_blob` methods.
def __omp_sniff_image_mime(bytes)
b = bytes.to_s.b
return "image/png" if b.start_with?("\x89PNG\r\n\x1a\n".b)
return "image/jpeg" if b.start_with?("\xFF\xD8\xFF".b)
nil
end
# Stringify keys, base64-encode image payloads, and scrub text payloads so the
# bundle is always JSON-safe before it reaches __omp_emit.
def __omp_normalize_bundle(hash)
bundle = {}
hash.each do |key, val|
k = key.to_s
bundle[k] =
if OMP_IMAGE_MIMES.include?(k)
__omp_image_payload(val)
elsif val.is_a?(String)
__omp_scrub(val)
else
val
end
end
bundle
end
# Guarantee a text/plain entry so the model always sees a textual hint, even for
# image-only bundles (mirrors the Python runner).
def __omp_finalize_bundle(bundle, value)
bundle["text/plain"] ||= __omp_scrub((value.inspect rescue value.class.name))
bundle
end
# Rich-display resolution for non-collection objects. Honors the repo
# `to_omp_mime` convention first, then the IRuby protocol
# (`to_iruby_mimebundle` -> [data, metadata], `to_iruby` -> [mime, data]) so plot
# and image objects (gruff, rubyplot, gnuplotrb, chunky_png, daru, ...) render
# inline — the Ruby analog of IPython's _repr_*_ methods. Returns nil when the
# value advertises no rich representation.
def __omp_rich_mime_bundle(value)
if value.respond_to?(:to_omp_mime)
mime = (value.to_omp_mime rescue nil)
return __omp_finalize_bundle(__omp_normalize_bundle(mime), value) if mime.is_a?(Hash) && !mime.empty?
end
if value.respond_to?(:to_iruby_mimebundle)
data =
begin
value.to_iruby_mimebundle
rescue ArgumentError
(value.to_iruby_mimebundle(include: []) rescue nil)
rescue StandardError
nil
end
data = data.first if data.is_a?(Array)
return __omp_finalize_bundle(__omp_normalize_bundle(data), value) if data.is_a?(Hash) && !data.empty?
end
if value.respond_to?(:to_iruby)
pair = (value.to_iruby rescue nil)
if pair.is_a?(Array) && pair.size == 2 && !pair[0].nil?
return __omp_finalize_bundle(__omp_normalize_bundle({ pair[0].to_s => pair[1] }), value)
end
end
# Last resort: probe well-known image emitters. Named methods (to_png/to_jpeg)
# are trusted; the generic to_blob is accepted only when its bytes sniff as an
# image. Covers gems that render via IRuby's registry rather than to_iruby
# (Gruff#to_blob, ChunkyPNG#to_blob, RMagick, Vips, ...).
if value.respond_to?(:to_png)
png = (value.to_png rescue nil)
return __omp_finalize_bundle({ "image/png" => __omp_image_payload(png) }, value) if png
end
jpeg_method = %i[to_jpeg to_jpg].find { |m| value.respond_to?(m) }
if jpeg_method
jpg = (value.public_send(jpeg_method) rescue nil)
return __omp_finalize_bundle({ "image/jpeg" => __omp_image_payload(jpg) }, value) if jpg
end
if value.respond_to?(:to_blob)
blob = (value.to_blob rescue nil)
if blob.is_a?(String) && (mime = __omp_sniff_image_mime(blob))
return __omp_finalize_bundle({ mime => __omp_image_payload(blob) }, value)
end
end
nil
end
# Build a Jupyter-style MIME bundle for a value. Strings render as plain text,
# Hash/Array render as JSON (plus a text/plain repr) so the model sees structure,
# and anything else falls back to its inspect string. Objects may opt into a
# richer bundle by defining `to_omp_mime` returning a Hash of mime => value.
# Hash/Array render as JSON (plus a text/plain repr) so the model sees structure.
# Other objects may expose a rich representation via `to_omp_mime` or the IRuby
# protocol (`to_iruby`/`to_iruby_mimebundle`); otherwise they fall back to inspect.
def __omp_mime_bundle(value)
case value
when String
@@ -151,12 +263,7 @@ def __omp_mime_bundle(value)
when nil
{ "text/plain" => "nil" }
else
if value.respond_to?(:to_omp_mime)
mime = (value.to_omp_mime rescue nil)
mime.is_a?(Hash) ? mime : { "text/plain" => __omp_scrub(value.inspect) }
else
{ "text/plain" => __omp_scrub(value.inspect) }
end
__omp_rich_mime_bundle(value) || { "text/plain" => __omp_scrub(value.inspect) }
end
end
@@ -157,7 +157,7 @@ export function resolveLocalUrlToPath(
}
/**
* On-disk roots the eval helpers (`read`/`write`/`append`) substitute for
* On-disk roots the eval helpers (`read`/`write`) substitute for
* internal-URL schemes so e.g. `write("local://x.md")` lands where a later
* `read local://x.md` resolves — instead of a literal `local:/` directory under
* the cwd (a stdlib `pathlib.Path`/`path.resolve` collapses `local://` to
@@ -169,6 +169,96 @@ export function buildEvalUrlRoots(options: LocalProtocolOptions): Record<string,
return { local: resolveLocalRoot(options) };
}
const LOCAL_WRITE_NOTE = "Use write path local://<file> to persist large intermediate artifacts across turns.";
type ResolvedLocalTarget =
| { kind: "listing"; root: string }
| { kind: "directory"; path: string }
| { kind: "file"; path: string; size: number };
/**
* Resolve a local:// URL to its on-disk target with realpath + containment
* checks on the root, parent, and target so symlinks cannot escape the session
* local root. Does NOT read or decode file contents — callers decide how to
* consume the resolved path. Shared by {@link LocalProtocolHandler.resolve} and
* {@link resolveLocalUrlToFile}.
*/
async function resolveLocalTarget(url: InternalUrl, opts: LocalProtocolOptions): Promise<ResolvedLocalTarget> {
const localRoot = path.resolve(resolveLocalRoot(opts));
await fs.mkdir(localRoot, { recursive: true });
let resolvedRoot: string;
try {
resolvedRoot = await fs.realpath(localRoot);
} catch (error) {
if (isEnoent(error)) {
throw new Error("Unable to initialize local:// root");
}
throw error;
}
const relativePath = extractRelativePath(url);
const targetPath = relativePath ? path.resolve(resolvedRoot, relativePath) : resolvedRoot;
ensureWithinRoot(targetPath, resolvedRoot);
if (targetPath === resolvedRoot) {
return { kind: "listing", root: resolvedRoot };
}
const parentDir = path.dirname(targetPath);
try {
const realParent = await fs.realpath(parentDir);
ensureWithinRoot(realParent, resolvedRoot);
} catch (error) {
if (!isEnoent(error)) throw error;
}
let realTargetPath: string;
try {
realTargetPath = await fs.realpath(targetPath);
} catch (error) {
if (isEnoent(error)) {
throw new Error(`Local file not found: ${url.href}`);
}
throw error;
}
ensureWithinRoot(realTargetPath, resolvedRoot);
const stat = await fs.stat(realTargetPath);
if (stat.isDirectory()) {
return { kind: "directory", path: realTargetPath };
}
if (!stat.isFile()) {
throw new Error(`local:// URL must resolve to a file or directory: ${url.href}`);
}
return { kind: "file", path: realTargetPath, size: stat.size };
}
/**
* Resolve a local:// URL to a regular on-disk file, applying the same
* realpath + containment guarantees as {@link LocalProtocolHandler.resolve}
* but WITHOUT reading or UTF-8-decoding its contents. Returns null when there
* is no active session or when the URL targets the root listing or a directory;
* throws the handler's not-found and "escapes local root" errors for missing
* files and symlink escapes.
*
* Options are resolved via {@link LocalProtocolHandler.resolveOptions} so the
* caller-options → override → registry order matches router resolution exactly.
* The read tool uses this to detect and emit image files from their real path
* before the text-only resource contract would decode the binary into mojibake.
*/
export async function resolveLocalUrlToFile(
input: string | InternalUrl,
context?: ResolveContext,
): Promise<{ path: string; size: number } | null> {
const opts = LocalProtocolHandler.resolveOptions(context);
if (!opts) return null;
const url = typeof input === "string" ? parseLocalUrl(input) : input;
const resolved = await resolveLocalTarget(url, opts);
return resolved.kind === "file" ? { path: resolved.path, size: resolved.size } : null;
}
/**
* Protocol handler for local:// URLs.
*
@@ -238,65 +328,22 @@ export class LocalProtocolHandler implements ProtocolHandler {
throw new Error("No session - local:// unavailable");
}
const localRoot = path.resolve(resolveLocalRoot(opts));
await fs.mkdir(localRoot, { recursive: true });
let resolvedRoot: string;
try {
resolvedRoot = await fs.realpath(localRoot);
} catch (error) {
if (isEnoent(error)) {
throw new Error("Unable to initialize local:// root");
}
throw error;
const resolved = await resolveLocalTarget(url, opts);
if (resolved.kind === "listing") {
return buildListing(url, resolved.root);
}
if (resolved.kind === "directory") {
return buildDirectoryResource(url.href, resolved.path, [LOCAL_WRITE_NOTE]);
}
const relativePath = extractRelativePath(url);
const targetPath = relativePath ? path.resolve(resolvedRoot, relativePath) : resolvedRoot;
ensureWithinRoot(targetPath, resolvedRoot);
if (targetPath === resolvedRoot) {
return buildListing(url, resolvedRoot);
}
const parentDir = path.dirname(targetPath);
try {
const realParent = await fs.realpath(parentDir);
ensureWithinRoot(realParent, resolvedRoot);
} catch (error) {
if (!isEnoent(error)) throw error;
}
let realTargetPath: string;
try {
realTargetPath = await fs.realpath(targetPath);
} catch (error) {
if (isEnoent(error)) {
throw new Error(`Local file not found: ${url.href}`);
}
throw error;
}
ensureWithinRoot(realTargetPath, resolvedRoot);
const stat = await fs.stat(realTargetPath);
if (stat.isDirectory()) {
return buildDirectoryResource(url.href, realTargetPath, [
"Use write path local://<file> to persist large intermediate artifacts across turns.",
]);
}
if (!stat.isFile()) {
throw new Error(`local:// URL must resolve to a file or directory: ${url.href}`);
}
const content = await Bun.file(realTargetPath).text();
const content = await Bun.file(resolved.path).text();
return {
url: url.href,
content,
contentType: getContentType(realTargetPath),
contentType: getContentType(resolved.path),
size: Buffer.byteLength(content, "utf-8"),
sourcePath: realTargetPath,
notes: ["Use write path local://<file> to persist large intermediate artifacts across turns."],
sourcePath: resolved.path,
notes: [LOCAL_WRITE_NOTE],
};
}
+15 -4
View File
@@ -945,6 +945,7 @@ async function buildSessionOptions(
interface RunRootCommandDependencies {
createAgentSession?: typeof createAgentSession;
discoverAuthStorage?: typeof discoverAuthStorage;
selectSession?: typeof selectSession;
runAcpMode?: RunAcpMode;
settings?: Settings;
forceSetupWizard?: boolean;
@@ -1131,7 +1132,8 @@ export async function runRootCommand(
// (see issue #1668).
if (typeof parsedArgs.resume === "string" && !sessionManager) {
writeStartupNotice(parsedArgs, `${chalk.dim("Resume cancelled: session is in another project.")}\n`);
return;
stopStartupWatchdog();
process.exit(0);
}
// Handle --resume (no value): show session picker
@@ -1147,17 +1149,26 @@ export async function runRootCommand(
preloadedAllSessions = await logger.time("SessionManager.listAll", SessionManager.listAll);
if (preloadedAllSessions.length === 0) {
writeStartupNotice(parsedArgs, `${chalk.dim("No sessions found")}\n`);
return;
stopStartupWatchdog();
process.exit(0);
}
}
pauseStartupWatchdog();
const selected = await logger.time("selectSession", selectSession, folderSessions, {
const selected = await logger.time("selectSession", deps.selectSession ?? selectSession, folderSessions, {
allSessions: preloadedAllSessions,
});
resumeStartupWatchdog();
if (!selected) {
writeStartupNotice(parsedArgs, `${chalk.dim("No session selected")}\n`);
return;
// Quit instead of returning: startup already armed long-lived handles
// (theme watcher + SIGWINCH/macOS appearance listeners via initTheme,
// settings save timer, model registry) that keep the event loop alive,
// so a bare return hangs the process after the picker leaves the alt
// screen. No session was built here, so there is nothing to flush. The
// in-session `/resume` picker (selector-controller.ts) takes a different
// onCancel that just closes the overlay — only this startup path exits.
stopStartupWatchdog();
process.exit(0);
}
// Resuming a session from another project: switch the process into that
// project's directory and refresh cwd-derived caches before the session is
@@ -468,8 +468,13 @@ function buildEvalStartText(args: unknown): string | undefined {
if (typeof args !== "object" || args === null || Array.isArray(args)) {
return undefined;
}
const cells = (args as EvalCellContainer).cells;
if (!Array.isArray(cells) || cells.length === 0) {
const container = args as EvalCellContainer & EvalCellLike;
const cells = Array.isArray(container.cells)
? container.cells
: typeof container.code === "string"
? [container]
: [];
if (cells.length === 0) {
return undefined;
}
const lines: string[] = [];
@@ -11,6 +11,7 @@
import {
Container,
Input,
matchesKey,
type SelectItem,
SelectList,
type SettingItem,
@@ -36,13 +37,18 @@ import { DynamicBorder } from "./dynamic-border";
/**
* Forwards a keystroke to `input`, but cancels via `onCancel` when the user presses Escape.
*
* Escape is decoded via `matchesKey` rather than a raw `\x1b` compare: inside the
* fullscreen settings overlay the kitty keyboard protocol is active (ghostty/kitty),
* where the Escape key arrives as the CSI-u sequence `\x1b[27u`, not a bare `\x1b`.
* The literal fallbacks preserve legacy single/double-escape on terminals without it.
*/
export function handleInputOrEscape(
data: string,
input: { handleInput(data: string): void },
onCancel: () => void,
): void {
if (data === "\x1b" || data === "\x1b\x1b") {
if (data === "\x1b" || data === "\x1b\x1b" || matchesKey(data, "escape")) {
onCancel();
return;
}
@@ -5,6 +5,7 @@ import {
Input,
matchesKey,
padding,
parseSgrMouse,
replaceTabs,
ScrollView,
Spacer,
@@ -161,6 +162,12 @@ export function mergeSessionRanking(
class SessionList implements Component {
#filteredSessions: SessionInfo[] = [];
#selectedIndex: number = 0;
// Maps a 0-based line within this list's own render to a filtered-session
// index, or undefined for chrome rows (search line, blanks, scrollbar gap).
// Rebuilt every render so the picker's mouse hit-testing tracks the live
// scroll window. Only consulted while the picker holds the alternate screen
// (where the overlay enables mouse tracking and paints from screen row 0).
#hitRows: (number | undefined)[] = [];
readonly #searchInput: Input;
onSelect?: (session: SessionInfo) => void;
onCancel?: () => void;
@@ -257,12 +264,32 @@ class SessionList implements Component {
}
}
/** Resolve a list-local rendered-line index to a filtered-session index. */
hitTestSession(line: number): number | undefined {
return this.#hitRows[line];
}
/** Wheel notch: move the selection one step (clamped, no wrap). */
handleWheel(delta: -1 | 1): void {
if (this.#filteredSessions.length === 0) return;
this.#selectedIndex = Math.max(0, Math.min(this.#filteredSessions.length - 1, this.#selectedIndex + delta));
}
/** Mouse click: select the session under the pointer and resume it. */
selectAndConfirm(index: number): void {
const session = this.#filteredSessions[index];
if (!session) return;
this.#selectedIndex = index;
this.onSelect?.(session);
}
invalidate(): void {
// No cached state to invalidate currently
}
render(width: number): readonly string[] {
const lines: string[] = [];
this.#hitRows = [];
// Render search input
lines.push(...this.#searchInput.render(width));
@@ -311,9 +338,11 @@ class SessionList implements Component {
// Each session block is built into sessionLines, then wrapped by ScrollView
// so the right-edge scrollbar is proportional at the physical-line level.
const sessionLines: string[] = [];
const sessionRowIndex: number[] = [];
const overflow = this.#filteredSessions.length > maxVisible;
const rowWidth = Math.max(0, width - (overflow ? 1 : 0));
for (let i = startIndex; i < endIndex; i++) {
const blockStart = sessionLines.length;
const session = this.#filteredSessions[i];
const isSelected = i === this.#selectedIndex;
@@ -363,6 +392,7 @@ class SessionList implements Component {
sessionLines.push(metadataLine);
sessionLines.push(""); // Blank line between sessions
for (let k = blockStart; k < sessionLines.length; k++) sessionRowIndex[k] = i;
}
// Wrap the rendered window in a ScrollView for a proportional right-edge bar.
@@ -375,16 +405,10 @@ class SessionList implements Component {
theme: { track: t => theme.fg("muted", t), thumb: t => theme.fg("accent", t) },
});
sv.setScrollOffset(Math.round(startIndex * linesPerItem));
lines.push(...sv.render(width));
// Add keybinding hint
lines.push("");
lines.push(
theme.fg(
"muted",
` [Del delete · Enter select · Tab ${this.#showCwd ? "current folder" : "all projects"} · Esc cancel]`,
),
);
const sessionRegionStart = lines.length;
const svLines = sv.render(width);
for (let k = 0; k < svLines.length; k++) this.#hitRows[sessionRegionStart + k] = sessionRowIndex[k];
lines.push(...svLines);
return lines;
}
@@ -462,6 +486,13 @@ export interface SessionSelectorOptions {
* Omitted only in tests; defaults to a conservative 24 rows.
*/
getTerminalRows?: () => number;
/**
* Fill the whole viewport and pin the footer (hint + bottom border) to the
* last rows, so the footer stops drifting as the list window changes height.
* Set by the standalone `--resume` picker (fullscreen alternate screen); the
* in-editor selector leaves it off and renders compactly.
*/
fillHeight?: boolean;
}
/**
@@ -470,6 +501,13 @@ export interface SessionSelectorOptions {
export class SessionSelectorComponent extends Container {
#sessionList: SessionList;
#confirmationDialog: HookSelectorComponent | null = null;
// Hosts whichever of `#sessionList` / `#confirmationDialog` is live this
// frame. The delete dialog REPLACES the list in this slot rather than being
// appended below the picker chrome, so the picker is always
// `chrome + max(list, dialog) + chrome` and never overflows the viewport
// (issue #3283: an overflowing dialog frame committed the header into
// scrollback, stranding it above the viewport once the dialog closed).
#contentSlot: Container;
#messageContainer: Container;
#headerText: Text;
#onDelete?: (session: SessionInfo) => Promise<boolean>;
@@ -479,6 +517,18 @@ export class SessionSelectorComponent extends Container {
#globalSessions: SessionInfo[] | null = null;
#scope: "folder" | "all" = "folder";
#toggling = false;
// 0-based line where the session list begins within this component's own
// render, captured each frame. The fullscreen picker overlay paints from
// screen row 0, so a mouse row maps to `row - #listLineOffset` inside the
// list. Only meaningful while the picker holds the alternate screen.
#listLineOffset = 0;
// 0-based line where the pinned footer begins; clicks at or below it never
// hit-test the list, so a footer click on a cramped (trimmed) frame can't
// resume a session scrolled off-screen.
#footerStart = 0;
readonly #getTerminalRows: () => number;
readonly #fillHeight: boolean;
readonly #bottomBorder = new DynamicBorder();
constructor(
sessions: SessionInfo[],
@@ -494,6 +544,8 @@ export class SessionSelectorComponent extends Container {
this.#loadAllSessions = options.loadAllSessions;
this.#folderSessions = sessions;
this.#globalSessions = options.allSessions ?? null;
this.#getTerminalRows = options.getTerminalRows ?? (() => 24);
this.#fillHeight = options.fillHeight ?? false;
// Add header
this.addChild(new Spacer(1));
this.#headerText = new Text(this.#headerLabel(), 1, 0);
@@ -517,11 +569,9 @@ export class SessionSelectorComponent extends Container {
void this.#toggleScope();
};
}
this.addChild(this.#sessionList);
// Add bottom border
this.addChild(new Spacer(1));
this.addChild(new DynamicBorder());
this.#contentSlot = new Container();
this.#contentSlot.addChild(this.#sessionList);
this.addChild(this.#contentSlot);
}
#headerLabel(): string {
@@ -582,6 +632,15 @@ export class SessionSelectorComponent extends Container {
#showDeleteConfirmation(session: SessionInfo): void {
const displayName = session.title || session.firstMessage.slice(0, 40) || session.id;
const closeDialog = () => {
this.#confirmationDialog = null;
// Restore the SessionList into the content slot so the picker is back
// to its normal layout on the very next render — the same frame the
// dialog disappears.
this.#contentSlot.clear();
this.#contentSlot.addChild(this.#sessionList);
this.#onRequestRender?.();
};
this.#confirmationDialog = new HookSelectorComponent(
`Delete session?\n${displayName}`,
["Yes", "No"],
@@ -597,25 +656,61 @@ export class SessionSelectorComponent extends Container {
this.#showError(err instanceof Error ? err.message : String(err));
}
}
// Close confirmation dialog
this.removeChild(this.#confirmationDialog!);
this.#confirmationDialog = null;
// Request rerender
this.#onRequestRender?.();
},
() => {
// Cancel - close confirmation dialog
this.removeChild(this.#confirmationDialog!);
this.#confirmationDialog = null;
// Request rerender
this.#onRequestRender?.();
closeDialog();
},
closeDialog,
);
// Show confirmation dialog
this.addChild(this.#confirmationDialog);
// Swap the SessionList out of the content slot and mount the dialog in its
// place: the dialog competes only with the SessionList's rendered budget,
// never the SessionList AND the picker chrome, so the picker frame stays
// inside the terminal viewport and the TUI never commits the header into
// scrollback (issue #3283).
this.#contentSlot.clear();
this.#contentSlot.addChild(this.#confirmationDialog);
this.#onRequestRender?.();
}
/**
* Concatenate the children's renders (like {@link Container}) while recording
* the line where the session list begins, so the fullscreen picker can hit-
* test mouse rows against the live list window. SessionList rebuilds its lines
* every frame, so Container's reference-memoization never applied here.
*
* In fill-height mode the body is padded (or, on a cramped terminal, trimmed)
* to leave exactly enough room for the footer at the screen bottom, so the
* footer is always visible and never drifts as the list window resizes. The
* in-editor selector just appends the footer directly.
*/
render(width: number): readonly string[] {
const lines: string[] = [];
for (const child of this.children) {
const childLines = child.render(width);
if (child === this.#contentSlot) this.#listLineOffset = lines.length;
for (const line of childLines) lines.push(line);
}
const footer = this.#footerLines(width);
if (this.#fillHeight) {
const target = Math.max(0, this.#getTerminalRows() - footer.length);
if (lines.length > target) lines.length = target;
else for (let i = lines.length; i < target; i++) lines.push("");
}
this.#footerStart = lines.length;
for (const line of footer) lines.push(line);
return lines;
}
/** Blank · keybinding hint · bottom border. Rendered by {@link render}. */
#footerLines(width: number): string[] {
const scopeHint = this.#scope === "all" ? "current folder" : "all projects";
const hint = theme.fg("muted", ` [Del delete · Enter select · Tab ${scopeHint} · Esc cancel]`);
return ["", hint, "", ...this.#bottomBorder.render(width)];
}
handleInput(keyData: string): void {
if (keyData.startsWith("\x1b[<")) {
this.#handleMouse(keyData);
return;
}
if (this.#confirmationDialog) {
this.#confirmationDialog.handleInput(keyData);
} else {
@@ -623,6 +718,25 @@ export class SessionSelectorComponent extends Container {
}
}
/**
* SGR mouse reports, delivered only while the picker holds the alternate
* screen (the fullscreen overlay enables tracking and paints from screen row
* 0). Wheel scrolls the list; a left click resumes the session under the
* pointer. Mouse is inert while the delete-confirmation dialog is open.
*/
#handleMouse(data: string): void {
if (this.#confirmationDialog) return;
const event = parseSgrMouse(data);
if (!event) return;
if (event.wheel !== null) {
this.#sessionList.handleWheel(event.wheel);
return;
}
if (!event.leftClick || event.row >= this.#footerStart) return;
const index = this.#sessionList.hitTestSession(event.row - this.#listLineOffset);
if (index !== undefined) this.#sessionList.selectAndConfirm(index);
}
getSessionList(): SessionList {
return this.#sessionList;
}
@@ -127,8 +127,13 @@ export function extractQuoteBlocks(text: string): QuoteBlock[] {
function extractEvalCode(args: unknown): { code: string; language: string } | undefined {
if (!args || typeof args !== "object") return undefined;
const cells = (args as { cells?: unknown }).cells;
if (!Array.isArray(cells)) return undefined;
const argsObj = args as { cells?: unknown; code?: unknown };
const cells = Array.isArray(argsObj.cells)
? argsObj.cells
: typeof argsObj.code === "string"
? [argsObj]
: undefined;
if (!cells) return undefined;
const codeBlocks: string[] = [];
let language = "python";
@@ -109,9 +109,9 @@ You MUST use the specialized tool over its shell equivalent:
{{#has tools "lsp"}}- Code intelligence → `{{toolRefs.lsp}}`.{{/has}}
{{#has tools "search"}}- Regex search → `{{toolRefs.search}}`, not `grep`, `rg`, or `awk`.{{/has}}
{{#has tools "find"}}- Globbing → `{{toolRefs.find}}`, not `ls **/*.ext` or `fd`.{{/has}}
{{#has tools "eval"}}- Quick compute → `{{toolRefs.eval}}`; you SHOULD go step by step.{{/has}}
{{#has tools "bash"}}- Use `{{toolRefs.bash}}` for terminal work—builds, tests, git, package managers—and pipelines that COMPUTE a fact: `wc -l`, `sort | uniq -c`, `comm`, `diff a b`, checksums. Commands shadowing the tools above are blocked.
- Litmus: produces a count, frequency, set difference, or checksum no tool returns → bash. Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}}
{{#has tools "eval"}}- Default for any compute: `{{toolRefs.eval}}` cells. Bash is the EXCEPTION — only single binary calls or short fact-computing pipelines (`wc -l`, `sort | uniq -c`, `diff`, checksums). The moment a command grows a loop, conditional, heredoc, `-e`/`-c` script, `$(…)` nesting, or >2 pipe stages, it's a program → `{{toolRefs.eval}}`. NEVER write multiline or inline-script bash.{{/has}}
{{#has tools "bash"}}- `{{toolRefs.bash}}`: real binaries and short fact pipelines only. Commands shadowing the specialized tools above are blocked.{{/has}}
{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash.{{#has tools "eval"}} Needs control flow, state, or fights shell quoting → `{{toolRefs.eval}}`.{{/has}} Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}}
{{#has tools "report_tool_issue"}}
<critical>
@@ -11,16 +11,16 @@ Worth it when the task benefits from decomposition + parallel coverage, or from
</when>
<helpers>
State persists across cells, so scout in one cell and fan out in the next. Every cell has:
State persists across eval calls, so scout in one call and fan out in the next. Every eval call has:
- `agent(prompt, *, agent_type="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, return_handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `return_handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts.
- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts.
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`.
- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out.
- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it.
- `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget.
Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across cells and turns for multi-phase work, reading each result before you decide the next phase.
Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across calls and turns for multi-phase work, reading each result before you decide the next phase.
</helpers>
<structure>
@@ -1,5 +1,19 @@
Runs bash in a shell session — terminal ops: git, bun, cargo, python.
# When to use bash — and when not to
Bash invokes **real binaries** with simple args. It is NOT a scripting surface.
Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a fact (`wc -l`, `sort | uniq -c`, `comm`, `diff`, a checksum, `git status`).
Anything below → `eval` cell, not bash:
- Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language
- Heredocs (`<<EOF`), `while`/`for`/`if`/`case` shell control flow
- `$(…)` command substitution nested inside another command
- Pipelines with more than two stages, or stages that need control flow or quote/JSON escaping
- Multiline commands, `&&`-chains mixing control flow
- Quote/JSON escaping that fights the shell
<instruction>
- `cwd` sets the working dir, not `cd dir && …`
- `env: { NAME: "…" }` for multiline / quote-heavy / untrusted values; reference `$NAME`
@@ -14,7 +28,9 @@ Runs bash in a shell session — terminal ops: git, bun, cargo, python.
</instruction>
<critical>
- Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.
- NEVER shell out to search content or files: `grep/rg` → `search`.
- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `find` tool (globbing). This is non-negotiable, even for a single quick listing.
- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://<id>`.
</critical>
+19 -19
View File
@@ -1,21 +1,23 @@
Run code in a persistent kernel using a list of cells.
Run one step of code in a persistent kernel.
<instruction>
Cells run in array order. State persists per language across cells, tool calls, and `task` subagents — stage helpers/datasets/clients once, subagents reuse directly, no re-import/serialize.
**One eval call = one cell = one logical step.** State persists per language across separate eval calls, tool calls, and `task` subagents — define helpers/datasets/clients in one call, then later calls reuse them directly.
Cell fields:
Work incrementally: imports in one call, define in the next, test, then use — each its own eval call. Re-run setup ONLY after `reset`, a kernel crash, or a `NameError`/`ReferenceError` proving the state is gone. Parallelize work *within* a cell with the `parallel(thunks)` helper, not by batching steps.
Fields:
- `language` — {{#if py}}`"py"` IPython kernel{{/if}}{{#ifAll py js}}, {{/ifAll}}{{#if js}}`"js"` persistent JavaScript VM{{/if}}{{#if rb}}{{#ifAny py js}}, {{/ifAny}}`"rb"` persistent Ruby kernel{{/if}}{{#if jl}}{{#ifAny py js rb}}, {{/ifAny}}`"jl"` persistent Julia kernel{{/if}}.
- `code` — cell body, verbatim. Newlines/quotes JSON-encoded; no fences, no headers.
- `title` (optional) — short transcript label (e.g. `"imports"`).
- `timeout` (optional) — per-cell seconds. Raise only for heavy compute or long non-agent tool calls.
- `reset` (optional) — wipe this cell's language kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}}
- `timeout` (optional) — seconds. Raise only for heavy compute or long non-agent tool calls.
- `reset` (optional) — wipe this language's kernel first.{{#ifAll py js}} Per-language: a `py` reset never touches the JS VM.{{/ifAll}}
Work incrementally — one logical step per cell (imports, define, test, use), many small cells per call; workflow notes in the assistant message or `title`, never in cell code.
{{#if py}}Live event loop: use top-level `await` directly; `asyncio.run(…)` raises "cannot be called from a running event loop".{{/if}}
{{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `tree(".", max_depth: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}}
{{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `tree(max_depth=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}}
Errors name the failing cell ("Cell 3 failed") — resubmit the fixed cell + any remaining.
{{#if js}}JS runs under **Bun**: Bun globals/APIs are available (`Bun.file`, `Bun.write`, `Bun.$`, `fetch`, `Buffer`); top-level `await`/`return` work directly.{{/if}}
{{#if rb}}Ruby: synchronous; helper options are keyword args (e.g. `output("id", limit: 2)`); the last expression auto-displays unless it is `nil`, an assignment, or a definition (like IRB).{{/if}}
{{#if jl}}Julia: synchronous; helper options are standard keyword args (e.g. `output("id", limit=2)`); the last expression auto-displays unless it is an assignment or a definition (like the Julia REPL).{{/if}}
On error, fix and re-run only the failing step — prior calls' state survives.
</instruction>
<prelude>
@@ -29,12 +31,6 @@ read(path, offset?=1, limit?=None) → str
File as text; offset/limit 1-indexed lines. Accepts `local://…`.
write(path, content) → str
Write file (creates parents) → resolved path. `local://…` persists across turns/subagents.
append(path, content) → str
Append → resolved path. Accepts `local://…`.
tree(path?=".", max_depth?=3, show_hidden?=False) → str
Directory tree.
diff(a, b) → str
Unified diff of two files.
env(key?=None, value?=None) → str | None | dict
No args → full env dict; one → value of `key`; two → set `key=value`, return value.
output(*ids, format?="raw", query?=None, offset?=None, limit?=None) → str | dict | list[dict]
@@ -43,9 +39,9 @@ tool.<name>(args) → unknown
Invoke any session tool; `args` = its parameter object.
completion(prompt, model?="default", system?=None, schema?=None) → str | dict
Oneshot, stateless (no history/tools). `model`: "smol" fast | "default" session | "slow" most capable. `schema` (JSON-Schema) → structured output, parsed object.
{{#if spawns}}agent(prompt, agent_type?="task", model?=None, label?=None, schema?=None, return_handle?=False) → str | dict
Run a subagent → final output. `agent_type`/`agentType` picks another discovered agent; `schema` as in completion(). Background via `local://` files named in the prompt. `return_handle`/`returnHandle` → DAG node dict { text, output, handle: "agent://<id>", id, agent } (parsed under `data` when `schema` set).
{{#if js}} JS: options are ONE trailing object — agent(prompt, { agentType, schema, returnHandle }).
{{#if spawns}}agent(prompt, agent?="task", model?=None, label?=None, schema?=None, handle?=False) → str | dict
Run a subagent → final output. `agent` picks another discovered agent; `schema` as in completion(). Background via `local://` files named in the prompt. `handle` → DAG node dict { text, output, handle: "agent://<id>", id, agent } (parsed under `data` when `schema` set).
{{#if js}} JS: options are ONE trailing object — agent(prompt, { agent, schema, handle }).
{{/if}}
{{/if}}
parallel(thunks) → list
@@ -63,10 +59,14 @@ budget → per-turn token budget
{{#if spawns}}
<dag>
Pipe handles through stage helpers to build a dependency graph — acyclic waves:
- **Name nodes.** Capture each `agent(…, {{#if py}}return_handle=True{{/if}}{{#if js}}{ returnHandle: true }{{/if}}{{#if jl}}return_handle=true{{/if}})` result; carries `handle` (`agent://<id>`) + `output`.
- **Name nodes.** Capture each `agent(…, {{#if py}}handle=True{{/if}}{{#if js}}{ handle: true }{{/if}}{{#if jl}}handle=true{{/if}})` result; carries `handle` (`agent://<id>`) + `output`.
- **Wire edges by reference.** Put an upstream node's `handle`/`output` in the dependent stage's prompt — large transcript never re-inlined. Bulk: `write("local://<name>.md", …)`, pass the URI.
- **`pipeline(items, *stages)` = staged waves**, barrier between stages (every item clears stage N before any enters N+1). **`parallel(thunks)` = one wave** of independent nodes.
- **Isolate failure.** A raising node re-raises the lowest-index error, aborts its wave; wrap risky nodes in try/except so a failure degrades only its dependent subtree, independent branches finish.
- **Acyclic only.** A node never waits on its own descendant.
</dag>
{{/if}}
<critical>
Prior top-level names (`data`, `sessions`, helpers, imports) survive into the next eval call — reuse them; NEVER re-import, re-require, or re-declare a helper. Re-read a file only if it may have changed since the last read. Re-run setup only after `reset`, a crash, or a `NameError`/`ReferenceError`.
</critical>
@@ -1,6 +1,6 @@
**Tasks referenced by verbatim content string, NEVER an auto-generated ID — no "task-1"/"task-N" exists. Pass the content text in the `task` field.**
Manages a phased task list. Pass `ops`: flat array of operations. Next pending task auto-promotes to `in_progress` on each completion. `pending` is a status, not an `op` — leave not-yet-started tasks implicit in `init`/`append`.
Next pending task auto-promotes to `in_progress` on each completion.
## Operations
@@ -1621,6 +1621,7 @@ export class AgentSession {
await this.#advisorRuntime.waitForCatchup(30000, threshold, signal);
}
}
await this.#maintainContextMidRun(messages, signal);
});
this.yieldQueue = new YieldQueue({
isStreaming: () => this.isStreaming,
@@ -2779,10 +2780,32 @@ export class AgentSession {
this.#lastAssistantMessage = undefined;
if (!msg) {
this.#lastSuccessfulYieldToolCallId = undefined;
logger.debug("agent_end maintenance routing", {
reason: "no-assistant-message",
goalModeEnabled: this.#goalModeState?.enabled === true,
goalStatus: this.#goalModeState?.goal.status,
});
await emitAgentEndNotification();
return;
}
const maintenanceRoute = (route: string, extra?: Record<string, unknown>) => {
logger.debug("agent_end maintenance routing", {
route,
stopReason: msg.stopReason,
provider: msg.provider,
model: msg.model,
contentBlocks: msg.content.length,
hasToolCalls: msg.content.some(content => content.type === "toolCall"),
hasText: msg.content.some(content => content.type === "text"),
goalModeEnabled: this.#goalModeState?.enabled === true,
goalStatus: this.#goalModeState?.goal.status,
successfulYield: this.#assistantEndedWithSuccessfulYield(msg),
...extra,
});
};
maintenanceRoute("entered");
// Invalidate GitHub Copilot credentials on auth failure so stale tokens
// aren't reused on the next request
if (
@@ -2796,27 +2819,61 @@ export class AgentSession {
if (this.#skipPostTurnMaintenanceAssistantTimestamp === msg.timestamp) {
this.#skipPostTurnMaintenanceAssistantTimestamp = undefined;
this.#lastSuccessfulYieldToolCallId = undefined;
maintenanceRoute("skip-post-turn-maintenance");
await emitAgentEndNotification();
return;
}
const activeGoal = this.#goalModeState?.enabled === true && this.#goalModeState.goal.status === "active";
if (this.#assistantEndedWithSuccessfulYield(msg)) {
this.#lastSuccessfulYieldToolCallId = undefined;
if (this.#goalModeState?.enabled && this.#goalModeState.goal.status === "active") {
if (activeGoal) {
maintenanceRoute("successful-yield-active-goal-checkCompaction");
const compactionTask = this.#checkCompaction(msg);
this.#trackPostPromptTask(compactionTask);
await compactionTask;
} else {
maintenanceRoute("successful-yield-no-active-goal");
}
await emitAgentEndNotification();
return;
}
this.#lastSuccessfulYieldToolCallId = undefined;
// Empty-stop cleanup MUST run before any compaction continuation: an
// empty toolUse stop must be stripped from active context + session
// history before we schedule another turn, otherwise the next
// Anthropic turn carries a tool_use block with no matching
// tool_result and corrupts message history. The handler also
// schedules its own retry, so a real empty stop never needs the
// active-goal threshold pre-empt below.
if (await this.#handleEmptyAssistantStop(msg)) {
maintenanceRoute("empty-stop-handled");
await emitAgentEndNotification();
return;
}
let compactionResult = COMPACTION_CHECK_NONE;
let checkedCompaction = false;
if (activeGoal) {
maintenanceRoute("active-goal-pre-empt-checkCompaction");
const compactionTask = this.#checkCompaction(msg);
this.#trackPostPromptTask(compactionTask);
compactionResult = await compactionTask;
checkedCompaction = true;
if (compactionResult.deferredHandoff || compactionResult.continuationScheduled) {
maintenanceRoute("active-goal-pre-empt-continuation-scheduled", {
deferredHandoff: compactionResult.deferredHandoff,
continuationScheduled: compactionResult.continuationScheduled,
});
this.#resolveRetry();
await emitAgentEndNotification();
return;
}
}
if (await this.#handleUnexpectedAssistantStop(msg)) {
maintenanceRoute("unexpected-stop-handled");
await emitAgentEndNotification();
return;
}
@@ -2856,9 +2913,12 @@ export class AgentSession {
}
this.#resolveRetry();
const compactionTask = this.#checkCompaction(msg);
this.#trackPostPromptTask(compactionTask);
const compactionResult = await compactionTask;
if (!checkedCompaction) {
maintenanceRoute("bottom-checkCompaction");
const compactionTask = this.#checkCompaction(msg);
this.#trackPostPromptTask(compactionTask);
compactionResult = await compactionTask;
}
// Check for incomplete todos only after a final assistant stop, not intermediate tool-use turns.
const hasToolCalls = msg.content.some(content => content.type === "toolCall");
if (hasToolCalls) {
@@ -8253,6 +8313,58 @@ export class AgentSession {
});
}
/**
* Compact active `/goal` runs that never settle to `agent_end`.
*
* Long autonomous goals can keep producing tool calls inside one agent run.
* The post-turn `agent_end` threshold check never fires in that shape, so
* context can grow until provider overflow. `onTurnEnd` is the safe boundary:
* tool results for the just-finished turn are already paired in
* `activeMessages`, the live array the agent loop reads before its next
* model call. Run maintenance here and splice the compacted state back into
* that array, mirroring [`AgentSession.#applyRewind`].
*/
async #maintainContextMidRun(activeMessages: AgentMessage[], signal?: AbortSignal): Promise<void> {
if (signal?.aborted || this.#isDisposed || this.isCompacting || this.isGeneratingHandoff) return;
if (!(this.#goalModeState?.enabled === true && this.#goalModeState.goal.status === "active")) return;
const model = this.model;
const contextWindow = model?.contextWindow ?? 0;
if (contextWindow <= 0) return;
const compactionSettings = this.settings.getGroup("compaction");
if (!compactionSettings.enabled || compactionSettings.strategy === "off") return;
const lastAssistant = [...activeMessages]
.reverse()
.find((message): message is AssistantMessage => message.role === "assistant");
if (!lastAssistant || lastAssistant.stopReason === "aborted" || lastAssistant.stopReason === "error") return;
const billedContextTokens = calculateContextTokens(lastAssistant.usage);
const storedContextTokens = this.#estimateStoredContextTokens();
const contextTokens = compactionContextTokens(billedContextTokens, storedContextTokens);
if (!shouldCompact(contextTokens, contextWindow, compactionSettings)) return;
const messagesBefore = activeMessages.length;
await this.#runAutoCompaction("threshold", false, false, false, {
autoContinue: false,
suppressContinuation: true,
triggerContextTokens: contextTokens,
});
if (signal?.aborted) return;
const compactedMessages = this.agent.state.messages;
if (compactedMessages !== activeMessages) {
activeMessages.splice(0, activeMessages.length, ...compactedMessages);
}
logger.debug("Mid-run goal compaction ran between tool-call turns", {
contextTokens,
contextWindow,
strategy: compactionSettings.strategy,
messagesBefore,
messagesAfter: activeMessages.length,
});
}
/**
* Check if context maintenance or promotion is needed and run it.
* Called after agent_end and before prompt submission.
@@ -8378,28 +8490,58 @@ export class AgentSession {
// Skip if this was an error (non-overflow errors don't have usage data)
if (assistantMessage.stopReason === "error") return COMPACTION_CHECK_NONE;
const pruneResult = await this.#pruneToolOutputs();
let contextTokens = calculateContextTokens(assistantMessage.usage);
if (supersedeResult) {
contextTokens = Math.max(0, contextTokens - supersedeResult.tokensSaved);
}
if (pruneResult) {
contextTokens = Math.max(0, contextTokens - pruneResult.tokensSaved);
}
// Floor by the real stored-conversation estimate so a payload-shrinking
// before_provider_request hook (e.g. a compression extension such as
// Headroom) can't deflate the provider-reported usage below the true
// history size and skip the threshold. The estimate runs after the prune
// passes above, so it reflects the post-prune message set.
contextTokens = compactionContextTokens(contextTokens, this.#estimateStoredContextTokens());
if (shouldCompact(contextTokens, contextWindow, compactionSettings)) {
const maintenanceTokensFreed = (supersedeResult?.tokensSaved ?? 0) + (pruneResult?.tokensSaved ?? 0);
const assistantUsageContextTokens = calculateContextTokens(assistantMessage.usage);
const storedContextTokens = this.#estimateStoredContextTokens();
// Pruning frees bytes for the NEXT prompt; it does not change the size of
// the prompt the LLM just billed for. Earlier revisions subtracted the
// per-turn supersede/prune `tokensSaved` from the threshold input, which
// let a long-running `/goal` session sit above `compaction.thresholdTokens`
// indefinitely whenever per-turn pruning saved enough to drop the
// post-prune estimate below the user-configured trigger — the visible
// context (anchored to the same provider billing) still showed >threshold,
// but `shouldCompact` no-op'd (#3174). Anchor the initial trigger on the
// last turn's billed context tokens, floored by the post-prune
// stored-conversation estimate so a payload-compression hook still can't
// deflate the trigger.
const contextTokens = compactionContextTokens(assistantUsageContextTokens, storedContextTokens);
const postMaintenanceContextTokens = compactionContextTokens(
Math.max(0, assistantUsageContextTokens - maintenanceTokensFreed),
storedContextTokens,
);
const thresholdTokens = resolveThresholdTokens(contextWindow, compactionSettings);
const shouldThresholdCompact = shouldCompact(contextTokens, contextWindow, compactionSettings);
logger.debug("Auto-compaction threshold decision", {
phase: "post-agent-end",
goalModeEnabled: this.#goalModeState?.enabled === true,
goalStatus: this.#goalModeState?.goal.status,
stopReason: assistantMessage.stopReason,
sameModel: sameModel === true,
contextWindow,
strategy: compactionSettings.strategy,
thresholdTokens,
assistantUsageContextTokens,
storedContextTokens,
resolvedContextTokens: contextTokens,
postMaintenanceContextTokens,
maintenanceTokensFreed,
shouldCompact: shouldThresholdCompact,
contextPromotionEnabled: this.settings.get("contextPromotion.enabled") === true,
});
if (shouldThresholdCompact) {
// Try promotion first — if a larger model is available, switch instead of compacting
const promoted = await this.#tryContextPromotion(assistantMessage);
if (!promoted) {
return await this.#runAutoCompaction("threshold", false, false, allowDefer, {
autoContinue,
triggerContextTokens: contextTokens,
triggerContextTokens: postMaintenanceContextTokens,
});
}
logger.debug("Auto-compaction threshold satisfied but context promotion took over", {
contextTokens,
contextWindow,
model: `${assistantMessage.provider}/${assistantMessage.model}`,
});
}
return COMPACTION_CHECK_NONE;
}
@@ -9523,13 +9665,15 @@ export class AgentSession {
willRetry: boolean,
deferred = false,
allowDefer = true,
options: { autoContinue?: boolean; triggerContextTokens?: number } = {},
options: { autoContinue?: boolean; triggerContextTokens?: number; suppressContinuation?: boolean } = {},
): Promise<CompactionCheckResult> {
const compactionSettings = this.settings.getGroup("compaction");
if (compactionSettings.strategy === "off") return COMPACTION_CHECK_NONE;
if (reason !== "idle" && !compactionSettings.enabled) return COMPACTION_CHECK_NONE;
const generation = this.#promptGeneration;
const shouldAutoContinue = options.autoContinue !== false && compactionSettings.autoContinue !== false;
const suppressContinuation = options.suppressContinuation === true;
const shouldAutoContinue =
!suppressContinuation && options.autoContinue !== false && compactionSettings.autoContinue !== false;
// Shake runs inline (cheap, no remote LLM). On overflow recovery, if shake
// reclaims nothing we fall through to the summary-compaction body below so
// the oversized input still gets resolved.
@@ -9540,6 +9684,7 @@ export class AgentSession {
generation,
shouldAutoContinue,
options.triggerContextTokens,
suppressContinuation,
);
if (outcome !== "fallback") return outcome;
}
@@ -9993,7 +10138,7 @@ export class AgentSession {
this.#scheduleAgentContinue({ delayMs: 100, generation });
continuationScheduled = true;
} else if (this.agent.hasQueuedMessages()) {
} else if (!suppressContinuation && this.agent.hasQueuedMessages()) {
// Auto-compaction can complete while follow-up/steering/custom messages are waiting.
// Kick the loop so queued messages are actually delivered.
this.#scheduleAgentContinue({
@@ -10053,6 +10198,7 @@ export class AgentSession {
generation: number,
autoContinue: boolean,
triggerContextTokens?: number,
suppressContinuation = false,
): Promise<CompactionCheckResult | "fallback"> {
const action = "shake";
this.#autoCompactionAbortController?.abort();
@@ -10083,15 +10229,16 @@ export class AgentSession {
// situation actually resolves; "idle" is exempt because its 60s+ timer
// re-checks usage before re-firing and cannot dead-loop on its own.
//
// #2275: the post-shake check MUST be anchored on the same metric that
// triggered compaction. The local estimator (`#estimatePendingPromptTokens`)
// undercounts thinking-signature payloads, so on thinking-heavy sessions it
// reads well below the provider-reported usage that fired the threshold.
// When that estimate slips under the threshold, the fallback never fires
// and the auto-continue prompt re-injects every turn. Prefer the trigger's
// own `contextTokens` (provider-anchored) when the caller supplies it, and
// add hysteresis (80% recovery band) so we don't oscillate at the boundary
// while shake keeps reclaiming a trickle of the previous turn's output.
// #2275: the post-shake check MUST stay provider-anchored when caller
// usage and local estimates diverge. The local estimator undercounts
// thinking-signature payloads, so thinking-heavy sessions can read well
// below the provider usage that fired the threshold. Prefer the caller's
// context figure when supplied, then subtract shake's own savings and add
// hysteresis (80% recovery band) so we don't oscillate at the boundary.
// Threshold callers pass the provider-billed trigger after accounting for
// any supersede/drop-useless pruning that already rewrote the next prompt;
// without that pre-shake savings, shake can fall through to context-full
// even though the post-prune history is already inside the recovery band.
const contextWindow = this.model?.contextWindow ?? 0;
const compactionSettings = this.settings.getGroup("compaction");
let stillOverThreshold = false;
@@ -10151,7 +10298,7 @@ export class AgentSession {
}
this.#scheduleAgentContinue({ delayMs: 100, generation });
continuationScheduled = true;
} else if (this.agent.hasQueuedMessages()) {
} else if (!suppressContinuation && this.agent.hasQueuedMessages()) {
this.#scheduleAgentContinue({
delayMs: 100,
generation,
+20
View File
@@ -153,6 +153,26 @@ export function getConfiguredThinkingLevelMetadata(level: ConfiguredThinkingLeve
return level === AUTO_THINKING ? AUTO_THINKING_METADATA : getThinkingLevelMetadata(level);
}
/**
* Thinking selectors accepted by the `--thinking` CLI flag, in display order:
* `off`, every concrete effort (`minimal`..`xhigh`), then `auto`. Single source
* for the flag's `options` list, shell completions, and the "invalid level"
* warning so all three stay in sync.
*/
export const CLI_THINKING_LEVELS: readonly string[] = [ThinkingLevel.Off, ...THINKING_EFFORTS, AUTO_THINKING];
/**
* Parses a `--thinking` CLI value. Accepts every {@link parseConfiguredThinkingLevel}
* selector (`off`, `auto`, `minimal`..`xhigh`, plus the `max` alias) but rejects
* `inherit`: an explicit `inherit` on the command line would suppress the
* settings/scoped-model fallback during startup resolution only to resolve back
* to the provider default, which is never what the user means.
*/
export function parseCliThinkingLevel(value: string | null | undefined): ConfiguredThinkingLevel | undefined {
const level = parseConfiguredThinkingLevel(value);
return level === ThinkingLevel.Inherit ? undefined : level;
}
/**
* Resolves an auto-classified effort against the active model's supported
* range. Unlike {@link clampThinkingLevelForModel}, `auto` never resolves below
+46 -40
View File
@@ -308,7 +308,7 @@ async function askSingleQuestion(
}
const customResult = await promptForCustomInput();
if (customResult.input === undefined) {
break;
continue;
}
customInput = customResult.input;
break;
@@ -332,51 +332,57 @@ async function askSingleQuestion(
}
selectedOptions = Array.from(selected);
} else {
const displayOptions = addRecommendedSuffix(questionOptions, recommended);
const optionsWithNavigation: ExtensionUISelectItem[] = [...displayOptions, OTHER_OPTION];
while (true) {
const displayOptions = addRecommendedSuffix(questionOptions, recommended);
const optionsWithNavigation: ExtensionUISelectItem[] = [...displayOptions, OTHER_OPTION];
let initialIndex = recommended;
const previouslySelected = selectedOptions[0];
if (previouslySelected) {
const selectedIndex = questionOptions.findIndex(option => option.label === previouslySelected);
if (selectedIndex >= 0) initialIndex = selectedIndex;
} else if (customInput !== undefined) {
initialIndex = displayOptions.length;
}
if (initialIndex !== undefined) {
const maxIndex = Math.max(optionsWithNavigation.length - 1, 0);
initialIndex = Math.max(0, Math.min(initialIndex, maxIndex));
}
const {
choice,
timedOut: selectTimedOut,
navigation: arrowNavigation,
} = await selectOption(promptWithProgress, optionsWithNavigation, initialIndex, {
selectionMarker: "radio",
markableCount: displayOptions.length,
});
timedOut = selectTimedOut;
if (arrowNavigation) {
return { selectedOptions, customInput, timedOut, navigation: arrowNavigation };
}
if (choice === undefined) {
if (!timedOut) {
return { selectedOptions, customInput, timedOut, cancelled: true };
let initialIndex = recommended;
const previouslySelected = selectedOptions[0];
if (previouslySelected) {
const selectedIndex = questionOptions.findIndex(option => option.label === previouslySelected);
if (selectedIndex >= 0) initialIndex = selectedIndex;
} else if (customInput !== undefined) {
initialIndex = displayOptions.length;
}
} else if (choice === OTHER_OPTION) {
if (!selectTimedOut) {
const customResult = await promptForCustomInput();
if (customResult.input !== undefined) {
customInput = customResult.input;
selectedOptions = [];
if (initialIndex !== undefined) {
const maxIndex = Math.max(optionsWithNavigation.length - 1, 0);
initialIndex = Math.max(0, Math.min(initialIndex, maxIndex));
}
const {
choice,
timedOut: selectTimedOut,
navigation: arrowNavigation,
} = await selectOption(promptWithProgress, optionsWithNavigation, initialIndex, {
selectionMarker: "radio",
markableCount: displayOptions.length,
});
timedOut = selectTimedOut;
if (arrowNavigation) {
return { selectedOptions, customInput, timedOut, navigation: arrowNavigation };
}
if (choice === undefined) {
if (!timedOut) {
return { selectedOptions, customInput, timedOut, cancelled: true };
}
// If editor was dismissed (undefined), keep prior selectedOptions/customInput intact
break;
}
if (choice === OTHER_OPTION) {
if (selectTimedOut) {
break;
}
const customResult = await promptForCustomInput();
if (customResult.input === undefined) {
continue;
}
customInput = customResult.input;
selectedOptions = [];
break;
}
} else {
selectedOptions = [stripRecommendedSuffix(choice)];
customInput = undefined;
break;
}
if (navigation?.allowForward) {
return { selectedOptions, customInput, timedOut, navigation: "forward" };
@@ -800,7 +800,7 @@ export class WorkerCore {
displays.push({ type: "text", text: safeJsonStringify(output.data) });
return;
}
// status — surface as compact JSON so helper side effects (read/write/tree) appear in
// status — surface as compact JSON so helper side effects (read/write/env) appear in
// the cell result alongside explicit display() output.
displays.push({ type: "text", text: safeJsonStringify(output.event) });
}
+5 -14
View File
@@ -56,6 +56,9 @@ interface EvalRenderCellArg {
}
interface EvalRenderArgs {
language?: string;
code?: string;
title?: string;
cells?: EvalRenderCellArg[];
__partialJson?: string;
}
@@ -81,8 +84,8 @@ function normalizeRenderLanguage(value: string | undefined): EvalLanguage {
}
function getRenderCells(args: EvalRenderArgs | undefined): EvalRenderCell[] {
const raw = args?.cells;
if (!Array.isArray(raw)) return [];
if (!args) return [];
const raw = Array.isArray(args.cells) ? args.cells : typeof args.code === "string" ? [args] : [];
const out: EvalRenderCell[] = [];
for (const cell of raw) {
if (!cell || typeof cell !== "object") continue;
@@ -238,14 +241,12 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string {
const opIcons: Record<string, AvailableIcon> = {
read: "icon.file",
write: "icon.file",
append: "icon.file",
cat: "icon.file",
touch: "icon.file",
ls: "icon.folder",
cd: "icon.folder",
pwd: "icon.folder",
mkdir: "icon.folder",
tree: "icon.folder",
git_status: "icon.git",
git_diff: "icon.git",
git_log: "icon.git",
@@ -277,7 +278,6 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string {
if (data.path) parts.push(`from ${shortenPath(String(data.path))}`);
break;
case "write":
case "append":
parts.push(`${data.chars ?? data.bytes ?? 0} chars`);
if (data.path) parts.push(`to ${shortenPath(String(data.path))}`);
break;
@@ -316,13 +316,6 @@ function formatStatusEvent(event: EvalStatusEvent, theme: Theme): string {
parts.push(`${data.lines} line${(data.lines as number) !== 1 ? "s" : ""}`);
if (data.staged) parts.push("(staged)");
break;
case "diff":
if (data.identical) {
parts.push("files identical");
} else {
parts.push("files differ");
}
break;
case "batch":
parts.push(`${data.files} file${(data.files as number) !== 1 ? "s" : ""} processed`);
break;
@@ -412,8 +405,6 @@ function formatStatusEventExpanded(event: EvalStatusEvent, theme: Theme): string
case "cat":
case "head":
case "tail":
case "tree":
case "diff":
case "git_diff":
case "sh":
if (data.preview) addPreview(String(data.preview));
+101 -104
View File
@@ -38,8 +38,6 @@ const EVAL_LANGUAGE_NAME: Record<EvalLanguageToken, string> = {
rb: "Ruby",
jl: "Julia",
};
const EVAL_CELLS_DESCRIPTION =
"cells executed in order. State persists within each language across cells and tool calls.";
/** Join names as an English "or" list: ["A"]→"A", ["A","B"]→"A or B", 3+→"A, B, or C". */
function joinWithOr(items: readonly string[]): string {
@@ -56,12 +54,12 @@ function describeCodeField(langs: readonly EvalLanguageToken[]): string {
const replLangs = langs.filter(lang => lang === "rb" || lang === "jl");
// No persistent REPL backends → keep the original py/js phrasing verbatim so the
// default (rb/jl off) wire schema stays byte-identical to the pre-feature one.
if (replLangs.length === 0) return "cell body, verbatim. Use top-level await freely.";
if (replLangs.length === 0) return "code to run in this eval call, verbatim. Use top-level await freely.";
const awaitLangs = langs.filter(lang => lang === "py" || lang === "js");
const clauses: string[] = [];
if (awaitLangs.length > 0) clauses.push(`Top-level \`await\` is available in ${awaitLangs.join("/")}`);
clauses.push(`${replLangs.join("/")} auto-display the last expression like a REPL`);
return `cell body, verbatim. ${clauses.join("; ")}.`;
return `code to run in this eval call, verbatim. ${clauses.join("; ")}.`;
}
/** One-line discovery summary listing the runtimes available this session. */
@@ -86,29 +84,24 @@ function enabledEvalLanguages(backends: EvalBackendsAllowance): EvalLanguageToke
const evalCellCommonFields = {
"title?": type("string").describe('short label shown in transcript (e.g. "imports", "load config")'),
"timeout?": type("number").describe("per-cell timeout in seconds"),
"reset?": type("boolean").describe(
"wipe this cell's language kernel before running. Other languages are untouched.",
),
"timeout?": type("number").describe("timeout for this eval call in seconds"),
"reset?": type("boolean").describe("wipe this language's kernel before running. Other languages are untouched."),
};
/**
* Per-cell input. Each cell runs in order; state persists within a language
* across cells and across tool calls. This static schema carries the full
* language union for typing; {@link buildEvalSchema} narrows the wire copy per
* session so disabled backends are never advertised to the model.
* Per-call input: a single cell. State persists within a language across
* separate eval calls and across tool calls, so each call is one logical step
* and later calls reuse what earlier ones defined. This static schema carries
* the full language union for typing; {@link buildEvalSchema} narrows the wire
* copy per session so disabled backends are never advertised to the model.
*/
const evalCellSchema = type({
language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)),
code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)),
...evalCellCommonFields,
});
export type EvalCellInput = typeof evalCellSchema.infer;
export const evalSchema = type({
cells: evalCellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION),
language: type("'py' | 'js' | 'rb' | 'jl'").describe(describeLanguageField(EVAL_LANGUAGE_ORDER)),
...evalCellCommonFields,
code: type("string").describe(describeCodeField(EVAL_LANGUAGE_ORDER)),
});
export type EvalToolParams = typeof evalSchema.infer;
export type EvalCellInput = EvalToolParams;
/**
* Build a session-scoped copy of the eval schema whose `language` enum and field
@@ -118,14 +111,11 @@ export type EvalToolParams = typeof evalSchema.infer;
* {@link evalSchema} (full union) remains the type-level source of truth.
*/
function buildEvalSchema(langs: readonly EvalLanguageToken[]): typeof evalSchema {
const cellSchema = type({
const schema = type({
language: type.enumerated(...langs).describe(describeLanguageField(langs)),
code: type("string").describe(describeCodeField(langs)),
...evalCellCommonFields,
});
const schema = type({
cells: cellSchema.array().atLeastLength(1).describe(EVAL_CELLS_DESCRIPTION),
});
return schema as unknown as typeof evalSchema;
}
@@ -290,22 +280,15 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
readonly approval = "exec" as const;
readonly formatApprovalDetails = (args: unknown): string[] => {
const params = args as Partial<EvalToolParams>;
const cells = Array.isArray(params.cells) ? params.cells : [];
const firstCell = cells[0] as Partial<EvalCellInput> | undefined;
if (!firstCell) return [];
const language =
typeof firstCell.language === "string" ? formatEvalInputLanguage(firstCell.language) : "javascript (default)";
const code = typeof firstCell.code === "string" ? firstCell.code : "";
const lines = [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`];
if (cells.length > 1) {
lines.push(`+${cells.length - 1} more cell${cells.length === 2 ? "" : "s"}`);
}
return lines;
typeof params.language === "string" ? formatEvalInputLanguage(params.language) : "javascript (default)";
const code = typeof params.code === "string" ? params.code : "";
return [`Language: ${language}`, `Code:\n${truncateForPrompt(code)}`];
};
get summary(): string {
return summarizeEvalLanguages(this.#enabledLanguages());
}
readonly loadMode = "discoverable";
readonly loadMode = "essential";
readonly label = "Eval";
get description(): string {
if (!this.session) return getEvalToolDescription();
@@ -320,25 +303,53 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
spawns: spawnsAllowed,
});
}
readonly examples: readonly ToolExample<typeof evalSchema.infer>[] = [
/** All reuse-chain examples; the `examples` getter filters by enabled languages. */
private static readonly ALL_EXAMPLES: readonly ToolExample<typeof evalSchema.infer>[] = [
{
caption: "First call — set up once",
call: {
cells: [
{
language: "py",
title: "imports",
timeout: 10,
code: "import json\nfrom pathlib import Path",
},
{
language: "py",
title: "load config",
code: "data = json.loads(read('package.json'))\ndisplay(data)",
},
],
language: "py",
title: "imports",
code: "import json\nfrom pathlib import Path",
},
},
{
caption: "Second call — reuse, do NOT re-import",
call: {
language: "py",
title: "load config",
code: "data = json.loads(read('package.json'))\ndisplay(data)",
},
},
{
caption: "Third call — reuse the loaded config",
call: {
language: "py",
title: "scan deps",
code: "display(sorted(data['dependencies']))",
},
},
{
caption: "Ruby first call — set up once",
call: {
language: "rb",
title: "setup",
code: "require 'json'\npkg_path = 'package.json'",
},
},
{
caption: "Ruby second call — reuse, do NOT re-require",
call: {
language: "rb",
title: "load config",
code: "pkg = JSON.parse(read(pkg_path))\ndisplay(pkg.keys.sort)",
},
},
];
get examples(): readonly ToolExample<typeof evalSchema.infer>[] {
const langs = new Set(this.#enabledLanguages());
return EvalTool.ALL_EXAMPLES.filter(ex => "call" in ex && langs.has(ex.call.language as EvalLanguageToken));
}
get parameters(): typeof evalSchema {
const langs = this.#enabledLanguages();
if (langs.length === 0 || langs.length === EVAL_LANGUAGE_ORDER.length) return evalSchema;
@@ -352,13 +363,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
readonly concurrency = "exclusive";
readonly strict = true;
readonly intent = (args: Partial<typeof evalSchema.infer>): string | undefined => {
const cells = Array.isArray(args.cells) ? args.cells : [];
const first = cells.find(c => c && typeof c === "object");
if (!first) return "evaluating";
const title = typeof first.title === "string" ? first.title : undefined;
const language = typeof first.language === "string" ? formatEvalInputLanguage(first.language) : "javascript";
const label = title || `running ${language}`;
return cells.length > 1 ? `${label} (+${cells.length - 1})` : label;
const title = typeof args.title === "string" ? args.title : undefined;
const language = typeof args.language === "string" ? formatEvalInputLanguage(args.language) : "javascript";
return title || `running ${language}`;
};
readonly #proxyExecutor?: EvalProxyExecutor;
@@ -398,27 +405,25 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
const session = this.session;
const excludeWebP = webpExclusionForModel(session.getActiveModel?.());
const cells: ResolvedEvalCell[] = [];
for (let i = 0; i < params.cells.length; i++) {
const cell = params.cells[i];
const language: EvalLanguage =
cell.language === "py"
? "python"
: cell.language === "rb"
? "ruby"
: cell.language === "jl"
? "julia"
: "js";
const resolved = await resolveBackend(session, language);
cells.push({
index: i,
title: cell.title,
code: cell.code,
timeoutMs: (cell.timeout ?? 30) * 1000,
reset: cell.reset ?? false,
const cellLanguage: EvalLanguage =
params.language === "py"
? "python"
: params.language === "rb"
? "ruby"
: params.language === "jl"
? "julia"
: "js";
const resolved = await resolveBackend(session, cellLanguage);
const cells: ResolvedEvalCell[] = [
{
index: 0,
title: params.title,
code: params.code,
timeoutMs: (params.timeout ?? 30) * 1000,
reset: params.reset ?? false,
resolved,
});
}
},
];
const languages = uniqueEvalLanguages(cells);
const notice = detailsNotice(cells);
const sessionAbortController = new AbortController();
@@ -453,6 +458,13 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
status: "pending",
}));
const cellOutputs: string[] = [];
// The cell currently inside backend.execute(). Streamed stdout is
// appended to its rendered `output` live so a long-running cell (e.g. a
// sleep loop) shows progress instead of nothing until it returns. A
// dedicated per-cell tail buffer keeps attribution correct and avoids
// double-counting against the aggregate `tailBuffer`; on completion the
// authoritative `cellResult.output` (below) overwrites this live tail.
let activeLiveCell: { result: EvalCellResult; buf: TailBuffer } | undefined;
const appendTail = (text: string) => {
tailBuffer.append(text);
@@ -502,6 +514,10 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
maxColumns: resolveOutputMaxColumns(session.settings),
onChunk: chunk => {
appendTail(chunk);
if (activeLiveCell) {
activeLiveCell.buf.append(chunk);
activeLiveCell.result.output = activeLiveCell.buf.text();
}
pushUpdate();
},
});
@@ -529,6 +545,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
cellResult.statusEvents = undefined;
cellResult.exitCode = undefined;
cellResult.durationMs = undefined;
activeLiveCell = { result: cellResult, buf: new TailBuffer(DEFAULT_MAX_BYTES * 2) };
pushUpdate();
const startTime = Date.now();
@@ -562,6 +579,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
});
} finally {
idle.dispose();
activeLiveCell = undefined;
}
const durationMs = Date.now() - startTime;
@@ -623,24 +641,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
cellResult.statusEvents = cellStatusEvents.length > 0 ? cellStatusEvents : undefined;
cellResult.hasMarkdown = cellHasMarkdown || undefined;
let combinedCellOutput = "";
if (cells.length > 1) {
const cellHeader = `[${i + 1}/${cells.length}]`;
const cellTitle = cell.title ? ` ${cell.title}` : "";
if (cellOutput) {
combinedCellOutput = `${cellHeader}${cellTitle}\n${cellOutput}`;
} else {
combinedCellOutput = `${cellHeader}${cellTitle} (ok)`;
}
cellOutputs.push(combinedCellOutput);
} else if (cellOutput) {
combinedCellOutput = cellOutput;
cellOutputs.push(combinedCellOutput);
}
if (combinedCellOutput) {
const prefix = cellOutputs.length > 1 ? "\n\n" : "";
appendTail(`${prefix}${combinedCellOutput}`);
if (cellOutput) {
cellOutputs.push(cellOutput);
appendTail(cellOutput);
}
if (result.cancelled) {
@@ -648,10 +651,7 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
pushUpdate();
const errorMsg = result.output || "Command aborted";
const combinedOutput = cellOutputs.join("\n\n");
const outputText =
cells.length > 1
? `${combinedOutput}\n\nCell ${i + 1} aborted: ${errorMsg}`
: combinedOutput || errorMsg;
const outputText = combinedOutput || errorMsg;
const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput);
const details: EvalToolDetails = {
@@ -674,12 +674,9 @@ export class EvalTool implements AgentTool<typeof evalSchema> {
cellResult.status = "error";
pushUpdate();
const combinedOutput = cellOutputs.join("\n\n");
const outputText =
cells.length > 1
? `${combinedOutput}\n\nCell ${i + 1} failed (exit code ${result.exitCode}). Earlier cells succeeded—their state persists. Fix only cell ${i + 1}.`
: combinedOutput
? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}`
: `Command exited with code ${result.exitCode}`;
const outputText = combinedOutput
? `${combinedOutput}\n\nCommand exited with code ${result.exitCode}`
: `Command exited with code ${result.exitCode}`;
const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput);
const details: EvalToolDetails = {
+8 -1
View File
@@ -377,7 +377,14 @@ export type ToolFactory = (session: ToolSession) => Tool | null | Promise<Tool |
export type BuiltinToolLoadMode = "essential" | "discoverable";
/** Default essential tool names when tools.essentialOverride is empty. */
export const DEFAULT_ESSENTIAL_TOOL_NAMES: readonly string[] = ["read", "bash", "edit", "write", "find"] as const;
export const DEFAULT_ESSENTIAL_TOOL_NAMES: readonly string[] = [
"read",
"bash",
"edit",
"write",
"find",
"eval",
] as const;
/**
* Resolve the active essential built-in tool names from settings.
+136 -60
View File
@@ -8,7 +8,7 @@ import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
import { glob, type SummaryResult, summarizeCode } from "@oh-my-pi/pi-natives";
import type { Component } from "@oh-my-pi/pi-tui";
import { Text } from "@oh-my-pi/pi-tui";
import { getRemoteDir, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
import { getRemoteDir, type ImageMetadata, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
import { type } from "arktype";
import { LRUCache } from "lru-cache/raw";
import {
@@ -22,7 +22,7 @@ import {
import { normalizeToLF } from "../edit/normalize";
import { isNotebookPath, readEditableNotebookText } from "../edit/notebook";
import type { RenderResultOptions } from "../extensibility/custom-tools/types";
import { InternalUrlRouter } from "../internal-urls";
import { InternalUrlRouter, resolveLocalUrlToFile } from "../internal-urls";
import { parseInternalUrl } from "../internal-urls/parse";
import type { InternalUrl } from "../internal-urls/types";
import { getLanguageFromPath, type Theme } from "../modes/theme/theme";
@@ -1112,6 +1112,79 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
.done();
}
/**
* Build content blocks for an on-disk image file: an `inspect_image`
* metadata note when inspection is enabled, otherwise the decoded image
* block. Shared by the plain-file read path and the `local://` image fast
* path so both honor `inspect_image.enabled`, the size cap, and auto-resize
* identically. Too-large / unsupported images surface as {@link ToolError}.
*/
async #loadImageContent(options: {
readPath: string;
absolutePath: string;
mimeType: string;
imageMetadata: ImageMetadata | null;
fileSize: number;
}): Promise<{ content: Array<TextContent | ImageContent>; details: ReadToolDetails; sourcePath: string }> {
const { readPath, absolutePath, mimeType, imageMetadata, fileSize } = options;
if (this.#inspectImageEnabled) {
const outputMime = imageMetadata?.mimeType ?? mimeType;
const metadataLines = [
"Image metadata:",
`- MIME: ${outputMime}`,
`- Bytes: ${fileSize} (${formatBytes(fileSize)})`,
imageMetadata?.width !== undefined && imageMetadata.height !== undefined
? `- Dimensions: ${imageMetadata.width}x${imageMetadata.height}`
: "- Dimensions: unknown",
imageMetadata?.channels !== undefined ? `- Channels: ${imageMetadata.channels}` : "- Channels: unknown",
imageMetadata?.hasAlpha === true
? "- Alpha: yes"
: imageMetadata?.hasAlpha === false
? "- Alpha: no"
: "- Alpha: unknown",
"",
`If you want to analyze the image, call inspect_image with path="${formatPathRelativeToCwd(
absolutePath,
this.session.cwd,
)}" and a question describing what to inspect and the desired output format.`,
];
return { content: [{ type: "text", text: metadataLines.join("\n") }], details: {}, sourcePath: absolutePath };
}
if (fileSize > MAX_IMAGE_SIZE) {
const sizeStr = formatBytes(fileSize);
const maxStr = formatBytes(MAX_IMAGE_SIZE);
throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`);
}
try {
const imageInput = await loadImageInput({
path: readPath,
cwd: this.session.cwd,
autoResize: this.#autoResizeImages,
maxBytes: MAX_IMAGE_SIZE,
resolvedPath: absolutePath,
detectedMimeType: mimeType,
excludeWebP: webpExclusionForModel(this.session.getActiveModel?.()),
});
if (!imageInput) {
throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`);
}
return {
content: [
{ type: "text", text: imageInput.textNote },
{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType },
],
details: {},
sourcePath: imageInput.resolvedPath,
};
} catch (error) {
if (error instanceof ImageInputTooLargeError) {
throw new ToolError(error.message);
}
throw error;
}
}
#buildInMemoryTextResult(
text: string,
offset: number | undefined,
@@ -2132,64 +2205,13 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
| undefined;
if (mimeType) {
if (this.#inspectImageEnabled) {
const metadata = imageMetadata;
const outputMime = metadata?.mimeType ?? mimeType;
const outputBytes = fileSize;
const metadataLines = [
"Image metadata:",
`- MIME: ${outputMime}`,
`- Bytes: ${outputBytes} (${formatBytes(outputBytes)})`,
metadata?.width !== undefined && metadata.height !== undefined
? `- Dimensions: ${metadata.width}x${metadata.height}`
: "- Dimensions: unknown",
metadata?.channels !== undefined ? `- Channels: ${metadata.channels}` : "- Channels: unknown",
metadata?.hasAlpha === true
? "- Alpha: yes"
: metadata?.hasAlpha === false
? "- Alpha: no"
: "- Alpha: unknown",
"",
`If you want to analyze the image, call inspect_image with path="${formatPathRelativeToCwd(
absolutePath,
this.session.cwd,
)}" and a question describing what to inspect and the desired output format.`,
];
content = [{ type: "text", text: metadataLines.join("\n") }];
details = {};
sourcePath = absolutePath;
} else {
if (fileSize > MAX_IMAGE_SIZE) {
const sizeStr = formatBytes(fileSize);
const maxStr = formatBytes(MAX_IMAGE_SIZE);
throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`);
}
try {
const imageInput = await loadImageInput({
path: readPath,
cwd: this.session.cwd,
autoResize: this.#autoResizeImages,
maxBytes: MAX_IMAGE_SIZE,
resolvedPath: absolutePath,
detectedMimeType: mimeType,
excludeWebP: webpExclusionForModel(this.session.getActiveModel?.()),
});
if (!imageInput) {
throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`);
}
content = [
{ type: "text", text: imageInput.textNote },
{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType },
];
details = {};
sourcePath = imageInput.resolvedPath;
} catch (error) {
if (error instanceof ImageInputTooLargeError) {
throw new ToolError(error.message);
}
throw error;
}
}
({ content, details, sourcePath } = await this.#loadImageContent({
readPath,
absolutePath,
mimeType,
imageMetadata,
fileSize,
}));
} else if (isNotebookPath(absolutePath) && !isRawSelector(parsed)) {
const notebookText = await readEditableNotebookText(absolutePath, localReadPath);
if (isMultiRange(parsed) && parsed.kind === "lines") {
@@ -2727,6 +2749,17 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
hasExtraction = hasPathExtraction || hasQueryExtraction;
}
// local:// files are real on-disk paths. Detect image files and emit a
// decoded image block before the text-only resource contract UTF-8
// decodes the binary into mojibake. The fast path returns null for
// non-images, directories, listings, or any resolution failure, so the
// text path below reproduces the router's not-found / symlink-escape
// behavior unchanged.
if (scheme === "local") {
const imageResult = await this.#tryReadLocalImage(urlMeta, signal);
if (imageResult) return imageResult;
}
// Reject line selectors when query extraction is used
if (hasExtraction && parsedSel.kind !== "none" && parsedSel.kind !== "raw") {
throw new ToolError("Cannot combine query extraction with line selectors");
@@ -2770,6 +2803,49 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
});
}
/**
* Fast path for `local://` image files. Resolves the URL to its real
* on-disk path with the same realpath + containment checks as
* {@link LocalProtocolHandler.resolve} (via {@link resolveLocalUrlToFile}),
* and — only when the target is a genuine image — emits a decoded image
* block. Returns null for non-images, directories, listings, or any
* resolution failure (not-found, symlink escape) so the caller falls back to
* normal text resolution, which reproduces the router's errors. Errors from
* a confirmed image (too large / unsupported) propagate rather than
* degrading into a corrupted text read.
*/
async #tryReadLocalImage(url: InternalUrl, signal?: AbortSignal): Promise<AgentToolResult<ReadToolDetails> | null> {
let file: { path: string; size: number } | null;
try {
file = await resolveLocalUrlToFile(url, {
cwd: this.session.cwd,
settings: this.session.settings,
signal,
localProtocolOptions: this.session.localProtocolOptions,
});
} catch {
// Not found / containment escape / no session — let the text path
// surface the router's canonical error.
return null;
}
if (!file) return null;
const imageMetadata = await readImageMetadata(file.path);
const mimeType = imageMetadata?.mimeType;
if (!mimeType) return null;
const { content, details, sourcePath } = await this.#loadImageContent({
readPath: url.href,
absolutePath: file.path,
mimeType,
imageMetadata,
fileSize: file.size,
});
const resultBuilder = toolResult(details).content(content).sourceInternal(url.href);
if (sourcePath) resultBuilder.sourcePath(sourcePath);
return resultBuilder.done();
}
/** Read directory contents as a formatted listing */
async #readDirectory(
absolutePath: string,
+60 -64
View File
@@ -52,21 +52,18 @@ const InitListEntry = type({
items: type("string").describe("task content").array().atLeastLength(1).describe("tasks for this phase"),
});
const TodoOpEntry = type({
const todoSchema = type({
op: TodoOp,
"list?": InitListEntry.array().describe("phased task list (init)"),
"task?": type("string").describe("task content"),
"phase?": type("string").describe("phase name"),
"items?": type("string").describe("task content").array().atLeastLength(1).describe("tasks to append"),
});
const todoSchema = type({
ops: TodoOpEntry.array().atLeastLength(1).describe("ordered todo operations"),
}).describe("apply ordered todo operations");
}).describe("apply a single todo operation");
type TodoParams = TodoSchema;
type TodoSchema = typeof todoSchema.infer;
type TodoOpEntryValue = TodoParams["ops"][number];
/** A single todo op entry (the params object itself). */
type TodoOpEntryValue = TodoParams;
// =============================================================================
// State helpers
@@ -402,10 +399,7 @@ function applyEntry(phases: TodoPhase[], entry: TodoOpEntryValue, errors: string
function applyParams(phases: TodoPhase[], params: TodoParams): { phases: TodoPhase[]; errors: string[] } {
const errors: string[] = [];
let next = phases;
for (const entry of params.ops) {
next = applyEntry(next, entry, errors);
}
const next = applyEntry(phases, params, errors);
normalizeInProgressTask(next);
return { phases: next, errors };
}
@@ -413,9 +407,15 @@ function applyParams(phases: TodoPhase[], params: TodoParams): { phases: TodoPha
/** Apply an array of `todo`-style ops to existing phases. Used by /todo slash command. */
export function applyOpsToPhases(
currentPhases: TodoPhase[],
ops: TodoParams["ops"],
ops: TodoParams[],
): { phases: TodoPhase[]; errors: string[] } {
return applyParams(clonePhases(currentPhases), { ops });
const errors: string[] = [];
let next = clonePhases(currentPhases);
for (const op of ops) {
next = applyEntry(next, op, errors);
}
normalizeInProgressTask(next);
return { phases: next, errors };
}
// =============================================================================
@@ -572,64 +572,44 @@ export class TodoTool implements AgentTool<typeof todoSchema, TodoToolDetails> {
{
caption: "Initial setup (multi-phase)",
call: {
ops: [
{
op: "init",
list: [
{ phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] },
{ phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] },
{ phase: "Verification", items: ["Run cargo test"] },
],
},
op: "init",
list: [
{ phase: "Foundation", items: ["Scaffold crate", "Wire workspace"] },
{ phase: "Auth", items: ["Port credential store", "Wire OAuth providers"] },
{ phase: "Verification", items: ["Run cargo test"] },
],
},
},
{
caption: "View current state (read-only)",
call: {
ops: [{ op: "view" }],
},
call: { op: "view" },
},
{
caption: "Initial setup (single phase)",
call: {
ops: [
{
op: "init",
list: [{ phase: "Implementation", items: ["Apply fix", "Run tests"] }],
},
],
op: "init",
list: [{ phase: "Implementation", items: ["Apply fix", "Run tests"] }],
},
},
{
caption: "Complete one task",
call: {
ops: [{ op: "done", task: "Wire workspace" }],
},
call: { op: "done", task: "Wire workspace" },
},
{
caption: "Complete a whole phase",
call: {
ops: [{ op: "done", phase: "Auth" }],
},
call: { op: "done", phase: "Auth" },
},
{
caption: "Remove all tasks",
call: {
ops: [{ op: "rm" }],
},
call: { op: "rm" },
},
{
caption: "Drop one task",
call: {
ops: [{ op: "drop", task: "Run cargo test" }],
},
call: { op: "drop", task: "Run cargo test" },
},
{
caption: "Append tasks to a phase",
call: {
ops: [{ op: "append", phase: "Auth", items: ["Handle retries", "Run tests"] }],
},
call: { op: "append", phase: "Auth", items: ["Handle retries", "Run tests"] },
},
];
readonly loadMode = "discoverable";
@@ -646,7 +626,7 @@ export class TodoTool implements AgentTool<typeof todoSchema, TodoToolDetails> {
): Promise<AgentToolResult<TodoToolDetails>> {
const previousPhases = clonePhases(this.session.getTodoPhases?.() ?? []);
// Pure-view calls are reads: no normalization, no state write.
const readOnly = params.ops.every(entry => entry.op === "view");
const readOnly = params.op === "view";
const { phases: updated, errors } = readOnly
? { phases: previousPhases, errors: [] as string[] }
: applyParams(clonePhases(previousPhases), params);
@@ -673,15 +653,32 @@ export class TodoTool implements AgentTool<typeof todoSchema, TodoToolDetails> {
// TUI Renderer
// =============================================================================
type TodoRenderArgs = {
ops?: Array<{
op?: string;
task?: string;
phase?: string;
items?: string[];
}>;
type TodoRenderOp = {
op?: string;
task?: string;
phase?: string;
items?: string[];
};
/** New single-op shape `{op,...}`; legacy `{ops:[...]}` still seen in old transcripts. */
type TodoRenderArgs = TodoRenderOp & {
ops?: TodoRenderOp[];
};
/**
* Normalize streaming/legacy render args to a flat op list. Accepts the new
* top-level `{op,...}` shape (returned as a one-element list), the legacy
* `{ops:[...]}` batch from old transcripts/collab-web, and partially-parsed
* streaming deltas (non-array `ops`, non-object entries) without crashing.
*/
function normalizeTodoArg(args: TodoRenderArgs | undefined): TodoRenderOp[] {
if (!args || typeof args !== "object") return [];
if (Array.isArray(args.ops)) {
return args.ops.filter((entry): entry is TodoRenderOp => !!entry && typeof entry === "object");
}
return typeof args.op === "string" ? [args] : [];
}
// =============================================================================
// Phase numbering (display-only)
// =============================================================================
@@ -794,7 +791,7 @@ function computeTouchedPhases(
for (const transition of completedTasks) touched.add(transition.phase);
// Phases explicitly named by the ops that ran. `init` replaces the whole
// list, so the entire plan is fresh and every phase counts as touched.
const ops = Array.isArray(args?.ops) ? args.ops : [];
const ops = normalizeTodoArg(args);
for (const op of ops) {
if (!op || typeof op !== "object") continue;
if (op.op === "init") {
@@ -823,18 +820,17 @@ function formatPhaseSummary(phase: TodoPhase, oneBasedIndex: number, uiTheme: Th
export const todoToolRenderer = {
renderCall(args: TodoRenderArgs, options: RenderResultOptions, uiTheme: Theme): Component {
// `args` here is the raw partially-parsed JSON from the streaming
// tool-call delta and may not satisfy `TodoRenderArgs` at runtime:
// `parseStreamingJson` can hand back `{ ops: "[" }` mid-delta, or
// entries that are `null` / strings before fields stream. Guard
// against non-array `ops` and non-object entries so a malformed
// delta never breaks the TUI render loop (#2005).
const opsList = Array.isArray(args?.ops) ? args.ops : [];
// `args` is the raw partially-parsed JSON from the streaming tool-call
// delta and may not satisfy `TodoRenderArgs` at runtime:
// `parseStreamingJson` can hand back `{ op: 1 }` mid-delta, or a legacy
// `{ ops: "[" }` shape before fields stream. `normalizeTodoArg` guards
// both the new single-op and legacy batch shapes so a malformed delta
// never breaks the TUI render loop (#2005).
const opsList = normalizeTodoArg(args);
const ops =
opsList.length === 0
? ["update"]
: opsList.map(entry => {
const e = entry && typeof entry === "object" ? entry : ({} as NonNullable<typeof entry>);
: opsList.map(e => {
const parts = [e.op ?? "update"];
if (e.task) parts.push(e.task);
if (e.phase) parts.push(e.phase);
+1 -1
View File
@@ -80,7 +80,7 @@ function formatHeader(options: CodeCellOptions, theme: Theme): { title: string;
parts.push(icon);
}
}
if (index !== undefined && total !== undefined) {
if (index !== undefined && total !== undefined && total > 1) {
parts.push(theme.fg("accent", `[${index + 1}/${total}]`));
}
if (title) {
@@ -3,6 +3,8 @@ import type { ImageContent } from "@oh-my-pi/pi-ai";
export interface ImageResizeOptions {
maxWidth?: number;
maxHeight?: number;
/** Smallest allowed edge length (px). Inputs below this are scaled up. */
minDimension?: number;
maxBytes?: number;
jpegQuality?: number;
excludeWebP?: boolean;
@@ -23,6 +25,13 @@ export interface ResizedImage {
// binding constraint once images are downsized to 1568px (Anthropic's internal threshold).
const DEFAULT_MAX_BYTES = 500 * 1024;
// Smallest edge length (px) vision backends reliably accept. They tile images into
// fixed patches (Anthropic uses 28px) and reject degenerate sub-patch images — e.g.
// the 1x1 PNG an empty chart render emits — with a hard 400 ("Could not process
// image") that can poison the whole request. 200px is the smallest size Anthropic
// documents as valid (200x200 = 64 visual tokens); undersized images are scaled up.
const DEFAULT_MIN_DIMENSION = 200;
const DEFAULT_OPTIONS: Required<Omit<ImageResizeOptions, "excludeWebP">> = {
// Anthropic's "internal recommended size" — Claude internally caps images at
// 1568px on the longest edge before vision processing.
@@ -30,6 +39,7 @@ const DEFAULT_OPTIONS: Required<Omit<ImageResizeOptions, "excludeWebP">> = {
maxHeight: 1568,
maxBytes: DEFAULT_MAX_BYTES,
jpegQuality: 80,
minDimension: DEFAULT_MIN_DIMENSION,
};
/**
@@ -87,7 +97,12 @@ export async function resizeImage(img: ImageContent, options?: ImageResizeOption
// still get JPEG-compressed.
const originalSize = inputBuffer.length;
const comfortableSize = opts.maxBytes / 4;
// Clamp the floor to the caps so an unusually small max can't demand an
// impossible "≥ min and ≤ max" target.
const minDimension = Math.min(opts.minDimension, opts.maxWidth, opts.maxHeight);
if (
originalWidth >= minDimension &&
originalHeight >= minDimension &&
originalWidth <= opts.maxWidth &&
originalHeight <= opts.maxHeight &&
originalSize <= comfortableSize &&
@@ -120,6 +135,21 @@ export async function resizeImage(img: ImageContent, options?: ImageResizeOption
targetHeight = opts.maxHeight;
}
// Lift undersized inputs up to the minimum. A uniform scale covers the
// common case (icons, the 1x1 chart) without distortion; an aspect ratio
// too extreme to satisfy both floor and cap falls back to stretching the
// lagging edge up to the floor via the default fit:"fill" resize.
if (targetWidth < minDimension || targetHeight < minDimension) {
const shortEdge = Math.min(targetWidth, targetHeight);
const upscale = Math.min(minDimension / shortEdge, opts.maxWidth / targetWidth, opts.maxHeight / targetHeight);
if (upscale > 1) {
targetWidth = Math.round(targetWidth * upscale);
targetHeight = Math.round(targetHeight * upscale);
}
targetWidth = Math.min(opts.maxWidth, Math.max(minDimension, targetWidth));
targetHeight = Math.min(opts.maxHeight, Math.max(minDimension, targetHeight));
}
// First-attempt encoder: try PNG and JPEG (+ WebP if not excluded) — return smallest.
// PNG wins for line art / few-color UI; JPEG wins for photographic content;
// WebP usually beats JPEG by 25–35% but is disabled when OMP_NO_WEBP is set
@@ -7,7 +7,7 @@
* them into a combined `answer` string on the SearchResponse.
*/
import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai";
import { settings } from "../../../config/settings";
import { getDefault, settings } from "../../../config/settings";
import { findApiKey, isSearchResponse } from "../../../exa/mcp-client";
import { parseSSE } from "../../../mcp/json-rpc";
import type { SearchResponse, SearchSource } from "../../../web/search/types";
@@ -18,6 +18,88 @@ import { SearchProvider } from "./base";
import { classifyProviderHttpError, withHardTimeout } from "./utils";
const EXA_API_URL = "https://api.exa.ai/search";
const DEFAULT_EXA_SEARCH_DELAY_MS = getDefault("exa.searchDelayMs");
let nextExaSearchRequestAt = 0;
let exaSearchThrottle = Promise.resolve();
function configuredExaSearchDelayMs(): number {
try {
const delayMs = settings.get("exa.searchDelayMs");
return Number.isFinite(delayMs) && delayMs > 0 ? Math.floor(delayMs) : 0;
} catch {
return DEFAULT_EXA_SEARCH_DELAY_MS;
}
}
function rejectWithAbortReason(reject: (reason?: unknown) => void, signal: AbortSignal): void {
try {
signal.throwIfAborted();
reject(new DOMException("The operation was aborted.", "AbortError"));
} catch (error) {
reject(error);
}
}
function abortableSleep(ms: number, signal: AbortSignal | undefined): Promise<void> {
if (ms <= 0) return Promise.resolve();
signal?.throwIfAborted();
const { promise, resolve, reject } = Promise.withResolvers<void>();
let timer: NodeJS.Timeout | undefined;
const cleanup = (): void => {
if (timer) {
clearTimeout(timer);
timer = undefined;
}
signal?.removeEventListener("abort", onAbort);
};
const onAbort = (): void => {
cleanup();
if (signal) rejectWithAbortReason(reject, signal);
};
timer = setTimeout(() => {
cleanup();
resolve();
}, ms);
signal?.addEventListener("abort", onAbort, { once: true });
if (signal?.aborted) onAbort();
return promise;
}
function waitUntilDoneOrAborted<T>(promise: Promise<T>, signal: AbortSignal | undefined): Promise<T> {
if (!signal) return promise;
signal.throwIfAborted();
const { promise: aborted, reject } = Promise.withResolvers<never>();
const onAbort = (): void => rejectWithAbortReason(reject, signal);
signal.addEventListener("abort", onAbort, { once: true });
return Promise.race([promise, aborted]).finally(() => {
signal.removeEventListener("abort", onAbort);
});
}
async function waitForExaSearchSlot(signal: AbortSignal | undefined): Promise<void> {
const delayMs = configuredExaSearchDelayMs();
if (delayMs <= 0) return;
const prior = exaSearchThrottle.catch(() => {});
const queued = prior.then(async () => {
signal?.throwIfAborted();
const waitMs = Math.max(0, nextExaSearchRequestAt - Date.now());
if (waitMs > 0) {
await abortableSleep(waitMs, signal);
}
signal?.throwIfAborted();
nextExaSearchRequestAt = Date.now() + delayMs;
});
exaSearchThrottle = queued.catch(() => {});
await waitUntilDoneOrAborted(queued, signal);
}
/** Reset Exa request pacing state for isolated provider tests. */
export function resetExaSearchThrottleForTest(): void {
nextExaSearchRequestAt = 0;
exaSearchThrottle = Promise.resolve();
}
type ExaSearchType = "neural" | "fast" | "auto" | "deep";
@@ -224,6 +306,7 @@ async function callExaSearch(apiKey: string, params: ExaSearchParams): Promise<E
const body = buildExaRequestBody(params);
const fetchImpl = params.fetch ?? fetch;
await waitForExaSearchSlot(params.signal);
const response = await fetchImpl(EXA_API_URL, {
method: "POST",
headers: {
@@ -260,6 +343,7 @@ async function callExaMcpSearch(params: ExaSearchParams): Promise<ExaSearchRespo
if (apiKey) query.set("exaApiKey", apiKey);
query.set("tools", "web_search_exa");
const fetchImpl = params.fetch ?? fetch;
await waitForExaSearchSlot(params.signal);
const response = await fetchImpl(`https://mcp.exa.ai/mcp?${query.toString()}`, {
method: "POST",
headers: {
@@ -247,7 +247,7 @@ describe("ACP event mapper", () => {
type: "tool_execution_start",
toolCallId: "tc-eval-start",
toolName: "eval",
args: { cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] },
args: { language: "js", title: "sum", code: "return 1 + 1;" },
intent: "sum",
} as AgentSessionEvent,
"session-1",
@@ -267,7 +267,7 @@ describe("ACP event mapper", () => {
expect(update.title).toBe("[js] sum\nreturn 1 + 1;");
expect(update.kind).toBe("execute");
expect(update.status).toBe("pending");
expect(update.rawInput).toEqual({ cells: [{ language: "js", title: "sum", code: "return 1 + 1;" }] });
expect(update.rawInput).toEqual({ language: "js", title: "sum", code: "return 1 + 1;" });
expect(update.content).toContainEqual({
type: "content",
content: { type: "text", text: "[js] sum\nreturn 1 + 1;" },
@@ -305,7 +305,7 @@ describe("ACP event mapper", () => {
type: "tool_execution_start",
toolCallId: "tc-eval-long-source",
toolName: "eval",
args: { cells: [{ language: "js", code: source }] },
args: { language: "js", code: source },
} as AgentSessionEvent,
"session-1",
);
@@ -1,6 +1,7 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as path from "node:path";
import { scheduler } from "node:timers/promises";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
@@ -10,6 +11,7 @@ import { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensi
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import * as unexpectedStopClassifier from "@oh-my-pi/pi-coding-agent/session/unexpected-stop-classifier";
import { getProjectAgentDir, TempDir, withTimeout } from "@oh-my-pi/pi-utils";
const runtimeSignalStoreKey = "__ompRuntimeSignals";
@@ -273,6 +275,382 @@ describe("AgentSession auto-compaction queue resume", () => {
expect(runtimeSignals.some(signal => signal.startsWith("compaction:end:"))).toBe(true);
});
it("triggers threshold compaction in active goals even when per-turn pruning shaves the post-prune estimate below threshold", async () => {
// Regression for #3174. Goal mode is the most common scenario: the agent
// runs many tool-result-heavy turns and the per-turn "useless" /
// "supersede" passes shave tokens off every check. Pre-fix
// `#checkCompaction` subtracted those savings from the threshold input, so
// with the reporter's fixed `compaction.thresholdTokens: 76384`, the
// threshold input fell below the trigger even when the provider-billed
// prompt (and the visible context anchored to it) sat above 90k tokens —
// auto-compaction silently no-op'd indefinitely while the loop kept
// running.
//
// This seeds one large `useless` tool result whose suffix sits inside the
// 8k cache-warm window so `#pruneStaleToolResults` actually returns ≥20k
// savings (well above the buggy code's mis-subtraction needed to drop
// 91000 below 76384). Compaction MUST still fire because the last turn's
// billed context tokens (91k) are above the configured threshold.
const now = Date.now();
// Seed: small user, small toolCall, ONE big useless tool result, then a
// handful of small turns that keep the suffix after the big result under
// the 8000-token cache-warm cutoff. The big result is the only viable
// prune candidate, and it alone saves well over 20k tokens — enough to
// drag the pre-fix threshold input from 91k well below 76384.
sessionManager.appendMessage({
role: "user",
content: "Investigate every module of the project.",
timestamp: now - 200,
});
const bigCallId = "call-big-useless";
sessionManager.appendMessage({
role: "assistant",
content: [{ type: "toolCall", id: bigCallId, name: "search", arguments: { pattern: "TODO" } }],
api: "anthropic-messages",
provider: "anthropic",
model: "claude-sonnet-4-5",
stopReason: "toolUse",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: now - 180,
});
sessionManager.appendMessage({
role: "toolResult",
toolCallId: bigCallId,
toolName: "search",
content: [{ type: "text", text: "match line\n".repeat(20000) }], // ~40k+ tokens
isError: false,
useless: true,
timestamp: now - 170,
});
// A few small follow-up turns so the big result's suffix stays inside the
// 8000-token cache-warm window. Each pair is well under a hundred tokens.
for (let i = 0; i < 4; i++) {
const smallId = `call-small-${i}`;
const ts = now - 160 + i * 2;
sessionManager.appendMessage({
role: "assistant",
content: [{ type: "toolCall", id: smallId, name: "read", arguments: { path: `note-${i}.md` } }],
api: "anthropic-messages",
provider: "anthropic",
model: "claude-sonnet-4-5",
stopReason: "toolUse",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: ts,
});
sessionManager.appendMessage({
role: "toolResult",
toolCallId: smallId,
toolName: "read",
content: [{ type: "text", text: `tiny note ${i}` }],
isError: false,
timestamp: ts + 1,
});
}
session.agent.replaceMessages(session.buildDisplaySessionContext().messages);
session.setGoalModeState({
enabled: true,
mode: "active",
goal: {
id: "goal-threshold-pruneable",
objective: "continue until compacted",
status: "active",
tokensUsed: 0,
timeUsedSeconds: 0,
createdAt: now,
updatedAt: now,
},
});
vi.spyOn(session.agent, "continue").mockImplementation(async () => {
session.agent.clearAllQueues();
});
session.settings.set("compaction.thresholdTokens", 76384);
session.settings.set("compaction.thresholdPercent", -1);
session.settings.set("compaction.strategy", "context-full");
session.settings.set("compaction.dropUseless", true);
session.settings.set("compaction.supersedeReads", true);
session.settings.set("compaction.keepRecentTokens", 10000);
session.settings.set("compaction.reserveTokens", 16384);
// Final assistant turn: billed at ~91k context tokens, just over the
// reporter's threshold. The pre-fix code would have subtracted ≥20k of
// prune savings and dropped the threshold input below 76384, skipping
// compaction. Post-fix it must trigger.
const finalAssistant = {
role: "assistant" as const,
content: [{ type: "text" as const, text: "Investigated module-7; continuing." }],
api: "anthropic-messages" as const,
provider: "anthropic" as const,
model: "claude-sonnet-4-5",
stopReason: "stop" as const,
usage: {
input: 5000,
output: 1000,
cacheRead: 85000,
cacheWrite: 0,
totalTokens: 91000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: now,
};
session.agent.emitExternalEvent({ type: "message_end", message: finalAssistant });
session.agent.emitExternalEvent({ type: "agent_end", messages: [finalAssistant] });
await session.waitForIdle();
const runtimeSignals = getRuntimeSignals();
expect(runtimeSignals).toContain("compaction:start:threshold");
expect(runtimeSignals.some(signal => signal.startsWith("compaction:end:"))).toBe(true);
});
it("runs active-goal threshold compaction before unexpected-stop retry continuation", async () => {
const now = Date.now();
session.setGoalModeState({
enabled: true,
mode: "active",
goal: {
id: "goal-unexpected-stop-threshold",
objective: "continue until compacted",
status: "active",
tokensUsed: 0,
timeUsedSeconds: 0,
createdAt: now,
updatedAt: now,
},
});
session.settings.set("compaction.thresholdTokens", 76384);
session.settings.set("compaction.thresholdPercent", -1);
session.settings.set("compaction.autoContinue", true);
session.settings.set("contextPromotion.enabled", false);
session.settings.set("features.unexpectedStopDetection", true);
session.settings.set("providers.unexpectedStopModel", "online");
vi.spyOn(unexpectedStopClassifier, "classifyUnexpectedStop").mockResolvedValue(true);
vi.spyOn(session.agent, "continue").mockImplementation(async () => {
session.agent.clearAllQueues();
});
const assistantMsg = {
role: "assistant" as const,
content: [{ type: "text" as const, text: "I should continue investigating another module." }],
api: "anthropic-messages" as const,
provider: "anthropic" as const,
model: "claude-sonnet-4-5",
stopReason: "stop" as const,
usage: {
input: 5000,
output: 1000,
cacheRead: 85000,
cacheWrite: 0,
totalTokens: 91000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: now,
};
session.agent.emitExternalEvent({ type: "message_end", message: assistantMsg });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMsg] });
await session.waitForIdle();
expect(getRuntimeSignals()).toContain("compaction:start:threshold");
});
it("resolves a pending retry before active-goal compaction continuation returns", async () => {
// Codex review on #3175: a retry can succeed with a non-empty text stop
// that is already over the active-goal compaction threshold. If the
// compaction pre-empt schedules its own continuation before the normal
// bottom-of-handler `#resolveRetry()` call runs, the session stays
// `isRetrying` and later prompt/idle gates remain blocked.
vi.useRealTimers();
const now = Date.now();
session.setGoalModeState({
enabled: true,
mode: "active",
goal: {
id: "goal-retry-threshold",
objective: "recover from retry and compact",
status: "active",
tokensUsed: 0,
timeUsedSeconds: 0,
createdAt: now,
updatedAt: now,
},
});
session.settings.set("compaction.thresholdTokens", 76384);
session.settings.set("compaction.thresholdPercent", -1);
session.settings.set("compaction.autoContinue", true);
session.settings.set("contextPromotion.enabled", false);
session.settings.set("retry.enabled", true);
session.settings.set("retry.baseDelayMs", 5);
session.settings.set("retry.maxDelayMs", 5_000);
session.settings.set("retry.maxRetries", 1);
session.settings.set("retry.modelFallback", false);
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
vi.spyOn(session.agent, "continue").mockImplementation(async () => {
session.agent.clearAllQueues();
});
const { promise: retryStarted, resolve: onRetryStarted } = Promise.withResolvers<void>();
const { promise: retryEnded, resolve: onRetryEnded } = Promise.withResolvers<void>();
const { promise: compactionDone, resolve: onCompactionDone } = Promise.withResolvers<void>();
session.subscribe(event => {
if (event.type === "auto_retry_start") onRetryStarted();
if (event.type === "auto_retry_end") onRetryEnded();
if (event.type === "auto_compaction_end") onCompactionDone();
});
const retryableError = {
role: "assistant" as const,
content: [{ type: "text" as const, text: "Transient provider failure." }],
api: "anthropic-messages" as const,
provider: "anthropic" as const,
model: "claude-sonnet-4-5",
stopReason: "error" as const,
errorMessage: "503 service unavailable: overloaded_error retry-after-ms=50",
usage: {
input: 100,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 100,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: now - 1,
};
session.agent.emitExternalEvent({ type: "message_end", message: retryableError });
session.agent.emitExternalEvent({ type: "agent_end", messages: [retryableError] });
await withTimeout(retryStarted, 1000, "Retry start timed out");
expect(session.isRetrying).toBe(true);
const recoveredOverThreshold = {
role: "assistant" as const,
content: [{ type: "text" as const, text: "Recovered; continuing the active goal." }],
api: "anthropic-messages" as const,
provider: "anthropic" as const,
model: "claude-sonnet-4-5",
stopReason: "stop" as const,
usage: {
input: 5000,
output: 1000,
cacheRead: 85000,
cacheWrite: 0,
totalTokens: 91000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: now,
};
session.agent.emitExternalEvent({ type: "message_end", message: recoveredOverThreshold });
await withTimeout(retryEnded, 1000, "Retry end timed out");
expect(session.isRetrying).toBe(true);
session.agent.emitExternalEvent({ type: "agent_end", messages: [recoveredOverThreshold] });
await withTimeout(compactionDone, 1000, "Compaction end timed out");
await session.waitForIdle();
expect(getRuntimeSignals()).toContain("compaction:start:threshold");
expect(session.isRetrying).toBe(false);
});
it("removes orphan toolUse assistant before active-goal threshold compaction continuation", async () => {
// Codex review on #3175: when an active goal turn is over threshold AND
// stops with an empty `toolUse` (no tool call), the new ordering must NOT
// skip `#handleEmptyAssistantStop` — that handler is the only path that
// strips the orphan assistant from active context + session history. If a
// compaction continuation runs with the orphan still in place, the next
// Anthropic turn carries a `tool_use` block with no matching
// `tool_result` and corrupts the message history.
const now = Date.now();
session.setGoalModeState({
enabled: true,
mode: "active",
goal: {
id: "goal-orphan-toolUse-threshold",
objective: "continue until compacted",
status: "active",
tokensUsed: 0,
timeUsedSeconds: 0,
createdAt: now,
updatedAt: now,
},
});
session.settings.set("compaction.thresholdTokens", 76384);
session.settings.set("compaction.thresholdPercent", -1);
session.settings.set("compaction.autoContinue", true);
session.settings.set("contextPromotion.enabled", false);
vi.spyOn(session.agent, "continue").mockImplementation(async () => {
session.agent.clearAllQueues();
});
const orphanToolUse = {
role: "assistant" as const,
// Empty toolUse stop: stopReason says a tool was requested but the
// content block is empty (no toolCall). This is the case the empty-stop
// cleanup defends against.
content: [] as never[],
api: "anthropic-messages" as const,
provider: "anthropic" as const,
model: "claude-sonnet-4-5",
stopReason: "toolUse" as const,
usage: {
input: 5000,
output: 1000,
cacheRead: 85000,
cacheWrite: 0,
totalTokens: 91000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: now,
};
session.agent.emitExternalEvent({ type: "message_end", message: orphanToolUse });
session.agent.emitExternalEvent({ type: "agent_end", messages: [orphanToolUse] });
await session.waitForIdle();
// Empty-stop cleanup short-circuits before any compaction continuation, so
// the threshold compaction MUST NOT fire on this turn — the next turn
// starts from the cleaned-up branch with the retry-reminder developer
// message instead. The pre-fix ordering let compaction reach
// `auto_compaction_start` first, scheduling a continuation while the
// orphan `toolUse` entry was still the session leaf.
const signals = getRuntimeSignals();
expect(signals).not.toContain("compaction:start:threshold");
// `#removeEmptyStopFromActiveContext` rewinds the session leaf past the
// orphan via `sessionManager.branch(parentId)` / `resetLeaf()`. If the
// cleanup is skipped, the orphan is still the leaf when the compaction
// continuation runs and the next Anthropic turn sends a `tool_use` block
// with no matching `tool_result`.
const branch = sessionManager.getBranch();
const orphanInBranch = branch.some(entry => {
if (entry.type !== "message") return false;
const message = entry.message as { role: string; stopReason?: string };
return message.role === "assistant" && message.stopReason === "toolUse";
});
expect(orphanInBranch).toBe(false);
});
it("has isCompacting true when the auto_compaction_start event fires", async () => {
// Defect 1: the compaction AbortController (which backs isCompacting) must be
// installed before auto_compaction_start is emitted. If it is installed after,
@@ -220,12 +220,8 @@ describe("AgentSession eager todo enforcement", () => {
it("initializes todos once, then continues within the same user turn", async () => {
scriptedResponses = [
createToolCallAssistantMessage("todo", {
ops: [
{
op: "init",
list: [{ phase: "List worktrees", items: ["List all git worktrees in the current repository"] }],
},
],
op: "init",
list: [{ phase: "List worktrees", items: ["List all git worktrees in the current repository"] }],
}),
createAssistantMessage("real user turn handled"),
];
@@ -0,0 +1,181 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core";
import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { GoalModeState } from "@oh-my-pi/pi-coding-agent/goals/state";
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { TempDir } from "@oh-my-pi/pi-utils";
import { type } from "arktype";
function activeGoalState(): GoalModeState {
const now = Date.now();
return {
enabled: true,
mode: "active",
goal: {
id: "goal-midrun-compaction",
objective: "Ship the release",
status: "active",
tokensUsed: 0,
timeUsedSeconds: 0,
createdAt: now,
updatedAt: now,
},
};
}
function highUsage(input: number) {
return {
input,
output: 100,
cacheRead: 0,
cacheWrite: 0,
totalTokens: input + 100,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
describe("AgentSession mid-run goal compaction", () => {
let tempDir: TempDir;
const cleanups: Array<() => Promise<void>> = [];
beforeEach(() => {
tempDir = TempDir.createSync("@pi-agent-goal-midrun-compaction-");
cleanups.length = 0;
});
afterEach(async () => {
for (const cleanup of cleanups) await cleanup();
cleanups.length = 0;
tempDir.removeSync();
vi.restoreAllMocks();
});
async function createHarness(settingsOverride: Record<string, unknown> = {}): Promise<{
session: AgentSession;
observedContexts: string[][];
}> {
const observedContexts: string[][] = [];
const model = getBundledModel("anthropic", "claude-sonnet-4-5");
if (!model) throw new Error("Expected claude-sonnet-4-5 model to exist");
const authStorage = await AuthStorage.create(path.join(tempDir.path(), `testauth-${cleanups.length}.db`));
authStorage.setRuntimeApiKey("anthropic", "test-key");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), `models-${cleanups.length}.yml`));
const settings = Settings.isolated({
"compaction.enabled": true,
"compaction.strategy": "context-full",
"compaction.autoContinue": true,
"compaction.thresholdTokens": 1000,
"compaction.thresholdPercent": -1,
"todo.enabled": false,
"todo.reminders": false,
...settingsOverride,
});
const sessionManager = SessionManager.inMemory(tempDir.path());
const mockBashTool: AgentTool = {
name: "bash",
label: "Bash",
description: "Mock bash tool",
parameters: type({}),
execute: async () => ({ content: [{ type: "text" as const, text: "tool output" }] }),
};
let call = 0;
const agent = new Agent({
getApiKey: () => "test-key",
initialState: { model, systemPrompt: ["Test"], tools: [mockBashTool], messages: [] },
convertToLlm,
streamFn: (_model, context) => {
const index = call++;
observedContexts.push(context.messages.map(message => JSON.stringify(message)));
const stream = new AssistantMessageEventStream();
const isToolTurn = index === 0;
const message = isToolTurn
? {
role: "assistant" as const,
content: [
{ type: "toolCall" as const, id: `tc-${index}`, name: "bash", arguments: { cmd: "ls" } },
],
api: "anthropic-messages" as const,
provider: "anthropic" as const,
model: "claude-sonnet-4-5",
usage: highUsage(50_000),
stopReason: "toolUse" as const,
timestamp: Date.now(),
}
: {
role: "assistant" as const,
content: [{ type: "text" as const, text: "All done." }],
api: "anthropic-messages" as const,
provider: "anthropic" as const,
model: "claude-sonnet-4-5",
usage: highUsage(200),
stopReason: "stop" as const,
timestamp: Date.now(),
};
queueMicrotask(() => {
stream.push({ type: "start", partial: message });
stream.push({ type: "done", reason: message.stopReason, message });
});
return stream;
},
});
const session = new AgentSession({
agent,
sessionManager,
settings,
modelRegistry,
toolRegistry: new Map([[mockBashTool.name, mockBashTool]]),
});
cleanups.push(async () => {
await session.dispose();
authStorage.close();
});
return { session, observedContexts };
}
it("compacts in place between tool-call turns during an active goal run", async () => {
const { session, observedContexts } = await createHarness();
session.setGoalModeState(activeGoalState());
const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async preparation => ({
summary: "MID-RUN-COMPACTED",
shortSummary: undefined,
firstKeptEntryId: preparation.firstKeptEntryId,
tokensBefore: preparation.tokensBefore,
details: {},
}));
await session.prompt("work on the release");
expect(compactSpy).toHaveBeenCalledTimes(1);
expect(observedContexts.length).toBeGreaterThanOrEqual(2);
expect(observedContexts[1].join("\n")).toContain("MID-RUN-COMPACTED");
});
it("does not compact mid-run when no goal is active", async () => {
const { session } = await createHarness();
const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async preparation => ({
summary: "SHOULD-NOT-RUN",
shortSummary: undefined,
firstKeptEntryId: preparation.firstKeptEntryId,
tokensBefore: preparation.tokensBefore,
details: {},
}));
await session.prompt("work on the release");
expect(compactSpy).not.toHaveBeenCalled();
});
});
@@ -426,7 +426,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
let toolExecutionSettled = false;
const toolExecution = EvalTool!
.execute("call-id", { cells: [{ language: "py", code: "print('tool')" }] }, undefined, undefined, undefined)
.execute("call-id", { language: "py", code: "print('tool')" }, undefined, undefined, undefined)
.finally(() => {
toolExecutionSettled = true;
});
@@ -652,13 +652,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
const disposeSession = session.dispose();
await expect(
EvalTool!.execute(
"call-id",
{ cells: [{ language: "py", code: "print('late')" }] },
undefined,
undefined,
undefined,
),
EvalTool!.execute("call-id", { language: "py", code: "print('late')" }, undefined, undefined, undefined),
).rejects.toThrow("Python execution is unavailable while session disposal is in progress");
await disposeSession;
expect(executeSpy).not.toHaveBeenCalled();
@@ -693,7 +687,7 @@ describe("AgentSession python cleanup", () => {
expect(EvalTool).toBeDefined();
const execution = EvalTool!.execute(
"call-id",
{ cells: [{ language: "py", code: "print('late after artifact')" }] },
{ language: "py", code: "print('late after artifact')" },
undefined,
undefined,
undefined,
@@ -14,6 +14,7 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import {
AUTO_THINKING,
clampAutoThinkingEffort,
parseCliThinkingLevel,
parseConfiguredThinkingLevel,
parseEffort,
parseThinkingLevel,
@@ -65,6 +66,14 @@ describe("auto thinking classifier helpers", () => {
expect(parseThinkingLevel(ThinkingLevel.Off)).toBe(ThinkingLevel.Off);
});
it("parses CLI --thinking selectors while rejecting inherit", () => {
expect(parseCliThinkingLevel(ThinkingLevel.Off)).toBe(ThinkingLevel.Off);
expect(parseCliThinkingLevel(AUTO_THINKING)).toBe(AUTO_THINKING);
expect(parseCliThinkingLevel("max")).toBe(ThinkingLevel.XHigh);
expect(parseCliThinkingLevel(ThinkingLevel.Inherit)).toBeUndefined();
expect(parseCliThinkingLevel("bogus")).toBeUndefined();
});
it("maps online 4-way classifier labels to effort levels", () => {
expect(parseDifficultyLevel("x-high")).toBe(Effort.XHigh);
expect(parseDifficultyLevel("The answer is HIGH.")).toBe(Effort.High);
@@ -974,7 +974,7 @@ describe("executeBash :async: background retention", () => {
});
it.skipIf(process.platform === "win32")(
"keeps a per-job :async: shell's background process alive across turns",
"keeps a per-job :async: shell's plain-`&` background process alive across turns",
async () => {
const pidFile = path.join(tmp, "pid");
const sleepBin = fs.existsSync("/bin/sleep") ? "/bin/sleep" : "sleep";
@@ -982,10 +982,11 @@ describe("executeBash :async: background retention", () => {
try {
// A per-job `:async:` key: its shell is removed from the reuse map at
// teardown, which would SIGKILL the backgrounded child (kill-on-drop).
// The retain logic keeps the shell alive while a background process is
// still running. `$!` is the external child's pid (nohup is a
// transparent background wrapper).
const res = await executeBash(`nohup ${sleepBin} 30 >/dev/null 2>&1 & echo $! > ${shellQuote(pidFile)}`, {
// A plain `&` job stays a child of the shell, so `liveBackgroundJobCount`
// sees it and the retain logic keeps the shell alive while the child
// runs. `$!` is the external child's own pid (no transparent wrapper to
// unwrap), so it is the process we assert on.
const res = await executeBash(`${sleepBin} 30 >/dev/null 2>&1 & echo $! > ${shellQuote(pidFile)}`, {
sessionKey: "retain-probe:async:job1",
cwd: tmp,
});
@@ -1012,4 +1013,48 @@ describe("executeBash :async: background retention", () => {
}
},
);
it.skipIf(process.platform === "win32")(
"keeps a nohup-detached background process alive across turns (reparenting)",
async () => {
const pidFile = path.join(tmp, "nohup-pid");
const sleepBin = fs.existsSync("/bin/sleep") ? "/bin/sleep" : "sleep";
let pid: number | undefined;
try {
// `nohup cmd &` is a transparent background wrapper: brush unwraps it and
// double-forks the operand so it reparents to init and survives teardown
// independently of the retain map. The shell only ever tracked the
// short-lived intermediate fork, so `$!` is NOT the surviving process —
// the operand writes its own pid before `exec`ing the long sleep, and
// that pid (unchanged across exec) is the one we assert stays alive.
const operand = `echo $$ > ${pidFile}; exec ${sleepBin} 30`;
const res = await executeBash(`nohup sh -c ${shellQuote(operand)} >/dev/null 2>&1 &`, {
sessionKey: "reparent-probe:async:job1",
cwd: tmp,
});
expect(res.cancelled).toBe(false);
await pollUntil(() => fs.existsSync(pidFile), Date.now() + 4000);
pid = Number.parseInt(fs.readFileSync(pidFile, "utf8").trim(), 10);
expect(Number.isInteger(pid)).toBe(true);
// A later turn on a different per-job shell must not have killed it.
await executeBash("true", { sessionKey: "reparent-probe:async:job2", cwd: tmp });
let alive = true;
try {
process.kill(pid, 0);
} catch {
alive = false;
}
expect(alive).toBe(true);
} finally {
if (pid !== undefined) {
try {
process.kill(pid, "SIGKILL");
} catch {}
}
}
},
);
});
@@ -1,6 +1,8 @@
import { describe, expect, it } from "bun:test";
import { ThinkingLevel } from "@oh-my-pi/pi-agent-core";
import { Effort } from "@oh-my-pi/pi-ai";
import { parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args";
import { AUTO_THINKING } from "@oh-my-pi/pi-coding-agent/thinking";
describe("parseArgs — --hide-thinking flag", () => {
it("parses --hide-thinking as a boolean flag", () => {
@@ -43,3 +45,21 @@ describe("parseArgs — --hide-thinking flag", () => {
expect(result.messages).toEqual([]);
});
});
describe("parseArgs — --thinking flag", () => {
it("accepts off so reasoning can be disabled from the CLI", () => {
expect(parseArgs(["--thinking", "off"]).thinking).toBe(ThinkingLevel.Off);
expect(parseArgs(["--thinking=off"]).thinking).toBe(ThinkingLevel.Off);
});
it("accepts auto, concrete efforts, and the max alias", () => {
expect(parseArgs(["--thinking", "auto"]).thinking).toBe(AUTO_THINKING);
expect(parseArgs(["--thinking", "medium"]).thinking).toBe(Effort.Medium);
expect(parseArgs(["--thinking", "max"]).thinking).toBe(ThinkingLevel.XHigh);
});
it("ignores invalid levels and the internal inherit selector", () => {
expect(parseArgs(["--thinking", "bogus"]).thinking).toBeUndefined();
expect(parseArgs(["--thinking", "inherit"]).thinking).toBeUndefined();
});
});
@@ -211,7 +211,7 @@ describe("omp completions (integration / drift)", () => {
}
expect(stdout).toContain("{-r,--resume}");
// Real enum option sets flow through unchanged.
expect(stdout).toContain(":value:(minimal low medium high xhigh)");
expect(stdout).toContain(":value:(off minimal low medium high xhigh auto)");
expect(stdout).toContain(":value:(always-ask write yolo)");
// Real subcommands present; dynamic callbacks wired.
expect(stdout).toContain("_omp_cmd_commit");
@@ -144,13 +144,10 @@ describe.skipIf(!SHOULD_RUN)("ruby runner subprocess", () => {
}
});
it("exposes prelude file + text helpers", async () => {
it("exposes prelude file helpers", async () => {
using tempDir = TempDir.createSync("@ruby-runner-prelude-");
const kernel = await RubyKernel.start({ cwd: tempDir.path() });
try {
const sorted = await executeRubyWithKernel(kernel, 'sort("b\\na\\nb", unique: true)', {});
expect(sorted.output).toContain("a\nb");
const written = await executeRubyWithKernel(kernel, 'write("note.txt", "hello"); read("note.txt")', {});
expect(written.output).toContain("hello");
} finally {
@@ -55,7 +55,7 @@ describe("runEvalAgent", () => {
getAgentId: () => "BridgeParent",
} as unknown as ToolSession;
await runEvalAgent({ prompt: "do work", agentType: "task" }, { session });
await runEvalAgent({ prompt: "do work", agent: "task" }, { session });
expect(runSubprocessSpy).toHaveBeenCalledTimes(1);
const options = runSubprocessSpy.mock.calls[0]?.[0];
@@ -0,0 +1,84 @@
/**
* Regression: cancelling the startup `--resume` session picker (e.g. pressing
* Esc) must terminate the process cleanly. Startup arms long-lived handles
* (theme/appearance listeners via initTheme, settings save timer, model
* registry), so the previous bare `return` left the event loop with live
* handles and the process hung after the picker left the alternate screen.
*
* The fix exits via `process.exit(0)` — matching the `--version`/`--export`
* early-exit convention in the same function. Only this startup call site
* exits; the in-session `/resume` picker (selector-controller.ts) keeps its own
* onCancel that just closes the overlay.
*/
import { describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { runRootCommand } from "@oh-my-pi/pi-coding-agent/main";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { TempDir } from "@oh-my-pi/pi-utils";
class ProcessExitSignal extends Error {
constructor(readonly code: number) {
super(`process.exit(${code})`);
this.name = "ProcessExitSignal";
}
}
describe("runRootCommand — startup --resume picker cancellation", () => {
it("exits cleanly (process.exit 0) when the picker is cancelled instead of returning and hanging", async () => {
using tempDir = TempDir.createSync("@omp-resume-cancel-");
const sessionDir = tempDir.path();
// One valid session so folderSessions is non-empty and the picker (not the
// "No sessions found" probe) is the path under test.
await Bun.write(
path.join(sessionDir, "existing.jsonl"),
`${JSON.stringify({ type: "session", id: "existing-session", cwd: sessionDir, timestamp: new Date().toISOString() })}\n`,
);
const authStorage = await AuthStorage.create(path.join(sessionDir, "auth.db"));
const settings = Settings.isolated({ "marketplace.autoUpdate": "off" });
// --print keeps initTheme non-interactive so no global appearance/SIGWINCH
// listeners leak into the rest of the suite; the picker branch is gated on
// `resume === true`, not on interactivity, so it still runs.
const parsed = parseArgs(["--resume", "--print"]);
parsed.noExtensions = true;
parsed.noSkills = true;
parsed.noRules = true;
parsed.noTools = true;
parsed.noLsp = true;
parsed.sessionDir = sessionDir;
const exitCodes: number[] = [];
vi.spyOn(process, "exit").mockImplementation(((code?: number) => {
exitCodes.push(code ?? 0);
throw new ProcessExitSignal(code ?? 0);
}) as typeof process.exit);
vi.spyOn(process.stdout, "write").mockImplementation(() => true);
let pickerCalled = false;
let thrown: unknown;
try {
await runRootCommand(parsed, ["--resume", "--print"], {
discoverAuthStorage: async () => authStorage,
settings,
selectSession: async () => {
pickerCalled = true;
return null; // user cancelled (Esc)
},
});
} catch (err) {
thrown = err;
} finally {
vi.restoreAllMocks();
authStorage.close();
}
expect(pickerCalled).toBe(true);
expect(thrown).toBeInstanceOf(ProcessExitSignal);
// Exactly one clean exit — proves the cancel branch terminates instead of
// falling through to session creation or returning into a hang.
expect(exitCodes).toEqual([0]);
}, 15_000);
});
@@ -0,0 +1,44 @@
import { afterEach, describe, expect, it } from "bun:test";
import { handleInputOrEscape } from "@oh-my-pi/pi-coding-agent/modes/components/plugin-settings";
import { setKittyProtocolActive } from "@oh-my-pi/pi-tui";
afterEach(() => {
setKittyProtocolActive(false);
});
describe("handleInputOrEscape", () => {
it("cancels on a kitty CSI-u escape (the fullscreen settings overlay encoding)", () => {
// Ghostty/kitty report Escape as `\x1b[27u` once the keyboard protocol is
// active (which it is inside the fullscreen settings overlay). A raw `\x1b`
// compare misses it, so Esc looked dead in the text-input submenu.
setKittyProtocolActive(true);
let cancelled = false;
const forwarded: string[] = [];
handleInputOrEscape("\x1b[27u", { handleInput: data => forwarded.push(data) }, () => {
cancelled = true;
});
expect(cancelled).toBe(true);
expect(forwarded).toEqual([]);
});
it("cancels on a legacy bare escape", () => {
let cancelled = false;
const forwarded: string[] = [];
handleInputOrEscape("\x1b", { handleInput: data => forwarded.push(data) }, () => {
cancelled = true;
});
expect(cancelled).toBe(true);
expect(forwarded).toEqual([]);
});
it("forwards a printable keystroke to the input instead of cancelling", () => {
setKittyProtocolActive(true);
let cancelled = false;
const forwarded: string[] = [];
handleInputOrEscape("g", { handleInput: data => forwarded.push(data) }, () => {
cancelled = true;
});
expect(cancelled).toBe(false);
expect(forwarded).toEqual(["g"]);
});
});
@@ -0,0 +1,140 @@
import { beforeAll, describe, expect, it } from "bun:test";
import { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-listing";
beforeAll(async () => {
await initTheme();
});
function makeSession(id: string, title: string | undefined): SessionInfo {
return {
path: `/work/${id}.jsonl`,
id,
cwd: "/work",
title,
created: new Date("2024-01-01T00:00:00Z"),
modified: new Date("2024-01-02T00:00:00Z"),
messageCount: 1,
size: 1024,
firstMessage: `body for ${id}`,
allMessagesText: `body for ${id}`,
};
}
/** SGR left-button press at a 1-based screen row (column is irrelevant for row hit-testing). */
function leftClick(row1Based: number, col1Based = 4): string {
return `\x1b[<0;${col1Based};${row1Based}M`;
}
/** SGR wheel notch: button 64 = up, 65 = down. */
function wheel(direction: "up" | "down"): string {
return `\x1b[<${direction === "down" ? 65 : 64};1;1M`;
}
function makeSelector(
sessions: SessionInfo[],
onSelect: (s: SessionInfo) => void,
rows = 40,
): SessionSelectorComponent {
return new SessionSelectorComponent(
sessions,
onSelect,
() => {},
() => {},
{
getTerminalRows: () => rows,
fillHeight: true,
},
);
}
describe("SessionSelectorComponent mouse", () => {
it("resumes the session under a left click", () => {
const sessions = [
makeSession("aaaa", "Alpha session"),
makeSession("bbbb", "Beta session"),
makeSession("cccc", "Gamma session"),
];
let picked: SessionInfo | undefined;
const selector = makeSelector(sessions, s => {
picked = s;
});
// Render first so the hit-test map and list offset reflect this frame.
const lines = selector.render(80);
const betaRow = lines.findIndex(line => line.includes("Beta session"));
expect(betaRow).toBeGreaterThanOrEqual(0);
// Mouse rows are 1-based; the fullscreen overlay paints from screen row 0.
selector.handleInput(leftClick(betaRow + 1));
expect(picked?.id).toBe("bbbb");
});
it("scrolls the selection with the wheel, then resumes it on Enter", () => {
const sessions = [
makeSession("aaaa", "Alpha session"),
makeSession("bbbb", "Beta session"),
makeSession("cccc", "Gamma session"),
];
let picked: SessionInfo | undefined;
const selector = makeSelector(sessions, s => {
picked = s;
});
selector.render(80);
// Selection starts at the first row; two notches down lands on Gamma.
selector.handleInput(wheel("down"));
selector.handleInput(wheel("down"));
selector.handleInput("\n");
expect(picked?.id).toBe("cccc");
});
it("ignores a click on the pinned footer (never resumes a hidden session)", () => {
const sessions = Array.from({ length: 20 }, (_, i) => makeSession(`s${i}`, `Title ${i}`));
let picked: SessionInfo | undefined;
const selector = makeSelector(
sessions,
s => {
picked = s;
},
40,
);
const lines = selector.render(80);
const footerRow = lines.findIndex(line => line.includes("Esc cancel"));
expect(footerRow).toBeGreaterThanOrEqual(0);
// Click directly on the footer hint row: must not resume anything.
selector.handleInput(leftClick(footerRow + 1));
expect(picked).toBeUndefined();
});
});
describe("SessionSelectorComponent fill-height footer", () => {
// First half titled (4 rows each), second half untitled (3 rows each), so the
// scrolled window changes height — the regression that made the footer drift.
function mixedSessions(count: number): SessionInfo[] {
return Array.from({ length: count }, (_, i) => makeSession(`s${i}`, i < count / 2 ? `Titled ${i}` : undefined));
}
it("fills the viewport and pins the footer to the bottom regardless of scroll", () => {
const rows = 40;
const selector = makeSelector(mixedSessions(20), () => {}, rows);
const top = selector.render(80);
const topHint = top.findIndex(line => line.includes("Esc cancel"));
expect(top.length).toBe(rows);
expect(topHint).toBe(rows - 3);
expect(top[rows - 1]!.trim().length).toBeGreaterThan(0); // bottom border on the last row
// Scroll to the bottom of the list (now an untitled window of a different
// height); the footer must not move.
for (let i = 0; i < 25; i++) selector.handleInput(wheel("down"));
const bottom = selector.render(80);
const bottomHint = bottom.findIndex(line => line.includes("Esc cancel"));
expect(bottom.length).toBe(rows);
expect(bottomHint).toBe(topHint);
expect(bottom[rows - 1]!.trim().length).toBeGreaterThan(0);
});
});
@@ -0,0 +1,166 @@
import { beforeAll, describe, expect, it } from "bun:test";
import { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-listing";
import { TUI } from "@oh-my-pi/pi-tui";
import { StressRenderScheduler } from "../../../../tui/test/render-stress-scheduler";
import { VirtualTerminal } from "../../../../tui/test/virtual-terminal";
beforeAll(() => {
initTheme();
});
function makeSessions(count: number): SessionInfo[] {
return Array.from({ length: count }, (_, i) => ({
path: `/work/SESSION_${i}.jsonl`,
id: `id-${i}`,
cwd: "/work",
title: `SESSION_${i}`,
created: new Date("2024-01-01T00:00:00Z"),
modified: new Date("2024-01-02T00:00:00Z"),
messageCount: 1,
size: 1024,
firstMessage: `body content ${i}`,
allMessagesText: `body content ${i}`,
}));
}
describe("issue #3283: /resume picker scrolls down after deleting a session", () => {
it("keeps the picker header pinned at the same viewport row before and after a delete", async () => {
const term = new VirtualTerminal(80, 24, 4096);
const scheduler = new StressRenderScheduler();
const tui = new TUI(term, undefined, { renderScheduler: scheduler });
const selector = new SessionSelectorComponent(
makeSessions(20),
() => {},
() => {},
() => {},
{
getTerminalRows: () => term.rows,
onDelete: async () => true,
},
);
selector.setOnRequestRender(() => tui.requestRender());
tui.addChild(selector);
tui.setFocus(selector);
try {
tui.start();
await scheduler.drain(term);
const headerRowBefore = term.getViewport().findIndex(row => Bun.stripANSI(row).includes("Resume Session"));
expect(headerRowBefore).toBeGreaterThanOrEqual(0);
// Press Delete (CSI 3 ~) to open the confirmation dialog, then
// Enter to accept "Yes".
selector.handleInput("\x1b[3~");
tui.requestRender();
await scheduler.drain(term);
selector.handleInput("\n");
// onDelete is async; let its microtasks flush before draining renders.
for (let i = 0; i < 8; i++) await Promise.resolve();
await scheduler.drain(term);
const viewport = term.getViewport().map(row => Bun.stripANSI(row).trimEnd());
const headerRowAfter = viewport.findIndex(row => row.includes("Resume Session"));
// Regression: dialog growing the frame and then shrinking must
// not push the picker header further down into the viewport
// (committed scrollback rows from the dialog frame).
expect(headerRowAfter).toBeGreaterThanOrEqual(0);
expect(headerRowAfter).toBe(headerRowBefore);
} finally {
tui.stop();
await term.flush();
}
});
it("keeps the picker header pinned even when the delete dialog is canceled", async () => {
const term = new VirtualTerminal(80, 24, 4096);
const scheduler = new StressRenderScheduler();
const tui = new TUI(term, undefined, { renderScheduler: scheduler });
const selector = new SessionSelectorComponent(
makeSessions(20),
() => {},
() => {},
() => {},
{
getTerminalRows: () => term.rows,
onDelete: async () => true,
},
);
selector.setOnRequestRender(() => tui.requestRender());
tui.addChild(selector);
tui.setFocus(selector);
try {
tui.start();
await scheduler.drain(term);
const headerRowBefore = term.getViewport().findIndex(row => Bun.stripANSI(row).includes("Resume Session"));
expect(headerRowBefore).toBeGreaterThanOrEqual(0);
// Open dialog, then Esc to cancel without deleting.
selector.handleInput("\x1b[3~");
tui.requestRender();
await scheduler.drain(term);
selector.handleInput("\x1b");
await scheduler.drain(term);
const viewport = term.getViewport().map(row => Bun.stripANSI(row).trimEnd());
const headerRowAfter = viewport.findIndex(row => row.includes("Resume Session"));
expect(headerRowAfter).toBe(headerRowBefore);
// Dialog gone, no scroll-down artefact.
expect(viewport.some(row => row.includes("Delete session?"))).toBe(false);
} finally {
tui.stop();
await term.flush();
}
});
it("keeps the picker frame bounded on a narrow terminal where the dialog title wraps", () => {
// Direct structural contract: even on a small terminal with a
// session name that wraps the dialog past any plausible fixed
// reserve, the picker's total rendered output stays within the
// terminal height. The dialog REPLACES the SessionList inside
// the picker frame, so its rows compete only with the
// SessionList's rendered budget — not with both the SessionList
// AND picker chrome (PR #3285 second review round). Without the
// structural fix the dialog could push the picker past the
// viewport once the SessionList reserve could not shrink any
// further, and the TUI committed the header into scrollback.
const longName = "a-very-very-very-very-long-session-title-that-must-wrap-on-a-narrow-terminal";
const sessions: SessionInfo[] = [
{
path: `/work/${longName}.jsonl`,
id: "id-long",
cwd: "/work",
title: longName,
created: new Date("2024-01-01T00:00:00Z"),
modified: new Date("2024-01-02T00:00:00Z"),
messageCount: 1,
size: 1024,
firstMessage: longName,
allMessagesText: longName,
},
...makeSessions(10),
];
const NARROW_WIDTH = 30;
const TERMINAL_ROWS = 24;
const selector = new SessionSelectorComponent(
sessions,
() => {},
() => {},
() => {},
{ getTerminalRows: () => TERMINAL_ROWS, onDelete: async () => true },
);
// Baseline: picker fits the viewport before the dialog opens.
expect(selector.render(NARROW_WIDTH).length).toBeLessThanOrEqual(TERMINAL_ROWS);
// Open the delete confirmation; the dialog takes the SessionList
// slot inside the picker frame. The picker's rendered total must
// STILL fit the viewport even though the dialog wraps past the
// previous 12-row reserve guess.
selector.handleInput("\x1b[3~");
expect(selector.render(NARROW_WIDTH).length).toBeLessThanOrEqual(TERMINAL_ROWS);
});
});
@@ -1,284 +0,0 @@
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import { SessionSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/session-selector";
import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller";
import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types";
import type { SessionInfo } from "@oh-my-pi/pi-coding-agent/session/session-listing";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { FileSessionStorage } from "@oh-my-pi/pi-coding-agent/session/session-storage";
type TestContext = InteractiveModeContext & {
editorContainer: {
children: unknown[];
clear: () => void;
addChild: (child: unknown) => void;
};
};
function makeSessionInfo(path: string): SessionInfo {
return {
path,
id: path,
cwd: "/tmp/project",
title: "Active session",
created: new Date("2025-01-01T00:00:00Z"),
modified: new Date("2025-01-01T00:00:00Z"),
messageCount: 1,
size: 0,
firstMessage: "hello",
allMessagesText: "hello",
};
}
function createContext(currentSessionFile: string): {
ctx: TestContext;
calls: string[];
setCurrentSessionFile: (path: string) => void;
showHookConfirm: (title: string, message: string) => Promise<boolean>;
newSession: () => Promise<boolean>;
} {
const calls: string[] = [];
let sessionFile = currentSessionFile;
const editorContainer = {
children: [] as unknown[],
clear() {
this.children = [];
calls.push("editorContainer.clear");
},
addChild(child: unknown) {
this.children.push(child);
calls.push("editorContainer.addChild");
},
};
const showHookConfirm = vi.fn(async () => true);
const newSession = vi.fn(async () => {
calls.push("session.newSession");
sessionFile = "/tmp/project/sessions/detached.jsonl";
return true;
});
const session = {
newSession,
switchSession: vi.fn(async () => true),
};
const ctx = {
editorContainer,
editor: {},
ui: {
setFocus: vi.fn(),
requestRender: vi.fn(() => {
calls.push("ui.requestRender");
}),
terminal: { columns: 120 },
},
session,
get viewSession() {
return session;
},
sessionManager: {
getCwd: () => "/tmp/project",
getSessionDir: () => "/tmp/project/sessions",
getSessionFile: () => sessionFile,
},
statusContainer: {
clear: vi.fn(() => {
calls.push("statusContainer.clear");
}),
},
pendingMessagesContainer: {
clear: vi.fn(() => {
calls.push("pendingMessagesContainer.clear");
}),
},
compactionQueuedMessages: [] as unknown[],
streamingComponent: { active: true },
streamingMessage: { active: true },
pendingTools: {
clear: vi.fn(() => {
calls.push("pendingTools.clear");
}),
},
loadingAnimation: {
stop: vi.fn(() => {
calls.push("loadingAnimation.stop");
}),
},
statusLine: {
invalidate: vi.fn(() => {
calls.push("statusLine.invalidate");
}),
setSessionStartTime: vi.fn(() => {
calls.push("statusLine.setSessionStartTime");
}),
},
updateEditorTopBorder: vi.fn(() => {
calls.push("updateEditorTopBorder");
}),
updateEditorBorderColor: vi.fn(() => {
calls.push("updateEditorBorderColor");
}),
renderInitialMessages: vi.fn(() => {
calls.push("renderInitialMessages");
}),
reloadTodos: vi.fn(async () => {
calls.push("reloadTodos");
}),
showStatus: vi.fn((message: string) => {
calls.push(`showStatus:${message}`);
}),
showError: vi.fn(),
showHookConfirm,
shutdown: vi.fn(async () => undefined),
clearTransientSessionUi() {
ctx.loadingAnimation?.stop();
ctx.statusContainer.clear();
ctx.pendingMessagesContainer.clear();
ctx.pendingTools.clear();
},
} as unknown as TestContext;
return {
ctx,
calls,
setCurrentSessionFile(path: string) {
sessionFile = path;
},
showHookConfirm,
newSession,
};
}
function renderText(selector: SessionSelectorComponent): string {
return selector.render(120).join("\n");
}
beforeAll(() => {
initTheme();
});
describe("SelectorController session deletion", () => {
beforeEach(() => {
vi.spyOn(SessionManager, "list").mockResolvedValue([]);
vi.spyOn(SessionManager, "listAll").mockResolvedValue([]);
});
afterEach(() => {
vi.restoreAllMocks();
});
it("detaches the active session before selector deletion removes it", async () => {
const activeSession = makeSessionInfo("/tmp/project/sessions/active.jsonl");
const { ctx, calls } = createContext(activeSession.path);
vi.spyOn(SessionManager, "list").mockResolvedValue([activeSession]);
const deleteSessionWithArtifacts = vi
.spyOn(FileSessionStorage.prototype, "deleteSessionWithArtifacts")
.mockImplementation(async sessionPath => {
calls.push(`delete:${sessionPath}`);
});
const controller = new SelectorController(ctx);
await controller.showSessionSelector();
const selector = ctx.editorContainer.children[0];
if (!(selector instanceof SessionSelectorComponent)) {
throw new Error("Expected session selector component");
}
const sessionList = selector.getSessionList() as unknown as {
onDeleteRequest?: (session: SessionInfo) => void;
};
sessionList.onDeleteRequest?.(activeSession);
selector.handleInput("\n");
await Bun.sleep(0);
expect(deleteSessionWithArtifacts).toHaveBeenCalledWith(activeSession.path);
expect(calls).toEqual([
"editorContainer.clear",
"editorContainer.addChild",
"ui.requestRender",
"session.newSession",
"loadingAnimation.stop",
"statusContainer.clear",
"pendingMessagesContainer.clear",
"pendingTools.clear",
"statusLine.invalidate",
"statusLine.setSessionStartTime",
"updateEditorTopBorder",
"updateEditorBorderColor",
"renderInitialMessages",
"reloadTodos",
"ui.requestRender",
`delete:${activeSession.path}`,
"ui.requestRender",
]);
expect(ctx.sessionManager.getSessionFile()).toBe("/tmp/project/sessions/detached.jsonl");
});
it("shows inline selector errors when session deletion fails after detach", async () => {
const activeSession = makeSessionInfo("/tmp/project/sessions/active.jsonl");
const { ctx, newSession } = createContext(activeSession.path);
vi.spyOn(SessionManager, "list").mockResolvedValue([activeSession]);
const deleteSessionWithArtifacts = vi
.spyOn(FileSessionStorage.prototype, "deleteSessionWithArtifacts")
.mockRejectedValue(new Error("disk failed"));
const controller = new SelectorController(ctx);
await controller.showSessionSelector();
const selector = ctx.editorContainer.children[0];
if (!(selector instanceof SessionSelectorComponent)) {
throw new Error("Expected session selector component");
}
const sessionList = selector.getSessionList() as unknown as {
onDeleteRequest?: (session: SessionInfo) => void;
};
sessionList.onDeleteRequest?.(activeSession);
selector.handleInput("\n");
await Bun.sleep(0);
expect(newSession).toHaveBeenCalledTimes(1);
expect(deleteSessionWithArtifacts).toHaveBeenCalledWith(activeSession.path);
expect(ctx.showError).not.toHaveBeenCalled();
expect(ctx.sessionManager.getSessionFile()).toBe("/tmp/project/sessions/detached.jsonl");
expect(renderText(selector)).toContain("Error: Failed to delete session: disk failed");
});
it("creates a fresh session before deleting via slash command and then shows the selector", async () => {
const activeSessionPath = "/tmp/project/sessions/active.jsonl";
const { ctx, calls, showHookConfirm, newSession } = createContext(activeSessionPath);
const deleteSessionWithArtifacts = vi
.spyOn(FileSessionStorage.prototype, "deleteSessionWithArtifacts")
.mockImplementation(async sessionPath => {
calls.push(`delete:${sessionPath}`);
});
const exists = vi.spyOn(FileSessionStorage.prototype, "exists").mockResolvedValue(true);
const controller = new SelectorController(ctx);
await controller.handleSessionDeleteCommand();
expect(exists).toHaveBeenCalledWith(activeSessionPath);
expect(showHookConfirm).toHaveBeenCalledWith(
"Delete Session",
"This will permanently delete the current session.\nYou will be returned to the session selector.",
);
expect(newSession).toHaveBeenCalledTimes(1);
expect(deleteSessionWithArtifacts).toHaveBeenCalledWith(activeSessionPath);
expect(calls).toEqual([
"session.newSession",
"loadingAnimation.stop",
"statusContainer.clear",
"pendingMessagesContainer.clear",
"pendingTools.clear",
"statusLine.invalidate",
"statusLine.setSessionStartTime",
"updateEditorTopBorder",
"updateEditorBorderColor",
"renderInitialMessages",
"reloadTodos",
"ui.requestRender",
`delete:${activeSessionPath}`,
"showStatus:Session deleted",
"editorContainer.clear",
"editorContainer.addChild",
"ui.requestRender",
]);
});
});
@@ -76,18 +76,25 @@ describe("extractLastCommand", () => {
expect(extractLastCommand(messages)).toEqual({ kind: "bash", code: "echo b", language: "bash" });
});
it("joins eval cell code and reports the cell language", () => {
it("extracts eval code from flat args and reports the language", () => {
const py = [
assistantCalls([{ name: "eval", arguments: { language: "py", code: "print(1)" } }]),
] as unknown as AgentMessage[];
expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)", language: "python" });
const js = [
assistantCalls([{ name: "eval", arguments: { language: "js", code: "log(1)" } }]),
] as unknown as AgentMessage[];
expect(extractLastCommand(js)?.language).toBe("javascript");
});
it("still joins legacy multi-cell eval args from older transcripts", () => {
const py = [
assistantCalls([
{ name: "eval", arguments: { cells: [{ language: "py", code: "print(1)" }, { code: "print(2)" }] } },
]),
] as unknown as AgentMessage[];
expect(extractLastCommand(py)).toEqual({ kind: "eval", code: "print(1)\n\nprint(2)", language: "python" });
const js = [
assistantCalls([{ name: "eval", arguments: { cells: [{ language: "js", code: "log(1)" }] } }]),
] as unknown as AgentMessage[];
expect(extractLastCommand(js)?.language).toBe("javascript");
});
});
@@ -242,7 +242,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => {
await Settings.init({ inMemory: true, overrides: { "terminal.showImages": true } });
setTerminalImageProtocol(ImageProtocol.Sixel);
const transcript = transcriptWith([
assistantToolCall("eval-image", "eval", { cells: [{ language: "py", code: "display(image)" }] }),
assistantToolCall("eval-image", "eval", { language: "py", code: "display(image)" }),
{
role: "toolResult",
toolCallId: "eval-image",
@@ -279,9 +279,7 @@ describe("UiHelpers.renderInitialMessages — image replay", () => {
isError: false,
timestamp: 2,
});
session.appendMessage(
assistantToolCall("eval-reopened", "eval", { cells: [{ language: "py", code: "display(image)" }] }),
);
session.appendMessage(assistantToolCall("eval-reopened", "eval", { language: "py", code: "display(image)" }));
session.appendMessage({
role: "toolResult",
toolCallId: "eval-reopened",
+64
View File
@@ -347,6 +347,70 @@ describe("AgentSession shake", () => {
expect(fullStart).toBeDefined();
});
it("counts pre-shake prune savings when deciding whether to fall back to context-full", async () => {
session.settings.set("compaction.strategy", "shake");
session.settings.set("compaction.thresholdTokens", 76384);
session.settings.set("compaction.thresholdPercent", -1);
session.settings.set("compaction.dropUseless", true);
session.settings.set("contextPromotion.enabled", false);
const now = Date.now();
sessionManager.appendMessage({
role: "user",
content: "Investigate every module of the project.",
timestamp: now - 200,
});
const bigCallId = "call-big-useless-for-shake";
sessionManager.appendMessage({
role: "assistant",
content: [{ type: "toolCall", id: bigCallId, name: "search", arguments: { pattern: "TODO" } }],
...apiInfo,
stopReason: "toolUse",
usage,
timestamp: now - 180,
});
sessionManager.appendMessage({
role: "toolResult",
toolCallId: bigCallId,
toolName: "search",
content: [{ type: "text", text: "match line\n".repeat(20000) }],
isError: false,
useless: true,
timestamp: now - 170,
});
session.agent.replaceMessages(session.buildDisplaySessionContext().messages);
const shakeSpy = vi
.spyOn(session, "shake")
.mockResolvedValue({ mode: "elide", toolResultsDropped: 1, blocksDropped: 0, tokensFreed: 100 });
const assistantMessage: AssistantMessage = {
role: "assistant",
content: [{ type: "text", text: "trigger" }],
...apiInfo,
stopReason: "stop",
usage: {
input: 5000,
output: 1000,
cacheRead: 85000,
cacheWrite: 0,
totalTokens: 91000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: now,
};
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
await Bun.sleep(50);
expect(shakeSpy).toHaveBeenCalledTimes(1);
const fullStart = events.find(
event => event.type === "auto_compaction_start" && (event as { action?: string }).action === "context-full",
);
expect(fullStart).toBeUndefined();
});
it("falls back after pre-prompt shake when the floored stored conversation remains over threshold", async () => {
session.settings.set("compaction.strategy", "shake");
session.settings.set("compaction.thresholdTokens", 8_000);
@@ -435,7 +435,9 @@ describe("streaming tool call preview height (bounded across renderers)", () =>
const hidden = total - window;
const longLines = Array.from({ length: total }, (_, i) => `line-${i}`);
const { lines, text } = renderPending("eval", {
cells: [{ language: "js", title: "big", code: longLines.map(line => `const ${line} = 1;`).join("\n") }],
language: "js",
title: "big",
code: longLines.map(line => `const ${line} = 1;`).join("\n"),
});
expect(lines.length, "eval code preview should stay bounded").toBeLessThan(window + 10);

Some files were not shown because too many files have changed in this diff Show More