Merge branch 'main' of github.com:can1357/oh-my-pi into fix-mcp-oauth-resource

This commit is contained in:
Clark Tomlinson
2026-06-10 16:03:48 -04:00
283 changed files with 36774 additions and 3842 deletions
Generated
+1
View File
@@ -2436,6 +2436,7 @@ dependencies = [
"pi-ast",
"pi-iso",
"pi-shell",
"png",
"portable-pty",
"rayon",
"regex",
+1
View File
@@ -274,6 +274,7 @@ image = { version = "0.25", default-features = false, features = [
"gif",
"webp",
] }
png = "0.18"
inferno = { version = "0.12", default-features = false }
syntect = { version = "5.3", default-features = false, features = [
"default-syntaxes",
+4
View File
@@ -482,12 +482,16 @@ For architecture and contribution guidelines, see [packages/coding-agent/DEVELOP
| Package | Description |
| --------------------------------------------------------- | -------------------------------------------------------------------------- |
| **[@oh-my-pi/pi-ai](packages/ai)** | Multi-provider LLM client with streaming and model/provider integration |
| **[@oh-my-pi/pi-catalog](packages/catalog)** | Model catalog: bundled model database, provider descriptors, and identity |
| **[@oh-my-pi/pi-agent-core](packages/agent)** | Agent runtime with tool calling and state management |
| **[@oh-my-pi/pi-coding-agent](packages/coding-agent)** | Interactive coding agent CLI and SDK |
| **[@oh-my-pi/pi-tui](packages/tui)** | Terminal UI library with differential rendering |
| **[@oh-my-pi/pi-natives](packages/natives)** | N-API bindings for grep, shell, image, text, syntax highlighting, and more |
| **[@oh-my-pi/omp-stats](packages/stats)** | Local observability dashboard for AI usage statistics |
| **[@oh-my-pi/pi-utils](packages/utils)** | Shared utilities (logging, streams, dirs/env/process helpers) |
| **[@oh-my-pi/hashline](packages/hashline)** | Line-anchored patch language and applier behind the `edit` tool |
| **[@oh-my-pi/pi-mnemopi](packages/mnemopi)** | Local SQLite memory engine for Oh My Pi agents |
| **[@oh-my-pi/pi-snapcompact](packages/snapcompact)** | SQuAD eval suite for snapcompact bitmap-frame context compression |
| **[@oh-my-pi/swarm-extension](packages/swarm-extension)** | Swarm orchestration extension package |
### Rust Crates
+17
View File
@@ -21,6 +21,7 @@
"@oh-my-pi/pi-catalog": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"@oh-my-pi/snapcompact": "catalog:",
"@opentelemetry/api": "catalog:",
},
"devDependencies": {
@@ -76,6 +77,7 @@
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-tui": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"@oh-my-pi/snapcompact": "catalog:",
"@opentelemetry/api": "catalog:",
"@opentelemetry/context-async-hooks": "catalog:",
"@opentelemetry/exporter-trace-otlp-proto": "catalog:",
@@ -141,6 +143,18 @@
"@types/bun": "catalog:",
},
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "15.10.12",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
},
"devDependencies": {
"@types/bun": "catalog:",
},
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "15.10.12",
@@ -276,6 +290,7 @@
"@oh-my-pi/pi-natives": "15.10.12",
"@oh-my-pi/pi-tui": "15.10.12",
"@oh-my-pi/pi-utils": "15.10.12",
"@oh-my-pi/snapcompact": "15.10.12",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -672,6 +687,8 @@
"@oh-my-pi/pi-utils": ["@oh-my-pi/pi-utils@workspace:packages/utils"],
"@oh-my-pi/snapcompact": ["@oh-my-pi/snapcompact@workspace:packages/snapcompact"],
"@oh-my-pi/swarm-extension": ["@oh-my-pi/swarm-extension@workspace:packages/swarm-extension"],
"@oh-my-pi/typescript-edit-benchmark": ["@oh-my-pi/typescript-edit-benchmark@workspace:packages/typescript-edit-benchmark"],
+1
View File
@@ -36,6 +36,7 @@ pi-ast.workspace = true
pi-iso.workspace = true
pi-shell.workspace = true
portable-pty.workspace = true
png.workspace = true
rayon.workspace = true
regex.workspace = true
serde.workspace = true
File diff suppressed because it is too large Load Diff
+255
View File
@@ -0,0 +1,255 @@
00001:E080EA2AEE0A0A00
00002:E080EA2AE40A0A00
00003:E080CA8AE40A0A00
00004:E080CE84E4040400
00005:E080CE8AEA0E0400
00006:E0A0EAAAAC0A0A00
00007:C0A0C8A8C8080E00
00008:C0A0CEA8CE020E00
00009:A0A0EEA4A4040400
0000A:80808E88EC080800
0000B:A0A0AEA444040400
0000C:E080CE888C080800
0000D:E0808E8AEE0C0A00
0000E:E080EE2AEA0A0E00
0000F:E080EE24E4040E00
00010:C0A0A8A8C8080E00
00011:C0A0A4ACC4040E00
00012:C0A0AEA2CE080E00
00013:C0A0AEA2C6020E00
00014:C0A0AAAACE020200
00015:E0A0AAAAAC0A0A00
00016:E080EA2AEE040400
00017:E080CC8AEC0A0C00
00018:E0808E8AEA0A0A00
00019:E080CA8EEA0A0A00
0001A:3C66663018001800
0001B:E080CE88E8080E00
0001C:E080CE888E020E00
0001D:E080AEA8EE020E00
0001E:E0A0EEC8AE020E00
0001F:A0A0AEA8EE020E00
00020:0000000000000000
00021:1818181818001800
00022:6666660000000000
00023:6C6CFE6CFE6C6C00
00024:183E603C067C1800
00025:00C6CC183066C600
00026:386C3876DCCC7600
00027:1818300000000000
00028:0C18303030180C00
00029:30180C0C0C183000
0002A:00663CFF3C660000
0002B:0018187E18180000
0002C:0000000000181830
0002D:0000007E00000000
0002E:0000000000181800
0002F:03060C183060C000
00030:3C666E7666663C00
00031:1838181818187E00
00032:3C660C1830607E00
00033:3C66061C06663C00
00034:1C3C6CCCFE0C0C00
00035:7E607C0606663C00
00036:1C30607C66663C00
00037:7E06060C18181800
00038:3C66663C66663C00
00039:3C66663E060C3800
0003A:0018180000181800
0003B:0018180000181830
0003C:0C18306030180C00
0003D:00007E007E000000
0003E:6030180C18306000
0003F:3C66060C18001800
00040:7CC6DEDEDEC07C00
00041:183C66667E666600
00042:7C66667C66667C00
00043:3C66606060663C00
00044:786C6666666C7800
00045:7E60607C60607E00
00046:7E60607C60606000
00047:3C66606E66663E00
00048:6666667E66666600
00049:7E18181818187E00
0004A:0606060606663C00
0004B:C6CCD8F0D8CCC600
0004C:6060606060607E00
0004D:C6EEFED6C6C6C600
0004E:C6E6F6DECEC6C600
0004F:3C66666666663C00
00050:7C66667C60606000
00051:3C666666666C3600
00052:7C66667C6C666600
00053:3C66603C06663C00
00054:7E18181818181800
00055:6666666666663C00
00056:66666666663C1800
00057:C6C6C6D6FEEEC600
00058:C3663C183C66C300
00059:C3663C1818181800
0005A:7E060C1830607E00
0005B:3C30303030303C00
0005C:C06030180C060300
0005D:3C0C0C0C0C0C3C00
0005E:10386CC600000000
0005F:00000000000000FF
00060:180C060000000000
00061:00003C063E663E00
00062:60607C6666667C00
00063:00003C6060603C00
00064:06063E6666663E00
00065:00003C667E603C00
00066:1C307C3030303000
00067:00003E66663E067C
00068:60607C6666666600
00069:1800381818181E00
0006A:0C000C0C0C0C0C78
0006B:6060666C786C6600
0006C:3818181818181E00
0006D:0000CCFED6D6C600
0006E:00007C6666666600
0006F:00003C6666663C00
00070:00007C66667C6060
00071:00003E66663E0606
00072:00007C6660606000
00073:00003E603C067C00
00074:30307E3030301E00
00075:0000666666663E00
00076:00006666663C1800
00077:0000C6C6D67C6C00
00078:0000C66C386CC600
00079:00006666663E063C
0007A:00007E0C18307E00
0007B:0E18187018180E00
0007C:1818181818181800
0007D:7018180E18187000
0007E:76DC000000000000
0007F:C0A0AEA4C4040400
00080:E0A0EEAAEA0A0E00
00081:E0A0E4ACE4040E00
00082:E0A0EEA2EE080E00
00083:E0A0EEA2EE020E00
00084:E0404E4AEA0A0A00
00085:E0A0A8A8A8080E00
00086:E080EE28EE020E00
00087:E080CE88EE020E00
00088:A0A0EEA8AE020E00
00089:A0A0E2A2A20A0E00
0008A:A0A0AEA84E020E00
0008B:E0A0EC8A8A0A0C00
0008C:E0A0EA8A8A0A0E00
0008D:C0A0CEA4A4040E00
0008E:E080EE22EE080E00
0008F:E080EE22E6020E00
00090:C0A0AEA8C8080E00
00091:E0A0E48C84040E00
00092:E0A0EE828E080E00
00093:E080EE28EC080E00
00094:E0808E88E8080E00
00095:A0E0EAAAAE0E0A00
00096:E080EE2AEE080800
00097:E080CE8AEE080800
00098:E0A0EE2AEE0A0E00
00099:E0A0EE2AEE020E00
0009A:E0A0EE2AEE0A0A00
0009B:E0808E88EE020E00
0009C:E080EE24E4040400
0009D:E0A0AEA8EE020E00
0009E:E0A0EA8E8E0A0A00
0009F:E0A0EEAAAE080800
000A0:0000000000000000
000A1:1800181818181800
000A2:00187ED8D87E1800
000A3:386C60F06066FC00
000A4:663C663C66000000
000A5:C3663C183C181800
000A6:1818180018181800
000A7:3C603C66663C063C
000A8:6600000000000000
000A9:7E819DB1B19D817E
000AA:3C6C6C3E007E0000
000AB:003366CC66330000
000AC:007E060600000000
000AD:0000003C00000000
000AE:7E81B9A5B9A5817E
000AF:7E00000000000000
000B0:3C663C0000000000
000B1:18187E1818007E00
000B2:7018306078000000
000B3:780C180C78000000
000B4:0C18300000000000
000B5:00006666667C60C0
000B6:3E7A7A3A1A1A1A00
000B7:0000001800000000
000B8:0000000000000818
000B9:3070303030000000
000BA:386C6C38007C0000
000BB:00CC663366CC0000
000BC:40C64C5A366ACF02
000BD:40C64C5E3366CC0F
000BE:C0266C3AF66ACF02
000BF:1800183060663C00
000C0:70003C667E666600
000C1:0E003C667E666600
000C2:1866003C667E6600
000C3:76DC003C667E6600
000C4:66003C667E666600
000C5:1818003C667E6600
000C6:3F6CCCFECCCCCF00
000C7:3C66606060663C18
000C8:7000FEC0F8C0FE00
000C9:0E00FEC0F8C0FE00
000CA:186600FEF0C0FE00
000CB:6600FEC0F8C0FE00
000CC:70007E1818187E00
000CD:0E007E1818187E00
000CE:1866007E18187E00
000CF:66007E1818187E00
000D0:786C66F6666C7800
000D1:76DC00C6F6DEC600
000D2:70007CC6C6C67C00
000D3:0E007CC6C6C67C00
000D4:1866007CC6C67C00
000D5:76DC007CC6C67C00
000D6:66007CC6C6C67C00
000D7:00C66C386CC60000
000D8:3E666E7E76667C00
000D9:7000C6C6C6C67C00
000DA:0E00C6C6C6C67C00
000DB:186600C6C6C67C00
000DC:6600C6C6C6C67C00
000DD:0E0066663C181800
000DE:C0C0FCC6FCC0C000
000DF:3C66666C66666C00
000E0:70003C063E663E00
000E1:0E003C063E663E00
000E2:1866003E66C67E00
000E3:76DC003E66C67E00
000E4:66003C063E663E00
000E5:1818003E66C67E00
000E6:00007E1B7FD87700
000E7:00003C6060603C18
000E8:70003C667E603C00
000E9:0E003C667E603C00
000EA:1866003C7E603C00
000EB:66003C667E603C00
000EC:7000381818183C00
000ED:0E00381818183C00
000EE:1866003818183C00
000EF:6600381818183C00
000F0:0C3E0C7CCCCC7800
000F1:76DC007C66666600
000F2:70003C6666663C00
000F3:0E003C6666663C00
000F4:1866003C66663C00
000F5:76DC003C66663C00
000F6:66003C6666663C00
000F7:1818007E00181800
000F8:00027CCED6E67C80
000F9:7000666666663E00
000FA:0E00666666663E00
000FB:1866006666663E00
000FC:6600666666663E00
000FD:0E006666663E063C
000FE:60607C66667C6060
000FF:66006666663E063C
+1
View File
@@ -36,6 +36,7 @@ pub mod highlight;
pub mod html;
pub mod keys;
pub mod sixel;
pub mod snapcompact;
pub use pi_ast::language;
pub mod power;
+595
View File
@@ -0,0 +1,595 @@
//! Snapcompact frame rendering.
//!
//! Rasterizes pre-normalized conversation text onto a square bitmap using one
//! of the bundled public-domain pixel fonts, then encodes it as PNG:
//!
//! - `5x8` — X.org BDF font (legacy shape).
//! - `8x8` — unscii-8 hex font (Latin-1 subset), the square cell that won the
//! snapcompact `SQuAD` evals.
//!
//! Shape controls, all eval-validated in `packages/snapcompact`:
//!
//! - **variant** — `sent` cycles glyph ink through six hues at sentence
//! boundaries; `bw` prints plain black ink (best for Anthropic readers).
//! - **lineRepeat** — prints every text line N times; copies after the first
//! sit on a pale highlight band. Redundancy coding: two looks per glyph at
//! half the density ("8x8r" shapes).
//! - **cellWidth/cellHeight** — target cell size. When it differs from the
//! font's natural cell, glyphs are rasterized at native size and the canvas
//! is Lanczos3-resampled to the target (anisotropic stretch, e.g. the
//! OpenAI-optimal "6x6u" shape), producing an anti-aliased RGB frame.
//!
//! Text normalization, frame chunking, provider shape selection, and archive
//! management live in `packages/agent/src/compaction/snapcompact.ts`; this
//! module is only the hot `text -> PNG bytes` path.
use std::{borrow::Cow, collections::HashMap, f32::consts::PI, sync::LazyLock};
use napi::bindgen_prelude::*;
use napi_derive::napi;
/// Upper bound on the frame edge: a hard stop against absurd allocations
/// (`size * size` pixel buffer), far above the 2576px production frame.
const MAX_FRAME_SIZE: u32 = 16384;
/// Indexed palette: 0 is the white background, 1-6 are the six dark sentence
/// hues from the eval renderer (HLS l=0.22 s=0.95, h ∈ {0, .08, .3, .5, .62,
/// .78}), 7 is plain black ink (`bw` variant), 8 is the pale highlight band
/// behind repeated line copies.
const PALETTE: [[u8; 3]; 9] = [
[255, 255, 255],
[109, 2, 2], // red
[109, 53, 2], // amber
[24, 109, 2], // green
[2, 109, 109], // teal
[2, 32, 109], // blue
[75, 2, 109], // violet
[0, 0, 0], // bw ink
[255, 247, 194], // repeat highlight band
];
const INK_COLORS: usize = 6;
const INK_BLACK: u8 = 7;
const BG_REPEAT: u8 = 8;
static FONT_5X8: LazyLock<Font> = LazyLock::new(|| parse_bdf(include_str!("fonts/5x8.bdf"), 5, 8));
static FONT_8X8: LazyLock<Font> = LazyLock::new(|| parse_hex(include_str!("fonts/unscii-8.hex")));
struct Glyph {
/// Glyph width in pixels (≤ 8 for the bundled fonts).
w: u8,
/// Glyph height in pixels.
h: i32,
xoff: i32,
yoff: i32,
/// One bitmask per bitmap row, MSB-leftmost.
rows: Vec<u8>,
}
struct Font {
/// Glyphs keyed by Unicode code point (ASCII + Latin-1 coverage).
glyphs: HashMap<u32, Glyph>,
ascent: i32,
/// Natural cell advance (x) in pixels.
cell_w: usize,
/// Natural cell pitch (y) in pixels.
cell_h: usize,
}
fn parse_bdf(text: &str, cell_w: usize, cell_h: usize) -> Font {
let mut glyphs = HashMap::new();
let mut ascent = 0i32;
let mut enc = -1i64;
let mut bbx = [0i32; 4];
let mut lines = text.lines();
while let Some(line) = lines.next() {
if let Some(rest) = line.strip_prefix("FONT_ASCENT") {
ascent = rest.trim().parse().unwrap_or(0);
} else if let Some(rest) = line.strip_prefix("ENCODING") {
enc = rest.trim().parse().unwrap_or(-1);
} else if let Some(rest) = line.strip_prefix("BBX") {
let mut parts = rest.split_ascii_whitespace();
for slot in &mut bbx {
*slot = parts.next().and_then(|part| part.parse().ok()).unwrap_or(0);
}
} else if line.starts_with("BITMAP") {
let mut rows = Vec::new();
for row in lines.by_ref() {
if row.starts_with("ENDCHAR") {
break;
}
rows.push(u8::from_str_radix(row.trim(), 16).unwrap_or(0));
}
if enc >= 0 {
glyphs.insert(enc as u32, Glyph {
w: bbx[0].clamp(0, 8) as u8,
h: bbx[1],
xoff: bbx[2],
yoff: bbx[3],
rows,
});
}
}
}
Font { glyphs, ascent, cell_w, cell_h }
}
/// Parse a unifont-style `.hex` font (`CODEPOINT:16-hex-digit bitmap`, one
/// byte per row of an 8x8 glyph). Baseline sits at row 7 (`ascent` 7 with a
/// one-pixel descender row), matching the eval renderer.
fn parse_hex(text: &str) -> Font {
let mut glyphs = HashMap::new();
for line in text.lines() {
let Some((cp, bits)) = line.split_once(':') else {
continue;
};
let Ok(enc) = u32::from_str_radix(cp.trim(), 16) else {
continue;
};
let bits = bits.trim();
if bits.len() != 16 {
continue;
}
let rows: Vec<u8> = (0..8)
.map(|i| u8::from_str_radix(&bits[i * 2..i * 2 + 2], 16).unwrap_or(0))
.collect();
glyphs.insert(enc, Glyph { w: 8, h: 8, xoff: 0, yoff: -1, rows });
}
Font { glyphs, ascent: 7, cell_w: 8, cell_h: 8 }
}
fn resolve_font(name: &str) -> Option<&'static Font> {
match name {
"5x8" => Some(&FONT_5X8),
"8x8" => Some(&FONT_8X8),
_ => None,
}
}
/// Frame grid geometry shared with the TypeScript caller.
struct Grid {
cols: usize,
rows: usize,
repeat: usize,
}
/// Rasterize `text` onto a `width` x `height` palette-indexed bitmap at the
/// font's natural cell size, row-major with no word wrap. Each text line is
/// printed `grid.repeat` times; copies after the first sit on the highlight
/// band. Ink cycles through six hues at sentence boundaries (terminator in
/// `.!?` followed by a space) unless `black_ink` pins it to black. Characters
/// beyond `cols * rows` are ignored; code points missing from the font leave
/// their cell blank.
fn render_bitmap(
text: &str,
width: usize,
height: usize,
font: &Font,
grid: &Grid,
black_ink: bool,
) -> Vec<u8> {
let mut pixels = vec![0u8; width * height]; // 0 = white background
let capacity = grid.cols * grid.rows;
if capacity == 0 {
return pixels;
}
if grid.repeat > 1 {
for row in 0..grid.rows {
for copy in 1..grid.repeat {
let band_top = (row * grid.repeat + copy) * font.cell_h;
for y in band_top..(band_top + font.cell_h).min(height) {
pixels[y * width..y * width + width].fill(BG_REPEAT);
}
}
}
}
let codes: Vec<u32> = text.chars().map(|ch| ch as u32).collect();
let count = codes.len().min(capacity);
let mut sentence = 0usize;
for i in 0..count {
let code = codes[i];
let ink = if black_ink {
INK_BLACK
} else {
(1 + sentence % INK_COLORS) as u8
};
if matches!(code, 0x2e | 0x21 | 0x3f) && codes.get(i + 1) == Some(&0x20) {
sentence += 1;
}
let Some(glyph) = font.glyphs.get(&code) else {
continue;
};
if glyph.rows.is_empty() {
continue;
}
let row = i / grid.cols;
let col = i - row * grid.cols;
let left = (col * font.cell_w) as i32 + glyph.xoff;
for copy in 0..grid.repeat {
let cell_top = ((row * grid.repeat + copy) * font.cell_h) as i32;
let top = cell_top + font.ascent - glyph.h - glyph.yoff;
for (r, &bits) in glyph.rows.iter().enumerate() {
if bits == 0 {
continue;
}
let y = top + r as i32;
if y < 0 || y >= height as i32 {
continue;
}
let row_base = y as usize * width;
for b in 0..glyph.w {
if bits & (0x80u8 >> b) != 0 {
let x = left + i32::from(b);
if x >= 0 && (x as usize) < width {
pixels[row_base + x as usize] = ink;
}
}
}
}
}
}
pixels
}
// ============================================================================
// Lanczos3 resampling (stretch shapes)
// ============================================================================
fn lanczos3(x: f32) -> f32 {
let x = x.abs();
if x < 1e-6 {
return 1.0;
}
if x >= 3.0 {
return 0.0;
}
let pix = PI * x;
(pix.sin() / pix) * ((pix / 3.0).sin() / (pix / 3.0))
}
/// Per-output-pixel kernel contributions for one axis, PIL-convention
/// (`center = (i + 0.5) * scale`, kernel stretched by `max(scale, 1)`,
/// weights normalized).
fn contributions(src_len: usize, dst_len: usize) -> Vec<(usize, Vec<f32>)> {
let scale = src_len as f32 / dst_len as f32;
let filt_scale = scale.max(1.0);
let support = 3.0 * filt_scale;
let mut out = Vec::with_capacity(dst_len);
for i in 0..dst_len {
let center = (i as f32 + 0.5) * scale;
let begin = ((center - support) as isize).max(0) as usize;
let end = ((center + support).ceil() as usize).min(src_len);
let mut weights = Vec::with_capacity(end - begin);
let mut total = 0.0f32;
for x in begin..end {
let w = lanczos3((x as f32 + 0.5 - center) / filt_scale);
weights.push(w);
total += w;
}
if total != 0.0 {
for w in &mut weights {
*w /= total;
}
}
out.push((begin, weights));
}
out
}
/// Separable Lanczos3 resize of an interleaved RGB f32 buffer.
fn resize_rgb(src: &[f32], sw: usize, sh: usize, dw: usize, dh: usize) -> Vec<f32> {
let horiz = contributions(sw, dw);
let mut tmp = vec![0f32; dw * sh * 3];
for y in 0..sh {
let src_row = &src[y * sw * 3..(y + 1) * sw * 3];
let dst_row = &mut tmp[y * dw * 3..(y + 1) * dw * 3];
for (x, (begin, weights)) in horiz.iter().enumerate() {
let mut acc = [0f32; 3];
for (k, &w) in weights.iter().enumerate() {
let s = (begin + k) * 3;
acc[0] = src_row[s].mul_add(w, acc[0]);
acc[1] = src_row[s + 1].mul_add(w, acc[1]);
acc[2] = src_row[s + 2].mul_add(w, acc[2]);
}
dst_row[x * 3..x * 3 + 3].copy_from_slice(&acc);
}
}
let vert = contributions(sh, dh);
let mut out = vec![0f32; dw * dh * 3];
for (y, (begin, weights)) in vert.iter().enumerate() {
let dst_row = &mut out[y * dw * 3..(y + 1) * dw * 3];
for (k, &w) in weights.iter().enumerate() {
let src_row = &tmp[(begin + k) * dw * 3..(begin + k + 1) * dw * 3];
for (d, &s) in dst_row.iter_mut().zip(src_row) {
*d = s.mul_add(w, *d);
}
}
}
out
}
// ============================================================================
// PNG encoding
// ============================================================================
/// Pack one-byte-per-pixel palette indices into 4-bit PNG scanline data
/// (two pixels per byte, high nibble first). With only 9 palette entries,
/// 4-bit depth halves the pre-deflate stream vs 8-bit.
fn pack_nibbles(pixels: &[u8], size: usize) -> Vec<u8> {
let row_bytes = size.div_ceil(2);
let mut packed = vec![0u8; row_bytes * size];
for y in 0..size {
let src = &pixels[y * size..(y + 1) * size];
let dst = &mut packed[y * row_bytes..(y + 1) * row_bytes];
for (x, &px) in src.iter().enumerate() {
dst[x / 2] |= px << (4 * (1 - x % 2));
}
}
packed
}
/// Encode a palette-indexed bitmap as a 4-bit indexed PNG with `None` row
/// filtering (the glyph bitmap is already minimal-entropy; filtering costs
/// encode time without helping deflate).
fn encode_indexed_png(
pixels: &[u8],
size: usize,
compression: png::Compression,
) -> Result<Vec<u8>> {
let mut palette = Vec::with_capacity(PALETTE.len() * 3);
for rgb in PALETTE {
palette.extend_from_slice(&rgb);
}
let mut out = Vec::new();
let mut encoder = png::Encoder::new(&mut out, size as u32, size as u32);
encoder.set_color(png::ColorType::Indexed);
encoder.set_depth(png::BitDepth::Four);
encoder.set_palette(Cow::Owned(palette));
encoder.set_compression(compression);
// MUST come after `set_compression`, which resets the filter to the
// compression level's default (`Adaptive` for `Balanced`).
encoder.set_filter(png::Filter::NoFilter);
let mut writer = encoder
.write_header()
.map_err(|err| Error::from_reason(format!("Failed to write PNG header: {err}")))?;
writer
.write_image_data(&pack_nibbles(pixels, size))
.map_err(|err| Error::from_reason(format!("Failed to write PNG data: {err}")))?;
writer
.finish()
.map_err(|err| Error::from_reason(format!("Failed to finish PNG stream: {err}")))?;
Ok(out)
}
/// Encode an interleaved RGB8 buffer as PNG. Stretched frames are
/// continuous-tone, so adaptive filtering (the `Balanced` default) helps.
fn encode_rgb_png(pixels: &[u8], size: usize, compression: png::Compression) -> Result<Vec<u8>> {
let mut out = Vec::new();
let mut encoder = png::Encoder::new(&mut out, size as u32, size as u32);
encoder.set_color(png::ColorType::Rgb);
encoder.set_depth(png::BitDepth::Eight);
encoder.set_compression(compression);
let mut writer = encoder
.write_header()
.map_err(|err| Error::from_reason(format!("Failed to write PNG header: {err}")))?;
writer
.write_image_data(pixels)
.map_err(|err| Error::from_reason(format!("Failed to write PNG data: {err}")))?;
writer
.finish()
.map_err(|err| Error::from_reason(format!("Failed to finish PNG stream: {err}")))?;
Ok(out)
}
// ============================================================================
// Entry point
// ============================================================================
/// Shape options for one snapcompact frame.
#[napi(object)]
#[derive(Default)]
pub struct SnapcompactRenderOptions {
/// Frame edge in pixels.
pub size: u32,
/// Bundled font: `"5x8"` (X.org BDF) or `"8x8"` (unscii-8). Default `"5x8"`.
pub font: Option<String>,
/// Target cell advance in pixels. Differing from the font's natural cell
/// triggers the Lanczos stretch path. Default: font natural width.
pub cell_width: Option<u32>,
/// Target cell pitch in pixels. Default: font natural height.
pub cell_height: Option<u32>,
/// Ink variant: `"sent"` (six-hue sentence cycling) or `"bw"` (black).
/// Default `"sent"`.
pub variant: Option<String>,
/// Print each text line this many times; copies after the first sit on a
/// pale highlight band. Default 1.
pub line_repeat: Option<u32>,
}
/// Render one snapcompact frame: print pre-normalized text onto a square
/// bitmap and encode it as PNG.
///
/// The glyph grid holds `floor(size/cellWidth) *
/// floor(size/cellHeight/lineRepeat)` characters; input beyond that is ignored
/// (the caller chunks text to capacity). Native-cell shapes encode as 4-bit
/// indexed PNG; stretched shapes (target cell != font cell) encode as RGB.
/// Returns the PNG bytes.
#[napi]
pub fn render_snapcompact_png(
text: String,
options: SnapcompactRenderOptions,
) -> Result<Uint8Array> {
let size = options.size;
if size == 0 || size > MAX_FRAME_SIZE {
return Err(Error::from_reason(format!(
"Invalid frame size {size}: expected 1..={MAX_FRAME_SIZE}"
)));
}
let font_name = options.font.as_deref().unwrap_or("5x8");
let font = resolve_font(font_name).ok_or_else(|| {
Error::from_reason(format!(
"Unknown snapcompact font {font_name:?}: expected \"5x8\" or \"8x8\""
))
})?;
let black_ink = match options.variant.as_deref().unwrap_or("sent") {
"sent" => false,
"bw" => true,
other => {
return Err(Error::from_reason(format!(
"Unknown snapcompact variant {other:?}: expected \"sent\" or \"bw\""
)));
},
};
let target_w = options.cell_width.unwrap_or(font.cell_w as u32).max(1) as usize;
let target_h = options.cell_height.unwrap_or(font.cell_h as u32).max(1) as usize;
let repeat = options.line_repeat.unwrap_or(1).max(1) as usize;
let size = size as usize;
let grid = Grid { cols: size / target_w, rows: size / target_h / repeat, repeat };
if grid.cols == 0 || grid.rows == 0 {
return Err(Error::from_reason(format!(
"Frame size {size} cannot fit a {target_w}x{target_h} cell grid (repeat {repeat})"
)));
}
if (target_w, target_h) == (font.cell_w, font.cell_h) {
// Native cell: rasterize straight onto the frame, indexed.
let pixels = render_bitmap(&text, size, size, font, &grid, black_ink);
return Ok(encode_indexed_png(&pixels, size, png::Compression::Balanced)?.into());
}
// Stretch shape: rasterize at the font's natural cell on a tight canvas,
// Lanczos3-resample to the target cell, paste onto the white frame.
let src_w = grid.cols * font.cell_w;
let src_h = grid.rows * grid.repeat * font.cell_h;
let dst_w = grid.cols * target_w;
let dst_h = grid.rows * grid.repeat * target_h;
let indexed = render_bitmap(&text, src_w, src_h, font, &grid, black_ink);
let mut rgb = vec![0f32; src_w * src_h * 3];
for (dst, &idx) in rgb.chunks_exact_mut(3).zip(&indexed) {
let [r, g, b] = PALETTE[idx as usize];
dst[0] = f32::from(r);
dst[1] = f32::from(g);
dst[2] = f32::from(b);
}
let resized = resize_rgb(&rgb, src_w, src_h, dst_w, dst_h);
let mut frame = vec![255u8; size * size * 3];
for y in 0..dst_h.min(size) {
let src_row = &resized[y * dst_w * 3..(y + 1) * dst_w * 3];
let dst_row = &mut frame[y * size * 3..];
for (d, &s) in dst_row[..dst_w.min(size) * 3].iter_mut().zip(src_row) {
*d = s.round().clamp(0.0, 255.0) as u8;
}
}
Ok(encode_rgb_png(&frame, size, png::Compression::Balanced)?.into())
}
#[cfg(test)]
mod tests {
use super::*;
fn opts(size: u32) -> SnapcompactRenderOptions {
SnapcompactRenderOptions { size, ..Default::default() }
}
#[test]
fn fonts_parse_ascii_coverage() {
for (font, ascent) in [(&*FONT_5X8, 7), (&*FONT_8X8, 7)] {
assert_eq!(font.ascent, ascent);
// Every printable ASCII char must have a glyph.
for cp in 0x20u32..0x7f {
assert!(font.glyphs.contains_key(&cp), "missing glyph for U+{cp:04X}");
}
}
}
#[test]
fn bitmap_inks_sentences_and_caps_capacity() {
// 40px -> 8 cols x 5 rows = 40 cells (5x8 font).
let grid = Grid { cols: 8, rows: 5, repeat: 1 };
let pixels = render_bitmap("Hi. Ok.", 40, 40, &FONT_5X8, &grid, false);
let inks: Vec<u8> = pixels.iter().copied().filter(|&p| p != 0).collect();
assert!(inks.contains(&1), "first sentence should use ink 1");
assert!(inks.contains(&2), "second sentence should use ink 2");
assert!(!inks.contains(&3), "no third sentence ink expected");
// Overflow input renders without panicking and stays in-bounds.
let overflow = render_bitmap(&"x".repeat(100), 40, 40, &FONT_5X8, &grid, false);
assert_eq!(overflow.len(), 40 * 40);
}
#[test]
fn bw_variant_prints_black_only() {
let grid = Grid { cols: 8, rows: 8, repeat: 1 };
let pixels = render_bitmap("Hi. Ok.", 64, 64, &FONT_8X8, &grid, true);
let inks: Vec<u8> = pixels.iter().copied().filter(|&p| p != 0).collect();
assert!(!inks.is_empty());
assert!(inks.iter().all(|&p| p == INK_BLACK), "bw must ink only black");
}
#[test]
fn line_repeat_duplicates_rows_on_highlight_bands() {
// 64px, 8x8 font, repeat 2 -> 8 cols x 4 unique rows.
let grid = Grid { cols: 8, rows: 4, repeat: 2 };
let pixels = render_bitmap("ABCDEFGH", 64, 64, &FONT_8X8, &grid, true);
// Copy band (rows 8..16) carries the highlight background.
assert!(pixels[9 * 64..10 * 64].contains(&BG_REPEAT), "duplicate band must be highlighted");
// Identical glyph ink in both copies: compare full 8-row bands modulo
// background.
for y in 0..8 {
for x in 0..64 {
let a = pixels[y * 64 + x];
let b = pixels[(y + 8) * 64 + x];
assert_eq!(a == INK_BLACK, b == INK_BLACK, "copy ink mismatch at ({x},{y})");
}
}
}
#[test]
fn render_native_is_indexed_and_stretch_is_rgb() {
let native = render_snapcompact_png("Hello world. Again.".into(), SnapcompactRenderOptions {
size: 128,
font: Some("8x8".into()),
variant: Some("bw".into()),
line_repeat: Some(2),
..Default::default()
})
.unwrap();
// PNG color type lives at byte 25 of the IHDR: 3 = indexed.
assert_eq!(native[25], 3);
let stretched =
render_snapcompact_png("Hello world. Again.".into(), SnapcompactRenderOptions {
size: 128,
font: Some("8x8".into()),
cell_width: Some(6),
cell_height: Some(6),
..Default::default()
})
.unwrap();
// 2 = truecolor RGB.
assert_eq!(stretched[25], 2);
// Stretched output must contain anti-aliased (non-extreme) pixels.
let legacy = render_snapcompact_png("Hi. Ok.".into(), opts(40)).unwrap();
assert_eq!(legacy[25], 3, "default shape stays the legacy 5x8 indexed path");
}
#[test]
fn rejects_bad_shapes() {
assert!(render_snapcompact_png("x".into(), opts(0)).is_err());
assert!(
render_snapcompact_png("x".into(), SnapcompactRenderOptions {
size: 64,
font: Some("9x9".into()),
..Default::default()
})
.is_err()
);
assert!(
render_snapcompact_png("x".into(), SnapcompactRenderOptions {
size: 64,
variant: Some("zebra".into()),
..Default::default()
})
.is_err()
);
}
}
+16 -1
View File
@@ -10,6 +10,7 @@ Both are persisted as session entries and converted back into user-context messa
## Key implementation files
- `packages/agent/src/compaction/compaction.ts` (context-full summarization and handoff generation)
- `packages/snapcompact/src/snapcompact.ts` (snapcompact strategy: history archived as dense bitmap images)
- `packages/agent/src/compaction/branch-summarization.ts`
- `packages/agent/src/compaction/pruning.ts`
- `packages/agent/src/compaction/utils.ts`
@@ -126,6 +127,20 @@ The automatic paths are intentionally different:
- Trigger: `runIdleCompaction()` when not streaming or already compacting.
- Uses `reason: "idle"` and does not auto-continue afterward.
### Snapcompact strategy
`compaction.strategy: "snapcompact"` replaces the LLM summarization call with a local, deterministic archival pass (`snapcompactCompact` from `@oh-my-pi/snapcompact`):
- The discarded history is serialized, whitespace-collapsed, and printed onto provider-aware square PNG frames using bundled public-domain pixel fonts. Anthropic-family and unknown APIs use repeated black `8x8` cells, Google uses repeated sentence-colored `8x8` cells, and OpenAI uses dense stretched `6x6` cells with `detail: "original"`.
- Frames persist under `CompactionEntry.preserveData.snapcompact` and are re-attached to the `compactionSummary` message as image blocks on every context rebuild; the entry's `summary` is a deterministic reading guide (grid geometry, role tags, truncation notes) plus the usual file-operation lists.
- Later compactions carry earlier frames forward. Beyond an 8-frame budget the archive fades from the middle out: the earliest frame (session head — the original request, or the filmed summary of older history) is pinned, and the oldest *unpinned* frames are evicted, so head and tail both survive. If the previous compaction was text-based, its summary is printed at the head of the frame archive as `[Summary of earlier history]`.
- No model, API key, or network is involved, so snapcompact is also safe for overflow recovery. It requires a vision-capable current model (`model.input` includes `"image"`); otherwise the run falls back to context-full and emits a warning notice (auto and manual paths). Manual `/compact` honors the strategy unless custom instructions are given (those imply a directed LLM summary).
- Rationale: the shape table comes from the snapcompact 200k-token evals in `packages/snapcompact`, where bitmap frames preserved QA recall at lower billed-token cost than raw text for vision-capable models.
### Display transcript
Compaction no longer visually restarts the conversation. The TUI renders the **display transcript** (`buildSessionContext({ transcript: true })` / `AgentSession.buildTranscriptSessionContext()`): every path entry in chronological order, with each compaction shown inline as a slim divider — `── 📷 compacted · ctrl+o ──` — at the point it fired. Expanding (ctrl+o) reveals the summary. Only the LLM context resets at the compaction boundary; the scrollback above the divider stays intact, including across session resume.
### Pre-compaction pruning
Before compaction checks, tool-result pruning may run (`pruneToolOutputs`).
@@ -373,7 +388,7 @@ Post-navigation event exposing new/old leaf and optional summary entry.
From `settings-schema.ts`:
- `compaction.enabled` = `true`
- `compaction.strategy` = `"context-full"` (`"handoff"` and `"off"` are also supported)
- `compaction.strategy` = `"context-full"` (`"handoff"`, `"shake"`, `"snapcompact"`, and `"off"` are also supported)
- `compaction.reserveTokens` = `16384`
- `compaction.keepRecentTokens` = `20000`
- `compaction.autoContinue` = `true`
@@ -2,6 +2,7 @@
"name": "hello-extension",
"version": "1.0.0",
"description": "Minimal oh-my-pi extension example",
"homepage": "https://omp.sh",
"omp": {
"extensions": ["./index.ts"]
}
@@ -1,6 +1,7 @@
{
"name": "my-plugin",
"version": "0.1.0",
"homepage": "https://omp.sh",
"omp": {
"extensions": ["./index.ts"]
}
@@ -2,6 +2,7 @@
"name": "safety-hook",
"version": "1.0.0",
"description": "oh-my-pi extension example: block rm -rf / via tool_call hook",
"homepage": "https://omp.sh",
"omp": {
"extensions": ["./index.ts"]
}
+1 -1
View File
@@ -172,7 +172,7 @@ delete 20
- `line N: \`insert\` needs at least one \`+TEXT\` body row.`
- `line N: \`replace block N:\` needs at least one \`+TEXT\` body row. To delete a block, use \`delete N..M\` with the block's line range.`
- Unresolvable `replace block N:` (apply / final-preview path only):
- `line N: \`replace block X:\` could not resolve a syntactic block beginning on line X. The language may be unsupported, the line may be blank or a closing delimiter, or the block may not parse. Use \`replace X..M:\` with the block's explicit end line instead.`
- `line N: \`replace block X:\` could not resolve a syntactic block beginning on line X. The language may be unsupported, the line may be blank or a closing delimiter, or the block may not parse. Use \`replace X..M:\` with the block's explicit end line instead.` — followed by a blank line and numbered `*`-marked context rows around line X (same shape as the mismatch preview).
- Delete with body:
- `line N: \`delete N..M\` does not take body rows. Remove the body, or use \`replace N..M:\`.`
- `line N: \`delete block N\` does not take body rows. Remove the body, or use \`replace block N:\` to replace the block.`
+4 -4
View File
@@ -138,7 +138,7 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod
- `await read(path, { offset?, limit? })`
- `await tree(path = ".", { maxDepth?, hidden? })`
- `sort(text, { reverse?, unique? })`, `uniq(text, { count? })`, `counter(items, { limit?, reverse? })`
- `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
- `await agent(prompt, { agentType?, model?, label?, schema? })`
- `await parallel([() => agent("a"), () => agent("b")])`
- `await pipeline(items, stage1, stage2)`
- `display(value)` behavior:
@@ -192,11 +192,11 @@ Both runtimes expose `completion()` — a single stateless completion against a
Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state.
- Signatures:
- JS: `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
- Python: `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)`
- JS: `await agent(prompt, { agentType?, model?, label?, schema? })`
- Python: `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)`
- `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work.
- `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply.
- `context` supplies shared background; `label` controls the `agent://<id>` output label prefix.
- Shared background is passed via files: write a `local://` file and reference it in the prompt. `label` controls the `agent://<id>` output label prefix.
- `schema` passes a JSON Schema to the subagent structured-output path. When present, the helper parses the final JSON text and returns an object.
- Spawn restrictions use `session.getSessionSpawns()` exactly like the `task` tool. Eval-driven subagent recursion is capped at depth 3.
- JS and Python both expose `parallel(thunks)` and `pipeline(items, ...stages)`; both use a bounded async/threaded pool whose width tracks the `task.maxConcurrency` setting (the same ceiling the `task` tool uses; `0` = run every item at once), preserve item order, and propagate rejections. The width is fetched live from the host via the `__concurrency__` bridge, so the helpers no longer take a `concurrency` argument.
+64 -85
View File
@@ -1,120 +1,99 @@
# irc
> Send short prose messages to other live agents in the current process.
> Send and receive messages between agents over a process-global mailbox bus.
## Source
- Entry: `packages/coding-agent/src/tools/irc.ts`
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/irc.md`
- Key collaborators:
- `packages/coding-agent/src/registry/agent-registry.ts` — process-global live agent directory.
- `packages/coding-agent/src/session/agent-session.ts` — side-channel reply generation and history injection.
- `packages/coding-agent/src/prompts/system/irc-incoming.md` — no-tools auto-reply prompt.
- `packages/coding-agent/src/tools/index.ts` — tool availability gating.
- `packages/coding-agent/src/config/settings-schema.ts` — `irc.enabled` default.
- `packages/coding-agent/src/irc/bus.ts` — process-global `IrcBus`: per-agent mailboxes, delivery, waiter matching.
- `packages/coding-agent/src/registry/agent-registry.ts` — process-global agent directory and status.
- `packages/coding-agent/src/registry/agent-lifecycle.ts` — revival of parked recipients on direct send.
- `packages/coding-agent/src/session/agent-session.ts` — `deliverIrcMessage(...)`: recipient-side injection and wake turns.
- `packages/coding-agent/src/prompts/system/irc-incoming.md` — incoming-message rendering for the recipient.
- `packages/coding-agent/src/config/settings-schema.ts` — `irc.enabled`, `irc.timeoutMs`.
- `packages/coding-agent/src/modes/controllers/event-controller.ts` — renders IRC events into chat UI.
- `packages/coding-agent/src/modes/utils/ui-helpers.ts` — formats `[IRC]` transcript lines.
- `packages/coding-agent/src/task/executor.ts` — carries `irc.enabled` into subagents.
## Inputs
### `op: "list"`
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `op` | `"list"` | Yes | Lists peers visible to the caller. |
### `op: "send"`
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `op` | `"send"` | Yes | Sends one message to one peer or to `"all"`. |
| `to` | `string` | Yes | Peer id such as `Main`, or `"all"` for broadcast. Whitespace is trimmed. |
| `message` | `string` | Yes | Message body. Whitespace is trimmed; empty-after-trim is rejected. |
| `awaitReply` | `boolean` | No | Wait for prose replies. Defaults to `true` for direct messages and `false` for `to: "all"`. |
| `op` | `"send" \| "wait" \| "inbox" \| "list"` | Yes | Operation. |
| `to` | `string` | `send` | Recipient agent id, or `"all"` for broadcast. Whitespace trimmed; self-send rejected. |
| `message` | `string` | `send` | Message body. Empty-after-trim is rejected. |
| `replyTo` | `string` | No | `send`: message id being answered. |
| `await` | `boolean` | No | `send`: after delivery, block until the next message from that peer arrives (round-trip sugar). Invalid with `to: "all"`. |
| `from` | `string` | No | `wait`: only accept a message from this agent id. |
| `timeoutMs` | `number` | No | `wait` / `send await:true`: timeout in milliseconds; `0` waits indefinitely. Defaults to `irc.timeoutMs`. |
| `peek` | `boolean` | No | `inbox`: list messages without consuming them. |
## Outputs
- Single-shot `AgentToolResult`; no streaming updates.
- `content` is one text block.
- `list` returns either `No other live agents.` or a bullet list headed by `<n> peer(s):`.
- `send` returns delivery summary text, then optional `## Replies`, `## Failed`, and `Unknown / unavailable peers:` sections.
- `details` is structured metadata:
- `list`: `{ op, from, peers, channels }`
- `send`: `{ op, from, to, delivered, replies?, failed?, notFound? }`
- The tool does not return raw IRC frames, message ids, or a transcript object.
- `content` is one text block:
- `list`: `No other agents.` or `<n> peer(s):` bullets — `id [displayName · kind · status]` plus unread count, parent, and last-activity age; a footer notes that parked agents are revived automatically when messaged.
- `send`: per-recipient delivery receipts (`injected` / `woken` / `revived` / `failed — <error>`); with `await: true`, the reply body or a clean no-reply timeout note.
- `wait`: the consumed message as `[<msgId>] <from>: <body>` (with a reply-to tag), or `No message within <duration>.`
- `inbox`: `Inbox empty.` or `<n> message(s):` bullets.
- `details: IrcDetails`: `{ op, from?, to?, receipts?, waited?, inbox?, peers? }`. `waited` is `null` when a wait timed out; `receipts` carry `{ to, outcome, error? }`.
## Flow
1. `IrcTool.createIf` only constructs the tool when `irc.enabled` is on and the session has both an `AgentRegistry` and `getAgentId` (`packages/coding-agent/src/tools/irc.ts`).
2. Tool discovery adds another gate in `packages/coding-agent/src/tools/index.ts`: if the caller is `Main` and `async.enabled` is off, `irc` is hidden because the main agent cannot talk to concurrent peers in sync mode.
3. `execute` resolves the process-global registry and sender id. Missing either returns a text error result instead of throwing.
4. `op: "list"` calls `registry.listVisibleTo(senderId)`, which exposes every other agent in flat namespace whose status is `running` or `idle` (`packages/coding-agent/src/registry/agent-registry.ts`).
5. `list` formats human-readable lines and returns `channels` as `['all', ...peerIds]`. These are logical targets only; there is no channel join state.
6. `op: "send"` trims `to` and `message`; missing values produce text errors.
7. `send` resolves targets:
- `to === "all"`: all visible peers.
- otherwise: one exact registry id, excluding self and excluding peers not in `running`/`idle`.
8. `send` chooses `awaitReply = params.awaitReply ?? !isBroadcast`.
9. Each target is dispatched in parallel via `target.session.respondAsBackground(...)`. One slow or failing peer does not block dispatch to the others.
10. `respondAsBackground` emits an `irc_message` session event, forwards a display-only relay to the main session UI, and either:
- queues just the incoming message for later history injection when `awaitReply === false`, or
- renders `packages/coding-agent/src/prompts/system/irc-incoming.md`, runs `runEphemeralTurn` with `toolChoice: "none"`, emits an auto-reply event, then queues both incoming and reply messages for history injection.
11. Deferred injection waits until the recipient is no longer streaming; `#flushPendingBackgroundExchanges` appends the custom messages through normal `message_start`/`message_end` external events so persistence and listeners see them.
12. Dispatch waits are bounded by `irc.timeoutMs` (default `120_000` ms). A value of `0` disables the local timeout; parent aborts still abort the dispatch.
13. `send` aggregates `delivered`, `replies`, `failed`, and `notFound`, then returns one text summary plus matching `details`.
1. `IrcTool.createIf` constructs the tool only when `irc.enabled` is on and the session has both an `AgentRegistry` and `getAgentId`. There is no longer a main-agent gate on `async.enabled` — the main agent is never sync-blocked.
2. `execute` resolves the registry and sender id; missing either returns a text error result instead of throwing.
3. `op: "list"`: `registry.list()` minus self and minus `aborted` agents — `parked` peers ARE listed. Each row includes the unread count from `IrcBus.unreadCount(...)` and last activity.
4. `op: "send"` validates `to`/`message`, rejects self-sends, and rejects `await` with `to: "all"`.
5. Target resolution: broadcasts fan out to `registry.listVisibleTo(senderId)` (live peers only — `running`/`idle`; reviving every parked agent on a broadcast would be a stampede). Direct sends go through the bus unfiltered, so a parked recipient is revived.
6. `IrcBus.send(...)` is fire-and-forget — it never blocks on the recipient generating anything. Delivery by recipient status:
- `running` → message enqueued and injected as a non-interrupting aside at the recipient's next step boundary (`AgentSession.deliverIrcMessage`, rendered from `irc-incoming.md`, persisted as an `irc:incoming` custom message) — receipt `injected`;
- `idle` (live session) → enqueued and a real turn is started — the message wakes the agent — receipt `woken`;
- `parked` → `AgentLifecycleManager.global().ensureLive(to)` revives the session first, then the wake path — receipt `revived`;
- resolution/revival failure → receipt `failed` with the error; other recipients still complete.
7. `send` with `await: true` then calls `IrcBus.wait(senderId, { from: to }, timeoutMs, signal)` and appends the reply (or a no-reply note suggesting `inbox`/`wait`) to the result.
8. `op: "wait"` blocks until a message for the caller (optionally filtered by `from`) arrives, consumes it, and returns it. Timeout returns a clean "no message" result, not an error.
9. `op: "inbox"` drains pending messages (or peeks with `peek: true`) without blocking.
10. Timeouts resolve as `params.timeoutMs ?? irc.timeoutMs`, normalized: `0` disables the timeout, negative/non-finite values fall back to the default `120_000`, positive values are truncated and clamped to ≥ 1 ms.
## Modes / Variants
- `list`: enumerate visible peers and logical channels.
- `send` direct message: one exact peer id, default synchronous auto-reply.
- `send` broadcast: `to: "all"`, default fire-and-forget (`awaitReply: false`) to every visible peer.
- `send` with `awaitReply: false`: recipient records the incoming message but does not generate a reply.
- `send` with `awaitReply: true`: recipient performs a no-tools ephemeral LLM turn and returns prose.
- `list`: enumerate peers with status (`running`/`idle`/`parked`), unread counts, and last activity.
- `send` direct: one exact peer id; wakes idle peers, revives parked ones.
- `send` broadcast: `to: "all"` to every live peer; parked peers are skipped.
- `send` + `await: true`: round-trip convenience — send, then wait for the next message from that peer. Replaces the old `awaitReply` auto-reply semantics without a fake reply.
- `wait`: block for an incoming message, optionally filtered by sender.
- `inbox`: non-blocking drain or peek.
## Side Effects
- Session state
- Reads from the process-global `AgentRegistry`.
- Emits `irc_message` session events on recipient sessions.
- Queues IRC custom messages into recipient persisted history after the current stream finishes.
- For non-main recipients, forwards display-only relay observations into the main session UI; these relays are not persisted to the main agent history.
- Subagents inherit `irc.enabled` from task executor settings.
- Reads the process-global `AgentRegistry`; direct sends to parked agents revive their sessions through the lifecycle manager.
- Persists `irc:incoming` custom messages into recipient history; replies are ordinary turns in the recipient's own session.
- Waking an idle/parked recipient starts a real agent turn (model requests, tool use) in that recipient.
- User-visible prompts / interactive UI
- IRC events render as `[IRC]` transcript lines in the TUI.
- Auto-replies are generated from `packages/coding-agent/src/prompts/system/irc-incoming.md` and explicitly forbid tool use.
- IRC events render as transcript cards in the TUI; the Agent Hub shows per-agent unread counts.
- Background work / cancellation
- `send` starts one background `respondAsBackground` call per target.
- The caller's `AbortSignal` is forwarded into each background reply turn. `irc.timeoutMs` creates a per-recipient `AbortController` and reports timeout failures per target.
- `send` itself never blocks on reply generation; only `wait` (and `await: true`) blocks, bounded by the resolved timeout and the caller's `AbortSignal`.
- Network
- No IRC server connection.
- When `awaitReply: true`, the recipient may make model-provider API calls through `runEphemeralTurn`.
- No IRC server connection. Woken recipients make their own model-provider calls as part of their turn.
- Filesystem
- No direct filesystem writes in the tool itself.
- No direct filesystem writes in the tool itself; recipient turns persist to their session JSONL as usual.
## Limits & Caps
- Availability gates:
- `irc.enabled` defaults to `true` in `packages/coding-agent/src/config/settings-schema.ts`.
- Main agent tool discovery suppresses `irc` when `async.enabled` is off (`packages/coding-agent/src/tools/index.ts`).
- Visibility scope: only peers in status `running` or `idle` are addressable via `listVisibleTo`.
- Reply execution:
- No tools are available in auto-reply turns (`toolChoice: "none"` in `runEphemeralTurn`).
- `irc.timeoutMs` defaults to `120_000`; `0` disables the timeout, non-finite values fall back to the default, and positive values are truncated and clamped to at least `1` ms.
- No retry, backoff, rate limit, or reply length cap is defined in `irc.ts`; behavior otherwise relies on the underlying model stream and any upstream API limits.
- Flush scheduling: deferred history injection polls every `50` ms while the recipient is still streaming (`#scheduleBackgroundExchangeFlush` in `packages/coding-agent/src/session/agent-session.ts`).
- Availability gates: `irc.enabled` (default `true`), an `AgentRegistry`, and a caller agent id.
- Mailboxes are bounded at 100 messages per agent (`MAILBOX_CAP` in `packages/coding-agent/src/irc/bus.ts`); oldest messages are dropped beyond the cap.
- `irc.timeoutMs` defaults to `120_000` and is the default `wait` / `send await:true` timeout; `0` disables the timeout, non-finite or negative values fall back to the default, positive values are truncated and clamped to at least `1` ms.
- Broadcast scope: live peers only (`running`/`idle`) via `listVisibleTo`; direct sends address any non-aborted agent, including parked ones.
## Errors
- The tool returns text errors, not thrown exceptions, for:
- The tool returns text errors (with `isError: true`), not thrown exceptions, for:
- missing registry: `IRC is unavailable in this session.`
- missing sender id: `IRC is unavailable: caller has no agent id.`
- missing `to`: `` `to` is required for op="send". ``
- missing `message`: `` `message` is required for op="send". ``
- unknown op: `Unknown irc op.`
- Unknown, self-addressed, non-running, and non-idle direct targets are reported under `details.notFound` and in the text footer `Unknown / unavailable peers:`.
- If a target has no attached session, it is treated as not found.
- Exceptions thrown by `respondAsBackground`, `runEphemeralTurn`, abort handling, or timeout handling are caught per-target and surfaced under `details.failed` as `{ id, error }`; other recipients still complete.
- If no target succeeds, `send` still returns normally with `No recipients received the message.` and optional `failed`/`notFound` metadata.
- missing `to` / `message` on `send`
- self-send: `Cannot send an IRC message to yourself.`
- `await` with `to: "all"`
- unknown op
- Per-recipient delivery failures surface as `failed` receipts with the error message; `send` is marked `isError` only when no recipient received the message.
- `wait` timeout is a normal result (`waited: null`), not an error.
## Notes
- This is IRC-like naming only. There are no servers, sockets, nick registration, auth handshakes, channels beyond `all`, or commands such as join/part/topic.
- Addressing is by exact agent id from the registry; there is no fuzzy lookup or aliasing.
- `channels` in `list` is synthetic output: `all` plus visible peer ids. Nothing is persisted across calls as channel membership.
- Persistence is per recipient history, not per sender history. The sender gets the tool result; the recipient later sees injected custom messages on its next turn.
- The main UI may show IRC relays for conversations it was not part of, but those relay records are explicitly display-only.
- Because reply generation snapshots in-flight assistant text, a recipient can answer based on partially streamed context.
- Direct self-messaging is rejected by resolving the target as unavailable.
- This is IRC-like naming only: no servers, sockets, channels, or join/part state. Addressing is by exact registry agent id.
- Replies are real turns by the recipient — the old ephemeral no-tools auto-reply (`awaitReply` / `respondAsBackground`) no longer exists. A recipient may keep working before answering; check `inbox` or `wait` again rather than re-sending.
- Wake-on-message is the revive primitive: messaging a parked agent is equivalent to resuming it (same `ensureLive` path as `task(resume:)` and the Agent Hub).
- Message ids are Snowflakes; pass them as `replyTo` to thread an answer to a specific message.
- Persistence is per recipient history: the sender gets receipts in the tool result; the recipient sees the injected `irc:incoming` message in its own transcript (visible via `history://<id>`).
+2 -3
View File
@@ -7,7 +7,6 @@
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/job.md`
- Key collaborators:
- `packages/coding-agent/src/async/job-manager.ts` — job registry, cancellation, delivery suppression.
- `packages/coding-agent/src/async/support.ts` — feature gating for background jobs.
- `packages/coding-agent/src/tools/bash.ts` — explicit async bash and auto-backgrounded bash jobs.
- `packages/coding-agent/src/task/index.ts` — async task-job scheduling.
- `packages/coding-agent/src/sdk.ts` — automatic follow-up delivery for unsuppressed completions.
@@ -45,7 +44,7 @@ Read-only snapshot path:
- Calling `job` with `list: true` returns a markdown summary of every job spawned by the calling agent (running + completed within retention) without waiting.
## Flow
1. `JobTool.createIf(...)` in `packages/coding-agent/src/tools/job.ts` only exposes the tool when `isBackgroundJobSupportEnabled(...)` returns true for either `async.enabled` or `bash.autoBackground.enabled`.
1. `JobTool` is registered unconditionally in `packages/coding-agent/src/tools/index.ts`; there is no `async.enabled` gate (the `task` tool always schedules background jobs).
2. `execute(...)` fetches `session.asyncJobManager`. If absent, it returns `Async execution is disabled; no background jobs are available.`
3. `cancel` ids are processed first:
- `manager.getJob(id)` missing → `not_found`.
@@ -83,7 +82,7 @@ Spawn paths that produce jobs:
- `async: true` always registers a `type: "bash"` job with `AsyncJobManager.register(...)` and returns a start message.
- auto-background mode (`bash.autoBackground.enabled`) starts the same managed job path for non-PTY commands, waits up to `min(bash.autoBackground.thresholdMs, timeoutMs - 1000)`, and if the command is still running returns a background-job start result instead of inline command output.
- `packages/coding-agent/src/task/index.ts`
- when `async.enabled` is on, the chosen agent is not blocking, and `tasks.length > 0`, each task item is registered as a `type: "task"` job.
- every `task` call (spawn or resume) registers one `type: "task"` job, unless the session has no job manager or the agent definition declares `blocking: true` (sync fallback).
Lifecycle and exact state names:
- Conceptual scheduling path: `pending` (only task-progress bookkeeping before work starts) → `running` → `completed` / `failed`; cancellation changes a running async job to `cancelled`.
+2 -2
View File
@@ -10,7 +10,7 @@
- `packages/coding-agent/src/tools/archive-reader.ts` — detect `archive.ext:inner/path`, index archives, list/read entries.
- `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite targets, parse selectors, render tables.
- `packages/coding-agent/src/tools/fetch.ts` — URL parsing, fetch/render pipeline, URL cache/artifacts.
- `packages/coding-agent/src/internal-urls/router.ts` — resolve `agent://`, `artifact://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`.
- `packages/coding-agent/src/internal-urls/router.ts` — resolve `agent://`, `artifact://`, `history://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`.
- `packages/coding-agent/src/edit/notebook.ts` — convert `.ipynb` to editable `# %% [...] cell:N` text.
- `packages/coding-agent/src/utils/file-display-mode.ts` — decide hashline vs line-number vs raw display.
- `packages/coding-agent/src/workspace-tree.ts` — render directory trees.
@@ -196,7 +196,7 @@ URL selectors are parsed separately in `packages/coding-agent/src/tools/fetch.ts
### Internal URLs
- `read` does not resolve these itself; it delegates to `session.internalRouter.resolve()`.
- Registered protocols are outside this file, but the router in `packages/coding-agent/src/internal-urls/router.ts` is built for `agent://`, `artifact://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, and `skill://`.
- Registered protocols are outside this file, but the router in `packages/coding-agent/src/internal-urls/router.ts` is built for `agent://`, `artifact://`, `history://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, and `skill://`.
- `#handleInternalUrl()` behavior:
- parses the URL with `parseInternalUrl()` so colons inside the host segment are legal
- for `agent://`, treats non-root path extraction or `?q=` extraction as a special no-pagination mode
+112 -166
View File
@@ -1,6 +1,6 @@
# task
> Launch subagents for parallel, optionally isolated work.
> Spawn one subagent per call to work in the background, or resume an existing one.
## Source
- Entry: `packages/coding-agent/src/task/index.ts`
@@ -9,217 +9,163 @@
- `packages/coding-agent/src/task/types.ts` — dynamic schema, progress/result types, output caps.
- `packages/coding-agent/src/task/discovery.ts` — discover project/user/plugin/bundled agents.
- `packages/coding-agent/src/task/agents.ts` — bundled agent definitions and frontmatter parsing.
- `packages/coding-agent/src/task/executor.ts` — create child sessions, run subagents, collect output.
- `packages/coding-agent/src/task/parallel.ts` — concurrency-limited scheduling and async semaphore.
- `packages/coding-agent/src/task/executor.ts` — create child sessions, run/resume subagents, collect output, hand finished sessions to the lifecycle manager.
- `packages/coding-agent/src/registry/agent-lifecycle.ts` — idle-TTL parking and revival of finished subagents.
- `packages/coding-agent/src/registry/agent-registry.ts` — process-global agent directory (`running | idle | parked | aborted`).
- `packages/coding-agent/src/async/job-manager.ts` — background job registration, progress, and result delivery.
- `packages/coding-agent/src/task/parallel.ts` — `Semaphore` used for the session-scoped concurrency bound.
- `packages/coding-agent/src/task/isolation-backend.ts` — isolation backend resolution and platform fallback.
- `packages/coding-agent/src/task/worktree.ts` — worktree / FUSE / ProjFS setup, patch capture, branch merge.
- `packages/coding-agent/src/task/output-manager.ts` — session-scoped `agent://` id allocation.
- `packages/coding-agent/src/task/simple-mode.ts` — `default` / `schema-free` / `independent` field gating.
- `packages/coding-agent/src/task/name-generator.ts` — default AdjectiveNoun agent ids.
- `packages/coding-agent/src/task/simple-mode.ts` — `default` / `schema-free` / `independent` schema gating.
- `packages/coding-agent/src/internal-urls/agent-protocol.ts` — resolve `agent://<id>` to saved subagent output.
- `packages/coding-agent/src/internal-urls/history-protocol.ts` — resolve `history://<id>` to a concise transcript.
- `packages/coding-agent/src/tools/index.ts` — tool registration and recursion-depth gating.
- `packages/coding-agent/src/sdk.ts` — child-session router/tool wiring and per-subagent `AgentOutputManager`.
- `docs/task-agent-discovery.md` — deeper discovery and precedence notes.
- `docs/handoff-generation-pipeline.md` — session artifact/handoff persistence patterns used by the wider session layer.
## Inputs
### Default mode (`task.simple = "default"`)
One call spawns (or resumes) exactly one subagent. There is no batch parameter and no shared `context` parameter — shared background goes into a `local://` file (e.g. `local://ctx.md`) that each assignment references; subagents share the parent's `local://` root.
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `agent` | `string` | Yes | Exact agent name for every task item. Resolved at execution time through `discoverAgents(...)`. |
| `tasks` | `Array<{ id: string; description: string; assignment: string }>` | Yes | Batch of small, self-contained task items. `id` max length 48 in schema; duplicate ids are rejected case-insensitively at runtime. |
| `context` | `string` | No | Shared background prepended to every subagent system prompt. Trimmed before use. |
| `schema` | `string` | No | JSON-encoded JTD schema. Overrides agent/session output schema when this mode allows task-level schemas. |
| `isolated` | `boolean` | No | Only present when the tool is created with isolation enabled. Requests isolated execution for the whole batch. |
| `agent` | `string` | Conditional | Agent type to spawn. Required unless `resume` is set; providing both is a validation error. |
| `resume` | `string` | Conditional | Existing agent id — revive the agent if parked and run a follow-up assignment in its existing session. Cannot be combined with `agent` or `isolated`. |
| `id` | `string` | No | Stable agent id, schema max length 48. Defaults to a generated AdjectiveNoun name. Uniquified per session by `AgentOutputManager`. |
| `description` | `string` | No | UI label only; the subagent never sees it. |
| `assignment` | `string` | Yes | The work — complete, self-contained instructions. Empty-after-trim is rejected. |
| `schema` | `string` | No | JSON-encoded JTD schema for the expected `yield` payload. Field exists only when `task.simple = "default"`. |
| `isolated` | `boolean` | No | Run in an isolated workspace and return patches. Field exists only when `task.isolation.mode` is not `none`. Isolated agents are NOT resumable. |
`tasks[].description` is UI-only. `tasks[].assignment` is the actual per-task instruction.
### Schema-free mode (`task.simple = "schema-free"`)
Same as default, except `schema` is rejected by `validateTaskModeParams(...)` in `packages/coding-agent/src/task/index.ts`.
### Independent mode (`task.simple = "independent"`)
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `agent` | `string` | Yes | Exact agent name. |
| `tasks` | `Array<{ id: string; description: string; assignment: string }>` | Yes | Same item shape, but each `assignment` must carry all required background because shared `context` is disabled. |
| `isolated` | `boolean` | No | Same conditional field as above. |
In this mode both `context` and `schema` are rejected.
Simple-mode gating (`task.simple`, one axis): `default` accepts the per-call `schema` override; `schema-free` and `independent` reject it (`validateTaskModeParams(...)`). `independent` additionally renders the subagent user prompt with the independent-mode flag. Agent frontmatter and inherited session schemas work in every mode.
## Outputs
The tool returns one text block plus `details: TaskToolDetails`.
`details` fields:
- `projectAgentsDir: string | null` — nearest discovered project `agents/` dir.
- `results: SingleResult[]` — one entry per task in input order for synchronous execution; empty for async-launch responses.
- `totalDurationMs: number`
- `usage?: Usage` — sum of per-subagent assistant-message usage.
- `outputPaths?: string[]` — written `.md` artifact paths for completed subagent outputs.
- `progress?: AgentProgress[]` — live or final per-task progress snapshots.
- `async?: { state: "running" | "completed" | "failed"; jobId: string; type: "task" }` — present for background execution updates/results.
Immediate (async) response — the normal case:
- `content`: `` Spawned agent `<id>` (job `<jobId>`). The result will be delivered when it yields. ... `` (or `Resumed agent ...`), plus a coordination hint (`irc` DM when enabled, otherwise `job`).
- `details`: `{ projectAgentsDir: null, results: [], totalDurationMs: 0, progress: [<seeded AgentProgress>], async: { state: "running", jobId, type: "task" } }`.
- Live progress keeps streaming into the same tool block via `onUpdate(...)`; the final result arrives later as an async-result injection into the parent conversation. The delivery text appends a resume hint: `` <id> is now idle — task(resume:"<id>") to continue it, transcript at history://<id> `` (aborted variant points at the transcript only).
Settled (sync-fallback or job-body) response:
- `content`: summary rendered from `packages/coding-agent/src/prompts/tools/task-summary.md` with a preview capped at 5000 chars; `agent://<id>` holds the full output.
- `details.results`: at most one `SingleResult`; `usage`, `outputPaths` populated.
`SingleResult` includes:
- identity: `index`, `id`, `agent`, `agentSource`, `description`, optional `assignment`
- status: `exitCode`, optional `error`, optional `aborted`, optional `abortReason`
- output: `output`, `stderr`, `truncated`, `durationMs`, `tokens`
- status: `exitCode`, optional `error`, optional `aborted`, optional `abortReason`, optional `retryFailure`
- output: `output`, `stderr`, `truncated`, `durationMs`, `tokens`, `requests`, optional `contextTokens`/`contextWindow`
- artifact metadata: `outputPath?`, `patchPath?`, `branchName?`, `nestedPatches?`, `outputMeta?`
- extracted tool data: `extractedToolData?` from registered subprocess tool handlers such as `yield` and `report_finding`
Artifacts and side channels:
- Every subagent with an artifacts dir writes `<id>.md`; `agent://<id>` resolves to that file.
- If the output file is JSON, `agent://<id>/<path>` and `agent://<id>?q=<query>` perform JSON extraction in `packages/coding-agent/src/internal-urls/agent-protocol.ts`.
- When the parent session persists artifacts, each subagent also gets `<id>.jsonl` session history.
- Isolated patch mode writes `<id>.patch` per successful task before merge.
- Async mode returns immediately after job registration, then emits `onUpdate(...)` progress snapshots and later hands completion to the session async-job pipeline.
- Every subagent with an artifacts dir writes `<id>.md`; `agent://<id>` resolves to that file. Resumes overwrite it per assignment.
- If the output file is JSON, `agent://<id>/<path>` and `agent://<id>?q=<query>` perform JSON extraction.
- Each subagent gets `<id>.jsonl` session history when the parent persists artifacts; `history://<id>` renders it as a concise transcript (works for live and parked agents).
- Isolated patch mode writes `<id>.patch` before merge.
## Flow
1. `TaskTool.create(...)` in `packages/coding-agent/src/task/index.ts` calls `discoverAgents(session.cwd)` once to build the dynamic prompt description from current agents and `task.simple` capabilities.
2. `execute(...)` validates mode-gated fields with `validateTaskModeParams(...)`.
3. It decides async vs sync:
- sync when `async.enabled` is false
- sync when the selected cached agent has `blocking === true`
- sync when `tasks.length === 0`
- otherwise async job scheduling
4. Async path:
- allocate unique output ids with `AgentOutputManager.allocateBatch(...)`
- create one async job per task through `session.asyncJobManager.register(...)`
- limit concurrent job bodies with `Semaphore(task.maxConcurrency)` from `packages/coding-agent/src/task/parallel.ts`
- each job body calls `#executeSync(...)` with a one-task batch and the preallocated id
- `onUpdate(...)` emits aggregate `progress` snapshots and `details.async`
5. Sync path (`#executeSync(...)`) rediscovers agents from disk via `discoverAgents(...)`, so runtime resolution can differ from the earlier prompt description.
6. It resolves the requested agent with `getAgent(...)`, rejects unknown or disabled agents, and enforces parent spawn policy plus `PI_BLOCKED_AGENT` self-recursion prevention.
7. It derives the effective output schema in priority order: task call `schema` (if allowed) → agent frontmatter `output` → inherited parent session schema.
8. It validates task ids: missing ids and case-insensitive duplicates are immediate errors.
9. If `isolated` was requested, it requires a git repo (`getRepoRoot(...)` / `captureBaseline(...)`) and resolves the actual backend through `resolveIsolationBackendForTaskExecution(...)`.
10. It chooses an artifacts dir from the parent session when available, otherwise a temp dir, and writes `context.md` there when `session.getCompactContext?.()` returns content.
11. It allocates unique ids again if the caller did not preallocate them, then builds `tasksWithUniqueIds`.
12. For each task, it seeds an `AgentProgress` entry and runs `runTask(...)` through `mapWithConcurrencyLimit(...)` using `task.maxConcurrency`.
13. Non-isolated `runTask(...)` calls `runSubprocess(...)` directly with parent cwd.
14. Isolated `runTask(...)`:
- creates an isolation workspace (`ensureWorktree(...)`, `ensureFuseOverlay(...)`, or `ensureProjfsOverlay(...)`)
- applies the captured baseline for worktrees
- runs `runSubprocess(...)` inside that workspace
- on success, either commits to a per-task branch (`mergeMode === "branch"`) or captures a patch with `captureDeltaPatch(...)`
- always cleans up the isolation workspace/backend
15. `runSubprocess(...)` in `packages/coding-agent/src/task/executor.ts` creates a child agent session with:
- isolated settings snapshot via `Settings.isolated(...)`, forcing `async.enabled = false` and `bash.autoBackground.enabled = false`
- child `agentId` / `parentTaskPrefix` equal to the allocated task id
- child internal URL router and `AgentOutputManager` from `packages/coding-agent/src/sdk.ts`
- the shared `context`, optional `context.md` reference, optional isolation worktree path, output schema, and IRC peer roster in the system prompt template
16. Child tool availability is derived from the agent definition plus runtime guards:
- explicit `agent.tools` if provided
- auto-add `task` when the agent has `spawns` and recursion depth allows it
- remove `task` at or past `task.maxRecursionDepth`
- expand `exec` to `eval` and `bash`
- strip parent-owned `todo` after session creation
17. `runSubprocess(...)` subscribes to child agent events, coalesces progress updates every 150 ms, forwards lifecycle/progress events on the parent event bus, and extracts tool data through `subprocessToolRegistry`.
18. The child must finish through the hidden `yield` tool. If it does not, `runSubprocess(...)` sends up to 3 reminder prompts; the last reminder forces `toolChoice = yield` when supported.
19. Finalization uses `finalizeSubprocessOutput(...)` to reconcile raw assistant text, `yield` payloads, structured schemas, `report_finding` data, and abort states. Output is truncated with `MAX_OUTPUT_BYTES` / `MAX_OUTPUT_LINES` before returning to the parent, but the full raw output is still written to `<id>.md`.
20. After all sync tasks finish, `#executeSync(...)` aggregates usage, collects artifact paths, and if isolation was used merges results back:
- branch mode: cherry-pick per-task branches with `mergeTaskBranches(...)`, then delete merged branches with `cleanupTaskBranches(...)`
- patch mode: combine non-empty patch artifacts, dry-check with `git.patch.canApplyText(...)`, then apply or leave manual artifacts
- nested repo patches are applied separately with `applyNestedPatches(...)`
21. The final text summary is rendered from `packages/coding-agent/src/prompts/tools/task-summary.md` and includes `agent://<id>` handles for outputs that exist.
1. `TaskTool.create(...)` discovers agents once per cwd through a process-level memo (`discoverAgentsForCreate`) to render the dynamic prompt description.
2. `execute(...)` repairs raw params (`repairTaskParams`), then validates: schema gating per `task.simple`, `agent` XOR `resume`, `resume` excludes `isolated`, non-empty `assignment`.
3. Sync fallback only when the session has no `AsyncJobManager` (orphaned host) or the selected agent definition declares `blocking: true`; the call then runs `#executeSync(...)` inline under the session-scoped semaphore.
4. Otherwise execution is always async:
- the agent id is resolved up front — `resume` must name a registered agent (else `ToolError` pointing at `irc` op:"list" and `history://<id>`); spawns allocate via `AgentOutputManager.allocate(params.id || generateTaskName())`;
- one `type: "task"` job is registered with `session.asyncJobManager` (`id` = agent id, `queued: true`, `ownerId` = caller agent id) and the tool returns immediately;
- the job body acquires the session-scoped `Semaphore` (one per `TaskTool` instance, sized from `task.maxConcurrency` at first use), marks the job running, runs `#executeSync(...)`, and reports progress through `buildAsyncDetails`/`onUpdate`;
- a failed or aborted run throws `TaskJobError` so the job lands `failed`, but the agent itself stays registered and interrogable.
5. `#executeSync(...)` dispatches: `resume` → `#executeResume(...)`, else `#runSpawn(...)`.
6. Resume path (`#executeResume`):
- `AgentLifecycleManager.global().ensureLive(resumeId)` returns the live session, reviving a parked one from its session JSONL; unknown ids or parked-without-reviver throw a `ToolError`;
- `resumeSubprocess(...)` in `packages/coding-agent/src/task/executor.ts` injects the rendered follow-up through the session's normal prompt path and drives it through the same monitor/yield/finalize pipeline as a spawn;
- the session is never disposed here — registry status settles back to `idle` (even on failure/abort) and the lifecycle manager re-arms the idle TTL.
7. Spawn path (`#runSpawn`) rediscovers agents from disk, so runtime resolution can differ from the create-time description.
8. It resolves the requested agent, rejects unknown or settings-disabled agents, and enforces parent spawn policy plus `PI_BLOCKED_AGENT` self-recursion prevention.
9. Output schema priority: task call `schema` (when `task.simple` allows) → agent frontmatter `output` → inherited parent session schema.
10. Plan mode swaps in an `effectiveAgent` with a read-only tool subset and plan-mode prompt; `runSubprocess(...)` receives the effective agent.
11. If `isolated`, it requires a git repo (`getRepoRoot(...)` / `captureBaseline(...)`) and resolves the backend through isolation-backend resolution with platform fallback.
12. Artifacts dir comes from the parent session file when available, otherwise a temp dir. When the session is executing an approved plan, the plan reference is handed to the subagent.
13. Non-isolated spawns call `runSubprocess(...)` directly with parent cwd; isolated spawns run inside the isolation workspace, then commit to a branch (`mergeMode === "branch"`) or capture a patch, and always clean up the workspace.
14. `runSubprocess(...)` creates a child agent session with an isolated settings snapshot (forcing `async.enabled = false` and `bash.autoBackground.enabled = false` — subagents are internally synchronous), child `agentId` equal to the allocated id, child internal URL router/`AgentOutputManager`, output schema, and the IRC peer roster in the system prompt.
15. Child tool availability: explicit `agent.tools` if provided; auto-add `task` when the agent has `spawns` and depth allows; strip `task` at `task.maxRecursionDepth`; expand `exec` to `eval` + `bash`; strip parent-owned `todo`.
16. The child must finish through the hidden `yield` tool; up to 3 reminder prompts, the last forcing `toolChoice = yield` when supported. `finalizeSubprocessOutput(...)` reconciles raw text, `yield` payloads, structured schemas, `report_finding` data, and abort states.
17. End-of-run lifecycle (keep-alive, in `runSubprocess`'s finalizer):
- hard abort (caller signal / wall-clock / budget) → registry status `aborted`, session disposed — terminal;
- isolated run → status `parked` without a reviver (workspace is merged + cleaned, so the session is not resumable; transcript stays readable via `history://`), then session disposed and detached;
- everything else (success and failure alike) → status `idle` with the live session attached, and `AgentLifecycleManager.global().adopt(id, { idleTtlMs, revive })` arms the park timer. The reviver reopens the session JSONL (park closed the writer, so the single-writer lock is taken cleanly).
18. Lifecycle thereafter: `idle` agents are parked after `task.agentIdleTtlMs` (session disposed; `AgentRef` + session file retained); messaging (`irc`), `task(resume:)`, or the Agent Hub revives them back to `idle`. `"Main"` is never parked.
## Modes / Variants
- Execution mode
- Sync inline execution — default path.
- Async background execution — one async job per task item when `async.enabled` is on and the chosen agent is not marked `blocking`.
- Simple mode
- `default` — accepts shared `context` and per-call `schema`.
- `schema-free` — accepts `context`, rejects `schema`.
- `independent` — rejects `context` and `schema`; each assignment stands alone.
- Isolation backend
- `none` — no isolation.
- `worktree` — detached git worktree plus baseline replay.
- `fuse-overlay` — Unix FUSE overlay mount.
- `fuse-projfs` — Windows ProjFS overlay.
- Isolation merge strategy
- Patch mode — capture/apply root patches, keep patch artifacts when application fails.
- Branch mode — commit each task onto `omp/task/<id>` branch, cherry-pick into parent, preserve failed branches for manual resolution.
- Agent source
- Project custom agents — nearest project config/plugin agent directories, first by source-family precedence.
- User custom agents — user config/plugin agent directories after project dirs of the same source family.
- Bundled agents — appended last from `packages/coding-agent/src/task/agents.ts`.
- Bundled agent types
- `explore` — read-only scout with structured handoff output.
- `plan` — architecture/planning agent; may spawn `explore`.
- `designer` — UI/UX specialist.
- `reviewer` — review agent with `report_finding` extraction.
- `task` — general-purpose worker with full capabilities.
- `quick_task` — low-reasoning mechanical worker using the same task prompt body.
- `librarian` — source-grounded external API/library researcher.
- `oracle` — senior-engineer implementation/debugging/general consultation agent.
- Always-async background job — default; spawn and resume both go through `AsyncJobManager`.
- Sync inline fallback — only when no job manager exists or the agent definition has `blocking: true`.
- Spawn vs resume
- `agent: "<type>"` — fresh subagent with a new (or caller-provided) id.
- `resume: "<id>"` — follow-up assignment in an existing session; revives a parked agent first. Transcript accretes; `agent://<id>` is overwritten per assignment.
- Simple mode (`task.simple`)
- `default` — accepts per-call `schema`.
- `schema-free` / `independent` — reject `schema`; `independent` also flags the subagent user prompt as independent-mode.
- Isolation backend: `none`, `worktree`, `fuse-overlay`, `fuse-projfs`.
- Isolation merge strategy: patch mode (capture/apply root patches) or branch mode (commit to `omp/task/<id>`, cherry-pick into parent).
- Agent source precedence: project custom agents, then user custom agents, then bundled agents (`explore`, `plan`, `designer`, `reviewer`, `task`, `quick_task`, `librarian`, `oracle`).
## Side Effects
- Filesystem
- Writes `context.md`, `<id>.jsonl`, and `<id>.md` under the session artifacts dir or a temp task dir.
- In isolated patch mode writes `<id>.patch` artifacts.
- Creates/removes worktrees or overlay mount directories.
- In branch mode creates temporary worktrees and task branches.
- Writes `<id>.jsonl` and `<id>.md` under the session artifacts dir or a temp task dir; isolated patch mode writes `<id>.patch`.
- Creates/removes worktrees or overlay mount directories; branch mode creates temporary worktrees and task branches.
- Network
- Child sessions may use whichever networked tools/models their active tool set permits.
- MCP proxy tools can call existing parent MCP connections with a 60_000 ms timeout.
- Subprocesses / native bindings
- `fuse-overlayfs` and `fusermount`/`fusermount3` for FUSE isolation.
- ProjFS native bindings via `@oh-my-pi/pi-natives` on Windows.
- `fuse-overlayfs` and `fusermount`/`fusermount3` for FUSE isolation; ProjFS native bindings on Windows.
- Git operations for baseline capture, patch apply, worktrees, branches, stash, cherry-pick, commits.
- Session state (transcript, memory, jobs, checkpoints, registries)
- Creates child `AgentSession` instances with isolated settings snapshots.
- Registers async jobs in `session.asyncJobManager` for background task mode.
- Creates child `AgentSession` instances with isolated settings snapshots; finished sessions stay registered in the process-global `AgentRegistry` as `idle`/`parked` until process teardown or explicit release.
- Registers one async job per call in `session.asyncJobManager`; completion is injected into the parent as an async-result message.
- Arms idle-TTL timers in `AgentLifecycleManager` (unref'd; they never hold the process open).
- Emits `task:subagent:event`, `task:subagent:progress`, and `task:subagent:lifecycle` on the parent event bus.
- Allocates session-scoped output ids through `AgentOutputManager` so `agent://` remains unique across invocations and resumes.
- Shares the parent `local://` root with subagents by passing `localProtocolOptions` through `createAgentSession(...)`.
- User-visible prompts / interactive UI
- Async mode streams aggregate progress updates.
- Missing-`yield` recovery sends up to three internal reminder prompts to the child session.
- Final summaries include `<system-notification>` blocks for isolation fallbacks or merge failures.
- Allocates session-scoped output ids through `AgentOutputManager` so `agent://` stays unique across invocations and resumes.
- Shares the parent `local://` root and `ArtifactManager` with subagents.
- Background work / cancellation
- Parent abort stops scheduling new work, aborts active child sessions, and marks unscheduled tasks as skipped.
- Async jobs keep their own cancellation via `AsyncJobManager`.
- `job cancel` (or parent tool-call abort) cancels the job; a hard-aborted run lands `aborted` and is torn down.
- Missing-`yield` recovery sends up to three internal reminder prompts to the child session.
## Limits & Caps
- Per-subagent output truncation: `MAX_OUTPUT_BYTES = 500_000` and `MAX_OUTPUT_LINES = 5000` in `packages/coding-agent/src/task/types.ts`. Full raw output is still written to `<id>.md` before truncation is returned to the caller.
- Progress coalescing in child execution: `PROGRESS_COALESCE_MS = 150` in `packages/coding-agent/src/task/executor.ts`.
- Recent output tail for progress: `RECENT_OUTPUT_TAIL_BYTES = 8 * 1024` and `recentOutput` keeps the last 8 non-empty lines in `packages/coding-agent/src/task/executor.ts`.
- Missing-`yield` reminder retries: `MAX_YIELD_RETRIES = 3` in `packages/coding-agent/src/task/executor.ts`.
- MCP proxy timeout: `MCP_CALL_TIMEOUT_MS = 60_000` in `packages/coding-agent/src/task/executor.ts`.
- Task id schema cap: `tasks[].id` `maxLength: 48` in `packages/coding-agent/src/task/types.ts`.
- Prompt text says ids should be `≤32` chars, but the runtime schema allows 48; this mismatch is real.
- Async/full sync parallelism both use `task.maxConcurrency` from settings:
- sync path: `mapWithConcurrencyLimit(...)`
- async path: `Semaphore(...)` around job bodies
- Recursion depth gate: `task.maxRecursionDepth` from settings; `packages/coding-agent/src/tools/index.ts` hides the `task` tool at or beyond the limit, and `runSubprocess(...)` also strips child `task` access at max depth.
- Final inline summary preview per task uses `fullOutputThreshold = 5000` chars in `packages/coding-agent/src/task/index.ts`; longer outputs are summarized while `agent://<id>` points to the full artifact.
- Concurrency: one session-scoped `Semaphore` sized from `task.maxConcurrency` at first use (later setting changes do not resize it) bounds concurrent subagents across parallel `task` calls — both async job bodies and the sync fallback acquire it.
- Idle TTL: `task.agentIdleTtlMs`, default `420_000` ms (7 min); `<= 0` disables parking and keeps idle sessions live until exit.
- Per-subagent output truncation: `MAX_OUTPUT_BYTES = 500_000` and `MAX_OUTPUT_LINES = 5000` in `packages/coding-agent/src/task/types.ts` (overridable via `PI_TASK_MAX_OUTPUT_BYTES` / `PI_TASK_MAX_OUTPUT_LINES`). Full raw output is still written to `<id>.md`.
- Progress coalescing: `PROGRESS_COALESCE_MS = 150`; recent-output tail: `RECENT_OUTPUT_TAIL_BYTES = 8 * 1024` (last 8 non-empty lines).
- Missing-`yield` reminder retries: `MAX_YIELD_RETRIES = 3`; MCP proxy timeout: `MCP_CALL_TIMEOUT_MS = 60_000` — both in `packages/coding-agent/src/task/executor.ts`.
- Agent id schema cap: `id` `maxLength: 48` in `packages/coding-agent/src/task/types.ts`. Prompt text says ids should be `≤32` chars; this mismatch is real.
- Soft request budget (`task.softRequestBudget`) and wall clock (`task.maxRuntimeMs`) apply to spawns and resumes alike.
- Recursion depth gate: `task.maxRecursionDepth`; `packages/coding-agent/src/tools/index.ts` hides the `task` tool at or beyond the limit, and `runSubprocess(...)` also strips child `task` access at max depth.
- Final inline summary preview uses `fullOutputThreshold = 5000` chars in `packages/coding-agent/src/task/index.ts`; `agent://<id>` points to the full artifact.
## Errors
- Most validation failures are returned as normal tool text with empty `results`, not thrown:
- invalid simple-mode fields
- unknown/disabled agent
- missing tasks
- missing/duplicate task ids
- spawn-policy denial
- requesting `isolated` while isolation mode is `none`
- Isolated execution without a git repo returns `Isolated task execution requires a git repository. ...`.
- Backend resolution can return a hard error (`ProjFS isolation initialization failed...`) or a non-fatal warning with fallback to `worktree`.
- `mapWithConcurrencyLimit(...)` fails fast on non-abort worker exceptions; already completed results are preserved only in the thrown path’s local state, not surfaced unless the caller catches and converts them.
- Child-session failures surface as `SingleResult.exitCode = 1` with `stderr`/`error` populated.
- Parameter validation failures are returned as normal tool text with empty `results`:
- `schema` outside `task.simple = "default"`
- both or neither of `agent` / `resume`
- `resume` combined with `isolated`
- missing/empty `assignment`
- unknown or settings-disabled agent, spawn-policy denial, requesting `isolated` while isolation mode is `none`
- `resume` of an id not in the registry throws a `ToolError` naming `irc` op:"list" and `history://<id>`.
- `ensureLive(...)` failures (agent parked without a reviver — e.g. an isolated run — or torn down) surface as `` Cannot resume "<id>": ... `` `ToolError`s.
- Isolated execution without a git repo returns `Isolated task execution requires a git repository. ...`; backend resolution can hard-error (ProjFS init) or warn and fall back to `worktree`.
- Job registration failure returns `Failed to start background task job: ...`.
- Child failures surface as `SingleResult.exitCode = 1` with `stderr`/`error` populated; the async job is marked failed but the delivery text still carries the output plus a resume/transcript hint.
- If the child omits `yield`, `finalizeSubprocessOutput(...)` injects warnings such as `SYSTEM WARNING: Subagent exited without calling yield tool after 3 reminders.`
- Async scheduling failures are accumulated per task; if no jobs start, the tool returns `Failed to start background task jobs: ...`.
- `agent://<id>` resolution errors are model-visible when another tool reads them: no session, no artifacts dir, missing id, conflicting extraction syntax, or invalid JSON for extraction.
## Notes
- Agent discovery precedence is first-wins by exact name: project dirs before user dirs within a source family, plugin agent dirs after config dirs, bundled agents last. See `packages/coding-agent/src/task/discovery.ts` and `docs/task-agent-discovery.md`.
- `TaskTool.create(...)` caches discovered agents only for description rendering and the async blocking-agent decision. `#executeSync(...)` rediscovers agents each call.
- Custom agent frontmatter can override bundled agents by name. Bundled definitions are embedded at build time in `packages/coding-agent/src/task/agents.ts`.
- Child sessions do not inherit conversation history automatically. The only built-in carry-over is shared `context`, optional `context.md`, workspace tree/skills/context files, and shared `local://` root.
- `Settings.isolated(...)` gives each child a session-isolated settings snapshot; tool enablement is recomputed inside the child session rather than sharing mutable parent tool state.
- When the parent passes `mcpManager`, child sessions disable standalone MCP discovery and instead get proxy tools that reuse the parent connections.
- Plan mode mutates an `effectiveAgent` with a read-only tool subset and plan-mode prompt text, but `runSubprocess(...)` is still invoked with `agent` rather than `effectiveAgent`. Model/thinking/schema overrides use the effective agent; prompt/tool/spawn restrictions do not fully flow through this call path.
- Branch-mode merge temporarily stashes the parent repo before cherry-picking task branches. A stash-pop conflict is treated as merge failure and leaves recovery state behind.
- Patch-mode only applies combined root patches if every successful task produced a patch and `git.patch.canApplyText(...)` succeeds.
- Nested git repos are handled separately from the root repo. They are copied into isolated worktrees, diffed independently, and merged later with `applyNestedPatches(...)` because parent git cannot track their file-level changes.
- `agent://` ids are name-based (`Task` first, `Task-2`/`Task-3` only when the name repeats, nested like `Parent.Child`) by `AgentOutputManager`; this is what prevents artifact collisions across repeated or nested task invocations.
- Parallelism is parallel `task` calls in one assistant message; the session-scoped semaphore bounds the fan-out. There is no batch array.
- Shared background convention: write it once to a `local://` file and reference that path in each assignment — subagents share the parent's `local://` root. This replaces the removed `context` parameter.
- Prefer `resume` over a fresh spawn for follow-up work: the resumed agent already holds the relevant context. `irc` op:"list" shows idle/parked candidates; `history://<id>` shows what an agent has done.
- Subagents are internally synchronous: the executor forces `async.enabled = false` and `bash.autoBackground.enabled = false` in the child settings snapshot, so there are no fire-and-forget grandchildren.
- Agent discovery precedence is first-wins by exact name: project dirs before user dirs within a source family, plugin agent dirs after config dirs, bundled agents last. Create-time discovery is memoized per cwd for the prompt description; execution-time discovery stays fresh.
- Child sessions do not inherit conversation history. Built-in carry-over is the workspace tree/skills/context files, the shared `local://` root, and the approved-plan reference when one exists.
- When the parent passes `mcpManager`, child sessions disable standalone MCP discovery and get proxy tools that reuse parent connections.
- Branch-mode merge temporarily stashes the parent repo before cherry-picking; a stash-pop conflict is treated as merge failure and leaves recovery state behind. Patch mode only applies the combined root patch when `git.patch.canApplyText(...)` succeeds; failures leave the `.patch` artifact for manual handling.
- Nested git repos are diffed independently inside isolated workspaces and merged separately with `applyNestedPatches(...)`.
- `agent://` ids are name-based (`Task` first, `Task-2`/`Task-3` only when the name repeats, nested like `Parent.Child`) by `AgentOutputManager`; this is what prevents artifact collisions across repeated or nested invocations.
+3
View File
@@ -1,5 +1,6 @@
{
"name": "omp-monorepo",
"homepage": "https://omp.sh",
"private": true,
"type": "module",
"packageManager": "bun@1.3.14",
@@ -30,6 +31,7 @@
"@oh-my-pi/pi-natives": "15.10.12",
"@oh-my-pi/pi-tui": "15.10.12",
"@oh-my-pi/pi-utils": "15.10.12",
"@oh-my-pi/snapcompact": "15.10.12",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -130,6 +132,7 @@
"stats:tools": "python3 scripts/session-stats/analyze.py tools",
"stats:edits": "python3 scripts/session-stats/analyze.py edits",
"stats:followups": "python3 scripts/session-stats/analyze.py followups",
"stats:audit": "bun scripts/session-stats/audit.ts",
"test:py": "python3 -m pytest -x python/omp-rpc/tests && python3 -m pytest -x python/robomp/tests",
"robomp:install": "pip install -e 'python/robomp[dev]'",
"robomp:serve": "python3 -m robomp serve",
+18 -1
View File
@@ -1,6 +1,23 @@
# Changelog
## [Unreleased]
### Breaking Changes
- Removed `compaction/index.ts` re-export of snapcompact helpers, so snapcompact utilities are no longer available from the agent compaction barrel and should be imported from `@oh-my-pi/snapcompact`
- Removed the `convertToLlm` alias export from `compaction/messages` — it duplicated `defaultConvertToLlm` under a second name. Import `defaultConvertToLlm` (array form) or the new `convertMessageToLlm` (single-message form) instead
### Added
- Added `convertMessageToLlm()`: the single-message core transformer behind `defaultConvertToLlm()`. Embedders with app-specific message roles should handle their own roles and delegate every core role (`user`/`developer`/`assistant`/`toolResult`/`custom`/`hookMessage`/`branchSummary`/`compactionSummary`) to it instead of duplicating the conversion — a duplicated `compactionSummary` case is how snapcompact frames once silently dropped off provider requests
- Added `pruneSupersededToolResults()` and the opt-in `PruneConfig.supersedeKey` hook so harnesses can prune stale tool results superseded by a newer read of the same file; superseded results are pruned ahead of age-based victims during overflow pruning and replaced with a `[Superseded by a newer read of this file]` placeholder. Without the new config, `pruneToolOutputs()` behavior is unchanged.
- Added `readToolSupersedeKey()` implementing the read-tool path/selector grammar (selector-free reads supersede range reads of the same file; URL-scheme paths exempt). Pruning honors prompt-cache economics: per-turn prunes only fire when the post-candidate suffix is small or the cache is cold (idle gap).
- Added the `snapcompact` compaction strategy via `@oh-my-pi/snapcompact`: instead of an LLM summary, discarded history is printed onto dense bitmap frames and re-attached to the compaction summary message as image blocks. `CompactionSummaryMessage` gains an optional `images` field, `estimateTokens()` charges per attached frame, and frames persist under `preserveData.snapcompact` with an 8-frame middle-out eviction budget.
- Snapcompact frames are now rendered in a provider-aware shape (`SNAPCOMPACT_SHAPES` + `resolveSnapcompactShape(api)`), following the snapcompact 200k-token monolithic evals: Anthropic-family and unknown APIs get `8x8r-bw` (unscii-8 square cells, black ink, every line printed twice with the copy on a pale highlight band — read at F1 parity with raw text at ~2x lower cost and the most refusal-robust), Google gets `8x8r-sent` (sentence-hue ink, ~2.9x cheaper), and OpenAI gets `6x6u-sent` (unscii Lanczos-stretched to 6x6 cells — OpenAI bills a flat ~2.9k tokens per image, so frame count is the only cost lever) with `detail: "original"` on the frame images. `snapcompactCompact()` accepts `model`/`shape` options, frames persist their shape metadata, mixed-shape archives (provider switches, legacy 5x8 frames) are flagged in the reading instructions, and `snapcompactGeometry()`/`renderSnapcompactFrame()` now take a shape
### Fixed
- Fixed queued steering messages being drained into an externally aborted run: interrupting mid-tool execution (e.g. Enter with a pending steer) dequeued the steer into the dying run — it landed in history without a response and the post-abort resume saw an empty queue, so the agent stopped instead of continuing. Steering/follow-up/aside queue polls are now skipped once the run's abort signal fires, leaving the queue intact for `Agent.continue()`.
- Fixed `<read-files>` compaction lists recording the same file once per line-range/raw selector (`src/foo.ts:50-200`, `:raw`, `:1-50:raw`, …): read-tool selectors are now stripped before tracking, so reads dedupe to the base path and match their write/edit path when splitting read-only vs modified lists. Selector-polluted lists stored by earlier compactions self-heal on the next compaction. `readToolSupersedeKey()` now shares the same splitter (`splitReadSelector()`), gaining the `..` range alias and `L`-prefix forms it previously missed.
## [15.10.12] - 2026-06-10
@@ -664,4 +681,4 @@ Initial release under @oh-my-pi scope. See previous releases at [badlogic/pi-mon
- `Agent` constructor now has all options optional (empty options use defaults).
- `queueMessage()` is now synchronous (no longer returns a Promise).
- `queueMessage()` is now synchronous (no longer returns a Promise).
+1
View File
@@ -39,6 +39,7 @@
"@oh-my-pi/pi-catalog": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"@oh-my-pi/snapcompact": "catalog:",
"@opentelemetry/api": "catalog:"
},
"devDependencies": {
+20 -7
View File
@@ -564,8 +564,10 @@ async function runLoopBody(
streamFn?: StreamFn,
): Promise<void> {
let firstTurn = true;
// Check for steering messages at start (user may have typed while waiting)
let pendingMessages: AgentMessage[] = (await config.getSteeringMessages?.()) || [];
// Check for steering messages at start (user may have typed while waiting).
// Skip when the run is already externally aborted — dequeuing would strand
// the messages in a run that is about to die.
let pendingMessages: AgentMessage[] = signal?.aborted ? [] : (await config.getSteeringMessages?.()) || [];
let harmonyRetryAttempt = 0;
let harmonyTruncateResumeCount = 0;
@@ -743,7 +745,12 @@ async function runLoopBody(
stream.push({ type: "turn_end", message, toolResults });
const steering = steeringMessagesFromExecution ?? ((await config.getSteeringMessages?.()) || []);
// On external abort (user interrupt), leave the steering queue intact: the
// session aborts then continues, delivering the queue into a fresh run.
// Draining it here would inject the messages right before a model call that
// instantly aborts — message lands in history, agent never responds.
const steering =
steeringMessagesFromExecution ?? (signal?.aborted ? [] : (await config.getSteeringMessages?.()) || []);
if (hasMoreToolCalls) {
// Mid-work: fold any non-interrupting asides into the next turn alongside steering.
const asides = resolveAsides(await config.getAsideMessages?.());
@@ -758,8 +765,9 @@ async function runLoopBody(
// Agent would stop here. Drain non-interrupting asides + follow-up messages.
await config.onBeforeYield?.();
const asideMessages = resolveAsides(await config.getAsideMessages?.());
const followUpMessages = (await config.getFollowUpMessages?.()) || [];
// Skip queue drains when externally aborted (same stranding hazard as above).
const asideMessages = signal?.aborted ? [] : resolveAsides(await config.getAsideMessages?.());
const followUpMessages = signal?.aborted ? [] : (await config.getFollowUpMessages?.()) || [];
if (asideMessages.length > 0 || followUpMessages.length > 0) {
// Set as pending so the inner loop processes them before stopping.
pendingMessages = [...asideMessages, ...followUpMessages];
@@ -1253,11 +1261,16 @@ async function executeToolCalls(
}));
const checkSteering = async (): Promise<void> => {
if (!shouldInterruptImmediately || !getSteeringMessages || interruptState.triggered) {
// `signal` (external/user abort) is checked separately from the internal
// steeringAbortController: once the run is externally aborted it is
// unwinding, and draining the steering queue here would strand the
// messages in the dying run instead of leaving them for the post-abort
// continue (interruptAndFlushQueuedMessages → Agent.continue()).
if (!shouldInterruptImmediately || !getSteeringMessages || interruptState.triggered || signal?.aborted) {
return;
}
const check = steeringCheckTail.then(async () => {
if (interruptState.triggered) return;
if (interruptState.triggered || signal?.aborted) return;
const steering = await getSteeringMessages();
if (steering.length > 0) {
steeringMessages = steering;
@@ -13,10 +13,10 @@ import { estimateTokens } from "./compaction";
import type { ReadonlySessionManager, SessionEntry } from "./entries";
import {
type ConvertToLlm,
convertToLlm,
createBranchSummaryMessage,
createCompactionSummaryMessage,
createCustomMessage,
defaultConvertToLlm,
} from "./messages";
import branchSummaryPrompt from "./prompts/branch-summary.md" with { type: "text" };
import branchSummaryPreamble from "./prompts/branch-summary-preamble.md" with { type: "text" };
@@ -27,6 +27,7 @@ import {
type FileOperations,
SUMMARIZATION_SYSTEM_PROMPT,
serializeConversation,
stripReadSelector,
upsertFileOperations,
} from "./utils";
@@ -214,7 +215,7 @@ export function prepareBranchEntries(entries: SessionEntry[], tokenBudget: numbe
if (entry.type === "branch_summary" && !entry.fromExtension && entry.details) {
const details = entry.details as BranchSummaryDetails;
if (Array.isArray(details.readFiles)) {
for (const f of details.readFiles) fileOps.read.add(f);
for (const f of details.readFiles) fileOps.read.add(stripReadSelector(f));
}
if (Array.isArray(details.modifiedFiles)) {
// Modified files go into both edited and written for proper deduplication
@@ -288,7 +289,7 @@ export async function generateBranchSummary(
// Transform to LLM-compatible messages, then serialize to text
// Serialization prevents the model from treating it as a conversation to continue
const llmMessages = (options.convertToLlm ?? convertToLlm)(messages);
const llmMessages = (options.convertToLlm ?? defaultConvertToLlm)(messages);
const conversationText = serializeConversation(llmMessages);
// Build prompt
+14 -8
View File
@@ -18,11 +18,12 @@ import {
import { clampThinkingLevelForModel } from "@oh-my-pi/pi-catalog/model-thinking";
import { countTokens } from "@oh-my-pi/pi-natives";
import { logger, prompt } from "@oh-my-pi/pi-utils";
import { SNAPCOMPACT_FRAME_TOKEN_ESTIMATE } from "@oh-my-pi/snapcompact";
import { type AgentTelemetry, instrumentedCompleteSimple } from "../telemetry";
import { ThinkingLevel } from "../thinking";
import type { AgentMessage } from "../types";
import type { CompactionEntry, SessionEntry } from "./entries";
import { type ConvertToLlm, convertToLlm, createBranchSummaryMessage, createCustomMessage } from "./messages";
import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages";
import {
buildOpenAiNativeHistory,
getPreservedOpenAiRemoteCompactionData,
@@ -45,6 +46,7 @@ import {
type FileOperations,
SUMMARIZATION_SYSTEM_PROMPT,
serializeConversation,
stripReadSelector,
upsertFileOperations,
} from "./utils";
@@ -74,7 +76,7 @@ function extractFileOperations(
if (!prevCompaction.fromExtension && prevCompaction.details) {
const details = prevCompaction.details as CompactionDetails;
if (Array.isArray(details.readFiles)) {
for (const f of details.readFiles) fileOps.read.add(f);
for (const f of details.readFiles) fileOps.read.add(stripReadSelector(f));
}
if (Array.isArray(details.modifiedFiles)) {
for (const f of details.modifiedFiles) fileOps.edited.add(f);
@@ -137,7 +139,7 @@ export interface CompactionResult<T = unknown> {
export interface CompactionSettings {
enabled: boolean;
strategy?: "context-full" | "handoff" | "shake" | "off";
strategy?: "context-full" | "handoff" | "shake" | "snapcompact" | "off";
thresholdPercent?: number;
thresholdTokens?: number;
reserveTokens: number;
@@ -310,6 +312,10 @@ export function estimateTokens(message: AgentMessage): number {
case "branchSummary":
case "compactionSummary": {
fragments.push(message.summary);
if (message.role === "compactionSummary" && message.images) {
// Snapcompact frames render at ≥1568px; providers bill the downscaled cap.
extra += message.images.length * SNAPCOMPACT_FRAME_TOKEN_ESTIMATE;
}
break;
}
default:
@@ -625,7 +631,7 @@ export async function generateSummary(
// Serialize conversation to text so model doesn't try to continue it
// Convert to LLM messages first (handles custom app messages when caller provides a transformer).
const llmMessages = (options?.convertToLlm ?? convertToLlm)(currentMessages);
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(currentMessages);
const conversationText = serializeConversation(llmMessages);
// Build the prompt with conversation wrapped in tags
@@ -724,7 +730,7 @@ export async function generateHandoff(
options: HandoffOptions,
signal?: AbortSignal,
): Promise<string> {
const llmMessages = (options.convertToLlm ?? convertToLlm)(messages);
const llmMessages = (options.convertToLlm ?? defaultConvertToLlm)(messages);
const requestMessages: Message[] = [
...llmMessages,
{
@@ -773,7 +779,7 @@ async function generateShortSummary(
options?: SummaryOptions,
): Promise<string> {
const maxTokens = Math.min(512, Math.floor(0.2 * reserveTokens));
const llmMessages = (options?.convertToLlm ?? convertToLlm)(recentMessages);
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(recentMessages);
const conversationText = serializeConversation(llmMessages);
let promptText = `<conversation>\n${conversationText}\n</conversation>\n\n`;
@@ -1010,7 +1016,7 @@ export async function compact(
? previousRemoteCompaction.replacementHistory
: undefined;
const remoteHistory = buildOpenAiNativeHistory(
(summaryOptions.convertToLlm ?? convertToLlm)(remoteMessages),
(summaryOptions.convertToLlm ?? defaultConvertToLlm)(remoteMessages),
model,
previousReplacementHistory,
);
@@ -1127,7 +1133,7 @@ async function generateTurnPrefixSummary(
): Promise<string> {
const maxTokens = Math.floor(0.5 * reserveTokens); // Smaller budget for turn prefix
const llmMessages = (options?.convertToLlm ?? convertToLlm)(messages);
const llmMessages = (options?.convertToLlm ?? defaultConvertToLlm)(messages);
const conversationText = serializeConversation(llmMessages);
const promptText = `<conversation>\n${conversationText}\n</conversation>\n\n${TURN_PREFIX_SUMMARIZATION_PROMPT}`;
const summarizationMessages = [
+78 -64
View File
@@ -51,6 +51,8 @@ export interface CompactionSummaryMessage {
shortSummary?: string;
tokensBefore: number;
providerPayload?: ProviderPayload;
/** Snapcompact frames archived by this compaction; appended as image blocks after the summary text. */
images?: ImageContent[];
timestamp: number;
}
@@ -98,6 +100,7 @@ export function createCompactionSummaryMessage(
timestamp: string,
shortSummary?: string,
providerPayload?: ProviderPayload,
images?: ImageContent[],
): CompactionSummaryMessage {
return {
role: "compactionSummary",
@@ -105,6 +108,7 @@ export function createCompactionSummaryMessage(
shortSummary,
tokensBefore,
providerPayload,
images: images && images.length > 0 ? images : undefined,
timestamp: new Date(timestamp).getTime(),
};
}
@@ -137,6 +141,79 @@ function isCoreCompactionMessage(message: AgentMessage): message is AgentMessage
);
}
/**
* Transform a single core-domain agent message to its LLM form; `undefined`
* drops it from the provider request.
*
* Single source of truth for the core roles (user/developer/assistant/
* toolResult) and the compaction messages owned by this package. Embedders
* with their own app messages (e.g. the coding agent) handle their custom
* roles and delegate every core role here — duplicating these cases is how
* snapcompact frames once silently fell off the provider request.
*/
export function convertMessageToLlm(message: AgentMessage): Message | undefined {
if (isCoreCompactionMessage(message)) {
switch (message.role) {
case "custom":
case "hookMessage": {
const content =
typeof message.content === "string"
? [{ type: "text" as const, text: message.content }]
: message.content;
return {
role: "developer",
content,
attribution: message.attribution,
timestamp: message.timestamp,
};
}
case "branchSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderBranchSummaryContext(message.summary),
},
],
attribution: "agent",
timestamp: message.timestamp,
};
case "compactionSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderCompactionSummaryContext(message.summary),
},
...(message.images ?? []),
],
attribution: "agent",
providerPayload: message.providerPayload,
timestamp: message.timestamp,
};
}
}
switch (message.role) {
case "user":
return { ...message, attribution: message.attribution ?? "user" };
case "developer":
return { ...message, attribution: message.attribution ?? "agent" };
case "assistant":
return message as AssistantMessage;
case "toolResult":
return {
...message,
content: getPrunedToolResultContent(message as ToolResultMessage),
attribution: message.attribution ?? "agent",
};
default:
return undefined;
}
}
/**
* Default compaction-domain transformer.
*
@@ -145,68 +222,5 @@ function isCoreCompactionMessage(message: AgentMessage): message is AgentMessage
* core LLM roles and the compaction messages owned by this package.
*/
export function defaultConvertToLlm(messages: AgentMessage[]): Message[] {
return messages
.map((message): Message | undefined => {
if (isCoreCompactionMessage(message)) {
switch (message.role) {
case "custom":
case "hookMessage": {
const content =
typeof message.content === "string"
? [{ type: "text" as const, text: message.content }]
: message.content;
return {
role: "developer",
content,
attribution: message.attribution,
timestamp: message.timestamp,
};
}
case "branchSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderBranchSummaryContext(message.summary),
},
],
attribution: "agent",
timestamp: message.timestamp,
};
case "compactionSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderCompactionSummaryContext(message.summary),
},
],
attribution: "agent",
providerPayload: message.providerPayload,
timestamp: message.timestamp,
};
}
}
switch (message.role) {
case "user":
return { ...message, attribution: message.attribution ?? "user" };
case "developer":
return { ...message, attribution: message.attribution ?? "agent" };
case "assistant":
return message as AssistantMessage;
case "toolResult":
return {
...message,
content: getPrunedToolResultContent(message as ToolResultMessage),
attribution: message.attribution ?? "agent",
};
default:
return undefined;
}
})
.filter(message => message !== undefined);
return messages.map(convertMessageToLlm).filter(message => message !== undefined);
}
export const convertToLlm = defaultConvertToLlm;
@@ -0,0 +1,17 @@
Prior conversation history has been archived verbatim onto {{frameCount}} snapcompact frame{{#if multipleFrames}}s{{/if}} — the bitmap image{{#if multipleFrames}}s{{/if}} attached below{{#if multipleFrames}}, ordered oldest to newest{{/if}}.
Reading a frame: monospace {{fontCell}} pixel font on a white background, {{cols}} characters per row, {{rows}} text rows per frame; read left to right, top to bottom. Text flows continuously with no word wrap, so words may break across row ends. Whitespace runs (including newlines) were collapsed to single spaces. {{#if sentenceInk}}Ink color cycles through six colors, advancing at sentence boundaries — a color change marks a new sentence.{{else}}Glyphs are plain black ink.{{/if}}{{#if lineRepeated}} Every text line is printed twice in a row — first on the white background, then repeated on a pale yellow band. The copies are identical: read each line once and use the duplicate only to double-check hard glyphs.{{/if}} Roles are tagged inline as [User]:, [Assistant]:, [Assistant thinking]:, [Assistant tool calls]:, and [Tool result]:.
{{#if mixedShapes}}
Older frames may use a different font, grid, or ink coloring than described above; the reading order is always the same (left to right, top to bottom, oldest frame first).
{{/if}}
{{#if includedPreviousSummary}}
The earliest frame begins with "[Summary of earlier history]" — a condensed digest of context that predates the archived conversation.
{{/if}}
{{#if truncatedChars}}
{{truncatedChars}} characters of older history were dropped to respect the frame budget. The first frame (session start) is always kept, so the missing span sits between the first frame and the next.
{{/if}}
Total archived: {{totalChars}} characters. Consult the frames whenever you need exact earlier details (user wording, decisions, file paths, tool output). If a region is hard to read, re-derive the fact from the workspace (re-read files, re-run commands) rather than guessing.
+174 -8
View File
@@ -3,7 +3,7 @@
*/
import type { ToolResultMessage } from "@oh-my-pi/pi-ai";
import type { AgentMessage } from "../types";
import type { AgentMessage, AgentToolCall } from "../types";
import { estimateTokens } from "./compaction";
import type { SessionEntry, SessionMessageEntry } from "./entries";
import {
@@ -12,6 +12,7 @@ import {
isSkillReadToolResult,
type ProtectedToolMatcher,
} from "./tool-protection";
import { splitReadSelector } from "./utils";
export interface PruneConfig {
/** Keep the most recent tool output tokens intact. */
@@ -20,6 +21,13 @@ export interface PruneConfig {
minimumSavings: number;
/** Tool-result protection matchers. String entries protect every result from that tool; predicates may inspect the paired tool call. */
protectedTools: ProtectedToolMatcher[];
/**
* Optional supersede key function (see {@link SupersedePruneConfig.supersedeKey}).
* When provided, superseded tool results are pruned first — even inside the
* `protectTokens` window — before age-based victims. Absent, behavior is
* unchanged.
*/
supersedeKey?: SupersedeKeyFn;
}
export const DEFAULT_PRUNE_CONFIG: PruneConfig = {
@@ -33,6 +41,34 @@ export interface PruneResult {
tokensSaved: number;
}
/** Exact placeholder written over a superseded tool result. */
export const SUPERSEDED_NOTICE = "[Superseded by a newer read of this file]";
/**
* Maps a tool call to a supersede key. Results sharing a key form a group in
* which every result except the newest is a supersede candidate. A key `K`
* additionally supersedes keys with prefix `K + "\u0000"` (selector-free read
* supersedes selector-carrying reads of the same base path). Return
* `undefined` to exempt a call from supersede grouping.
*/
export type SupersedeKeyFn = (toolName: string, args: Record<string, unknown>) => string | undefined;
export interface SupersedePruneConfig {
/** Supersede key function; results sharing a key supersede older ones. */
supersedeKey: SupersedeKeyFn;
/** Prune a candidate now when all messages after it total at most this many estimated tokens. Default 8 000. */
suffixTokenLimit?: number;
/** Prune all candidates when the last message is at least this old (prompt cache is cold anyway). Default 30 min. */
idleFlushMs?: number;
/** Clock override for tests. */
now?: number;
/** Tool-result protection matchers (same contract as {@link PruneConfig.protectedTools}). */
protectedTools: ProtectedToolMatcher[];
}
const DEFAULT_SUFFIX_TOKEN_LIMIT = 8_000;
const DEFAULT_IDLE_FLUSH_MS = 30 * 60_000;
function createPrunedNotice(tokens: number): string {
return `[Output truncated - ${tokens} tokens]`;
}
@@ -44,18 +80,121 @@ function getToolResultMessage(entry: SessionEntry): ToolResultMessage | undefine
return message as ToolResultMessage;
}
function estimatePrunedSavings(tokens: number): number {
const noticeTokens = Math.ceil(createPrunedNotice(tokens).length / 4);
function estimatePrunedSavings(tokens: number, notice: string): number {
const noticeTokens = Math.ceil(notice.length / 4);
return Math.max(0, tokens - noticeTokens);
}
interface SupersedeCandidate {
entry: SessionMessageEntry;
message: ToolResultMessage;
/** Index of the entry within the `entries` array. */
index: number;
tokens: number;
}
/**
* Collect superseded tool results: for every unpruned, unprotected tool result
* whose paired call resolves a supersede key, a LATER result with the same key
* — or with a key that is the `"\u0000"`-prefix parent of this one — marks it
* superseded. Returned in message order.
*/
function collectSupersededResults(
entries: readonly SessionEntry[],
toolCallsById: ReadonlyMap<string, AgentToolCall>,
supersedeKey: SupersedeKeyFn,
protectedTools: readonly ProtectedToolMatcher[],
): SupersedeCandidate[] {
const candidates: SupersedeCandidate[] = [];
const seenKeys = new Set<string>();
for (let i = entries.length - 1; i >= 0; i--) {
const entry = entries[i];
const message = getToolResultMessage(entry);
if (!message || message.prunedAt !== undefined) continue;
const toolCall = toolCallsById.get(message.toolCallId);
if (!toolCall) continue;
if (isProtectedToolResult(message, toolCall, protectedTools)) continue;
const key = supersedeKey(toolCall.name, toolCall.arguments as Record<string, unknown>);
if (key === undefined) continue;
const separator = key.indexOf("\u0000");
const superseded = seenKeys.has(key) || (separator >= 0 && seenKeys.has(key.slice(0, separator)));
seenKeys.add(key);
if (!superseded) continue;
candidates.push({
entry: entry as SessionMessageEntry,
message,
index: i,
tokens: estimateTokens(message as AgentMessage),
});
}
return candidates.reverse();
}
/**
* Prune superseded tool results (e.g. stale `read` outputs replaced by a newer
* read of the same file). Cheap, incremental, and prompt-cache-aware: a
* candidate is pruned now only when the suffix after it is small (tail case —
* the read→edit→read loop) or when the context has been idle long enough that
* the provider cache is cold anyway (then ALL candidates flush).
*/
export function pruneSupersededToolResults(entries: SessionEntry[], config: SupersedePruneConfig): PruneResult {
const toolCallsById = collectToolCallsById(entries);
const candidates = collectSupersededResults(entries, toolCallsById, config.supersedeKey, config.protectedTools);
if (candidates.length === 0) return { prunedCount: 0, tokensSaved: 0 };
const now = config.now ?? Date.now();
let lastMessageTimestamp: number | undefined;
for (let i = entries.length - 1; i >= 0; i--) {
const entry = entries[i];
if (entry.type !== "message") continue;
const timestamp = (entry.message as AgentMessage).timestamp;
if (typeof timestamp === "number") lastMessageTimestamp = timestamp;
break;
}
const idle =
lastMessageTimestamp !== undefined && now - lastMessageTimestamp >= (config.idleFlushMs ?? DEFAULT_IDLE_FLUSH_MS);
let toPrune: SupersedeCandidate[];
if (idle) {
toPrune = candidates;
} else {
const suffixTokenLimit = config.suffixTokenLimit ?? DEFAULT_SUFFIX_TOKEN_LIMIT;
// suffixTokens[i] = estimated tokens of all messages strictly after entry i.
const suffixTokens = new Array<number>(entries.length);
let accumulated = 0;
for (let i = entries.length - 1; i >= 0; i--) {
suffixTokens[i] = accumulated;
const entry = entries[i];
if (entry.type === "message") accumulated += estimateTokens(entry.message as AgentMessage);
}
toPrune = candidates.filter(candidate => suffixTokens[candidate.index] <= suffixTokenLimit);
}
if (toPrune.length === 0) return { prunedCount: 0, tokensSaved: 0 };
const prunedAt = Date.now();
let tokensSaved = 0;
for (const candidate of toPrune) {
candidate.message.content = [{ type: "text", text: SUPERSEDED_NOTICE }];
candidate.message.prunedAt = prunedAt;
tokensSaved += estimatePrunedSavings(candidate.tokens, SUPERSEDED_NOTICE);
}
return { prunedCount: toPrune.length, tokensSaved };
}
export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig = DEFAULT_PRUNE_CONFIG): PruneResult {
let accumulatedTokens = 0;
let tokensSaved = 0;
let prunedCount = 0;
const candidates: Array<{ entry: SessionMessageEntry; tokens: number }> = [];
const candidates: Array<{ entry: SessionMessageEntry; tokens: number; superseded: boolean }> = [];
const toolCallsById = collectToolCallsById(entries);
const supersededMessages = config.supersedeKey
? new Set(
collectSupersededResults(entries, toolCallsById, config.supersedeKey, config.protectedTools).map(
candidate => candidate.message,
),
)
: undefined;
for (let i = entries.length - 1; i >= 0; i--) {
const entry = entries[i];
@@ -70,17 +209,23 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
continue;
}
if (accumulatedTokens < config.protectTokens || isProtected) {
// Superseded results are pruned first: they bypass the protect window
// (a stale copy of re-read content is dead weight at any age).
const superseded = supersededMessages?.has(message) ?? false;
if (!superseded && (accumulatedTokens < config.protectTokens || isProtected)) {
accumulatedTokens += tokens;
continue;
}
candidates.push({ entry: entry as SessionMessageEntry, tokens });
candidates.push({ entry: entry as SessionMessageEntry, tokens, superseded });
accumulatedTokens += tokens;
}
for (const candidate of candidates) {
tokensSaved += estimatePrunedSavings(candidate.tokens);
tokensSaved += estimatePrunedSavings(
candidate.tokens,
candidate.superseded ? SUPERSEDED_NOTICE : createPrunedNotice(candidate.tokens),
);
}
if (tokensSaved < config.minimumSavings || candidates.length === 0) {
@@ -90,10 +235,31 @@ export function pruneToolOutputs(entries: SessionEntry[], config: PruneConfig =
const prunedAt = Date.now();
for (const candidate of candidates) {
const message = candidate.entry.message as ToolResultMessage;
message.content = [{ type: "text", text: createPrunedNotice(candidate.tokens) }];
message.content = [
{ type: "text", text: candidate.superseded ? SUPERSEDED_NOTICE : createPrunedNotice(candidate.tokens) },
];
message.prunedAt = prunedAt;
prunedCount++;
}
return { prunedCount, tokensSaved };
}
/**
* Supersede key for the `read` tool: the file path with the trailing line/raw
* selector stripped (the read tool's own splitter grammar via
* {@link splitReadSelector}, e.g. `src/foo.ts:50-200`, `:2-4:raw`).
* Internal/URL-scheme paths (`skill://…`, `https://…`) are exempt.
* Selector-free reads key on the bare path; selector-carrying reads key on
* `path + "\u0000" + selector`, so two reads collide only when the newer is
* selector-free or the selectors are identical (the pass's prefix rule lets a
* bare-path read supersede selector-carrying reads of the same file).
*/
export function readToolSupersedeKey(toolName: string, args: Record<string, unknown>): string | undefined {
if (toolName !== "read") return undefined;
const path = args.path;
if (typeof path !== "string" || path.length === 0) return undefined;
if (path.includes("://")) return undefined;
const { path: base, sel } = splitReadSelector(path);
return sel === undefined ? base : `${base}\u0000${sel}`;
}
+50 -1
View File
@@ -26,6 +26,55 @@ export function createFileOps(): FileOperations {
};
}
// Read-tool selector grammar, mirrored from the conservative filesystem splitter in
// packages/coding-agent/src/tools/path-utils.ts (splitPathAndSel). Keep in sync.
// A trailing `:chunk` is a selector only when it is a line-range list
// (`50`, `50-200`, `50+10`, `5-16,960-973`, `..` alias), `raw`, or `conflicts` —
// alone or as a `range:raw` / `raw:range` compound.
const RANGE_CHUNK_SRC = String.raw`L?\d+(?:(?:[-+]|\.\.)L?\d+|-|\.\.)?`;
const RANGE_LIST_SRC = `${RANGE_CHUNK_SRC}(?:,${RANGE_CHUNK_SRC})*`;
const READ_SELECTOR_RE = new RegExp(`^(?:${RANGE_LIST_SRC}|raw|conflicts)$`, "i");
const READ_RANGE_ONLY_RE = new RegExp(`^${RANGE_LIST_SRC}$`, "i");
const READ_RAW_ONLY_RE = /^raw$/i;
/**
* Split a read-tool path into its base path and trailing selector, mirroring the
* read tool's own splitter. Single source of the grammar in this package: the
* file-operations list strips selectors via {@link stripReadSelector}, and the
* supersede-prune pass keys on both parts via `readToolSupersedeKey`.
*/
export function splitReadSelector(path: string): { path: string; sel?: string } {
const colon = path.lastIndexOf(":");
if (colon <= 0) return { path };
const candidate = path.slice(colon + 1);
if (!READ_SELECTOR_RE.test(candidate)) return { path };
let base = path.slice(0, colon);
let sel = candidate;
// Compound trailing selector: `path:1-50:raw` or `path:raw:1-50`.
const inner = base.lastIndexOf(":");
if (inner > 0) {
const innerCandidate = base.slice(inner + 1);
const innerIsRaw = READ_RAW_ONLY_RE.test(innerCandidate);
const outerIsRaw = READ_RAW_ONLY_RE.test(candidate);
const innerIsRange = READ_RANGE_ONLY_RE.test(innerCandidate);
const outerIsRange = READ_RANGE_ONLY_RE.test(candidate);
if ((innerIsRaw && outerIsRange) || (innerIsRange && outerIsRaw)) {
sel = `${innerCandidate}:${candidate}`;
base = base.slice(0, inner);
}
}
return { path: base, sel };
}
/**
* Strip a trailing read-tool selector (`:50-200`, `:raw`, `:1-50:raw`, `:conflicts`, …)
* so the same file read with different line ranges dedupes to one `<read-files>` entry
* and matches its write/edit path when computing read-only vs modified lists.
*/
export function stripReadSelector(path: string): string {
return splitReadSelector(path).path;
}
/**
* Extract file operations from tool calls in an assistant message.
*/
@@ -46,7 +95,7 @@ export function extractFileOpsFromMessage(message: AgentMessage, fileOps: FileOp
switch (block.name) {
case "read":
fileOps.read.add(path);
fileOps.read.add(stripReadSelector(path));
break;
case "write":
fileOps.written.add(path);
@@ -0,0 +1,76 @@
import { describe, expect, it } from "bun:test";
import {
computeFileLists,
createFileOps,
extractFileOpsFromMessage,
formatFileOperations,
stripReadSelector,
} from "../src/compaction/utils";
import { createAssistantMessage } from "./helpers";
function readCall(id: string, path: string) {
return { type: "toolCall" as const, id, name: "read", arguments: { path } };
}
describe("stripReadSelector", () => {
it("strips line-range and raw selectors in every supported shape", () => {
expect(stripReadSelector("src/foo.ts:50")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:50-")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:50-200")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:50+150")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:5-16,960-973")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:2724..2727")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:raw")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:conflicts")).toBe("src/foo.ts");
// Compound raw+range, either order.
expect(stripReadSelector("src/foo.ts:100-170:raw")).toBe("src/foo.ts");
expect(stripReadSelector("src/foo.ts:raw:2-4")).toBe("src/foo.ts");
});
it("keeps archive member paths, stripping only the trailing selector", () => {
expect(stripReadSelector("archive.zip:dir/file.ts:50-60")).toBe("archive.zip:dir/file.ts");
expect(stripReadSelector("archive.zip:dir/file.ts")).toBe("archive.zip:dir/file.ts");
});
it("leaves non-selector colons untouched", () => {
expect(stripReadSelector("db.sqlite:users")).toBe("db.sqlite:users");
expect(stripReadSelector("local://ctx.md")).toBe("local://ctx.md");
expect(stripReadSelector("https://example.com/page")).toBe("https://example.com/page");
expect(stripReadSelector("src/foo.ts")).toBe("src/foo.ts");
});
});
describe("extractFileOpsFromMessage", () => {
it("dedupes the same file read through different selectors to one entry", () => {
const fileOps = createFileOps();
const message = createAssistantMessage([
readCall("r1", "docs/compaction.md:100-170:raw"),
readCall("r2", "docs/compaction.md:8-16,128-139,384-388"),
readCall("r3", "docs/compaction.md:raw"),
readCall("r4", "docs/compaction.md"),
]);
extractFileOpsFromMessage(message, fileOps);
expect([...fileOps.read]).toEqual(["docs/compaction.md"]);
});
it("matches selector-suffixed reads against modified paths", () => {
const fileOps = createFileOps();
const message = createAssistantMessage([
readCall("r1", "src/login.ts:30-80"),
{ type: "toolCall" as const, id: "w1", name: "write", arguments: { path: "src/login.ts" } },
]);
extractFileOpsFromMessage(message, fileOps);
const { readFiles, modifiedFiles } = computeFileLists(fileOps);
expect(readFiles).toEqual([]);
expect(modifiedFiles).toEqual(["src/login.ts"]);
});
});
describe("formatFileOperations", () => {
it("renders one path per line, not literal \\n separators", () => {
const rendered = formatFileOperations(["a.ts", "b.ts"], ["c.ts"]);
expect(rendered).toContain("<read-files>\na.ts\nb.ts\n</read-files>");
expect(rendered).toContain("<modified-files>\nc.ts\n</modified-files>");
expect(rendered).not.toContain("\\n");
});
});
@@ -0,0 +1,44 @@
import { describe, expect, it } from "bun:test";
import type { ImageContent } from "@oh-my-pi/pi-ai";
import { SNAPCOMPACT_FRAME_TOKEN_ESTIMATE } from "@oh-my-pi/snapcompact";
import { estimateTokens } from "../src/compaction/compaction";
import { createCompactionSummaryMessage, defaultConvertToLlm } from "../src/compaction/messages";
describe("compaction summary message with snapcompact frames", () => {
const images: ImageContent[] = [
{ type: "image", data: "ZmFrZQ==", mimeType: "image/png" },
{ type: "image", data: "ZmFrZTI=", mimeType: "image/png" },
];
it("estimateTokens charges per attached frame", () => {
const bare = createCompactionSummaryMessage("summary text", 1000, new Date().toISOString());
const withFrames = createCompactionSummaryMessage(
"summary text",
1000,
new Date().toISOString(),
undefined,
undefined,
images,
);
expect(estimateTokens(withFrames) - estimateTokens(bare)).toBe(2 * SNAPCOMPACT_FRAME_TOKEN_ESTIMATE);
});
it("defaultConvertToLlm appends frames as image blocks after the summary text", () => {
const message = createCompactionSummaryMessage(
"the snapcompact archive",
1000,
new Date().toISOString(),
undefined,
undefined,
images,
);
const [converted] = defaultConvertToLlm([message]);
expect(converted.role).toBe("user");
const content = converted.content as Array<{ type: string; text?: string; data?: string }>;
expect(content.length).toBe(3);
expect(content[0].type).toBe("text");
expect(content[0].text).toContain("the snapcompact archive");
expect(content[1]).toEqual(images[0]);
expect(content[2]).toEqual(images[1]);
});
});
+346
View File
@@ -0,0 +1,346 @@
import { describe, expect, test } from "bun:test";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import type { SessionEntry, SessionMessageEntry } from "@oh-my-pi/pi-agent-core/compaction";
import {
DEFAULT_PRUNE_CONFIG,
pruneSupersededToolResults,
pruneToolOutputs,
readToolSupersedeKey,
SUPERSEDED_NOTICE,
type SupersedePruneConfig,
} from "@oh-my-pi/pi-agent-core/compaction";
import type { ProtectedToolContext } from "@oh-my-pi/pi-agent-core/compaction/tool-protection";
import type { AssistantMessage, TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai";
let idCounter = 0;
function nextId(): string {
return `entry-${idCounter++}`;
}
function messageEntry(message: AgentMessage, timestamp: number): SessionMessageEntry {
return { type: "message", id: nextId(), parentId: null, timestamp: new Date(timestamp).toISOString(), message };
}
function assistantMessage(content: AssistantMessage["content"], timestamp: number): AssistantMessage {
return {
role: "assistant",
content,
timestamp,
provider: "mock",
model: "mock",
api: "mock",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "stop",
};
}
function toolResultMessage(toolName: string, toolCallId: string, text: string, timestamp: number): ToolResultMessage {
return {
role: "toolResult",
toolCallId,
toolName,
content: [{ type: "text", text }],
isError: false,
timestamp,
};
}
/** Assistant toolCall entry + paired toolResult entry for one read. */
function readPair(path: string, text: string, timestamp: number): [SessionMessageEntry, SessionMessageEntry] {
const callId = `call-${idCounter++}`;
return [
messageEntry(
assistantMessage([{ type: "toolCall", id: callId, name: "read", arguments: { path } }], timestamp),
timestamp,
),
messageEntry(toolResultMessage("read", callId, text, timestamp), timestamp),
];
}
function textEntry(text: string, timestamp: number): SessionMessageEntry {
return messageEntry(assistantMessage([{ type: "text", text }], timestamp), timestamp);
}
function resultText(entry: SessionEntry): string {
const message = (entry as SessionMessageEntry).message as ToolResultMessage;
return (message.content[0] as TextContent).text;
}
function resultMessage(entry: SessionEntry): ToolResultMessage {
return (entry as SessionMessageEntry).message as ToolResultMessage;
}
function cfg(over: Partial<SupersedePruneConfig> = {}): SupersedePruneConfig {
return { supersedeKey: readToolSupersedeKey, protectedTools: [], ...over };
}
const T0 = Date.UTC(2026, 5, 10, 12, 0, 0);
const FILE_CONTENT = "export function alpha() { return 1; }\n".repeat(50);
// Comfortably above any small suffixTokenLimit used below.
const BIG_TEXT = "const value = computeSomething(12345);\n".repeat(500);
describe("readToolSupersedeKey", () => {
test("bare path keys on itself; non-read and non-string paths are exempt", () => {
expect(readToolSupersedeKey("read", { path: "src/foo.ts" })).toBe("src/foo.ts");
expect(readToolSupersedeKey("bash", { path: "src/foo.ts" })).toBeUndefined();
expect(readToolSupersedeKey("read", { path: 42 })).toBeUndefined();
expect(readToolSupersedeKey("read", {})).toBeUndefined();
});
test("URL/internal schemes are exempt", () => {
expect(readToolSupersedeKey("read", { path: "skill://react" })).toBeUndefined();
expect(readToolSupersedeKey("read", { path: "https://example.com/page" })).toBeUndefined();
});
test("strips trailing selectors into a \\u0000-separated key", () => {
expect(readToolSupersedeKey("read", { path: "src/foo.ts:50-200" })).toBe("src/foo.ts\u000050-200");
expect(readToolSupersedeKey("read", { path: "src/foo.ts:raw" })).toBe("src/foo.ts\u0000raw");
expect(readToolSupersedeKey("read", { path: "src/foo.ts:conflicts" })).toBe("src/foo.ts\u0000conflicts");
expect(readToolSupersedeKey("read", { path: "src/foo.ts:2-4:raw" })).toBe("src/foo.ts\u00002-4:raw");
expect(readToolSupersedeKey("read", { path: "src/foo.ts:5-16,960-973" })).toBe("src/foo.ts\u00005-16,960-973");
expect(readToolSupersedeKey("read", { path: "src/foo.ts:50+150" })).toBe("src/foo.ts\u000050+150");
});
test("does not strip non-selector colon segments", () => {
expect(readToolSupersedeKey("read", { path: "db.sqlite:users" })).toBe("db.sqlite:users");
expect(readToolSupersedeKey("read", { path: "db.sqlite:users:42" })).toBe("db.sqlite:users\u000042");
});
});
describe("pruneSupersededToolResults — tail case", () => {
test("(a) older identical-path read pruned with exact placeholder when suffix small", () => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
const entries: SessionEntry[] = [call1, result1, call2, result2];
const result = pruneSupersededToolResults(entries, cfg({ now: T0 + 1_000 }));
expect(result.prunedCount).toBe(1);
expect(result.tokensSaved).toBeGreaterThan(0);
expect(resultText(result1)).toBe("[Superseded by a newer read of this file]");
expect(resultText(result1)).toBe(SUPERSEDED_NOTICE);
expect(resultMessage(result1).prunedAt).toBeDefined();
// Latest read untouched.
expect(resultText(result2)).toBe(FILE_CONTENT);
expect(resultMessage(result2).prunedAt).toBeUndefined();
});
test("(b) NOT pruned when suffix exceeds limit and no idle gap", () => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
const big = textEntry(BIG_TEXT, T0 + 2_000);
const entries: SessionEntry[] = [call1, result1, call2, result2, big];
const result = pruneSupersededToolResults(entries, cfg({ suffixTokenLimit: 200, now: T0 + 2_000 }));
expect(result.prunedCount).toBe(0);
expect(result.tokensSaved).toBe(0);
expect(resultText(result1)).toBe(FILE_CONTENT);
expect(resultMessage(result1).prunedAt).toBeUndefined();
expect(resultText(result2)).toBe(FILE_CONTENT);
});
test("(c) idle gap prunes all candidates regardless of suffix", () => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/bar.ts", FILE_CONTENT, T0 + 1_000);
const [call3, result3] = readPair("src/foo.ts", FILE_CONTENT, T0 + 2_000);
const [call4, result4] = readPair("src/bar.ts", FILE_CONTENT, T0 + 3_000);
const big = textEntry(BIG_TEXT, T0 + 4_000);
const entries: SessionEntry[] = [call1, result1, call2, result2, call3, result3, call4, result4, big];
// Suffix limit 0 would block every candidate; only the idle gap fires.
const result = pruneSupersededToolResults(
entries,
cfg({ suffixTokenLimit: 0, idleFlushMs: 30 * 60_000, now: T0 + 4_000 + 30 * 60_000 }),
);
expect(result.prunedCount).toBe(2);
expect(resultText(result1)).toBe(SUPERSEDED_NOTICE);
expect(resultText(result2)).toBe(SUPERSEDED_NOTICE);
expect(resultText(result3)).toBe(FILE_CONTENT);
expect(resultText(result4)).toBe(FILE_CONTENT);
});
test("no idle flush when gap is below the threshold", () => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
const big = textEntry(BIG_TEXT, T0 + 2_000);
const entries: SessionEntry[] = [call1, result1, call2, result2, big];
const result = pruneSupersededToolResults(
entries,
cfg({ suffixTokenLimit: 0, idleFlushMs: 30 * 60_000, now: T0 + 2_000 + 29 * 60_000 }),
);
expect(result.prunedCount).toBe(0);
expect(resultText(result1)).toBe(FILE_CONTENT);
expect(resultText(result2)).toBe(FILE_CONTENT);
});
});
describe("pruneSupersededToolResults — selectors", () => {
test("(d) different range selectors do not supersede each other; a later selector-free read supersedes them", () => {
const [callA, resultA] = readPair("src/foo.ts:50-200", FILE_CONTENT, T0);
const [callB, resultB] = readPair("src/foo.ts:10-20", FILE_CONTENT, T0 + 1_000);
let entries: SessionEntry[] = [callA, resultA, callB, resultB];
// Different selectors: no candidates.
let result = pruneSupersededToolResults(entries, cfg({ now: T0 + 1_000 }));
expect(result.prunedCount).toBe(0);
expect(resultText(resultA)).toBe(FILE_CONTENT);
expect(resultText(resultB)).toBe(FILE_CONTENT);
// Identical selector strings DO supersede.
const [callA2, resultA2] = readPair("src/foo.ts:50-200", FILE_CONTENT, T0 + 2_000);
entries = [...entries, callA2, resultA2];
result = pruneSupersededToolResults(entries, cfg({ now: T0 + 2_000 }));
expect(result.prunedCount).toBe(1);
expect(resultText(resultA)).toBe(SUPERSEDED_NOTICE);
expect(resultText(resultB)).toBe(FILE_CONTENT);
expect(resultText(resultA2)).toBe(FILE_CONTENT);
// A later selector-free read supersedes every selector-carrying read of the base path.
const [callFull, resultFull] = readPair("src/foo.ts", FILE_CONTENT, T0 + 3_000);
entries = [...entries, callFull, resultFull];
result = pruneSupersededToolResults(entries, cfg({ now: T0 + 3_000 }));
expect(result.prunedCount).toBe(2);
expect(resultText(resultB)).toBe(SUPERSEDED_NOTICE);
expect(resultText(resultA2)).toBe(SUPERSEDED_NOTICE);
expect(resultText(resultFull)).toBe(FILE_CONTENT);
});
test("a selector-carrying read does NOT supersede an earlier selector-free read", () => {
const [callFull, resultFull] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [callRange, resultRange] = readPair("src/foo.ts:50-200", FILE_CONTENT, T0 + 1_000);
const entries: SessionEntry[] = [callFull, resultFull, callRange, resultRange];
const result = pruneSupersededToolResults(entries, cfg({ now: T0 + 1_000 }));
expect(result.prunedCount).toBe(0);
expect(resultText(resultFull)).toBe(FILE_CONTENT);
expect(resultText(resultRange)).toBe(FILE_CONTENT);
});
});
describe("pruneSupersededToolResults — protection & latest", () => {
test("(e) latest read never pruned, even with idle flush", () => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
const [call3, result3] = readPair("src/foo.ts", FILE_CONTENT, T0 + 2_000);
const entries: SessionEntry[] = [call1, result1, call2, result2, call3, result3];
const result = pruneSupersededToolResults(entries, cfg({ now: T0 + 2_000 + 60 * 60_000 }));
expect(result.prunedCount).toBe(2);
expect(resultText(result1)).toBe(SUPERSEDED_NOTICE);
expect(resultText(result2)).toBe(SUPERSEDED_NOTICE);
expect(resultText(result3)).toBe(FILE_CONTENT);
expect(resultMessage(result3).prunedAt).toBeUndefined();
});
test("(f) protected tool results never pruned", () => {
const protectPlan = ({ toolCall }: ProtectedToolContext): boolean =>
(toolCall?.arguments as Record<string, unknown> | undefined)?.path === "plan.md";
const [planCall1, planResult1] = readPair("plan.md", FILE_CONTENT, T0);
const [fooCall1, fooResult1] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
const [planCall2, planResult2] = readPair("plan.md", FILE_CONTENT, T0 + 2_000);
const [fooCall2, fooResult2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 3_000);
const entries: SessionEntry[] = [
planCall1,
planResult1,
fooCall1,
fooResult1,
planCall2,
planResult2,
fooCall2,
fooResult2,
];
const result = pruneSupersededToolResults(entries, cfg({ protectedTools: [protectPlan], now: T0 + 3_000 }));
expect(result.prunedCount).toBe(1);
expect(resultText(planResult1)).toBe(FILE_CONTENT);
expect(resultText(planResult2)).toBe(FILE_CONTENT);
expect(resultText(fooResult1)).toBe(SUPERSEDED_NOTICE);
expect(resultText(fooResult2)).toBe(FILE_CONTENT);
});
test("already-pruned results are ignored as candidates and as superseders", () => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
resultMessage(result2).prunedAt = T0 + 1_500;
const entries: SessionEntry[] = [call1, result1, call2, result2];
// The only newer same-key read is itself pruned -> result1 has no live superseder.
const result = pruneSupersededToolResults(entries, cfg({ now: T0 + 2_000 }));
expect(result.prunedCount).toBe(0);
expect(resultText(result1)).toBe(FILE_CONTENT);
});
});
describe("pruneToolOutputs — supersede priority fold", () => {
test("with supersedeKey, superseded results bypass the protect window and get the supersede placeholder", () => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
const entries: SessionEntry[] = [call1, result1, call2, result2];
const result = pruneToolOutputs(entries, {
protectTokens: 1_000_000, // everything inside the protect window
minimumSavings: 0,
protectedTools: [],
supersedeKey: readToolSupersedeKey,
});
expect(result.prunedCount).toBe(1);
expect(resultText(result1)).toBe(SUPERSEDED_NOTICE);
expect(resultText(result2)).toBe(FILE_CONTENT);
});
test("(g) without supersedeKey, behavior is unchanged (regression guard)", () => {
const buildEntries = (): {
entries: SessionEntry[];
oldResult: SessionMessageEntry;
newResult: SessionMessageEntry;
} => {
const [call1, result1] = readPair("src/foo.ts", FILE_CONTENT, T0);
const [call2, result2] = readPair("src/foo.ts", FILE_CONTENT, T0 + 1_000);
return { entries: [call1, result1, call2, result2], oldResult: result1, newResult: result2 };
};
// Protect window covers everything: nothing pruned, superseded reads included.
const protectedFixture = buildEntries();
const protectedRun = pruneToolOutputs(protectedFixture.entries, {
protectTokens: 1_000_000,
minimumSavings: 0,
protectedTools: [],
});
expect(protectedRun).toEqual({ prunedCount: 0, tokensSaved: 0 });
expect(resultText(protectedFixture.oldResult)).toBe(FILE_CONTENT);
expect(resultText(protectedFixture.newResult)).toBe(FILE_CONTENT);
// Protect window empty: every result past it pruned with the legacy
// truncation placeholder — never the supersede placeholder.
const unprotectedFixture = buildEntries();
const unprotectedRun = pruneToolOutputs(unprotectedFixture.entries, {
protectTokens: 0,
minimumSavings: 0,
protectedTools: [],
});
expect(unprotectedRun.prunedCount).toBe(2);
expect(resultText(unprotectedFixture.oldResult)).toMatch(/^\[Output truncated - \d+ tokens\]$/);
expect(resultText(unprotectedFixture.newResult)).toMatch(/^\[Output truncated - \d+ tokens\]$/);
// Default config shape is untouched.
expect(DEFAULT_PRUNE_CONFIG.supersedeKey).toBeUndefined();
expect(DEFAULT_PRUNE_CONFIG.protectTokens).toBe(40_000);
expect(DEFAULT_PRUNE_CONFIG.minimumSavings).toBe(20_000);
});
});
+8
View File
@@ -2,6 +2,14 @@
## [Unreleased]
### Added
- Added optional `ImageContent.detail` (`"auto" | "low" | "high" | "original"`): an OpenAI resolution hint forwarded by the `openai-responses` serializers (default stays `auto`) and by `openai-completions` for the values Chat Completions supports. `"original"` preserves native resolution — required for snapcompact frames, whose pixel-font glyphs do not survive the default downscale. Providers without a detail knob ignore the field.
### Fixed
- Cross-model `anthropic-messages → anthropic-messages` continuations now preserve prior assistant turns' reasoning chains end-to-end: every prior `thinking`/`redactedThinking` block survives (not just the latest surviving assistant), and third-party ↔ third-party replays keep their signatures intact so the reasoning chain stays signed for the next turn. Signatures are stripped (and any `redacted_thinking` sibling without a native landing spot is dropped) only when an official Anthropic endpoint is on either end of the replay — official Anthropic cryptographically binds reasoning signatures to its key+session+model, while compatible reasoning endpoints (Z.AI, DeepSeek, custom anthropic-messages providers configured via `models.yaml`) treat them as opaque continuation hints. Source-side official detection uses the canonical catalog provider id `"anthropic"` (assistant messages carry no `baseUrl`); target-side detection reuses the baked `compat.officialEndpoint` flag. Latest-turn byte-for-byte behavior (Anthropic's "thinking blocks in the latest assistant message cannot be modified" rule) and existing aborted/errored last-block sanitization are unchanged. ([#2257](https://github.com/can1357/oh-my-pi/issues/2257), [#2265](https://github.com/can1357/oh-my-pi/issues/2265))
## [15.10.12] - 2026-06-10
### Added
@@ -1651,6 +1651,8 @@ export function convertMessages(
type: "image_url",
image_url: {
url: `data:${item.mimeType};base64,${item.data}`,
// Chat Completions has no "original"; omit it (provider default).
...(item.detail && item.detail !== "original" ? { detail: item.detail } : {}),
},
} satisfies ChatCompletionContentPartImage);
} else {
@@ -289,7 +289,7 @@ export function convertResponsesInputContent(
for (const item of imageBlocks) {
normalizedContent.push({
type: "input_image",
detail: "auto",
detail: item.detail ?? "auto",
image_url: `data:${item.mimeType};base64,${item.data}`,
} satisfies ResponseInputImage);
}
@@ -448,7 +448,7 @@ export function appendResponsesToolResultMessages<TApi extends Api>(
if (block.type === "image") {
contentParts.push({
type: "input_image",
detail: "auto",
detail: block.detail ?? "auto",
image_url: `data:${block.mimeType};base64,${block.data}`,
} satisfies ResponseInputImage);
}
@@ -139,6 +139,10 @@ function getLatestSurvivingAssistantIndex(messages: readonly Message[]): number
return -1;
}
function isAnthropicMessagesModel(model: Model): model is Model<"anthropic-messages"> {
return model.api === "anthropic-messages";
}
/**
* Normalize tool call ID for cross-provider compatibility.
* OpenAI Responses API generates IDs that are 450+ chars with special characters like `|`.
@@ -184,10 +188,45 @@ export function transformMessages<TApi extends Api>(
assistantMsg.api === model.api &&
assistantMsg.model === model.id;
const mustPreserveLatestAnthropicThinking =
index === latestSurvivingAssistantIndex &&
model.api === "anthropic-messages" &&
assistantMsg.api === "anthropic-messages";
const isAnthropicTarget = isAnthropicMessagesModel(model);
// Anthropic's all-or-none contract on prior-turn thinking blocks
// applies to every `anthropic-messages → anthropic-messages` replay,
// not just the latest assistant turn. The legacy
// `mustPreserveLatestAnthropicThinking` flag only honored it for the
// latest turn; every prior turn fell through to the cross-API
// text-demotion path whenever the conversation crossed a model id,
// silently dropping the reasoning chain on continuation for custom
// anthropic-messages providers configured via `models.yaml` and
// session-level model swaps (#2257).
const isAnthropicReplay = isAnthropicTarget && assistantMsg.api === "anthropic-messages";
const isLatestSurvivingAssistant = index === latestSurvivingAssistantIndex;
// Signature policy is a second axis. Anthropic cryptographically
// binds reasoning signatures to its key+session+model, so cross-model
// signatures must be stripped whenever official Anthropic is on
// either end of the replay:
// * official → 3p: the 3p target can't reverify the signature;
// keeping it leaks private continuation metadata for no benefit.
// * 3p → official: official rejects a foreign signature outright.
// * official → official cross-model: the new model rejects the
// previous model's signature.
// 3p ↔ 3p replays preserve signatures because compatible providers
// (Z.AI, DeepSeek, custom `models.yaml` providers) treat them as
// opaque continuation hints rather than verified material; stripping
// degrades the reasoning chain into unsigned/text on the next turn
// (#2265). Source-side official detection uses the canonical catalog
// provider id `"anthropic"` because assistant messages carry no
// `baseUrl` — a user who manually points `provider: "anthropic"` at
// a custom proxy via `models.yaml` will see signatures stripped, the
// conservative direction (degraded reasoning, not broken requests).
const isOfficialAnthropicSource = isAnthropicReplay && assistantMsg.provider === "anthropic";
const isOfficialAnthropicTarget = isAnthropicTarget && model.compat.officialEndpoint;
const officialAnthropicInvolved = isOfficialAnthropicSource || isOfficialAnthropicTarget;
// Compatible Anthropic-messages reasoning targets that accept
// unsigned thinking natively (Z.AI, DeepSeek, the generic
// `reasoning && !official` case in the compat builder). Used to keep
// `redacted_thinking` siblings beside unsigned visible thinking on
// targets that won't text-demote it.
const replaysUnsignedAnthropicThinking = isAnthropicTarget && model.compat.replayUnsignedThinking;
// Thinking signatures can be untrustworthy for two distinct reasons with very
// different blast radii:
//
@@ -226,11 +265,37 @@ export function transformMessages<TApi extends Api>(
// untrustworthy signature so the encoder can downgrade the block to text.
const signatureUntrustworthy =
abandonedToolUse || (invalidStopReason && blockIndex === lastBlockIndex);
const sanitized =
let sanitized: typeof block =
signatureUntrustworthy && block.thinkingSignature
? { ...block, thinkingSignature: undefined }
: block;
if (mustPreserveLatestAnthropicThinking) return abandonedToolUse ? block : sanitized;
if (isAnthropicReplay) {
// Latest abandoned turn: Anthropic's byte-for-byte rule forbids
// even stripping a signature on the latest message.
if (isLatestSurvivingAssistant && abandonedToolUse) return block;
// Cross-model prior turns crossing an official Anthropic endpoint
// must strip the source signature so the downstream encoder
// applies its `replayUnsignedThinking` policy (unsigned thinking
// is emitted natively on Anthropic-compatible reasoning endpoints
// and demoted to text on official Anthropic). 3p ↔ 3p replays
// keep the signature so the reasoning chain stays signed on
// continuation (#2265).
if (
!isLatestSurvivingAssistant &&
!isSameModel &&
officialAnthropicInvolved &&
sanitized.thinkingSignature
) {
sanitized = { ...sanitized, thinkingSignature: undefined };
}
// Drop blocks with neither a signature anchor nor any text —
// nothing for the next turn to replay.
if (!sanitized.thinkingSignature && (!sanitized.thinking || sanitized.thinking.trim() === "")) {
return [];
}
return sanitized;
}
// Cross-API target: keep the existing text-demotion fallback.
// For same model: keep thinking blocks with signatures (needed for replay)
// even if the thinking text is empty (OpenAI encrypted reasoning)
if (isSameModel && sanitized.thinkingSignature) return sanitized;
@@ -244,7 +309,15 @@ export function transformMessages<TApi extends Api>(
}
if (block.type === "redactedThinking") {
if (mustPreserveLatestAnthropicThinking) return block;
// Redacted thinking is native-only. Keep it for same-model
// signed replay, the latest byte-for-byte Anthropic turn, or
// compatible targets that will also emit sibling unsigned
// thinking natively. Drop it when the visible thinking was
// cross-model stripped and will be demoted to text.
if (isAnthropicReplay) {
if (isSameModel || isLatestSurvivingAssistant || replaysUnsignedAnthropicThinking) return block;
return [];
}
if (isSameModel) return block;
return [];
}
+6
View File
@@ -409,6 +409,12 @@ export interface ImageContent {
type: "image";
data: string; // base64 encoded image data
mimeType: string; // e.g., "image/jpeg", "image/png"
/**
* OpenAI-only resolution hint. `"original"` preserves native resolution
* (required for snapcompact frames, whose glyphs do not survive the
* default `auto` downscale). Providers without a detail knob ignore it.
*/
detail?: "auto" | "low" | "high" | "original";
}
export interface ToolCall {
@@ -0,0 +1,342 @@
import { describe, expect, it } from "bun:test";
import { convertAnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic";
import type {
AssistantMessage,
Message,
Model,
ModelSpec,
ToolResultMessage,
UserMessage,
} from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
/**
* Cross-model `anthropic-messages` continuations must preserve the prior
* turn's reasoning chain. Anthropic enforces an all-or-none contract on
* thinking blocks ("if you include thinking blocks in prior assistant turns,
* you must include ALL thinking blocks (including redacted ones)") but the
* legacy transform only honored that for the LATEST surviving assistant.
* Every earlier turn fell through to the cross-API text-demotion path
* whenever the conversation crossed a model boundary — silently dropping the
* reasoning chain on continuation for custom anthropic-messages providers
* configured via `models.yaml` and for session-level model swaps (#2257).
*
* The signature policy is a second axis: official Anthropic cryptographically
* binds signatures to its key+session+model, so cross-model signatures must
* be stripped (and matching redacted siblings dropped) whenever either side
* of the replay is official Anthropic. Third-party endpoints (Z.AI, DeepSeek,
* custom anthropic-messages providers) treat signatures as opaque
* continuation hints they pass through unchanged, so 3p ↔ 3p replays
* preserve them as-is to keep the reasoning chain signed for the next
* turn (#2265).
*/
function makeAnthropicModel(overrides: Partial<ModelSpec<"anthropic-messages">> = {}): Model<"anthropic-messages"> {
return buildModel({
api: "anthropic-messages",
provider: "custom-anthropic",
id: "reasoning-model",
name: "Reasoning Anthropic-Compatible Model",
baseUrl: "https://llm.example.com/anthropic",
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
maxTokens: 8_192,
contextWindow: 200_000,
reasoning: true,
...overrides,
} as ModelSpec<"anthropic-messages">);
}
function makeUser(text: string): UserMessage {
return { role: "user", content: text, timestamp: 0 };
}
function makeAssistant(
content: AssistantMessage["content"],
overrides: Partial<AssistantMessage> = {},
): AssistantMessage {
return {
role: "assistant",
content,
api: "anthropic-messages",
provider: "custom-anthropic",
model: "reasoning-model",
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "toolUse",
timestamp: 0,
...overrides,
};
}
function toolResult(toolCallId: string, text: string): ToolResultMessage {
return {
role: "toolResult",
toolCallId,
toolName: "read",
content: [{ type: "text", text }],
isError: false,
timestamp: 0,
};
}
interface WireThinkingBlock {
type: "thinking";
thinking: string;
signature: string;
}
interface WireTextBlock {
type: "text";
text: string;
}
interface WireRedactedBlock {
type: "redacted_thinking";
data: string;
}
interface WireToolUseBlock {
type: "tool_use";
id: string;
name: string;
input: Record<string, unknown>;
}
type WireBlock =
| WireThinkingBlock
| WireTextBlock
| WireRedactedBlock
| WireToolUseBlock
| { type: string; [key: string]: unknown };
describe("Anthropic prior-turn thinking preservation (#2257, #2265)", () => {
it("preserves the prior thinking block as native `thinking` across compatible endpoints", () => {
// Source v1, target v2, both on the same custom anthropic-messages
// provider. The first assistant turn is PRIOR, so the latest-only
// preservation path doesn't help — without the fix the prior thinking
// block is demoted to plain `text` and the reasoning chain disappears.
const target = makeAnthropicModel({ id: "reasoning-model-v2" });
const priorThinkingText = "Plan: read README, then summarize.";
const messages: Message[] = [
makeUser("Summarize README"),
makeAssistant(
[
{ type: "thinking", thinking: priorThinkingText, thinkingSignature: "sig_v1" },
{ type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "README.md" } },
],
{ model: "reasoning-model-v1" },
),
toolResult("toolu_prior", "README body"),
makeAssistant(
[
{ type: "thinking", thinking: "Got the body, now translating", thinkingSignature: "sig_v2" },
{ type: "text", text: "Voici le résumé en français." },
],
{ model: "reasoning-model-v2", stopReason: "stop" },
),
makeUser("Now translate it to Spanish"),
];
const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant");
expect(assistants).toHaveLength(2);
const priorBlocks = assistants[0].content as WireBlock[];
const thinking = priorBlocks.find(b => b.type === "thinking") as WireThinkingBlock | undefined;
expect(thinking).toBeDefined();
expect(thinking?.thinking).toBe(priorThinkingText);
// 3p ↔ 3p replay: the source signature is opaque continuation metadata
// that compatible endpoints pass through. Stripping it (the pre-fix
// behavior) silently demotes the reasoning chain on the next turn.
expect(thinking?.signature).toBe("sig_v1");
// And the paired tool_use must still be present right after it.
const toolUse = priorBlocks.find(b => b.type === "tool_use") as WireToolUseBlock | undefined;
expect(toolUse?.id).toBe("toolu_prior");
});
it("keeps the signature on prior turns when the source model matches the target", () => {
// Same provider+api+id throughout: signatures are valid and must ride
// the wire untouched (prompt-cache stability + Anthropic's all-or-none
// invariant).
const target = makeAnthropicModel();
const messages: Message[] = [
makeUser("Summarize README"),
makeAssistant([
{ type: "thinking", thinking: "plan", thinkingSignature: "sig_same" },
{ type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "README.md" } },
]),
toolResult("toolu_prior", "README body"),
makeAssistant(
[
{ type: "thinking", thinking: "summarising", thinkingSignature: "sig_latest" },
{ type: "text", text: "summary" },
],
{ stopReason: "stop" },
),
makeUser("And now in Spanish"),
];
const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant");
const priorBlocks = assistants[0].content as WireBlock[];
const thinking = priorBlocks.find(b => b.type === "thinking") as WireThinkingBlock | undefined;
expect(thinking?.thinking).toBe("plan");
expect(thinking?.signature).toBe("sig_same");
});
it("preserves redacted_thinking blocks from prior anthropic-messages turns", () => {
// Anthropic's "include ALL thinking blocks (including redacted ones)"
// rule means redacted_thinking from earlier turns must survive whenever
// any thinking content from the same turn is replayed.
const target = makeAnthropicModel({ id: "reasoning-model-v2" });
const messages: Message[] = [
makeUser("Summarize README"),
makeAssistant(
[
{ type: "thinking", thinking: "visible reasoning", thinkingSignature: "sig" },
{ type: "redactedThinking", data: "encrypted-blob" },
{ type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "README.md" } },
],
{ model: "reasoning-model-v1" },
),
toolResult("toolu_prior", "README body"),
makeAssistant(
[
{ type: "thinking", thinking: "later", thinkingSignature: "sig_latest" },
{ type: "text", text: "summary" },
],
{ model: "reasoning-model-v2", stopReason: "stop" },
),
makeUser("Translate"),
];
const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant");
const priorBlocks = assistants[0].content as WireBlock[];
const redacted = priorBlocks.find(b => b.type === "redacted_thinking") as WireRedactedBlock | undefined;
expect(redacted).toBeDefined();
expect(redacted?.data).toBe("encrypted-blob");
});
it("strips foreign signatures and drops redacted_thinking when the target is official Anthropic", () => {
// 3p → official Anthropic. The official endpoint rejects foreign
// signatures cryptographically, and `replayUnsignedThinking: false`
// demotes the unsigned visible thinking to text downstream, so the
// matching redacted sibling must not remain as a lone native
// redacted_thinking block.
const target = makeAnthropicModel({
provider: "anthropic",
id: "claude-sonnet-4-6",
baseUrl: "https://api.anthropic.com",
});
const messages: Message[] = [
makeUser("Summarize README"),
makeAssistant(
[
{ type: "thinking", thinking: "visible reasoning", thinkingSignature: "sig_custom" },
{ type: "redactedThinking", data: "foreign-encrypted-blob" },
{ type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "README.md" } },
],
{ model: "reasoning-model-v1" },
),
toolResult("toolu_prior", "README body"),
makeAssistant(
[
{ type: "thinking", thinking: "official latest", thinkingSignature: "sig_latest" },
{ type: "text", text: "summary" },
],
{
provider: "anthropic",
model: "claude-sonnet-4-6",
stopReason: "stop",
},
),
makeUser("Translate"),
];
const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant");
const priorBlocks = assistants[0].content as WireBlock[];
const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined;
expect(text?.text).toBe("visible reasoning");
expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined();
expect(priorBlocks.find(b => b.type === "redacted_thinking")).toBeUndefined();
});
it("strips official Anthropic source signatures on cross-model replay to a 3p target", () => {
// official Anthropic → 3p. Anthropic's signature is bound to the
// issuing model+session, so the 3p target cannot reverify or
// meaningfully continue from it; passing it through would leak
// private continuation metadata for no benefit. The unsigned thinking
// is still emitted natively because the 3p target's compat advertises
// `replayUnsignedThinking: true`.
const target = makeAnthropicModel({ id: "reasoning-model-v2" });
const messages: Message[] = [
makeUser("Summarize README"),
makeAssistant(
[
{ type: "thinking", thinking: "anthropic reasoning", thinkingSignature: "sig_anthropic" },
{ type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "README.md" } },
],
{ provider: "anthropic", model: "claude-sonnet-4-6" },
),
toolResult("toolu_prior", "README body"),
makeAssistant(
[
{ type: "thinking", thinking: "v2 reasoning", thinkingSignature: "sig_v2" },
{ type: "text", text: "summary" },
],
{ model: "reasoning-model-v2", stopReason: "stop" },
),
makeUser("Translate"),
];
const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant");
const priorBlocks = assistants[0].content as WireBlock[];
const thinking = priorBlocks.find(b => b.type === "thinking") as WireThinkingBlock | undefined;
expect(thinking?.thinking).toBe("anthropic reasoning");
expect(thinking?.signature).toBe("");
});
it("does not promote prior unsigned thinking from non-anthropic sources to thinking blocks", () => {
// Cross-API replay: prior turn came from OpenAI-responses with no
// Anthropic signature. The all-or-none rule scope is per-API; we must
// not invent thinking blocks for a turn whose source can't sign them —
// the existing cross-API text demotion is the right behavior.
const target = makeAnthropicModel();
const messages: Message[] = [
makeUser("Summarize README"),
makeAssistant(
[
{ type: "thinking", thinking: "openai chain-of-thought", thinkingSignature: "" },
{ type: "toolCall", id: "toolu_prior", name: "read", arguments: { path: "README.md" } },
],
{
api: "openai-responses",
provider: "openai",
model: "o1-preview",
} as Partial<AssistantMessage>,
),
toolResult("toolu_prior", "README body"),
makeAssistant(
[
{ type: "thinking", thinking: "anthropic latest", thinkingSignature: "sig_latest" },
{ type: "text", text: "summary" },
],
{ stopReason: "stop" },
),
makeUser("Translate"),
];
const params = convertAnthropicMessages(messages, target, false);
const assistants = params.filter(p => p.role === "assistant");
const priorBlocks = assistants[0].content as WireBlock[];
expect(priorBlocks.find(b => b.type === "thinking")).toBeUndefined();
// Reasoning text still survives on the wire (as text, via the existing
// cross-API demotion path).
const text = priorBlocks.find(b => b.type === "text") as WireTextBlock | undefined;
expect(text?.text).toBe("openai chain-of-thought");
});
});
+46 -2
View File
@@ -5,6 +5,50 @@
### Fixed
- Fixed MCP OAuth authorization and token requests to include the required `resource` indicator for the target MCP server.
### Breaking Changes
- The `task` tool now spawns exactly one subagent per call and always runs it in the background: the batch `tasks[]` array and shared `context` parameter are removed — fan out with parallel `task` calls, share background via a `local://` file referenced in each assignment, and receive results as async job deliveries (block with `job poll` only when genuinely needed)
- Reworked `irc` to `send`/`wait`/`inbox`/`list` ops over a per-agent mailbox bus: the blocking `awaitReply` auto-reply turn is removed — `send` is fire-and-forget with delivery receipts, and replies are real turns by the recipient observed via `wait` (or the `send` `await: true` sugar)
- Removed the `context` argument from eval `agent()` in both the JS and Python preludes: pass shared background via a `local://` file referenced in the prompt
- Replaced the standalone session-observer overlay with the Agent Hub: `app.session.observe` (`ctrl+s`) now opens the hub, whose chat view absorbed the observer's transcript renderer
### Added
- Added pre-TUI startup input capture so users could type while interactive sessions initialize and keep their draft while the application loads
- Snapcompact compaction now passes the session model so frames render in the provider-optimal shape (unscii `8x8r-bw` for Anthropic-family/unknown APIs, `8x8r-sent` for Google, Lanczos-stretched `6x6u-sent` with `detail: "original"` for OpenAI), per the snapcompact 200k-token evals
- Added queued submission replay so Enter presses made before startup completion are submitted automatically once interactive mode begins
- Added per-turn supersede pruning of stale `read` results: when a file is re-read, older copies of the same path/selector are pruned from context at cache-favorable moments (small suffix, idle gap, or alongside overflow pruning). Gated by the new `compaction.supersedeReads` setting (default on)
- Added soft request budgets for task subagents (explore/quick_task 40, others 90, configurable via `task.softRequestBudget`, 0 disables): crossing the budget injects a one-time wrap-up steer into the child; crossing 1.5× aborts the run gracefully
- Added cancelled/aborted subagent salvage: instead of `(no output)`, merged task results now carry the child's last activity snippet plus request/token stats, and per-child stats lines include request counts
- Added a repeat-read notice to the `read` tool: the third and later reads of the same file in a session append a one-line note suggesting range re-reads or the context echoed in edit results
- Added a hard inline byte cap (~50KB) at the bash and browser tool-result boundaries with head/tail elision and an `artifact://` footer for the full output, closing paths that previously let 100KB+ results land inline
- Added the Agent Hub overlay (`ctrl+s`, `alt+a`, or double-tap left arrow on an empty editor): a live table of registered subagents (status, unread IRC count, current task, last activity) with per-agent chat — Enter opens a transcript + input line that steers a running agent, prompts an idle one, and revives a parked one; `r` revives and `x` aborts/releases the selected agent
- Added the `snapcompact` compaction strategy (`compaction.strategy: "snapcompact"`): history is archived onto dense bitmap "snapcompact" frames a vision model reads back directly, instead of an LLM-generated summary — instant, free, and verbatim. Auto compaction (including overflow recovery) and manual `/compact` both honor it; falls back to context-full with a visible warning notice when the current model is text-only (e.g. Codex API surfaces) or when `/compact` is given custom instructions. Frames survive context rebuilds and later compactions (budget eviction is middle-out: the session-head frame is pinned); the expanded compaction message notes the attached frame count
- Added a persistent subagent lifecycle: finished subagents stay live as `idle`, are parked to disk after `task.agentIdleTtlMs` (default 7 minutes; `0` keeps them live until exit), and are revived automatically when messaged, resumed, or prompted from the Agent Hub
- Added `task(resume: "<id>")` to revive an idle or parked subagent and run a follow-up assignment in its existing session, keeping its accumulated context
- Added the `history://` protocol: `history://` lists every registered agent and `history://<agentId>` renders a concise markdown transcript (tool calls collapsed to one line each, thinking elided) for live and parked agents alike
- Added an IRC mailbox bus with bounded per-agent inboxes: `irc` `wait` blocks until a matching message arrives, `inbox` drains or peeks pending messages, and sending to an idle or parked agent wakes or revives it for a real turn
- Added a dedicated TUI renderer for the `irc` tool: directional send/receive headers with delivery-outcome coloring, quoted message bodies with expand-aware truncation, per-recipient receipt trees for broadcasts and failures, and status-badged peer listings with unread counts
### Changed
- Changed interactive startup to carry the startup editor state into the live prompt so text entered during splash is preserved in the editor when the TUI takes over
- Changed the compaction UX so the conversation no longer visually restarts: the TUI renders the full-history display transcript (`buildSessionContext({ transcript: true })`), with each compaction shown as a slim inline divider — `── 📷 compacted · ctrl+o ──` — at the point it fired; expanding (ctrl+o) reveals the summary and snapcompact frame count. Applies to live compaction, `/compact`, `/tree` navigation, and session resume
- Changed model-scope display during startup to appear as an in-UI information notification instead of a direct stdout line
- Changed `async.enabled` to gate async bash commands only — the `task` tool now runs asynchronously regardless of the setting
- Changed `irc.timeoutMs` to be the default timeout for `irc` `wait` and `send` with `await: true`
### Fixed
- Fixed startup Ctrl+C handling in pre-TUI mode so it now clears typed text before exiting on a second press
- Fixed npm CLI distribution bundles by embedding the stats dashboard client bundle so dashboard assets are served in prebuilt installs
- Fixed the CLI smoke-test command to start the stats server and verify dashboard HTML is served, catching bundled-asset regressions
- Added verification of a `<div id="root"></div>` and `index.js` in smoke-test dashboard responses
- Restored the checkmark glyph on ask-tool custom answers and the multi-select "Done selecting" option, which a status-glyph sweep had swapped for the ask tool icon
### Fixed
- Fixed the `thinking.autoPending` statusbar indicator using question-mark glyphs (`▣?`, nf-md-help_box, `[?]`) in every symbol preset, which made the auto-thinking pending state indistinguishable from a terminal missing-glyph fallback. Replaced with clear loading indicators (`⟳`, fa-circle-o-notch, `[~]`) ([#2267](https://github.com/can1357/oh-my-pi/issues/2267)).
## [15.10.12] - 2026-06-10
@@ -24,7 +68,7 @@
### Changed
- Bash execution now preserves minimized shell output inline while saving the untouched capture as an `artifact://…` footer when shell minimization rewrites a command's output.
- Task tool live progress now renders finished subagents first and keeps unfinished (pending/running) ones pinned at the bottom of the list.
- Task tool agent lists now render in runtime-ascending order in both the live progress view (finished agents, sorted by runtime, above pending/running ones) and the finalized result view, so rows no longer reshuffle when the call finalizes.
- `OutputSink` artifact files (`~/.omp/agent/artifacts/<id>.<tool>.log`) are unbounded by default again, so `artifact://<id>` references preserve the complete raw stream. The head + rolling-tail capping machinery from [#2081](https://github.com/can1357/oh-my-pi/issues/2081) (with its `[ARTIFACT TRUNCATED: …]` close notice) remains available as an opt-in via `artifactMaxBytes`, and the head window now closes permanently on first overflow so later small chunks cannot be written out of order before the tail replay.
### Fixed
@@ -9975,4 +10019,4 @@ Initial public release.
- Git branch display in footer
- Message queueing during streaming responses
- OAuth integration for Gmail and Google Calendar access
- HTML export with syntax highlighting and collapsible sections
- HTML export with syntax highlighting and collapsible sections
@@ -1,6 +1,7 @@
{
"name": "pi-extension-with-deps",
"version": "1.0.0",
"homepage": "https://omp.sh",
"type": "module",
"omp": {
"extensions": [
+1
View File
@@ -56,6 +56,7 @@
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-tui": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
"@oh-my-pi/snapcompact": "catalog:",
"@opentelemetry/api": "catalog:",
"@opentelemetry/context-async-hooks": "catalog:",
"@opentelemetry/exporter-trace-otlp-proto": "catalog:",
+28 -19
View File
@@ -51,25 +51,34 @@ async function cleanBundleOutputs(): Promise<void> {
async function main(): Promise<void> {
const start = Bun.nanoseconds();
await cleanBundleOutputs();
await runCommand([
"bun",
"build",
"--target=bun",
"--outdir",
"dist",
"--minify-whitespace",
"--minify-syntax",
"--keep-names",
"--external",
"mupdf",
"--external",
"@oh-my-pi/pi-natives",
"--external",
"@huggingface/transformers",
"--define",
'process.env.PI_BUNDLED="true"',
"./src/cli.ts",
]);
// The npm bundle ships no stats dashboard sources or prebuilt dist/client,
// so embed the dashboard archive the same way compiled binaries do
// (scripts/build-binary.ts). Reset afterwards to keep the checked-in
// placeholder empty.
await runCommand(["bun", "--cwd=../stats", "scripts/generate-client-bundle.ts", "--generate"]);
try {
await runCommand([
"bun",
"build",
"--target=bun",
"--outdir",
"dist",
"--minify-whitespace",
"--minify-syntax",
"--keep-names",
"--external",
"mupdf",
"--external",
"@oh-my-pi/pi-natives",
"--external",
"@huggingface/transformers",
"--define",
'process.env.PI_BUNDLED="true"',
"./src/cli.ts",
]);
} finally {
await runCommand(["bun", "--cwd=../stats", "scripts/generate-client-bundle.ts", "--reset"]);
}
await ensureShebang();
const stat = await fs.stat(cliPath);
const elapsedMs = (Bun.nanoseconds() - start) / 1_000_000;
-1
View File
@@ -1,2 +1 @@
export * from "./job-manager";
export * from "./support";
@@ -1,5 +0,0 @@
import type { Settings } from "../config/settings";
export function isBackgroundJobSupportEnabled(settings: Pick<Settings, "get">): boolean {
return settings.get("async.enabled") || settings.get("bash.autoBackground.enabled");
}
+20 -6
View File
@@ -43,19 +43,33 @@ async function showHelp(config: CliConfig): Promise<void> {
}
}
/**
* Smoke-test entry. Spawns bundled workers, pings them, exits.
* Smoke-test entry. Spawns bundled workers, serves the stats dashboard once,
* pings everything, then exits.
*
* Purpose: catch the silent worker-load regressions that hit compiled
* binaries (issues #1011 and #1027). Version/help paths do not spawn worker
* modules on a fresh install, so this probe is the minimal end-to-end test
* that proves `new Worker(...)` resolves and bundled worker modules evaluate.
* Purpose: catch the silent worker-load and bundled-asset regressions that hit
* compiled binaries and the npm CLI bundle. Version/help paths do not spawn
* worker modules or serve dashboard assets on a fresh install, so this probe is
* the minimal end-to-end test that proves those distribution-only paths work.
* Wired into `scripts/install-tests/run-ci.sh` so binary / source-link /
* tarball installs all exercise it on every CI run.
*/
async function runSmokeTest(): Promise<void> {
const { smokeTestSyncWorker } = await import("@oh-my-pi/omp-stats");
const { smokeTestSyncWorker, startServer } = await import("@oh-my-pi/omp-stats");
const { smokeTestTinyTitleWorker } = await import("./tiny/title-client");
await smokeTestSyncWorker();
const statsServer = await startServer(0);
try {
const response = await fetch(`http://127.0.0.1:${statsServer.port}/`);
if (!response.ok) throw new Error(`stats dashboard smoke failed: HTTP ${response.status}`);
const html = await response.text();
if (!html.includes('<div id="root"></div>') || !html.includes("index.js")) {
throw new Error("stats dashboard smoke failed: dashboard HTML was not served");
}
} finally {
statsServer.stop();
}
await smokeTestTinyTitleWorker();
process.stdout.write("smoke-test: ok\n");
}
+1 -1
View File
@@ -69,7 +69,7 @@ function fakeToolFor(name: string, fixture: GalleryFixture | undefined): AgentTo
if (!fixture?.label && !fixture?.editMode && !fixture?.customRendered) return undefined;
const tool: Record<string, unknown> = { name, label: fixture.label ?? name, mode: fixture.editMode };
if (fixture.customRendered) {
const renderer = toolRenderers[name] as
const renderer = toolRenderers[fixture.renderer ?? name] as
| { renderCall?: unknown; renderResult?: unknown; mergeCallAndResult?: unknown; inline?: unknown }
| undefined;
if (renderer) {
@@ -1,62 +1,76 @@
// Gallery fixtures for the agentic orchestration tools (task, goal, job).
// Gallery fixtures for the agentic orchestration tools (task, irc, goal, job).
import type { Usage } from "@oh-my-pi/pi-ai";
import type { TaskToolDetails } from "../../task/types";
import type { IrcDetails } from "../../tools/irc";
import type { GalleryFixture } from "./types";
/** Message/activity timestamps are offsets from load time so gallery ages stay plausible. */
const FIXTURE_NOW = Date.now();
/** Plausible cumulative usage for a fixture subagent run. */
const fixtureUsage = (tokens: { input: number; output: number }, costTotal: number): Usage => ({
input: tokens.input,
output: tokens.output,
cacheRead: 0,
cacheWrite: 0,
totalTokens: tokens.input + tokens.output,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: costTotal },
});
export const agenticFixtures: Record<string, GalleryFixture> = {
task: {
label: "Task",
customRendered: true,
// Streaming: agent chosen, first task fully arrived, second still landing.
// Streaming: agent chosen, assignment still landing.
streamingArgs: {
agent: "task",
tasks: [
{
id: "AuthLoader",
description: "Load auth middleware",
assignment: "Read packages/server/src/auth/*.ts and summarize the session-cookie flow.",
},
{ id: "RateLimiter", description: "Audit rate limiter" },
],
id: "AuthLoader",
description: "Load auth middleware",
assignment: "Read packages/server/src/auth/*.ts and summarize the session-cookie",
},
args: {
agent: "task",
context: [
"# Goal",
"Harden the HTTP auth stack before the release cut.",
"# Constraints",
"Touch only files under packages/server/src/auth/. Do not run gates.",
].join("\n"),
tasks: [
{
id: "AuthLoader",
description: "Load auth middleware",
assignment:
"Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.",
},
{
id: "RateLimiter",
description: "Audit rate limiter",
assignment:
"Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.",
},
{
id: "TokenRotation",
description: "Check token rotation",
assignment:
"Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.",
},
],
id: "AuthLoader",
description: "Load auth middleware",
assignment:
"Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.",
},
result: {
content: [
{
type: "text",
text: "3 agents completed: AuthLoader, RateLimiter, TokenRotation.",
text: "Agent AuthLoader completed.",
},
],
details: {
projectAgentsDir: null,
totalDurationMs: 48_200,
usage: { cost: { total: 0.34 } },
usage: fixtureUsage({ input: 52_600, output: 8_800 }, 0.12),
progress: [
{
index: 0,
id: "AuthLoader",
agent: "task",
agentSource: "bundled",
status: "completed",
task: "Read packages/server/src/auth/session.ts and middleware.ts",
description: "Load auth middleware",
lastIntent: "Documenting session-cookie flow",
recentTools: [
{ tool: "read", args: "packages/server/src/auth/session.ts", endMs: 1_749_200_040_000 },
{ tool: "read", args: "packages/server/src/auth/middleware.ts", endMs: 1_749_200_052_000 },
],
recentOutput: ["Session validation runs in middleware.ts:42 via verifySessionCookie()."],
toolCount: 9,
requests: 6,
tokens: 61_400,
contextTokens: 23_100,
contextWindow: 200_000,
cost: 0.12,
durationMs: 41_900,
resolvedModel: "anthropic/claude-sonnet",
},
],
results: [
{
index: 0,
@@ -77,100 +91,31 @@ export const agenticFixtures: Record<string, GalleryFixture> = {
truncated: false,
durationMs: 41_900,
tokens: 61_400,
requests: 6,
contextTokens: 23_100,
contextWindow: 200_000,
resolvedModel: "anthropic/claude-sonnet",
usage: { cost: { total: 0.12 } },
usage: fixtureUsage({ input: 52_600, output: 8_800 }, 0.12),
outputMeta: { lineCount: 3, charCount: 214 },
},
{
index: 1,
id: "RateLimiter",
agent: "task",
agentSource: "bundled",
description: "Audit rate limiter",
task: "Inspect packages/server/src/auth/rate-limit.ts",
assignment:
"Inspect packages/server/src/auth/rate-limit.ts. Confirm the 429 path sets Retry-After and report gaps.",
exitCode: 0,
output: [
"rate-limit.ts uses a fixed-window counter keyed by client IP.",
"429 responses set Retry-After (rate-limit.ts:57).",
"Gap: no per-account limit, so a botnet across IPs bypasses the cap.",
].join("\n"),
stderr: "",
truncated: false,
durationMs: 38_500,
tokens: 54_800,
contextTokens: 19_700,
contextWindow: 200_000,
resolvedModel: "anthropic/claude-sonnet",
usage: { cost: { total: 0.1 } },
outputMeta: { lineCount: 3, charCount: 198 },
},
{
index: 2,
id: "TokenRotation",
agent: "task",
agentSource: "bundled",
description: "Check token rotation",
task: "Trace refresh-token rotation in packages/server/src/auth/tokens.ts",
assignment:
"Trace refresh-token rotation in packages/server/src/auth/tokens.ts and flag any reuse window.",
exitCode: 0,
output: [
"Refresh tokens rotate on every use (tokens.ts:120) and the old jti is revoked.",
"Reuse of a rotated token triggers full-family revocation — no reuse window found.",
].join("\n"),
stderr: "",
truncated: false,
durationMs: 48_200,
tokens: 49_200,
contextTokens: 17_500,
contextWindow: 200_000,
resolvedModel: "anthropic/claude-sonnet",
usage: { cost: { total: 0.12 } },
outputMeta: { lineCount: 2, charCount: 160 },
},
],
},
} satisfies TaskToolDetails,
},
errorResult: {
isError: true,
content: [
{
type: "text",
text: "1 of 3 agents failed: RateLimiter.",
text: "Agent RateLimiter failed.",
},
],
details: {
projectAgentsDir: null,
totalDurationMs: 39_400,
usage: { cost: { total: 0.21 } },
totalDurationMs: 9_800,
usage: fixtureUsage({ input: 10_900, output: 1_400 }, 0.1),
results: [
{
index: 0,
id: "AuthLoader",
agent: "task",
agentSource: "bundled",
description: "Load auth middleware",
task: "Read packages/server/src/auth/session.ts and middleware.ts",
assignment:
"Read packages/server/src/auth/session.ts and middleware.ts, then document the session-cookie validation flow and any TODOs.",
exitCode: 0,
output: "Session validation runs in middleware.ts:42 via verifySessionCookie().",
stderr: "",
truncated: false,
durationMs: 31_200,
tokens: 58_100,
contextTokens: 21_900,
contextWindow: 200_000,
resolvedModel: "anthropic/claude-sonnet",
usage: { cost: { total: 0.11 } },
outputMeta: { lineCount: 1, charCount: 70 },
},
{
index: 1,
id: "RateLimiter",
agent: "task",
agentSource: "bundled",
@@ -184,15 +129,243 @@ export const agenticFixtures: Record<string, GalleryFixture> = {
truncated: false,
durationMs: 9_800,
tokens: 12_300,
requests: 3,
contextTokens: 6_400,
contextWindow: 200_000,
resolvedModel: "anthropic/claude-sonnet",
usage: { cost: { total: 0.1 } },
usage: fixtureUsage({ input: 10_900, output: 1_400 }, 0.1),
error: "Subagent exited 1: target file packages/server/src/auth/rate-limit.ts does not exist.",
outputMeta: { lineCount: 0, charCount: 0 },
},
],
},
} satisfies TaskToolDetails,
},
},
// Resume: follow-up assignment into an existing (idle or parked) agent.
task_resume: {
label: "Task (resume)",
customRendered: true,
renderer: "task",
// Streaming: resume target known; the follow-up assignment still landing.
streamingArgs: {
resume: "AuthLoader",
assignment: "Follow up: does the sliding-expiration TODO affect",
},
args: {
resume: "AuthLoader",
assignment:
"Follow up: does the sliding-expiration TODO at session.ts:88 affect the refresh-token path? Document the answer.",
},
result: {
content: [{ type: "text", text: "Agent AuthLoader completed." }],
details: {
projectAgentsDir: null,
totalDurationMs: 22_400,
usage: fixtureUsage({ input: 30_200, output: 4_100 }, 0.07),
results: [
{
index: 0,
id: "AuthLoader",
agent: "task",
agentSource: "bundled",
task: "Follow up: does the sliding-expiration TODO at session.ts:88 affect the refresh-token path?",
assignment:
"Follow up: does the sliding-expiration TODO at session.ts:88 affect the refresh-token path? Document the answer.",
exitCode: 0,
output:
"No — refresh tokens bypass the sliding window: refreshSession() re-issues the cookie unconditionally (session.ts:131).",
stderr: "",
truncated: false,
durationMs: 19_700,
tokens: 34_300,
requests: 4,
contextTokens: 31_800,
contextWindow: 200_000,
resolvedModel: "anthropic/claude-sonnet",
usage: fixtureUsage({ input: 30_200, output: 4_100 }, 0.07),
outputMeta: { lineCount: 1, charCount: 118 },
},
],
} satisfies TaskToolDetails,
},
errorResult: {
isError: true,
content: [
{
type: "text",
text: 'No agent "AuthLoader" to resume — it ran isolated and is not revivable. See history:// for the agent index.',
},
],
},
},
irc: {
label: "IRC",
// Streaming: recipient known; the message body still arriving.
streamingArgs: { op: "send", to: "AuthLoader", message: "Are you still touching" },
args: {
op: "send",
to: "AuthLoader",
message: "Are you still touching src/server/auth.ts? I need to add a 401 path.",
await: true,
},
result: {
content: [
{
type: "text",
text: [
"Delivered to 1 peer(s):",
"- AuthLoader: revived",
"",
"Reply from AuthLoader:",
"Done with auth.ts — go ahead, just rebase past my session-store rename.",
].join("\n"),
},
],
details: {
op: "send",
from: "Main",
to: "AuthLoader",
receipts: [{ to: "AuthLoader", outcome: "revived" }],
waited: {
id: "7181122334455667789",
from: "AuthLoader",
to: "Main",
body: "Done with auth.ts — go ahead, just rebase past my session-store rename.",
ts: FIXTURE_NOW - 5_000,
replyTo: "7181122334455667788",
},
} satisfies IrcDetails,
},
errorResult: {
isError: true,
content: [
{
type: "text",
text: 'No recipients received the message.\n- RateLimiter: failed — unknown agent "RateLimiter"',
},
],
details: {
op: "send",
from: "Main",
to: "RateLimiter",
receipts: [{ to: "RateLimiter", outcome: "failed", error: 'unknown agent "RateLimiter"' }],
} satisfies IrcDetails,
},
},
irc_wait: {
label: "IRC (wait)",
customRendered: true,
renderer: "irc",
streamingArgs: { op: "wait", from: "AuthLoader" },
args: { op: "wait", from: "AuthLoader", timeoutMs: 60_000 },
result: {
content: [
{
type: "text",
text: "[7181122334455667790] AuthLoader: session-store rename is merged; auth.ts is yours.",
},
],
details: {
op: "wait",
from: "Main",
waited: {
id: "7181122334455667790",
from: "AuthLoader",
to: "Main",
body: "session-store rename is merged; auth.ts is yours.",
ts: FIXTURE_NOW - 30_000,
},
} satisfies IrcDetails,
},
},
irc_inbox: {
label: "IRC (inbox)",
customRendered: true,
renderer: "irc",
streamingArgs: { op: "inbox" },
args: { op: "inbox", peek: true },
result: {
content: [
{
type: "text",
text: [
"2 unread message(s):",
"- [7181122334455667791] AuthLoader: hub table reads unreadCount — ping me when the bus lands.",
"- [7181122334455667792] RateLimiter (reply to 7181122334455667791): bus is in; receipts carry outcome.",
].join("\n"),
},
],
details: {
op: "inbox",
from: "Main",
inbox: [
{
id: "7181122334455667791",
from: "AuthLoader",
to: "Main",
body: "hub table reads unreadCount — ping me when the bus lands.",
ts: FIXTURE_NOW - 4 * 60_000,
},
{
id: "7181122334455667792",
from: "RateLimiter",
to: "Main",
body: "bus is in; receipts carry outcome.",
ts: FIXTURE_NOW - 60_000,
replyTo: "7181122334455667791",
},
],
} satisfies IrcDetails,
},
},
irc_list: {
label: "IRC (list)",
customRendered: true,
renderer: "irc",
streamingArgs: { op: "list" },
args: { op: "list" },
result: {
content: [
{
type: "text",
text: [
"2 peer(s):",
"- AuthLoader [task · sub · idle] — parent Main, active 2m ago",
"- RateLimiter [task · sub · parked] — unread 2, parent Main, active 12m ago",
"",
"Parked agents are revived automatically when you message them.",
].join("\n"),
},
],
details: {
op: "list",
from: "Main",
peers: [
{
id: "AuthLoader",
displayName: "task",
kind: "sub",
status: "idle",
parentId: "Main",
unread: 0,
lastActivity: FIXTURE_NOW - 2 * 60_000,
},
{
id: "RateLimiter",
displayName: "task",
kind: "sub",
status: "parked",
parentId: "Main",
unread: 2,
lastActivity: FIXTURE_NOW - 12 * 60_000,
},
],
} satisfies IrcDetails,
},
},
@@ -36,6 +36,11 @@ export interface GalleryFixture {
* real one keeps the gallery honest for these tools.
*/
customRendered?: boolean;
/**
* Renderer-registry key to use when the fixture key is a variant of a tool
* (e.g. `task_resume` → `task`). Defaults to the fixture key.
*/
renderer?: string;
/**
* Arguments shown during the streaming state — a partial view of {@link args}
* as if the tool-call JSON were still arriving. May include `__partialJson`
@@ -59,29 +59,46 @@ export function createAnalyzeFileTool(options: {
label: "Analyze Files",
description: "Spawn quick_task agents to analyze files.",
parameters: analyzeFileSchema,
async execute(toolCallId, params, onUpdate, ctx, signal) {
async execute(toolCallId, params, _onUpdate, ctx, signal) {
const toolSession = buildToolSession(ctx, options);
// The hand-built ToolSession carries no asyncJobManager, so every
// execute() below takes the task tool's sync fallback and resolves
// with the subagent's result inline — exactly what this flow needs.
// The tool's session semaphore bounds the parallel fan-out.
const taskTool = await TaskTool.create(toolSession);
const numstat = options.state.overview?.numstat ?? [];
const tasks = params.files.map((file, index) => {
const relatedFiles = formatRelatedFiles(params.files, file, numstat);
const assignment = prompt.render(analyzeFilePrompt, {
file,
goal: params.goal,
related_files: relatedFiles,
});
return {
id: `AnalyzeFile${index + 1}`,
description: `Analyze ${file}`,
assignment,
};
});
const taskParams: TaskParams = {
agent: "quick_task",
schema: JSON.stringify(analyzeFileOutputSchema),
tasks,
const schema = JSON.stringify(analyzeFileOutputSchema);
const analyses = await Promise.all(
params.files.map((file, index) => {
const relatedFiles = formatRelatedFiles(params.files, file, numstat);
const assignment = prompt.render(analyzeFilePrompt, {
file,
goal: params.goal,
related_files: relatedFiles,
});
const taskParams: TaskParams = {
agent: "quick_task",
id: `AnalyzeFile${index + 1}`,
description: `Analyze ${file}`,
assignment,
schema,
};
return taskTool.execute(`${toolCallId}-${index + 1}`, taskParams, signal);
}),
);
const results = analyses.flatMap(analysis => analysis.details?.results ?? []);
const text = analyses
.map(analysis => analysis.content.find(part => part.type === "text")?.text ?? "")
.filter(Boolean)
.join("\n\n");
return {
content: [{ type: "text", text: text || "(no output)" }],
details: {
projectAgentsDir: null,
results,
totalDurationMs: analyses.reduce((sum, analysis) => sum + (analysis.details?.totalDurationMs ?? 0), 0),
},
};
return taskTool.execute(toolCallId, taskParams, signal, onUpdate);
},
};
}
@@ -36,6 +36,7 @@ interface AppKeybindings {
"app.clipboard.pasteTextRaw": true;
"app.clipboard.copyLine": true;
"app.clipboard.copyPrompt": true;
"app.agents.hub": true;
"app.session.new": true;
"app.session.tree": true;
"app.session.fork": true;
@@ -166,9 +167,13 @@ export const KEYBINDINGS = {
defaultKeys: [],
description: "Resume session",
},
"app.agents.hub": {
defaultKeys: "alt+a",
description: "Open the agent hub",
},
"app.session.observe": {
defaultKeys: "ctrl+s",
description: "Observe subagent sessions",
description: "Open the agent hub",
},
"app.session.togglePath": {
defaultKeys: "ctrl+p",
@@ -1177,13 +1177,13 @@ export const SETTINGS_SCHEMA = {
"compaction.strategy": {
type: "enum",
values: ["context-full", "handoff", "shake", "off"] as const,
values: ["context-full", "handoff", "shake", "snapcompact", "off"] as const,
default: "context-full",
ui: {
tab: "context",
label: "Compaction Strategy",
description:
"Choose in-place context-full maintenance, auto-handoff, surgical shake (drop heavy content), or disable auto maintenance (off)",
"Choose in-place context-full maintenance, auto-handoff, surgical shake (drop heavy content), snapcompact (archive history as dense images), or disable auto maintenance (off)",
options: [
{
value: "context-full",
@@ -1196,6 +1196,11 @@ export const SETTINGS_SCHEMA = {
label: "Shake",
description: "Drop heavy content (tool results + large blocks) in place; recover via artifact",
},
{
value: "snapcompact",
label: "Snapcompact",
description: "Archive history onto dense bitmap images the model reads back; no LLM call",
},
{
value: "off",
label: "Off",
@@ -1326,6 +1331,17 @@ export const SETTINGS_SCHEMA = {
],
},
},
"compaction.supersedeReads": {
type: "boolean",
default: true,
ui: {
tab: "context",
label: "Supersede Stale Reads",
description: "Prune older read results when the same file is read again (cache-aware, runs every turn)",
},
},
// Branch summaries
"branchSummary.enabled": {
type: "boolean",
@@ -2305,8 +2321,7 @@ export const SETTINGS_SCHEMA = {
ui: {
tab: "tools",
label: "IRC Timeout",
description:
"Drop IRC messages whose recipient does not respond within this many milliseconds (0 disables the timeout)",
description: "Default timeout for irc wait (and send await:true) in milliseconds; 0 disables the timeout",
options: [
{ value: "0", label: "Disabled" },
{ value: "30000", label: "30 seconds" },
@@ -2501,7 +2516,7 @@ export const SETTINGS_SCHEMA = {
ui: {
tab: "tools",
label: "Async Execution",
description: "Enable async bash commands and background task execution",
description: "Enable async bash commands",
},
},
@@ -2841,6 +2856,34 @@ export const SETTINGS_SCHEMA = {
},
},
"task.agentIdleTtlMs": {
type: "number",
default: 420_000,
ui: {
tab: "tasks",
label: "Agent Idle TTL",
description:
"How long an idle subagent stays live in memory before being parked to disk (ms). Parked agents are revived automatically when messaged or resumed. 0 keeps idle agents live until exit.",
},
},
"task.softRequestBudget": {
type: "number",
default: 90,
ui: {
tab: "tasks",
label: "Soft Subagent Request Budget",
description:
"Soft per-subagent request budget (assistant requests per run). Crossing it injects one steering notice asking the subagent to wrap up; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled explore/quick_task agents use a lower built-in budget.",
options: [
{ value: "0", label: "Disabled" },
{ value: "40", label: "40 requests" },
{ value: "90", label: "90 requests", description: "Default" },
{ value: "150", label: "150 requests" },
],
},
},
"task.disabledAgents": {
type: "array",
default: [] as string[],
@@ -3353,7 +3396,7 @@ export type TreeFilterMode = SettingValue<"treeFilterMode">;
export interface CompactionSettings {
enabled: boolean;
strategy: "context-full" | "handoff" | "shake" | "off";
strategy: "context-full" | "handoff" | "shake" | "snapcompact" | "off";
thresholdPercent: number;
thresholdTokens: number;
reserveTokens: number;
@@ -3365,6 +3408,7 @@ export interface CompactionSettings {
idleEnabled: boolean;
idleThresholdTokens: number;
idleTimeoutSeconds: number;
supersedeReads: boolean;
}
export interface ContextPromotionSettings {
@@ -99,6 +99,7 @@ function singleResult(options: ExecutorOptions, overrides: Partial<SingleResult>
truncated: false,
durationMs: 1,
tokens: 0,
requests: 0,
...overrides,
};
}
@@ -178,7 +179,7 @@ describe("runEvalAgent", () => {
expect(runSpy).not.toHaveBeenCalled();
});
it("passes the parent execution context and only sets outputSchema when schema is supplied", async () => {
it("passes parent execution options and only sets outputSchema when schema is supplied", async () => {
mockAgents();
const runSpy = vi.spyOn(taskExecutor, "runSubprocess").mockImplementation(async options => singleResult(options));
const abortController = new AbortController();
@@ -186,7 +187,7 @@ describe("runEvalAgent", () => {
const session = makeSession({ depth: 2, activeModel: "p/current", modelString: "p/fallback" });
await runEvalAgent(
{ prompt: " hello ", context: " context ", label: "My Agent", model: "p/override", schema },
{ prompt: " hello ", label: "My Agent", model: "p/override", schema },
{ session, signal: abortController.signal },
);
await runEvalAgent({ prompt: "plain" }, { session });
@@ -199,7 +200,6 @@ describe("runEvalAgent", () => {
expect(firstOptions.parentActiveModelPattern).toBe("p/current");
expect(firstOptions.outputSchema).toBe(schema);
expect(firstOptions.assignment).toBe("hello");
expect(firstOptions.context).toBe("context");
expect(firstOptions.description).toBe("My Agent");
expect(firstOptions.modelOverride).toEqual(["p/override"]);
expect(secondOptions.outputSchema).toBeUndefined();
@@ -542,6 +542,7 @@ describe("agent() through eval runtimes", () => {
recentOutput: [],
toolCount: 0,
tokens: 0,
requests: 0,
cost: 0,
durationMs: 0,
...overrides,
@@ -674,6 +675,7 @@ describe("agent() through eval runtimes", () => {
recentOutput: [],
toolCount: i,
tokens: 0,
requests: 0,
cost: 0,
durationMs: i * 10,
});
+2 -15
View File
@@ -34,7 +34,6 @@ const agentArgsSchema = z.object({
prompt: z.string().min(1, "prompt must be a non-empty string"),
agentType: z.string().min(1).optional(),
model: z.union([z.string().min(1), z.array(z.string().min(1)).min(1)]).optional(),
context: z.string().optional(),
label: z.string().optional(),
schema: z.unknown().optional(),
});
@@ -43,7 +42,6 @@ interface EvalAgentArgs {
prompt: string;
agentType?: string;
model?: string | string[];
context?: string;
label?: string;
schema?: unknown;
}
@@ -135,20 +133,12 @@ function getOutputManager(session: ToolSession): AgentOutputManager {
async function getArtifacts(session: ToolSession): Promise<{
sessionFile: string | null;
artifactsDir: string;
contextFile?: string;
}> {
const sessionFile = session.getSessionFile();
const sessionArtifactsDir = sessionFile ? sessionFile.slice(0, -6) : null;
const artifactsDir = sessionArtifactsDir ?? path.join(os.tmpdir(), `omp-eval-agent-${Snowflake.next()}`);
await fs.mkdir(artifactsDir, { recursive: true });
const shouldWriteConversationContext = session.settings.get("irc.enabled") !== true;
const compactContext = shouldWriteConversationContext ? session.getCompactContext?.() : undefined;
if (!compactContext) return { sessionFile, artifactsDir };
const contextFile = path.join(artifactsDir, "context.md");
await Bun.write(contextFile, compactContext);
return { sessionFile, artifactsDir, contextFile };
return { sessionFile, artifactsDir };
}
function emitProgressStatus(emitStatus: ((event: JsStatusEvent) => void) | undefined, progress: AgentProgress): void {
@@ -246,11 +236,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
};
const parentArtifactManager = options.session.getArtifactManager?.() ?? undefined;
const mcpManager = options.session.mcpManager ?? MCPManager.instance();
const { sessionFile, artifactsDir, contextFile } = await getArtifacts(options.session);
const { sessionFile, artifactsDir } = await getArtifacts(options.session);
const outputManager = getOutputManager(options.session);
const id = await outputManager.allocate(outputIdBase(parsed.label, agentName));
const assignment = parsed.prompt.trim();
const context = trimToUndefined(parsed.context);
// Suspend eval timeout accounting while the subagent owns control. The
// timeout clock restarts once the bridge returns to the cell runtime.
const result = await withBridgeTimeoutPause(options.emitStatus, () =>
@@ -259,7 +248,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
agent: effectiveAgent,
task: renderSubagentPrompt(assignment),
assignment,
context,
description: trimToUndefined(parsed.label),
index: 0,
id,
@@ -271,7 +259,6 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
sessionFile,
persistArtifacts: Boolean(sessionFile),
artifactsDir,
contextFile,
// Eval `agent()` subagents are short-lived programmatic helpers (data
// collection, structured output, parallel() fan-out). LSP server
// cold-start costs tens of seconds and is pure overhead here, so it is
@@ -65,7 +65,7 @@ if (!globalThis.__omp_js_prelude_loaded__) {
};
const agent = async (prompt, opts, ...rest) => {
const o = optionsArg("agent", opts, rest, "{ agentType, model, context, label, schema }");
const o = optionsArg("agent", opts, rest, "{ agentType, model, label, schema }");
const res = await globalThis.__omp_call_tool__("__agent__", { prompt, ...o });
const text = res && typeof res === "object" ? res.text : res;
return hasOwn(o, "schema") ? JSON.parse(text) : text;
+5 -6
View File
@@ -519,21 +519,20 @@ if "__omp_prelude_loaded__" not in globals():
text = res.get("text") if isinstance(res, dict) else res
return json.loads(text) if schema is not None else text
def agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None):
def agent(prompt, *, agent_type="task", model=None, label=None, schema=None):
"""Run a subagent and return its final output.
`agent_type` selects the subagent definition (default "task"). Pass
`model` to override that agent's model, `context` for shared background,
`label` for the output artifact id, and `schema` to request structured
JSON output; when `schema` is supplied the parsed object is returned.
`model` to override that agent's model, `label` for the output artifact
id, and `schema` to request structured JSON output; when `schema` is
supplied the parsed object is returned. Share background by writing a
local:// file and referencing it in the prompt.
"""
args = {"prompt": prompt}
if agent_type is not None:
args["agentType"] = agent_type
if model is not None:
args["model"] = model
if context is not None:
args["context"] = context
if label is not None:
args["label"] = label
if schema is not None:
File diff suppressed because one or more lines are too long
@@ -1023,18 +1023,18 @@
}
function renderTask(name, args, result, ctx) {
const agent = str(args.agent) || '?';
const tasks = Array.isArray(args.tasks) ? args.tasks : [];
const badges = ['agent=' + agent, tasks.length + ' subtask' + (tasks.length === 1 ? '' : 's')];
const badges = [];
if (args.resume) badges.push('resume=' + str(args.resume));
else badges.push('agent=' + (str(args.agent) || '?'));
if (args.id) badges.push('id=' + str(args.id));
if (args.isolated) badges.push('isolated');
let html = toolHead('task', '', badges);
if (tasks.length) {
const description = str(args.description);
const assignment = str(args.assignment);
if (description || assignment) {
html += '<div class="tool-args">';
for (const t of tasks) {
const id = t?.id ? escapeHtml(String(t.id)) : '?';
const desc = t?.description ? escapeHtml(String(t.description)) : '';
html += '<div class="tool-arg"><span class="tool-arg-key">' + id + '</span> ' + desc + '</div>';
}
if (description) html += '<div class="tool-arg"><span class="tool-arg-key">' + escapeHtml(description) + '</span></div>';
if (assignment) html += '<div class="tool-arg">' + escapeHtml(assignment) + '</div>';
html += '</div>';
}
if (result) {
@@ -1479,13 +1479,38 @@
}
function renderIrc(name, args, result, ctx) {
const op = str(args.op) || '?';
const details = result && result.details ? result.details : null;
const op = str(args.op) || (details && str(details.op)) || '?';
const badges = [op];
if (args.to) badges.push('to=' + args.to);
if (args.awaitReply === false) badges.push('no-reply');
if (args.to) badges.push('to=' + str(args.to));
if (op === 'wait' && args.from) badges.push('from=' + str(args.from));
if (args.await) badges.push('await');
if (args.peek) badges.push('peek');
let html = toolHead('irc', '', badges);
if (args.message) html += '<div class="tool-output"><div>' + escapeHtml(String(args.message)) + '</div></div>';
if (result) {
let renderedDetails = false;
if (details && Array.isArray(details.receipts) && details.receipts.length) {
html += '<div class="tool-args">';
for (const receipt of details.receipts) {
const outcome = escapeHtml(String(receipt.outcome)) + (receipt.error ? ' — ' + escapeHtml(String(receipt.error)) : '');
html += '<div class="tool-arg"><span class="tool-arg-key">' + escapeHtml(String(receipt.to)) + '</span> ' + outcome + '</div>';
}
html += '</div>';
renderedDetails = true;
}
if (details && details.waited) {
html += '<div class="tool-output"><div>' + escapeHtml(String(details.waited.from)) + ': ' + escapeHtml(String(details.waited.body)) + '</div></div>';
renderedDetails = true;
}
if (details && Array.isArray(details.inbox) && details.inbox.length) {
html += '<div class="tool-args">';
for (const msg of details.inbox) {
html += '<div class="tool-arg"><span class="tool-arg-key">' + escapeHtml(String(msg.from)) + '</span> ' + escapeHtml(String(msg.body)) + '</div>';
}
html += '</div>';
renderedDetails = true;
}
if (!renderedDetails && result) {
const output = ctx.getResultText();
if (output) html += formatExpandableOutput(output, 8);
}
@@ -103,11 +103,11 @@ export type CustomToolSessionEvent =
| {
reason: "auto_compaction_start";
trigger: "threshold" | "overflow" | "idle" | "incomplete";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
}
| {
reason: "auto_compaction_end";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
result: CompactionResult | undefined;
aborted: boolean;
willRetry: boolean;
@@ -204,13 +204,13 @@ export interface TurnEndEvent {
export interface AutoCompactionStartEvent {
type: "auto_compaction_start";
reason: "threshold" | "overflow" | "idle" | "incomplete";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
}
/** Fired when auto-compaction ends */
export interface AutoCompactionEndEvent {
type: "auto_compaction_end";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
result: CompactionResult | undefined;
aborted: boolean;
willRetry: boolean;
@@ -0,0 +1,113 @@
/**
* Protocol handler for history:// URLs.
*
* Exposes agent transcripts as concise markdown. Live refs render from the
* in-memory message array; parked refs (session disposed, sessionFile
* retained) load read-only from the JSONL session file — no writer, no lock.
*
* URL forms:
* - history:// - Index of all registry agents (id, status, kind, last activity)
* - history://<agentId> - Concise markdown transcript of that agent
*/
import type { AgentRef } from "../registry/agent-registry";
import { AgentRegistry } from "../registry/agent-registry";
import { formatSessionHistoryMarkdown } from "../session/session-history-format";
import { loadSessionMessagesReadOnly } from "../session/session-manager";
import type { InternalResource, InternalUrl, ProtocolHandler, UrlCompletion } from "./types";
/** Humanize a last-activity timestamp as `Ns/Nm/Nh/Nd ago`. */
function formatAgo(timestamp: number): string {
const diffMs = Math.max(0, Date.now() - timestamp);
const secs = Math.floor(diffMs / 1000);
if (secs < 60) return `${secs}s ago`;
const mins = Math.floor(secs / 60);
if (mins < 60) return `${mins}m ago`;
const hours = Math.floor(mins / 60);
if (hours < 24) return `${hours}h ago`;
return `${Math.floor(hours / 24)}d ago`;
}
/**
* Handler for history:// URLs.
*
* Resolves agent ids against the global AgentRegistry, serving transcripts
* for both live and parked agents.
*/
export class HistoryProtocolHandler implements ProtocolHandler {
readonly scheme = "history";
readonly immutable = false;
async resolve(url: InternalUrl): Promise<InternalResource> {
const agentId = url.rawHost || url.hostname;
const registry = AgentRegistry.global();
if (!agentId) {
const content = this.#renderIndex(registry.list());
return {
url: url.href,
content,
contentType: "text/markdown",
size: Buffer.byteLength(content, "utf-8"),
};
}
let ref = registry.get(agentId);
if (!ref) {
// Case-insensitive fallback: agent ids are human-typed (e.g. AuthLoader).
const lower = agentId.toLowerCase();
ref = registry.list().find(candidate => candidate.id.toLowerCase() === lower);
}
if (!ref) {
const known = registry.list().map(candidate => candidate.id);
const knownStr = known.length > 0 ? known.join(", ") : "none";
throw new Error(`Unknown agent: ${agentId}\nKnown agents: ${knownStr}\nList all with history://`);
}
const notes: string[] = [];
let messages: unknown[];
if (ref.session) {
messages = ref.session.messages;
notes.push("Source: live session");
} else if (ref.sessionFile) {
messages = await loadSessionMessagesReadOnly(ref.sessionFile);
notes.push(`Source: session file (read-only, ${ref.status})`);
} else {
throw new Error(`Agent ${ref.id} has no transcript: session is gone and no session file was retained`);
}
const content = formatSessionHistoryMarkdown(messages, { title: `${ref.id} (${ref.status})` });
return {
url: url.href,
content,
contentType: "text/markdown",
size: Buffer.byteLength(content, "utf-8"),
sourcePath: ref.sessionFile ?? undefined,
notes,
};
}
#renderIndex(refs: AgentRef[]): string {
const lines: string[] = ["# Agents", ""];
if (refs.length === 0) {
lines.push("No agents registered.");
return `${lines.join("\n")}\n`;
}
lines.push("| id | status | kind | parent | last activity |", "|---|---|---|---|---|");
for (const ref of refs) {
lines.push(
`| ${ref.id} | ${ref.status} | ${ref.kind} | ${ref.parentId ?? "—"} | ${formatAgo(ref.lastActivity)} |`,
);
}
lines.push("", "Read a transcript with `read history://<id>`.");
return `${lines.join("\n")}\n`;
}
async complete(): Promise<UrlCompletion[]> {
return AgentRegistry.global()
.list()
.map(ref => ({
value: ref.id,
description: `${ref.status} · ${ref.kind}${ref.parentId ? ` · parent ${ref.parentId}` : ""}`,
}));
}
}
@@ -10,6 +10,7 @@
export * from "./agent-protocol";
export * from "./artifact-protocol";
export * from "./history-protocol";
export * from "./issue-pr-protocol";
export * from "./json-query";
export * from "./local-protocol";
@@ -1,5 +1,5 @@
/**
* Internal URL router for internal protocols (`agent://`, `artifact://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`).
* Internal URL router for internal protocols (`agent://`, `artifact://`, `history://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`).
*
* One process-global router with one handler per scheme. Access via
* `InternalUrlRouter.instance()`. Handlers are stateless; per-session and
@@ -7,6 +7,7 @@
*/
import { AgentProtocolHandler } from "./agent-protocol";
import { ArtifactProtocolHandler } from "./artifact-protocol";
import { HistoryProtocolHandler } from "./history-protocol";
import { IssueProtocolHandler, PrProtocolHandler } from "./issue-pr-protocol";
import { LocalProtocolHandler } from "./local-protocol";
import { McpProtocolHandler } from "./mcp-protocol";
@@ -35,6 +36,7 @@ export class InternalUrlRouter {
this.register(new McpProtocolHandler());
this.register(new IssueProtocolHandler());
this.register(new PrProtocolHandler());
this.register(new HistoryProtocolHandler());
}
/** Process-global router instance. */
@@ -1,7 +1,7 @@
/**
* Types for the internal URL routing system.
*
* Internal URLs (`agent://`, `artifact://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`) are resolved by tools like read,
* Internal URLs (`agent://`, `artifact://`, `history://`, `issue://`, `local://`, `mcp://`, `memory://`, `omp://`, `pr://`, `rule://`, `skill://`, and `vault://`) are resolved by tools like read,
* providing access to agent outputs and server resources without exposing filesystem paths.
*/
+275
View File
@@ -0,0 +1,275 @@
/**
* IrcBus - Process-global mailbox bus for agent-to-agent messaging.
*
* Replaces the old auto-reply model: a `send` never blocks on the recipient
* generating anything. Delivery resolves the recipient via the global
* AgentRegistry — parked agents are revived through the
* AgentLifecycleManager, idle agents are woken with a real turn, and busy
* agents receive the message as a non-interrupting aside at the next step
* boundary (see AgentSession.deliverIrcMessage). Replies are real turns by
* the recipient, observed via `wait`.
*/
import { logger, Snowflake } from "@oh-my-pi/pi-utils";
import { AgentLifecycleManager } from "../registry/agent-lifecycle";
import { AgentRegistry, MAIN_AGENT_ID } from "../registry/agent-registry";
import type { CustomMessage } from "../session/messages";
export interface IrcMessage {
id: string;
/** Sender agent id. */
from: string;
/** Recipient agent id (resolved; "all" is expanded by the tool, not stored). */
to: string;
body: string;
ts: number;
/** Message id being answered. */
replyTo?: string;
}
export interface IrcDeliveryReceipt {
to: string;
outcome: "injected" | "woken" | "revived" | "failed";
error?: string;
}
interface IrcWaiter {
from?: string;
resolve: (msg: IrcMessage) => void;
cancel: () => void;
}
/** Mailbox cap per agent; oldest messages are dropped beyond it. */
const MAILBOX_CAP = 100;
export class IrcBus {
static #global: IrcBus | undefined;
static global(): IrcBus {
if (!IrcBus.#global) {
IrcBus.#global = new IrcBus();
}
return IrcBus.#global;
}
/** Reset the global bus. Test-only. */
static resetGlobalForTests(): void {
IrcBus.#global = undefined;
}
readonly #registry: AgentRegistry;
readonly #lifecycle: () => AgentLifecycleManager;
readonly #mailboxes = new Map<string, IrcMessage[]>();
readonly #waiters = new Map<string, IrcWaiter[]>();
constructor(registry: AgentRegistry = AgentRegistry.global(), lifecycle?: AgentLifecycleManager) {
this.#registry = registry;
// Lazy: the lifecycle global self-constructs against the global registry,
// so only touch it when a parked recipient actually needs reviving.
this.#lifecycle = () => lifecycle ?? AgentLifecycleManager.global();
}
/**
* Fire-and-forget delivery. Never blocks on the recipient generating
* anything: the receipt reports how the message reached the recipient
* (waiter/aside = "injected", idle wake = "woken", park revival =
* "revived"), not what they did with it.
*/
async send(msg: Omit<IrcMessage, "id" | "ts">): Promise<IrcDeliveryReceipt> {
const message: IrcMessage = { ...msg, id: Snowflake.next(), ts: Date.now() };
const ref = this.#registry.get(message.to);
if (!ref || ref.status === "aborted") {
return { to: message.to, outcome: "failed", error: `Unknown or terminated agent "${message.to}".` };
}
let revived = false;
if (ref.status === "parked") {
try {
await this.#lifecycle().ensureLive(message.to);
revived = true;
} catch (error) {
return {
to: message.to,
outcome: "failed",
error: error instanceof Error ? error.message : String(error),
};
}
}
// A pending `wait` from the recipient consumes the message directly —
// it is returned from their irc tool call and never hits the inbox or
// the session injection path.
const waiter = this.#takeMatchingWaiter(message.to, message.from);
if (waiter) {
waiter.resolve(message);
this.#relayToMainUi(message);
return { to: message.to, outcome: revived ? "revived" : "injected" };
}
const session = this.#registry.get(message.to)?.session;
if (!session) {
return { to: message.to, outcome: "failed", error: `Agent "${message.to}" has no live session.` };
}
this.#enqueue(message);
try {
const delivery = await session.deliverIrcMessage(message);
this.#relayToMainUi(message);
return { to: message.to, outcome: revived ? "revived" : delivery };
} catch (error) {
return {
to: message.to,
outcome: "failed",
error: error instanceof Error ? error.message : String(error),
};
}
}
/**
* Block until a message for `agentId` (optionally from `filter.from`)
* arrives; consume + return it. Null on timeout (`timeoutMs <= 0` waits
* forever). Rejects when `signal` aborts.
*/
async wait(
agentId: string,
filter: { from?: string },
timeoutMs: number,
signal?: AbortSignal,
): Promise<IrcMessage | null> {
if (signal?.aborted) {
throw signal.reason instanceof Error ? signal.reason : new Error("IRC wait aborted");
}
// Already-pending mail satisfies the wait without parking a waiter.
const pending = this.#takeFromMailbox(agentId, filter.from);
if (pending) return pending;
const { promise, resolve, reject } = Promise.withResolvers<IrcMessage | null>();
let timer: NodeJS.Timeout | undefined;
let onAbort: (() => void) | undefined;
const waiter: IrcWaiter = {
from: filter.from,
resolve: msg => {
cleanup();
resolve(msg);
},
cancel: () => {
cleanup();
},
};
const cleanup = (): void => {
this.#removeWaiter(agentId, waiter);
clearTimeout(timer);
if (signal && onAbort) signal.removeEventListener("abort", onAbort);
};
if (signal) {
onAbort = () => {
cleanup();
reject(signal.reason instanceof Error ? signal.reason : new Error("IRC wait aborted"));
};
signal.addEventListener("abort", onAbort, { once: true });
}
if (timeoutMs > 0) {
timer = setTimeout(() => {
cleanup();
resolve(null);
}, timeoutMs);
timer.unref?.();
}
let waiters = this.#waiters.get(agentId);
if (!waiters) {
waiters = [];
this.#waiters.set(agentId, waiters);
}
waiters.push(waiter);
return promise;
}
/** Drain (or peek) pending messages for `agentId`. */
inbox(agentId: string, opts?: { peek?: boolean }): IrcMessage[] {
const mailbox = this.#mailboxes.get(agentId);
if (!mailbox || mailbox.length === 0) return [];
if (opts?.peek) return [...mailbox];
this.#mailboxes.delete(agentId);
return mailbox;
}
unreadCount(agentId: string): number {
return this.#mailboxes.get(agentId)?.length ?? 0;
}
#enqueue(message: IrcMessage): void {
let mailbox = this.#mailboxes.get(message.to);
if (!mailbox) {
mailbox = [];
this.#mailboxes.set(message.to, mailbox);
}
mailbox.push(message);
if (mailbox.length > MAILBOX_CAP) {
const dropped = mailbox.shift();
logger.debug("IrcBus: mailbox full, dropped oldest message", {
agentId: message.to,
droppedId: dropped?.id,
droppedFrom: dropped?.from,
});
}
}
/** Resolve the OLDEST waiter for `agentId` whose from-filter accepts `from`. */
#takeMatchingWaiter(agentId: string, from: string): IrcWaiter | undefined {
const waiters = this.#waiters.get(agentId);
if (!waiters) return undefined;
const index = waiters.findIndex(waiter => !waiter.from || waiter.from === from);
if (index === -1) return undefined;
const [waiter] = waiters.splice(index, 1);
if (waiters.length === 0) this.#waiters.delete(agentId);
return waiter;
}
#removeWaiter(agentId: string, waiter: IrcWaiter): void {
const waiters = this.#waiters.get(agentId);
if (!waiters) return;
const index = waiters.indexOf(waiter);
if (index !== -1) waiters.splice(index, 1);
if (waiters.length === 0) this.#waiters.delete(agentId);
}
#takeFromMailbox(agentId: string, from?: string): IrcMessage | undefined {
const mailbox = this.#mailboxes.get(agentId);
if (!mailbox) return undefined;
const index = from ? mailbox.findIndex(msg => msg.from === from) : 0;
if (index === -1 || mailbox.length === 0) return undefined;
const [message] = mailbox.splice(index, 1);
if (mailbox.length === 0) this.#mailboxes.delete(agentId);
return message;
}
/**
* Surface agent↔agent traffic as a display-only card on the main session
* UI. Skipped when the main agent is the recipient — its own
* `deliverIrcMessage` (or `wait` tool result) already shows the message.
*/
#relayToMainUi(message: IrcMessage): void {
if (message.to === MAIN_AGENT_ID) return;
const mainSession = this.#registry.get(MAIN_AGENT_ID)?.session;
if (!mainSession) return;
const record: CustomMessage = {
role: "custom",
customType: "irc:relay",
content: `[IRC \`${message.from}\` → \`${message.to}\`]\n\n${message.body}`,
display: true,
details: { from: message.from, to: message.to, body: message.body },
attribution: "agent",
timestamp: message.ts,
};
try {
mainSession.emitIrcRelayObservation(record);
} catch (error) {
// Display-only forwarding must never affect delivery semantics.
logger.debug("IrcBus: main UI relay failed", { to: message.to, error: String(error) });
}
}
}
+51 -27
View File
@@ -51,10 +51,10 @@ import { ExtensionRunner } from "./extensibility/extensions/runner";
import type { ExtensionUIContext } from "./extensibility/extensions/types";
import { scheduleMarketplaceAutoUpdate } from "./extensibility/plugins/marketplace-auto-update";
import type { MCPManager } from "./mcp";
import { WelcomeComponent } from "./modes/components/welcome";
import { InteractiveMode } from "./modes/interactive-mode";
import type { PrintModeOptions } from "./modes/print-mode";
import { CURRENT_SETUP_VERSION } from "./modes/setup-version";
import { StartupInput } from "./modes/startup-input";
import { initTheme, stopThemeWatcher } from "./modes/theme/theme";
import type { SubmittedUserInput } from "./modes/types";
import {
@@ -97,33 +97,31 @@ function maybeShowStartupSplash(options: {
modelName?: string;
providerName?: string;
lspServers?: LspStartupServerInfo[];
}): void {
if (!options.isInteractive) return;
if (options.resuming || options.quiet) return;
if ($env.PI_TIMING) return;
if (!process.stdin.isTTY || !process.stdout.isTTY) return;
}): StartupInput | undefined {
if (!options.isInteractive) return undefined;
if (options.resuming || options.quiet) return undefined;
if ($env.PI_TIMING) return undefined;
if (!process.stdin.isTTY || !process.stdout.isTTY) return undefined;
// First-run launches go straight into the setup wizard, which paints its own
// splash — keep the minimal two-line notice there.
if (options.setupPending) {
process.stdout.write(`${chalk.dim(`omp ${options.version}`)}\n${chalk.dim("Initializing session…")}\n`);
return;
return undefined;
}
// Render the same welcome box the TUI paints first: recent sessions as a
// loading placeholder (the fixed slot count keeps the box height stable) and
// the logo held on the intro animation's first frame so the in-TUI intro
// continues from the frame shown here. Clearing the screen first puts the
// box at the same origin the TUI's first full paint (clearScrollback) uses,
// so the live welcome replaces this frame in place without shifting.
const welcome = new WelcomeComponent(
options.version,
options.modelName ?? "",
options.providerName ?? "",
null,
options.lspServers ?? [],
);
welcome.holdIntroFirstFrame();
const lines = welcome.render(process.stdout.columns || 80);
process.stdout.write(`\x1b[2J\x1b[H\x1b[3J\n${lines.join("\n")}\n`);
// Paint the same welcome box the TUI paints first (recent sessions as a
// loading placeholder, logo held on the intro animation's first frame) plus
// a live editor, and start capturing raw stdin so the user can type — and
// even submit — while the session loads in the background. The TUI's first
// full paint (clearScrollback) replaces this frame in place; the editor
// instance itself is handed to InteractiveMode so nothing typed is lost.
const startupInput = new StartupInput({
version: options.version,
modelName: options.modelName ?? "",
providerName: options.providerName ?? "",
lspServers: options.lspServers ?? [],
});
startupInput.start();
return startupInput;
}
async function checkForNewVersion(currentVersion: string): Promise<string | undefined> {
@@ -372,6 +370,7 @@ async function runInteractiveMode(
initialMessage?: string,
initialImages?: ImageContent[],
titleSystemPrompt?: string,
startupInput?: StartupInput,
): Promise<void> {
const mode = new InteractiveMode(
session,
@@ -382,6 +381,7 @@ async function runInteractiveMode(
mcpManager,
eventBus,
titleSystemPrompt,
startupInput?.editor,
);
// Cold-launch gate: the full setup wizard (every scene + the overlay and
@@ -400,6 +400,12 @@ async function runInteractiveMode(
})
: [];
// Hand the terminal over: stop the pre-TUI capture (restoring cooked mode so
// ProcessTerminal records the correct prior raw state) right before the TUI
// grabs stdin. Keystrokes typed during init's awaits stay OS-buffered and
// flow into the TUI once it resumes stdin.
startupInput?.detach();
await mode.init({
suppressWelcomeIntro: resuming || setupScenes.length > 0,
clearInitialTerminalHistory: true,
@@ -425,7 +431,7 @@ async function runInteractiveMode(
// Every in-process session load also uses `clearTerminalHistory`; cold launch
// follows the same clean-cutover path instead of preserving a previous run's
// transcript above the fresh one.
mode.renderInitialMessages(undefined, { preserveExistingChat: true, clearTerminalHistory: true });
mode.renderInitialMessages({ preserveExistingChat: true, clearTerminalHistory: true });
for (const notify of notifs) {
if (!notify) {
@@ -460,8 +466,19 @@ async function runInteractiveMode(
}
}
const startupSubmissions = [...(startupInput?.queuedSubmissions ?? [])];
while (true) {
const input = await mode.getUserInput();
const inputPromise = mode.getUserInput();
const queuedText = startupSubmissions.shift();
if (queuedText !== undefined) {
// Replay through the real submit pipeline (slash commands, bash/python
// modes, title generation) exactly as if Enter were pressed now.
await mode.editor.onSubmit?.(queuedText);
// Handled inline (e.g. a slash command) without consuming the pending
// input wait — replay the next queued item on the next iteration.
if (mode.onInputCallback) continue;
}
const input = await inputPromise;
await submitInteractiveInput(mode, session, input);
}
}
@@ -1264,7 +1281,7 @@ export async function runRootCommand(
settingsInstance.get("lsp.lazy") ? "available" : "connecting",
)
: [];
maybeShowStartupSplash({
const startupInput = maybeShowStartupSplash({
isInteractive,
resuming: Boolean(parsedArgs.continue || parsedArgs.resume || parsedArgs.fork),
quiet: settingsInstance.get("startup.quiet"),
@@ -1275,6 +1292,9 @@ export async function runRootCommand(
lspServers: splashLspServers,
});
// TEMP-SMOKE: stretch the startup gap for VHS verification. REMOVE.
if (process.env.PI_DEBUG_SLOW_START) await Bun.sleep(Number(process.env.PI_DEBUG_SLOW_START));
const { session, setToolUIContext, modelFallbackMessage, lspServers, mcpManager } = await createSession({
...sessionOptions,
eventBus,
@@ -1322,7 +1342,10 @@ export async function runRootCommand(
return `${scopedModel.model.id}${thinkingStr}`;
})
.join(", ");
process.stdout.write(`${chalk.dim(`Model scope: ${modelList} ${chalk.gray("(Ctrl+P to cycle)")}`)}\n`);
// Routed through the TUI (not stdout): the startup capture owns the
// terminal in raw mode here, and the TUI's first clearScrollback paint
// would wipe a pre-TUI line anyway.
notifs.push({ kind: "info", message: `Model scope: ${modelList} (Ctrl+P to cycle)` });
}
if ($env.PI_TIMING) {
@@ -1350,6 +1373,7 @@ export async function runRootCommand(
initialMessage,
initialImages,
titleSystemPrompt,
startupInput,
);
} else {
// Branch-only single-shot runner: keep print-mode code out of normal interactive startup.
@@ -1,51 +1,87 @@
import { Box, Markdown, Spacer, Text } from "@oh-my-pi/pi-tui";
import { Box, type Component, Markdown } from "@oh-my-pi/pi-tui";
import { getMarkdownTheme, theme } from "../../modes/theme/theme";
import type { CompactionSummaryMessage } from "../../session/messages";
/**
* Component that renders a compaction message with collapsed/expanded state.
* Uses same background color as hook messages for visual consistency.
* Compaction point in the transcript, rendered as a slim horizontal divider:
*
* ──────── 📷 compacted · ctrl+o ────────
*
* The conversation above the divider stays visible (display transcript keeps
* full history); only the LLM context was reset. Expanding (ctrl+o) reveals
* the compaction summary below the divider.
*/
export class CompactionSummaryMessageComponent extends Box {
export class CompactionSummaryMessageComponent implements Component {
#expanded = false;
#cache?: { width: number; lines: string[] };
#detail?: Box;
constructor(private readonly message: CompactionSummaryMessage) {
super(1, 1, t => theme.bg("customMessageBg", t));
this.#updateDisplay();
}
constructor(private readonly message: CompactionSummaryMessage) {}
setExpanded(expanded: boolean): void {
if (this.#expanded === expanded) return;
this.#expanded = expanded;
this.#updateDisplay();
this.#cache = undefined;
}
override invalidate(): void {
super.invalidate();
this.#updateDisplay();
invalidate(): void {
this.#cache = undefined;
// Theme may have changed — rebuild the detail box lazily on next render.
this.#detail = undefined;
}
#updateDisplay(): void {
this.clear();
const tokenStr = this.message.tokensBefore.toLocaleString();
const label = theme.fg("customMessageLabel", theme.bold("[compaction]"));
this.addChild(new Text(label, 0, 0));
this.addChild(new Spacer(1));
if (this.#expanded) {
const header = `**Compacted from ${tokenStr} tokens**\n\n`;
this.addChild(
new Markdown(header + this.message.summary, 0, 0, getMarkdownTheme(), {
color: (text: string) => theme.fg("customMessageText", text),
}),
);
} else {
this.addChild(
new Text(theme.fg("customMessageText", `Compacted from ${tokenStr} tokens (ctrl+o to expand)`), 0, 0),
);
if (this.message.shortSummary) {
this.addChild(new Text(theme.fg("customMessageText", this.message.shortSummary), 0, 1));
}
render(width: number): readonly string[] {
width = Math.max(1, width);
if (this.#cache?.width === width) {
return this.#cache.lines;
}
const lines = this.#expanded
? ["", this.#divider(width), "", ...this.#detailBox().render(width)]
: ["", this.#divider(width), ""];
this.#cache = { width, lines };
return lines;
}
#divider(width: number): string {
const rule = theme.tree.horizontal;
const label = `${theme.icon.camera} compacted`;
// sep.dot ships pre-padded (" · "); trim so the hint joins with single spaces.
const hint = `${theme.sep.dot.trim()} ctrl+o`;
const plainWidth = Bun.stringWidth(`${label} ${hint}`, { countAnsiEscapeCodes: false });
// ` label hint ` framed by rules on both sides.
const remaining = width - plainWidth - 2;
if (remaining < 4) {
// Too narrow for a framed rule — emit the bare label.
return theme.fg("muted", label);
}
const left = Math.floor(remaining / 2);
const right = remaining - left;
return (
theme.fg("dim", rule.repeat(left)) +
` ${theme.fg("muted", label)} ${theme.fg("dim", hint)} ` +
theme.fg("dim", rule.repeat(right))
);
}
#detailBox(): Box {
if (this.#detail) return this.#detail;
const box = new Box(1, 1, t => theme.bg("customMessageBg", t));
const tokenStr = this.message.tokensBefore.toLocaleString();
const frameCount = this.message.images?.length ?? 0;
const frameNote =
frameCount > 0 ? `\n\n_${frameCount} snapcompact frame${frameCount === 1 ? "" : "s"} attached_` : "";
box.addChild(
new Markdown(
`**Compacted from ${tokenStr} tokens**\n\n${this.message.summary}${frameNote}`,
0,
0,
getMarkdownTheme(),
{
color: (text: string) => theme.fg("customMessageText", text),
},
),
);
this.#detail = box;
return box;
}
}
@@ -175,6 +175,8 @@ export class CustomEditor extends Editor {
onDequeue?: () => void;
/** Called when Caps Lock is pressed. */
onCapsLock?: () => void;
/** Called when left-arrow is pressed while the editor is empty (cursor necessarily at start). */
onLeftAtStart?: () => void;
/** Custom key handlers from extensions and non-built-in app actions. */
#customKeyHandlers = new Map<KeyId, () => void>();
@@ -257,6 +259,14 @@ export class CustomEditor extends Editor {
const parsedKey = parseKey(data);
const canonical = parsedKey !== undefined ? canonicalKeyId(parsedKey) : undefined;
// Left-arrow on an empty editor: surface for the agent-hub double-tap
// gesture. Plain "left" only — modified arrows and any in-text cursor
// movement fall through to normal handling.
if (canonical === "left" && this.onLeftAtStart && this.getText().trim() === "") {
this.onLeftAtStart();
return;
}
if (canonical !== undefined) {
// Intercept configured image paste (async - fires and handles result)
if (this.#matchesAction(canonical, "app.clipboard.pasteImage") && this.onPasteImage) {
@@ -19,6 +19,7 @@ import type { Theme } from "../../modes/theme/theme";
import { theme } from "../../modes/theme/theme";
import { BASH_DEFAULT_PREVIEW_LINES } from "../../tools/bash";
import { EVAL_DEFAULT_PREVIEW_LINES } from "../../tools/eval";
import { isWaitingPollDetails } from "../../tools/job";
import {
formatArgsInline,
JSON_TREE_MAX_DEPTH_COLLAPSED,
@@ -194,6 +195,11 @@ export class ToolExecutionComponent extends Container {
// sealed the block stays in the transcript's repaintable live region so a
// late result still repaints instead of stranding the streaming preview.
#sealed = false;
// A `job` poll result whose watched jobs are all still running. Such a
// block never finalizes (stays in the transcript live region) so a
// follow-up `job` call can displace it instead of stacking another
// "waiting on N jobs" frame. Cleared by `seal()`.
#displaceable = false;
#renderState: {
spinnerFrame?: number;
expanded: boolean;
@@ -359,6 +365,11 @@ export class ToolExecutionComponent extends Container {
): void {
this.#result = result;
this.#isPartial = isPartial;
// A `job` poll that found every watched job still running is transient
// "still waiting" chrome; keep the block displaceable so the next `job`
// call replaces it instead of stacking another waiting frame (see the
// event controller's displaceable-poll bookkeeping).
this.#displaceable = this.#toolName === "job" && result.isError !== true && isWaitingPollDetails(result.details);
// When tool is complete, ensure args are marked complete so spinner stops
if (!isPartial) {
this.#argsComplete = true;
@@ -425,7 +436,11 @@ export class ToolExecutionComponent extends Container {
(this.#result?.details as { async?: { state?: string } } | undefined)?.async?.state === "running";
const isBackgroundAsyncTask = this.#toolName === "task" && isBackgroundAsyncRunning;
const isPartialTask = this.#isPartial && this.#toolName === "task" && !isBackgroundAsyncTask;
const needsSpinner = isStreamingArgs || isPartialTask;
// A displaceable waiting poll keeps its spinner ticking: it reads as one
// persistent live poll, and the changing leading glyph keeps the
// transcript's stable-prefix ratchet from committing rows of a block
// that a follow-up `job` call may remove.
const needsSpinner = isStreamingArgs || isPartialTask || this.isDisplaceableBlock();
if (needsSpinner && !this.#spinnerInterval) {
const now = performance.now();
const frameCount = theme.spinnerFrames.length;
@@ -513,6 +528,9 @@ export class ToolExecutionComponent extends Container {
isTranscriptBlockFinalized(): boolean {
if (this.#sealed) return true;
if (this.#result === undefined) return false;
// A displaceable waiting poll stays live: its rows are kept out of
// native scrollback so a follow-up `job` call can remove the block.
if (this.#displaceable) return false;
if (!this.#isPartial) return true;
// Partial result: a background async tool is accepted to freeze (the agent
// continues while it runs and would otherwise pin an unbounded live region);
@@ -528,11 +546,23 @@ export class ToolExecutionComponent extends Container {
seal(): void {
if (this.#sealed) return;
this.#sealed = true;
this.#displaceable = false;
this.stopAnimation();
this.#updateDisplay();
this.#ui.requestRender();
}
/**
* Whether this block is a waiting `job` poll (every watched job still
* running) that has not been sealed. Such a block never finalized, so none
* of its rows entered native scrollback (the ticking spinner keeps the
* stable-prefix ratchet at zero) and the whole block can be removed when a
* follow-up `job` call supersedes it.
*/
isDisplaceableBlock(): boolean {
return this.#displaceable && !this.#sealed;
}
/**
* Stop spinner animation and cleanup resources.
*/
@@ -77,6 +77,11 @@ export class EventController {
// Insertion-ordered IRC cards not yet retired; values are the transcript
// components each card contributed (see #retireIrcCard for the guard).
#liveIrcCards = new Map<string, Component[]>();
// Most recent `job` tool block whose result still had every watched job
// running. Kept un-finalized (live) so the next `job` call displaces it —
// one persistent poll instead of a stack of "waiting on N jobs" frames —
// and sealed in place the moment anything else lands below it.
#displaceablePollComponent: ToolExecutionComponent | undefined = undefined;
#streamingReveal: StreamingRevealController;
#handlers: AgentSessionEventHandlers;
@@ -282,6 +287,7 @@ export class EventController {
const signature = `${textContent}\u0000${imageCount}`;
this.#resetReadGroup();
this.#resolveDisplaceablePoll();
const wasOptimistic = this.ctx.optimisticUserMessageSignature === signature;
const wasLocallySubmitted = this.ctx.locallySubmittedUserSignatures.delete(signature) || wasOptimistic;
if (!wasOptimistic) {
@@ -389,6 +395,28 @@ export class EventController {
}
}
/**
* Resolve the pending displaceable poll block before the next block lands.
* A follow-up `job` call displaces it — the stale "waiting on N jobs" frame
* is removed so repeated polls read as one persistent poll — while anything
* else seals it in place as final history. Removal is safe only because a
* displaceable block never finalizes: commits stop at the first live block,
* so none of its rows have entered native scrollback (see
* ToolExecutionComponent.isDisplaceableBlock).
*/
#resolveDisplaceablePoll(nextToolName?: string): void {
const previous = this.#displaceablePollComponent;
if (!previous) return;
this.#displaceablePollComponent = undefined;
if (nextToolName === "job" && previous.isDisplaceableBlock()) {
this.ctx.chatContainer.removeChild(previous);
}
// Sealing stops the waiting-poll spinner and freezes the block (for a
// just-removed component it only clears the animation timer).
previous.seal();
this.ctx.ui.requestRender();
}
async #handleNotice(event: Extract<AgentSessionEvent, { type: "notice" }>): Promise<void> {
const message = event.source ? `${event.source}: ${event.message}` : event.message;
if (event.level === "error") {
@@ -444,6 +472,7 @@ export class EventController {
continue;
}
if (!readArgsTargetInternalUrl(content.arguments)) {
if (!this.ctx.pendingTools.has(content.id)) this.#resolveDisplaceablePoll(content.name);
this.#trackReadToolCall(content.id, content.arguments);
const component = this.ctx.pendingTools.get(content.id);
if (component) {
@@ -465,6 +494,7 @@ export class EventController {
? { ...content.arguments, __partialJson: content.partialJson }
: content.arguments;
if (!this.ctx.pendingTools.has(content.id)) {
this.#resolveDisplaceablePoll(content.name);
this.#resetReadGroup();
const tool = this.ctx.session.getToolByName(content.name);
const component = new ToolExecutionComponent(
@@ -561,6 +591,9 @@ export class EventController {
component.seal();
}
}
// These calls will never produce a result either, so the tracked
// waiting poll cannot be displaced anymore — freeze it in place.
this.#resolveDisplaceablePoll();
}
this.#lastAssistantComponent = this.ctx.streamingComponent;
this.#lastAssistantComponent.setUsageInfo(event.message.usage);
@@ -589,6 +622,7 @@ export class EventController {
async #handleToolExecutionStart(event: Extract<AgentSessionEvent, { type: "tool_execution_start" }>): Promise<void> {
this.#updateWorkingMessageFromIntent(event.intent);
if (!this.ctx.pendingTools.has(event.toolCallId)) {
this.#resolveDisplaceablePoll(event.toolName);
if (event.toolName === "read" && readArgsHaveTarget(event.args) && !readArgsTargetInternalUrl(event.args)) {
this.#trackReadToolCall(event.toolCallId, event.args);
const component = this.ctx.pendingTools.get(event.toolCallId);
@@ -697,6 +731,14 @@ export class EventController {
this.ctx.pendingTools.delete(event.toolCallId);
this.#backgroundToolCallIds.delete(event.toolCallId);
}
if (
event.toolName === "job" &&
component instanceof ToolExecutionComponent &&
component.isDisplaceableBlock()
) {
// Remember the waiting poll so the next `job` call can displace it.
this.#displaceablePollComponent = component;
}
this.ctx.ui.requestRender();
}
}
@@ -759,6 +801,9 @@ export class EventController {
this.#readToolCallArgs.clear();
this.#readToolCallAssistantComponents.clear();
this.#resetReadGroup();
// The turn is over: nothing else lands this turn, so the waiting poll is
// final history — seal it instead of letting its spinner tick while idle.
this.#resolveDisplaceablePoll();
this.#lastAssistantComponent = undefined;
this.ctx.ui.requestRender();
this.#scheduleIdleCompaction();
@@ -140,7 +140,7 @@ export class ExtensionUiController {
reload: async () => {
await this.ctx.session.reload();
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
this.ctx.showStatus("Reloaded session");
},
@@ -197,7 +197,7 @@ export class ExtensionUiController {
// Update UI
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
this.ctx.editor.setText(result.selectedText);
this.ctx.showStatus("Branched to new session");
@@ -212,7 +212,7 @@ export class ExtensionUiController {
// Update UI
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
if (result.editorText && !this.ctx.editor.getText().trim()) {
this.ctx.editor.setText(result.editorText);
@@ -230,7 +230,7 @@ export class ExtensionUiController {
}
setSessionTerminalTitle(this.ctx.sessionManager.getSessionName(), this.ctx.sessionManager.getCwd());
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
return { cancelled: false };
},
@@ -376,7 +376,7 @@ export class ExtensionUiController {
reload: async () => {
await this.ctx.session.reload();
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
this.ctx.showStatus("Reloaded session");
},
@@ -426,7 +426,7 @@ export class ExtensionUiController {
// Update UI
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
this.ctx.editor.setText(result.selectedText);
this.ctx.showStatus("Branched to new session");
@@ -441,7 +441,7 @@ export class ExtensionUiController {
// Update UI
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
if (result.editorText && !this.ctx.editor.getText().trim()) {
this.ctx.editor.setText(result.editorText);
@@ -458,7 +458,7 @@ export class ExtensionUiController {
return { cancelled: true };
}
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
return { cancelled: false };
},
@@ -235,10 +235,26 @@ export class InputController {
for (const key of this.ctx.keybindings.getKeys("app.clipboard.copyLine")) {
this.ctx.editor.setCustomKeyHandler(key, () => this.handleCopyCurrentLine());
}
for (const key of this.ctx.keybindings.getKeys("app.session.observe")) {
this.ctx.editor.setCustomKeyHandler(key, () => this.ctx.showSessionObserver());
const hubKeys = new Set([
...this.ctx.keybindings.getKeys("app.agents.hub"),
...this.ctx.keybindings.getKeys("app.session.observe"),
]);
for (const key of hubKeys) {
this.ctx.editor.setCustomKeyHandler(key, () => this.ctx.showAgentHub());
}
// Double-tap left arrow on an empty editor opens the agent hub — same
// 500ms window as the double-escape state machine above.
this.ctx.editor.onLeftAtStart = () => {
const now = Date.now();
if (now - this.ctx.lastLeftTapTime < 500) {
this.ctx.lastLeftTapTime = 0;
this.ctx.showAgentHub();
} else {
this.ctx.lastLeftTapTime = now;
}
};
this.#setupEnhancedPaste();
this.ctx.editor.onChange = (text: string) => {
@@ -40,6 +40,7 @@ import { shortenPath } from "../../tools/render-utils";
import { copyToClipboard } from "../../utils/clipboard";
import { setSessionTerminalTitle } from "../../utils/title-generator";
import { AgentDashboard } from "../components/agent-dashboard";
import { AgentHubOverlayComponent } from "../components/agent-hub";
import { AssistantMessageComponent } from "../components/assistant-message";
import { CopySelectorComponent } from "../components/copy-selector";
import { ExtensionDashboard } from "../components/extensions";
@@ -47,7 +48,6 @@ import { HistorySearchComponent } from "../components/history-search";
import { ModelSelectorComponent } from "../components/model-selector";
import { OAuthSelectorComponent } from "../components/oauth-selector";
import { PluginSelectorComponent } from "../components/plugin-selector";
import { SessionObserverOverlayComponent } from "../components/session-observer-overlay";
import { SessionSelectorComponent } from "../components/session-selector";
import { SettingsSelectorComponent } from "../components/settings-selector";
import { ToolExecutionComponent } from "../components/tool-execution";
@@ -578,7 +578,7 @@ export class SelectorController {
}
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
this.ctx.editor.setText(result.selectedText);
done();
this.ctx.showStatus("Branched to new session");
@@ -719,9 +719,10 @@ export class SelectorController {
return;
}
// Update UI — pass the context built by navigateTree to skip a second O(N) walk.
// Update UI — rebuild the display transcript for the new leaf (the
// context from navigateTree is the LLM context, not the transcript).
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(result.sessionContext, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
if (result.editorText && !this.ctx.editor.getText().trim()) {
this.ctx.editor.setText(result.editorText);
@@ -846,7 +847,7 @@ export class SelectorController {
this.ctx.statusLine.setSessionStartTime(Date.now());
this.ctx.updateEditorTopBorder();
this.ctx.updateEditorBorderColor();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
this.ctx.ui.requestRender(true, { clearScrollback: true });
return true;
@@ -871,7 +872,7 @@ export class SelectorController {
// Clear and re-render the chat
this.ctx.chatContainer.clear();
this.ctx.renderInitialMessages(undefined, { clearTerminalHistory: true });
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
await this.ctx.reloadTodos();
this.ctx.showStatus(movedProject ? `Resumed session in ${shortenPath(newCwd)}` : "Resumed session");
}
@@ -1074,31 +1075,34 @@ export class SelectorController {
});
}
showSessionObserver(registry: SessionObserverRegistry): void {
const observeKeys = this.ctx.keybindings.getKeys("app.session.observe");
let cleanup: (() => void) | undefined;
showAgentHub(observers: SessionObserverRegistry): void {
const hubKeys = [
...this.ctx.keybindings.getKeys("app.agents.hub"),
...this.ctx.keybindings.getKeys("app.session.observe"),
];
let hub: AgentHubOverlayComponent | undefined;
let overlayHandle: OverlayHandle | undefined;
const done = () => {
cleanup?.();
hub?.dispose();
overlayHandle?.hide();
this.ctx.ui.requestRender();
};
const selector = new SessionObserverOverlayComponent(registry, done, observeKeys);
cleanup = registry.onChange(() => {
selector.refreshFromRegistry();
this.ctx.ui.requestRender();
hub = new AgentHubOverlayComponent({
observers,
hubKeys,
onDone: done,
requestRender: () => this.ctx.ui.requestRender(),
});
overlayHandle = this.ctx.ui.showOverlay(selector, {
overlayHandle = this.ctx.ui.showOverlay(hub, {
anchor: "bottom-center",
width: "100%",
maxHeight: "100%",
margin: 0,
});
this.ctx.ui.setFocus(selector);
this.ctx.ui.setFocus(hub);
this.ctx.ui.requestRender();
}
}
@@ -327,6 +327,7 @@ export class InteractiveMode implements InteractiveModeContext {
#pendingSubmissionDispose: (() => void) | undefined;
lastSigintTime = 0;
lastEscapeTime = 0;
lastLeftTapTime = 0;
shutdownRequested = false;
#isShuttingDown = false;
hookSelector: HookSelectorComponent | undefined = undefined;
@@ -388,6 +389,7 @@ export class InteractiveMode implements InteractiveModeContext {
mcpManager?: MCPManager,
eventBus?: EventBus,
titleSystemPrompt?: string,
startupEditor?: CustomEditor,
) {
this.session = session;
this.sessionManager = session.sessionManager;
@@ -422,7 +424,9 @@ export class InteractiveMode implements InteractiveModeContext {
this.btwContainer = new Container();
this.omfgContainer = new Container();
this.errorBannerContainer = new Container();
this.editor = new CustomEditor(getEditorTheme());
// Adopt the pre-TUI startup editor when provided: typed text, cursor,
// paste buffers, and undo history carry over from the splash phase.
this.editor = startupEditor ?? new CustomEditor(getEditorTheme());
this.editor.setUseTerminalCursor(this.ui.getShowHardwareCursor());
this.editor.setAutocompleteMaxVisible(settings.get("autocompleteMaxVisible"));
this.editor.onAutocompleteCancel = () => {
@@ -1088,7 +1092,9 @@ export class InteractiveMode implements InteractiveModeContext {
rebuildChatFromMessages(): void {
this.chatContainer.clear();
const context = this.session.buildDisplaySessionContext();
// Full-history transcript: compactions render as inline dividers instead
// of restarting the visible conversation (the LLM context still resets).
const context = this.session.buildTranscriptSessionContext();
this.renderSessionContext(context);
}
@@ -2880,11 +2886,8 @@ export class InteractiveMode implements InteractiveModeContext {
this.#uiHelpers.renderSessionContext(sessionContext, options);
}
renderInitialMessages(
prebuiltContext?: SessionContext,
options?: { preserveExistingChat?: boolean; clearTerminalHistory?: boolean },
): void {
this.#uiHelpers.renderInitialMessages(prebuiltContext, options);
renderInitialMessages(options?: { preserveExistingChat?: boolean; clearTerminalHistory?: boolean }): void {
this.#uiHelpers.renderInitialMessages(options);
}
getUserMessageText(message: Message): string {
@@ -3068,13 +3071,8 @@ export class InteractiveMode implements InteractiveModeContext {
await this.#selectorController.showDebugSelector();
}
showSessionObserver(): void {
const sessions = this.#observerRegistry.getSessions();
if (sessions.length <= 1) {
this.showStatus("No active subagent sessions");
return;
}
this.#selectorController.showSessionObserver(this.#observerRegistry);
showAgentHub(): void {
this.#selectorController.showAgentHub(this.#observerRegistry);
}
resetObserverRegistry(): void {
@@ -0,0 +1,192 @@
import { StdinBuffer, truncateToWidth } from "@oh-my-pi/pi-tui";
import { postmortem } from "@oh-my-pi/pi-utils";
import { CustomEditor } from "./components/custom-editor";
import { type LspServerInfo, WelcomeComponent } from "./components/welcome";
import { getEditorTheme, theme } from "./theme/theme";
/** Synchronized-output guards (DEC 2026); unsupported terminals ignore them. */
const SYNC_BEGIN = "\x1b[?2026h";
const SYNC_END = "\x1b[?2026l";
export interface StartupInputOptions {
version: string;
modelName: string;
providerName: string;
lspServers: LspServerInfo[];
}
/**
* Pre-TUI live input phase. Paints the same frame the TUI's first full paint
* will produce — welcome box held on the intro's first frame, blank chat area,
* editor box — and runs a real {@link CustomEditor} against raw stdin while
* session creation continues in the background.
*
* The editor instance is handed to InteractiveMode at construction, so typed
* text, cursor position, paste buffers, and undo history carry seamlessly into
* the live UI. Enter submissions made before the session is ready are queued
* (rendered dimmed in the chat area, where the real transcript will appear)
* and replayed through the real submit pipeline once the input loop starts.
*
* Handoff contract: {@link detach} must run before `ProcessTerminal.start()`
* grabs stdin — it restores cooked mode (so the terminal records the correct
* prior raw state) and pauses stdin, leaving OS-buffered keystrokes to flow
* into the TUI once it resumes.
*/
export class StartupInput {
readonly editor: CustomEditor;
readonly #welcome: WelcomeComponent;
#queued: string[] = [];
#stdinBuffer: StdinBuffer | undefined;
#dataListener: ((chunk: string) => void) | undefined;
#resizeListener: (() => void) | undefined;
#unregisterCleanup: (() => void) | undefined;
#started = false;
#detached = false;
#wasRaw = false;
constructor(options: StartupInputOptions) {
this.#welcome = new WelcomeComponent(
options.version,
options.modelName,
options.providerName,
null,
options.lspServers,
);
// Freeze the logo on the intro's first frame so the in-TUI intro picks up
// exactly where this frame leaves off.
this.#welcome.holdIntroFirstFrame();
this.editor = new CustomEditor(getEditorTheme());
// `Editor.#submitValue` expands paste markers, trims, and clears the
// buffer before invoking onSubmit, so `text` is final plain text.
this.editor.onSubmit = text => {
if (text) this.#queued.push(text);
this.#paintLiveRegion();
};
// Ctrl+C: clear typed text; on an empty editor abort startup (pre-TUI raw
// mode swallows SIGINT, so this is the muscle-memory escape hatch).
this.editor.onClear = () => {
if (this.editor.getText()) {
this.editor.setText("");
this.#paintLiveRegion();
} else {
this.#exit(130);
}
};
// Ctrl+D: same exit semantics as the live UI on an idle session.
this.editor.onExit = () => this.#exit(0);
}
/** Enter submissions captured before the session was ready, in order. */
get queuedSubmissions(): readonly string[] {
return this.#queued;
}
/** Grab stdin (raw mode), paint the initial frame, and start echoing input. */
start(): void {
if (this.#started) return;
this.#started = true;
this.#wasRaw = process.stdin.isRaw === true;
process.stdin.setRawMode?.(true);
process.stdin.setEncoding("utf8");
process.stdin.resume();
// Same sequence-splitting pipeline ProcessTerminal uses, so the editor
// receives single key events and bracketed pastes arrive re-wrapped.
const buffer = new StdinBuffer({ timeout: 50 });
buffer.on("data", sequence => this.feedInput(sequence));
buffer.on("paste", content => this.feedInput(`\x1b[200~${content}\x1b[201~`));
this.#stdinBuffer = buffer;
this.#dataListener = chunk => buffer.process(chunk);
process.stdin.on("data", this.#dataListener);
this.#resizeListener = () => this.#paintFull();
process.stdout.on("resize", this.#resizeListener);
// Crash safety: a fatal error before handoff must not leave the user's
// terminal in raw mode with a hidden cursor.
this.#unregisterCleanup = postmortem.register("startup-input-restore", () => this.#restoreTerminal());
// Bracketed paste on; hardware cursor off (the editor draws its own).
process.stdout.write("\x1b[?2004h\x1b[?25l");
this.#paintFull();
}
/** Route one complete input sequence into the editor and refresh the frame. */
feedInput(sequence: string): void {
this.editor.handleInput(sequence);
this.#paintLiveRegion();
}
/**
* Stop capturing and hand the terminal to the TUI. Restores cooked mode so
* `ProcessTerminal.start()` records the correct prior state, and pauses
* stdin so keystrokes typed during the remaining init await flow into the
* TUI once it resumes. The painted frame is left in place — the TUI's first
* full paint replaces it at the same origin.
*/
detach(): void {
this.#unregisterCleanup?.();
this.#unregisterCleanup = undefined;
this.editor.onSubmit = undefined;
this.editor.onClear = undefined;
this.editor.onExit = undefined;
this.#restoreTerminal();
}
#restoreTerminal(): void {
if (!this.#started || this.#detached) return;
this.#detached = true;
if (this.#dataListener) process.stdin.off("data", this.#dataListener);
if (this.#resizeListener) process.stdout.off("resize", this.#resizeListener);
this.#stdinBuffer?.removeAllListeners();
process.stdin.pause();
process.stdin.setRawMode?.(this.#wasRaw);
process.stdout.write("\x1b[?2004l\x1b[?25h");
}
#exit(code: number): void {
this.detach();
process.stdout.write("\r\n");
void postmortem.quit(code);
}
/**
* Compose the full frame, mirroring the TUI's first-paint layout: Spacer,
* welcome box, Spacer, chat area (queued submissions), hook Spacer, editor.
* `liveRegionIndex` marks the first row that changes with input; everything
* above it is the stable welcome prefix.
*/
#frameRows(width: number): { rows: string[]; liveRegionIndex: number } {
const rows: string[] = ["", ...this.#welcome.render(width), ""];
const liveRegionIndex = rows.length;
for (const text of this.#queued) {
rows.push(truncateToWidth(theme.fg("dim", ` › ${text.replace(/\s+/g, " ")}`), Math.max(0, width - 1)));
}
rows.push("");
const terminalRows = process.stdout.rows || 24;
this.editor.setMaxHeight(Math.max(3, Math.min(10, terminalRows - rows.length - 2)));
rows.push(...this.editor.render(width));
return { rows, liveRegionIndex };
}
#paintFull(): void {
if (!this.#started || this.#detached) return;
const width = process.stdout.columns || 80;
const { rows } = this.#frameRows(width);
// Raw mode disables ONLCR; emit explicit CR+LF between rows.
process.stdout.write(`${SYNC_BEGIN}\x1b[2J\x1b[H\x1b[3J${rows.join("\r\n")}${SYNC_END}`);
}
#paintLiveRegion(): void {
if (!this.#started || this.#detached) return;
const width = process.stdout.columns || 80;
const { rows, liveRegionIndex } = this.#frameRows(width);
if (rows.length >= (process.stdout.rows || 24)) {
// Frame taller than the viewport: the initial write scrolled, so
// absolute row addressing no longer maps to the frame. Repaint all.
this.#paintFull();
return;
}
const region = rows.slice(liveRegionIndex).join("\r\n");
process.stdout.write(`${SYNC_BEGIN}\x1b[${liveRegionIndex + 1};1H\x1b[0J${region}${SYNC_END}`);
}
}
+18 -5
View File
@@ -129,6 +129,8 @@ export type SymbolKey =
| "icon.extensionInstruction"
// STT
| "icon.mic"
// Compaction divider
| "icon.camera"
// Thinking Levels
| "thinking.minimal"
| "thinking.low"
@@ -220,7 +222,8 @@ export type SymbolKey =
| "tool.resolve"
| "tool.review"
| "tool.inspectImage"
| "tool.goal";
| "tool.goal"
| "tool.irc";
type SymbolMap = Record<SymbolKey, string>;
@@ -322,13 +325,15 @@ const UNICODE_SYMBOLS: SymbolMap = {
"icon.extensionInstruction": "📘",
// STT
"icon.mic": "🎤",
// Compaction divider
"icon.camera": "📷",
// Thinking levels
"thinking.minimal": "◔ min",
"thinking.low": "◑ low",
"thinking.medium": "◒ med",
"thinking.high": "◕ high",
"thinking.xhigh": "◉ xhigh",
"thinking.autoPending": "▣?",
"thinking.autoPending": "⟳",
// Checkboxes
"checkbox.checked": "☑",
"checkbox.unchecked": "☐",
@@ -414,6 +419,7 @@ const UNICODE_SYMBOLS: SymbolMap = {
"tool.review": "◉",
"tool.inspectImage": "🖼",
"tool.goal": "◎",
"tool.irc": "✉",
};
const NERD_SYMBOLS: SymbolMap = {
@@ -599,6 +605,8 @@ const NERD_SYMBOLS: SymbolMap = {
"icon.extensionInstruction": "\uf02d",
// STT - fa-microphone
"icon.mic": "\uf130",
// Compaction divider - fa-camera-retro
"icon.camera": "\uf083",
// Thinking Levels - emoji labels
// pick: 🤨 min | alt:  min  min
"thinking.minimal": "\u{F0E7} min",
@@ -610,8 +618,8 @@ const NERD_SYMBOLS: SymbolMap = {
"thinking.high": "\u{F111} high",
// pick: 🧠 xhi | alt:  xhi  xhi
"thinking.xhigh": "\u{F06D} xhi",
// pick: 󰞋 (nf-md-help_box) | alt:  [?]
"thinking.autoPending": "\u{f078b}",
// pick: (fa-circle-o-notch) | alt: 󰂼 (nf-md-cached) ⟳
"thinking.autoPending": "\uf1ce",
// Checkboxes
// pick:  | alt:  
"checkbox.checked": "\uf14a",
@@ -708,6 +716,7 @@ const NERD_SYMBOLS: SymbolMap = {
"tool.review": "\uEA70",
"tool.inspectImage": "\uEAEA",
"tool.goal": "\uEBF8",
"tool.irc": "\uF086",
};
const ASCII_SYMBOLS: SymbolMap = {
@@ -808,13 +817,15 @@ const ASCII_SYMBOLS: SymbolMap = {
"icon.extensionInstruction": "IN",
// STT
"icon.mic": "MIC",
// Compaction divider
"icon.camera": "[o]",
// Thinking Levels
"thinking.minimal": "[min]",
"thinking.low": "[low]",
"thinking.medium": "[med]",
"thinking.high": "[high]",
"thinking.xhigh": "[xhi]",
"thinking.autoPending": "[?]",
"thinking.autoPending": "[~]",
// Checkboxes
"checkbox.checked": "[x]",
"checkbox.unchecked": "[ ]",
@@ -898,6 +909,7 @@ const ASCII_SYMBOLS: SymbolMap = {
"tool.review": "rev",
"tool.inspectImage": "[i]",
"tool.goal": "(o)",
"tool.irc": "irc",
};
const SYMBOL_PRESETS: Record<SymbolPreset, SymbolMap> = {
@@ -1686,6 +1698,7 @@ export class Theme {
extensionContextFile: this.#symbols["icon.extensionContextFile"],
extensionInstruction: this.#symbols["icon.extensionInstruction"],
mic: this.#symbols["icon.mic"],
camera: this.#symbols["icon.camera"],
};
}
+3 -5
View File
@@ -136,6 +136,7 @@ export interface InteractiveModeContext {
locallySubmittedUserSignatures: Set<string>;
lastSigintTime: number;
lastEscapeTime: number;
lastLeftTapTime: number;
shutdownRequested: boolean;
hookSelector: HookSelectorComponent | undefined;
hookInput: HookInputComponent | undefined;
@@ -225,10 +226,7 @@ export interface InteractiveModeContext {
sessionContext: SessionContext,
options?: { updateFooter?: boolean; populateHistory?: boolean },
): void;
renderInitialMessages(
prebuiltContext?: SessionContext,
options?: { preserveExistingChat?: boolean; clearTerminalHistory?: boolean },
): void;
renderInitialMessages(options?: { preserveExistingChat?: boolean; clearTerminalHistory?: boolean }): void;
getUserMessageText(message: Message): string;
findLastAssistantMessage(): AssistantMessage | undefined;
extractAssistantText(message: AssistantMessage): string;
@@ -292,7 +290,7 @@ export interface InteractiveModeContext {
showProviderSetup(): Promise<void>;
showHookConfirm(title: string, message: string): Promise<boolean>;
showDebugSelector(): Promise<void>;
showSessionObserver(): void;
showAgentHub(): void;
resetObserverRegistry(): void;
// Input handling
@@ -50,6 +50,7 @@ export function buildHotkeysMarkdown(bindings: HotkeysMarkdownBindings): string
`| \`${appKey(bindings, "app.editor.external")}\` | Edit message in external editor |`,
`| \`${appKey(bindings, "app.clipboard.pasteImage")}\` | Paste image from clipboard |`,
`| \`${appKey(bindings, "app.stt.toggle")}\` | Toggle speech-to-text recording |`,
`| \`${appKey(bindings, "app.agents.hub")}\` / \`${appKey(bindings, "app.session.observe")}\` / double-tap \`←\` (empty editor) | Open the agent hub |`,
"| `#` | Open prompt actions |",
"| `/` | Slash commands |",
"| `!` | Run bash command |",
@@ -190,19 +190,13 @@ export class UiHelpers {
this.ctx.chatContainer.addChild(component);
break;
}
if (
message.customType === "irc:incoming" ||
message.customType === "irc:autoreply" ||
message.customType === "irc:relay"
) {
if (message.customType === "irc:incoming" || message.customType === "irc:relay") {
const details = (
message as CustomMessage<{
from?: string;
to?: string;
message?: string;
reply?: string;
body?: string;
kind?: "message" | "reply";
}>
).details;
let arrow: string;
@@ -211,10 +205,6 @@ export class UiHelpers {
const peer = details?.from ?? "?";
body = details?.message ?? "";
arrow = `⇦ ${peer}`;
} else if (message.customType === "irc:autoreply") {
const peer = details?.to ?? "?";
body = details?.reply ?? "";
arrow = `⇨ ${peer}`;
} else {
const from = details?.from ?? "?";
const to = details?.to ?? "?";
@@ -337,13 +327,23 @@ export class UiHelpers {
let readGroup: ReadToolGroupComponent | null = null;
const readToolCallArgs = new Map<string, Record<string, unknown>>();
const readToolCallAssistantComponents = new Map<string, AssistantMessageComponent>();
const deferredMessages: AgentMessage[] = [];
for (const message of sessionContext.messages) {
// Defer compaction summaries so they render at the bottom (visible after scroll)
if (message.role === "compactionSummary") {
deferredMessages.push(message);
continue;
// Rebuild-time mirror of the event controller's displaceable-poll
// bookkeeping: a `job` poll that found every watched job still running is
// superseded by the next `job` call, so a rebuilt transcript collapses a
// repeated-poll run to its final snapshot instead of replaying the spam.
let waitingPoll: ToolExecutionComponent | null = null;
const resolveWaitingPoll = (nextToolName?: string) => {
const previous = waitingPoll;
if (!previous) return;
waitingPoll = null;
if (nextToolName === "job" && previous.isDisplaceableBlock()) {
this.ctx.chatContainer.removeChild(previous);
}
// Sealing freezes the block and stops the waiting-poll spinner that
// updateResult armed.
previous.seal();
};
for (const message of sessionContext.messages) {
// Assistant messages need special handling for tool calls
if (message.role === "assistant") {
this.ctx.addMessageToChat(message);
@@ -379,6 +379,7 @@ export class UiHelpers {
if (content.type !== "toolCall") {
continue;
}
resolveWaitingPoll(content.name);
if (
content.name === "read" &&
@@ -493,8 +494,17 @@ export class UiHelpers {
if (component) {
component.updateResult(message, false, message.toolCallId);
this.ctx.pendingTools.delete(message.toolCallId);
if (
message.toolName === "job" &&
component instanceof ToolExecutionComponent &&
component.isDisplaceableBlock()
) {
waitingPoll = component;
}
}
} else {
// A user prompt closes the displacement window, same as the live path.
if (message.role === "user") resolveWaitingPoll();
// All other messages use standard rendering
this.ctx.addMessageToChat(message, options);
}
@@ -504,17 +514,15 @@ export class UiHelpers {
// rebuilt group freezes (even with a never-persisted result) and commits to
// native scrollback like every other historical block.
readGroup?.seal();
// Render deferred messages (compaction summaries) at the bottom so they're visible
for (const message of deferredMessages) {
this.ctx.addMessageToChat(message, options);
}
// A trailing waiting poll is final history on rebuild; seal it so it
// freezes (and its spinner timer stops) like every other block.
resolveWaitingPoll();
this.ctx.pendingTools.clear();
this.ctx.ui.requestRender();
}
renderInitialMessages(prebuiltContext?: SessionContext, options: RenderInitialMessagesOptions = {}): void {
renderInitialMessages(options: RenderInitialMessagesOptions = {}): void {
// This path is used to rebuild the visible chat transcript (e.g. after custom/debug UI).
// Clear existing rendered chat first to avoid duplicating the full session in the container.
// On a non-preserving rebuild the existing blocks are discarded for good, so
@@ -530,8 +538,9 @@ export class UiHelpers {
this.ctx.pendingBashComponents = [];
this.ctx.pendingPythonComponents = [];
// Reuse a pre-built context when available (e.g. from navigateTree) to avoid a second O(N) walk.
const context = prebuiltContext ?? this.ctx.sessionManager.buildSessionContext();
// Display always uses the full-history transcript: compactions show as
// inline dividers instead of restarting the visible conversation.
const context = this.ctx.session.buildTranscriptSessionContext();
this.ctx.renderSessionContext(context, {
updateFooter: true,
populateHistory: true,
@@ -1,8 +1,7 @@
<irc>
You received an IRC message from agent `{{from}}`.
Incoming IRC message from agent `{{from}}`{{#if replyTo}} (replying to {{replyTo}}){{/if}}:
Reply briefly and directly using the conversation context already available to you. NEVER call tools. The reply you write is delivered back to `{{from}}` as your answer.
Message:
{{message}}
If a response is expected, reply with the `irc` tool (`op: "send"`, `to: "{{from}}"`) — you may finish your current step first. Nobody replies on your behalf.
</irc>
@@ -8,7 +8,7 @@ You decompose, dispatch, verify, and iterate. Substantial and parallelizable wor
<rules>
1. **NEVER yield until everything is closed.** A phase finishing is *not* a yield point — launch the next phase in the same turn. Stop only when every requested item is verifiably done, or you hit a concrete [blocked] state that genuinely requires the user.
2. **Enumerate the full surface before dispatching.** If the request references audits, plans, checklists, phase lists, or file lists, expand them into a flat set of items in `todo`. "Most of them" or "the important ones" is failure. Re-read the source documents — NEVER work from memory.
3. **Parallelize maximally; NEVER launch a one-off task.** Every set of edits with disjoint file scope MUST ship as one `task` batch — fan the work as wide as it decomposes. A single-task batch for divisible work is a failure: split it. If you are about to dispatch exactly one subagent, stop — either there is more to run alongside it (find it and batch them) or the change is small enough to make inline yourself (do it). Serialize only when one subagent produces a contract (types, schema, shared module) the next consumes — and state the dependency when you do.
3. **Parallelize maximally; NEVER launch a one-off task.** Every set of edits with disjoint file scope MUST ship as parallel `task` calls in one message — fan the work as wide as it decomposes. Dispatching divisible work one call at a time, serially, is a failure: split it and dispatch together. If you are about to dispatch exactly one subagent, stop — either there is more to run alongside it (find it and dispatch them together) or the change is small enough to make inline yourself (do it). Serialize only when one subagent produces a contract (types, schema, shared module) the next consumes — and state the dependency when you do.
4. **Each `task` assignment is self-contained.** Subagents have no shared context. Spell out: target files (≤3–5 explicit paths, no globs), the change with APIs and patterns, edge cases, and observable acceptance criteria. NEVER assume they read the same plan you did.
5. **Verify after every phase before launching the next.** Run the appropriate gate: `bun check` for types, package-scoped `bun test` for behavior, `lsp diagnostics` for changed files. If a phase introduced breakage, dispatch fix-up subagents *before* moving on. NEVER declare a phase done on a red tree.
6. **Commit policy.** If the request asks for commits or the repo workflow expects them, commit after each green phase with a focused message. NEVER commit a red tree. NEVER commit work the user did not ask to commit.
@@ -21,7 +21,7 @@ You decompose, dispatch, verify, and iterate. Substantial and parallelizable wor
<workflow>
1. **Ingest.** Read every referenced file (audits, plans, prior agent output, current branch state). Run `git status` to see uncommitted changes.
2. **Plan.** Materialize the full work surface in `todo` as ordered phases. Within each phase, list the parallelizable units.
3. **Dispatch phase.** Launch all parallel `task` subagents in one call. Wait for the batch.
3. **Dispatch phase.** Launch all parallel `task` subagents in one message, then collect every result (async results / `job poll`) before moving on.
4. **Verify phase.** Run the gates. On failure, dispatch fix-up subagents and re-verify. Do not advance with a red gate.
5. **Commit phase** (if applicable). Focused message naming the phase.
6. **Advance.** Mark the phase done in `todo`, immediately start the next phase. No summary message between phases — keep going.
@@ -3,13 +3,6 @@ ROLE
{{agent}}
{{#if context}}
CONTEXT
===================================
{{context}}
{{/if}}
{{#if planReference}}
PLAN
===================================
@@ -32,11 +25,6 @@ You are working in an isolated working tree at `{{worktree}}` for this sub-task.
You NEVER modify files outside this tree or in the original repository.
{{/if}}
{{#if contextFile}}
# Conversation Context
If you need additional information, your conversation with the user is in {{contextFile}} — `read` its tail or `search` it for relevant terms.
{{/if}}
{{#if ircPeers}}
# IRC Peers
You can reach other live agents via the `irc` tool. Your id is `{{ircSelfId}}`. Currently visible peers:
@@ -149,6 +149,7 @@ With most FS/bash-like tools, static references to them will automatically resol
- `agent://<id>`: full agent output artifact
- `/<path>`: JSON field extraction
- `artifact://<id>`: Artifact content
- `history://<agentId>`: agent transcript as concise markdown; bare `history://` lists agents
- `local://<name>.md`: Plan artifacts and shared content with subagents
{{#if hasObsidian}}
- `vault://<vault>/<path>`: Obsidian vault content (read/edit). `vault://` lists vaults; `vault://_/…` targets the active vault. File-scoped `?op=outline|backlinks|links|tags|properties|tasks|base|…`; vault-scoped `?op=search&q=…|daily|tasks|orphans|unresolved|bases|…`.
@@ -13,8 +13,8 @@ Worth it when the task benefits from decomposition + parallel coverage, or from
<helpers>
State persists across cells, so scout in one cell and fan out in the next. Every cell has:
- `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `context` is shared background; `label` names the artifact. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep.
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool runs as wide as a `task` tool batch — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
- `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent_type` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep.
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`.
- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out.
- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it.
@@ -46,9 +46,9 @@ tool.<name>(args) → unknown
Invoke any session tool by name. `args` is the tool's parameter object.
completion(prompt, model?="default", system?=None, schema?=None) → str | dict
Oneshot, stateless completion (no history, no tools). `model` picks a tier: "smol" (fast), "default" (this session's model), "slow" (most capable). Pass `system` for a system prompt. Pass a JSON-Schema `schema` to force structured output and get the parsed object back; otherwise returns the completion text.
{{#if spawns}}agent(prompt, agent_type?="task", model?=None, context?=None, label?=None, schema?=None) → str | dict
Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back.
{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, context, schema }).
{{#if spawns}}agent(prompt, agent_type?="task", model?=None, label?=None, schema?=None) → str | dict
Run a subagent and return its final output. Defaults to the bundled "task" agent; pass `agent_type`/`agentType` for another discovered agent. Pass a JSON-Schema `schema` to force structured output and get the parsed object back. Share background by writing a `local://` file and referencing it in the prompt.
{{#if js}} In JS, pass options as one trailing object — never positional: agent(prompt, { agentType, schema }).
{{/if}}
{{/if}}
parallel(thunks) → list
+29 -19
View File
@@ -1,11 +1,15 @@
Sends short text messages to other live agents in this process and receives their prose replies.
Sends short text messages to other agents in this process and receives theirs.
<instruction>
- The main agent is addressable as `Main`. Subagents reuse their task id (e.g. `AuthLoader`, or `AuthLoader-2` when the name repeats).
- `op: "list"` returns the current set of visible peers. Use it before sending if you are not sure who is live.
- `op: "send"` delivers `message` to `to`. `to` may be a specific id or `"all"` to broadcast.
- Replies are generated on a side channel that does not wait for the recipient's main loop, so it is safe to IRC an agent that is mid tool call.
- The exchange (question + auto-reply) is injected into the recipient's history; they see it on their next turn and can follow up.
- `op: "list"` — every addressable peer with status (`running` | `idle` | `parked`), unread count, parent, and last activity. Use it before sending if you are not sure who exists.
- `op: "send"` — fire-and-forget delivery of `message` to `to` (a peer id, or `"all"` to broadcast to live peers). Returns per-recipient receipts immediately; it NEVER waits for the recipient to act. Receipt outcomes: `injected` (recipient was mid-turn; message folded in at their next step boundary), `woken` (idle recipient started a turn), `revived` (parked recipient was brought back and woken), `failed`.
- Messaging an `idle` or `parked` peer is how you wake it — there is no separate revive call.
- `send` with `await: true` — convenience round-trip: send, then block until the next message from that peer arrives (or the timeout passes). Invalid with `to: "all"`.
- `op: "wait"` — block until a message arrives (optionally only `from` a specific peer); consumes and returns it. A timeout is a clean "no message" result, not an error.
- `op: "inbox"` — drain pending messages without blocking (`peek: true` to leave them unread).
- `replyTo` — set it to the id of the message you are answering so the sender can correlate.
- Nobody answers on a peer's behalf anymore: a reply only arrives when the recipient actually sends one. For background on what a peer has been doing, `read` `history://<id>` instead of interrogating them.
</instruction>
<when_to_use>
@@ -21,29 +25,35 @@ NEVER use `irc` for: routine progress updates, things a tool call can verify, or
<etiquette>
These rules apply to both sending and replying.
- **Plain prose only.** NEVER send structured JSON status payloads (e.g. `{"type":"task_completed",…}`). Write a normal sentence: "Done with the auth refactor — left a TODO in `src/server/auth.ts` for the rate limiter."
- **NEVER quote the message you are replying to.** Lead with the answer.
- **Use IRC, not terminal tools, to learn about peers.** NEVER `grep` artifacts, read other sessions' JSONL files, or shell-poke to figure out what another agent is doing. DM them.
- **One round-trip is enough.** Replies arrive synchronously when the recipient is reachable. NEVER follow up with "did you get my message?". If `delivered` is empty or the result was `failed`, the peer is unavailable — move on or report the blocker; NEVER retry in a loop.
- **NEVER quote the message you are replying to.** Lead with the answer; set `replyTo` instead.
- **Use IRC, not terminal tools, to learn about peers.** NEVER `grep` artifacts, read other sessions' JSONL files, or shell-poke to figure out what another agent is doing. DM them, or `read` `history://<id>`.
- **Send, then keep working.** `send` returns immediately — only `wait` (or `await: true`) when you genuinely cannot proceed without the answer. NEVER follow up with "did you get my message?"; a `failed` receipt means the peer is unreachable — move on or report the blocker; NEVER retry in a loop.
- **Answer when a response is expected.** When an incoming message asks something, reply with `irc send` to the sender (you may finish your current step first).
- **Stay terse.** A DM is a chat message, not a memo. One question per send. Share file paths and artifacts via `local://` / `memory://` / `artifact://` URLs instead of pasting blobs.
- **Address peers by id.** Use the exact id from `op: "list"` (e.g. `AuthLoader`, `Main`). NEVER invent friendly names.
- **NEVER IRC for things a tool would answer.** If a `read`, `grep`, or build command resolves the question, do that first.
- **Answer incoming IRC messages before continuing.** Address the question directly; do not repeat it back to the user.
</etiquette>
<output>
- `send`: returns each recipient that received the message and any prose replies that arrived.
- `list`: returns peers and channels visible to the caller.
- `send`: per-recipient delivery receipts (`injected` / `woken` / `revived` / `failed`); with `await: true`, also the reply (or a timeout notice).
- `wait`: the consumed message, or a clean timeout notice.
- `inbox`: pending messages, oldest first.
- `list`: peers with status, unread count, parent, and last activity.
</output>
<examples>
# List peers
`{"op": "list"}`
# Direct message to the main agent (waits for prose reply)
`{"op": "send", "to": "Main", "message": "Should I prefer JWT or session cookies for the auth flow?"}`
# Unexpected state — ask the originator
`{"op": "send", "to": "Main", "message": "Assignment says edit src/auth/jwt.ts but the file does not exist. Is the new path src/server/auth/jwt.ts?"}`
# Blocked by a peer — ask them directly
`{"op": "send", "to": "AuthLoader", "message": "Are you still touching src/server/auth.ts? I need to add a 401 path; OK to proceed or should I wait?"}`
# Broadcast to discover who owns something (no replies, just informs them)
`{"op": "send", "to": "all", "message": "About to refactor src/server/middleware/*. Anyone already in there?", "awaitReply": false}`
# Fire-and-forget DM — keep working, check inbox later
`{"op": "send", "to": "AuthLoader", "message": "Are you still touching src/server/auth.ts? I need to add a 401 path."}`
# Round-trip when you cannot proceed without the answer
`{"op": "send", "to": "Main", "message": "Should I prefer JWT or session cookies for the auth flow?", "await": true}`
# Wake a parked agent (same send — the bus revives it)
`{"op": "send", "to": "SchemaMigrator", "message": "The users table changed again; please re-check your migration."}`
# Block until a specific peer answers
`{"op": "wait", "from": "AuthLoader", "timeoutMs": 60000}`
# Drain pending messages
`{"op": "inbox"}`
# Broadcast to live peers (no replies expected)
`{"op": "send", "to": "all", "message": "About to refactor src/server/middleware/*. Anyone already in there?"}`
</examples>
@@ -8,7 +8,7 @@ Read files, directories, archives, SQLite databases, images, documents, internal
## Parameters
- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`), or URL. Append `:<sel>` for line ranges, raw mode, or special modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`).
- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `history://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`), or URL. Append `:<sel>` for line ranges, raw mode, or special modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`).
## Selectors
@@ -74,7 +74,7 @@ For `.sqlite`, `.sqlite3`, `.db`, `.db3`:
# Internal URIs
`skill://<name>`, `agent://<id>`, `artifact://<id>`, `memory://root`, `rule://<name>`, `local://<name>.md`, `vault://<vault>/<path>`, `mcp://<uri>`, `omp://<doc>.md`, `issue://<N>`, and `pr://<N>` resolve transparently and accept the same line selectors as filesystem paths. Use `artifact://<id>` to recover full output that a previous bash/eval/tool result spilled or truncated.
`skill://<name>`, `agent://<id>`, `artifact://<id>`, `history://<agentId>`, `memory://root`, `rule://<name>`, `local://<name>.md`, `vault://<vault>/<path>`, `mcp://<uri>`, `omp://<doc>.md`, `issue://<N>`, and `pr://<N>` resolve transparently and accept the same line selectors as filesystem paths. Use `artifact://<id>` to recover full output that a previous bash/eval/tool result spilled or truncated. `history://<agentId>` is an agent's transcript as concise markdown; bare `history://` lists agents.
<critical>
- You MUST use `read` for every file, directory, archive, and URL inspection. `cat`, `head`, `tail`, `less`, `more`, `ls`, `tar`, `unzip`, `curl`, `wget` are FORBIDDEN — any such bash call is a bug, regardless of how short or convenient it looks.
@@ -1,28 +1,17 @@
<task-summary>
<header>{{successCount}}/{{totalCount}} succeeded{{#if hasCancelledNote}} ({{cancelledCount}} cancelled){{/if}} [{{duration}}]</header>
{{#each summaries}}
<agent id="{{id}}" agent="{{agent}}">
<status>{{status}}</status>
<task-result id="{{id}}" agent="{{agentName}}" status="{{status}}" duration="{{duration}}">
{{#if meta}}<meta lines="{{meta.lineCount}}" size="{{meta.charSize}}" />{{/if}}
{{#if truncated}}
<preview full-path="agent://{{id}}">
<preview full-output="agent://{{id}}">
{{preview}}
</preview>
{{else}}
<result>
<output>
{{preview}}
</result>
</output>
{{/if}}
</agent>
{{#unless @last}}
---
{{/unless}}
{{/each}}
{{#if mergeSummary}}
<merge-summary>
{{mergeSummary}}
</merge-summary>
{{/if}}
</task-summary>
</task-result>
+22 -35
View File
@@ -1,43 +1,37 @@
Launches subagents to parallelize workflows.
Spawns ONE subagent per call to work in the background, or resumes an existing one.
{{#if asyncEnabled}}
- Results are delivered automatically when complete.
- The tool result lists the assigned task ids (e.g. `AuthLoader`) — those are the live agent ids.
- Spawning is non-blocking: the call returns immediately with the agent id and a job id; the result is delivered automatically when the agent yields.
- Parallelism = multiple `task` calls in one assistant message. Concurrency is bounded at {{MAX_CONCURRENCY}} running subagents per session.
- If genuinely blocked on a result, wait with `job poll`; otherwise keep working. `job cancel` terminates a task and **cannot carry a message** — only for stalled/abandoned work.
{{#if ircEnabled}}
- Coordinate with running tasks via `irc` using those ids. `job cancel` terminates a task and **cannot carry a message** — only use it for stalled/abandoned work.
- If genuinely blocked on completion, wait with `job poll`; otherwise keep working.
{{else}}
- If genuinely blocked on completion, wait with `job poll`; otherwise keep working.
- Use `job list` to snapshot manager state; `cancel: [id]` only to actually stop a stuck task.
{{/if}}
- Coordinate with running agents via `irc` using their ids. Agents reach you and their siblings live the same way.
{{/if}}
{{#if ircEnabled}}
Subagents have no conversation history, but they can reach you and their siblings live via the `irc` tool. Front-load every fact, file path, and direction they need in {{#if contextEnabled}}`context` or `assignment`{{else}}each `assignment`{{/if}}.
{{else}}
Subagents have no conversation history. Every fact, file path, and direction they need MUST be explicit in {{#if contextEnabled}}`context` or `assignment`{{else}}each `assignment`{{/if}}.
{{/if}}
<lifecycle>
- Finished agents stay alive: `idle` first, then `parked` after a TTL — both remain addressable and revivable.
- `resume: "<id>"` revives an idle/parked agent and runs a follow-up assignment in its existing session. **Prefer resuming an agent that already holds the relevant context over spawning fresh**{{#if ircEnabled}} — check `irc` op:"list" for candidates{{/if}}.
- `history://<id>` is the agent's transcript; `agent://<id>` its latest output artifact.
</lifecycle>
<parameters>
- `agent`: agent type for all tasks
- `tasks`: tasks to execute in parallel
- `.id`: CamelCase, ≤32 chars
- `.description`: UI label only — subagent never sees it
- `.assignment`: complete self-contained instructions; one-liners and missing acceptance criteria are PROHIBITED
{{#if contextEnabled}}- `context`: shared background prepended to every assignment; session-specific only{{/if}}
- `agent`: agent type to spawn; omit when `resume` is set
- `resume`: existing agent id — continue that agent instead of spawning (cannot combine with `agent` or `isolated`)
- `id`: stable agent id, CamelCase, ≤32 chars; generated when omitted
- `description`: UI label only — subagent never sees it
- `assignment`: complete self-contained instructions; one-liners and missing acceptance criteria are PROHIBITED
{{#if customSchemaEnabled}}- `schema`: JTD schema for expected structured output (do not put format rules in assignments){{/if}}
{{#if isolationEnabled}}- `isolated`: run in isolated env; use when tasks edit overlapping files{{/if}}
{{#if isolationEnabled}}- `isolated`: run in isolated env; returns patches. Isolated agents are NOT resumable{{/if}}
</parameters>
<rules>
- **Maximize batch width.** Spawn the widest parallel set the work decomposes into. NEVER spawn a single-task batch for divisible work, or defer work that could have been concurrent.
- **Subagents do not verify, lint, or format.** Every assignment MUST instruct the subagent to skip all gates, formatters, and project-wide build/test/lint. You run them once at the end across the union of changed files — avoids redundant runs and racing formatter passes.
- **Maximize fan-out.** Issue the widest set of parallel `task` calls the work decomposes into. NEVER serialize work that could run concurrently.
- **Subagents do not verify, lint, or format.** Every assignment MUST instruct the subagent to skip all gates, formatters, and project-wide build/test/lint. You run them once at the end across the union of changed files.
- No globs, no "update all", no package-wide scope. Fan out.
- NEVER slow down or serialize because tasks might overlap on some files. Agents resolve collisions among themselves in real time.
- Pass large payloads via `local://<path>` URIs, not inline. {{#if contextEnabled}} (other than the context){{/if}}
{{#if contextEnabled}}- Put shared constraints in `context` once; do not duplicate across assignments.{{/if}}
- Subagents have no conversation history. Every fact, file path, and direction they need MUST be explicit in the `assignment`.
- **Shared background**: write it ONCE to a `local://` file (e.g. `local://ctx.md`) and reference that path in each assignment. Pass large payloads via `local://<path>` URIs, not inline.
- Prefer agents that investigate **and** edit in one pass; only spin a read-only discovery step when affected files are genuinely unknown.
- **Read-only agents**: Agents tagged READ-ONLY (e.g. `explore`) have no edit/write/command tools. NEVER hand them an assignment that requires changing files or running commands — they cannot do it and the turn is wasted. Use them to investigate and report back; do the edits yourself or delegate to a writing agent (`task`, `oracle`, `designer`).
- **Read-only agents**: Agents tagged READ-ONLY (e.g. `explore`) have no edit/write/command tools. NEVER hand them an assignment that requires changing files or running commands. Use them to investigate and report back; do the edits yourself or delegate to a writing agent (`task`, `oracle`, `designer`).
- **No reasoning offload**: NEVER offload reasoning, analysis, design, or decision-making to `quick_task` or `explore` — they run minimal-effort / small models for mechanical lookups and data collection only. Keep judgment and synthesis in your own context; delegate hard thinking to `task`, `plan`, or `oracle`.
</rules>
@@ -51,16 +45,9 @@ Test: can task B run correctly without seeing A's output? If no, sequence A →
Sequential when one task produces a contract (types, API, schema, core module) the other consumes.
Parallel when tasks touch disjoint files or are independent refactors/tests.
{{/if}}
Sequenced follow-ups SHOULD `resume` the agent that produced the prerequisite — it already holds the context.
</parallelization>
{{#if contextEnabled}}
<context-fmt>
# Goal ← one sentence: what the batch accomplishes
# Constraints ← MUST/NEVER rules and session decisions
# Contract ← exact types/signatures if tasks share an interface
</context-fmt>
{{/if}}
<assignment-fmt>
# Target ← exact files and symbols; explicit non-goals
# Change ← step-by-step add/remove/rename; APIs and patterns
@@ -0,0 +1,218 @@
/**
* AgentLifecycleManager - Owns the idle → parked → revived lifecycle of
* adopted subagents.
*
* The task executor hands a finished agent over via {@link AgentLifecycleManager.adopt};
* from then on the manager arms a TTL timer whenever the agent goes `idle`,
* parks it on expiry (disposes the live session, keeps the AgentRef +
* sessionFile), and revives it on demand through
* {@link AgentLifecycleManager.ensureLive}. Only this manager flips
* `parked` ↔ `idle`.
*/
import { logger } from "@oh-my-pi/pi-utils";
import type { AgentSession } from "../session/agent-session";
import { AgentRegistry, MAIN_AGENT_ID, type RegistryEvent } from "./agent-registry";
export type AgentReviver = () => Promise<AgentSession>;
export interface AdoptOptions {
/** TTL before an idle agent is parked. <= 0 disables parking. */
idleTtlMs: number;
/** Recreates a live AgentSession from the ref's sessionFile. Absent => not resumable after park (e.g. isolated runs). */
revive?: AgentReviver;
}
interface AdoptedAgent {
idleTtlMs: number;
revive?: AgentReviver;
timer?: NodeJS.Timeout;
}
export class AgentLifecycleManager {
static #global: AgentLifecycleManager | undefined;
static global(): AgentLifecycleManager {
if (!AgentLifecycleManager.#global) {
AgentLifecycleManager.#global = new AgentLifecycleManager();
}
return AgentLifecycleManager.#global;
}
/** Reset the global manager. Test-only. */
static resetGlobalForTests(): void {
const current = AgentLifecycleManager.#global;
if (current) {
current.#unsubscribe?.();
current.#unsubscribe = undefined;
for (const adopted of current.#adopted.values()) {
clearTimeout(adopted.timer);
}
current.#adopted.clear();
current.#revivals.clear();
current.#parking.clear();
}
AgentLifecycleManager.#global = undefined;
}
readonly #registry: AgentRegistry;
readonly #adopted = new Map<string, AdoptedAgent>();
/** Ids whose session is being disposed by {@link park} right now. */
readonly #parking = new Set<string>();
/** In-flight revives, so concurrent {@link ensureLive} calls coalesce. */
readonly #revivals = new Map<string, Promise<AgentSession>>();
#unsubscribe: (() => void) | undefined;
constructor(registry: AgentRegistry = AgentRegistry.global()) {
this.#registry = registry;
this.#unsubscribe = registry.onChange(event => this.#onRegistryEvent(event));
}
/**
* Take ownership of a finished subagent. Caller has already set registry
* status to "idle". Arms the TTL timer (idleTtlMs <= 0 adopts without one).
*/
adopt(id: string, opts: AdoptOptions): void {
if (id === MAIN_AGENT_ID) return;
if (!this.#registry.get(id)) {
logger.warn("AgentLifecycleManager.adopt: unknown agent id", { id });
return;
}
const existing = this.#adopted.get(id);
clearTimeout(existing?.timer);
const adopted: AdoptedAgent = { idleTtlMs: opts.idleTtlMs, revive: opts.revive };
this.#adopted.set(id, adopted);
this.#armTimer(id, adopted);
}
/** True if the id is adopted (parked or live). */
has(id: string): boolean {
return this.#adopted.has(id);
}
/** True while {@link park} is disposing this agent's session (lets dispose hooks distinguish park from teardown). */
isParking(id: string): boolean {
return this.#parking.has(id);
}
/**
* Dispose the live session, detach it from the registry, and mark the
* agent `parked`. No-op unless the id is adopted and live.
*/
async park(id: string): Promise<void> {
const adopted = this.#adopted.get(id);
if (!adopted) return;
const ref = this.#registry.get(id);
if (!ref?.session) return;
if (adopted.timer) {
clearTimeout(adopted.timer);
adopted.timer = undefined;
}
this.#parking.add(id);
try {
try {
await ref.session.dispose();
} catch (error) {
logger.warn("AgentLifecycleManager.park: session dispose failed", { id, error: String(error) });
}
this.#registry.detachSession(id);
this.#registry.setStatus(id, "parked");
} finally {
this.#parking.delete(id);
}
}
/**
* Return the live session, reviving from the sessionFile if parked.
* Throws a plain Error if the id is unknown or parked without a reviver.
* Concurrent calls share one in-flight revive.
*/
async ensureLive(id: string): Promise<AgentSession> {
const ref = this.#registry.get(id);
if (!ref) {
throw new Error(
`Unknown agent "${id}" — it was never registered or has been released. If a transcript exists, read history://${id}.`,
);
}
if (ref.session) return ref.session;
const inflight = this.#revivals.get(id);
if (inflight) return inflight;
const adopted = this.#adopted.get(id);
if (ref.status !== "parked" || !adopted?.revive) {
throw new Error(
`Agent "${id}" is ${ref.status} and cannot be revived${adopted?.revive ? "" : " (no reviver registered)"}. Its transcript remains readable at history://${id}.`,
);
}
const revival = this.#revive(id, adopted.revive, ref.sessionFile);
this.#revivals.set(id, revival);
try {
return await revival;
} finally {
this.#revivals.delete(id);
}
}
/** Hard removal: dispose if live, unregister from registry, drop timers. */
async release(id: string): Promise<void> {
const adopted = this.#adopted.get(id);
clearTimeout(adopted?.timer);
this.#adopted.delete(id);
const ref = this.#registry.get(id);
if (ref?.session) {
try {
await ref.session.dispose();
} catch (error) {
logger.warn("AgentLifecycleManager.release: session dispose failed", { id, error: String(error) });
}
}
this.#registry.unregister(id);
}
/** Teardown everything (process exit / main session dispose). */
async dispose(): Promise<void> {
this.#unsubscribe?.();
this.#unsubscribe = undefined;
const ids = [...this.#adopted.keys()];
await Promise.all(ids.map(id => this.release(id)));
this.#revivals.clear();
this.#parking.clear();
}
async #revive(id: string, revive: AgentReviver, sessionFile: string | null): Promise<AgentSession> {
const session = await revive();
this.#registry.attachSession(id, session, sessionFile);
// Emits status_changed → "idle", which re-arms the TTL timer below.
this.#registry.setStatus(id, "idle");
return session;
}
#armTimer(id: string, adopted: AdoptedAgent): void {
if (adopted.idleTtlMs <= 0) return;
clearTimeout(adopted.timer);
const timer = setTimeout(() => {
adopted.timer = undefined;
void this.park(id);
}, adopted.idleTtlMs);
timer.unref?.();
adopted.timer = timer;
}
#onRegistryEvent(event: RegistryEvent): void {
const adopted = this.#adopted.get(event.ref.id);
if (!adopted) return;
if (event.type === "removed") {
clearTimeout(adopted.timer);
this.#adopted.delete(event.ref.id);
return;
}
if (event.type !== "status_changed") return;
if (event.ref.status === "running") {
if (adopted.timer) {
clearTimeout(adopted.timer);
adopted.timer = undefined;
}
} else if (event.ref.status === "idle") {
this.#armTimer(event.ref.id, adopted);
}
}
}
@@ -1,16 +1,26 @@
/**
* AgentRegistry - Process-global registry of live AgentSession instances.
* AgentRegistry - Process-global registry of agents (the main session plus
* every subagent), keyed by stable id.
*
* Tracks every alive agent (the main session plus every subagent) so the
* `irc` tool can address peers by id. Sessions are registered explicitly at
* creation and removed when the owner releases them.
* Tracks each agent's status and (when live) its AgentSession so peers can be
* addressed by id (`irc`, `task resume`, `history://`). Sessions are
* registered explicitly at creation; finished agents stay registered as
* `idle` (live) or `parked` (session disposed, ref + sessionFile retained for
* revival) and are only removed on explicit release/teardown.
*/
import type { AgentSession } from "../session/agent-session";
export const MAIN_AGENT_ID = "Main";
export type AgentStatus = "running" | "idle" | "completed" | "aborted";
/**
* - `running`: a turn is in flight.
* - `idle`: live AgentSession in memory, awaiting work. Finished agents are
* `idle`, not removed.
* - `parked`: session disposed; AgentRef + sessionFile retained, revivable.
* - `aborted`: hard-killed, terminal.
*/
export type AgentStatus = "running" | "idle" | "parked" | "aborted";
export type AgentKind = "main" | "sub";
export interface AgentRef {
@@ -19,6 +29,7 @@ export interface AgentRef {
kind: AgentKind;
parentId?: string;
status: AgentStatus;
/** Null exactly when parked/aborted. */
session: AgentSession | null;
sessionFile: string | null;
createdAt: number;
+29 -9
View File
@@ -34,7 +34,7 @@ import {
Snowflake,
} from "@oh-my-pi/pi-utils";
import chalk from "chalk";
import { type AsyncJob, AsyncJobManager, isBackgroundJobSupportEnabled } from "./async";
import { type AsyncJob, AsyncJobManager } from "./async";
import { loadCapability } from "./capability";
import { type Rule, ruleCapability, setActiveRules } from "./capability/rule";
import { bucketRules } from "./capability/rule-buckets";
@@ -93,6 +93,7 @@ import { createSessionMemoryRuntimeContext, resolveMemoryBackend } from "./memor
import type { MnemopiSessionState } from "./mnemopi/state";
import asyncResultTemplate from "./prompts/tools/async-result.md" with { type: "text" };
import lateDiagnosticTemplate from "./prompts/tools/lsp-late-diagnostic.md" with { type: "text" };
import { AgentLifecycleManager } from "./registry/agent-lifecycle";
import { AgentRegistry, MAIN_AGENT_ID } from "./registry/agent-registry";
import {
collectEnvSecrets,
@@ -1293,7 +1294,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
let hasSession = false;
let hasRegistered = false;
const enableLsp = options.enableLsp ?? true;
const backgroundJobsEnabled = isBackgroundJobSupportEnabled(settings);
const asyncMaxJobs = Math.min(100, Math.max(1, settings.get("async.maxJobs") ?? 100));
const ASYNC_INLINE_RESULT_MAX_CHARS = 12_000;
const ASYNC_PREVIEW_MAX_CHARS = 4_000;
@@ -1326,7 +1326,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
// (issue #1923). The `instance()` guard means later sessions also skip
// constructing an orphaned manager that nothing would ever route to.
const asyncJobManager =
backgroundJobsEnabled && !options.parentTaskPrefix && !AsyncJobManager.instance()
!options.parentTaskPrefix && !AsyncJobManager.instance()
? new AsyncJobManager({
maxRunningJobs: asyncMaxJobs,
onJobComplete: async (jobId, result, job) => {
@@ -1351,6 +1351,17 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
const resolvedAgentId = options.agentId ?? options.parentTaskPrefix ?? MAIN_AGENT_ID;
const resolvedAgentDisplayName =
options.agentDisplayName ?? ((options.taskDepth ?? 0) > 0 || options.parentTaskPrefix ? "sub" : "main");
const agentKind = (options.taskDepth ?? 0) > 0 || options.parentTaskPrefix ? ("sub" as const) : ("main" as const);
/**
* Forget the agent ref on teardown — unless the agent is being parked (or is
* already parked). Parking disposes the session but keeps the ref addressable
* (history://, revive); only process teardown / explicit kill unregisters.
*/
const unregisterUnlessParked = (): void => {
if (agentRegistry.get(resolvedAgentId)?.status === "parked") return;
if (AgentLifecycleManager.global().isParking(resolvedAgentId)) return;
agentRegistry.unregister(resolvedAgentId);
};
const evalKernelOwnerId = `agent-session:${Snowflake.next()}`;
try {
@@ -1409,7 +1420,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
getTurnBudget: () => sessionManager.getTurnBudget(),
recordEvalSubagentUsage: output => sessionManager.recordEvalSubagentOutput(output),
getClientBridge: () => session?.clientBridge,
getCompactContext: () => session.formatCompactContext(),
queueDeferredDiagnostics: entry => session?.yieldQueue.enqueue(LSP_LATE_DIAGNOSTIC_MESSAGE_TYPE, entry),
bumpFileMutationVersion: path => {
const next = (fileMutationVersions.get(path) ?? 0) + 1;
@@ -2083,7 +2093,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
agentRegistry.register({
id: resolvedAgentId,
displayName: resolvedAgentDisplayName,
kind: (options.taskDepth ?? 0) > 0 || options.parentTaskPrefix ? "sub" : "main",
kind: agentKind,
parentId: options.parentTaskPrefix,
session: null,
sessionFile: sessionManager.getSessionFile() ?? null,
@@ -2320,7 +2330,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
ttsrManager,
obfuscator,
agentId: resolvedAgentId,
agentRegistry,
providerSessionId: options.providerSessionId,
parentEvalSessionId: options.parentEvalSessionId,
});
@@ -2341,15 +2350,26 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
// Attach the live session to the pre-registered ref so peers can route IRC
// messages here. Refresh sessionFile in case it was unavailable at pre-register
// time. The dispose wrapper below unregisters on teardown.
// time. The dispose wrapper below unregisters on teardown (unless parked).
agentRegistry.attachSession(resolvedAgentId, session, sessionManager.getSessionFile() ?? null);
{
const originalDispose = session.dispose.bind(session);
session.dispose = async () => {
try {
// Reject new session work (Python/eval starts) the moment disposal
// begins — the lifecycle await below opens an async gap before
// AgentSession.dispose() would otherwise set its guards.
session.beginDispose();
if (agentKind === "main") {
// Top-level teardown owns the global agent lifecycle: park timers,
// adopted subagent sessions, revivers. Tear it down while shared
// resources (kernels, MCP, LSP) are still live. Subagent disposal
// must NOT touch the global lifecycle.
await AgentLifecycleManager.global().dispose();
}
await originalDispose();
} finally {
agentRegistry.unregister(resolvedAgentId);
unregisterUnlessParked();
unsubscribeCredentialDisabled?.();
}
};
@@ -2502,7 +2522,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
if (hasSession) {
await session.dispose();
} else {
if (hasRegistered) agentRegistry.unregister(resolvedAgentId);
if (hasRegistered) unregisterUnlessParked();
if (asyncJobManager) {
if (AsyncJobManager.instance() === asyncJobManager) {
AsyncJobManager.setInstance(undefined);
+193 -230
View File
@@ -29,6 +29,7 @@ import {
type AgentState,
type AgentTool,
AppendOnlyContextManager,
type AsideMessage,
resolveTelemetry,
ThinkingLevel,
} from "@oh-my-pi/pi-agent-core";
@@ -55,7 +56,12 @@ import {
type SummaryOptions,
shouldCompact,
} from "@oh-my-pi/pi-agent-core/compaction";
import { DEFAULT_PRUNE_CONFIG, pruneToolOutputs } from "@oh-my-pi/pi-agent-core/compaction/pruning";
import {
DEFAULT_PRUNE_CONFIG,
pruneSupersededToolResults,
pruneToolOutputs,
readToolSupersedeKey,
} from "@oh-my-pi/pi-agent-core/compaction/pruning";
import type { ProtectedToolMatcher } from "@oh-my-pi/pi-agent-core/compaction/tool-protection";
import type {
AssistantMessage,
@@ -100,6 +106,7 @@ import {
relativePathWithinRoot,
Snowflake,
} from "@oh-my-pi/pi-utils";
import { snapcompactCompact } from "@oh-my-pi/snapcompact";
import { type AsyncJob, type AsyncJobDeliveryState, AsyncJobManager } from "../async";
import { classifyDifficulty } from "../auto-thinking/classifier";
import { reset as resetCapabilities } from "../capability";
@@ -163,6 +170,7 @@ import { GoalRuntime } from "../goals/runtime";
import type { Goal, GoalModeState } from "../goals/state";
import type { HindsightSessionState } from "../hindsight/state";
import { type LocalProtocolOptions, resolveLocalUrlToPath } from "../internal-urls";
import type { IrcMessage } from "../irc/bus";
import { resolveMemoryBackend } from "../memory-backend";
import { getMnemopiSessionState, type MnemopiSessionState, setMnemopiSessionState } from "../mnemopi/state";
import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate";
@@ -184,7 +192,6 @@ import planModeToolDecisionReminderPrompt from "../prompts/system/plan-mode-tool
};
import ttsrInterruptTemplate from "../prompts/system/ttsr-interrupt.md" with { type: "text" };
import ttsrToolReminderTemplate from "../prompts/system/ttsr-tool-reminder.md" with { type: "text" };
import { type AgentRegistry, MAIN_AGENT_ID } from "../registry/agent-registry";
import {
deobfuscateSessionContext,
obfuscateProviderContext,
@@ -230,10 +237,8 @@ import type { AuthStorage } from "./auth-storage";
import type { ClientBridge, ClientBridgePermissionOption, ClientBridgePermissionOutcome } from "./client-bridge";
import {
type BashExecutionMessage,
type CompactionSummaryMessage,
type CustomMessage,
convertToLlm,
type FileMentionMessage,
type PythonExecutionMessage,
readPendingDisplayTag,
SILENT_ABORT_MARKER,
@@ -259,11 +264,11 @@ export type AgentSessionEvent =
| {
type: "auto_compaction_start";
reason: "threshold" | "overflow" | "idle" | "incomplete";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
}
| {
type: "auto_compaction_end";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
result: CompactionResult | undefined;
aborted: boolean;
willRetry: boolean;
@@ -413,8 +418,6 @@ export interface AgentSessionConfig {
asyncJobManager?: AsyncJobManager;
/** Agent identity (registry id like "Main" or "Alice") used for IRC routing. */
agentId?: string;
/** Shared agent registry (for forwarding IRC observations to the main session UI). */
agentRegistry?: AgentRegistry;
/**
* Override the provider-facing session ID for all API requests from this session.
* When absent, `sessionManager.getSessionId()` is used. Needed when benchmark or
@@ -557,15 +560,15 @@ function formatRetryFallbackBaseSelector(selector: RetryFallbackSelector): strin
return `${selector.provider}/${selector.id}`;
}
const IRC_REPLY_MAX_BYTES = 4096;
const EPHEMERAL_REPLY_MAX_BYTES = 4096;
/**
* Collapse degenerate IRC ephemeral replies before they hit the relay.
* Collapse degenerate ephemeral replies (/btw, /omfg side-channel turns).
* Models occasionally loop on a single line (~16 reports of N-times-repeated
* replies); compress runs longer than 3 down to one instance + `[…N×]`, then
* cap at 4 KiB so a runaway reply can't flood the channel.
*/
function dedupeIrcReply(text: string): string {
function dedupeEphemeralReply(text: string): string {
if (!text) return text;
const lines = text.split("\n");
const out: string[] = [];
@@ -582,11 +585,11 @@ function dedupeIrcReply(text: string): string {
i = j;
}
let result = out.join("\n");
if (Buffer.byteLength(result, "utf8") > IRC_REPLY_MAX_BYTES) {
if (Buffer.byteLength(result, "utf8") > EPHEMERAL_REPLY_MAX_BYTES) {
// Trim by characters until we're under the byte budget — handles multi-byte
// glyphs at the boundary without splitting them.
const suffix = "\n[…truncated]";
const budget = IRC_REPLY_MAX_BYTES - Buffer.byteLength(suffix, "utf8");
const budget = EPHEMERAL_REPLY_MAX_BYTES - Buffer.byteLength(suffix, "utf8");
while (Buffer.byteLength(result, "utf8") > budget) {
result = result.slice(0, -1);
}
@@ -941,13 +944,11 @@ export class AgentSession {
#activeEvalExecutions = new Set<Promise<unknown>>();
#evalExecutionDisposing = false;
// Background-channel IRC exchanges queued while the recipient was streaming.
// Drained into history (via emitExternalEvent) once the recipient becomes idle.
#pendingBackgroundExchanges: CustomMessage[][] = [];
#scheduledBackgroundExchangeFlush = false;
// Agent identity + registry for IRC relay forwarding to the main session UI.
// Incoming IRC messages received while a turn was streaming; drained as
// non-interrupting asides at the next step boundary (see the aside provider).
#pendingIrcAsides: CustomMessage[] = [];
// Agent identity (registry id) used for IRC routing and job ownership.
#agentId: string | undefined;
#agentRegistry: AgentRegistry | undefined;
#providerSessionId: string | undefined;
#freshProviderSessionId: string | undefined;
#isDisposed = false;
@@ -1204,7 +1205,13 @@ export class AgentSession {
// Background-job completions / late diagnostics are pulled into the run at
// each step boundary as non-interrupting asides (see Agent.getAsideMessages),
// so they reach the model between requests without waiting for a yield.
this.agent.setAsideMessageProvider(() => this.yieldQueue.drainLazy());
this.agent.setAsideMessageProvider(() => {
const pendingIrc = this.#pendingIrcAsides;
this.#pendingIrcAsides = [];
const thunks: AsideMessage[] = pendingIrc.map(record => () => record);
thunks.push(...this.yieldQueue.drainLazy());
return thunks;
});
this.#convertToLlm = config.convertToLlm ?? convertToLlm;
this.#rebuildSystemPrompt = config.rebuildSystemPrompt;
this.#getMcpServerInstructions = config.getMcpServerInstructions;
@@ -1235,7 +1242,6 @@ export class AgentSession {
this.#ttsrManager = config.ttsrManager;
this.#obfuscator = config.obfuscator;
this.#agentId = config.agentId;
this.#agentRegistry = config.agentRegistry;
this.#providerSessionId = config.providerSessionId;
this.agent.setAssistantMessageEventInterceptor((message, assistantMessageEvent) => {
const event: AgentEvent = {
@@ -3090,16 +3096,29 @@ export class AgentSession {
state.resetConversationTracking();
}
/**
* Synchronously mark the session as disposing so new work is rejected
* immediately: Python/eval starts throw, queued asides are dropped, and the
* aside provider is detached. Idempotent; `dispose()` runs it first.
*
* Wrappers that await other teardown before delegating to `dispose()` MUST
* call this before their first await — otherwise work started in that async
* gap slips past the disposal guards.
*/
beginDispose(): void {
this.#isDisposed = true;
this.#pendingIrcAsides = [];
this.yieldQueue.clear();
this.agent.setAsideMessageProvider(undefined);
this.#evalExecutionDisposing = true;
}
/**
* Remove all listeners, flush pending writes, and disconnect from agent.
* Call this when completely done with the session.
*/
async dispose(): Promise<void> {
this.#isDisposed = true;
this.#pendingBackgroundExchanges = [];
this.yieldQueue.clear();
this.agent.setAsideMessageProvider(undefined);
this.#evalExecutionDisposing = true;
this.beginDispose();
try {
if (this.#extensionRunner?.hasHandlers("session_shutdown")) {
await this.#extensionRunner.emit({ type: "session_shutdown" });
@@ -4032,6 +4051,16 @@ export class AgentSession {
return deobfuscateSessionContext(this.sessionManager.buildSessionContext(), this.#obfuscator);
}
/**
* Full-history transcript for TUI display: every path entry in
* chronological order with compactions rendered inline at the point they
* fired (instead of replacing prior history). Display-only — NEVER feed
* the result to `agent.replaceMessages` or a provider.
*/
buildTranscriptSessionContext(): SessionContext {
return deobfuscateSessionContext(this.sessionManager.buildSessionContext({ transcript: true }), this.#obfuscator);
}
#obfuscateForProvider<T>(value: T): T {
if (!this.#obfuscator?.hasSecrets()) return value;
return this.#obfuscator.obfuscateObject(value);
@@ -4634,7 +4663,7 @@ export class AgentSession {
// Flush any pending bash messages before the new prompt
this.#flushPendingBashMessages();
this.#flushPendingPythonMessages();
this.#flushPendingBackgroundExchanges();
this.#flushPendingIrcAsides();
// Reset todo reminder count on new user prompt
this.#todoReminderCount = 0;
@@ -6048,6 +6077,35 @@ export class AgentSession {
return result;
}
/**
* Per-turn supersede pass: prune older `read` results that a newer read of
* the same file has made stale. Cache-aware (only fires when the suffix
* after a candidate is small or the session has been idle long enough that
* the provider prompt cache is cold), so it is cheap to run every turn.
* Gated on the `compaction.supersedeReads` setting.
*/
async #pruneSupersededReads(): Promise<{ prunedCount: number; tokensSaved: number } | undefined> {
if (!this.settings.getGroup("compaction").supersedeReads) return undefined;
const branchEntries = this.sessionManager.getBranch();
const result = pruneSupersededToolResults(
branchEntries,
this.#withPlanProtection({
supersedeKey: readToolSupersedeKey,
protectedTools: [...DEFAULT_PRUNE_CONFIG.protectedTools],
}),
);
if (result.prunedCount === 0) {
return undefined;
}
await this.sessionManager.rewriteEntries();
const sessionContext = this.buildDisplaySessionContext();
this.agent.replaceMessages(sessionContext.messages);
this.#syncTodoPhasesFromBranch();
this.#closeCodexProviderSessionsForHistoryRewrite();
return result;
}
/**
* Strip image content blocks from every message on the current branch and
* persist the rewrite. Walks `SessionManager.getBranch()` in place — both
@@ -6237,6 +6295,20 @@ export class AgentSession {
const compactionPrep = await this.#prepareCompactionFromHooks(preparation, hookCompaction);
// Strategy honored on manual /compact too. Custom instructions imply a
// directed LLM summary; a text-only model cannot read the frames back —
// both take the summarizer path (the latter loudly).
const wantsSnapcompact =
compactionPrep.kind !== "fromHook" && compactionSettings.strategy === "snapcompact" && !customInstructions;
const snapcompactReady = wantsSnapcompact && this.model.input.includes("image");
if (wantsSnapcompact && !snapcompactReady) {
this.emitNotice(
"warning",
`snapcompact needs a vision-capable model (${this.model.id} is text-only) — using an LLM summary instead`,
"compaction",
);
}
let summary: string;
let shortSummary: string | undefined;
let firstKeptEntryId: string;
@@ -6250,6 +6322,14 @@ export class AgentSession {
tokensBefore = compactionPrep.tokensBefore;
details = compactionPrep.details;
preserveData = compactionPrep.preserveData;
} else if (snapcompactReady) {
const snapcompactResult = await snapcompactCompact(preparation, { convertToLlm, model: this.model });
summary = snapcompactResult.summary;
shortSummary = snapcompactResult.shortSummary;
firstKeptEntryId = snapcompactResult.firstKeptEntryId;
tokensBefore = snapcompactResult.tokensBefore;
details = snapcompactResult.details;
preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) };
} else {
// Generate compaction result. Only convert known abort-shaped
// rejections (AbortError raised while the abort signal is set,
@@ -6680,6 +6760,10 @@ export class AgentSession {
return false;
}
// Supersede pass runs every turn, before any threshold gating: it is cheap
// (bails when no candidate) and independent of the compaction setting.
const supersedeResult = await this.#pruneSupersededReads();
const compactionSettings = this.settings.getGroup("compaction");
if (!compactionSettings.enabled || compactionSettings.strategy === "off") return false;
@@ -6688,6 +6772,9 @@ export class AgentSession {
if (assistantMessage.stopReason === "error") return false;
const pruneResult = await this.#pruneToolOutputs();
let contextTokens = calculateContextTokens(assistantMessage.usage);
if (supersedeResult) {
contextTokens = Math.max(0, contextTokens - supersedeResult.tokensSaved);
}
if (pruneResult) {
contextTokens = Math.max(0, contextTokens - pruneResult.tokensSaved);
}
@@ -7578,9 +7665,25 @@ export class AgentSession {
// "overflow" forces context-full because the input itself is broken — a handoff
// LLM call would hit the same overflow. "incomplete" is an output-side problem,
// so a handoff request on the existing context is still viable.
let action: "context-full" | "handoff" =
// so a handoff request on the existing context is still viable. Snapcompact is
// safe for every reason (it makes no LLM call at all) but requires a vision
// model to be worth anything — fall back to context-full otherwise.
let action: "context-full" | "handoff" | "snapcompact" =
compactionSettings.strategy === "handoff" && reason !== "overflow" ? "handoff" : "context-full";
if (compactionSettings.strategy === "snapcompact") {
if (this.model?.input.includes("image")) {
action = "snapcompact";
} else {
logger.warn("Snapcompact compaction requires a vision-capable model; falling back to context-full", {
model: this.model?.id,
});
this.emitNotice(
"warning",
`snapcompact needs a vision-capable model (${this.model?.id ?? "unknown"} is text-only) — using an LLM summary instead`,
"compaction",
);
}
}
await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action });
// Abort any older auto-compaction before installing this run's controller.
this.#autoCompactionAbortController?.abort();
@@ -7719,6 +7822,16 @@ export class AgentSession {
tokensBefore = compactionPrep.tokensBefore;
details = compactionPrep.details;
preserveData = compactionPrep.preserveData;
} else if (action === "snapcompact") {
// Local, deterministic: render discarded history onto PNG frames.
// No model candidates, no API key, no retry loop.
const snapcompactResult = await snapcompactCompact(preparation, { convertToLlm, model: this.model });
summary = snapcompactResult.summary;
shortSummary = snapcompactResult.shortSummary;
firstKeptEntryId = snapcompactResult.firstKeptEntryId;
tokensBefore = snapcompactResult.tokensBefore;
details = snapcompactResult.details;
preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) };
} else {
const candidates = this.#getCompactionModelCandidates(availableModels);
const retrySettings = this.settings.getGroup("retry");
@@ -8926,118 +9039,56 @@ export class AgentSession {
}
// =========================================================================
// Background-Channel IRC Exchanges
// IRC Delivery
// =========================================================================
/**
* Generate an ephemeral reply to a background message (e.g. an IRC ping from
* another agent) using this session's current model + system prompt + history.
* Deliver an IRC message into this session (recipient side; called by the
* IrcBus). Emits the `irc_message` session event for UI cards and injects
* the rendered message into the model's context as an `irc:incoming`
* custom message:
*
* The incoming message is queued for injection into the recipient's persisted
* history immediately so timeouts/abort still preserve delivery. The reply is
* computed via a side-channel `streamSimple` call (analogous to `/btw`) so it
* never blocks on the recipient's in-flight tool calls. When a reply is
* generated, it is queued separately. Injection happens immediately when the
* session is idle, otherwise it is deferred until streaming ends.
* - mid-turn → queued on the aside channel and folded in at the next step
* boundary (non-interrupting, like async-result deliveries) → "injected";
* - idle → starts a real turn with the message so the recipient wakes
* → "woken".
*
* Never blocks on the recipient's turn: the wake turn is fire-and-forget.
*/
async respondAsBackground(args: {
from: string;
message: string;
awaitReply?: boolean;
signal?: AbortSignal;
}): Promise<{ replyText: string | null }> {
const awaitReply = args.awaitReply !== false;
const incomingTimestamp = Date.now();
const incomingRecord: CustomMessage = {
async deliverIrcMessage(msg: IrcMessage): Promise<"injected" | "woken"> {
if (this.#isDisposed) {
throw new Error("Recipient session is disposed.");
}
const record: CustomMessage = {
role: "custom",
customType: "irc:incoming",
content: `[IRC \`${args.from}\` → you]\n\n${args.message}`,
content: prompt.render(ircIncomingTemplate, {
from: msg.from,
message: msg.body,
replyTo: msg.replyTo ?? "",
}),
display: true,
details: { from: args.from, message: args.message },
details: { id: msg.id, from: msg.from, message: msg.body, ...(msg.replyTo ? { replyTo: msg.replyTo } : {}) },
attribution: "agent",
timestamp: incomingTimestamp,
timestamp: msg.ts,
};
void this.#emitSessionEvent({ type: "irc_message", message: incomingRecord });
this.#forwardIrcRelayToMain({
from: args.from,
to: this.#agentId ?? "?",
body: args.message,
kind: "message",
timestamp: incomingTimestamp,
});
this.#queueBackgroundExchangeInjection([incomingRecord]);
if (!awaitReply) {
return { replyText: null };
void this.#emitSessionEvent({ type: "irc_message", message: record });
if (this.isStreaming) {
this.#pendingIrcAsides.push(record);
return "injected";
}
const incomingPrompt = prompt.render(ircIncomingTemplate, {
from: args.from,
message: args.message,
// Idle: same wake primitive the yield queue uses for async-result
// delivery — prompt the agent directly so a real turn runs.
this.agent.prompt(record).catch(error => {
logger.warn("IRC wake turn failed", { from: msg.from, to: msg.to, error: String(error) });
});
const { replyText } = await this.runEphemeralTurn({
promptText: incomingPrompt,
signal: args.signal,
});
const replyRecord: CustomMessage = {
role: "custom",
customType: "irc:autoreply",
content: `[IRC you → \`${args.from}\` (auto)]\n\n${replyText}`,
display: true,
details: { to: args.from, reply: replyText },
attribution: "agent",
timestamp: Date.now(),
};
void this.#emitSessionEvent({ type: "irc_message", message: replyRecord });
this.#forwardIrcRelayToMain({
from: this.#agentId ?? "?",
to: args.from,
body: replyText,
kind: "reply",
timestamp: replyRecord.timestamp,
});
this.#queueBackgroundExchangeInjection([replyRecord]);
return { replyText };
}
/**
* Forward an IRC exchange observation to the main agent's session UI so the
* user can see every IRC conversation in the main transcript, even when the
* main agent is not a direct participant. The relay record is display-only:
* it is NOT injected into the main agent's persisted history.
*/
#forwardIrcRelayToMain(args: {
from: string;
to: string;
body: string;
kind: "message" | "reply";
timestamp: number;
}): void {
const registry = this.#agentRegistry;
if (!registry) return;
// If this session is the main agent, the local emit already reached the main UI.
if (this.#agentId === MAIN_AGENT_ID) return;
const mainRef = registry.get(MAIN_AGENT_ID);
const mainSession = mainRef?.session;
if (!mainSession || mainSession === this) return;
const arrow = args.kind === "reply" ? "→ (auto)" : "→";
const relayRecord: CustomMessage = {
role: "custom",
customType: "irc:relay",
content: `[IRC \`${args.from}\` ${arrow} \`${args.to}\`]\n\n${args.body}`,
display: true,
details: { from: args.from, to: args.to, body: args.body, kind: args.kind },
attribution: "agent",
timestamp: args.timestamp,
};
mainSession.emitIrcRelayObservation(relayRecord);
return "woken";
}
/**
* Emit an IRC relay observation event on this session for UI rendering only.
* Does not persist the record to history. Public so other sessions can forward.
* Does not persist the record to history. Called by the IrcBus to surface
* agent↔agent traffic on the main session.
*/
emitIrcRelayObservation(record: CustomMessage): void {
void this.#emitSessionEvent({ type: "irc_message", message: record });
@@ -9049,7 +9100,7 @@ export class AgentSession {
* does not block on, or interfere with, any in-flight main turn. The
* session's history and persisted state are NOT modified by this call.
*
* Used by `respondAsBackground` (IRC) and `BtwController` (`/btw`) to share
* Used by `BtwController` (`/btw`) and `OmfgController` (`/omfg`) to share
* the snapshot + stream pipeline. The snapshot includes any in-flight
* streaming assistant text so the model sees the half-finished response
* rather than missing context.
@@ -9137,7 +9188,7 @@ export class AgentSession {
args.onTextDelta(replyText.slice(emittedReplyText.length));
}
return {
replyText: args.dedupeReply === false ? replyText.trim() : dedupeIrcReply(replyText.trim()),
replyText: args.dedupeReply === false ? replyText.trim() : dedupeEphemeralReply(replyText.trim()),
assistantMessage,
};
}
@@ -9188,46 +9239,21 @@ export class AgentSession {
return messages;
}
#queueBackgroundExchangeInjection(messages: CustomMessage[]): void {
this.#pendingBackgroundExchanges.push(messages);
if (!this.isStreaming) {
this.#flushPendingBackgroundExchanges();
return;
}
this.#scheduleBackgroundExchangeFlush();
}
#scheduleBackgroundExchangeFlush(): void {
if (this.#scheduledBackgroundExchangeFlush) return;
this.#scheduledBackgroundExchangeFlush = true;
const attempt = (): void => {
if (this.#pendingBackgroundExchanges.length === 0 || this.#isDisposed) {
this.#pendingBackgroundExchanges = [];
this.#scheduledBackgroundExchangeFlush = false;
return;
}
if (this.isStreaming) {
setTimeout(attempt, 50);
return;
}
this.#scheduledBackgroundExchangeFlush = false;
this.#flushPendingBackgroundExchanges();
};
setTimeout(attempt, 0);
}
#flushPendingBackgroundExchanges(): void {
if (this.#pendingBackgroundExchanges.length === 0) return;
const batches = this.#pendingBackgroundExchanges;
this.#pendingBackgroundExchanges = [];
for (const batch of batches) {
for (const msg of batch) {
// emitExternalEvent on message_end appends to agent state and dispatches
// to all session listeners, which in turn handle TUI rendering and
// sessionManager persistence via #handleAgentEvent.
this.agent.emitExternalEvent({ type: "message_start", message: msg });
this.agent.emitExternalEvent({ type: "message_end", message: msg });
}
/**
* Persist any IRC asides that missed their step-boundary injection (the
* message landed after the turn's last aside drain). Called at the start
* of the next prompt so the model still sees them.
*/
#flushPendingIrcAsides(): void {
if (this.#pendingIrcAsides.length === 0) return;
const records = this.#pendingIrcAsides;
this.#pendingIrcAsides = [];
for (const record of records) {
// emitExternalEvent on message_end appends to agent state and dispatches
// to all session listeners, which in turn handle TUI rendering and
// sessionManager persistence via #handleAgentEvent.
this.agent.emitExternalEvent({ type: "message_start", message: record });
this.agent.emitExternalEvent({ type: "message_end", message: record });
}
}
@@ -10032,69 +10058,6 @@ export class AgentSession {
});
}
/**
* Format the conversation as compact context for subagents.
* Includes only user messages and assistant text responses.
* Excludes: system prompt, tool definitions, tool calls/results, thinking blocks.
*/
formatCompactContext(): string {
const lines: string[] = [];
lines.push("# Conversation Context");
lines.push("");
lines.push(
"This is a summary of the parent conversation. Read this if you need additional context about what was discussed or decided.",
);
lines.push("");
for (const msg of this.messages) {
if (msg.role === "user" || msg.role === "developer") {
lines.push(msg.role === "developer" ? "## Developer" : "## User");
lines.push("");
if (typeof msg.content === "string") {
lines.push(msg.content);
} else {
for (const c of msg.content) {
if (c.type === "text") {
lines.push(c.text);
} else if (c.type === "image") {
lines.push("[Image attached]");
}
}
}
lines.push("");
} else if (msg.role === "assistant") {
const assistantMsg = msg as AssistantMessage;
// Only include text content, skip tool calls and thinking
const textParts: string[] = [];
for (const c of assistantMsg.content) {
if (c.type === "text" && c.text.trim()) {
textParts.push(c.text);
}
}
if (textParts.length > 0) {
lines.push("## Assistant");
lines.push("");
lines.push(textParts.join("\n\n"));
lines.push("");
}
} else if (msg.role === "fileMention") {
const fileMsg = msg as FileMentionMessage;
const paths = fileMsg.files.map(f => f.path).join(", ");
lines.push(`[Files referenced: ${paths}]`);
lines.push("");
} else if (msg.role === "compactionSummary") {
const compactMsg = msg as CompactionSummaryMessage;
lines.push("## Earlier Context (Summarized)");
lines.push("");
lines.push(compactMsg.summary);
lines.push("");
}
// Skip: toolResult, bashExecution, pythonExecution, branchSummary, custom, hookMessage
}
return lines.join("\n").trim();
}
// =========================================================================
// Extension System
// =========================================================================
+11 -78
View File
@@ -8,8 +8,7 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import {
type BranchSummaryMessage,
type CompactionSummaryMessage,
renderBranchSummaryContext,
renderCompactionSummaryContext,
convertMessageToLlm,
} from "@oh-my-pi/pi-agent-core/compaction/messages";
import type {
AssistantMessage,
@@ -17,7 +16,6 @@ import type {
Message,
MessageAttribution,
TextContent,
ToolResultMessage,
UserMessage,
} from "@oh-my-pi/pi-ai";
import { prompt } from "@oh-my-pi/pi-utils";
@@ -28,6 +26,7 @@ export {
type CompactionSummaryMessage,
createBranchSummaryMessage,
createCompactionSummaryMessage,
createCustomMessage,
} from "@oh-my-pi/pi-agent-core/compaction/messages";
import type { OutputMeta } from "../tools/output-meta";
@@ -59,7 +58,7 @@ export interface SkillPromptDetails {
*
* Consumers: `AgentSession.#handleAgentEvent` (stamper) writes this value;
* `EventController.#handleMessageEnd`, `AssistantMessageComponent`,
* `ui-helpers.addMessageToChat` (renderers), `SessionObserverOverlay
* `ui-helpers.addMessageToChat` (renderers), `AgentHubOverlayComponent
* #buildTranscriptLines`, `runPrintMode`, and `AcpAgent#replayAssistantMessage`
* (fallback error emission) read it via `isSilentAbort`. */
export const SILENT_ABORT_MARKER = "__omp.silent_abort__";
@@ -220,15 +219,6 @@ export function wrapSteeringForModel(messages: AgentMessage[]): AgentMessage[] {
return wrappedMessages ?? messages;
}
function getPrunedToolResultContent(message: ToolResultMessage): (TextContent | ImageContent)[] {
if (message.prunedAt === undefined) {
return message.content;
}
const textBlocks = message.content.filter((content): content is TextContent => content.type === "text");
const text = textBlocks.map(block => block.text).join("") || "[Output truncated]";
return [{ type: "text", text }];
}
/** Result of filtering image blocks out of a `(TextContent | ImageContent)[]` array. */
interface StripContentResult {
content: (TextContent | ImageContent)[];
@@ -478,26 +468,6 @@ export function sanitizeRehydratedOpenAIResponsesAssistantMessage(message: Assis
};
}
/** Convert CustomMessageEntry to AgentMessage format */
export function createCustomMessage(
customType: string,
content: string | (TextContent | ImageContent)[],
display: boolean,
details: unknown | undefined,
timestamp: string,
attribution?: MessageAttribution,
): CustomMessage {
return {
role: "custom",
customType,
content,
display,
details,
attribution,
timestamp: new Date(timestamp).getTime(),
};
}
/**
* Transform AgentMessages (including custom types) to LLM-compatible Messages.
*
@@ -530,43 +500,6 @@ export function convertToLlm(messages: AgentMessage[]): Message[] {
attribution: "user",
timestamp: m.timestamp,
};
case "custom":
case "hookMessage": {
const content = typeof m.content === "string" ? [{ type: "text" as const, text: m.content }] : m.content;
const role = "developer";
const attribution = m.attribution;
return {
role,
content,
attribution,
timestamp: m.timestamp,
};
}
case "branchSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderBranchSummaryContext(m.summary),
},
],
attribution: "agent",
timestamp: m.timestamp,
};
case "compactionSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderCompactionSummaryContext(m.summary),
},
],
attribution: "agent",
providerPayload: m.providerPayload,
timestamp: m.timestamp,
};
case "fileMention": {
const fileContents = m.files
.map(file => {
@@ -587,18 +520,18 @@ export function convertToLlm(messages: AgentMessage[]): Message[] {
timestamp: m.timestamp,
};
}
case "custom":
case "hookMessage":
case "branchSummary":
case "compactionSummary":
case "user":
return { ...m, attribution: m.attribution ?? "user" };
case "developer":
return { ...m, attribution: m.attribution ?? "agent" };
case "assistant":
return m;
case "toolResult":
return {
...m,
content: getPrunedToolResultContent(m as ToolResultMessage),
attribution: m.attribution ?? "agent",
};
// Core roles share one transformer with agent-core —
// duplicating them here is how snapcompact frames once
// silently fell off the provider request.
return convertMessageToLlm(m);
default:
m satisfies never;
return undefined;
@@ -0,0 +1,246 @@
/**
* Concise markdown transcript serializer for `history://` URLs.
*
* Unlike `session-dump-format.ts` (verbose `/dump` export), this emits a
* compressed transcript: full user/assistant/developer text, tool call +
* result pairs collapsed to single lines, thinking elided, custom messages
* as one-liners. No system prompt, no tool catalog, no config sections.
*/
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import { INTENT_FIELD } from "@oh-my-pi/pi-agent-core";
import type { AssistantMessage, ImageContent, TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai";
import type {
BashExecutionMessage,
BranchSummaryMessage,
CompactionSummaryMessage,
CustomMessage,
FileMentionMessage,
HookMessage,
PythonExecutionMessage,
} from "./messages";
export interface HistoryFormatOptions {
/** Optional H1 prepended to the transcript. */
title?: string;
}
/** Max length of the primary-arg summary inside `→ tool(...)` lines. */
const PRIMARY_ARG_MAX = 120;
/** Per-tool preference order for the most informative scalar argument. */
const PRIMARY_ARG_KEYS = [
"path",
"file_path",
"filePath",
"command",
"cmd",
"pattern",
"url",
"query",
"prompt",
"assignment",
"message",
"op",
"name",
"id",
] as const;
/** Collapse whitespace runs and truncate to `max` chars with an ellipsis. */
function oneLine(text: string, max = PRIMARY_ARG_MAX): string {
const flat = text.replace(/\s+/g, " ").trim();
return flat.length > max ? `${flat.slice(0, max - 1)}…` : flat;
}
/** Join the text blocks of a string-or-blocks content field. Images become `[image]`. */
function contentToText(content: string | readonly (TextContent | ImageContent)[]): string {
if (typeof content === "string") return content;
const parts: string[] = [];
for (const block of content) {
if (block.type === "text") parts.push(block.text);
else parts.push("[image]");
}
return parts.join("\n");
}
function lineCount(text: string): number {
if (!text) return 0;
return text.split("\n").length;
}
/** Pick the most informative scalar argument of a tool call. */
function primaryArg(args: Record<string, unknown> | undefined): string {
if (!args || typeof args !== "object") return "";
for (const key of PRIMARY_ARG_KEYS) {
const value = args[key];
if (typeof value === "string" && value.length > 0) return oneLine(value);
if (Array.isArray(value) && value.length > 0 && value.every(v => typeof v === "string")) {
return oneLine(value.join(", "));
}
}
// Fallback: first non-intent string arg, then a compact JSON of the args.
const rest: Record<string, unknown> = {};
let restCount = 0;
for (const key in args) {
if (key === INTENT_FIELD) continue;
const value = args[key];
if (typeof value === "string" && value.length > 0) return oneLine(value);
rest[key] = value;
restCount++;
}
if (restCount === 0) return "";
try {
return oneLine(JSON.stringify(rest));
} catch {
return "";
}
}
/** One line per tool call: `→ read(src/foo.ts:50-80) ⇒ ok · 31 lines`. */
function toolCallLine(
name: string,
args: Record<string, unknown> | undefined,
result: ToolResultMessage | undefined,
): string {
const head = `→ ${name}(${primaryArg(args)})`;
if (!result) return `${head} ⇒ pending`;
const text = contentToText(result.content);
const lines = lineCount(text);
const count = `${lines} ${lines === 1 ? "line" : "lines"}`;
if (result.isError) {
const firstLine = oneLine(text.split("\n", 1)[0] ?? "");
return firstLine ? `${head} ⇒ error · ${count} — ${firstLine}` : `${head} ⇒ error · ${count}`;
}
return `${head} ⇒ ok · ${count}`;
}
/** One line for a user-initiated `!`/`$` execution. */
function executionLine(
kind: "bash" | "python",
source: string,
msg: BashExecutionMessage | PythonExecutionMessage,
): string {
const status = msg.cancelled
? "cancelled"
: msg.exitCode !== undefined && msg.exitCode !== 0
? `error · exit ${msg.exitCode}`
: "ok";
const lines = lineCount(msg.output);
return `→ ${kind}! ${oneLine(source)} ⇒ ${status} · ${lines} ${lines === 1 ? "line" : "lines"}`;
}
/** One-liner for custom/hook messages: `[irc] A → B: body…`. */
function customOneLiner(msg: CustomMessage | HookMessage): string {
const details = (msg.details ?? {}) as Record<string, unknown>;
const str = (key: string): string => (typeof details[key] === "string" ? (details[key] as string) : "");
switch (msg.customType) {
case "irc:incoming":
return `[irc] ${str("from") || "?"} → me: ${oneLine(str("message"))}`;
case "irc:relay":
return `[irc] ${str("from") || "?"} → ${str("to") || "?"}: ${oneLine(str("body"))}`;
case "async-result": {
const jobs = Array.isArray(details.jobs) && details.jobs.length > 0 ? details.jobs : [details];
const labels = jobs
.map(job => {
const j = (job ?? {}) as Record<string, unknown>;
return typeof j.label === "string" && j.label ? j.label : typeof j.jobId === "string" ? j.jobId : "job";
})
.join(", ");
return `[async-result] ${oneLine(labels)}`;
}
default:
return `[${msg.customType}] ${oneLine(contentToText(msg.content))}`;
}
}
/**
* Format a session's message array as a concise markdown transcript.
*
* `messages` is the session's in-memory message array (or the read-only
* equivalent loaded from a session file) — the same shapes
* `session-dump-format.ts` consumes.
*/
export function formatSessionHistoryMarkdown(messages: unknown[], opts?: HistoryFormatOptions): string {
const typed = messages as AgentMessage[];
const lines: string[] = [];
if (opts?.title) {
lines.push(`# ${opts.title}`, "");
}
// Index tool results by call id so each toolCall collapses to one line.
const resultsByCallId = new Map<string, ToolResultMessage>();
for (const msg of typed) {
if (msg.role === "toolResult") {
resultsByCallId.set(msg.toolCallId, msg);
}
}
const consumed = new Set<string>();
for (const msg of typed) {
switch (msg.role) {
case "user":
case "developer": {
const text = contentToText(msg.content);
if (!text.trim()) break;
lines.push(`## ${msg.role}`, "", text, "");
break;
}
case "assistant": {
const assistantMsg = msg as AssistantMessage;
const body: string[] = [];
for (const block of assistantMsg.content) {
if (block.type === "text") {
if (block.text.trim()) body.push(block.text);
} else if (block.type === "toolCall") {
const result = resultsByCallId.get(block.id);
if (result) consumed.add(block.id);
body.push(toolCallLine(block.name, block.arguments, result));
}
// thinking / redactedThinking elided entirely
}
if (body.length === 0) break;
lines.push("## assistant", "", ...body, "");
break;
}
case "toolResult": {
// Normally consumed by its toolCall; orphans (e.g. truncated history) get their own line.
if (consumed.has(msg.toolCallId)) break;
lines.push(toolCallLine(msg.toolName, undefined, msg), "");
break;
}
case "bashExecution": {
const bashMsg = msg as BashExecutionMessage;
if (bashMsg.excludeFromContext) break;
lines.push(executionLine("bash", bashMsg.command, bashMsg), "");
break;
}
case "pythonExecution": {
const pythonMsg = msg as PythonExecutionMessage;
if (pythonMsg.excludeFromContext) break;
lines.push(executionLine("python", pythonMsg.code, pythonMsg), "");
break;
}
case "custom":
case "hookMessage": {
lines.push(customOneLiner(msg as CustomMessage | HookMessage), "");
break;
}
case "branchSummary": {
const branchMsg = msg as BranchSummaryMessage;
lines.push(`[branch] from ${branchMsg.fromId}: ${oneLine(branchMsg.summary)}`, "");
break;
}
case "compactionSummary": {
const compactMsg = msg as CompactionSummaryMessage;
lines.push(`[compaction] ${oneLine(compactMsg.summary)}`, "");
break;
}
case "fileMention": {
const fileMsg = msg as FileMentionMessage;
lines.push(`[file-mention] ${oneLine(fileMsg.files.map(f => f.path).join(", "))}`, "");
break;
}
}
}
return `${lines.join("\n").trim()}\n`;
}
@@ -27,6 +27,7 @@ import {
Snowflake,
toError,
} from "@oh-my-pi/pi-utils";
import { getPreservedSnapcompactArchive, snapcompactImages } from "@oh-my-pi/snapcompact";
import { ArtifactManager } from "./artifacts";
import {
type BlobPutOptions,
@@ -544,6 +545,17 @@ export function getLatestCompactionEntry(entries: SessionEntry[]): CompactionEnt
return null;
}
export interface BuildSessionContextOptions {
/**
* Build the full-history display transcript instead of the LLM context:
* every path entry in chronological order, with each compaction emitted
* inline as a `compactionSummary` message at the position it fired rather
* than replacing the history before it. Display-only — never send the
* result to a provider.
*/
transcript?: boolean;
}
/**
* Build the session context from entries using tree traversal.
* If leafId is provided, walks from that entry to root.
@@ -553,6 +565,7 @@ export function buildSessionContext(
entries: SessionEntry[],
leafId?: string | null,
byId?: Map<string, SessionEntry>,
options?: BuildSessionContextOptions,
): SessionContext {
// Build uuid index if not available
if (!byId) {
@@ -692,7 +705,29 @@ export function buildSessionContext(
}
};
if (compaction) {
if (options?.transcript) {
// Display transcript: every entry in chronological order. Compactions do
// not erase prior history here — each renders inline (as a divider in the
// TUI) at the point it fired, with any snapcompact frames re-attached so
// the component can report them.
for (const entry of path) {
if (entry.type === "compaction") {
const snapcompactArchive = getPreservedSnapcompactArchive(entry.preserveData);
messages.push(
createCompactionSummaryMessage(
entry.summary,
entry.tokensBefore,
entry.timestamp,
entry.shortSummary,
undefined,
snapcompactArchive ? snapcompactImages(snapcompactArchive) : undefined,
),
);
} else {
appendMessage(entry);
}
}
} else if (compaction) {
const providerPayload: ProviderPayload | undefined = (() => {
const candidate = compaction.preserveData?.openaiRemoteCompaction;
if (!candidate || typeof candidate !== "object") return undefined;
@@ -707,7 +742,9 @@ export function buildSessionContext(
})();
const remoteReplacementHistory = providerPayload?.items;
// Emit summary first
// Emit summary first; re-attach any archived snapcompact frames so the
// model can keep reading the archived history after every context rebuild.
const snapcompactArchive = getPreservedSnapcompactArchive(compaction.preserveData);
messages.push(
createCompactionSummaryMessage(
compaction.summary,
@@ -715,6 +752,7 @@ export function buildSessionContext(
compaction.timestamp,
compaction.shortSummary,
providerPayload,
snapcompactArchive ? snapcompactImages(snapcompactArchive) : undefined,
),
);
@@ -957,6 +995,21 @@ async function resolveBlobRefsInEntries(entries: FileEntry[], blobStore: BlobSto
await Promise.all(promises);
}
/**
* Read-only message view of a session file: load entries, migrate to the
* current version, resolve blob refs, and build the context along the
* persisted leaf path (last entry). Does NOT create a writer or take the
* session lock — safe to call against a file another session is writing.
*/
export async function loadSessionMessagesReadOnly(filePath: string): Promise<AgentMessage[]> {
const entries = await loadEntriesFromFile(filePath);
if (entries.length === 0) return [];
migrateToCurrentVersion(entries);
await resolveBlobRefsInEntries(entries, new BlobStore(getBlobsDir()));
const sessionEntries = entries.filter((e): e is SessionEntry => e.type !== "session");
return buildSessionContext(sessionEntries).messages;
}
/**
* Lightweight metadata for a session file, used in session picker UI.
* Uses lazy getters to defer string formatting until actually displayed.
@@ -3205,11 +3258,12 @@ export class SessionManager {
}
/**
* Build the session context (what gets sent to the LLM).
* Build the session context (what gets sent to the LLM), or — with
* `{ transcript: true }` — the full-history display transcript.
* Uses tree traversal from current leaf.
*/
buildSessionContext(): SessionContext {
return buildSessionContext(this.getEntries(), this.#leafId, this.#byId);
buildSessionContext(options?: BuildSessionContextOptions): SessionContext {
return buildSessionContext(this.getEntries(), this.#leafId, this.#byId, options);
}
/** Strip stale OpenAI Responses assistant replay metadata from loaded in-memory entries. */
@@ -570,6 +570,66 @@ export function truncateMiddle(content: string, options: TruncationOptions = {})
};
}
// =============================================================================
// Inline byte cap — final defense at the tool-result boundary
// =============================================================================
/** Options for {@link enforceInlineByteCap}. */
export interface InlineByteCapOptions {
/** Inline byte budget. Defaults to {@link DEFAULT_MAX_BYTES}. */
maxBytes?: number;
/** What the text is, for the elision marker (e.g. "bash output"). */
label: string;
/**
* Persist the full text as a session artifact. When an artifact id is
* returned, a `[raw output: artifact://<id>]` footer is appended so the
* elided bytes stay recoverable.
*/
saveArtifact?: (full: string) => string | undefined | Promise<string | undefined>;
}
/** Drop the partial last line of a head window (keep it if there is no newline at all). */
function trimHeadToLineBoundary(text: string): string {
const idx = text.lastIndexOf(NL);
return idx > 0 ? text.substring(0, idx) : text;
}
/** Drop the partial first line of a tail window (keep it if there is no newline at all). */
function trimTailToLineBoundary(text: string): string {
const idx = text.indexOf(NL);
if (idx < 0 || idx === text.length - 1) return text;
return text.substring(idx + 1);
}
/**
* Final-defense inline size guard for tool results.
*
* No-op when `text` fits within `maxBytes` (the common path). Otherwise keeps
* ~60% of the budget from the head and ~25% from the tail — cut on line
* boundaries, never splitting a multi-byte UTF-8 sequence — with an elision
* marker between. The remaining ~15% is slack for the marker and the optional
* `[raw output: artifact://<id>]` footer, so the result stays under `maxBytes`.
*/
export async function enforceInlineByteCap(text: string, options: InlineByteCapOptions): Promise<string> {
const maxBytes = options.maxBytes ?? DEFAULT_MAX_BYTES;
if (maxBytes <= 0) return text;
const totalBytes = Buffer.byteLength(text, "utf-8");
if (totalBytes <= maxBytes) return text;
const head = trimHeadToLineBoundary(truncateHeadBytes(text, Math.floor(maxBytes * 0.6)).text);
const tail = trimTailToLineBoundary(truncateTailBytes(text, Math.floor(maxBytes * 0.25)).text);
const elidedBytes = Math.max(0, totalBytes - Buffer.byteLength(head, "utf-8") - Buffer.byteLength(tail, "utf-8"));
const marker = `[… elided ${elidedBytes} bytes of ${options.label} …]`;
let composed = `${head}\n${marker}\n${tail}`;
const artifactId = await options.saveArtifact?.(text);
if (artifactId) {
const sep = composed.endsWith(NL) ? "" : NL;
composed += `${sep}[raw output: artifact://${artifactId}]`;
}
return composed;
}
// =============================================================================
// TailBuffer — ring-style tail buffer with lazy joining
// =============================================================================
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -85,15 +85,4 @@ export class AgentOutputManager {
await this.#ensureInitialized();
return this.#allocateUnique(id);
}
/**
* Allocate unique IDs for a batch of tasks.
*
* @param ids Array of requested IDs
* @returns Array of unique IDs in same order
*/
async allocateBatch(ids: string[]): Promise<string[]> {
await this.#ensureInitialized();
return ids.map(id => this.#allocateUnique(id));
}
}

Some files were not shown because too many files have changed in this diff Show More