From a9229b0f26b73d9451ba94aaeecd81d5999dfd75 Mon Sep 17 00:00:00 2001 From: can1357 Date: Fri, 12 Jun 2026 09:11:41 +0200 Subject: [PATCH] feat(snapcompact): clipped snapcompact PNG frame height to rendered text rows - Adjusted snapcompact rendering to compute used rows from text, dim toggles, and doc line breaks, then derive output height from usedRows x lineRepeat x cellHeight. - Updated indexed and RGB PNG encoders to accept explicit canvas width and height so native renders now emit non-square frames matching actual content. - Expanded Rust and TypeScript tests and updated docs/changelogs to assert and describe variable-height frame behavior. --- crates/pi-natives/src/snapcompact.rs | 129 ++++++++++++++---- docs/compaction.md | 2 +- packages/natives/CHANGELOG.md | 4 + packages/natives/native/index.d.ts | 14 +- packages/snapcompact/CHANGELOG.md | 1 + packages/snapcompact/src/snapcompact.ts | 7 +- packages/snapcompact/test/snapcompact.test.ts | 4 +- 7 files changed, 127 insertions(+), 34 deletions(-) diff --git a/crates/pi-natives/src/snapcompact.rs b/crates/pi-natives/src/snapcompact.rs index ad5023347..32a612a92 100644 --- a/crates/pi-natives/src/snapcompact.rs +++ b/crates/pi-natives/src/snapcompact.rs @@ -1,7 +1,8 @@ //! Snapcompact frame rendering. //! -//! Rasterizes pre-normalized conversation text onto a square bitmap using one -//! of the bundled public-domain pixel fonts, then encodes it as PNG: +//! Rasterizes pre-normalized conversation text onto a `size`-wide bitmap +//! (height hugs the rows the text actually needs) using one of the bundled +//! public-domain pixel fonts, then encodes it as PNG: //! //! - `5x8` — X.org BDF font (legacy shape). //! - `8x8` — unscii-8 hex font (Latin-1 subset), the square cell that won the @@ -188,6 +189,25 @@ struct Grid { cell_h: usize, } +/// Grid rows the text actually occupies, so the canvas height hugs the +/// content instead of padding the frame to a full square. Mirrors the +/// renderers' cell accounting: dim toggles are zero-width, every other code +/// point (including `U+2588` and glyphs missing from the font) consumes one +/// cell; doc layout fills one row per `\n`-separated line down the first +/// column before spilling into the second. +fn used_rows(text: &str, grid: &Grid, doc: bool) -> usize { + let rows = if doc { + text.split('\n').count() + } else { + let cells = text + .chars() + .filter(|&ch| !matches!(ch as u32, DIM_ON | DIM_OFF)) + .count(); + cells.div_ceil(grid.cols) + }; + rows.clamp(1, grid.rows) +} + /// Paint the pale highlight bands behind line copies after the first. fn fill_repeat_bands(pixels: &mut [u8], width: usize, height: usize, grid: &Grid) { if grid.repeat <= 1 { @@ -517,11 +537,11 @@ fn resize_rgb(src: &[f32], sw: usize, sh: usize, dw: usize, dh: usize) -> Vec Vec { - let row_bytes = size.div_ceil(2); - let mut packed = vec![0u8; row_bytes * size]; - for y in 0..size { - let src = &pixels[y * size..(y + 1) * size]; +fn pack_nibbles(pixels: &[u8], width: usize, height: usize) -> Vec { + let row_bytes = width.div_ceil(2); + let mut packed = vec![0u8; row_bytes * height]; + for y in 0..height { + let src = &pixels[y * width..(y + 1) * width]; let dst = &mut packed[y * row_bytes..(y + 1) * row_bytes]; for (x, &px) in src.iter().enumerate() { dst[x / 2] |= px << (4 * (1 - x % 2)); @@ -535,7 +555,8 @@ fn pack_nibbles(pixels: &[u8], size: usize) -> Vec { /// encode time without helping deflate). fn encode_indexed_png( pixels: &[u8], - size: usize, + width: usize, + height: usize, compression: png::Compression, ) -> Result> { let mut palette = Vec::with_capacity(PALETTE.len() * 3); @@ -543,7 +564,7 @@ fn encode_indexed_png( palette.extend_from_slice(&rgb); } let mut out = Vec::new(); - let mut encoder = png::Encoder::new(&mut out, size as u32, size as u32); + let mut encoder = png::Encoder::new(&mut out, width as u32, height as u32); encoder.set_color(png::ColorType::Indexed); encoder.set_depth(png::BitDepth::Four); encoder.set_palette(Cow::Owned(palette)); @@ -555,7 +576,7 @@ fn encode_indexed_png( .write_header() .map_err(|err| Error::from_reason(format!("Failed to write PNG header: {err}")))?; writer - .write_image_data(&pack_nibbles(pixels, size)) + .write_image_data(&pack_nibbles(pixels, width, height)) .map_err(|err| Error::from_reason(format!("Failed to write PNG data: {err}")))?; writer .finish() @@ -565,9 +586,14 @@ fn encode_indexed_png( /// Encode an interleaved RGB8 buffer as PNG. Stretched frames are /// continuous-tone, so adaptive filtering (the `Balanced` default) helps. -fn encode_rgb_png(pixels: &[u8], size: usize, compression: png::Compression) -> Result> { +fn encode_rgb_png( + pixels: &[u8], + width: usize, + height: usize, + compression: png::Compression, +) -> Result> { let mut out = Vec::new(); - let mut encoder = png::Encoder::new(&mut out, size as u32, size as u32); + let mut encoder = png::Encoder::new(&mut out, width as u32, height as u32); encoder.set_color(png::ColorType::Rgb); encoder.set_depth(png::BitDepth::Eight); encoder.set_compression(compression); @@ -591,7 +617,9 @@ fn encode_rgb_png(pixels: &[u8], size: usize, compression: png::Compression) -> #[napi(object)] #[derive(Default)] pub struct SnapcompactRenderOptions { - /// Frame edge in pixels. + /// Frame width in pixels; also bounds the grid rows + /// (`floor(size/cellHeight/lineRepeat)`). Output height hugs the rows the + /// text actually uses instead of padding to a square. pub size: u32, /// Bundled font: `"5x8"`, `"6x12"`, `"8x13"` (X.org BDF) or `"8x8"` /// (unscii-8). Default `"5x8"`. @@ -618,10 +646,12 @@ pub struct SnapcompactRenderOptions { pub columns: Option, } -/// Render one snapcompact frame: print pre-normalized text onto a square -/// bitmap and encode it as PNG. +/// Render one snapcompact frame: print pre-normalized text onto a +/// `size`-wide bitmap and encode it as PNG. /// -/// The glyph grid holds `floor(size/cellWidth) * +/// The bitmap height hugs the rows the text actually occupies +/// (`usedRows * lineRepeat * cellHeight`), so a partially filled frame never +/// pays for blank padding rows. The glyph grid holds `floor(size/cellWidth) * /// floor(size/cellHeight/lineRepeat)` characters; input beyond that is ignored /// (the caller chunks text to capacity). Native-cell shapes encode as 4-bit /// indexed PNG; stretched shapes (target cell != font cell) encode as RGB. @@ -681,6 +711,10 @@ pub fn render_snapcompact_png( "Frame size {size} cannot fit a {target_w}x{target_h} cell grid (repeat {repeat})" ))); } + // Tight canvas: width stays the frame edge (the reading geometry the + // caller derives cols from), height hugs the rows the text needs. + let used = used_rows(&text, &grid, doc); + let height = used * grid.repeat * grid.cell_h; let stretch = options.stretch != Some(false) && (target_w, target_h) != (font.cell_w, font.cell_h); @@ -689,12 +723,12 @@ pub fn render_snapcompact_png( // cell box (the natural cell, or natural glyphs on a padded pitch // when `stretch: false`). let pixels = if doc { - render_doc_bitmap(&text, size, size, font, &grid, black_ink) + render_doc_bitmap(&text, size, height, font, &grid, black_ink) } else { - render_bitmap(&text, size, size, font, &grid, black_ink) + render_bitmap(&text, size, height, font, &grid, black_ink) }; return Ok(STANDARD - .encode(encode_indexed_png(&pixels, size, png::Compression::Balanced)?) + .encode(encode_indexed_png(&pixels, size, height, png::Compression::Balanced)?) .into()); } @@ -703,9 +737,9 @@ pub fn render_snapcompact_png( // resample to the target cell, paste onto the white frame. let native = Grid { cell_w: font.cell_w, cell_h: font.cell_h, ..grid }; let src_w = grid.cols * font.cell_w; - let src_h = grid.rows * grid.repeat * font.cell_h; + let src_h = used * grid.repeat * font.cell_h; let dst_w = grid.cols * target_w; - let dst_h = grid.rows * grid.repeat * target_h; + let dst_h = used * grid.repeat * target_h; let indexed = if doc { render_doc_bitmap(&text, src_w, src_h, font, &native, black_ink) } else { @@ -719,8 +753,8 @@ pub fn render_snapcompact_png( dst[2] = f32::from(b); } let resized = resize_rgb(&rgb, src_w, src_h, dst_w, dst_h); - let mut frame = vec![255u8; size * size * 3]; - for y in 0..dst_h.min(size) { + let mut frame = vec![255u8; size * dst_h * 3]; + for y in 0..dst_h { let src_row = &resized[y * dst_w * 3..(y + 1) * dst_w * 3]; let dst_row = &mut frame[y * size * 3..]; for (d, &s) in dst_row[..dst_w.min(size) * 3].iter_mut().zip(src_row) { @@ -728,7 +762,7 @@ pub fn render_snapcompact_png( } } Ok(STANDARD - .encode(encode_rgb_png(&frame, size, png::Compression::Balanced)?) + .encode(encode_rgb_png(&frame, size, dst_h, png::Compression::Balanced)?) .into()) } @@ -940,7 +974,9 @@ mod tests { assert_eq!(png[25], 3, "8on16 must stay indexed"); // IHDR width/height live at bytes 16..24, big-endian. let dim = |off: usize| u32::from_be_bytes(png[off..off + 4].try_into().unwrap()); - assert_eq!((dim(16), dim(20)), (128, 128), "declared geometry must match"); + // "Hello there. General Kenobi!" is 28 chars on a 16-col grid: 2 rows + // of the 16px pitch — the height hugs them instead of padding to 128. + assert_eq!((dim(16), dim(20)), (128, 32), "declared geometry must match"); // Glyph ink must sit in the top 13px of every 16px pitch row. let grid = Grid { cols: 16, rows: 8, repeat: 1, cell_w: 8, cell_h: 16 }; @@ -990,6 +1026,49 @@ mod tests { assert!(!inks.contains(&2), "grid mode must not advance hue across newline"); } + #[test] + fn frame_height_hugs_used_rows() { + let dims = |png: &[u8]| { + let dim = |off: usize| u32::from_be_bytes(png[off..off + 4].try_into().unwrap()); + (dim(16), dim(20)) + }; + let render = |text: &str, opts: SnapcompactRenderOptions| { + png_bytes(render_snapcompact_png(text.into(), opts).unwrap()) + }; + let opts_8x8 = + || SnapcompactRenderOptions { size: 64, font: Some("8x8".into()), ..Default::default() }; + // 8 cols of 8x8 cells: 10 chars span 2 rows -> 16px tall. + assert_eq!(dims(&render("0123456789", opts_8x8())), (64, 16)); + // Dim toggles are zero-width and must not add a row. + assert_eq!(dims(&render("\u{e}01234567\u{f}", opts_8x8())), (64, 8)); + // Capacity-filling text keeps the full grid height. + assert_eq!(dims(&render(&"x".repeat(64), opts_8x8())), (64, 64)); + // Repeat shapes hug `usedRows * repeat` copy bands. + let repeated = + render("0123456789", SnapcompactRenderOptions { line_repeat: Some(2), ..opts_8x8() }); + assert_eq!(dims(&repeated), (64, 32)); + // Doc layout counts `\n` lines down the first column. + let doc = render("Hello there.\nSecond line", SnapcompactRenderOptions { + size: 256, + font: Some("8x13".into()), + cell_width: Some(8), + cell_height: Some(16), + stretch: Some(false), + columns: Some(2), + ..Default::default() + }); + assert_eq!(dims(&doc), (256, 32)); + // The stretch path hugs too (RGB output, 6x6 target cells). + let stretched = render("0123456789ab", SnapcompactRenderOptions { + size: 60, + font: Some("8x8".into()), + cell_width: Some(6), + cell_height: Some(6), + ..Default::default() + }); + assert_eq!(dims(&stretched), (60, 12)); + } + #[test] fn columns_validates_and_renders_doc_frames() { assert!( diff --git a/docs/compaction.md b/docs/compaction.md index 7ac1150a3..54d589b25 100644 --- a/docs/compaction.md +++ b/docs/compaction.md @@ -131,7 +131,7 @@ The automatic paths are intentionally different: `compaction.strategy: "snapcompact"` replaces the LLM summarization call with a local, deterministic archival pass (`compact` from `@oh-my-pi/snapcompact`): -- The discarded history is serialized, whitespace-collapsed, and printed onto model-aware square PNG frames using bundled public-domain pixel fonts. The shape resolves from the **model id** when the model line was measured — Claude reads X.org `6x12` glyphs with dimmed stopwords (`6x12-dim`), Gemini reads two word-wrapped columns of `8x13` glyphs with sentence-hue ink and dimmed stopwords (`doc-8on16-sent-dim`), GPT/Kimi/GLM read `8x13` glyphs on a 16px pitch (`8on16-bw`) — so a Claude routed through Vertex or OpenRouter keeps its Claude shape. Unmeasured models fall back to their wire API family (Anthropic-family/unknown → `6x12-dim`, Google → `doc-8on16-sent-dim`, OpenAI-compatible → `8on16-bw`); billing (token estimate, OpenAI's `detail: "original"` hint) always follows the API carrying the request. The `snapcompact.shape` setting (default `auto`) forces one of the research-eval variants instead: square grids (`8x8r`/`8x8u`/`6x6u`/`5x8` × sentence-hue/black ink) or the per-model eval winners (`6x12-dim`, `8x13-bw`, `8on16-bw`, and the two-column word-wrapped `doc-8on16-bw`/`-sent`/`-sent-dim`, where `dim` prints stopwords in gray). A forced variant keeps its geometry but is re-priced for the target provider's image billing. The same setting governs inline system-prompt/tool-result imaging (`snapcompact.systemPrompt`, `snapcompact.toolResults`). +- The discarded history is serialized, whitespace-collapsed, and printed onto model-aware PNG frames (frame width fixed per shape; frame height hugs the rows actually printed) using bundled public-domain pixel fonts. The shape resolves from the **model id** when the model line was measured — Claude reads X.org `6x12` glyphs with dimmed stopwords (`6x12-dim`), Gemini reads two word-wrapped columns of `8x13` glyphs with sentence-hue ink and dimmed stopwords (`doc-8on16-sent-dim`), GPT/Kimi/GLM read `8x13` glyphs on a 16px pitch (`8on16-bw`) — so a Claude routed through Vertex or OpenRouter keeps its Claude shape. Unmeasured models fall back to their wire API family (Anthropic-family/unknown → `6x12-dim`, Google → `doc-8on16-sent-dim`, OpenAI-compatible → `8on16-bw`); billing (token estimate, OpenAI's `detail: "original"` hint) always follows the API carrying the request. The `snapcompact.shape` setting (default `auto`) forces one of the research-eval variants instead: square grids (`8x8r`/`8x8u`/`6x6u`/`5x8` × sentence-hue/black ink) or the per-model eval winners (`6x12-dim`, `8x13-bw`, `8on16-bw`, and the two-column word-wrapped `doc-8on16-bw`/`-sent`/`-sent-dim`, where `dim` prints stopwords in gray). A forced variant keeps its geometry but is re-priced for the target provider's image billing. The same setting governs inline system-prompt/tool-result imaging (`snapcompact.systemPrompt`, `snapcompact.toolResults`). - Serialization keeps the archive conversation-dense: tool results are truncated head+tail (default 2,000 chars at a 0.6 head ratio), tool-call argument values are capped per value (500) and per call (2,000), and tool output is printed in dim gray ink so conversation reads louder than tool noise. All budgets and the dimming are configurable via `SerializeOptions` (`toolResultMaxChars`, `toolArgMaxChars`, `toolCallMaxChars`, `truncateHeadRatio`, `dimToolResults`). - Frames persist under `CompactionEntry.preserveData.snapcompact` and are re-attached to the `compactionSummary` message as image blocks on every context rebuild; the entry's `summary` is a deterministic reading guide (grid geometry, role tags, truncation notes) plus the usual file-operation lists. - Later compactions carry earlier frames forward. The frame budget is provider-aware (`providerFrameBudget`): the per-provider image cap clamped to 8 (`MAX_FRAMES`) — OpenRouter hard-caps requests at 8 images and silently drops the excess, unknown providers get a safe floor of 5. Beyond the budget the archive fades from the middle out: the earliest frame (session head — the original request, or the filmed summary of older history) is pinned, and the oldest *unpinned* frames are evicted. Pages of the *current* compaction that no longer fit are never rendered or dropped — the newest unframed slice survives verbatim as a text tail on the summary (`Archive.textTail`, capped at two frame capacities with middle elision) and is folded back into frames by the next compaction. If the previous compaction was text-based, its summary is printed at the head of the frame archive as `[Summary of earlier history]`. diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index e0344dec0..87d0c21bc 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -7,6 +7,10 @@ - Added the X.org misc `6x12` and `8x13` BDF fonts (public domain, vendored in `crates/pi-natives/src/fonts/`) to `renderSnapcompactPng`, alongside two new options for the snapcompact eval-winner shapes: `stretch: false` renders glyphs at natural size on the requested cell box while keeping the 4-bit indexed encoder (e.g. 8x13 glyphs on an 8x16 pitch, the "8on16" shapes), and `columns: 2` flows pre-wrapped newline-separated lines down two newspaper columns with a 3-cell gutter (the "doc" shapes); in doc mode sentence hues also advance across a terminator followed by a newline - Added a line-break marker to `renderSnapcompactPng`: `U+2588` (FULL BLOCK) fills its entire cell box with pitch-black ink in both grid and doc layouts, ignoring the sentence hue and dim state, and counts as a sentence boundary after a `.`/`!`/`?` terminator +### Changed + +- `renderSnapcompactPng` now clips the frame height to the text: the PNG stays `size` pixels wide but is only `usedRows * lineRepeat * cellHeight` tall (dim toggles are zero-width; doc layout counts `\n`-separated lines), so a partially filled frame no longer pads to a full square of blank rows + ## [15.11.4] - 2026-06-12 ### Fixed diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 4ef09b3c9..aa3e9a7c3 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -1285,10 +1285,12 @@ export interface PtyStartOptions { export declare function readImageFromClipboard(): Promise /** - * Render one snapcompact frame: print pre-normalized text onto a square - * bitmap and encode it as PNG. + * Render one snapcompact frame: print pre-normalized text onto a + * `size`-wide bitmap and encode it as PNG. * - * The glyph grid holds `floor(size/cellWidth) * + * The bitmap height hugs the rows the text actually occupies + * (`usedRows * lineRepeat * cellHeight`), so a partially filled frame never + * pays for blank padding rows. The glyph grid holds `floor(size/cellWidth) * * floor(size/cellHeight/lineRepeat)` characters; input beyond that is ignored * (the caller chunks text to capacity). Native-cell shapes encode as 4-bit * indexed PNG; stretched shapes (target cell != font cell) encode as RGB. @@ -1433,7 +1435,11 @@ export declare function sliceWithWidth(line: string, startCol: number, length: n /** Shape options for one snapcompact frame. */ export interface SnapcompactRenderOptions { - /** Frame edge in pixels. */ + /** + * Frame width in pixels; also bounds the grid rows + * (`floor(size/cellHeight/lineRepeat)`). Output height hugs the rows the + * text actually uses instead of padding to a square. + */ size: number /** * Bundled font: `"5x8"`, `"6x12"`, `"8x13"` (X.org BDF) or `"8x8"` diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index 4bb6ca83a..da095ab62 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -14,6 +14,7 @@ ### Changed +- Frames are no longer padded to a square: the native renderer clips each PNG's height to the text rows actually printed, so a partially filled frame (typically the newest) bills only the pixel rows it uses - **Changed the OpenAI default shape from `6x6u-sent` to `8on16-bw`.** A production-regime mono eval (gpt-5.5, the full 800k-char SQuAD flow in one request, n=50) scored the old dense default f1 .602 vs .851 for `8on16-bw` rendered by the production pipeline, at near-equal total cost (the dense cells burned the frame savings on reasoning tokens); chunked exp14 had already scored `8on16-bw` .906. `SHAPES.openaiDense` is renamed to `SHAPES.openai` - **Changed the Google default shape from `8x8r-sent` to `doc-8on16-sent-dim`.** Production-rendered mono eval on gemini-3.5-flash (400k chars, one request, n=25): f1 .900 vs .853 for the repeated grid at lower cost, agreeing with the chunked round-2 winner - **Changed the Anthropic default shape from `8x8r-bw` to `6x12-dim`.** Production mono eval on claude-fable (400k chars, one request, n=25): f1 .840 vs .877 for the repeated grid — within noise — at 37% lower cost (12 frames instead of 21 per 400k chars), with clean completions in every probe; opus reads the same trade (.800 vs .833 at 42% lower cost) diff --git a/packages/snapcompact/src/snapcompact.ts b/packages/snapcompact/src/snapcompact.ts index de1fdc17e..2a5e7f9df 100644 --- a/packages/snapcompact/src/snapcompact.ts +++ b/packages/snapcompact/src/snapcompact.ts @@ -2,9 +2,10 @@ * Snapcompact compaction: archive conversation history as dense bitmap images. * * Instead of asking an LLM to summarize discarded history, the serialized - * conversation is rendered into square PNG frames of pixel-font text that - * vision models read back directly, like an archivist at a snapcompact frame - * reader. + * conversation is rendered into PNG frames of pixel-font text that vision + * models read back directly, like an archivist at a snapcompact frame + * reader. Frames are `frameSize` wide; their height hugs the text rows + * actually printed, so a partially filled frame never bills blank rows. * * The frame shape is provider-aware, following the snapcompact SQuAD evals * (`packages/snapcompact`, 200k-token monolithic runs): diff --git a/packages/snapcompact/test/snapcompact.test.ts b/packages/snapcompact/test/snapcompact.test.ts index 26f3b80fd..804361ce4 100644 --- a/packages/snapcompact/test/snapcompact.test.ts +++ b/packages/snapcompact/test/snapcompact.test.ts @@ -274,7 +274,9 @@ describe("render", () => { const decoded = decodePng(Buffer.from(frame.data, "base64")); expect(decoded.width).toBe(TEST_FRAME_SIZE); - expect(decoded.height).toBe(TEST_FRAME_SIZE); + // 40 chars on a 64-col grid: one 8px text row; height hugs it instead + // of padding the frame to a 320px square. + expect(decoded.height).toBe(8); expect(decoded.colorType).toBe(3); // indexed color // Two sentences → glyphs printed in ink 1 then ink 2; background stays 0.