diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 34d4ff807..a5610ea9c 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -9,6 +9,7 @@ ### Changed - Compaction, handoff, short-summary, and branch-summarization helpers now accept an `ApiKey` (static string or resolver) instead of a pre-resolved string, so a 401 mid-compaction force-refreshes and rotates the credential through the central auth-retry policy before any model-level fallback. The remote OpenAI compaction request is wrapped in `withAuth` and its HTTP failures now carry `.status`, so the retry classifier actually fires on remote-compaction 401s. +- `transformProviderContext` now receives the dispatch model as a second argument (`(context, model) => Context`), so per-request transforms can gate on model capabilities (vision input, provider, API family). Existing single-argument implementations keep working unchanged. - Remote-compaction and summarization failures now throw pi-ai's typed `ProviderHttpError` instead of mutating plain `Error`s with a `.status` property; the generic `requestRemoteCompaction` error now carries `.status` (and response headers) too. ## [15.11.2] - 2026-06-11 diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index e044ba6a5..5ecf0aeb3 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -880,7 +880,7 @@ async function streamAssistantResponse( }; } if (config.transformProviderContext) { - llmContext = config.transformProviderContext(llmContext); + llmContext = config.transformProviderContext(llmContext, config.model); } const streamFunction = streamFn || streamSimple; diff --git a/packages/agent/src/agent.ts b/packages/agent/src/agent.ts index 8c50fa620..35868b464 100644 --- a/packages/agent/src/agent.ts +++ b/packages/agent/src/agent.ts @@ -98,7 +98,7 @@ export interface AgentOptions { * Optional transform applied after provider context assembly and before * telemetry capture/provider send. */ - transformProviderContext?: (context: Context) => Context; + transformProviderContext?: (context: Context, model: Model) => Context; /** * Steering mode: "all" = send all steering messages at once, "one-at-a-time" = one per turn @@ -285,7 +285,7 @@ export class Agent { #abortController?: AbortController; #convertToLlm: (messages: AgentMessage[]) => Message[] | Promise; #transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => Promise; - #transformProviderContext?: (context: Context) => Context; + #transformProviderContext?: (context: Context, model: Model) => Context; #steeringQueue: AgentMessage[] = []; #followUpQueue: AgentMessage[] = []; #steeringMode: "all" | "one-at-a-time"; diff --git a/packages/agent/src/types.ts b/packages/agent/src/types.ts index d37929820..d9b3afcc6 100644 --- a/packages/agent/src/types.ts +++ b/packages/agent/src/types.ts @@ -113,7 +113,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions { * normalization, and append-only context handling, but before telemetry capture * and provider send. */ - transformProviderContext?: (context: Context) => Context; + transformProviderContext?: (context: Context, model: Model) => Context; /** * Resolves an API key dynamically for each LLM call. diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index a704d4834..29e1b54a0 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -9,6 +9,7 @@ ### Added - `ModelRegistry.resolver` now accepts a model directly — `resolver(model, sessionId)` — deriving `provider`, `baseUrl`, and `modelId` from it; all model-scoped call sites migrated from the verbose `resolver(model.provider, { sessionId, baseUrl, modelId })` form. +- Added experimental `snapcompact.systemPrompt` and `snapcompact.toolResults` settings (off by default, `/settings` → Context → Experimental) that render the system prompt and large historical tool results as dense snapcompact PNG frames on vision-capable models to cut token cost. Frames are built per-request in the provider-context transform, cached across turns, capped by a per-provider image budget, and gated on a token-savings estimate — they never reach `session.jsonl`. ### Changed diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 724b8a9f3..87de785eb 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -104,7 +104,7 @@ export const TAB_GROUPS: Record = { "Startup & Updates", "Power (macOS)", ], - context: ["General", "Compaction", "Rules (TTSR)"], + context: ["General", "Compaction", "Rules (TTSR)", "Experimental"], memory: ["General", "Mnemopi", "Hindsight"], files: ["Editing", "Reading", "Read Summaries", "LSP"], shell: ["Bash", "Eval & Python"], @@ -1540,6 +1540,31 @@ export const SETTINGS_SCHEMA = { }, }, + // Experimental: snapcompact inline imaging (transient, per-request; never persisted) + "snapcompact.systemPrompt": { + type: "boolean", + default: false, + ui: { + tab: "context", + group: "Experimental", + label: "Snapcompact System Prompt", + description: + "Experimental: render the system prompt as dense PNG image(s) and attach to the first user message (vision models only). Saves tokens; loses system-prompt prompt caching.", + }, + }, + + "snapcompact.toolResults": { + type: "boolean", + default: false, + ui: { + tab: "context", + group: "Experimental", + label: "Snapcompact Tool Results", + description: + "Experimental: render large historical tool results as dense PNG image(s) instead of text (vision models only). Saves tokens on accumulated read/search output.", + }, + }, + // Branch summaries "branchSummary.enabled": { type: "boolean", diff --git a/packages/coding-agent/src/prompts/system/snapcompact-system-frames-note.md b/packages/coding-agent/src/prompts/system/snapcompact-system-frames-note.md new file mode 100644 index 000000000..2283cf6f4 --- /dev/null +++ b/packages/coding-agent/src/prompts/system/snapcompact-system-frames-note.md @@ -0,0 +1 @@ +=== OPERATING INSTRUCTIONS — read the image(s) below as your system prompt === diff --git a/packages/coding-agent/src/prompts/system/snapcompact-system-stub.md b/packages/coding-agent/src/prompts/system/snapcompact-system-stub.md new file mode 100644 index 000000000..48d6e2c12 --- /dev/null +++ b/packages/coding-agent/src/prompts/system/snapcompact-system-stub.md @@ -0,0 +1 @@ +Your full operating instructions are attached as PNG image(s) at the start of the first user message. Read every frame carefully, in order, and follow them as your authoritative system prompt before doing anything else. diff --git a/packages/coding-agent/src/prompts/system/snapcompact-toolresult-note.md b/packages/coding-agent/src/prompts/system/snapcompact-toolresult-note.md new file mode 100644 index 000000000..1908bfba4 --- /dev/null +++ b/packages/coding-agent/src/prompts/system/snapcompact-toolresult-note.md @@ -0,0 +1 @@ +[Rasterized] diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 05e6664bd..ba2479ea4 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -9,6 +9,7 @@ import { type ThinkingLevel, } from "@oh-my-pi/pi-agent-core"; import { + type Context, type CredentialDisabledEvent, type Message, type Model, @@ -121,6 +122,7 @@ import { wrapSteeringForModel, } from "./session/messages"; import { getRestorableSessionModels, SessionManager } from "./session/session-manager"; +import { SnapcompactInlineTransformer } from "./session/snapcompact-inline"; import { closeAllConnections } from "./ssh/connection-manager"; import { unmountAll } from "./ssh/sshfs-mount"; import { @@ -2156,6 +2158,24 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} const withContext = await extensionRunner.emitContext(messages); return wrapSteeringForModel(withContext); }; + // Per-request provider-context transforms. Obfuscate FIRST so secrets are + // redacted from text before snapcompact rasterizes it into PNG frames. + // Both operate on the transient outgoing Context only — never persisted. + const snapcompactInline = + settings.get("snapcompact.systemPrompt") || settings.get("snapcompact.toolResults") + ? new SnapcompactInlineTransformer({ + renderSystemPrompt: settings.get("snapcompact.systemPrompt"), + renderToolResults: settings.get("snapcompact.toolResults"), + }) + : undefined; + const transformProviderContext = + obfuscator || snapcompactInline + ? (context: Context, transformModel: Model): Context => { + let transformed = obfuscator ? obfuscateProviderContext(obfuscator, context) : context; + if (snapcompactInline) transformed = snapcompactInline.transform(transformed, transformModel); + return transformed; + } + : undefined; const onPayload = async (payload: unknown, _model?: Model) => { return await extensionRunner.emitBeforeProviderRequest(payload); }; @@ -2196,7 +2216,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} sessionId: providerSessionId, promptCacheKey: options.providerPromptCacheKey, transformContext, - transformProviderContext: obfuscator ? context => obfuscateProviderContext(obfuscator, context) : undefined, + transformProviderContext, steeringMode: settings.get("steeringMode") ?? "one-at-a-time", followUpMode: settings.get("followUpMode") ?? "one-at-a-time", interruptMode: settings.get("interruptMode") ?? "immediate", diff --git a/packages/coding-agent/src/session/snapcompact-inline.ts b/packages/coding-agent/src/session/snapcompact-inline.ts new file mode 100644 index 000000000..423272e17 --- /dev/null +++ b/packages/coding-agent/src/session/snapcompact-inline.ts @@ -0,0 +1,187 @@ +/** + * Snapcompact inline imaging: per-request transform that swaps the system + * prompt and/or large historical tool results for dense PNG frames on + * vision-capable models. + * + * Runs inside the agent loop's `transformProviderContext` hook — after the + * persisted history is converted to the outgoing `Context`, before the + * provider stream call. It only ever builds NEW message objects/arrays; the + * input context shares `content` array references with the persisted + * `SessionMessageEntry` messages, so mutation would leak rendered images + * into session.jsonl. + */ +import type { Context, ImageContent, Model, TextContent, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai"; +import { countTokens } from "@oh-my-pi/pi-natives"; +import { + renderSnapcompactFrames, + resolveSnapcompactShape, + type SnapcompactShape, + snapcompactFrameCount, +} from "@oh-my-pi/snapcompact"; +import systemFramesNote from "../prompts/system/snapcompact-system-frames-note.md" with { type: "text" }; +import systemStub from "../prompts/system/snapcompact-system-stub.md" with { type: "text" }; +import toolResultNote from "../prompts/system/snapcompact-toolresult-note.md" with { type: "text" }; + +export interface SnapcompactInlineOptions { + renderSystemPrompt: boolean; + renderToolResults: boolean; +} + +/** + * Image-count budget per provider. Snapcompact frames are 1568px (<2000px) so + * dimension/size limits never bind; only COUNT does. Strictest mainstream is + * Groq (~5), so unknown providers get the safe floor. + */ +const INLINE_IMAGE_BUDGET_BY_PROVIDER: Record = { + anthropic: 90, + "amazon-bedrock": 90, + openai: 200, + google: 200, + "google-vertex": 200, + "google-gemini-cli": 200, +}; +const DEFAULT_INLINE_IMAGE_BUDGET = 5; +const MAX_SYSTEM_PROMPT_FRAMES = 6; +/** Tool results under this many tokens are never rasterized — the swap can't + * save enough to justify trading crisp text for an image. */ +const MIN_TOOL_RESULT_TOKENS = 3000; +/** Render only if imageTokens <= textTokens * SAVINGS_MARGIN. */ +const SAVINGS_MARGIN = 0.9; + +/** Count image blocks already present across all message contents. */ +function countContextImages(context: Context): number { + let count = 0; + for (const message of context.messages) { + const content = message.content; + if (typeof content === "string") continue; + for (const block of content) { + if (block.type === "image") count++; + } + } + return count; +} + +function isTextContent(block: TextContent | ImageContent): block is TextContent { + return block.type === "text"; +} + +/** Image tokens must undercut text tokens by the margin to be worth rendering. */ +function passesSavingsGate(frames: number, shape: SnapcompactShape, textTokens: number): boolean { + return frames * shape.frameTokenEstimate <= textTokens * SAVINGS_MARGIN; +} + +interface FrameCacheEntry { + hash: number | bigint; + frames: ImageContent[]; +} + +/** + * Stateless with respect to the model (passed per call, so mid-session model + * switches re-resolve shape and budget); stateful only for the render caches, + * which live as long as the session's Agent. + */ +export class SnapcompactInlineTransformer { + /** Rendered tool-result frames keyed by toolCallId. */ + #toolCache = new Map(); + #systemCache?: FrameCacheEntry; + + constructor(private readonly options: SnapcompactInlineOptions) {} + + transform(context: Context, model: Model): Context { + // Vision gate: providers silently DROP images on text-only models — + // rendering would lose the content entirely. + if (!model.input.includes("image")) return context; + + const shape = resolveSnapcompactShape(model.api); + let budget = + (INLINE_IMAGE_BUDGET_BY_PROVIDER[model.provider] ?? DEFAULT_INLINE_IMAGE_BUDGET) - countContextImages(context); + if (budget <= 0) return context; + + const messages = [...context.messages]; + let changed = false; + + if (this.options.renderToolResults) { + const toolResultIndices: number[] = []; + const liveToolCallIds = new Set(); + for (let i = 0; i < messages.length; i++) { + const message = messages[i]; + if (message.role !== "toolResult") continue; + toolResultIndices.push(i); + liveToolCallIds.add(message.toolCallId); + } + // Oldest-first for cache-stable bytes; skip the LAST tool result so + // the freshest output stays crisp text. + for (let k = 0; k < toolResultIndices.length - 1 && budget > 0; k++) { + const index = toolResultIndices[k]; + const message = messages[index] as ToolResultMessage; + // Don't re-image results that already carry images (screenshots etc.). + if (message.content.some(block => block.type === "image")) continue; + const text = message.content + .filter(isTextContent) + .map(block => block.text) + .join("\n"); + const textTokens = countTokens(text); + if (textTokens < MIN_TOOL_RESULT_TOKENS) continue; + const needed = snapcompactFrameCount(text, { shape }); + if (needed === 0 || needed > budget) continue; + if (!passesSavingsGate(needed, shape, textTokens)) continue; + const frames = this.#framesFor(this.#toolCache, message.toolCallId, text, shape); + messages[index] = { ...message, content: [{ type: "text", text: toolResultNote }, ...frames] }; + budget -= frames.length; + changed = true; + } + // Drop cache entries for tool calls no longer in the context + // (compacted away) so the cache stays bounded by live history. + for (const key of this.#toolCache.keys()) { + if (!liveToolCallIds.has(key)) this.#toolCache.delete(key); + } + } + + let systemPrompt = context.systemPrompt; + if (this.options.renderSystemPrompt && context.systemPrompt?.length && budget > 0) { + const joined = context.systemPrompt.join("\n\n"); + const needed = snapcompactFrameCount(joined, { shape }); + const userIndex = messages.findIndex(message => message.role === "user"); + if ( + needed > 0 && + needed <= Math.min(budget, MAX_SYSTEM_PROMPT_FRAMES) && + passesSavingsGate(needed, shape, countTokens(joined)) && + // No user message to carry the frames → leave the prompt as text. + userIndex >= 0 + ) { + const hash = Bun.hash(joined); + let cached = this.#systemCache; + if (!cached || cached.hash !== hash) { + cached = { + hash, + frames: renderSnapcompactFrames(joined, { shape, maxFrames: MAX_SYSTEM_PROMPT_FRAMES }), + }; + this.#systemCache = cached; + } + const frames = cached.frames; + const original = messages[userIndex] as UserMessage; + const originalContent: (TextContent | ImageContent)[] = + typeof original.content === "string" ? [{ type: "text", text: original.content }] : original.content; + messages[userIndex] = { + ...original, + content: [{ type: "text", text: systemFramesNote }, ...frames, ...originalContent], + }; + systemPrompt = [systemStub]; + budget -= frames.length; + changed = true; + } + } + + if (!changed) return context; + return { ...context, systemPrompt, messages }; + } + + #framesFor(cache: Map, key: string, text: string, shape: SnapcompactShape): ImageContent[] { + const hash = Bun.hash(text); + const cached = cache.get(key); + if (cached && cached.hash === hash) return cached.frames; + const frames = renderSnapcompactFrames(text, { shape }); + cache.set(key, { hash, frames }); + return frames; + } +} diff --git a/packages/coding-agent/test/snapcompact-inline.test.ts b/packages/coding-agent/test/snapcompact-inline.test.ts new file mode 100644 index 000000000..2d26a1313 --- /dev/null +++ b/packages/coding-agent/test/snapcompact-inline.test.ts @@ -0,0 +1,227 @@ +import { describe, expect, it, spyOn } from "bun:test"; +import type { Context, ImageContent, Message, TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; +import { SnapcompactInlineTransformer } from "@oh-my-pi/pi-coding-agent/session/snapcompact-inline"; +import * as snapcompact from "@oh-my-pi/snapcompact"; + +/** + * Token-dense deterministic word salad. 3000 words ≈ 20.6k normalized chars + * → 2 anthropic-shape frames (capacity 19208) whose ~6600 estimated image + * tokens clear the savings gate against ~8900 text tokens. + */ +function denseText(words: number): string { + return Array.from({ length: words }, (_, i) => `w${(i * 7919) % 100000}`).join(" "); +} + +const LARGE = denseText(3000); +const SMALL = "12 lines OK"; + +function toolResult(id: string, text: string): ToolResultMessage { + return { + role: "toolResult", + toolCallId: id, + toolName: "read", + content: [{ type: "text", text }], + isError: false, + timestamp: 0, + }; +} + +function userMessage(text: string): Message { + return { role: "user", content: text, timestamp: 0 }; +} + +function makeModel( + overrides: { + provider?: string; + input?: ("text" | "image")[]; + api?: "anthropic-messages" | "google-generative-ai"; + } = {}, +) { + return buildModel({ + id: "test-model", + name: "Test Model", + api: overrides.api ?? "anthropic-messages", + provider: overrides.provider ?? "anthropic", + baseUrl: "https://example.invalid", + reasoning: false, + input: overrides.input ?? ["text", "image"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }); +} + +function makeContext(): Context { + return { + systemPrompt: ["You are a coding agent.", "Follow the rules."], + messages: [ + userMessage("first user prompt"), + toolResult("call_1", LARGE), + toolResult("call_2", SMALL), + toolResult("call_3", LARGE), + ], + }; +} + +function imageCount(context: Context): number { + let count = 0; + for (const message of context.messages) { + if (typeof message.content === "string") continue; + for (const block of message.content) if (block.type === "image") count++; + } + return count; +} + +describe("SnapcompactInlineTransformer", () => { + it("is a no-op for text-only models", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: true, renderToolResults: true }); + const context = makeContext(); + expect(transformer.transform(context, makeModel({ input: ["text"] }))).toBe(context); + }); + + it("images large historical tool results, keeping small and most-recent ones as text", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: false, renderToolResults: true }); + const context = makeContext(); + const result = transformer.transform(context, makeModel()); + + // Large historical result → leading text note + image frames. + const imaged = result.messages[1] as ToolResultMessage; + expect(imaged.content[0].type).toBe("text"); + expect(imaged.content.length).toBeGreaterThan(1); + expect(imaged.content.slice(1).every(block => block.type === "image")).toBe(true); + for (const block of imaged.content.slice(1) as ImageContent[]) { + expect(block.mimeType).toBe("image/png"); + expect(block.data.length).toBeGreaterThan(0); + } + + // Small result fails the savings gate; the most-recent stays crisp text. + expect(result.messages[2]).toBe(context.messages[2]); + expect(result.messages[3]).toBe(context.messages[3]); + expect((result.messages[3] as ToolResultMessage).content[0]).toEqual({ type: "text", text: LARGE }); + + // System prompt untouched when only tool results are enabled. + expect(result.systemPrompt).toBe(context.systemPrompt); + }); + + it("never mutates the input context (persisted history shares these references)", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: true, renderToolResults: true }); + const context = makeContext(); + const originalMessages = context.messages; + const originalSystemPrompt = context.systemPrompt; + const original = context.messages[1] as ToolResultMessage; + const originalContent = original.content; + + const result = transformer.transform(context, makeModel()); + expect(result).not.toBe(context); + + expect(context.messages).toBe(originalMessages); + expect(context.systemPrompt).toBe(originalSystemPrompt); + expect(context.systemPrompt).toEqual(["You are a coding agent.", "Follow the rules."]); + expect(original.content).toBe(originalContent); + expect(originalContent).toEqual([{ type: "text", text: LARGE }]); + expect((context.messages[0] as { content: string }).content).toBe("first user prompt"); + }); + + it("leaves tool results that already carry images untouched", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: false, renderToolResults: true }); + const withImage: ToolResultMessage = { + ...toolResult("call_img", LARGE), + content: [ + { type: "text", text: LARGE }, + { type: "image", data: "aGk=", mimeType: "image/png" }, + ], + }; + const context: Context = { + messages: [userMessage("hi"), withImage, toolResult("call_tail", LARGE)], + }; + const result = transformer.transform(context, makeModel()); + expect(result.messages[1]).toBe(withImage); + }); + + it("replaces a large system prompt with a stub and rides frames on the first user message", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: true, renderToolResults: false }); + const longPrompt = denseText(3000); + const context: Context = { + systemPrompt: [longPrompt], + messages: [userMessage("do the thing"), toolResult("call_1", SMALL)], + }; + const result = transformer.transform(context, makeModel()); + + expect(result.systemPrompt).toHaveLength(1); + expect(result.systemPrompt![0]).not.toBe(longPrompt); + expect(result.systemPrompt![0].length).toBeLessThan(500); + + const carrier = result.messages[0] as { content: (TextContent | ImageContent)[] }; + expect(carrier.content[0].type).toBe("text"); + const images = carrier.content.filter(block => block.type === "image"); + expect(images.length).toBeGreaterThan(0); + // Original user text survives at the tail. + expect(carrier.content[carrier.content.length - 1]).toEqual({ type: "text", text: "do the thing" }); + }); + + it("keeps a small system prompt as text and skips when no user message exists", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: true, renderToolResults: false }); + const small: Context = { systemPrompt: ["Be terse."], messages: [userMessage("hi")] }; + expect(transformer.transform(small, makeModel())).toBe(small); + + const noUser: Context = { systemPrompt: [denseText(3000)], messages: [toolResult("call_1", SMALL)] }; + expect(transformer.transform(noUser, makeModel())).toBe(noUser); + }); + + it("never rasterizes tool results under the 3k-token floor, even when frames are cheaper", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: false, renderToolResults: true }); + // ~1.5k tokens: the google shape estimates 1 frame ≈ 1100 tokens, so the + // savings gate alone would rasterize this — the floor must keep it text. + const midsize = denseText(500); + const context: Context = { + messages: [userMessage("go"), toolResult("call_1", midsize), toolResult("call_2", LARGE)], + }; + const result = transformer.transform(context, makeModel({ api: "google-generative-ai", provider: "google" })); + expect(result).toBe(context); + }); + + it("respects the per-provider image budget for unknown providers", () => { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: false, renderToolResults: true }); + const context: Context = { + messages: [ + userMessage("go"), + toolResult("call_1", LARGE), + toolResult("call_2", LARGE), + toolResult("call_3", LARGE), + toolResult("call_4", LARGE), + ], + }; + // Unknown provider → default budget 5. Each LARGE needs 2 frames: + // call_1 (2) + call_2 (2) fit, call_3 needs 2 > 1 remaining → text. + const result = transformer.transform(context, makeModel({ provider: "groq" })); + expect(imageCount(result)).toBeLessThanOrEqual(5); + expect(result.messages[3]).toBe(context.messages[3]); + expect(result.messages[4]).toBe(context.messages[4]); + }); + + it("caches renders across turns: identical input does not re-rasterize", () => { + const spy = spyOn(snapcompact, "renderSnapcompactFrames"); + try { + const transformer = new SnapcompactInlineTransformer({ renderSystemPrompt: true, renderToolResults: true }); + const context = makeContext(); + const model = makeModel(); + + const first = transformer.transform(context, model); + const callsAfterFirst = spy.mock.calls.length; + expect(callsAfterFirst).toBeGreaterThan(0); + + const second = transformer.transform(context, model); + expect(spy.mock.calls.length).toBe(callsAfterFirst); + + const firstFrames = (first.messages[1] as ToolResultMessage).content.slice(1); + const secondFrames = (second.messages[1] as ToolResultMessage).content.slice(1); + expect(secondFrames.length).toBe(firstFrames.length); + for (let i = 0; i < firstFrames.length; i++) { + expect(secondFrames[i]).toBe(firstFrames[i]); + } + } finally { + spy.mockRestore(); + } + }); +}); diff --git a/packages/snapcompact/CHANGELOG.md b/packages/snapcompact/CHANGELOG.md index ba7085212..708c44645 100644 --- a/packages/snapcompact/CHANGELOG.md +++ b/packages/snapcompact/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added `renderSnapcompactFrames()` for paging arbitrary text into snapcompact PNG frames as LLM image blocks, and `snapcompactFrameCount()` for predicting the frame count without rendering + ## [15.11.0] - 2026-06-10 ### Breaking Changes diff --git a/packages/snapcompact/src/snapcompact.ts b/packages/snapcompact/src/snapcompact.ts index cddf2923e..f5afc8959 100644 --- a/packages/snapcompact/src/snapcompact.ts +++ b/packages/snapcompact/src/snapcompact.ts @@ -579,6 +579,52 @@ export function renderSnapcompactFrame( return { data, cols, rows, chars }; } +/** Options for {@link renderSnapcompactFrames} and {@link snapcompactFrameCount}. */ +export interface RenderSnapcompactFramesOptions { + /** Explicit shape; wins over `model`. */ + shape?: SnapcompactShape; + /** Model whose `api` selects the eval-optimal shape. */ + model?: Pick; + /** Frame edge in px; defaults to the shape's `frameSize`. */ + frameSize?: number; + /** Hard cap on frames produced; omit for unbounded (caller decides usage). */ + maxFrames?: number; +} + +/** + * Render arbitrary text into snapcompact PNG frames as LLM image blocks + * (first page first). Synchronous: safe to call from per-request transforms. + * Empty/whitespace-only input yields no frames. + */ +export function renderSnapcompactFrames(text: string, options?: RenderSnapcompactFramesOptions): ImageContent[] { + const shape = options?.shape ?? resolveSnapcompactShape(options?.model?.api); + const frameSize = options?.frameSize ?? shape.frameSize; + const geometry = snapcompactGeometry(shape, frameSize); + const normalized = normalizeForSnapcompact(text); + const frames: ImageContent[] = []; + for (let offset = 0; offset < normalized.length; offset += geometry.capacity) { + if (options?.maxFrames !== undefined && frames.length >= options.maxFrames) break; + const rendered = renderSnapcompactFrame(normalized.slice(offset, offset + geometry.capacity), shape, frameSize); + frames.push({ + type: "image", + data: rendered.data, + mimeType: "image/png", + ...(shape.imageDetail ? { detail: shape.imageDetail } : {}), + }); + } + return frames; +} + +/** Frames needed to hold `text` at the given shape/size, without rendering. */ +export function snapcompactFrameCount( + text: string, + options?: Pick, +): number { + const shape = options?.shape ?? resolveSnapcompactShape(options?.model?.api); + const geometry = snapcompactGeometry(shape, options?.frameSize ?? shape.frameSize); + return Math.ceil(normalizeForSnapcompact(text).length / geometry.capacity); +} + // ============================================================================ // Archive helpers // ============================================================================ diff --git a/packages/snapcompact/test/snapcompact.test.ts b/packages/snapcompact/test/snapcompact.test.ts index bf7c9b067..be0f02972 100644 --- a/packages/snapcompact/test/snapcompact.test.ts +++ b/packages/snapcompact/test/snapcompact.test.ts @@ -6,6 +6,7 @@ import { isSnapcompactShape, normalizeForSnapcompact, renderSnapcompactFrame, + renderSnapcompactFrames, resolveSnapcompactShape, SNAPCOMPACT_DIM_OFF, SNAPCOMPACT_DIM_ON, @@ -16,6 +17,7 @@ import { type SnapcompactCompactionResult, serializeSnapcompactConversation, snapcompactCompact, + snapcompactFrameCount, snapcompactGeometry, snapcompactImages, } from "../src"; @@ -240,6 +242,49 @@ describe("renderSnapcompactFrame", () => { }); }); +describe("renderSnapcompactFrames", () => { + it("returns no frames for empty or whitespace-only input", () => { + expect(renderSnapcompactFrames("", { shape: SNAPCOMPACT_SHAPES.anthropic, frameSize: TEST_FRAME_SIZE })).toEqual( + [], + ); + expect( + renderSnapcompactFrames(" \n\t ", { shape: SNAPCOMPACT_SHAPES.anthropic, frameSize: TEST_FRAME_SIZE }), + ).toEqual([]); + expect(snapcompactFrameCount("", { shape: SNAPCOMPACT_SHAPES.anthropic, frameSize: TEST_FRAME_SIZE })).toBe(0); + }); + + it("pages text into image blocks matching the predicted frame count", () => { + const shape = SNAPCOMPACT_SHAPES.anthropic; + const { capacity } = snapcompactGeometry(shape, TEST_FRAME_SIZE); + + const short = renderSnapcompactFrames("hello world", { shape, frameSize: TEST_FRAME_SIZE }); + expect(short).toHaveLength(1); + expect(short[0].type).toBe("image"); + expect(short[0].mimeType).toBe("image/png"); + expect(short[0].data.length).toBeGreaterThan(0); + + const text = "x".repeat(capacity * 2 + 10); + const frames = renderSnapcompactFrames(text, { shape, frameSize: TEST_FRAME_SIZE }); + expect(frames).toHaveLength(3); + expect(snapcompactFrameCount(text, { shape, frameSize: TEST_FRAME_SIZE })).toBe(3); + }); + + it("honors maxFrames and propagates the shape's detail hint", () => { + const shape = SNAPCOMPACT_SHAPES.openaiDense; + const { capacity } = snapcompactGeometry(shape, TEST_FRAME_SIZE); + const frames = renderSnapcompactFrames("x".repeat(capacity * 3), { + shape, + frameSize: TEST_FRAME_SIZE, + maxFrames: 2, + }); + expect(frames).toHaveLength(2); + // openaiDense carries imageDetail: "original"; anthropic carries none. + expect(frames[0].detail).toBe("original"); + const bw = renderSnapcompactFrames("hi", { shape: SNAPCOMPACT_SHAPES.anthropic, frameSize: TEST_FRAME_SIZE }); + expect(bw[0].detail).toBeUndefined(); + }); +}); + describe("serializeSnapcompactConversation", () => { it("truncates oversized tool results keeping head and tail", () => { const text = `HEAD-${"x".repeat(5000)}-TAIL`;