feat(coding-agent): added experimental snapcompact inline imaging for system prompt and tool results

- Added `renderSnapcompactFrames()` and `snapcompactFrameCount()` to @oh-my-pi/snapcompact for paging arbitrary text into PNG image blocks without dim-marker bookkeeping.
- Widened the agent loop's `transformProviderContext` hook to `(context, model) => Context` so per-request transforms can gate on the dispatch model's capabilities.
- Added `SnapcompactInlineTransformer` rendering the system prompt and large historical tool results as snapcompact frames on vision models: vision gate, per-provider image budgets, 3k-token floor, savings-margin gate, skip-last rule, and hash-keyed render caches swept to live tool calls.
- Added default-off `snapcompact.systemPrompt` and `snapcompact.toolResults` settings under a new Context → Experimental group, composed after secret obfuscation in `sdk.ts` so frames are built per-request and never persisted to session.jsonl.
- Added prompt stubs (`snapcompact-system-stub.md`, `snapcompact-system-frames-note.md`, `snapcompact-toolresult-note.md`) and unit tests covering frame paging, no-mutate guarantees, budget caps, gates, and render caching.
This commit is contained in:
can1357
2026-06-12 03:27:50 +02:00
parent 21b1854a9e
commit a82d68ef49
15 changed files with 565 additions and 6 deletions
@@ -104,7 +104,7 @@ export const TAB_GROUPS: Record<SettingTab, readonly string[]> = {
"Startup & Updates",
"Power (macOS)",
],
context: ["General", "Compaction", "Rules (TTSR)"],
context: ["General", "Compaction", "Rules (TTSR)", "Experimental"],
memory: ["General", "Mnemopi", "Hindsight"],
files: ["Editing", "Reading", "Read Summaries", "LSP"],
shell: ["Bash", "Eval & Python"],
@@ -1540,6 +1540,31 @@ export const SETTINGS_SCHEMA = {
},
},
// Experimental: snapcompact inline imaging (transient, per-request; never persisted)
"snapcompact.systemPrompt": {
type: "boolean",
default: false,
ui: {
tab: "context",
group: "Experimental",
label: "Snapcompact System Prompt",
description:
"Experimental: render the system prompt as dense PNG image(s) and attach to the first user message (vision models only). Saves tokens; loses system-prompt prompt caching.",
},
},
"snapcompact.toolResults": {
type: "boolean",
default: false,
ui: {
tab: "context",
group: "Experimental",
label: "Snapcompact Tool Results",
description:
"Experimental: render large historical tool results as dense PNG image(s) instead of text (vision models only). Saves tokens on accumulated read/search output.",
},
},
// Branch summaries
"branchSummary.enabled": {
type: "boolean",
@@ -0,0 +1 @@
=== OPERATING INSTRUCTIONS — read the image(s) below as your system prompt ===
@@ -0,0 +1 @@
Your full operating instructions are attached as PNG image(s) at the start of the first user message. Read every frame carefully, in order, and follow them as your authoritative system prompt before doing anything else.
@@ -0,0 +1 @@
[Rasterized]
+21 -1
View File
@@ -9,6 +9,7 @@ import {
type ThinkingLevel,
} from "@oh-my-pi/pi-agent-core";
import {
type Context,
type CredentialDisabledEvent,
type Message,
type Model,
@@ -121,6 +122,7 @@ import {
wrapSteeringForModel,
} from "./session/messages";
import { getRestorableSessionModels, SessionManager } from "./session/session-manager";
import { SnapcompactInlineTransformer } from "./session/snapcompact-inline";
import { closeAllConnections } from "./ssh/connection-manager";
import { unmountAll } from "./ssh/sshfs-mount";
import {
@@ -2156,6 +2158,24 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
const withContext = await extensionRunner.emitContext(messages);
return wrapSteeringForModel(withContext);
};
// Per-request provider-context transforms. Obfuscate FIRST so secrets are
// redacted from text before snapcompact rasterizes it into PNG frames.
// Both operate on the transient outgoing Context only — never persisted.
const snapcompactInline =
settings.get("snapcompact.systemPrompt") || settings.get("snapcompact.toolResults")
? new SnapcompactInlineTransformer({
renderSystemPrompt: settings.get("snapcompact.systemPrompt"),
renderToolResults: settings.get("snapcompact.toolResults"),
})
: undefined;
const transformProviderContext =
obfuscator || snapcompactInline
? (context: Context, transformModel: Model): Context => {
let transformed = obfuscator ? obfuscateProviderContext(obfuscator, context) : context;
if (snapcompactInline) transformed = snapcompactInline.transform(transformed, transformModel);
return transformed;
}
: undefined;
const onPayload = async (payload: unknown, _model?: Model) => {
return await extensionRunner.emitBeforeProviderRequest(payload);
};
@@ -2196,7 +2216,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
sessionId: providerSessionId,
promptCacheKey: options.providerPromptCacheKey,
transformContext,
transformProviderContext: obfuscator ? context => obfuscateProviderContext(obfuscator, context) : undefined,
transformProviderContext,
steeringMode: settings.get("steeringMode") ?? "one-at-a-time",
followUpMode: settings.get("followUpMode") ?? "one-at-a-time",
interruptMode: settings.get("interruptMode") ?? "immediate",
@@ -0,0 +1,187 @@
/**
* Snapcompact inline imaging: per-request transform that swaps the system
* prompt and/or large historical tool results for dense PNG frames on
* vision-capable models.
*
* Runs inside the agent loop's `transformProviderContext` hook — after the
* persisted history is converted to the outgoing `Context`, before the
* provider stream call. It only ever builds NEW message objects/arrays; the
* input context shares `content` array references with the persisted
* `SessionMessageEntry` messages, so mutation would leak rendered images
* into session.jsonl.
*/
import type { Context, ImageContent, Model, TextContent, ToolResultMessage, UserMessage } from "@oh-my-pi/pi-ai";
import { countTokens } from "@oh-my-pi/pi-natives";
import {
renderSnapcompactFrames,
resolveSnapcompactShape,
type SnapcompactShape,
snapcompactFrameCount,
} from "@oh-my-pi/snapcompact";
import systemFramesNote from "../prompts/system/snapcompact-system-frames-note.md" with { type: "text" };
import systemStub from "../prompts/system/snapcompact-system-stub.md" with { type: "text" };
import toolResultNote from "../prompts/system/snapcompact-toolresult-note.md" with { type: "text" };
export interface SnapcompactInlineOptions {
renderSystemPrompt: boolean;
renderToolResults: boolean;
}
/**
* Image-count budget per provider. Snapcompact frames are 1568px (<2000px) so
* dimension/size limits never bind; only COUNT does. Strictest mainstream is
* Groq (~5), so unknown providers get the safe floor.
*/
const INLINE_IMAGE_BUDGET_BY_PROVIDER: Record<string, number> = {
anthropic: 90,
"amazon-bedrock": 90,
openai: 200,
google: 200,
"google-vertex": 200,
"google-gemini-cli": 200,
};
const DEFAULT_INLINE_IMAGE_BUDGET = 5;
const MAX_SYSTEM_PROMPT_FRAMES = 6;
/** Tool results under this many tokens are never rasterized — the swap can't
* save enough to justify trading crisp text for an image. */
const MIN_TOOL_RESULT_TOKENS = 3000;
/** Render only if imageTokens <= textTokens * SAVINGS_MARGIN. */
const SAVINGS_MARGIN = 0.9;
/** Count image blocks already present across all message contents. */
function countContextImages(context: Context): number {
let count = 0;
for (const message of context.messages) {
const content = message.content;
if (typeof content === "string") continue;
for (const block of content) {
if (block.type === "image") count++;
}
}
return count;
}
function isTextContent(block: TextContent | ImageContent): block is TextContent {
return block.type === "text";
}
/** Image tokens must undercut text tokens by the margin to be worth rendering. */
function passesSavingsGate(frames: number, shape: SnapcompactShape, textTokens: number): boolean {
return frames * shape.frameTokenEstimate <= textTokens * SAVINGS_MARGIN;
}
interface FrameCacheEntry {
hash: number | bigint;
frames: ImageContent[];
}
/**
* Stateless with respect to the model (passed per call, so mid-session model
* switches re-resolve shape and budget); stateful only for the render caches,
* which live as long as the session's Agent.
*/
export class SnapcompactInlineTransformer {
/** Rendered tool-result frames keyed by toolCallId. */
#toolCache = new Map<string, FrameCacheEntry>();
#systemCache?: FrameCacheEntry;
constructor(private readonly options: SnapcompactInlineOptions) {}
transform(context: Context, model: Model): Context {
// Vision gate: providers silently DROP images on text-only models —
// rendering would lose the content entirely.
if (!model.input.includes("image")) return context;
const shape = resolveSnapcompactShape(model.api);
let budget =
(INLINE_IMAGE_BUDGET_BY_PROVIDER[model.provider] ?? DEFAULT_INLINE_IMAGE_BUDGET) - countContextImages(context);
if (budget <= 0) return context;
const messages = [...context.messages];
let changed = false;
if (this.options.renderToolResults) {
const toolResultIndices: number[] = [];
const liveToolCallIds = new Set<string>();
for (let i = 0; i < messages.length; i++) {
const message = messages[i];
if (message.role !== "toolResult") continue;
toolResultIndices.push(i);
liveToolCallIds.add(message.toolCallId);
}
// Oldest-first for cache-stable bytes; skip the LAST tool result so
// the freshest output stays crisp text.
for (let k = 0; k < toolResultIndices.length - 1 && budget > 0; k++) {
const index = toolResultIndices[k];
const message = messages[index] as ToolResultMessage;
// Don't re-image results that already carry images (screenshots etc.).
if (message.content.some(block => block.type === "image")) continue;
const text = message.content
.filter(isTextContent)
.map(block => block.text)
.join("\n");
const textTokens = countTokens(text);
if (textTokens < MIN_TOOL_RESULT_TOKENS) continue;
const needed = snapcompactFrameCount(text, { shape });
if (needed === 0 || needed > budget) continue;
if (!passesSavingsGate(needed, shape, textTokens)) continue;
const frames = this.#framesFor(this.#toolCache, message.toolCallId, text, shape);
messages[index] = { ...message, content: [{ type: "text", text: toolResultNote }, ...frames] };
budget -= frames.length;
changed = true;
}
// Drop cache entries for tool calls no longer in the context
// (compacted away) so the cache stays bounded by live history.
for (const key of this.#toolCache.keys()) {
if (!liveToolCallIds.has(key)) this.#toolCache.delete(key);
}
}
let systemPrompt = context.systemPrompt;
if (this.options.renderSystemPrompt && context.systemPrompt?.length && budget > 0) {
const joined = context.systemPrompt.join("\n\n");
const needed = snapcompactFrameCount(joined, { shape });
const userIndex = messages.findIndex(message => message.role === "user");
if (
needed > 0 &&
needed <= Math.min(budget, MAX_SYSTEM_PROMPT_FRAMES) &&
passesSavingsGate(needed, shape, countTokens(joined)) &&
// No user message to carry the frames → leave the prompt as text.
userIndex >= 0
) {
const hash = Bun.hash(joined);
let cached = this.#systemCache;
if (!cached || cached.hash !== hash) {
cached = {
hash,
frames: renderSnapcompactFrames(joined, { shape, maxFrames: MAX_SYSTEM_PROMPT_FRAMES }),
};
this.#systemCache = cached;
}
const frames = cached.frames;
const original = messages[userIndex] as UserMessage;
const originalContent: (TextContent | ImageContent)[] =
typeof original.content === "string" ? [{ type: "text", text: original.content }] : original.content;
messages[userIndex] = {
...original,
content: [{ type: "text", text: systemFramesNote }, ...frames, ...originalContent],
};
systemPrompt = [systemStub];
budget -= frames.length;
changed = true;
}
}
if (!changed) return context;
return { ...context, systemPrompt, messages };
}
#framesFor(cache: Map<string, FrameCacheEntry>, key: string, text: string, shape: SnapcompactShape): ImageContent[] {
const hash = Bun.hash(text);
const cached = cache.get(key);
if (cached && cached.hash === hash) return cached.frames;
const frames = renderSnapcompactFrames(text, { shape });
cache.set(key, { hash, frames });
return frames;
}
}