feat(coding-agent): integrated snapcompact strategy and per-turn supersede pruning

Adds compaction.strategy: "snapcompact" to the schema and the AgentSession routing: when chosen, both manual /compact (without custom instructions) and auto compaction call snapcompactCompact() to archive history as PNG frames instead of an LLM summary. Falls back to context-full with a visible warning notice when the current model is text-only or when /compact gets custom instructions. CustomTool and shared-event payloads carry the new action through. \n\nAlso wires the per-turn supersede pass: #pruneSupersededReads() runs every turn before threshold gating (cache-aware: only fires when the post-candidate suffix is small or the prompt cache is cold), prunes older read results superseded by a newer read of the same file, rewrites the session, and accounts the saved tokens in the next compaction decision. Gated by compaction.supersedeReads (default on).\n\nsession/messages.ts now delegates the core role conversion to agent-core's convertMessageToLlm so snapcompact image blocks flow through the LLM-context conversion path.
This commit is contained in:
can1357
2026-06-10 17:52:49 +02:00
parent a64ff00cb8
commit 9f62c7904a
7 changed files with 212 additions and 91 deletions
@@ -1177,13 +1177,13 @@ export const SETTINGS_SCHEMA = {
"compaction.strategy": {
type: "enum",
values: ["context-full", "handoff", "shake", "off"] as const,
values: ["context-full", "handoff", "shake", "snapcompact", "off"] as const,
default: "context-full",
ui: {
tab: "context",
label: "Compaction Strategy",
description:
"Choose in-place context-full maintenance, auto-handoff, surgical shake (drop heavy content), or disable auto maintenance (off)",
"Choose in-place context-full maintenance, auto-handoff, surgical shake (drop heavy content), snapcompact (archive history as dense images), or disable auto maintenance (off)",
options: [
{
value: "context-full",
@@ -1196,6 +1196,11 @@ export const SETTINGS_SCHEMA = {
label: "Shake",
description: "Drop heavy content (tool results + large blocks) in place; recover via artifact",
},
{
value: "snapcompact",
label: "Snapcompact",
description: "Archive history onto dense bitmap images the model reads back; no LLM call",
},
{
value: "off",
label: "Off",
@@ -3363,7 +3368,7 @@ export type TreeFilterMode = SettingValue<"treeFilterMode">;
export interface CompactionSettings {
enabled: boolean;
strategy: "context-full" | "handoff" | "shake" | "off";
strategy: "context-full" | "handoff" | "shake" | "snapcompact" | "off";
thresholdPercent: number;
thresholdTokens: number;
reserveTokens: number;
@@ -103,11 +103,11 @@ export type CustomToolSessionEvent =
| {
reason: "auto_compaction_start";
trigger: "threshold" | "overflow" | "idle" | "incomplete";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
}
| {
reason: "auto_compaction_end";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
result: CompactionResult | undefined;
aborted: boolean;
willRetry: boolean;
@@ -204,13 +204,13 @@ export interface TurnEndEvent {
export interface AutoCompactionStartEvent {
type: "auto_compaction_start";
reason: "threshold" | "overflow" | "idle" | "incomplete";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
}
/** Fired when auto-compaction ends */
export interface AutoCompactionEndEvent {
type: "auto_compaction_end";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
result: CompactionResult | undefined;
aborted: boolean;
willRetry: boolean;
@@ -55,8 +55,14 @@ import {
type ShakeRegion,
type SummaryOptions,
shouldCompact,
snapcompactCompact,
} from "@oh-my-pi/pi-agent-core/compaction";
import { DEFAULT_PRUNE_CONFIG, pruneToolOutputs } from "@oh-my-pi/pi-agent-core/compaction/pruning";
import {
DEFAULT_PRUNE_CONFIG,
pruneSupersededToolResults,
pruneToolOutputs,
readToolSupersedeKey,
} from "@oh-my-pi/pi-agent-core/compaction/pruning";
import type { ProtectedToolMatcher } from "@oh-my-pi/pi-agent-core/compaction/tool-protection";
import type {
AssistantMessage,
@@ -258,11 +264,11 @@ export type AgentSessionEvent =
| {
type: "auto_compaction_start";
reason: "threshold" | "overflow" | "idle" | "incomplete";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
}
| {
type: "auto_compaction_end";
action: "context-full" | "handoff" | "shake";
action: "context-full" | "handoff" | "shake" | "snapcompact";
result: CompactionResult | undefined;
aborted: boolean;
willRetry: boolean;
@@ -6071,6 +6077,35 @@ export class AgentSession {
return result;
}
/**
* Per-turn supersede pass: prune older `read` results that a newer read of
* the same file has made stale. Cache-aware (only fires when the suffix
* after a candidate is small or the session has been idle long enough that
* the provider prompt cache is cold), so it is cheap to run every turn.
* Gated on the `compaction.supersedeReads` setting.
*/
async #pruneSupersededReads(): Promise<{ prunedCount: number; tokensSaved: number } | undefined> {
if (!this.settings.getGroup("compaction").supersedeReads) return undefined;
const branchEntries = this.sessionManager.getBranch();
const result = pruneSupersededToolResults(
branchEntries,
this.#withPlanProtection({
supersedeKey: readToolSupersedeKey,
protectedTools: [...DEFAULT_PRUNE_CONFIG.protectedTools],
}),
);
if (result.prunedCount === 0) {
return undefined;
}
await this.sessionManager.rewriteEntries();
const sessionContext = this.buildDisplaySessionContext();
this.agent.replaceMessages(sessionContext.messages);
this.#syncTodoPhasesFromBranch();
this.#closeCodexProviderSessionsForHistoryRewrite();
return result;
}
/**
* Strip image content blocks from every message on the current branch and
* persist the rewrite. Walks `SessionManager.getBranch()` in place — both
@@ -6260,6 +6295,20 @@ export class AgentSession {
const compactionPrep = await this.#prepareCompactionFromHooks(preparation, hookCompaction);
// Strategy honored on manual /compact too. Custom instructions imply a
// directed LLM summary; a text-only model cannot read the frames back —
// both take the summarizer path (the latter loudly).
const wantsSnapcompact =
compactionPrep.kind !== "fromHook" && compactionSettings.strategy === "snapcompact" && !customInstructions;
const snapcompactReady = wantsSnapcompact && this.model.input.includes("image");
if (wantsSnapcompact && !snapcompactReady) {
this.emitNotice(
"warning",
`snapcompact needs a vision-capable model (${this.model.id} is text-only) — using an LLM summary instead`,
"compaction",
);
}
let summary: string;
let shortSummary: string | undefined;
let firstKeptEntryId: string;
@@ -6273,6 +6322,14 @@ export class AgentSession {
tokensBefore = compactionPrep.tokensBefore;
details = compactionPrep.details;
preserveData = compactionPrep.preserveData;
} else if (snapcompactReady) {
const snapcompactResult = await snapcompactCompact(preparation, { convertToLlm });
summary = snapcompactResult.summary;
shortSummary = snapcompactResult.shortSummary;
firstKeptEntryId = snapcompactResult.firstKeptEntryId;
tokensBefore = snapcompactResult.tokensBefore;
details = snapcompactResult.details;
preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) };
} else {
// Generate compaction result. Only convert known abort-shaped
// rejections (AbortError raised while the abort signal is set,
@@ -6703,6 +6760,10 @@ export class AgentSession {
return false;
}
// Supersede pass runs every turn, before any threshold gating: it is cheap
// (bails when no candidate) and independent of the compaction setting.
const supersedeResult = await this.#pruneSupersededReads();
const compactionSettings = this.settings.getGroup("compaction");
if (!compactionSettings.enabled || compactionSettings.strategy === "off") return false;
@@ -6711,6 +6772,9 @@ export class AgentSession {
if (assistantMessage.stopReason === "error") return false;
const pruneResult = await this.#pruneToolOutputs();
let contextTokens = calculateContextTokens(assistantMessage.usage);
if (supersedeResult) {
contextTokens = Math.max(0, contextTokens - supersedeResult.tokensSaved);
}
if (pruneResult) {
contextTokens = Math.max(0, contextTokens - pruneResult.tokensSaved);
}
@@ -7601,9 +7665,25 @@ export class AgentSession {
// "overflow" forces context-full because the input itself is broken — a handoff
// LLM call would hit the same overflow. "incomplete" is an output-side problem,
// so a handoff request on the existing context is still viable.
let action: "context-full" | "handoff" =
// so a handoff request on the existing context is still viable. Snapcompact is
// safe for every reason (it makes no LLM call at all) but requires a vision
// model to be worth anything — fall back to context-full otherwise.
let action: "context-full" | "handoff" | "snapcompact" =
compactionSettings.strategy === "handoff" && reason !== "overflow" ? "handoff" : "context-full";
if (compactionSettings.strategy === "snapcompact") {
if (this.model?.input.includes("image")) {
action = "snapcompact";
} else {
logger.warn("Snapcompact compaction requires a vision-capable model; falling back to context-full", {
model: this.model?.id,
});
this.emitNotice(
"warning",
`snapcompact needs a vision-capable model (${this.model?.id ?? "unknown"} is text-only) — using an LLM summary instead`,
"compaction",
);
}
}
await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action });
// Abort any older auto-compaction before installing this run's controller.
this.#autoCompactionAbortController?.abort();
@@ -7742,6 +7822,16 @@ export class AgentSession {
tokensBefore = compactionPrep.tokensBefore;
details = compactionPrep.details;
preserveData = compactionPrep.preserveData;
} else if (action === "snapcompact") {
// Local, deterministic: render discarded history onto PNG frames.
// No model candidates, no API key, no retry loop.
const snapcompactResult = await snapcompactCompact(preparation, { convertToLlm });
summary = snapcompactResult.summary;
shortSummary = snapcompactResult.shortSummary;
firstKeptEntryId = snapcompactResult.firstKeptEntryId;
tokensBefore = snapcompactResult.tokensBefore;
details = snapcompactResult.details;
preserveData = { ...(compactionPrep.preserveData ?? {}), ...(snapcompactResult.preserveData ?? {}) };
} else {
const candidates = this.#getCompactionModelCandidates(availableModels);
const retrySettings = this.settings.getGroup("retry");
+11 -78
View File
@@ -8,8 +8,7 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import {
type BranchSummaryMessage,
type CompactionSummaryMessage,
renderBranchSummaryContext,
renderCompactionSummaryContext,
convertMessageToLlm,
} from "@oh-my-pi/pi-agent-core/compaction/messages";
import type {
AssistantMessage,
@@ -17,7 +16,6 @@ import type {
Message,
MessageAttribution,
TextContent,
ToolResultMessage,
UserMessage,
} from "@oh-my-pi/pi-ai";
import { prompt } from "@oh-my-pi/pi-utils";
@@ -28,6 +26,7 @@ export {
type CompactionSummaryMessage,
createBranchSummaryMessage,
createCompactionSummaryMessage,
createCustomMessage,
} from "@oh-my-pi/pi-agent-core/compaction/messages";
import type { OutputMeta } from "../tools/output-meta";
@@ -59,7 +58,7 @@ export interface SkillPromptDetails {
*
* Consumers: `AgentSession.#handleAgentEvent` (stamper) writes this value;
* `EventController.#handleMessageEnd`, `AssistantMessageComponent`,
* `ui-helpers.addMessageToChat` (renderers), `SessionObserverOverlay
* `ui-helpers.addMessageToChat` (renderers), `AgentHubOverlayComponent
* #buildTranscriptLines`, `runPrintMode`, and `AcpAgent#replayAssistantMessage`
* (fallback error emission) read it via `isSilentAbort`. */
export const SILENT_ABORT_MARKER = "__omp.silent_abort__";
@@ -220,15 +219,6 @@ export function wrapSteeringForModel(messages: AgentMessage[]): AgentMessage[] {
return wrappedMessages ?? messages;
}
function getPrunedToolResultContent(message: ToolResultMessage): (TextContent | ImageContent)[] {
if (message.prunedAt === undefined) {
return message.content;
}
const textBlocks = message.content.filter((content): content is TextContent => content.type === "text");
const text = textBlocks.map(block => block.text).join("") || "[Output truncated]";
return [{ type: "text", text }];
}
/** Result of filtering image blocks out of a `(TextContent | ImageContent)[]` array. */
interface StripContentResult {
content: (TextContent | ImageContent)[];
@@ -478,26 +468,6 @@ export function sanitizeRehydratedOpenAIResponsesAssistantMessage(message: Assis
};
}
/** Convert CustomMessageEntry to AgentMessage format */
export function createCustomMessage(
customType: string,
content: string | (TextContent | ImageContent)[],
display: boolean,
details: unknown | undefined,
timestamp: string,
attribution?: MessageAttribution,
): CustomMessage {
return {
role: "custom",
customType,
content,
display,
details,
attribution,
timestamp: new Date(timestamp).getTime(),
};
}
/**
* Transform AgentMessages (including custom types) to LLM-compatible Messages.
*
@@ -530,43 +500,6 @@ export function convertToLlm(messages: AgentMessage[]): Message[] {
attribution: "user",
timestamp: m.timestamp,
};
case "custom":
case "hookMessage": {
const content = typeof m.content === "string" ? [{ type: "text" as const, text: m.content }] : m.content;
const role = "developer";
const attribution = m.attribution;
return {
role,
content,
attribution,
timestamp: m.timestamp,
};
}
case "branchSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderBranchSummaryContext(m.summary),
},
],
attribution: "agent",
timestamp: m.timestamp,
};
case "compactionSummary":
return {
role: "user",
content: [
{
type: "text" as const,
text: renderCompactionSummaryContext(m.summary),
},
],
attribution: "agent",
providerPayload: m.providerPayload,
timestamp: m.timestamp,
};
case "fileMention": {
const fileContents = m.files
.map(file => {
@@ -587,18 +520,18 @@ export function convertToLlm(messages: AgentMessage[]): Message[] {
timestamp: m.timestamp,
};
}
case "custom":
case "hookMessage":
case "branchSummary":
case "compactionSummary":
case "user":
return { ...m, attribution: m.attribution ?? "user" };
case "developer":
return { ...m, attribution: m.attribution ?? "agent" };
case "assistant":
return m;
case "toolResult":
return {
...m,
content: getPrunedToolResultContent(m as ToolResultMessage),
attribution: m.attribution ?? "agent",
};
// Core roles share one transformer with agent-core —
// duplicating them here is how snapcompact frames once
// silently fell off the provider request.
return convertMessageToLlm(m);
default:
m satisfies never;
return undefined;
@@ -814,6 +814,58 @@ describe("buildSessionContext", () => {
expect((loaded.messages[0] as any).summary).toContain("Summary of 1,a,2,b");
});
it("re-attaches snapcompact frames from preserveData as compaction summary images", () => {
const u1 = createMessageEntry(createUserMessage("1"));
const a1 = createMessageEntry(createAssistantMessage("a"));
const u2 = createMessageEntry(createUserMessage("2"));
const frame = { data: "ZmFrZQ==", mimeType: "image/png", cols: 64, rows: 40, chars: 4 };
const compaction: CompactionEntry = {
...createCompactionEntry("Filmed summary", u2.id),
preserveData: { snapcompact: { frames: [frame], totalChars: 4, truncatedChars: 0 } },
};
const u3 = createMessageEntry(createUserMessage("3"));
const loaded = buildSessionContext([u1, a1, u2, compaction, u3]);
const summaryMessage = loaded.messages[0] as { role: string; images?: unknown };
expect(summaryMessage.role).toBe("compactionSummary");
expect(summaryMessage.images).toEqual([{ type: "image", data: "ZmFrZQ==", mimeType: "image/png" }]);
});
it("transcript option keeps full history with every compaction inline at its position", () => {
const u1 = createMessageEntry(createUserMessage("1"));
const a1 = createMessageEntry(createAssistantMessage("a"));
const compact1 = createCompactionEntry("First summary", u1.id);
const u2 = createMessageEntry(createUserMessage("2"));
const frame = { data: "ZmFrZQ==", mimeType: "image/png", cols: 64, rows: 40, chars: 4 };
const compact2: CompactionEntry = {
...createCompactionEntry("Second summary", u2.id),
preserveData: { snapcompact: { frames: [frame], totalChars: 4, truncatedChars: 0 } },
};
const u3 = createMessageEntry(createUserMessage("3"));
const entries: SessionEntry[] = [u1, a1, compact1, u2, compact2, u3];
const transcript = buildSessionContext(entries, undefined, undefined, { transcript: true });
// Nothing erased: every message survives, compactions sit where they fired.
expect(transcript.messages.map(m => m.role)).toEqual([
"user",
"assistant",
"compactionSummary",
"user",
"compactionSummary",
"user",
]);
const first = transcript.messages[2] as { summary: string };
const second = transcript.messages[4] as { summary: string; images?: unknown };
expect(first.summary).toContain("First summary");
expect(second.summary).toContain("Second summary");
// Snapcompact frames ride along in the transcript too.
expect(second.images).toEqual([{ type: "image", data: "ZmFrZQ==", mimeType: "image/png" }]);
// LLM context is untouched by the option: latest compaction replaces history.
const llm = buildSessionContext(entries);
expect(llm.messages.map(m => m.role)).toEqual(["compactionSummary", "user", "user"]);
});
it("should handle multiple compactions (only latest matters)", () => {
// First batch
const u1 = createMessageEntry(createUserMessage("1"));
@@ -1,6 +1,6 @@
import { describe, expect, it } from "bun:test";
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import type { ImageContent, Message } from "@oh-my-pi/pi-ai";
import type { ImageContent, Message, TextContent } from "@oh-my-pi/pi-ai";
import { inferCopilotInitiator } from "@oh-my-pi/pi-ai/providers/github-copilot-headers";
import { convertToLlm, wrapSteeringForModel } from "@oh-my-pi/pi-coding-agent/session/messages";
@@ -13,6 +13,47 @@ function expectAttribution(message: Message | undefined, expected: "user" | "age
expect(message.attribution).toBe(expected);
}
describe("convertToLlm compaction summary", () => {
it("appends snapcompact frames as image blocks after the summary text", () => {
// Regression: the live session uses THIS converter (not agent-core's
// defaultConvertToLlm). Dropping the frames here silently severs the
// archive from the provider request — the model sees a summary that
// references attached frames that never arrive.
const images: ImageContent[] = [
{ type: "image", data: "ZmFrZQ==", mimeType: "image/png" },
{ type: "image", data: "ZmFrZTI=", mimeType: "image/png" },
];
const messages: AgentMessage[] = [
{
role: "compactionSummary",
summary: "the film archive",
tokensBefore: 1000,
images,
timestamp: Date.now(),
},
];
const converted = convertToLlm(messages);
expect(converted).toHaveLength(1);
expect(converted[0]?.role).toBe("user");
const content = converted[0]?.content as Array<TextContent | ImageContent>;
expect(content).toHaveLength(3);
expect(content[0].type).toBe("text");
expect((content[0] as TextContent).text).toContain("the film archive");
expect(content[1]).toEqual(images[0]);
expect(content[2]).toEqual(images[1]);
});
it("emits text-only content when no frames are archived", () => {
const messages: AgentMessage[] = [
{ role: "compactionSummary", summary: "plain summary", tokensBefore: 1000, timestamp: Date.now() },
];
const converted = convertToLlm(messages);
expect((converted[0]?.content as unknown[]).length).toBe(1);
});
});
describe("convertToLlm custom message mapping", () => {
it("maps custom messages to developer role with explicit agent attribution", () => {
const messages: AgentMessage[] = [