Files
oh-my-pi/packages/coding-agent/test/agent-session-snapcompact-budget.test.ts
T
omp-eval 78e469088d test(agent): decouple snapcompact skip test from provider-dependent LLM fallback
The "skips snapcompact entirely" case asserted .rejects.toThrow() on
session.compact(), but after the maxFrames<1 skip the manual /compact path
falls through to the LLM summarizer whose outcome is provider/network
dependent (resolves when a summary lands, rejects only without
credentials/network). In a sandbox it resolves and blows the 5s default
timeout. Tolerate either outcome and pin only the deterministic skip
contract (snapcompact.compact not invoked + the kept-history notice), with
a generous explicit timeout. Addresses chatgpt-codex P2 review on #3249.
2026-06-24 18:26:19 +02:00

289 lines
13 KiB
TypeScript
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
/**
* Regression test for issue #3247.
*
* Snapcompact's bundled `MAX_FRAMES_DEFAULT = 80` × `FRAME_TOKEN_ESTIMATE = 5024`
* ≈ 402k tokens worth of frames. On any sub-1M-token window (e.g. Claude
* Sonnet 4.5's 200k), passing the default cap to `snapcompact.compact()` made
* the post-render projection in `AgentSession` always overflow the budget,
* emit the "snapcompact could not bring the context under the limit" warning
* on every threshold tick, and downgrade to an LLM summary. The fix sizes the
* `maxFrames` cap from the live model window (window − reserve − non-message
* overhead − kept-recent − summary-text reserve) before calling
* `snapcompact.compact()`.
*
* The contract this test defends: for a 200k-window vision model with sane
* kept-recent traffic, AgentSession MUST pass a budget-sized `maxFrames`
* (smaller than `MAX_FRAMES_DEFAULT`, and with `maxFrames × FRAME_TOKEN_ESTIMATE`
* inside the resolved budget) so the projection accepts the snapcompact
* result instead of falling back to the LLM summarizer.
*/
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { Agent } from "@oh-my-pi/pi-agent-core";
import { effectiveReserveTokens, estimateTokens, prepareCompaction } from "@oh-my-pi/pi-agent-core/compaction";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { computeNonMessageTokens } from "@oh-my-pi/pi-coding-agent/modes/utils/context-usage";
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { TempDir } from "@oh-my-pi/pi-utils";
import * as snapcompact from "@oh-my-pi/snapcompact";
describe("AgentSession snapcompact frame-budget sizing", () => {
let tempDir: TempDir;
let session: AgentSession;
let sessionManager: SessionManager;
let authStorage: AuthStorage;
let modelRegistry: ModelRegistry;
beforeEach(async () => {
tempDir = TempDir.createSync("@pi-snapcompact-budget-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
authStorage.setRuntimeApiKey("anthropic", "test-key");
modelRegistry = new ModelRegistry(authStorage);
sessionManager = SessionManager.create(tempDir.path(), tempDir.path());
const model = getBundledModel("anthropic", "claude-sonnet-4-5");
if (!model) throw new Error("Expected bundled claude-sonnet-4-5 model");
// Sanity: the contract only holds for vision models with a window
// genuinely smaller than the snapcompact upper bound. If the bundled
// catalog ever raises Sonnet's window past 1M, this test no longer
// covers the failure mode the fix targets.
expect(model.input).toContain("image");
expect(model.contextWindow).toBeLessThan(1_000_000);
const agent = new Agent({
initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] },
});
// Seed a representative long-running session: many turn-pairs with
// substantial filler so prepareCompaction() splits the branch into
// "discard + summarize" (oldest) vs "kept-recent" (newest).
const filler = "the quick brown fox jumps over the lazy dog. ".repeat(64);
for (let i = 0; i < 64; i++) {
sessionManager.appendMessage({
role: "user",
content: [{ type: "text", text: `turn ${i}: ${filler}` }],
timestamp: Date.now() - (64 - i) * 1000,
});
sessionManager.appendMessage({
role: "assistant",
content: [{ type: "text", text: `reply ${i}: ${filler}` }],
api: "anthropic-messages",
provider: "anthropic",
model: "claude-sonnet-4-5",
stopReason: "stop",
usage: {
input: 1000,
output: 1000,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 2000,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
timestamp: Date.now() - (64 - i) * 1000 + 100,
});
}
session = new AgentSession({
agent,
sessionManager,
settings: Settings.isolated({
"compaction.strategy": "snapcompact",
"compaction.autoContinue": false,
// Force a small kept-recent window so the seeded conversation
// definitely splits into discard + kept and prepareCompaction()
// returns a non-empty preparation.
"compaction.keepRecentTokens": 4000,
}),
modelRegistry,
});
});
afterEach(async () => {
try {
await session?.dispose();
} finally {
authStorage?.close();
await tempDir?.remove();
vi.restoreAllMocks();
}
});
it("passes a maxFrames whose full projection (frames + text edges + base) fits the budget", async () => {
// Tighten kept-recent into the realistic ~100k-token range. Without
// it, the helper has so much headroom that even a flawed (too-large)
// cap reserve passes the `maxFrames × FRAME_TOKEN_ESTIMATE < budget`
// check by accident. Reviewer chatgpt-codex on #3249 cited the exact
// failure mode: ~120k headroom on Anthropic 11on16-bw chose 23 frames
// under the previous 4k-reserve helper, but `23 × 5024 + 7k text
// edges + 2k summary template + base` then exceeded the same headroom.
const model = session.model;
if (!model) throw new Error("Expected model to be set on session");
const ctxWindow = model.contextWindow ?? 0;
expect(ctxWindow).toBeGreaterThan(0);
const settings = { enabled: true as const, reserveTokens: 16384, keepRecentTokens: 4000 };
const reserve = effectiveReserveTokens(ctxWindow, settings);
const budget = ctxWindow - reserve;
// Filler tuned so `baseTokens ≈ 100k`, leaving ~70k headroom — the
// regime where a shape-aware cap reserve actually matters.
const targetRecentTokens = 100_000;
const filler = "x".repeat(targetRecentTokens * 4);
sessionManager.appendMessage({
role: "user",
content: [{ type: "text", text: filler }],
timestamp: Date.now(),
});
const branchEntries = sessionManager.getBranch();
const firstKeptEntry = branchEntries[branchEntries.length - 1];
if (!firstKeptEntry?.id) throw new Error("Expected branch entry with id");
const compactSpy = vi.spyOn(snapcompact, "compact").mockResolvedValue({
summary: "stubbed snapcompact",
shortSummary: "stub",
firstKeptEntryId: firstKeptEntry.id,
tokensBefore: 100_000,
details: { readFiles: [], modifiedFiles: [] },
preserveData: {
snapcompact: { frames: [], totalChars: 0, truncatedChars: 0 },
},
});
await session.compact(undefined, { mode: "snapcompact" });
expect(compactSpy).toHaveBeenCalledTimes(1);
const opts = compactSpy.mock.calls[0]?.[1];
expect(opts).toBeDefined();
const maxFrames = opts?.maxFrames;
expect(maxFrames).toBeDefined();
expect(maxFrames).toBeLessThan(snapcompact.MAX_FRAMES_DEFAULT);
expect(maxFrames).toBeGreaterThan(0);
// Verify the FULL projection — base (non-message + kept-recent) +
// frame-bearing summary cost — fits the budget. The projection
// {@link #projectSnapcompactContextTokens} mirrors what the auto and
// manual paths charge: countTokens(summary + textHead + textTail) +
// numFrames × FRAME_TOKEN_ESTIMATE + non-message + kept-recent.
const preparation = prepareCompaction(branchEntries, settings);
if (!preparation) throw new Error("Expected non-empty preparation");
let baseTokens = computeNonMessageTokens(session);
for (const message of preparation.recentMessages) {
baseTokens += estimateTokens(message);
}
const shape = snapcompact.resolveShape(model);
const edgeCap = snapcompact.geometry(shape).capacity;
// Worst-case `textHead + textTail` tokenized at the cl100k 4-chars/token
// baseline, plus a 2k allowance for the snapcompact summary template
// (intro + FILES section + grid notes).
const worstCaseEdgeTokens = Math.ceil((2 * edgeCap) / 4) + 2000;
const fullProjection = baseTokens + (maxFrames ?? 0) * snapcompact.FRAME_TOKEN_ESTIMATE + worstCaseEdgeTokens;
expect(fullProjection).toBeLessThanOrEqual(budget);
});
it("skips snapcompact entirely when kept-recent already exceeds the budget", async () => {
// Append one synthetic message large enough to overflow the model window
// on its own (kept by findCutPoint since keepRecentTokens=4000 falls
// well short of it). Snapcompact CANNOT fit even a single frame; the
// session MUST skip it instead of running and emitting "could not bring
// the context under the limit" every tick.
const model = session.model;
if (!model) throw new Error("Expected model");
const ctxWindow = model.contextWindow ?? 0;
const huge = "a".repeat(ctxWindow * 4);
sessionManager.appendMessage({
role: "user",
content: [{ type: "text", text: huge }],
timestamp: Date.now(),
});
const compactSpy = vi.spyOn(snapcompact, "compact");
const notices: { level: string; message: string }[] = [];
session.subscribe(event => {
if (event.type === "notice") {
notices.push({ level: event.level, message: event.message });
}
});
// After the skip, the manual /compact path falls through to the LLM
// summarizer; its outcome (resolve once a summary lands, reject when no
// credentials/network are available) is provider-dependent and NOT this
// test's contract. Tolerate either so the suite never depends on network
// or auth (chatgpt-codex review on #3249). The skip itself is the pin.
await session.compact(undefined, { mode: "snapcompact" }).then(
() => {},
() => {},
);
// snapcompact.compact() MUST NOT be invoked when the budget cannot
// fit even one frame — running it just to reject the result and
// re-emit the warning is the exact loop issue #3247 reports.
expect(compactSpy).not.toHaveBeenCalled();
// The user-facing notice MUST explain the kept-history overflow rather
// than the misleading "could not bring the context under the limit"
// (which implied snapcompact had run and produced an oversized result).
expect(notices.some(n => n.level === "warning" && n.message.includes("kept history"))).toBe(true);
}, 30_000);
it("still invokes snapcompact with maxFrames=1 when residual headroom is below the summary-text reserve", async () => {
// Reviewer (chatgpt-codex on #3249, second pass): when kept-recent +
// non-message leaves SOME real headroom but less than the 4k
// SUMMARY_TEXT_RESERVE the helper holds back to size frame caps, the
// previous revision still went negative and returned 0 (skipped
// snapcompact). But a text-only snapcompact archive (the
// `text.length <= 2 * edgeCap` short-circuit in `planArchive`)
// typically costs only a few hundred tokens of summary lead, far
// below 4k. The skip decision MUST use raw `baseTokens >= totalBudget`
// — the cap reserve applies only to the maxFrames math, not the skip.
const model = session.model;
if (!model) throw new Error("Expected model");
const ctxWindow = model.contextWindow ?? 0;
// Tune kept-recent so the residual `totalBudget − baseTokens` is
// 1500 tokens — strictly positive, but well below the 4k cap reserve.
// The previous helper would compute frameBudget = 1500 − 4000 = −2500
// and return 0; the fixed helper returns 1 because the residual is
// positive and the text-only archive can still fit.
const reserve = Math.max(Math.floor(ctxWindow * 0.15), 16384);
const headroomTokens = 1500;
const targetRecentTokens = ctxWindow - reserve - headroomTokens;
// Rough 4-chars-per-token rule for the tiktoken estimator on ASCII.
const filler = "x".repeat(targetRecentTokens * 4);
sessionManager.appendMessage({
role: "user",
content: [{ type: "text", text: filler }],
timestamp: Date.now(),
});
const branchEntries = sessionManager.getBranch();
const lastEntry = branchEntries[branchEntries.length - 1];
if (!lastEntry?.id) throw new Error("Expected branch entry with id");
const compactSpy = vi.spyOn(snapcompact, "compact").mockResolvedValue({
summary: "stubbed snapcompact",
shortSummary: "stub",
firstKeptEntryId: lastEntry.id,
tokensBefore: 100_000,
// Text-only archive: zero frames, modest text edges. The projection
// charges 0 for frames, so the post-compaction context fits.
details: { readFiles: [], modifiedFiles: [] },
preserveData: {
snapcompact: { frames: [], totalChars: 1000, truncatedChars: 0 },
},
});
await session.compact(undefined, { mode: "snapcompact" });
expect(compactSpy).toHaveBeenCalledTimes(1);
const opts = compactSpy.mock.calls[0]?.[1];
// Snapcompact MUST be invoked with the floor cap, never skipped,
// even though one frame charge would overflow the budget — the
// text-only `planArchive` path makes this case recoverable.
expect(opts?.maxFrames).toBe(1);
});
});