From 14bd572f49bcfdaa576be031803d32da4b3a0758 Mon Sep 17 00:00:00 2001 From: can1357 Date: Sun, 31 May 2026 14:14:37 +0200 Subject: [PATCH] refactor(shake): removed shake-summary mode and local-model compressor - Dropped `summarizeShakeRegions`, the shake-summary prompt, and related types. - Removed `shake-summary` compaction strategy and `providers.shakeSummaryModel` setting. - Migrated existing `shake-summary` configs to plain `shake` on load. - Simplified `/shake` to `elide` and `images` modes only. --- docs/local-models.md | 50 -------- packages/agent/CHANGELOG.md | 4 + packages/agent/src/compaction/compaction.ts | 2 +- .../src/compaction/prompts/shake-summary.md | 29 ----- packages/agent/src/compaction/shake.ts | 117 +----------------- packages/agent/test/shake.test.ts | 68 +--------- packages/coding-agent/CHANGELOG.md | 4 + .../src/config/settings-schema.ts | 27 +--- packages/coding-agent/src/config/settings.ts | 10 ++ .../src/extensibility/custom-tools/types.ts | 4 +- .../src/extensibility/shared-events.ts | 4 +- .../modes/controllers/command-controller.ts | 46 ++----- .../src/modes/controllers/event-controller.ts | 10 +- .../coding-agent/src/session/agent-session.ts | 104 ++-------------- .../coding-agent/src/session/shake-types.ts | 9 +- .../src/slash-commands/builtin-registry.ts | 6 +- packages/coding-agent/src/tiny/models.ts | 28 ----- .../coding-agent/src/tiny/title-client.ts | 12 +- .../coding-agent/src/tiny/title-protocol.ts | 12 +- packages/coding-agent/src/tiny/worker.ts | 22 +--- packages/coding-agent/test/shake.test.ts | 51 -------- .../test/slash-commands/shake.test.ts | 8 +- 22 files changed, 71 insertions(+), 556 deletions(-) delete mode 100644 packages/agent/src/compaction/prompts/shake-summary.md diff --git a/docs/local-models.md b/docs/local-models.md index 5d53c878e..68502d27b 100644 --- a/docs/local-models.md +++ b/docs/local-models.md @@ -135,56 +135,6 @@ wins that task. - `String(item)` produces `[object Object]` on object array items. - The line-fallback drops items `<=10` chars, so a correct short fact like `Name: Can` is discarded. -## Task 3: Shake-summary compression (`providers.shakeSummaryModel`) - -**Task**: extractively compress aged heavy tool-result regions for `/shake summary` and the -`shake-summary` auto-compaction strategy. This path is strictly local/on-device and always keeps an -`artifact://` recovery link, so the model must prefer faithful omission over invented detail — the -full original is one fetch away. - -**Grain (validated against practice)**: the summary is **per tool result** and **extractive** — what -the result established (paths, identifiers, signatures, error messages, exit codes, commands kept -verbatim), not a free-form "what happened" narrative. Industry consensus (Factory.ai compression -evals, LangChain Deep Agents, Manus, Anthropic's agent loop) is that per-tool/per-phase extractive -summaries with an external artifact pointer preserve attribution and recoverability far better than a -single whole-trajectory narrative; "what happened" prose belongs in a separate global session-state -layer, not the per-result path. A per-result "what happened" summary is near-contentless (the tool -call already says what ran), which is exactly what the bench showed. - -**Bench**: dev script `scripts/bench-shake-summary.ts` against one real `-Projects-pi` transcript, -driving the shared tiny-model worker directly (q4, CPU, greedy). It captures coverage (regions that -parse to a `` summary), compression ratio, latency, and — for unparseable outputs — the raw -completion, so format failures are judgeable instead of vanishing. Representative aged **read** result -(`auth-storage.ts`, 3002 lines, middle-truncated to a 32 KB prompt sample). Artifacts: -`/tmp/shake-bench-lfm2-350m.json`, `/tmp/shake-bench-prefill*.json`. - -**Findings**: -- **Model floor is ~1B.** LFM2-350M loads fastest (~0.3 s) but on a long read it *hallucinated* - fictional code in a markdown fence instead of extracting, and never emitted the `` format — - unusable for a faithful record. Sub-1B models pattern-match "rewrite the code" and confabulate. -- **Prefill fixes format, not comprehension.** Pinning the assistant turn open with the output tag - (a recognized SLM technique) forced LFM2-350M/700M to emit a well-formed block, but the *content* - stayed empty/garbage (`409`, `The`, `This`). A content-bearing prefix (`The tool returned `) makes - it worse — it biases a garden-path completion. Prefill must pin format only. -- **Single-shot whole-region input breaks the capable model.** Feeding ~10 K tokens in one completion - crashed Qwen3-1.7B's q4 ONNX build ("Unknown failure"); the production path avoids this by batching - at `DEFAULT_BATCH_TOKEN_BUDGET` (4 K), which Qwen3-1.7B handles cleanly. - -**Recommendation**: keep **Qwen3-1.7B** as the shake-summary default. It is not the fastest, but the -task values faithfulness over prettiness — invented line numbers, paths, or commands are worse than -terse omission since the artifact remains recoverable — and Qwen3-1.7B is the smallest candidate that -extracts faithfully without confabulating. Incremental background precompute (see below) amortizes its -latency outside the foreground compaction path. A format-only prefill (``) is a -low-risk future reliability win for the local models; the worker already supports it. - -**Shipped local options**: `qwen3-1.7b` (recommended), `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`. -**Default**: `qwen3-1.7b`. - -**Instant compaction**: aged eligible tool results past the shake protect window are summarized in the -background (off-thread worker) as they age out and cached on the message (`ToolResultMessage.shakeSummary`, -keyed by `toolCallId` + content hash + model). A warm `/shake summary` then reuses the cache and issues -zero foreground `complete` calls; cache entries invalidate on content-hash or model-key change and are -skipped once `prunedAt` is set. ## Integration notes diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index de14686bd..03655ff3b 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Removed + +- Removed the local-model `summarizeShakeRegions` compressor and related shake-summary prompt/types; shake now only provides mechanical artifact-backed elision primitives. + ## [15.7.3] - 2026-05-31 ### Added diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index a0f70afc6..47c9a58ac 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -135,7 +135,7 @@ export interface CompactionResult { export interface CompactionSettings { enabled: boolean; - strategy?: "context-full" | "handoff" | "shake" | "shake-summary" | "off"; + strategy?: "context-full" | "handoff" | "shake" | "off"; thresholdPercent?: number; thresholdTokens?: number; reserveTokens: number; diff --git a/packages/agent/src/compaction/prompts/shake-summary.md b/packages/agent/src/compaction/prompts/shake-summary.md deleted file mode 100644 index 9a4e18f9c..000000000 --- a/packages/agent/src/compaction/prompts/shake-summary.md +++ /dev/null @@ -1,29 +0,0 @@ -You compress heavy regions of a coding-agent conversation so they take less context while staying faithful. Each region is a tool result or a large code/markup block that is being dropped from live context. - -You will receive regions wrapped as: - -``` - - -...original content... - -... - -``` - -For EACH input region, emit one compressed block: - -``` - -...compressed content... - -``` - -Rules: - -- EXTRACT, do not rewrite. Keep exact file paths, identifiers, symbol names, signatures, line numbers, error messages, exit codes, command names, URLs, and concrete decisions verbatim. Never invent, rename, or "clean up" any of them. -- Drop only redundancy: repeated boilerplate, decorative output, long unchanged spans, ASCII art, progress bars, and filler prose. -- Preserve the gist: what the region established, what it found, what changed, and any value the agent may still need to recall. -- Be terse. Prefer short lines and fragments over sentences. Aim well under the original size. -- Emit exactly one `` element per input region, reusing the same `index`. Output nothing outside the `` elements — no preamble, no commentary. -- If a region holds nothing worth keeping, emit `(no salient content)`. diff --git a/packages/agent/src/compaction/shake.ts b/packages/agent/src/compaction/shake.ts index 6de9366e9..f979d659e 100644 --- a/packages/agent/src/compaction/shake.ts +++ b/packages/agent/src/compaction/shake.ts @@ -1,23 +1,20 @@ /** * Context-reducing surgical compaction ("shake"). * - * `shake` drops heavy content out of the live context mechanically rather than - * via an LLM summary: whole tool-call results and large fenced/XML blocks are - * replaced with short placeholders (elide) or extractive compressions - * (summary). This module is the pure layer — region detection and in-place - * mutation only. Artifact offload, LLM calls, persistence, and provider-session - * teardown are orchestrated by the caller (`AgentSession.shake`). + * `shake` drops heavy content out of the live context mechanically: whole + * tool-call results and large fenced/XML blocks are replaced with short + * placeholders. This module is the pure layer — region detection and in-place + * mutation only. Artifact offload, persistence, and provider-session teardown + * are orchestrated by the caller (`AgentSession.shake`). * * Layering mirrors `pruning.ts`: no I/O here. */ import type { TextContent, ToolResultMessage } from "@oh-my-pi/pi-ai"; import { countTokens } from "@oh-my-pi/pi-natives"; -import { prompt } from "@oh-my-pi/pi-utils"; import type { AgentMessage } from "../types"; import { estimateTokens } from "./compaction"; import type { CustomMessageEntry, SessionEntry, SessionMessageEntry } from "./entries"; -import shakeSummaryPrompt from "./prompts/shake-summary.md" with { type: "text" }; import { collectToolCallsById, isProtectedToolResult, @@ -407,107 +404,3 @@ export function applyShakeRegions(items: Array<{ region: ShakeRegion; replacemen }); for (const { region, replacement } of ordered) applyShakeRegion(region, replacement); } - -// ============================================================================ -// Summary-mode compressor -// ============================================================================ - -const SHAKE_SUMMARY_PROMPT = prompt.render(shakeSummaryPrompt); - -/** One region handed to the summary compressor. */ -export interface ShakeSummaryItem { - index: number; - label: string; - text: string; -} - -/** - * Completion backend for shake summary. Mirrors the on-device local client - * (`tinyModelClient.complete`): a single prompt in, text out, or `null` when - * the model is unavailable / produced nothing. Injected by the caller so this - * module stays I/O-free and provider-agnostic — shake summary runs on a local - * model, never a remote/cloud LLM. - */ -export type ShakeSummaryComplete = ( - prompt: string, - options: { maxTokens: number; signal?: AbortSignal }, -) => Promise; - -export interface ShakeSummaryOptions { - signal?: AbortSignal; - /** Approximate input-token budget per completion call. */ - batchTokenBudget?: number; -} - -// Local models run small context windows; keep batches modest. -const DEFAULT_BATCH_TOKEN_BUDGET = 4_000; - -function buildSummaryPrompt(items: ShakeSummaryItem[]): string { - const parts: string[] = [SHAKE_SUMMARY_PROMPT, "", ""]; - for (const item of items) { - parts.push(``, item.text, ""); - } - parts.push(""); - return parts.join("\n"); -} - -function parseSummaryResponse(responseText: string, indices: number[]): Map { - const result = new Map(); - for (const index of indices) { - const pattern = new RegExp(`]*>([\\s\\S]*?)`); - const match = pattern.exec(responseText); - if (!match) continue; - const text = match[1].trim(); - if (text.length > 0) result.set(index, text); - } - return result; -} - -/** - * Extractively compress shake regions with a local model. - * - * Batches regions by `batchTokenBudget`, issues one `complete` call per batch, - * and parses the delimited `…` output leniently. - * Regions the backend omits/empties — and every region in a batch the backend - * returns `null` for (model unavailable / nothing produced) — are simply absent - * from the returned map, so the caller falls back to an elide placeholder. - * Propagates whatever `complete` throws. - */ -export async function summarizeShakeRegions( - items: ShakeSummaryItem[], - complete: ShakeSummaryComplete, - options: ShakeSummaryOptions = {}, -): Promise> { - const budget = options.batchTokenBudget ?? DEFAULT_BATCH_TOKEN_BUDGET; - const summaries = new Map(); - if (items.length === 0) return summaries; - - const batches: ShakeSummaryItem[][] = []; - let current: ShakeSummaryItem[] = []; - let currentTokens = 0; - for (const item of items) { - const itemTokens = item.text.length === 0 ? 0 : countTokens(item.text); - if (current.length > 0 && currentTokens + itemTokens > budget) { - batches.push(current); - current = []; - currentTokens = 0; - } - current.push(item); - currentTokens += itemTokens; - } - if (current.length > 0) batches.push(current); - - for (const batch of batches) { - const batchTokens = batch.reduce((sum, item) => sum + (item.text.length === 0 ? 0 : countTokens(item.text)), 0); - const maxTokens = Math.min(2_048, Math.max(256, Math.floor(batchTokens / 2))); - const text = await complete(buildSummaryPrompt(batch), { maxTokens, signal: options.signal }); - if (!text) continue; - const parsed = parseSummaryResponse( - text, - batch.map(item => item.index), - ); - for (const [index, value] of parsed) summaries.set(index, value); - } - - return summaries; -} diff --git a/packages/agent/test/shake.test.ts b/packages/agent/test/shake.test.ts index 48b11e74f..8c8d0dba5 100644 --- a/packages/agent/test/shake.test.ts +++ b/packages/agent/test/shake.test.ts @@ -1,12 +1,6 @@ import { describe, expect, test } from "bun:test"; import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; -import type { - SessionEntry, - SessionMessageEntry, - ShakeConfig, - ShakeSummaryComplete, - ShakeSummaryItem, -} from "@oh-my-pi/pi-agent-core/compaction"; +import type { SessionEntry, SessionMessageEntry, ShakeConfig } from "@oh-my-pi/pi-agent-core/compaction"; import { AGGRESSIVE_SHAKE_CONFIG, applyShakeRegion, @@ -14,7 +8,6 @@ import { collectShakeRegions, DEFAULT_SHAKE_CONFIG, estimateTokens, - summarizeShakeRegions, } from "@oh-my-pi/pi-agent-core/compaction"; import type { AssistantMessage, TextContent, ToolCall, ToolResultMessage } from "@oh-my-pi/pi-ai"; @@ -226,62 +219,3 @@ describe("shake config presets", () => { expect(collectShakeRegions([] as SessionEntry[], AGGRESSIVE_SHAKE_CONFIG)).toHaveLength(0); }); }); - -describe("summarizeShakeRegions — local-model compressor", () => { - const items: ShakeSummaryItem[] = [ - { index: 0, label: "bash", text: "alpha ".repeat(40) }, - { index: 1, label: "read", text: "beta ".repeat(40) }, - ]; - - test("parses delimited per-region output from the local backend", async () => { - const complete: ShakeSummaryComplete = async () => - 'compressed A\ncompressed B'; - const summaries = await summarizeShakeRegions(items, complete); - expect(summaries.get(0)).toBe("compressed A"); - expect(summaries.get(1)).toBe("compressed B"); - }); - - test("omits regions the model did not emit (caller elides them)", async () => { - const complete: ShakeSummaryComplete = async () => 'only A'; - const summaries = await summarizeShakeRegions(items, complete); - expect(summaries.get(0)).toBe("only A"); - expect(summaries.has(1)).toBe(false); - }); - - test("returns an empty map when the local model is unavailable (null)", async () => { - const complete: ShakeSummaryComplete = async () => null; - const summaries = await summarizeShakeRegions(items, complete); - expect(summaries.size).toBe(0); - }); - - test("feeds the configured maxTokens and the rendered region prompt to the backend", async () => { - let seenPrompt = ""; - let seenMaxTokens = 0; - const complete: ShakeSummaryComplete = async (prompt, opts) => { - seenPrompt = prompt; - seenMaxTokens = opts.maxTokens; - return 'x\ny'; - }; - await summarizeShakeRegions(items, complete); - expect(seenPrompt).toContain(''); - expect(seenPrompt).toContain(''); - expect(seenMaxTokens).toBeGreaterThanOrEqual(256); - }); - - test("splits regions across batches by token budget (one call per batch)", async () => { - const big: ShakeSummaryItem[] = [ - { index: 0, label: "bash", text: "word ".repeat(400) }, - { index: 1, label: "read", text: "word ".repeat(400) }, - ]; - let calls = 0; - const complete: ShakeSummaryComplete = async prompt => { - calls++; - // Echo back whichever index this batch carried. - return /index="0"/.test(prompt) ? 'a' : 'b'; - }; - const summaries = await summarizeShakeRegions(big, complete, { batchTokenBudget: 200 }); - expect(calls).toBe(2); - expect(summaries.get(0)).toBe("a"); - expect(summaries.get(1)).toBe("b"); - }); -}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 52a951793..bf03fccee 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Removed + +- Removed `/shake summary`, the `shake-summary` auto-compaction strategy, and the `providers.shakeSummaryModel` setting. Use `/shake` or `compaction.strategy: shake` for mechanical artifact-backed elision without local-model CPU. + ## [15.7.3] - 2026-05-31 ### Added diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 965aee208..52b3fbc7e 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -14,12 +14,9 @@ import { import { AUTO_THINKING_MODEL_OPTIONS, AUTO_THINKING_MODEL_VALUES, - DEFAULT_SHAKE_SUMMARY_MODEL_KEY, ONLINE_AUTO_THINKING_MODEL_KEY, ONLINE_MEMORY_MODEL_KEY, ONLINE_TINY_TITLE_MODEL_KEY, - SHAKE_SUMMARY_MODEL_OPTIONS, - SHAKE_SUMMARY_MODEL_VALUES, TINY_MEMORY_MODEL_OPTIONS, TINY_MEMORY_MODEL_VALUES, TINY_TITLE_MODEL_OPTIONS, @@ -1139,13 +1136,13 @@ export const SETTINGS_SCHEMA = { "compaction.strategy": { type: "enum", - values: ["context-full", "handoff", "shake", "shake-summary", "off"] as const, + values: ["context-full", "handoff", "shake", "off"] as const, default: "context-full", ui: { tab: "context", label: "Compaction Strategy", description: - "Choose in-place context-full maintenance, auto-handoff, surgical shake (drop heavy content), shake with local-model summaries, or disable auto maintenance (off)", + "Choose in-place context-full maintenance, auto-handoff, surgical shake (drop heavy content), or disable auto maintenance (off)", options: [ { value: "context-full", @@ -1158,11 +1155,6 @@ export const SETTINGS_SCHEMA = { label: "Shake", description: "Drop heavy content (tool results + large blocks) in place; recover via artifact", }, - { - value: "shake-summary", - label: "Shake (summary)", - description: "Shake, but compress heavy regions with a local on-device model instead of dropping", - }, { value: "off", label: "Off", @@ -3018,19 +3010,6 @@ export const SETTINGS_SCHEMA = { }, }, - "providers.shakeSummaryModel": { - type: "enum", - values: SHAKE_SUMMARY_MODEL_VALUES, - default: DEFAULT_SHAKE_SUMMARY_MODEL_KEY, - ui: { - tab: "context", - label: "Shake Summary Model", - description: - "Local on-device model used by /shake summary and the shake-summary compaction strategy to compress heavy regions. Runs entirely on-device; downloads on first use. Falls back to plain elide when unavailable.", - options: SHAKE_SUMMARY_MODEL_OPTIONS, - }, - }, - "providers.kimiApiFormat": { type: "enum", values: ["openai", "anthropic"] as const, @@ -3319,7 +3298,7 @@ export type TreeFilterMode = SettingValue<"treeFilterMode">; export interface CompactionSettings { enabled: boolean; - strategy: "context-full" | "handoff" | "shake" | "shake-summary" | "off"; + strategy: "context-full" | "handoff" | "shake" | "off"; thresholdPercent: number; thresholdTokens: number; reserveTokens: number; diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index b7d5451a5..62ed1325b 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -663,6 +663,16 @@ export class Settings { raw["edit.mode"] = "hashline"; } + // compaction.strategy: removed local-model shake-summary mode; plain shake + // keeps the same mechanical artifact-backed reduction without background CPU. + const compactionObj = raw.compaction as Record | undefined; + if (compactionObj?.strategy === "shake-summary") { + compactionObj.strategy = "shake"; + } + if (raw["compaction.strategy"] === "shake-summary") { + raw["compaction.strategy"] = "shake"; + } + // statusLine: rename "plan_mode" segment to "mode" const statusLineObj = raw.statusLine as Record | undefined; if (statusLineObj) { diff --git a/packages/coding-agent/src/extensibility/custom-tools/types.ts b/packages/coding-agent/src/extensibility/custom-tools/types.ts index 31f4ff774..3115eae8c 100644 --- a/packages/coding-agent/src/extensibility/custom-tools/types.ts +++ b/packages/coding-agent/src/extensibility/custom-tools/types.ts @@ -101,11 +101,11 @@ export type CustomToolSessionEvent = | { reason: "auto_compaction_start"; trigger: "threshold" | "overflow" | "idle" | "incomplete"; - action: "context-full" | "handoff" | "shake" | "shake-summary"; + action: "context-full" | "handoff" | "shake"; } | { reason: "auto_compaction_end"; - action: "context-full" | "handoff" | "shake" | "shake-summary"; + action: "context-full" | "handoff" | "shake"; result: CompactionResult | undefined; aborted: boolean; willRetry: boolean; diff --git a/packages/coding-agent/src/extensibility/shared-events.ts b/packages/coding-agent/src/extensibility/shared-events.ts index 72adf5328..0ee574ef4 100644 --- a/packages/coding-agent/src/extensibility/shared-events.ts +++ b/packages/coding-agent/src/extensibility/shared-events.ts @@ -204,13 +204,13 @@ export interface TurnEndEvent { export interface AutoCompactionStartEvent { type: "auto_compaction_start"; reason: "threshold" | "overflow" | "idle" | "incomplete"; - action: "context-full" | "handoff" | "shake" | "shake-summary"; + action: "context-full" | "handoff" | "shake"; } /** Fired when auto-compaction ends */ export interface AutoCompactionEndEvent { type: "auto_compaction_end"; - action: "context-full" | "handoff" | "shake" | "shake-summary"; + action: "context-full" | "handoff" | "shake"; result: CompactionResult | undefined; aborted: boolean; willRetry: boolean; diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index e82d98702..96c1526df 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -1124,48 +1124,16 @@ export class CommandController { } /** - * TUI handler for `/shake`. `elide`/`images` are instant structural drops; - * `summary` runs the local on-device compressor behind a cancelable loader - * (Esc aborts via `abortCompaction`). Rebuilds the chat and reports counts. + * TUI handler for `/shake`. `elide` drops heavy structural content and + * `images` strips image blocks. Rebuilds the chat and reports counts. */ async handleShakeCommand(mode: ShakeMode): Promise { let result: ShakeResult; - if (mode === "summary") { - if (this.ctx.loadingAnimation) { - this.ctx.loadingAnimation.stop(); - this.ctx.loadingAnimation = undefined; - } - this.ctx.statusContainer.clear(); - const originalOnEscape = this.ctx.editor.onEscape; - this.ctx.editor.onEscape = () => { - this.ctx.session.abortCompaction(); - }; - const loader = new Loader( - this.ctx.ui, - spinner => theme.fg("accent", spinner), - text => theme.fg("muted", text), - "Shaking context (summary)… (esc to cancel)", - getSymbolTheme().spinnerFrames, - ); - this.ctx.statusContainer.addChild(loader); - this.ctx.ui.requestRender(); - try { - result = await this.ctx.session.shake("summary"); - } catch (error) { - this.ctx.showError(`Shake failed: ${error instanceof Error ? error.message : String(error)}`); - return; - } finally { - loader.stop(); - this.ctx.statusContainer.clear(); - this.ctx.editor.onEscape = originalOnEscape; - } - } else { - try { - result = await this.ctx.session.shake(mode); - } catch (error) { - this.ctx.showError(`Shake failed: ${error instanceof Error ? error.message : String(error)}`); - return; - } + try { + result = await this.ctx.session.shake(mode); + } catch (error) { + this.ctx.showError(`Shake failed: ${error instanceof Error ? error.message : String(error)}`); + return; } const dropped = result.toolResultsDropped + result.blocksDropped + (result.imagesDropped ?? 0); diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index 999d36df4..603c3ebae 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -647,9 +647,7 @@ export class EventController { ? "Auto-handoff" : event.action === "shake" ? "Auto-shake" - : event.action === "shake-summary" - ? "Auto-shake (summary)" - : "Auto context-full maintenance"; + : "Auto context-full maintenance"; this.ctx.autoCompactionLoader = new Loader( this.ctx.ui, spinner => theme.fg("accent", spinner), @@ -673,7 +671,7 @@ export class EventController { this.ctx.statusContainer.clear(); } const isHandoffAction = event.action === "handoff"; - const isShakeAction = event.action === "shake" || event.action === "shake-summary"; + const isShakeAction = event.action === "shake"; if (event.aborted) { this.ctx.showStatus( isHandoffAction @@ -690,9 +688,7 @@ export class EventController { this.ctx.rebuildChatFromMessages(); this.ctx.statusLine.invalidate(); this.ctx.updateEditorTopBorder(); - this.ctx.showStatus( - event.action === "shake-summary" ? "Auto-shake (summary) completed" : "Auto-shake completed", - ); + this.ctx.showStatus("Auto-shake completed"); } } else if (event.result) { this.ctx.rebuildChatFromMessages(); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index a1e1678f7..5a587d0d9 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -51,11 +51,8 @@ import { prepareCompaction, type ShakeConfig, type ShakeRegion, - type ShakeSummaryComplete, - type ShakeSummaryItem, type SummaryOptions, shouldCompact, - summarizeShakeRegions, } from "@oh-my-pi/pi-agent-core/compaction"; import { DEFAULT_PRUNE_CONFIG, pruneToolOutputs } from "@oh-my-pi/pi-agent-core/compaction/pruning"; import type { @@ -185,8 +182,7 @@ import { resolveThinkingLevelForModel, toReasoningEffort, } from "../thinking"; -import { isTinyMemoryLocalModelKey } from "../tiny/models"; -import { shutdownTinyTitleClient, tinyModelClient } from "../tiny/title-client"; +import { shutdownTinyTitleClient } from "../tiny/title-client"; import { buildDiscoverableToolSearchIndex, collectDiscoverableTools, @@ -241,11 +237,11 @@ export type AgentSessionEvent = | { type: "auto_compaction_start"; reason: "threshold" | "overflow" | "idle" | "incomplete"; - action: "context-full" | "handoff" | "shake" | "shake-summary"; + action: "context-full" | "handoff" | "shake"; } | { type: "auto_compaction_end"; - action: "context-full" | "handoff" | "shake" | "shake-summary"; + action: "context-full" | "handoff" | "shake"; result: CompactionResult | undefined; aborted: boolean; willRetry: boolean; @@ -5659,10 +5655,6 @@ export class AgentSession { * - `images` delegates to {@link dropImages}. * - `elide` replaces whole tool-call results and large fenced/XML blocks * with short placeholders that embed an `artifact://` recovery link. - * - `summary` extractively compresses the same regions with the configured - * local on-device model (`providers.shakeSummaryModel`), falling back to - * the elide placeholder per region (or wholesale when the local model is - * unavailable). Never calls a remote/cloud LLM. * * Mutates the branch in place, persists via `rewriteEntries`, replays the * rebuilt context through the agent, and tears down provider sessions that @@ -5683,29 +5675,7 @@ export class AgentSession { } const artifactId = await this.#saveShakeArtifact(regions); - let replacements: string[]; - if (mode === "summary") { - // Manual `/shake summary` installs the compaction controller so Esc / - // `abortCompaction()` can cancel the local-model pass; the auto-shake path - // passes its own signal and manages `#autoCompactionAbortController`. - let controller: AbortController | undefined; - let signal = opts.signal; - if (!signal) { - if (this.#compactionAbortController) throw new Error("Compaction already in progress"); - controller = new AbortController(); - this.#compactionAbortController = controller; - signal = controller.signal; - } - try { - replacements = await this.#buildShakeSummaryReplacements(regions, artifactId, signal); - } finally { - if (controller && this.#compactionAbortController === controller) { - this.#compactionAbortController = undefined; - } - } - } else { - replacements = regions.map((region, index) => this.#shakeElidePlaceholder(region, index, artifactId)); - } + const replacements = regions.map((region, index) => this.#shakeElidePlaceholder(region, index, artifactId)); let toolResultsDropped = 0; let blocksDropped = 0; @@ -5762,56 +5732,6 @@ export class AgentSession { } } - /** - * Build per-region replacements for summary mode using the configured local - * on-device model (`providers.shakeSummaryModel`) via {@link tinyModelClient}. - * Shake summary never calls a remote/cloud LLM. When the configured model is - * not a known local key, every region falls back to the elide placeholder. - * Otherwise compresses via {@link summarizeShakeRegions}; per region, uses - * the parsed summary (with a recovery footer) or the elide placeholder when - * the local model omitted it / was unavailable. Any thrown failure degrades - * the whole batch to elide so the reduction still happens. - */ - async #buildShakeSummaryReplacements( - regions: ShakeRegion[], - artifactId: string | undefined, - signal: AbortSignal | undefined, - ): Promise { - const elide = (): string[] => - regions.map((region, index) => this.#shakeElidePlaceholder(region, index, artifactId)); - - const modelKey = this.settings.get("providers.shakeSummaryModel"); - if (!isTinyMemoryLocalModelKey(modelKey)) return elide(); - - const items: ShakeSummaryItem[] = regions.map((region, index) => ({ - index, - label: region.label, - text: region.originalText, - })); - - const complete: ShakeSummaryComplete = (promptText, opts) => - tinyModelClient.complete(modelKey, promptText, { maxTokens: opts.maxTokens, signal: opts.signal }); - - let summaries: Map; - try { - summaries = await summarizeShakeRegions(items, complete, { signal }); - } catch (error) { - logger.warn("Shake summary compression failed; falling back to elide", { - error: error instanceof Error ? error.message : String(error), - }); - return elide(); - } - - return regions.map((region, index) => { - const summary = summaries.get(index); - if (!summary) return this.#shakeElidePlaceholder(region, index, artifactId); - if (artifactId) { - return `${summary}\n\n[recover full: artifact://${artifactId} (region ${index + 1})]`; - } - return summary; - }); - } - /** * Manually compact the session context. * Aborts current agent operation first. @@ -7025,11 +6945,11 @@ export class AgentSession { if (reason !== "idle" && !compactionSettings.enabled) return false; const generation = this.#promptGeneration; - // Shake strategies run inline (cheap, no remote LLM). On overflow recovery, - // if shake reclaims nothing we fall through to the summary-compaction body - // below so the oversized input still gets resolved. - if (compactionSettings.strategy === "shake" || compactionSettings.strategy === "shake-summary") { - const outcome = await this.#runAutoShake(reason, compactionSettings.strategy, willRetry, generation); + // Shake runs inline (cheap, no remote LLM). On overflow recovery, if shake + // reclaims nothing we fall through to the summary-compaction body below so + // the oversized input still gets resolved. + if (compactionSettings.strategy === "shake") { + const outcome = await this.#runAutoShake(reason, willRetry, generation); if (outcome !== "fallback") return false; } // "overflow" and "incomplete" force inline execution because they are recovery @@ -7426,12 +7346,10 @@ export class AgentSession { */ async #runAutoShake( reason: "overflow" | "threshold" | "idle" | "incomplete", - strategy: "shake" | "shake-summary", willRetry: boolean, generation: number, ): Promise<"handled" | "fallback"> { - const action = strategy === "shake-summary" ? "shake-summary" : "shake"; - const mode = strategy === "shake-summary" ? "summary" : "elide"; + const action = "shake"; await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action }); this.#autoCompactionAbortController?.abort(); const controller = new AbortController(); @@ -7439,7 +7357,7 @@ export class AgentSession { const signal = controller.signal; const compactionSettings = this.settings.getGroup("compaction"); try { - const result = await this.shake(mode, { config: DEFAULT_SHAKE_CONFIG, signal }); + const result = await this.shake("elide", { config: DEFAULT_SHAKE_CONFIG, signal }); if (signal.aborted) { await this.#emitSessionEvent({ type: "auto_compaction_end", diff --git a/packages/coding-agent/src/session/shake-types.ts b/packages/coding-agent/src/session/shake-types.ts index 71a2a5035..ee73aecb5 100644 --- a/packages/coding-agent/src/session/shake-types.ts +++ b/packages/coding-agent/src/session/shake-types.ts @@ -6,14 +6,14 @@ */ /** Mode selector for `AgentSession.shake`. */ -export type ShakeMode = "elide" | "summary" | "images"; +export type ShakeMode = "elide" | "images"; /** Outcome of an `AgentSession.shake` run. */ export interface ShakeResult { mode: ShakeMode; - /** Whole tool-call results dropped/compressed. */ + /** Whole tool-call results dropped. */ toolResultsDropped: number; - /** Large fenced/XML blocks dropped/compressed. */ + /** Large fenced/XML blocks dropped. */ blocksDropped: number; /** Image blocks removed (images mode only). */ imagesDropped?: number; @@ -39,6 +39,5 @@ export function formatShakeSummary(result: ShakeResult): string { parts.push(`${result.blocksDropped} block${result.blocksDropped === 1 ? "" : "s"}`); } if (parts.length === 0) return "Nothing to shake."; - const verb = result.mode === "summary" ? "Compressed" : "Shook"; - return `${verb} ${parts.join(" + ")} (~${result.tokensFreed} tokens freed).`; + return `Shook ${parts.join(" + ")} (~${result.tokensFreed} tokens freed).`; } diff --git a/packages/coding-agent/src/slash-commands/builtin-registry.ts b/packages/coding-agent/src/slash-commands/builtin-registry.ts index 86b8af258..b5330eb9c 100644 --- a/packages/coding-agent/src/slash-commands/builtin-registry.ts +++ b/packages/coding-agent/src/slash-commands/builtin-registry.ts @@ -62,9 +62,8 @@ const shutdownHandlerTui = (_command: ParsedSlashCommand, runtime: TuiSlashComma function parseShakeMode(args: string): ShakeMode | { error: string } { const verb = args.trim().toLowerCase(); if (verb === "" || verb === "elide") return "elide"; - if (verb === "summary") return "summary"; if (verb === "images") return "images"; - return { error: `Unknown /shake mode "${verb}". Use elide, summary, or images.` }; + return { error: `Unknown /shake mode "${verb}". Use elide or images.` }; } const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ @@ -826,10 +825,9 @@ const BUILTIN_SLASH_COMMAND_REGISTRY: ReadonlyArray = [ acpDescription: "Shake heavy content out of the conversation context", subcommands: [ { name: "elide", description: "Strip tool results + large blocks (default)" }, - { name: "summary", description: "Compress heavy regions with a local on-device model" }, { name: "images", description: "Strip image blocks" }, ], - acpInputHint: "[elide|summary|images]", + acpInputHint: "[elide|images]", allowArgs: true, handle: async (command, runtime) => { const mode = parseShakeMode(command.args); diff --git a/packages/coding-agent/src/tiny/models.ts b/packages/coding-agent/src/tiny/models.ts index aa8a66773..e1ec19cf0 100644 --- a/packages/coding-agent/src/tiny/models.ts +++ b/packages/coding-agent/src/tiny/models.ts @@ -196,34 +196,6 @@ export function getTinyMemoryModelSpec(key: TinyMemoryLocalModelKey): (typeof TI return spec; } -/** - * Shake-summary models. Shake's `summary` mode (and the `shake-summary` - * compaction strategy) compress heavy regions strictly on-device — there is no - * online/remote option, so this registry reuses the local memory models only. - */ -export const SHAKE_SUMMARY_MODEL_VALUES = [ - "qwen3-1.7b", - "gemma-3-1b", - "qwen2.5-1.5b", - "lfm2-1.2b", -] as const satisfies readonly TinyMemoryLocalModelKey[]; - -export type ShakeSummaryModelKey = (typeof SHAKE_SUMMARY_MODEL_VALUES)[number]; - -// Guard: every local memory model is offered for shake summary (catches drift). -type MissingShakeSummaryValue = Exclude; -const SHAKE_SUMMARY_MODEL_VALUES_MATCH_REGISTRY: MissingShakeSummaryValue extends never ? true : never = true; -void SHAKE_SUMMARY_MODEL_VALUES_MATCH_REGISTRY; - -export const SHAKE_SUMMARY_MODEL_OPTIONS = TINY_MEMORY_LOCAL_MODELS.map(model => ({ - value: model.key, - label: model.label, - description: model.description, -})) satisfies ReadonlyArray<{ value: ShakeSummaryModelKey; label: string; description: string }>; - -/** Default shake-summary local model when none is named. */ -export const DEFAULT_SHAKE_SUMMARY_MODEL_KEY: ShakeSummaryModelKey = DEFAULT_MEMORY_LOCAL_MODEL_KEY; - /** Any local model key (title or memory), used by the shared inference worker. */ export type TinyLocalModelKey = TinyTitleLocalModelKey | TinyMemoryLocalModelKey; diff --git a/packages/coding-agent/src/tiny/title-client.ts b/packages/coding-agent/src/tiny/title-client.ts index 7da519e3b..1382f40bd 100644 --- a/packages/coding-agent/src/tiny/title-client.ts +++ b/packages/coding-agent/src/tiny/title-client.ts @@ -211,7 +211,7 @@ export class TinyTitleClient { async complete( modelKey: string, prompt: string, - options: { maxTokens?: number; signal?: AbortSignal; prefill?: string; stop?: string } = {}, + options: { maxTokens?: number; signal?: AbortSignal } = {}, ): Promise { if (!isTinyMemoryLocalModelKey(modelKey)) return null; if (options.signal?.aborted) return null; @@ -229,15 +229,7 @@ export class TinyTitleClient { }; options.signal?.addEventListener("abort", abort, { once: true }); try { - worker.send({ - type: "complete", - id, - modelKey, - prompt, - maxTokens: options.maxTokens, - prefill: options.prefill, - stop: options.stop, - }); + worker.send({ type: "complete", id, modelKey, prompt, maxTokens: options.maxTokens }); return await promise; } finally { options.signal?.removeEventListener("abort", abort); diff --git a/packages/coding-agent/src/tiny/title-protocol.ts b/packages/coding-agent/src/tiny/title-protocol.ts index 49e95b258..9267a0b89 100644 --- a/packages/coding-agent/src/tiny/title-protocol.ts +++ b/packages/coding-agent/src/tiny/title-protocol.ts @@ -30,17 +30,7 @@ export interface TinyTitleProgressEvent { export type TinyTitleWorkerInbound = | { type: "ping"; id: string } | { type: "generate"; id: string; modelKey: TinyTitleLocalModelKey; message: string } - | { - type: "complete"; - id: string; - modelKey: TinyLocalModelKey; - prompt: string; - maxTokens?: number; - /** Optional assistant-turn prefix appended after the generation prompt to pin output format. */ - prefill?: string; - /** Optional literal stop string; generation halts once it appears in the decoded tail. */ - stop?: string; - } + | { type: "complete"; id: string; modelKey: TinyLocalModelKey; prompt: string; maxTokens?: number } | { type: "download"; id: string; modelKey: TinyLocalModelKey } | { type: "close" }; diff --git a/packages/coding-agent/src/tiny/worker.ts b/packages/coding-agent/src/tiny/worker.ts index 3e068919b..2d7a1fa83 100644 --- a/packages/coding-agent/src/tiny/worker.ts +++ b/packages/coding-agent/src/tiny/worker.ts @@ -461,15 +461,14 @@ async function generateTitle( return extractTinyTitle(output[0]?.generated_text ?? ""); } -function buildCompletionPrompt(generator: TextGenerationPipeline, promptText: string, prefill?: string): string { +function buildCompletionPrompt(generator: TextGenerationPipeline, promptText: string): string { const chat = [{ role: "user", content: promptText }]; const chatTemplateOptions = { add_generation_prompt: true, tokenize: false, enable_thinking: false, }; - const base = generator.tokenizer.apply_chat_template(chat, chatTemplateOptions) as string; - return prefill ? `${base}${prefill}` : base; + return `${generator.tokenizer.apply_chat_template(chat, chatTemplateOptions)}`; } /** @@ -484,27 +483,18 @@ async function generateCompletion( modelKey: TinyLocalModelKey, promptText: string, maxTokens: number | undefined, - prefill?: string, - stop?: string, ): Promise { const generator = await loadPipeline(modelKey, transport, requestId); - const text = buildCompletionPrompt(generator, promptText, prefill); + const text = buildCompletionPrompt(generator, promptText); const requested = maxTokens ?? MEMORY_COMPLETION_MAX_NEW_TOKENS; const maxNewTokens = Math.min(Math.max(1, requested), MEMORY_COMPLETION_MAX_NEW_TOKENS); - const transformers = stop ? await loadTransformers(transport, requestId, modelKey) : undefined; const output = (await generator(text, { max_new_tokens: maxNewTokens, do_sample: false, return_full_text: false, - ...(transformers && stop - ? { stopping_criteria: createStopOnTextCriteria(transformers, generator.tokenizer, stop) } - : {}), })) as TextGenerationStringOutput; - const generated = output[0]?.generated_text ?? ""; - // Re-attach the forced prefix so the caller's parser sees the full assistant turn, - // including the opening tag it pinned via `prefill`. - const full = `${prefill ?? ""}${generated}`.trim(); - return full === "" ? null : full; + const generated = (output[0]?.generated_text ?? "").trim(); + return generated === "" ? null : generated; } function releasePipelines(): void { @@ -548,8 +538,6 @@ async function handleQueuedRequest( request.modelKey, request.prompt, request.maxTokens, - request.prefill, - request.stop, ); transport.send({ type: "completion", id: request.id, text }); return; diff --git a/packages/coding-agent/test/shake.test.ts b/packages/coding-agent/test/shake.test.ts index b10d66f09..42c503f1a 100644 --- a/packages/coding-agent/test/shake.test.ts +++ b/packages/coding-agent/test/shake.test.ts @@ -8,7 +8,6 @@ import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; -import { tinyModelClient } from "@oh-my-pi/pi-coding-agent/tiny/title-client"; import { TempDir } from "@oh-my-pi/pi-utils"; const usage = { @@ -144,56 +143,6 @@ describe("AgentSession shake", () => { }); }); - describe("summary (local model)", () => { - it("replaces regions with the local model's parsed compression", async () => { - seedHeavyToolResult("Y".repeat(4000)); - const completeSpy = vi - .spyOn(tinyModelClient, "complete") - .mockResolvedValue('compressed bash output'); - - const result = await session.shake("summary"); - - expect(result.mode).toBe("summary"); - expect(result.toolResultsDropped).toBe(1); - expect(completeSpy).toHaveBeenCalledTimes(1); - // The configured local model key (default qwen3-1.7b) is the first arg. - expect(completeSpy.mock.calls[0][0]).toBe("qwen3-1.7b"); - - const [tr] = branchToolResults(); - const text = tr.content.map(b => (b.type === "text" ? b.text : "")).join(""); - expect(text).toContain("compressed bash output"); - }); - - it("falls back to the elide placeholder when the local model is unavailable", async () => { - seedHeavyToolResult("Z".repeat(4000)); - const completeSpy = vi.spyOn(tinyModelClient, "complete").mockResolvedValue(null); - - const result = await session.shake("summary"); - - expect(completeSpy).toHaveBeenCalled(); - expect(result.toolResultsDropped).toBe(1); - const [tr] = branchToolResults(); - const text = tr.content.map(b => (b.type === "text" ? b.text : "")).join(""); - expect(text).toContain("shaken"); - expect(text).toContain("artifact://"); - }); - - it("falls back to elide per region the local model omits", async () => { - seedHeavyToolResult("A".repeat(4000)); - seedHeavyToolResult("B".repeat(4000)); - // Only region 0 is summarized; region 1 is omitted → elide fallback. - vi.spyOn(tinyModelClient, "complete").mockResolvedValue('summary of A'); - - const result = await session.shake("summary"); - - expect(result.toolResultsDropped).toBe(2); - const results = branchToolResults(); - const texts = results.map(tr => tr.content.map(b => (b.type === "text" ? b.text : "")).join("")); - expect(texts.some(t => t.includes("summary of A"))).toBe(true); - expect(texts.some(t => t.includes("shaken"))).toBe(true); - }); - }); - describe("protected tools", () => { it("never shakes skill results", async () => { seedHeavyToolResult("S".repeat(4000), "skill"); diff --git a/packages/coding-agent/test/slash-commands/shake.test.ts b/packages/coding-agent/test/slash-commands/shake.test.ts index f7b313095..9edc3beb1 100644 --- a/packages/coding-agent/test/slash-commands/shake.test.ts +++ b/packages/coding-agent/test/slash-commands/shake.test.ts @@ -44,7 +44,7 @@ describe("/shake dispatch (ACP)", () => { }); it("parses each explicit mode", async () => { - for (const mode of ["elide", "summary", "images"] as const) { + for (const mode of ["elide", "images"] as const) { const h = acpRuntime(); await executeAcpBuiltinSlashCommand(`/shake ${mode}`, h.runtime); expect(h.shake).toHaveBeenCalledWith(mode); @@ -62,7 +62,7 @@ describe("/shake dispatch (ACP)", () => { it("is advertised to ACP clients with the mode hint", () => { const advertised = ACP_BUILTIN_SLASH_COMMANDS.find(c => c.name === "shake"); expect(advertised).toBeDefined(); - expect(advertised?.input?.hint).toBe("[elide|summary|images]"); + expect(advertised?.input?.hint).toBe("[elide|images]"); }); it("advertises /shake images as the image-stripping path and no longer advertises /drop-images", () => { @@ -74,10 +74,10 @@ describe("/shake dispatch (ACP)", () => { describe("/shake dispatch (TUI)", () => { it("routes the parsed mode to handleShakeCommand and clears the editor", async () => { const h = tuiRuntime(); - const handled = await executeBuiltinSlashCommand("/shake summary", h.runtime); + const handled = await executeBuiltinSlashCommand("/shake images", h.runtime); expect(handled).toBe(true); expect(h.setText).toHaveBeenCalledWith(""); - expect(h.handleShakeCommand).toHaveBeenCalledWith("summary"); + expect(h.handleShakeCommand).toHaveBeenCalledWith("images"); }); it("defaults to elide for a bare /shake", async () => {